Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
47 lines
2.2 KiB
Python
47 lines
2.2 KiB
Python
"""Score the recall-recovery methods against the blind labels (SELECTION.md, "Recovering recall").
|
|
|
|
python3 evaluate_recover.py
|
|
"""
|
|
import collections, json, pathlib
|
|
|
|
mapping = json.loads(pathlib.Path('recover-map.json').read_text())
|
|
labels = json.loads(pathlib.Path('labels-recover.json').read_text())
|
|
assert all(labels[r]['label'] in (0, 1, 2) for r in mapping), 'labels-recover.json is incomplete'
|
|
N_SCENES = 838
|
|
|
|
stats = {}
|
|
for method in ('B', 'M1', 'M2', 'M3'):
|
|
ids = [r for r, e in mapping.items() if method in e['methods']]
|
|
lab = [labels[r]['label'] for r in ids]
|
|
names = [labels[r]['name'].strip().lower() for r in ids if labels[r]['label'] >= 1]
|
|
scenes = collections.Counter(i for r in ids for i in mapping[r]['members'])
|
|
stats[method] = {
|
|
'accepted': len(ids),
|
|
'precision1plus': sum(l >= 1 for l in lab) / len(ids) if ids else 0.0,
|
|
'precision2': sum(l == 2 for l in lab) / len(ids) if ids else 0.0,
|
|
'label2': sum(l == 2 for l in lab),
|
|
'distinct_situations': len(set(names)),
|
|
'repeated_names': sum(c - 1 for c in collections.Counter(names).values() if c > 1),
|
|
'coverage': len(scenes) / N_SCENES,
|
|
'scenes_in_several': sum(v > 1 for v in scenes.values()),
|
|
'ids': ids,
|
|
}
|
|
|
|
b = stats['B']
|
|
print(f"{'method':6} {'acc':>4} {'P1+':>5} {'P2':>5} {'#2':>3} {'distinct':>8} {'dup names':>9} {'cover':>6} {'multi':>5} success")
|
|
for m, s in stats.items():
|
|
ok = (m != 'B' and s['accepted'] >= 2 * b['accepted'] and s['precision1plus'] >= 0.80
|
|
and s['distinct_situations'] >= 2 * b['distinct_situations'])
|
|
print(f"{m:6} {s['accepted']:4} {s['precision1plus']:5.2f} {s['precision2']:5.2f} {s['label2']:3} "
|
|
f"{s['distinct_situations']:8} {s['repeated_names']:9} {s['coverage']:6.0%} {s['scenes_in_several']:5} "
|
|
f"{'-' if m == 'B' else ('PASS' if ok else 'fail')}")
|
|
|
|
print('\nper method, accepted clusters:')
|
|
for m, s in stats.items():
|
|
print(f'\n{m}:')
|
|
for r in s['ids']:
|
|
e = mapping[r]
|
|
print(f" {r} size {len(e['members']):2} label {labels[r]['label']} {labels[r]['name'] or '-'}")
|
|
pathlib.Path('recover-result.json').write_text(json.dumps(
|
|
{m: {k: v for k, v in s.items() if k != 'ids'} for m, s in stats.items()}, indent=1))
|