Files
TheLadder/tools/story-to-pack/probe/evaluate_recover.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

47 lines
2.2 KiB
Python

"""Score the recall-recovery methods against the blind labels (SELECTION.md, "Recovering recall").
python3 evaluate_recover.py
"""
import collections, json, pathlib
mapping = json.loads(pathlib.Path('recover-map.json').read_text())
labels = json.loads(pathlib.Path('labels-recover.json').read_text())
assert all(labels[r]['label'] in (0, 1, 2) for r in mapping), 'labels-recover.json is incomplete'
N_SCENES = 838
stats = {}
for method in ('B', 'M1', 'M2', 'M3'):
ids = [r for r, e in mapping.items() if method in e['methods']]
lab = [labels[r]['label'] for r in ids]
names = [labels[r]['name'].strip().lower() for r in ids if labels[r]['label'] >= 1]
scenes = collections.Counter(i for r in ids for i in mapping[r]['members'])
stats[method] = {
'accepted': len(ids),
'precision1plus': sum(l >= 1 for l in lab) / len(ids) if ids else 0.0,
'precision2': sum(l == 2 for l in lab) / len(ids) if ids else 0.0,
'label2': sum(l == 2 for l in lab),
'distinct_situations': len(set(names)),
'repeated_names': sum(c - 1 for c in collections.Counter(names).values() if c > 1),
'coverage': len(scenes) / N_SCENES,
'scenes_in_several': sum(v > 1 for v in scenes.values()),
'ids': ids,
}
b = stats['B']
print(f"{'method':6} {'acc':>4} {'P1+':>5} {'P2':>5} {'#2':>3} {'distinct':>8} {'dup names':>9} {'cover':>6} {'multi':>5} success")
for m, s in stats.items():
ok = (m != 'B' and s['accepted'] >= 2 * b['accepted'] and s['precision1plus'] >= 0.80
and s['distinct_situations'] >= 2 * b['distinct_situations'])
print(f"{m:6} {s['accepted']:4} {s['precision1plus']:5.2f} {s['precision2']:5.2f} {s['label2']:3} "
f"{s['distinct_situations']:8} {s['repeated_names']:9} {s['coverage']:6.0%} {s['scenes_in_several']:5} "
f"{'-' if m == 'B' else ('PASS' if ok else 'fail')}")
print('\nper method, accepted clusters:')
for m, s in stats.items():
print(f'\n{m}:')
for r in s['ids']:
e = mapping[r]
print(f" {r} size {len(e['members']):2} label {labels[r]['label']} {labels[r]['name'] or '-'}")
pathlib.Path('recover-result.json').write_text(json.dumps(
{m: {k: v for k, v in s.items() if k != 'ids'} for m, s in stats.items()}, indent=1))