Files
TheLadder/tools/story-to-pack/probe/evaluate_judge.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

91 lines
4.8 KiB
Python

"""Score G4 (model-judged FIT) against the blind labels (SELECTION.md, "G4").
python3 evaluate_judge.py qwen2.5:3b-instruct
"""
import hashlib, json, pathlib, statistics, sys
import gate2 as g
MODEL = sys.argv[1]
judged = json.loads(pathlib.Path(f"judge-{MODEL.replace(':', '_').replace('/', '_')}.json").read_text())
G2 = json.loads(pathlib.Path('gate2-frozen.json').read_text())
def key(members):
return hashlib.sha1(','.join(map(str, sorted(members))).encode()).hexdigest()[:16]
sources = [('dev', 'candidates-dev.json', 'labels-dev.json', False), ('test', 'candidates-test.json', 'labels-test.json', False),
('recover', 'recover-map.json', 'labels-recover.json', True), ('validate', 'validate-map.json', 'labels-validate.json', True),
('validate3', 'validate3-map.json', 'labels-validate3.json', True)]
rows = []
for name, cfile, lfile, is_map in sources:
cand = json.loads(pathlib.Path(cfile).read_text())
lab = json.loads(pathlib.Path(lfile).read_text())
for cid, v in cand.items():
members = v['members'] if is_map else v
j = judged.get(key(members))
if j is None: continue
sig = g.signals(members)
rows.append({'set': name, 'id': cid, 'label': lab[cid]['label'], 'name': lab[cid]['name'],
'size': len(members), 'fit': j['fit'], 'parsed': j['parsed'], 'situation': j['situation'],
'g2': g.g2_passes(sig, G2)})
def score(sub, accept):
acc = [r for r in sub if accept(r)]
n1 = sum(r['label'] >= 1 for r in acc)
return {'accepted': len(acc), 'label1plus': n1, 'label2': sum(r['label'] == 2 for r in acc),
'P1+': n1 / len(acc) if acc else 0.0,
'recall1+': n1 / max(1, sum(r['label'] >= 1 for r in sub))}
def line(label, s):
return (f" {label:22} accepted {s['accepted']:3} >=1 {s['label1plus']:3} label2 {s['label2']:2} "
f"P1+ {s['P1+']:.2f} recall {s['recall1+']:.2f}")
dev = [r for r in rows if r['set'] == 'dev']
best = None
for theta in (0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0):
s = score(dev, lambda r: r['fit'] >= theta)
if s['accepted'] == 0 or s['P1+'] < 0.85: continue
k = (s['label1plus'], s['label2'], theta)
if best is None or k > best[0]: best = (k, theta)
theta = best[1] if best else None
print(f'{MODEL}: {len(rows)} labelled cluster rows; parse failures {sum(not r["parsed"] for r in rows)}')
print(f'theta chosen on dev: {theta}' + ('' if theta is not None else ' (no theta reaches P1+ >= 0.85 on dev)'))
for t in (0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0):
print(line(f'dev FIT >= {t}', score(dev, lambda r: r['fit'] >= t)))
print('\nmean FIT by label (all rows):')
for lab in (0, 1, 2):
f = [r['fit'] for r in rows if r['label'] == lab]
print(f' label {lab}: n={len(f):3} mean {statistics.mean(f):.2f} median {statistics.median(f):.2f}')
if theta is not None:
g4 = lambda r: r['fit'] >= theta
for title, names in (('PRIMARY: validate + validate3', ('validate', 'validate3')), ('secondary: test + recover', ('test', 'recover'))):
sub = [r for r in rows if r['set'] in names]
print(f'\n{title}: {len(sub)} clusters, {sum(r["label"] >= 1 for r in sub)} labelled >=1')
print(line('all', score(sub, lambda r: True)))
print(line('G2', score(sub, lambda r: r['g2'])))
print(line(f'G4 FIT >= {theta}', score(sub, g4)))
print(line('G4 AND G2', score(sub, lambda r: g4(r) and r['g2'])))
for band, lo, hi in (('<=9', 0, 9), ('10-25', 10, 25), ('>=26', 26, 999)):
b = [r for r in sub if lo <= r['size'] <= hi]
print(line(f' G4 size {band}', score(b, g4)) + f' (of {len(b)})')
prim = [r for r in rows if r['set'] in ('validate', 'validate3')]
s4, s2 = score(prim, g4), score(prim, lambda r: r['g2'])
small = score([r for r in prim if r['size'] <= 9], g4)
c1 = s4['P1+'] >= 0.80
c2 = s4['label1plus'] > s2['label1plus']
c3 = small['accepted'] < 3 or small['P1+'] >= 0.67
print(f'\ncriterion 1 (P1+ >= 0.80): {"PASS" if c1 else "fail"} ({s4["P1+"]:.2f})')
print(f'criterion 2 (more >=1 than G2): {"PASS" if c2 else "fail"} ({s4["label1plus"]} vs {s2["label1plus"]})')
print(f'criterion 3 (<=9 members P1+ >= 0.67 if >= 3 accepted): {"PASS" if c3 else "fail"} '
f'({small["accepted"]} accepted, P1+ {small["P1+"]:.2f})')
print(f'OVERALL: {"PASS" if c1 and c2 and c3 else "fail"}')
print('\naccepted on the primary set, model situation vs label name:')
for r in sorted(prim, key=lambda r: -r['fit']):
if g4(r):
print(f" {r['id']} size {r['size']:2} label {r['label']} fit {r['fit']:.2f} model: {r['situation'][:55]:55} | label: {r['name']}")
pathlib.Path(f"judge-result-{MODEL.replace(':', '_')}.json").write_text(json.dumps(
{'theta': theta, 'primary_G4': s4, 'primary_G2': s2, 'small': small, 'criteria': [c1, c2, c3]}, indent=1))