Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
91 lines
4.8 KiB
Python
91 lines
4.8 KiB
Python
"""Score G4 (model-judged FIT) against the blind labels (SELECTION.md, "G4").
|
|
|
|
python3 evaluate_judge.py qwen2.5:3b-instruct
|
|
"""
|
|
import hashlib, json, pathlib, statistics, sys
|
|
import gate2 as g
|
|
|
|
MODEL = sys.argv[1]
|
|
judged = json.loads(pathlib.Path(f"judge-{MODEL.replace(':', '_').replace('/', '_')}.json").read_text())
|
|
G2 = json.loads(pathlib.Path('gate2-frozen.json').read_text())
|
|
|
|
def key(members):
|
|
return hashlib.sha1(','.join(map(str, sorted(members))).encode()).hexdigest()[:16]
|
|
|
|
sources = [('dev', 'candidates-dev.json', 'labels-dev.json', False), ('test', 'candidates-test.json', 'labels-test.json', False),
|
|
('recover', 'recover-map.json', 'labels-recover.json', True), ('validate', 'validate-map.json', 'labels-validate.json', True),
|
|
('validate3', 'validate3-map.json', 'labels-validate3.json', True)]
|
|
rows = []
|
|
for name, cfile, lfile, is_map in sources:
|
|
cand = json.loads(pathlib.Path(cfile).read_text())
|
|
lab = json.loads(pathlib.Path(lfile).read_text())
|
|
for cid, v in cand.items():
|
|
members = v['members'] if is_map else v
|
|
j = judged.get(key(members))
|
|
if j is None: continue
|
|
sig = g.signals(members)
|
|
rows.append({'set': name, 'id': cid, 'label': lab[cid]['label'], 'name': lab[cid]['name'],
|
|
'size': len(members), 'fit': j['fit'], 'parsed': j['parsed'], 'situation': j['situation'],
|
|
'g2': g.g2_passes(sig, G2)})
|
|
|
|
def score(sub, accept):
|
|
acc = [r for r in sub if accept(r)]
|
|
n1 = sum(r['label'] >= 1 for r in acc)
|
|
return {'accepted': len(acc), 'label1plus': n1, 'label2': sum(r['label'] == 2 for r in acc),
|
|
'P1+': n1 / len(acc) if acc else 0.0,
|
|
'recall1+': n1 / max(1, sum(r['label'] >= 1 for r in sub))}
|
|
|
|
def line(label, s):
|
|
return (f" {label:22} accepted {s['accepted']:3} >=1 {s['label1plus']:3} label2 {s['label2']:2} "
|
|
f"P1+ {s['P1+']:.2f} recall {s['recall1+']:.2f}")
|
|
|
|
dev = [r for r in rows if r['set'] == 'dev']
|
|
best = None
|
|
for theta in (0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0):
|
|
s = score(dev, lambda r: r['fit'] >= theta)
|
|
if s['accepted'] == 0 or s['P1+'] < 0.85: continue
|
|
k = (s['label1plus'], s['label2'], theta)
|
|
if best is None or k > best[0]: best = (k, theta)
|
|
theta = best[1] if best else None
|
|
print(f'{MODEL}: {len(rows)} labelled cluster rows; parse failures {sum(not r["parsed"] for r in rows)}')
|
|
print(f'theta chosen on dev: {theta}' + ('' if theta is not None else ' (no theta reaches P1+ >= 0.85 on dev)'))
|
|
for t in (0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0):
|
|
print(line(f'dev FIT >= {t}', score(dev, lambda r: r['fit'] >= t)))
|
|
|
|
print('\nmean FIT by label (all rows):')
|
|
for lab in (0, 1, 2):
|
|
f = [r['fit'] for r in rows if r['label'] == lab]
|
|
print(f' label {lab}: n={len(f):3} mean {statistics.mean(f):.2f} median {statistics.median(f):.2f}')
|
|
|
|
if theta is not None:
|
|
g4 = lambda r: r['fit'] >= theta
|
|
for title, names in (('PRIMARY: validate + validate3', ('validate', 'validate3')), ('secondary: test + recover', ('test', 'recover'))):
|
|
sub = [r for r in rows if r['set'] in names]
|
|
print(f'\n{title}: {len(sub)} clusters, {sum(r["label"] >= 1 for r in sub)} labelled >=1')
|
|
print(line('all', score(sub, lambda r: True)))
|
|
print(line('G2', score(sub, lambda r: r['g2'])))
|
|
print(line(f'G4 FIT >= {theta}', score(sub, g4)))
|
|
print(line('G4 AND G2', score(sub, lambda r: g4(r) and r['g2'])))
|
|
for band, lo, hi in (('<=9', 0, 9), ('10-25', 10, 25), ('>=26', 26, 999)):
|
|
b = [r for r in sub if lo <= r['size'] <= hi]
|
|
print(line(f' G4 size {band}', score(b, g4)) + f' (of {len(b)})')
|
|
|
|
prim = [r for r in rows if r['set'] in ('validate', 'validate3')]
|
|
s4, s2 = score(prim, g4), score(prim, lambda r: r['g2'])
|
|
small = score([r for r in prim if r['size'] <= 9], g4)
|
|
c1 = s4['P1+'] >= 0.80
|
|
c2 = s4['label1plus'] > s2['label1plus']
|
|
c3 = small['accepted'] < 3 or small['P1+'] >= 0.67
|
|
print(f'\ncriterion 1 (P1+ >= 0.80): {"PASS" if c1 else "fail"} ({s4["P1+"]:.2f})')
|
|
print(f'criterion 2 (more >=1 than G2): {"PASS" if c2 else "fail"} ({s4["label1plus"]} vs {s2["label1plus"]})')
|
|
print(f'criterion 3 (<=9 members P1+ >= 0.67 if >= 3 accepted): {"PASS" if c3 else "fail"} '
|
|
f'({small["accepted"]} accepted, P1+ {small["P1+"]:.2f})')
|
|
print(f'OVERALL: {"PASS" if c1 and c2 and c3 else "fail"}')
|
|
|
|
print('\naccepted on the primary set, model situation vs label name:')
|
|
for r in sorted(prim, key=lambda r: -r['fit']):
|
|
if g4(r):
|
|
print(f" {r['id']} size {r['size']:2} label {r['label']} fit {r['fit']:.2f} model: {r['situation'][:55]:55} | label: {r['name']}")
|
|
pathlib.Path(f"judge-result-{MODEL.replace(':', '_')}.json").write_text(json.dumps(
|
|
{'theta': theta, 'primary_G4': s4, 'primary_G2': s2, 'small': small, 'criteria': [c1, c2, c3]}, indent=1))
|