Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
109 lines
5.2 KiB
Python
109 lines
5.2 KiB
Python
"""Validate G2 on fresh data (SELECTION.md, "Validation on fresh data").
|
|
|
|
python3 validate_gate2.py build # V1, V2, V3 -> blind sheet-validate.txt, validate-map.json
|
|
python3 validate_gate2.py score # after labels-validate.json is filled in
|
|
"""
|
|
import json, pathlib, random, sys
|
|
import gate2 as g
|
|
|
|
mode = sys.argv[1] if len(sys.argv) > 1 else 'build'
|
|
GATE = json.loads(pathlib.Path('gate2-frozen.json').read_text())
|
|
GROW_CAP = 45
|
|
|
|
if mode == 'build':
|
|
entries = {} # member tuple -> {'sets': [...], 'stop': ...}
|
|
|
|
def add(members, tag, stop=None):
|
|
e = entries.setdefault(tuple(sorted(members)), {'sets': [], 'stop': None})
|
|
e['sets'].append(tag)
|
|
if stop: e['stop'] = stop
|
|
|
|
for tag, k, seed in (('V1', 100, 15), ('V2', 40, 14)):
|
|
cands = [c for c in g.kmeans(k, seed) if g.eligible(c)]
|
|
sample = random.Random(seed).sample(cands, min(30, len(cands)))
|
|
for c in sample: add(c, tag)
|
|
print(f'{tag}: k={k} seed={seed}: {len(cands)} candidates, sampled {len(sample)}', flush=True)
|
|
|
|
cands = [c for c in g.kmeans(60, 16) if g.eligible(c)]
|
|
accepted = [c for c in cands if g.g2_passes(g.signals(c), GATE)]
|
|
stops = {'gate': 0, 'cap': 0}
|
|
for c in accepted:
|
|
m = list(c)
|
|
stop = 'cap'
|
|
while len(m) < GROW_CAP:
|
|
cen, members = g.centroid(m), set(m)
|
|
j = max((i for i in range(g.n) if i not in members), key=lambda i: g.dot(g.V[i], cen))
|
|
if not g.g2_passes(g.signals(m + [j]), GATE):
|
|
stop = 'gate'
|
|
break
|
|
m.append(j)
|
|
stops[stop] += 1
|
|
add(m, 'V3', stop)
|
|
print(f'V3: k=60 seed=16: {len(cands)} candidates, G2 accepts {len(accepted)}; '
|
|
f'growth stopped by gate {stops["gate"]}, by cap {stops["cap"]}', flush=True)
|
|
|
|
keys = list(entries)
|
|
rnd = random.Random(1616)
|
|
rnd.shuffle(keys)
|
|
mapping = {f'X{i + 1:03d}': {'members': list(k), **entries[k]} for i, k in enumerate(keys)}
|
|
pathlib.Path('validate-map.json').write_text(json.dumps(mapping, indent=1))
|
|
lines = []
|
|
for xid, e in mapping.items():
|
|
shown = e['members'][:]
|
|
rnd.shuffle(shown)
|
|
lines.append(f'=== {xid}')
|
|
lines.extend(f' - {g.text[i]}' for i in shown)
|
|
lines.append('')
|
|
pathlib.Path('sheet-validate.txt').write_text('\n'.join(lines), encoding='utf-8')
|
|
lab = pathlib.Path('labels-validate.json')
|
|
if not lab.exists():
|
|
lab.write_text(json.dumps({x: {'label': None, 'name': ''} for x in mapping}, indent=1))
|
|
print(f'blind sheet: {len(mapping)} clusters, {sum(len(e["members"]) for e in mapping.values())} lines')
|
|
|
|
elif mode == 'score':
|
|
mapping = json.loads(pathlib.Path('validate-map.json').read_text())
|
|
labels = json.loads(pathlib.Path('labels-validate.json').read_text())
|
|
assert all(labels[x]['label'] in (0, 1, 2) for x in mapping), 'labels-validate.json is incomplete'
|
|
rows = {x: dict(g.signals(e['members']), label=labels[x]['label'], name=labels[x]['name'],
|
|
sets=e['sets'], stop=e['stop']) for x, e in mapping.items()}
|
|
|
|
def score(ids, accept):
|
|
acc = [x for x in ids if accept(rows[x])]
|
|
n1 = sum(rows[x]['label'] >= 1 for x in acc)
|
|
n2 = sum(rows[x]['label'] == 2 for x in acc)
|
|
return len(acc), n1, n2, (n1 / len(acc) if acc else 0.0)
|
|
|
|
ok = {}
|
|
for tag in ('V1', 'V2'):
|
|
ids = [x for x, r in rows.items() if tag in r['sets']]
|
|
base = sum(rows[x]['label'] >= 1 for x in ids)
|
|
print(f'{tag}: {len(ids)} sampled, {base} labelled >=1 (base rate {base / len(ids):.2f})')
|
|
for name, accept in (('old gate', g.old_passes), ('G2', lambda r: g.g2_passes(r, GATE))):
|
|
a, n1, n2, p = score(ids, accept)
|
|
print(f' {name:9} accepted {a:2} label>=1 {n1:2} label2 {n2} P1+ {p:.2f}')
|
|
ok[tag] = score(ids, lambda r: g.g2_passes(r, GATE))
|
|
old_v1 = score([x for x, r in rows.items() if 'V1' in r['sets']], g.old_passes)[0]
|
|
|
|
v3 = [x for x, r in rows.items() if 'V3' in r['sets']]
|
|
by_gate = sum(rows[x]['stop'] == 'gate' for x in v3)
|
|
n1 = sum(rows[x]['label'] >= 1 for x in v3)
|
|
p3 = n1 / len(v3) if v3 else 0.0
|
|
print(f'V3: {len(v3)} grown clusters, stopped by gate {by_gate}, by cap {len(v3) - by_gate}, '
|
|
f'label>=1 {n1}, P1+ {p3:.2f}')
|
|
|
|
c1 = all(ok[t][0] >= 3 and ok[t][3] >= 0.80 for t in ('V1', 'V2'))
|
|
c2 = ok['V1'][0] > old_v1
|
|
c3 = bool(v3) and by_gate > len(v3) / 2 and p3 >= 0.80
|
|
print(f'\ncriterion 1 (P1+ >= 0.80 and >= 3 accepted on V1 and V2): {"PASS" if c1 else "fail"}')
|
|
print(f'criterion 2 (G2 accepts more small clusters than old gate on V1): {"PASS" if c2 else "fail"}')
|
|
print(f'criterion 3 (V3 mostly gate-stopped, P1+ >= 0.80): {"PASS" if c3 else "fail"}')
|
|
print(f'OVERALL: {"PASS" if c1 and c2 and c3 else "fail"}')
|
|
|
|
print('\nper cluster:')
|
|
for x, r in sorted(rows.items()):
|
|
print(f" {x} {','.join(r['sets']):6} size {r['size']:2} label {r['label']} "
|
|
f"old {'Y' if g.old_passes(r) else '-'} G2 {'Y' if g.g2_passes(r, GATE) else '-'} "
|
|
f"stop {r['stop'] or '-':4} {r['name'] or ''}")
|
|
pathlib.Path('validate-result.json').write_text(json.dumps(
|
|
{'criteria': [c1, c2, c3], 'V1': ok['V1'], 'V2': ok['V2'], 'V3': [len(v3), by_gate, n1, p3]}, indent=1))
|