Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
47 lines
1.9 KiB
Python
47 lines
1.9 KiB
Python
"""Dump candidate clusters to blind label sheets (SELECTION.md, protocol steps 1-3).
|
|
|
|
python3 make_sheets.py
|
|
|
|
Writes, for development (seed 11) and test (seed 12):
|
|
candidates-<set>.json cluster id -> member chunk indices (what signals are computed on)
|
|
sheet-<set>.txt per cluster, member summaries only, shuffled; no titles, no scores
|
|
labels-<set>.json an empty template to fill in: {id: {"label": null, "name": ""}}
|
|
|
|
Cluster ids are assigned in a shuffled order so position carries no information.
|
|
"""
|
|
import json, pathlib, random, sys
|
|
|
|
K = 60
|
|
SETS = {'dev': 11, 'test': 12}
|
|
|
|
sys.argv = ['measure.py', 'summary_embeddings.json']
|
|
exec(pathlib.Path('measure.py').read_text(encoding='utf-8').split("print(f'{path.name}")[0])
|
|
|
|
summaries = json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries']
|
|
text = {x['chunk']: x['summary'] for x in summaries}
|
|
|
|
for name, seed in SETS.items():
|
|
cands = []
|
|
for c in kmeans(K, seed):
|
|
size, nst, dom, _ = describe(c)
|
|
if size >= MIN_SIZE and nst >= MIN_STORIES and dom <= MAX_DOMINANT:
|
|
cands.append(sorted(c))
|
|
rnd = random.Random(seed * 1000 + 7)
|
|
rnd.shuffle(cands)
|
|
prefix = name[0].upper()
|
|
ids = {f'{prefix}{i + 1:02d}': c for i, c in enumerate(cands)}
|
|
|
|
pathlib.Path(f'candidates-{name}.json').write_text(json.dumps(ids, indent=1))
|
|
lines = []
|
|
for cid, members in ids.items():
|
|
shown = members[:]
|
|
rnd.shuffle(shown)
|
|
lines.append(f'=== {cid}')
|
|
lines.extend(f' - {text[i]}' for i in shown)
|
|
lines.append('')
|
|
pathlib.Path(f'sheet-{name}.txt').write_text('\n'.join(lines), encoding='utf-8')
|
|
labels = pathlib.Path(f'labels-{name}.json')
|
|
if not labels.exists():
|
|
labels.write_text(json.dumps({cid: {'label': None, 'name': ''} for cid in ids}, indent=1))
|
|
print(f'{name}: seed {seed}, {len(ids)} candidate clusters, {sum(map(len, ids.values()))} scenes')
|