"""Dump candidate clusters to blind label sheets (SELECTION.md, protocol steps 1-3). python3 make_sheets.py Writes, for development (seed 11) and test (seed 12): candidates-.json cluster id -> member chunk indices (what signals are computed on) sheet-.txt per cluster, member summaries only, shuffled; no titles, no scores labels-.json an empty template to fill in: {id: {"label": null, "name": ""}} Cluster ids are assigned in a shuffled order so position carries no information. """ import json, pathlib, random, sys K = 60 SETS = {'dev': 11, 'test': 12} sys.argv = ['measure.py', 'summary_embeddings.json'] exec(pathlib.Path('measure.py').read_text(encoding='utf-8').split("print(f'{path.name}")[0]) summaries = json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries'] text = {x['chunk']: x['summary'] for x in summaries} for name, seed in SETS.items(): cands = [] for c in kmeans(K, seed): size, nst, dom, _ = describe(c) if size >= MIN_SIZE and nst >= MIN_STORIES and dom <= MAX_DOMINANT: cands.append(sorted(c)) rnd = random.Random(seed * 1000 + 7) rnd.shuffle(cands) prefix = name[0].upper() ids = {f'{prefix}{i + 1:02d}': c for i, c in enumerate(cands)} pathlib.Path(f'candidates-{name}.json').write_text(json.dumps(ids, indent=1)) lines = [] for cid, members in ids.items(): shown = members[:] rnd.shuffle(shown) lines.append(f'=== {cid}') lines.extend(f' - {text[i]}' for i in shown) lines.append('') pathlib.Path(f'sheet-{name}.txt').write_text('\n'.join(lines), encoding='utf-8') labels = pathlib.Path(f'labels-{name}.json') if not labels.exists(): labels.write_text(json.dumps({cid: {'label': None, 'name': ''} for cid in ids}, indent=1)) print(f'{name}: seed {seed}, {len(ids)} candidate clusters, {sum(map(len, ids.values()))} scenes')