Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
98 lines
4.8 KiB
Python
98 lines
4.8 KiB
Python
"""Compute the pre-registered selection signals S1-S6 (SELECTION.md) for a candidates file.
|
||
|
||
python3 signals.py candidates-dev.json # writes signals-dev.json
|
||
python3 signals.py candidates-test.json # writes signals-test.json
|
||
|
||
No inference: summary vectors, prose vectors, summary text and story ids only.
|
||
S5 needs 12 extra k-means runs; they are computed once and cached in coassoc.json.
|
||
"""
|
||
import collections, json, math, pathlib, random, re, statistics, sys
|
||
|
||
cand_path = pathlib.Path(sys.argv[1])
|
||
set_name = cand_path.stem.replace('candidates-', '')
|
||
PARTITION_SEED = {'dev': 11, 'test': 12}[set_name]
|
||
PARTITION_K = 60
|
||
|
||
sys.argv = ['measure.py', 'summary_embeddings.json']
|
||
exec(pathlib.Path('measure.py').read_text(encoding='utf-8').split("print(f'{path.name}")[0])
|
||
# now defined: chunks, V (unit summary vectors), n, dim, story, dot, norm, kmeans, describe
|
||
|
||
prose_raw = json.loads(pathlib.Path('embeddings.json').read_text())
|
||
P = [norm(v) for v in (prose_raw['vectors'] if isinstance(prose_raw, dict) else prose_raw)[:n]]
|
||
summaries = json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries']
|
||
text = {x['chunk']: x['summary'] for x in summaries}
|
||
candidates = json.loads(cand_path.read_text())
|
||
|
||
# Sanity: the candidates must be exactly this partition's recurring clusters.
|
||
partition = [sorted(c) for c in kmeans(PARTITION_K, PARTITION_SEED)]
|
||
assert set(map(tuple, candidates.values())) <= set(map(tuple, partition)), 'candidates do not match the partition'
|
||
sized_coh = [describe(c)[3] for c in partition if len(c) >= MIN_SIZE]
|
||
tight_cut = statistics.median(sized_coh)
|
||
|
||
# S5: co-association across runs that are neither the dev nor the test partition.
|
||
CO_RUNS = [(k, s) for k in (50, 60, 70, 80, 90, 100) for s in (21, 22)]
|
||
co_path = pathlib.Path('coassoc.json')
|
||
if co_path.exists():
|
||
co = json.loads(co_path.read_text())
|
||
else:
|
||
co = [[0] * n for _ in range(n)]
|
||
for k, s in CO_RUNS:
|
||
for g in kmeans(k, s):
|
||
for a in g:
|
||
row = co[a]
|
||
for b in g: row[b] += 1
|
||
print(f' co-association run k={k} seed={s}', flush=True)
|
||
co_path.write_text(json.dumps(co))
|
||
|
||
# S4 baseline: mean prose cosine over random cross-story pairs in the whole corpus.
|
||
rnd = random.Random(5)
|
||
base = []
|
||
while len(base) < 20000:
|
||
a, b = rnd.randrange(n), rnd.randrange(n)
|
||
if story[a] != story[b]: base.append(dot(P[a], P[b]))
|
||
prose_base = sum(base) / len(base)
|
||
|
||
STOP = set("""a an the and or but of to in on at for with by from as into over under about after before
|
||
while during than then that this these those it its his her hers him he she they them their theirs we our
|
||
you your i me my is are was were be been being has have had do does did not no nor so such who whom whose
|
||
which what when where why how all any each both either neither one two some other others another more most
|
||
less least very just only also even yet still now up down out off away back again once own same too can
|
||
could will would shall should may might must upon despite toward towards between among amid amidst
|
||
someone someone's someone’s""".split())
|
||
docs = [set(re.findall(r"[a-z][a-z'’]+", text[i].lower())) - STOP for i in range(n)]
|
||
df = collections.Counter(w for d in docs for w in d)
|
||
|
||
def signals(members):
|
||
size, nst, dom, coh = describe(members)
|
||
cross, within, pcross, coa = [], [], [], []
|
||
for x in range(len(members)):
|
||
for y in range(x + 1, len(members)):
|
||
a, b = members[x], members[y]
|
||
coa.append(co[a][b] / len(CO_RUNS))
|
||
if story[a] == story[b]:
|
||
within.append(dot(V[a], V[b]))
|
||
else:
|
||
cross.append(dot(V[a], V[b])); pcross.append(dot(P[a], P[b]))
|
||
s1 = sum(cross) / len(cross)
|
||
s2 = s1 / (sum(within) / len(within)) if within else 1.0
|
||
counts = collections.Counter(story[i] for i in members)
|
||
s3 = (1 / sum((c / size) ** 2 for c in counts.values())) / size
|
||
s4 = sum(pcross) / len(pcross) - prose_base
|
||
s5 = sum(coa) / len(coa)
|
||
cw = collections.Counter(w for i in members for w in docs[i])
|
||
best = None
|
||
for w, c in cw.items():
|
||
if c < 2: continue
|
||
score = math.log((c + .5) / (size - c + .5)) - math.log((df[w] + .5) / (n - df[w] + .5))
|
||
if best is None or score > best[0]: best = (score, w, c)
|
||
s6, word = (best[2] / size, best[1]) if best else (0.0, '')
|
||
return {'size': size, 'stories': nst, 'dominant': round(dom, 4), 'coherence': round(coh, 4),
|
||
'B0_tight': coh >= tight_cut,
|
||
'S1': round(s1, 4), 'S2': round(s2, 4), 'S3': round(s3, 4), 'S4': round(s4, 4),
|
||
'S5': round(s5, 4), 'S6': round(s6, 4), 'top_word': word}
|
||
|
||
out = {cid: signals(members) for cid, members in candidates.items()}
|
||
out_path = pathlib.Path(f'signals-{set_name}.json')
|
||
out_path.write_text(json.dumps(out, indent=1))
|
||
print(f'{set_name}: {len(out)} clusters; prose cross-story baseline {prose_base:.4f}; wrote {out_path}')
|