Files
TheLadder/tools/story-to-pack/probe/signals.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

98 lines
4.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Compute the pre-registered selection signals S1-S6 (SELECTION.md) for a candidates file.
python3 signals.py candidates-dev.json # writes signals-dev.json
python3 signals.py candidates-test.json # writes signals-test.json
No inference: summary vectors, prose vectors, summary text and story ids only.
S5 needs 12 extra k-means runs; they are computed once and cached in coassoc.json.
"""
import collections, json, math, pathlib, random, re, statistics, sys
cand_path = pathlib.Path(sys.argv[1])
set_name = cand_path.stem.replace('candidates-', '')
PARTITION_SEED = {'dev': 11, 'test': 12}[set_name]
PARTITION_K = 60
sys.argv = ['measure.py', 'summary_embeddings.json']
exec(pathlib.Path('measure.py').read_text(encoding='utf-8').split("print(f'{path.name}")[0])
# now defined: chunks, V (unit summary vectors), n, dim, story, dot, norm, kmeans, describe
prose_raw = json.loads(pathlib.Path('embeddings.json').read_text())
P = [norm(v) for v in (prose_raw['vectors'] if isinstance(prose_raw, dict) else prose_raw)[:n]]
summaries = json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries']
text = {x['chunk']: x['summary'] for x in summaries}
candidates = json.loads(cand_path.read_text())
# Sanity: the candidates must be exactly this partition's recurring clusters.
partition = [sorted(c) for c in kmeans(PARTITION_K, PARTITION_SEED)]
assert set(map(tuple, candidates.values())) <= set(map(tuple, partition)), 'candidates do not match the partition'
sized_coh = [describe(c)[3] for c in partition if len(c) >= MIN_SIZE]
tight_cut = statistics.median(sized_coh)
# S5: co-association across runs that are neither the dev nor the test partition.
CO_RUNS = [(k, s) for k in (50, 60, 70, 80, 90, 100) for s in (21, 22)]
co_path = pathlib.Path('coassoc.json')
if co_path.exists():
co = json.loads(co_path.read_text())
else:
co = [[0] * n for _ in range(n)]
for k, s in CO_RUNS:
for g in kmeans(k, s):
for a in g:
row = co[a]
for b in g: row[b] += 1
print(f' co-association run k={k} seed={s}', flush=True)
co_path.write_text(json.dumps(co))
# S4 baseline: mean prose cosine over random cross-story pairs in the whole corpus.
rnd = random.Random(5)
base = []
while len(base) < 20000:
a, b = rnd.randrange(n), rnd.randrange(n)
if story[a] != story[b]: base.append(dot(P[a], P[b]))
prose_base = sum(base) / len(base)
STOP = set("""a an the and or but of to in on at for with by from as into over under about after before
while during than then that this these those it its his her hers him he she they them their theirs we our
you your i me my is are was were be been being has have had do does did not no nor so such who whom whose
which what when where why how all any each both either neither one two some other others another more most
less least very just only also even yet still now up down out off away back again once own same too can
could will would shall should may might must upon despite toward towards between among amid amidst
someone someone's someone’s""".split())
docs = [set(re.findall(r"[a-z][a-z'’]+", text[i].lower())) - STOP for i in range(n)]
df = collections.Counter(w for d in docs for w in d)
def signals(members):
size, nst, dom, coh = describe(members)
cross, within, pcross, coa = [], [], [], []
for x in range(len(members)):
for y in range(x + 1, len(members)):
a, b = members[x], members[y]
coa.append(co[a][b] / len(CO_RUNS))
if story[a] == story[b]:
within.append(dot(V[a], V[b]))
else:
cross.append(dot(V[a], V[b])); pcross.append(dot(P[a], P[b]))
s1 = sum(cross) / len(cross)
s2 = s1 / (sum(within) / len(within)) if within else 1.0
counts = collections.Counter(story[i] for i in members)
s3 = (1 / sum((c / size) ** 2 for c in counts.values())) / size
s4 = sum(pcross) / len(pcross) - prose_base
s5 = sum(coa) / len(coa)
cw = collections.Counter(w for i in members for w in docs[i])
best = None
for w, c in cw.items():
if c < 2: continue
score = math.log((c + .5) / (size - c + .5)) - math.log((df[w] + .5) / (n - df[w] + .5))
if best is None or score > best[0]: best = (score, w, c)
s6, word = (best[2] / size, best[1]) if best else (0.0, '')
return {'size': size, 'stories': nst, 'dominant': round(dom, 4), 'coherence': round(coh, 4),
'B0_tight': coh >= tight_cut,
'S1': round(s1, 4), 'S2': round(s2, 4), 'S3': round(s3, 4), 'S4': round(s4, 4),
'S5': round(s5, 4), 'S6': round(s6, 4), 'top_word': word}
out = {cid: signals(members) for cid, members in candidates.items()}
out_path = pathlib.Path(f'signals-{set_name}.json')
out_path.write_text(json.dumps(out, indent=1))
print(f'{set_name}: {len(out)} clusters; prose cross-story baseline {prose_base:.4f}; wrote {out_path}')