Files
TheLadder/tools/story-to-pack/probe/measure.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

104 lines
4.7 KiB
Python

"""Do the clusters in an embedding file capture recurring SITUATIONS?
python3 measure.py embeddings.json # prose embeddings (the 2026-09-10 baseline)
python3 measure.py summary_embeddings.json # summarise-then-embed (the fix under test)
The 2026-09-10 probe counted distinct stories per cluster and was fooled: a
cluster reporting 5 stories was routinely 60-79% ONE story plus stragglers.
So this reports concentration, not just spread.
dominant share = largest single-story count / cluster size
recurring = size >= MIN_SIZE, >= MIN_STORIES distinct stories,
and dominant share <= MAX_DOMINANT
Pass mark: see DESIGN.md, "Assumptions under test". Judged at k = 60 and
k = 100 only. Results are the median over SEEDS, so one lucky (or unlucky)
initialisation cannot decide the test -- the 2026-09-10 "2% coverage" was
seed 11 alone; seeds 12 and 13 give 8-10%.
Pure Python -- no numpy on this machine. Only chunk indices that have an
embedding are used, so a partial (pilot) file is fine.
"""
import json, math, pathlib, random, statistics, sys
KS = (25, 40, 60, 100)
SEEDS = (11, 12, 13)
MIN_SIZE, MIN_STORIES, MAX_DOMINANT = 5, 4, 0.40
path = pathlib.Path(sys.argv[1] if len(sys.argv) > 1 else 'embeddings.json')
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
raw = json.loads(path.read_text(encoding='utf-8'))
vecs = raw['vectors'] if isinstance(raw, dict) else raw
n = min(len(chunks), len(vecs))
def norm(v):
m = math.sqrt(sum(x * x for x in v)) or 1.0
return [x / m for x in v]
V = [norm(v) for v in vecs[:n]]
dim = len(V[0])
story = [chunks[i]['story'] for i in range(n)]
def dot(a, b):
s = 0.0
for x, y in zip(a, b): s += x * y
return s
def kmeans(K, seed, iters=15):
rnd = random.Random(seed)
cent = [V[i] for i in rnd.sample(range(n), K)]
assign = [-1] * n
for _ in range(iters):
moved = 0
for i, v in enumerate(V):
best, bs = 0, -2.0
for k, c in enumerate(cent):
s = dot(v, c)
if s > bs: bs, best = s, k
if assign[i] != best: assign[i] = best; moved += 1
if moved == 0: break
groups = {}
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
for k, mem in groups.items():
cent[k] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)])
groups = {}
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
return list(groups.values())
def describe(cluster):
counts = {}
for i in cluster: counts[story[i]] = counts.get(story[i], 0) + 1
cen = norm([sum(V[i][d] for i in cluster) / len(cluster) for d in range(dim)])
coh = sum(dot(V[i], cen) for i in cluster) / len(cluster)
return len(cluster), len(counts), max(counts.values()) / len(cluster), coh
# "Tight" is relative to the file's own median cluster coherence. An absolute
# cutoff (the 2026-09-10 probe used 0.83) does not transfer between embedding
# files: one-sentence summaries and 300-word scenes sit at different cosines.
# Without it, loose clusters held together by prose register count as recurring.
print(f'{path.name}: {n} chunks from {len(set(story))} stories, seeds {SEEDS}\n')
print(f'{"k":>4} {"sized>=%d" % MIN_SIZE:>9} {"median dom":>11} {"recurring":>10} {"coverage":>9} {"tight rec":>10} {"tight cov":>10}')
print('-' * 70)
report = {'file': path.name, 'chunks': n, 'rows': []}
for K in KS:
if K * 2 > n: continue
per_seed = []
for seed in SEEDS:
rows = [describe(c) for c in kmeans(K, seed)]
sized = [r for r in rows if r[0] >= MIN_SIZE]
rec = [r for r in sized if r[1] >= MIN_STORIES and r[2] <= MAX_DOMINANT]
cut = statistics.median(r[3] for r in sized) if sized else 1.0
tight = [r for r in rec if r[3] >= cut]
per_seed.append((len(sized),
statistics.median(r[2] for r in sized) if sized else 1.0,
len(rec), sum(r[0] for r in rec) / n,
len(tight), sum(r[0] for r in tight) / n))
med = [statistics.median(s[j] for s in per_seed) for j in range(6)]
print(f'{K:>4} {med[0]:>9.0f} {med[1]:>11.2f} {med[2]:>10.0f} {med[3]:>8.0%} {med[4]:>10.0f} {med[5]:>9.0%}')
report['rows'].append({'k': K, 'sized': med[0], 'median_dominant': med[1],
'recurring': med[2], 'coverage': med[3],
'tight_recurring': med[4], 'tight_coverage': med[5]})
out = path.with_name(path.stem + '.measure.json')
out.write_text(json.dumps(report, indent=1))
print(f'\nrecurring = size>={MIN_SIZE}, stories>={MIN_STORIES}, dominant share<={MAX_DOMINANT:.0%}; '
f'tight = coherence at or above this file\'s median cluster. Judge at k>=60. Wrote {out.name}')