Files
TheLadder/tools/story-to-pack/probe/sweep.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

66 lines
2.6 KiB
Python

"""How many clusters are BOTH situationally sharp and drawn from many stories?
Sharp = high mean cosine to centroid (a real situation, not prose register)
Spread = drawn from >= 4 different stories (recurring, not a single scene)
A pack needs ~15 per stage that are both.
"""
import json, math, pathlib, random
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
raw = json.loads(pathlib.Path('embeddings.json').read_text(encoding='utf-8'))
n = min(len(chunks), len(raw))
def norm(v):
m = math.sqrt(sum(x * x for x in v)) or 1.0
return [x / m for x in v]
V = [norm(v) for v in raw[:n]]
dim = len(V[0])
story = [chunks[i]['story'] for i in range(n)]
def dot(a, b):
s = 0.0
for x, y in zip(a, b): s += x * y
return s
def kmeans(K, seed=11, iters=12):
random.seed(seed)
cent = [V[i] for i in random.sample(range(n), K)]
assign = [-1] * n
for _ in range(iters):
moved = 0
for i, v in enumerate(V):
best, bs = 0, -2.0
for k, c in enumerate(cent):
s = dot(v, c)
if s > bs: bs, best = s, k
if assign[i] != best: assign[i] = best; moved += 1
if moved == 0: break
groups = {}
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
for k in range(K):
mem = groups.get(k)
if not mem: continue
cent[k] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)])
groups = {}
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
out = []
for k, idx in groups.items():
coh = sum(dot(V[i], cent[k]) for i in idx) / len(idx)
out.append({'size': len(idx), 'stories': len({story[i] for i in idx}), 'coh': coh, 'idx': idx})
return out
print(f'{"k":>4} {"clusters":>9} {"sharp>=.83":>11} {"spread>=4":>10} {"BOTH":>6} {"median size":>12}')
print('-' * 60)
best = None
for K in (25, 40, 60, 80, 100):
cl = kmeans(K)
sharp = [c for c in cl if c['coh'] >= 0.83]
spread = [c for c in cl if c['stories'] >= 4]
both = [c for c in cl if c['coh'] >= 0.83 and c['stories'] >= 4]
sizes = sorted(c['size'] for c in cl)
print(f'{K:>4} {len(cl):>9} {len(sharp):>11} {len(spread):>10} {len(both):>6} {sizes[len(sizes)//2]:>12}')
if best is None or len(both) > len(best[1]): best = (K, both, cl)
pathlib.Path('best_k.json').write_text(json.dumps({'k': best[0],
'both': [{'size': c['size'], 'stories': c['stories'], 'coh': c['coh'], 'idx': c['idx']} for c in best[1]]}))
print(f'\nbest k = {best[0]} with {len(best[1])} clusters that are both sharp and recurring')