Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
66 lines
2.6 KiB
Python
66 lines
2.6 KiB
Python
"""How many clusters are BOTH situationally sharp and drawn from many stories?
|
|
|
|
Sharp = high mean cosine to centroid (a real situation, not prose register)
|
|
Spread = drawn from >= 4 different stories (recurring, not a single scene)
|
|
A pack needs ~15 per stage that are both.
|
|
"""
|
|
import json, math, pathlib, random
|
|
|
|
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
|
|
raw = json.loads(pathlib.Path('embeddings.json').read_text(encoding='utf-8'))
|
|
n = min(len(chunks), len(raw))
|
|
|
|
def norm(v):
|
|
m = math.sqrt(sum(x * x for x in v)) or 1.0
|
|
return [x / m for x in v]
|
|
V = [norm(v) for v in raw[:n]]
|
|
dim = len(V[0])
|
|
story = [chunks[i]['story'] for i in range(n)]
|
|
|
|
def dot(a, b):
|
|
s = 0.0
|
|
for x, y in zip(a, b): s += x * y
|
|
return s
|
|
|
|
def kmeans(K, seed=11, iters=12):
|
|
random.seed(seed)
|
|
cent = [V[i] for i in random.sample(range(n), K)]
|
|
assign = [-1] * n
|
|
for _ in range(iters):
|
|
moved = 0
|
|
for i, v in enumerate(V):
|
|
best, bs = 0, -2.0
|
|
for k, c in enumerate(cent):
|
|
s = dot(v, c)
|
|
if s > bs: bs, best = s, k
|
|
if assign[i] != best: assign[i] = best; moved += 1
|
|
if moved == 0: break
|
|
groups = {}
|
|
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
|
|
for k in range(K):
|
|
mem = groups.get(k)
|
|
if not mem: continue
|
|
cent[k] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)])
|
|
groups = {}
|
|
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
|
|
out = []
|
|
for k, idx in groups.items():
|
|
coh = sum(dot(V[i], cent[k]) for i in idx) / len(idx)
|
|
out.append({'size': len(idx), 'stories': len({story[i] for i in idx}), 'coh': coh, 'idx': idx})
|
|
return out
|
|
|
|
print(f'{"k":>4} {"clusters":>9} {"sharp>=.83":>11} {"spread>=4":>10} {"BOTH":>6} {"median size":>12}')
|
|
print('-' * 60)
|
|
best = None
|
|
for K in (25, 40, 60, 80, 100):
|
|
cl = kmeans(K)
|
|
sharp = [c for c in cl if c['coh'] >= 0.83]
|
|
spread = [c for c in cl if c['stories'] >= 4]
|
|
both = [c for c in cl if c['coh'] >= 0.83 and c['stories'] >= 4]
|
|
sizes = sorted(c['size'] for c in cl)
|
|
print(f'{K:>4} {len(cl):>9} {len(sharp):>11} {len(spread):>10} {len(both):>6} {sizes[len(sizes)//2]:>12}')
|
|
if best is None or len(both) > len(best[1]): best = (K, both, cl)
|
|
pathlib.Path('best_k.json').write_text(json.dumps({'k': best[0],
|
|
'both': [{'size': c['size'], 'stories': c['stories'], 'coh': c['coh'], 'idx': c['idx']} for c in best[1]]}))
|
|
print(f'\nbest k = {best[0]} with {len(best[1])} clusters that are both sharp and recurring')
|