"""How many clusters are BOTH situationally sharp and drawn from many stories? Sharp = high mean cosine to centroid (a real situation, not prose register) Spread = drawn from >= 4 different stories (recurring, not a single scene) A pack needs ~15 per stage that are both. """ import json, math, pathlib, random chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8')) raw = json.loads(pathlib.Path('embeddings.json').read_text(encoding='utf-8')) n = min(len(chunks), len(raw)) def norm(v): m = math.sqrt(sum(x * x for x in v)) or 1.0 return [x / m for x in v] V = [norm(v) for v in raw[:n]] dim = len(V[0]) story = [chunks[i]['story'] for i in range(n)] def dot(a, b): s = 0.0 for x, y in zip(a, b): s += x * y return s def kmeans(K, seed=11, iters=12): random.seed(seed) cent = [V[i] for i in random.sample(range(n), K)] assign = [-1] * n for _ in range(iters): moved = 0 for i, v in enumerate(V): best, bs = 0, -2.0 for k, c in enumerate(cent): s = dot(v, c) if s > bs: bs, best = s, k if assign[i] != best: assign[i] = best; moved += 1 if moved == 0: break groups = {} for i, k in enumerate(assign): groups.setdefault(k, []).append(i) for k in range(K): mem = groups.get(k) if not mem: continue cent[k] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)]) groups = {} for i, k in enumerate(assign): groups.setdefault(k, []).append(i) out = [] for k, idx in groups.items(): coh = sum(dot(V[i], cent[k]) for i in idx) / len(idx) out.append({'size': len(idx), 'stories': len({story[i] for i in idx}), 'coh': coh, 'idx': idx}) return out print(f'{"k":>4} {"clusters":>9} {"sharp>=.83":>11} {"spread>=4":>10} {"BOTH":>6} {"median size":>12}') print('-' * 60) best = None for K in (25, 40, 60, 80, 100): cl = kmeans(K) sharp = [c for c in cl if c['coh'] >= 0.83] spread = [c for c in cl if c['stories'] >= 4] both = [c for c in cl if c['coh'] >= 0.83 and c['stories'] >= 4] sizes = sorted(c['size'] for c in cl) print(f'{K:>4} {len(cl):>9} {len(sharp):>11} {len(spread):>10} {len(both):>6} {sizes[len(sizes)//2]:>12}') if best is None or len(both) > len(best[1]): best = (K, both, cl) pathlib.Path('best_k.json').write_text(json.dumps({'k': best[0], 'both': [{'size': c['size'], 'stories': c['stories'], 'coh': c['coh'], 'idx': c['idx']} for c in best[1]]})) print(f'\nbest k = {best[0]} with {len(best[1])} clusters that are both sharp and recurring')