Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
104 lines
4.7 KiB
Python
104 lines
4.7 KiB
Python
"""Do the clusters in an embedding file capture recurring SITUATIONS?
|
|
|
|
python3 measure.py embeddings.json # prose embeddings (the 2026-09-10 baseline)
|
|
python3 measure.py summary_embeddings.json # summarise-then-embed (the fix under test)
|
|
|
|
The 2026-09-10 probe counted distinct stories per cluster and was fooled: a
|
|
cluster reporting 5 stories was routinely 60-79% ONE story plus stragglers.
|
|
So this reports concentration, not just spread.
|
|
|
|
dominant share = largest single-story count / cluster size
|
|
recurring = size >= MIN_SIZE, >= MIN_STORIES distinct stories,
|
|
and dominant share <= MAX_DOMINANT
|
|
|
|
Pass mark: see DESIGN.md, "Assumptions under test". Judged at k = 60 and
|
|
k = 100 only. Results are the median over SEEDS, so one lucky (or unlucky)
|
|
initialisation cannot decide the test -- the 2026-09-10 "2% coverage" was
|
|
seed 11 alone; seeds 12 and 13 give 8-10%.
|
|
|
|
Pure Python -- no numpy on this machine. Only chunk indices that have an
|
|
embedding are used, so a partial (pilot) file is fine.
|
|
"""
|
|
import json, math, pathlib, random, statistics, sys
|
|
|
|
KS = (25, 40, 60, 100)
|
|
SEEDS = (11, 12, 13)
|
|
MIN_SIZE, MIN_STORIES, MAX_DOMINANT = 5, 4, 0.40
|
|
|
|
path = pathlib.Path(sys.argv[1] if len(sys.argv) > 1 else 'embeddings.json')
|
|
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
|
|
raw = json.loads(path.read_text(encoding='utf-8'))
|
|
vecs = raw['vectors'] if isinstance(raw, dict) else raw
|
|
n = min(len(chunks), len(vecs))
|
|
|
|
def norm(v):
|
|
m = math.sqrt(sum(x * x for x in v)) or 1.0
|
|
return [x / m for x in v]
|
|
V = [norm(v) for v in vecs[:n]]
|
|
dim = len(V[0])
|
|
story = [chunks[i]['story'] for i in range(n)]
|
|
|
|
def dot(a, b):
|
|
s = 0.0
|
|
for x, y in zip(a, b): s += x * y
|
|
return s
|
|
|
|
def kmeans(K, seed, iters=15):
|
|
rnd = random.Random(seed)
|
|
cent = [V[i] for i in rnd.sample(range(n), K)]
|
|
assign = [-1] * n
|
|
for _ in range(iters):
|
|
moved = 0
|
|
for i, v in enumerate(V):
|
|
best, bs = 0, -2.0
|
|
for k, c in enumerate(cent):
|
|
s = dot(v, c)
|
|
if s > bs: bs, best = s, k
|
|
if assign[i] != best: assign[i] = best; moved += 1
|
|
if moved == 0: break
|
|
groups = {}
|
|
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
|
|
for k, mem in groups.items():
|
|
cent[k] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)])
|
|
groups = {}
|
|
for i, k in enumerate(assign): groups.setdefault(k, []).append(i)
|
|
return list(groups.values())
|
|
|
|
def describe(cluster):
|
|
counts = {}
|
|
for i in cluster: counts[story[i]] = counts.get(story[i], 0) + 1
|
|
cen = norm([sum(V[i][d] for i in cluster) / len(cluster) for d in range(dim)])
|
|
coh = sum(dot(V[i], cen) for i in cluster) / len(cluster)
|
|
return len(cluster), len(counts), max(counts.values()) / len(cluster), coh
|
|
|
|
# "Tight" is relative to the file's own median cluster coherence. An absolute
|
|
# cutoff (the 2026-09-10 probe used 0.83) does not transfer between embedding
|
|
# files: one-sentence summaries and 300-word scenes sit at different cosines.
|
|
# Without it, loose clusters held together by prose register count as recurring.
|
|
print(f'{path.name}: {n} chunks from {len(set(story))} stories, seeds {SEEDS}\n')
|
|
print(f'{"k":>4} {"sized>=%d" % MIN_SIZE:>9} {"median dom":>11} {"recurring":>10} {"coverage":>9} {"tight rec":>10} {"tight cov":>10}')
|
|
print('-' * 70)
|
|
report = {'file': path.name, 'chunks': n, 'rows': []}
|
|
for K in KS:
|
|
if K * 2 > n: continue
|
|
per_seed = []
|
|
for seed in SEEDS:
|
|
rows = [describe(c) for c in kmeans(K, seed)]
|
|
sized = [r for r in rows if r[0] >= MIN_SIZE]
|
|
rec = [r for r in sized if r[1] >= MIN_STORIES and r[2] <= MAX_DOMINANT]
|
|
cut = statistics.median(r[3] for r in sized) if sized else 1.0
|
|
tight = [r for r in rec if r[3] >= cut]
|
|
per_seed.append((len(sized),
|
|
statistics.median(r[2] for r in sized) if sized else 1.0,
|
|
len(rec), sum(r[0] for r in rec) / n,
|
|
len(tight), sum(r[0] for r in tight) / n))
|
|
med = [statistics.median(s[j] for s in per_seed) for j in range(6)]
|
|
print(f'{K:>4} {med[0]:>9.0f} {med[1]:>11.2f} {med[2]:>10.0f} {med[3]:>8.0%} {med[4]:>10.0f} {med[5]:>9.0%}')
|
|
report['rows'].append({'k': K, 'sized': med[0], 'median_dominant': med[1],
|
|
'recurring': med[2], 'coverage': med[3],
|
|
'tight_recurring': med[4], 'tight_coverage': med[5]})
|
|
out = path.with_name(path.stem + '.measure.json')
|
|
out.write_text(json.dumps(report, indent=1))
|
|
print(f'\nrecurring = size>={MIN_SIZE}, stories>={MIN_STORIES}, dominant share<={MAX_DOMINANT:.0%}; '
|
|
f'tight = coherence at or above this file\'s median cluster. Judge at k>=60. Wrote {out.name}')
|