Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
183 lines
7.1 KiB
Python
183 lines
7.1 KiB
Python
"""G2, the size-invariant gate: signals, the old gate for comparison, and k-means. No inference.
|
||
|
||
Import it from the probe directory. On first import it builds and caches pairwise similarities
|
||
(sim.pkl) and the same-size random baseline for Z1 (null.pkl).
|
||
|
||
S1 mean summary cosine over cross-story member pairs
|
||
Z1 S1 as a z-score against random member sets of the same size
|
||
MMIN lowest member-level mean cosine to members of other stories
|
||
W largest share among words whose sharing is significant (hypergeometric, Bonferroni 0.01)
|
||
"""
|
||
import collections, json, math, pathlib, pickle, random, re
|
||
|
||
HERE = pathlib.Path(__file__).resolve().parent
|
||
MIN_SIZE, MIN_STORIES, MAX_DOMINANT = 5, 4, 0.40
|
||
|
||
chunks = json.loads((HERE / 'chunks.json').read_text(encoding='utf-8'))
|
||
_raw = json.loads((HERE / 'summary_embeddings.json').read_text())
|
||
_vecs = _raw['vectors'] if isinstance(_raw, dict) else _raw
|
||
n = min(len(chunks), len(_vecs))
|
||
|
||
def norm(v):
|
||
m = math.sqrt(sum(x * x for x in v)) or 1.0
|
||
return [x / m for x in v]
|
||
|
||
V = [norm(v) for v in _vecs[:n]]
|
||
dim = len(V[0])
|
||
story = [chunks[i]['story'] for i in range(n)]
|
||
text = {x['chunk']: x['summary'] for x in
|
||
json.loads((HERE / 'summaries_clean.json').read_text(encoding='utf-8'))['summaries']}
|
||
|
||
def dot(a, b):
|
||
s = 0.0
|
||
for x, y in zip(a, b): s += x * y
|
||
return s
|
||
|
||
_sim_path = HERE / 'sim.pkl'
|
||
if _sim_path.exists():
|
||
SIM = pickle.loads(_sim_path.read_bytes())
|
||
else:
|
||
SIM = [[0.0] * n for _ in range(n)]
|
||
for a in range(n):
|
||
va, row = V[a], SIM[a]
|
||
for b in range(a, n):
|
||
s = dot(va, V[b])
|
||
row[b] = s
|
||
SIM[b][a] = s
|
||
_sim_path.write_bytes(pickle.dumps(SIM))
|
||
|
||
# Identical to signals.py.
|
||
STOP = set("""a an the and or but of to in on at for with by from as into over under about after before
|
||
while during than then that this these those it its his her hers him he she they them their theirs we our
|
||
you your i me my is are was were be been being has have had do does did not no nor so such who whom whose
|
||
which what when where why how all any each both either neither one two some other others another more most
|
||
less least very just only also even yet still now up down out off away back again once own same too can
|
||
could will would shall should may might must upon despite toward towards between among amid amidst
|
||
someone someone's someone’s""".split())
|
||
docs = [set(re.findall(r"[a-z][a-z'’]+", text[i].lower())) - STOP for i in range(n)]
|
||
df = collections.Counter(w for d in docs for w in d)
|
||
ALPHA = 0.01 / len(df)
|
||
|
||
def describe(m):
|
||
counts = collections.Counter(story[i] for i in m)
|
||
return len(m), len(counts), max(counts.values()) / len(m)
|
||
|
||
def eligible(m):
|
||
size, nst, dom = describe(m)
|
||
return size >= MIN_SIZE and nst >= MIN_STORIES and dom <= MAX_DOMINANT
|
||
|
||
def cohesion(m):
|
||
"""(S1, MMIN)."""
|
||
tot, cnt, lowest = 0.0, 0, None
|
||
for a in m:
|
||
s, c, row = 0.0, 0, SIM[a]
|
||
for b in m:
|
||
if story[a] != story[b]:
|
||
s += row[b]; c += 1
|
||
if c:
|
||
tot += s; cnt += c
|
||
mean = s / c
|
||
lowest = mean if lowest is None else min(lowest, mean)
|
||
return (tot / cnt if cnt else 0.0), (lowest if lowest is not None else 0.0)
|
||
|
||
# Same-size random baseline for Z1.
|
||
_null_path = HERE / 'null.pkl'
|
||
if _null_path.exists():
|
||
NULL = pickle.loads(_null_path.read_bytes())
|
||
else:
|
||
rnd, NULL = random.Random(2026), {}
|
||
for m in range(5, 151):
|
||
samples, need = [], 300 if m <= 40 else 120
|
||
while len(samples) < need:
|
||
members = rnd.sample(range(n), m)
|
||
if eligible(members):
|
||
samples.append(cohesion(members)[0])
|
||
mu = sum(samples) / len(samples)
|
||
sd = math.sqrt(sum((x - mu) ** 2 for x in samples) / (len(samples) - 1))
|
||
NULL[m] = (mu, sd)
|
||
_null_path.write_bytes(pickle.dumps(NULL))
|
||
|
||
def _lchoose(a, b):
|
||
return math.lgamma(a + 1) - math.lgamma(b + 1) - math.lgamma(a - b + 1)
|
||
|
||
def hyper_sf(c, m, k, total):
|
||
"""P(X >= c) for X ~ Hypergeometric(total, k successes, m draws)."""
|
||
top = min(m, k)
|
||
if c > top: return 0.0
|
||
logs = [_lchoose(k, x) + _lchoose(total - k, m - x) - _lchoose(total, m)
|
||
for x in range(c, top + 1) if m - x <= total - k]
|
||
if not logs: return 0.0
|
||
mx = max(logs)
|
||
return math.exp(mx) * sum(math.exp(l - mx) for l in logs)
|
||
|
||
def word_dominance(m):
|
||
"""(W, word): largest member share among significantly shared words."""
|
||
size = len(m)
|
||
cw = collections.Counter(w for i in m for w in docs[i])
|
||
best = (0.0, '')
|
||
for w, c in cw.items():
|
||
if c < 2: continue
|
||
if hyper_sf(c, size, df[w], n) <= ALPHA and c / size > best[0]:
|
||
best = (c / size, w)
|
||
return best
|
||
|
||
def old_s6(m):
|
||
"""The first gate's S6, identical to signals.py."""
|
||
size = len(m)
|
||
cw = collections.Counter(w for i in m for w in docs[i])
|
||
best = None
|
||
for w, c in cw.items():
|
||
if c < 2: continue
|
||
score = math.log((c + .5) / (size - c + .5)) - math.log((df[w] + .5) / (n - df[w] + .5))
|
||
if best is None or score > best[0]: best = (score, c)
|
||
return best[1] / size if best else 0.0
|
||
|
||
def signals(m):
|
||
size, nst, dom = describe(m)
|
||
s1, mmin = cohesion(m)
|
||
mu, sd = NULL[min(max(size, 5), 150)]
|
||
w, word = word_dominance(m)
|
||
return {'size': size, 'stories': nst, 'dominant': dom, 'S1': s1, 'Z1': (s1 - mu) / sd,
|
||
'MMIN': mmin, 'W': w, 'W_word': word, 'S6_old': old_s6(m)}
|
||
|
||
OLD_GATE = json.loads((HERE / 'rule-frozen.json').read_text())
|
||
_OLD = {s: t for s, _, t in OLD_GATE['terms']}
|
||
|
||
def old_passes(sig):
|
||
return sig['size'] >= MIN_SIZE and sig['stories'] >= MIN_STORIES and sig['dominant'] <= MAX_DOMINANT \
|
||
and sig['S1'] >= _OLD['S1'] and sig['S6_old'] <= _OLD['S6']
|
||
|
||
def g2_passes(sig, gate):
|
||
if not (sig['size'] >= MIN_SIZE and sig['stories'] >= MIN_STORIES and sig['dominant'] <= MAX_DOMINANT):
|
||
return False
|
||
if gate.get('S1') is not None and sig['S1'] < gate['S1']: return False
|
||
if gate.get('Z1') is not None and sig['Z1'] < gate['Z1']: return False
|
||
if gate.get('MMIN') is not None and sig['MMIN'] < gate['MMIN']: return False
|
||
if gate.get('W') is not None and sig['W'] >= gate['W']: return False
|
||
return True
|
||
|
||
def kmeans(k, seed, iters=15):
|
||
"""Identical to measure.py's k-means over the summary vectors."""
|
||
rnd = random.Random(seed)
|
||
cent = [V[i] for i in rnd.sample(range(n), k)]
|
||
assign = [-1] * n
|
||
for _ in range(iters):
|
||
moved = 0
|
||
for i, v in enumerate(V):
|
||
best, bs = 0, -2.0
|
||
for j, c in enumerate(cent):
|
||
s = dot(v, c)
|
||
if s > bs: bs, best = s, j
|
||
if assign[i] != best: assign[i] = best; moved += 1
|
||
if moved == 0: break
|
||
groups = collections.defaultdict(list)
|
||
for i, j in enumerate(assign): groups[j].append(i)
|
||
for j, mem in groups.items():
|
||
cent[j] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)])
|
||
groups = collections.defaultdict(list)
|
||
for i, j in enumerate(assign): groups[j].append(i)
|
||
return [sorted(g) for g in groups.values()]
|
||
|
||
def centroid(m):
|
||
return norm([sum(V[i][d] for i in m) / len(m) for d in range(dim)])
|