Files
TheLadder/tools/story-to-pack/probe/gate2.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

183 lines
7.1 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""G2, the size-invariant gate: signals, the old gate for comparison, and k-means. No inference.
Import it from the probe directory. On first import it builds and caches pairwise similarities
(sim.pkl) and the same-size random baseline for Z1 (null.pkl).
S1 mean summary cosine over cross-story member pairs
Z1 S1 as a z-score against random member sets of the same size
MMIN lowest member-level mean cosine to members of other stories
W largest share among words whose sharing is significant (hypergeometric, Bonferroni 0.01)
"""
import collections, json, math, pathlib, pickle, random, re
HERE = pathlib.Path(__file__).resolve().parent
MIN_SIZE, MIN_STORIES, MAX_DOMINANT = 5, 4, 0.40
chunks = json.loads((HERE / 'chunks.json').read_text(encoding='utf-8'))
_raw = json.loads((HERE / 'summary_embeddings.json').read_text())
_vecs = _raw['vectors'] if isinstance(_raw, dict) else _raw
n = min(len(chunks), len(_vecs))
def norm(v):
m = math.sqrt(sum(x * x for x in v)) or 1.0
return [x / m for x in v]
V = [norm(v) for v in _vecs[:n]]
dim = len(V[0])
story = [chunks[i]['story'] for i in range(n)]
text = {x['chunk']: x['summary'] for x in
json.loads((HERE / 'summaries_clean.json').read_text(encoding='utf-8'))['summaries']}
def dot(a, b):
s = 0.0
for x, y in zip(a, b): s += x * y
return s
_sim_path = HERE / 'sim.pkl'
if _sim_path.exists():
SIM = pickle.loads(_sim_path.read_bytes())
else:
SIM = [[0.0] * n for _ in range(n)]
for a in range(n):
va, row = V[a], SIM[a]
for b in range(a, n):
s = dot(va, V[b])
row[b] = s
SIM[b][a] = s
_sim_path.write_bytes(pickle.dumps(SIM))
# Identical to signals.py.
STOP = set("""a an the and or but of to in on at for with by from as into over under about after before
while during than then that this these those it its his her hers him he she they them their theirs we our
you your i me my is are was were be been being has have had do does did not no nor so such who whom whose
which what when where why how all any each both either neither one two some other others another more most
less least very just only also even yet still now up down out off away back again once own same too can
could will would shall should may might must upon despite toward towards between among amid amidst
someone someone's someone’s""".split())
docs = [set(re.findall(r"[a-z][a-z'’]+", text[i].lower())) - STOP for i in range(n)]
df = collections.Counter(w for d in docs for w in d)
ALPHA = 0.01 / len(df)
def describe(m):
counts = collections.Counter(story[i] for i in m)
return len(m), len(counts), max(counts.values()) / len(m)
def eligible(m):
size, nst, dom = describe(m)
return size >= MIN_SIZE and nst >= MIN_STORIES and dom <= MAX_DOMINANT
def cohesion(m):
"""(S1, MMIN)."""
tot, cnt, lowest = 0.0, 0, None
for a in m:
s, c, row = 0.0, 0, SIM[a]
for b in m:
if story[a] != story[b]:
s += row[b]; c += 1
if c:
tot += s; cnt += c
mean = s / c
lowest = mean if lowest is None else min(lowest, mean)
return (tot / cnt if cnt else 0.0), (lowest if lowest is not None else 0.0)
# Same-size random baseline for Z1.
_null_path = HERE / 'null.pkl'
if _null_path.exists():
NULL = pickle.loads(_null_path.read_bytes())
else:
rnd, NULL = random.Random(2026), {}
for m in range(5, 151):
samples, need = [], 300 if m <= 40 else 120
while len(samples) < need:
members = rnd.sample(range(n), m)
if eligible(members):
samples.append(cohesion(members)[0])
mu = sum(samples) / len(samples)
sd = math.sqrt(sum((x - mu) ** 2 for x in samples) / (len(samples) - 1))
NULL[m] = (mu, sd)
_null_path.write_bytes(pickle.dumps(NULL))
def _lchoose(a, b):
return math.lgamma(a + 1) - math.lgamma(b + 1) - math.lgamma(a - b + 1)
def hyper_sf(c, m, k, total):
"""P(X >= c) for X ~ Hypergeometric(total, k successes, m draws)."""
top = min(m, k)
if c > top: return 0.0
logs = [_lchoose(k, x) + _lchoose(total - k, m - x) - _lchoose(total, m)
for x in range(c, top + 1) if m - x <= total - k]
if not logs: return 0.0
mx = max(logs)
return math.exp(mx) * sum(math.exp(l - mx) for l in logs)
def word_dominance(m):
"""(W, word): largest member share among significantly shared words."""
size = len(m)
cw = collections.Counter(w for i in m for w in docs[i])
best = (0.0, '')
for w, c in cw.items():
if c < 2: continue
if hyper_sf(c, size, df[w], n) <= ALPHA and c / size > best[0]:
best = (c / size, w)
return best
def old_s6(m):
"""The first gate's S6, identical to signals.py."""
size = len(m)
cw = collections.Counter(w for i in m for w in docs[i])
best = None
for w, c in cw.items():
if c < 2: continue
score = math.log((c + .5) / (size - c + .5)) - math.log((df[w] + .5) / (n - df[w] + .5))
if best is None or score > best[0]: best = (score, c)
return best[1] / size if best else 0.0
def signals(m):
size, nst, dom = describe(m)
s1, mmin = cohesion(m)
mu, sd = NULL[min(max(size, 5), 150)]
w, word = word_dominance(m)
return {'size': size, 'stories': nst, 'dominant': dom, 'S1': s1, 'Z1': (s1 - mu) / sd,
'MMIN': mmin, 'W': w, 'W_word': word, 'S6_old': old_s6(m)}
OLD_GATE = json.loads((HERE / 'rule-frozen.json').read_text())
_OLD = {s: t for s, _, t in OLD_GATE['terms']}
def old_passes(sig):
return sig['size'] >= MIN_SIZE and sig['stories'] >= MIN_STORIES and sig['dominant'] <= MAX_DOMINANT \
and sig['S1'] >= _OLD['S1'] and sig['S6_old'] <= _OLD['S6']
def g2_passes(sig, gate):
if not (sig['size'] >= MIN_SIZE and sig['stories'] >= MIN_STORIES and sig['dominant'] <= MAX_DOMINANT):
return False
if gate.get('S1') is not None and sig['S1'] < gate['S1']: return False
if gate.get('Z1') is not None and sig['Z1'] < gate['Z1']: return False
if gate.get('MMIN') is not None and sig['MMIN'] < gate['MMIN']: return False
if gate.get('W') is not None and sig['W'] >= gate['W']: return False
return True
def kmeans(k, seed, iters=15):
"""Identical to measure.py's k-means over the summary vectors."""
rnd = random.Random(seed)
cent = [V[i] for i in rnd.sample(range(n), k)]
assign = [-1] * n
for _ in range(iters):
moved = 0
for i, v in enumerate(V):
best, bs = 0, -2.0
for j, c in enumerate(cent):
s = dot(v, c)
if s > bs: bs, best = s, j
if assign[i] != best: assign[i] = best; moved += 1
if moved == 0: break
groups = collections.defaultdict(list)
for i, j in enumerate(assign): groups[j].append(i)
for j, mem in groups.items():
cent[j] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)])
groups = collections.defaultdict(list)
for i, j in enumerate(assign): groups[j].append(i)
return [sorted(g) for g in groups.values()]
def centroid(m):
return norm([sum(V[i][d] for i in m) / len(m) for d in range(dim)])