"""G2, the size-invariant gate: signals, the old gate for comparison, and k-means. No inference. Import it from the probe directory. On first import it builds and caches pairwise similarities (sim.pkl) and the same-size random baseline for Z1 (null.pkl). S1 mean summary cosine over cross-story member pairs Z1 S1 as a z-score against random member sets of the same size MMIN lowest member-level mean cosine to members of other stories W largest share among words whose sharing is significant (hypergeometric, Bonferroni 0.01) """ import collections, json, math, pathlib, pickle, random, re HERE = pathlib.Path(__file__).resolve().parent MIN_SIZE, MIN_STORIES, MAX_DOMINANT = 5, 4, 0.40 chunks = json.loads((HERE / 'chunks.json').read_text(encoding='utf-8')) _raw = json.loads((HERE / 'summary_embeddings.json').read_text()) _vecs = _raw['vectors'] if isinstance(_raw, dict) else _raw n = min(len(chunks), len(_vecs)) def norm(v): m = math.sqrt(sum(x * x for x in v)) or 1.0 return [x / m for x in v] V = [norm(v) for v in _vecs[:n]] dim = len(V[0]) story = [chunks[i]['story'] for i in range(n)] text = {x['chunk']: x['summary'] for x in json.loads((HERE / 'summaries_clean.json').read_text(encoding='utf-8'))['summaries']} def dot(a, b): s = 0.0 for x, y in zip(a, b): s += x * y return s _sim_path = HERE / 'sim.pkl' if _sim_path.exists(): SIM = pickle.loads(_sim_path.read_bytes()) else: SIM = [[0.0] * n for _ in range(n)] for a in range(n): va, row = V[a], SIM[a] for b in range(a, n): s = dot(va, V[b]) row[b] = s SIM[b][a] = s _sim_path.write_bytes(pickle.dumps(SIM)) # Identical to signals.py. STOP = set("""a an the and or but of to in on at for with by from as into over under about after before while during than then that this these those it its his her hers him he she they them their theirs we our you your i me my is are was were be been being has have had do does did not no nor so such who whom whose which what when where why how all any each both either neither one two some other others another more most less least very just only also even yet still now up down out off away back again once own same too can could will would shall should may might must upon despite toward towards between among amid amidst someone someone's someone’s""".split()) docs = [set(re.findall(r"[a-z][a-z'’]+", text[i].lower())) - STOP for i in range(n)] df = collections.Counter(w for d in docs for w in d) ALPHA = 0.01 / len(df) def describe(m): counts = collections.Counter(story[i] for i in m) return len(m), len(counts), max(counts.values()) / len(m) def eligible(m): size, nst, dom = describe(m) return size >= MIN_SIZE and nst >= MIN_STORIES and dom <= MAX_DOMINANT def cohesion(m): """(S1, MMIN).""" tot, cnt, lowest = 0.0, 0, None for a in m: s, c, row = 0.0, 0, SIM[a] for b in m: if story[a] != story[b]: s += row[b]; c += 1 if c: tot += s; cnt += c mean = s / c lowest = mean if lowest is None else min(lowest, mean) return (tot / cnt if cnt else 0.0), (lowest if lowest is not None else 0.0) # Same-size random baseline for Z1. _null_path = HERE / 'null.pkl' if _null_path.exists(): NULL = pickle.loads(_null_path.read_bytes()) else: rnd, NULL = random.Random(2026), {} for m in range(5, 151): samples, need = [], 300 if m <= 40 else 120 while len(samples) < need: members = rnd.sample(range(n), m) if eligible(members): samples.append(cohesion(members)[0]) mu = sum(samples) / len(samples) sd = math.sqrt(sum((x - mu) ** 2 for x in samples) / (len(samples) - 1)) NULL[m] = (mu, sd) _null_path.write_bytes(pickle.dumps(NULL)) def _lchoose(a, b): return math.lgamma(a + 1) - math.lgamma(b + 1) - math.lgamma(a - b + 1) def hyper_sf(c, m, k, total): """P(X >= c) for X ~ Hypergeometric(total, k successes, m draws).""" top = min(m, k) if c > top: return 0.0 logs = [_lchoose(k, x) + _lchoose(total - k, m - x) - _lchoose(total, m) for x in range(c, top + 1) if m - x <= total - k] if not logs: return 0.0 mx = max(logs) return math.exp(mx) * sum(math.exp(l - mx) for l in logs) def word_dominance(m): """(W, word): largest member share among significantly shared words.""" size = len(m) cw = collections.Counter(w for i in m for w in docs[i]) best = (0.0, '') for w, c in cw.items(): if c < 2: continue if hyper_sf(c, size, df[w], n) <= ALPHA and c / size > best[0]: best = (c / size, w) return best def old_s6(m): """The first gate's S6, identical to signals.py.""" size = len(m) cw = collections.Counter(w for i in m for w in docs[i]) best = None for w, c in cw.items(): if c < 2: continue score = math.log((c + .5) / (size - c + .5)) - math.log((df[w] + .5) / (n - df[w] + .5)) if best is None or score > best[0]: best = (score, c) return best[1] / size if best else 0.0 def signals(m): size, nst, dom = describe(m) s1, mmin = cohesion(m) mu, sd = NULL[min(max(size, 5), 150)] w, word = word_dominance(m) return {'size': size, 'stories': nst, 'dominant': dom, 'S1': s1, 'Z1': (s1 - mu) / sd, 'MMIN': mmin, 'W': w, 'W_word': word, 'S6_old': old_s6(m)} OLD_GATE = json.loads((HERE / 'rule-frozen.json').read_text()) _OLD = {s: t for s, _, t in OLD_GATE['terms']} def old_passes(sig): return sig['size'] >= MIN_SIZE and sig['stories'] >= MIN_STORIES and sig['dominant'] <= MAX_DOMINANT \ and sig['S1'] >= _OLD['S1'] and sig['S6_old'] <= _OLD['S6'] def g2_passes(sig, gate): if not (sig['size'] >= MIN_SIZE and sig['stories'] >= MIN_STORIES and sig['dominant'] <= MAX_DOMINANT): return False if gate.get('S1') is not None and sig['S1'] < gate['S1']: return False if gate.get('Z1') is not None and sig['Z1'] < gate['Z1']: return False if gate.get('MMIN') is not None and sig['MMIN'] < gate['MMIN']: return False if gate.get('W') is not None and sig['W'] >= gate['W']: return False return True def kmeans(k, seed, iters=15): """Identical to measure.py's k-means over the summary vectors.""" rnd = random.Random(seed) cent = [V[i] for i in rnd.sample(range(n), k)] assign = [-1] * n for _ in range(iters): moved = 0 for i, v in enumerate(V): best, bs = 0, -2.0 for j, c in enumerate(cent): s = dot(v, c) if s > bs: bs, best = s, j if assign[i] != best: assign[i] = best; moved += 1 if moved == 0: break groups = collections.defaultdict(list) for i, j in enumerate(assign): groups[j].append(i) for j, mem in groups.items(): cent[j] = norm([sum(V[i][d] for i in mem) / len(mem) for d in range(dim)]) groups = collections.defaultdict(list) for i, j in enumerate(assign): groups[j].append(i) return [sorted(g) for g in groups.values()] def centroid(m): return norm([sum(V[i][d] for i in m) / len(m) for d in range(dim)])