The pipeline's embed-and-cluster step is dead, and this commit holds both the evidence for that and the step proposed to replace it. Predicaments. Scenes are re-described as "what the person is up against", with no names, jobs or places, then embedded and clustered (redescribe.py, topic_words.py, topic_share.py). The pilot chose qwen3:14b over 3b by reading both side by side. Two defects the pilot exposed are fixed: split.py missed titles in quotes and a contents subtitle after a dash, so three stories had been merged into their neighbours, and strip_names.py read New York place names as people. The corrected corpus is probe/v2 (97 stories, 839 scenes); carry_summaries.py reuses the 829 unchanged v1 summaries. Topic share fell from 20% to 13% at k=60, short of the pre-registered 10%. Hand references. Three corpora were read scene by scene and written up by hand, under the same prompt rules the local models get, as a baseline to judge them against: O. Henry (probe/v2/claude, 839 scenes, 20 situations), Wharton's Descent of Man (probe/wharton, 262 scenes, 16 groups) and Jacobs's The Lady of the Barge (probe/jacobs, 157 scenes, 19 groups). Each has its own README and a readable page. No inference was used for any of them. Catalogue. probe/catalogue maps every hand group in the three references onto 36 situation entries, with an answer key per corpus and one recurrence rule applied to all three. classify.py assigns a scene one entry or none, leave-one-corpus- out; score.py checks it against the key, with a self-test on random labels. Why clustering is out: hand-written predicaments, embedded and clustered exactly as the model's were, agree with the hand grouping at ARI 0.05 — no better than the 14B text's 0.07. Better rewriting cannot rescue it. Embeddings cannot even shortlist: the hand label is the nearest entry 13% of the time and in the top 8 half the time. The classification runs are not here. The dev and test runs are pre-registered in probe/catalogue/README.md with the bar set beforehand, and are blocked on the inference host, whose GPU has fallen off the PCIe bus three times. The 30-scene partial output in out/ is not a result. Review page. The situation review is now a browser page rather than JSON edited by hand (review_page.py, review_page_logic.cjs with Node tests, format schema v2). It has never been rendered in a real browser. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014BygvsUXV9eU6oHkTCkKZ1
88 lines
5.3 KiB
Python
88 lines
5.3 KiB
Python
"""Compare Claude's hand pass with the local pipeline, on files already on disk. No inference.
|
|
|
|
cd probe/v2 && python3 claude/compare.py > claude/compare.log
|
|
|
|
1. Text: leakage of topic words (jobs, relationships, places) and capitalised names into the
|
|
one-sentence predicaments, Claude vs qwen3:14b (same prompt rules).
|
|
2. Grouping: Claude's hand situations vs the k-means groups on the 14B predicament embeddings
|
|
(k = 60; seed 24 is the review seed, 11-13 the measurement seeds).
|
|
"""
|
|
import collections, json, math, pathlib, re, statistics, sys
|
|
HERE = pathlib.Path(__file__).resolve().parent
|
|
sys.path.insert(0, str(HERE)); sys.path.insert(0, str(HERE.parent.parent))
|
|
from assign import ASSIGN
|
|
from topic_words import TOPIC_WORDS, PLACE_WORDS
|
|
|
|
chunks = json.load(open('chunks.json'))
|
|
n = len(chunks)
|
|
mine = {x['chunk']: x['summary'] for x in json.load(open(HERE / 'predicaments-claude.json'))['summaries']}
|
|
q14 = {x['chunk']: x['summary'] for x in json.load(open('predicaments_clean.json'))['summaries']}
|
|
q14raw = {x['chunk']: x.get('raw', x['summary']) for x in json.load(open('predicaments.json'))['summaries']}
|
|
|
|
def words(s): return re.findall(r"[a-zé']+", s.lower())
|
|
def caps_mid(s):
|
|
toks = re.findall(r"[A-Za-z']+", s)
|
|
return [t for t in toks[1:] if t[0].isupper() and t not in ('I',)]
|
|
|
|
print('== 1. Predicament text (839 scenes each) ==')
|
|
for name, d in (('claude', mine), ('qwen3:14b raw', q14raw), ('qwen3:14b after strip_names', q14)):
|
|
tw = [set(words(d[i])) & TOPIC_WORDS for i in range(n)]
|
|
pw = [set(words(d[i])) & PLACE_WORDS for i in range(n)]
|
|
cm = [caps_mid(d[i]) for i in range(n)]
|
|
lens = [len(d[i].split()) for i in range(n)]
|
|
some = sum(1 for i in range(n) if d[i].lower().startswith('someone'))
|
|
print(f'{name:30s} median words {statistics.median(lens):>4} over 20 words {sum(l > 20 for l in lens):>3} '
|
|
f'with a topic word {sum(bool(x) for x in tw)/n:5.1%} with a place word {sum(bool(x) for x in pw)/n:5.1%} '
|
|
f'capitalised mid-sentence {sum(bool(x) for x in cm):>3} starts "Someone" {some/n:5.0%}')
|
|
top = collections.Counter(w for x in tw for w in x).most_common(12)
|
|
print(' ' * 32 + 'commonest topic words: ' + ', '.join(f'{w} {c}' for w, c in top))
|
|
|
|
# ---- 2. grouping agreement --------------------------------------------------------------
|
|
import os; EMB_FILE = os.environ.get('EMB_FILE', 'predicament_embeddings.json'); print('embeddings:', EMB_FILE); sys.argv = ['measure.py', EMB_FILE]
|
|
src = (HERE.parent.parent / 'measure.py').read_text(encoding='utf-8').split("print(f'{path.name}")[0]
|
|
exec(src) # defines V, story, kmeans, describe, MIN_SIZE, MIN_STORIES, MAX_DOMINANT
|
|
|
|
def ari(labels_a, labels_b):
|
|
pairs = collections.Counter(zip(labels_a, labels_b))
|
|
a = collections.Counter(labels_a); b = collections.Counter(labels_b)
|
|
c2 = lambda x: x * (x - 1) / 2
|
|
idx = sum(c2(v) for v in pairs.values()); sa = sum(c2(v) for v in a.values()); sb = sum(c2(v) for v in b.values())
|
|
exp = sa * sb / c2(len(labels_a)); mx = (sa + sb) / 2
|
|
return (idx - exp) / (mx - exp) if mx != exp else 0.0
|
|
|
|
print('\n== 2. Claude hand situations vs 14B k-means groups (k = 60) ==')
|
|
print(f'Claude assigned {len(ASSIGN)} of {n} scenes to {len(set(ASSIGN.values()))} situations; the rest carry no recurring situation.')
|
|
for seed in (24, 11, 12, 13):
|
|
groups = kmeans(60, seed)
|
|
lab = {}
|
|
for gi, g in enumerate(groups):
|
|
for i in g: lab[i] = gi
|
|
assigned = sorted(ASSIGN)
|
|
a = ari([ASSIGN[i] for i in assigned], [lab[i] for i in assigned])
|
|
cand = [g for g in groups if (lambda s: s[0] >= MIN_SIZE and s[1] >= MIN_STORIES and s[2] <= MAX_DOMINANT)(describe(g))]
|
|
purities, unassigned = [], []
|
|
for g in cand:
|
|
sits = collections.Counter(ASSIGN[i] for i in g if i in ASSIGN)
|
|
purities.append(sits.most_common(1)[0][1] / len(g) if sits else 0.0)
|
|
unassigned.append(sum(1 for i in g if i not in ASSIGN) / len(g))
|
|
print(f'seed {seed:>2}: ARI on assigned scenes {a:+.3f}; {len(cand)} candidate groups: median share of a group that is its '
|
|
f'commonest Claude situation {statistics.median(purities):.0%}, median share with no situation {statistics.median(unassigned):.0%}')
|
|
if seed == 24:
|
|
spread = {}
|
|
for s in sorted(set(ASSIGN.values())):
|
|
mem = [i for i in ASSIGN if ASSIGN[i] == s]
|
|
cl = collections.Counter(lab[i] for i in mem)
|
|
spread[s] = (len(mem), len(cl), cl.most_common(1)[0][1] / len(mem))
|
|
print(' how each Claude situation is scattered across the 14B groups at seed 24:')
|
|
for s, (m, k, top) in sorted(spread.items(), key=lambda kv: -kv[1][0]):
|
|
print(f' {s:18s} {m:3d} scenes over {k:2d} groups; largest single group holds {top:4.0%}')
|
|
rows = []
|
|
for g in cand:
|
|
sits = collections.Counter(ASSIGN[i] for i in g if i in ASSIGN)
|
|
top = sits.most_common(1)[0] if sits else ('-', 0)
|
|
rows.append((top[1] / len(g), len(g), top[0], sits))
|
|
print(' the 14B candidate groups, purest first (share = scenes carrying the commonest Claude situation):')
|
|
for share, size, s, sits in sorted(rows, reverse=True):
|
|
other = ', '.join(f'{k} {v}' for k, v in sits.most_common()[1:4])
|
|
print(f' {share:4.0%} of {size:2d} {s:18s} {("also " + other) if other else ""}')
|