The pipeline's embed-and-cluster step is dead, and this commit holds both the evidence for that and the step proposed to replace it. Predicaments. Scenes are re-described as "what the person is up against", with no names, jobs or places, then embedded and clustered (redescribe.py, topic_words.py, topic_share.py). The pilot chose qwen3:14b over 3b by reading both side by side. Two defects the pilot exposed are fixed: split.py missed titles in quotes and a contents subtitle after a dash, so three stories had been merged into their neighbours, and strip_names.py read New York place names as people. The corrected corpus is probe/v2 (97 stories, 839 scenes); carry_summaries.py reuses the 829 unchanged v1 summaries. Topic share fell from 20% to 13% at k=60, short of the pre-registered 10%. Hand references. Three corpora were read scene by scene and written up by hand, under the same prompt rules the local models get, as a baseline to judge them against: O. Henry (probe/v2/claude, 839 scenes, 20 situations), Wharton's Descent of Man (probe/wharton, 262 scenes, 16 groups) and Jacobs's The Lady of the Barge (probe/jacobs, 157 scenes, 19 groups). Each has its own README and a readable page. No inference was used for any of them. Catalogue. probe/catalogue maps every hand group in the three references onto 36 situation entries, with an answer key per corpus and one recurrence rule applied to all three. classify.py assigns a scene one entry or none, leave-one-corpus- out; score.py checks it against the key, with a self-test on random labels. Why clustering is out: hand-written predicaments, embedded and clustered exactly as the model's were, agree with the hand grouping at ARI 0.05 — no better than the 14B text's 0.07. Better rewriting cannot rescue it. Embeddings cannot even shortlist: the hand label is the nearest entry 13% of the time and in the top 8 half the time. The classification runs are not here. The dev and test runs are pre-registered in probe/catalogue/README.md with the bar set beforehand, and are blocked on the inference host, whose GPU has fallen off the PCIe bus three times. The 30-scene partial output in out/ is not a result. Review page. The situation review is now a browser page rather than JSON edited by hand (review_page.py, review_page_logic.cjs with Node tests, format schema v2). It has never been rendered in a real browser. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014BygvsUXV9eU6oHkTCkKZ1
101 lines
5.1 KiB
Python
101 lines
5.1 KiB
Python
"""Build a review page: the groups gate G2 accepts, plus the nearest misses.
|
|
|
|
python3 review.py # k = 60, seed 24 -> review/k60-s24/
|
|
python3 review.py --k 60 --seed 31 --borderline 8
|
|
|
|
Writes into the run directory:
|
|
REVIEW.html open it in a browser: the whole review happens there
|
|
candidates.json what was found, for apply_review.py to check the saved review against
|
|
|
|
When the review is saved from the page:
|
|
python3 apply_review.py review/k60-s24 -> situations.json
|
|
|
|
No inference: this uses the summaries and embeddings already on disk.
|
|
"""
|
|
import argparse, datetime, hashlib, json, pathlib
|
|
import gate2 as g
|
|
import review_format as rf
|
|
import review_page
|
|
|
|
|
|
def gate_failures(sig, gate):
|
|
out = []
|
|
if gate.get('S1') is not None and sig['S1'] < gate['S1']:
|
|
out.append(f"cohesion S1 {sig['S1']:.3f}, needs {gate['S1']:.3f}")
|
|
if gate.get('Z1') is not None and sig['Z1'] < gate['Z1']:
|
|
out.append(f"cohesion against same-size random sets Z1 {sig['Z1']:.1f}, needs {gate['Z1']}")
|
|
if gate.get('MMIN') is not None and sig['MMIN'] < gate['MMIN']:
|
|
out.append(f"weakest scene MMIN {sig['MMIN']:.3f}, needs {gate['MMIN']:.3f}")
|
|
if gate.get('W') is not None and sig['W'] >= gate['W']:
|
|
out.append(f"the word \"{sig['W_word']}\" is in {sig['W']:.0%} of summaries, limit {gate['W']:.0%}")
|
|
return out
|
|
|
|
|
|
def shortfall(sig, gate):
|
|
"""How far a rejected group is from passing; smaller is closer."""
|
|
s = 0.0
|
|
if gate.get('S1') is not None: s += max(0.0, gate['S1'] - sig['S1']) / 0.01
|
|
if gate.get('Z1') is not None: s += max(0.0, gate['Z1'] - sig['Z1'])
|
|
if gate.get('W') is not None and sig['W'] >= gate['W']: s += 10 * (sig['W'] - gate['W'] + 0.01)
|
|
return s
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
ap.add_argument('--k', type=int, default=60, help='number of k-means clusters (default 60)')
|
|
ap.add_argument('--seed', type=int, default=24, help='k-means seed (default 24)')
|
|
ap.add_argument('--borderline', type=int, default=8, help='near misses offered as optional extras (default 8)')
|
|
ap.add_argument('--out', help='run directory (default review/k<K>-s<SEED>)')
|
|
args = ap.parse_args()
|
|
|
|
run_id = f'k{args.k}-s{args.seed}'
|
|
out = pathlib.Path(args.out or f'review/{run_id}')
|
|
|
|
gate = json.loads(pathlib.Path('gate2-frozen.json').read_text())
|
|
found = [c for c in g.kmeans(args.k, args.seed) if g.eligible(c)]
|
|
scored = [(c, g.signals(c)) for c in found]
|
|
accepted = sorted([cs for cs in scored if g.g2_passes(cs[1], gate)], key=lambda cs: -cs[1]['S1'])
|
|
rejected = sorted([cs for cs in scored if not g.g2_passes(cs[1], gate)], key=lambda cs: shortfall(cs[1], gate))
|
|
borderline = rejected[:args.borderline]
|
|
|
|
listed = [(f'C{i + 1:02d}', 'accepted', c, s) for i, (c, s) in enumerate(accepted)]
|
|
listed += [(f'B{i + 1:02d}', 'borderline', c, s) for i, (c, s) in enumerate(borderline)]
|
|
|
|
candidates = {}
|
|
for cid, kind, members, sig in listed:
|
|
member_rows = [{'member': rf.member_id(cid, n + 1), 'chunk': chunk, 'story': g.chunks[chunk]['story'],
|
|
'title': g.chunks[chunk]['title'].title(), 'summary': g.text[chunk]}
|
|
for n, chunk in enumerate(sorted(members))]
|
|
candidates[cid] = {
|
|
'kind': kind,
|
|
'stats': {'scenes': sig['size'], 'stories': sig['stories'], 'largest_story_share': round(sig['dominant'], 3),
|
|
'S1': round(sig['S1'], 4), 'Z1': round(sig['Z1'], 2), 'W': round(sig['W'], 3), 'W_word': sig['W_word']},
|
|
'gate_failures': gate_failures(sig, gate),
|
|
'role_hints': review_page.role_hints(member_rows),
|
|
'members': member_rows,
|
|
}
|
|
# Ties a saved review to exactly these groups, so progress for a different set of groups
|
|
# that happens to share the run name is never loaded into this page.
|
|
fingerprint = hashlib.sha1(json.dumps({cid: [m['chunk'] for m in c['members']] for cid, c in candidates.items()},
|
|
sort_keys=True).encode()).hexdigest()[:12]
|
|
document = {
|
|
'schema_version': rf.SCHEMA_VERSION,
|
|
'run': {'id': run_id, 'k': args.k, 'seed': args.seed, 'fingerprint': fingerprint,
|
|
'created': datetime.datetime.now().astimezone().isoformat(timespec='seconds'),
|
|
'gate': {'name': 'G2', **gate}, 'scenes': g.n, 'stories': len(set(g.story)),
|
|
'candidates_found': len(found), 'accepted': len(accepted), 'borderline': len(borderline)},
|
|
'candidates': candidates,
|
|
}
|
|
|
|
out.mkdir(parents=True, exist_ok=True)
|
|
(out / 'candidates.json').write_text(json.dumps(document, indent=1, ensure_ascii=False), encoding='utf-8')
|
|
(out / 'REVIEW.html').write_text(review_page.render_page(document), encoding='utf-8')
|
|
|
|
print(f'{run_id}: {len(found)} candidate groups, G2 accepts {len(accepted)}, {len(borderline)} optional extras')
|
|
print(f'open this in a browser: {(out / "REVIEW.html").resolve()}')
|
|
print(f'when the review is saved: python3 apply_review.py {out}')
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main()
|