Replace clustering with a catalogue, and hand-write the references to judge it against
The pipeline's embed-and-cluster step is dead, and this commit holds both the evidence for that and the step proposed to replace it. Predicaments. Scenes are re-described as "what the person is up against", with no names, jobs or places, then embedded and clustered (redescribe.py, topic_words.py, topic_share.py). The pilot chose qwen3:14b over 3b by reading both side by side. Two defects the pilot exposed are fixed: split.py missed titles in quotes and a contents subtitle after a dash, so three stories had been merged into their neighbours, and strip_names.py read New York place names as people. The corrected corpus is probe/v2 (97 stories, 839 scenes); carry_summaries.py reuses the 829 unchanged v1 summaries. Topic share fell from 20% to 13% at k=60, short of the pre-registered 10%. Hand references. Three corpora were read scene by scene and written up by hand, under the same prompt rules the local models get, as a baseline to judge them against: O. Henry (probe/v2/claude, 839 scenes, 20 situations), Wharton's Descent of Man (probe/wharton, 262 scenes, 16 groups) and Jacobs's The Lady of the Barge (probe/jacobs, 157 scenes, 19 groups). Each has its own README and a readable page. No inference was used for any of them. Catalogue. probe/catalogue maps every hand group in the three references onto 36 situation entries, with an answer key per corpus and one recurrence rule applied to all three. classify.py assigns a scene one entry or none, leave-one-corpus- out; score.py checks it against the key, with a self-test on random labels. Why clustering is out: hand-written predicaments, embedded and clustered exactly as the model's were, agree with the hand grouping at ARI 0.05 — no better than the 14B text's 0.07. Better rewriting cannot rescue it. Embeddings cannot even shortlist: the hand label is the nearest entry 13% of the time and in the top 8 half the time. The classification runs are not here. The dev and test runs are pre-registered in probe/catalogue/README.md with the bar set beforehand, and are blocked on the inference host, whose GPU has fallen off the PCIe bus three times. The 30-scene partial output in out/ is not a result. Review page. The situation review is now a browser page rather than JSON edited by hand (review_page.py, review_page_logic.cjs with Node tests, format schema v2). It has never been rendered in a real browser. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014BygvsUXV9eU6oHkTCkKZ1
This commit is contained in:
co-authored by
Claude Opus 5
parent
fa3769d0fe
commit
e5617b86ba
@@ -1,50 +1,75 @@
|
||||
"""Check a filled-in review form and turn it into situations.json.
|
||||
"""Check a saved review and turn it into situations.json.
|
||||
|
||||
python3 apply_review.py review/k60-s24
|
||||
python3 apply_review.py review/k60-s24 # finds the saved review itself
|
||||
python3 apply_review.py review/k60-s24 ~/Downloads/review-k60-s24.json
|
||||
|
||||
Reads candidates.json and review.json from that directory. If the form is complete and consistent, it
|
||||
writes situations.json there and exits 0. Otherwise it prints every problem at once, writes nothing,
|
||||
and exits 1.
|
||||
Without a file named, it uses the newest review-<run>*.json in the run directory, and
|
||||
otherwise the newest one in ~/Downloads, where a browser saves it. It prints every problem
|
||||
at once and writes nothing until the review passes; then it writes situations.json into the
|
||||
run directory.
|
||||
"""
|
||||
import datetime, json, pathlib, sys
|
||||
import datetime, json, os, pathlib, sys
|
||||
import review_format as rf
|
||||
|
||||
|
||||
def downloads_dir():
|
||||
return pathlib.Path(os.environ.get('STP_DOWNLOADS', pathlib.Path.home() / 'Downloads'))
|
||||
|
||||
|
||||
def find_review(run_dir, run_id):
|
||||
for folder in (run_dir, downloads_dir()):
|
||||
if folder.is_dir():
|
||||
found = sorted(folder.glob(f'review-{run_id}*.json'), key=lambda p: p.stat().st_mtime, reverse=True)
|
||||
if found:
|
||||
return found[0]
|
||||
return None
|
||||
|
||||
|
||||
def main(argv):
|
||||
if len(argv) != 2:
|
||||
if len(argv) not in (2, 3):
|
||||
print(__doc__.strip())
|
||||
return 2
|
||||
run_dir = pathlib.Path(argv[1])
|
||||
cand_path, form_path = run_dir / 'candidates.json', run_dir / 'review.json'
|
||||
for p in (cand_path, form_path):
|
||||
if not p.exists():
|
||||
print(f'{p} does not exist. Run review.py first, or check the directory name.')
|
||||
return 1
|
||||
cand_path = run_dir / 'candidates.json'
|
||||
if not cand_path.exists():
|
||||
print(f'{run_dir} has no candidates.json. Check the folder name, or run review.py first.')
|
||||
return 1
|
||||
candidates = json.loads(cand_path.read_text(encoding='utf-8'))
|
||||
run_id = candidates['run']['id']
|
||||
|
||||
review_path = pathlib.Path(argv[2]).expanduser() if len(argv) == 3 else find_review(run_dir, run_id)
|
||||
if review_path is None or not review_path.exists():
|
||||
print(f'No saved review found for run {run_id}.')
|
||||
print(f'Open {run_dir / "REVIEW.html"} in a browser, and at the end press "Save review file".')
|
||||
print(f'It saves review-{run_id}.json into Downloads, where this looks for it.')
|
||||
return 1
|
||||
print(f'reading {review_path}')
|
||||
try:
|
||||
form = json.loads(form_path.read_text(encoding='utf-8'))
|
||||
except json.JSONDecodeError as e:
|
||||
print(f'{form_path} is not valid JSON: line {e.lineno}, column {e.colno}: {e.msg}.')
|
||||
print('Common causes: a missing comma between fields, a trailing comma before } or ], or a quote left open.')
|
||||
form = json.loads(review_path.read_text(encoding='utf-8'))
|
||||
except json.JSONDecodeError:
|
||||
print('That file is damaged and cannot be read. Save the review again from REVIEW.html; '
|
||||
'your progress is still in the browser.')
|
||||
return 1
|
||||
|
||||
errors, warnings, result = rf.validate(candidates, form)
|
||||
for w in warnings:
|
||||
print(f'warning: {w}')
|
||||
print(f'note: {w}')
|
||||
if errors:
|
||||
print(f'\n{len(errors)} problem(s) to fix in {form_path}; nothing was written:')
|
||||
print(f'\nThe review is not finished. {len(errors)} thing(s) to sort out in REVIEW.html, '
|
||||
f'then save the review file again:')
|
||||
for e in errors:
|
||||
print(f' - {e}')
|
||||
return 1
|
||||
|
||||
result['applied'] = datetime.datetime.now().astimezone().isoformat(timespec='seconds')
|
||||
result['review_file'] = str(review_path)
|
||||
out = run_dir / 'situations.json'
|
||||
out.write_text(json.dumps(result, indent=1, ensure_ascii=False), encoding='utf-8')
|
||||
c = result['counts']
|
||||
print(f'\n{out}: {c["situations"]} situations from {c["accepted"]} accepted and {c["merged"]} merged '
|
||||
f'candidates; {c["rejected"]} rejected, {c["skipped"]} skipped.')
|
||||
print(f'\n{out}: {c["situations"]} situations ({c["kept"]} kept, {c["same"]} merged in, '
|
||||
f'{c["dropped"]} dropped, {c["left_out"]} extras left out).')
|
||||
for s in result['situations']:
|
||||
print(f' {s["id"]} ({s["size"]} scenes, {s["stories"]} stories, from {"+".join(s["from"])}): {s["situation"]}')
|
||||
print(f' {s["id"]} {s["situation"]} ({s["size"]} scenes, {s["stories"]} stories)')
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user