Files
TheLadder/tools/story-to-pack/probe/test_review.py
T
JesseMarkowitzandClaude Opus 5 e5617b86ba Replace clustering with a catalogue, and hand-write the references to judge it against
The pipeline's embed-and-cluster step is dead, and this commit holds both the
evidence for that and the step proposed to replace it.

Predicaments. Scenes are re-described as "what the person is up against", with
no names, jobs or places, then embedded and clustered (redescribe.py,
topic_words.py, topic_share.py). The pilot chose qwen3:14b over 3b by reading
both side by side. Two defects the pilot exposed are fixed: split.py missed
titles in quotes and a contents subtitle after a dash, so three stories had been
merged into their neighbours, and strip_names.py read New York place names as
people. The corrected corpus is probe/v2 (97 stories, 839 scenes);
carry_summaries.py reuses the 829 unchanged v1 summaries. Topic share fell from
20% to 13% at k=60, short of the pre-registered 10%.

Hand references. Three corpora were read scene by scene and written up by hand,
under the same prompt rules the local models get, as a baseline to judge them
against: O. Henry (probe/v2/claude, 839 scenes, 20 situations), Wharton's
Descent of Man (probe/wharton, 262 scenes, 16 groups) and Jacobs's The Lady of
the Barge (probe/jacobs, 157 scenes, 19 groups). Each has its own README and a
readable page. No inference was used for any of them.

Catalogue. probe/catalogue maps every hand group in the three references onto 36
situation entries, with an answer key per corpus and one recurrence rule applied
to all three. classify.py assigns a scene one entry or none, leave-one-corpus-
out; score.py checks it against the key, with a self-test on random labels.

Why clustering is out: hand-written predicaments, embedded and clustered exactly
as the model's were, agree with the hand grouping at ARI 0.05 — no better than
the 14B text's 0.07. Better rewriting cannot rescue it. Embeddings cannot even
shortlist: the hand label is the nearest entry 13% of the time and in the top 8
half the time.

The classification runs are not here. The dev and test runs are pre-registered
in probe/catalogue/README.md with the bar set beforehand, and are blocked on the
inference host, whose GPU has fallen off the PCIe bus three times. The 30-scene
partial output in out/ is not a result.

Review page. The situation review is now a browser page rather than JSON edited
by hand (review_page.py, review_page_logic.cjs with Node tests, format schema
v2). It has never been rendered in a real browser.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014BygvsUXV9eU6oHkTCkKZ1
2026-09-20 17:19:42 -04:00

174 lines
8.1 KiB
Python

"""Tests for the saved review's checks, the page renderer, and apply_review.py.
python3 -m unittest test_review
"""
import json, os, pathlib, subprocess, sys, tempfile, unittest
import review_format as rf
import review_page
HERE = pathlib.Path(__file__).resolve().parent
def group(cid, kind, stories):
"""A synthetic group whose n-th scene comes from stories[n]."""
return {'kind': kind, 'stats': {'stories': len(set(stories))}, 'gate_failures': [], 'role_hints': [],
'members': [{'member': rf.member_id(cid, n + 1), 'chunk': abs(hash((cid, n))) % 100000, 'story': s,
'title': f'Story {s}', 'summary': f'{cid} scene {n + 1}'} for n, s in enumerate(stories)]}
def fixture():
return {'schema_version': rf.SCHEMA_VERSION,
'run': {'id': 'k60-s99', 'fingerprint': 'abc123', 'scenes': 21, 'stories': 13},
'candidates': {'C01': group('C01', 'accepted', [1, 2, 3, 4, 5, 6, 7, 8]),
'C02': group('C02', 'accepted', [9, 10, 11, 12, 13, 14]),
'B01': group('B01', 'borderline', [1, 3, 5, 7, 9, 11, 13])}}
def blank_review(cands):
return {'schema_version': rf.SCHEMA_VERSION, 'run': cands['run']['id'], 'fingerprint': cands['run']['fingerprint'],
'entries': [rf.blank_entry(cid, c['kind']) for cid, c in cands['candidates'].items()]}
def finished(cands):
review = blank_review(cands)
e = {x['id']: x for x in review['entries']}
e['C01'].update(decision='keep', actor='lodger', wants='more time to pay', counterpart='a landlady',
excluded=['C01.02'])
e['C02'].update(decision='same', same_as='C01')
return review, e
class ReviewChecks(unittest.TestCase):
def test_a_blank_review_asks_only_for_the_required_groups(self):
cands = fixture()
errors, _, result = rf.validate(cands, blank_review(cands))
self.assertIsNone(result)
self.assertTrue(any(err.startswith('Group 1: not decided yet') for err in errors), errors)
self.assertTrue(any(err.startswith('Group 2: not decided yet') for err in errors), errors)
self.assertFalse(any('Extra group' in err for err in errors))
def test_a_finished_review_merges_and_leaves_out_unticked_scenes(self):
cands = fixture()
review, _ = finished(cands)
errors, _, result = rf.validate(cands, review)
self.assertEqual(errors, [])
self.assertEqual(len(result['situations']), 1)
s = result['situations'][0]
self.assertEqual(s['situation'], 'A lodger wants more time to pay from a landlady.')
self.assertEqual(s['from'], ['C01', 'C02'])
self.assertEqual(s['size'], 8 - 1 + 6)
self.assertNotIn('C01.02', [m['member'] for m in s['members']])
self.assertEqual(result['left_out'], ['B01'])
def test_a_kept_group_must_be_named(self):
cands = fixture()
review, e = finished(cands)
e['C01']['wants'] = ' '
errors, _, _ = rf.validate(cands, review)
self.assertTrue(any('Group 1: kept, but its name is not finished' in err and 'what they want' in err
for err in errors), errors)
def test_same_as_a_group_that_is_not_kept(self):
cands = fixture()
review, e = finished(cands)
e['C01']['decision'] = 'drop'
errors, _, _ = rf.validate(cands, review)
self.assertTrue(any(err.startswith('Group 2: marked the same as Group 1, but Group 1 is not kept')
for err in errors), errors)
def test_too_few_ticked_scenes_are_refused_with_what_to_do(self):
cands = fixture()
review, e = finished(cands)
e['C02'].update(decision='keep', same_as=None, actor='clerk', wants='a raise', counterpart='an employer',
excluded=['C02.01', 'C02.02', 'C02.03'])
errors, _, result = rf.validate(cands, review)
self.assertIsNone(result)
self.assertTrue(any(err.startswith('Group 2: 3 ticked scenes from 3 stories') and 'Tick more scenes' in err
for err in errors), errors)
def test_an_older_review_file_is_explained(self):
cands = fixture()
review, _ = finished(cands)
review['schema_version'] = 1
errors, _, _ = rf.validate(cands, review)
self.assertEqual(len(errors), 1)
self.assertIn('different version of the review tool', errors[0])
def test_a_review_for_different_groups_is_refused(self):
cands = fixture()
review, _ = finished(cands)
review['fingerprint'] = 'zzz999'
errors, _, _ = rf.validate(cands, review)
self.assertTrue(any('different set of groups' in err for err in errors), errors)
def test_an_extra_group_cannot_be_dropped_only_left_out(self):
cands = fixture()
review, e = finished(cands)
e['B01']['decision'] = 'drop'
errors, _, _ = rf.validate(cands, review)
self.assertTrue(any(err.startswith('Extra group 1') for err in errors), errors)
class PageRendering(unittest.TestCase):
def test_the_page_embeds_everything_and_nothing_can_close_its_scripts_early(self):
cands = fixture()
cands['candidates']['C01']['members'][0]['summary'] = 'A trap </script><b>bold</b>'
page = review_page.render_page(cands)
self.assertNotIn('/*__DATA__*/', page)
self.assertNotIn('/*__LOGIC__*/', page)
self.assertNotIn('__RUN_ID__', page)
self.assertEqual(page.count('</script>'), 2, 'only the two real script tags may close')
self.assertIn('Situation review · k60-s99', page)
def test_role_words_are_suggested_from_the_scenes(self):
members = [{'summary': 'A lodger begs the landlady for time.'},
{'summary': 'The landlady turns a lodger out.'},
{'summary': 'Someone hides from the landlady.'}]
self.assertEqual(review_page.role_hints(members)[:2], ['landlady', 'lodger'])
class ApplyReview(unittest.TestCase):
def run_apply(self, run_dir, *extra, downloads):
env = dict(os.environ, STP_DOWNLOADS=str(downloads))
return subprocess.run([sys.executable, 'apply_review.py', str(run_dir), *extra], cwd=HERE,
capture_output=True, text=True, env=env)
def test_it_finds_the_saved_review_in_downloads_and_writes_nothing_until_it_passes(self):
cands = fixture()
with tempfile.TemporaryDirectory() as tmp:
run_dir, downloads = pathlib.Path(tmp, 'run'), pathlib.Path(tmp, 'Downloads')
run_dir.mkdir(); downloads.mkdir()
(run_dir / 'candidates.json').write_text(json.dumps(cands))
nothing = self.run_apply(run_dir, downloads=downloads)
self.assertEqual(nothing.returncode, 1)
self.assertIn('No saved review found', nothing.stdout)
(downloads / 'review-k60-s99.json').write_text(json.dumps(blank_review(cands)))
unfinished = self.run_apply(run_dir, downloads=downloads)
self.assertEqual(unfinished.returncode, 1)
self.assertIn('The review is not finished', unfinished.stdout)
self.assertFalse((run_dir / 'situations.json').exists())
review, _ = finished(cands)
(downloads / 'review-k60-s99 (1).json').write_text(json.dumps(review))
done = self.run_apply(run_dir, downloads=downloads)
self.assertEqual(done.returncode, 0, done.stdout + done.stderr)
self.assertIn('review-k60-s99 (1).json', done.stdout, 'the newest save is the one read')
written = json.loads((run_dir / 'situations.json').read_text())
self.assertEqual(written['situations'][0]['from'], ['C01', 'C02'])
def test_a_damaged_file_is_explained(self):
cands = fixture()
with tempfile.TemporaryDirectory() as tmp:
run_dir = pathlib.Path(tmp)
(run_dir / 'candidates.json').write_text(json.dumps(cands))
(run_dir / 'review-k60-s99.json').write_text('{"entries": [ {"id": "C01",} ]}')
r = self.run_apply(run_dir, downloads=pathlib.Path(tmp, 'none'))
self.assertEqual(r.returncode, 1)
self.assertIn('damaged', r.stdout)
if __name__ == '__main__':
unittest.main()