The pipeline's embed-and-cluster step is dead, and this commit holds both the evidence for that and the step proposed to replace it. Predicaments. Scenes are re-described as "what the person is up against", with no names, jobs or places, then embedded and clustered (redescribe.py, topic_words.py, topic_share.py). The pilot chose qwen3:14b over 3b by reading both side by side. Two defects the pilot exposed are fixed: split.py missed titles in quotes and a contents subtitle after a dash, so three stories had been merged into their neighbours, and strip_names.py read New York place names as people. The corrected corpus is probe/v2 (97 stories, 839 scenes); carry_summaries.py reuses the 829 unchanged v1 summaries. Topic share fell from 20% to 13% at k=60, short of the pre-registered 10%. Hand references. Three corpora were read scene by scene and written up by hand, under the same prompt rules the local models get, as a baseline to judge them against: O. Henry (probe/v2/claude, 839 scenes, 20 situations), Wharton's Descent of Man (probe/wharton, 262 scenes, 16 groups) and Jacobs's The Lady of the Barge (probe/jacobs, 157 scenes, 19 groups). Each has its own README and a readable page. No inference was used for any of them. Catalogue. probe/catalogue maps every hand group in the three references onto 36 situation entries, with an answer key per corpus and one recurrence rule applied to all three. classify.py assigns a scene one entry or none, leave-one-corpus- out; score.py checks it against the key, with a self-test on random labels. Why clustering is out: hand-written predicaments, embedded and clustered exactly as the model's were, agree with the hand grouping at ARI 0.05 — no better than the 14B text's 0.07. Better rewriting cannot rescue it. Embeddings cannot even shortlist: the hand label is the nearest entry 13% of the time and in the top 8 half the time. The classification runs are not here. The dev and test runs are pre-registered in probe/catalogue/README.md with the bar set beforehand, and are blocked on the inference host, whose GPU has fallen off the PCIe bus three times. The 30-scene partial output in out/ is not a result. Review page. The situation review is now a browser page rather than JSON edited by hand (review_page.py, review_page_logic.cjs with Node tests, format schema v2). It has never been rendered in a real browser. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014BygvsUXV9eU6oHkTCkKZ1
174 lines
8.1 KiB
Python
174 lines
8.1 KiB
Python
"""Tests for the saved review's checks, the page renderer, and apply_review.py.
|
|
|
|
python3 -m unittest test_review
|
|
"""
|
|
import json, os, pathlib, subprocess, sys, tempfile, unittest
|
|
import review_format as rf
|
|
import review_page
|
|
|
|
HERE = pathlib.Path(__file__).resolve().parent
|
|
|
|
|
|
def group(cid, kind, stories):
|
|
"""A synthetic group whose n-th scene comes from stories[n]."""
|
|
return {'kind': kind, 'stats': {'stories': len(set(stories))}, 'gate_failures': [], 'role_hints': [],
|
|
'members': [{'member': rf.member_id(cid, n + 1), 'chunk': abs(hash((cid, n))) % 100000, 'story': s,
|
|
'title': f'Story {s}', 'summary': f'{cid} scene {n + 1}'} for n, s in enumerate(stories)]}
|
|
|
|
|
|
def fixture():
|
|
return {'schema_version': rf.SCHEMA_VERSION,
|
|
'run': {'id': 'k60-s99', 'fingerprint': 'abc123', 'scenes': 21, 'stories': 13},
|
|
'candidates': {'C01': group('C01', 'accepted', [1, 2, 3, 4, 5, 6, 7, 8]),
|
|
'C02': group('C02', 'accepted', [9, 10, 11, 12, 13, 14]),
|
|
'B01': group('B01', 'borderline', [1, 3, 5, 7, 9, 11, 13])}}
|
|
|
|
|
|
def blank_review(cands):
|
|
return {'schema_version': rf.SCHEMA_VERSION, 'run': cands['run']['id'], 'fingerprint': cands['run']['fingerprint'],
|
|
'entries': [rf.blank_entry(cid, c['kind']) for cid, c in cands['candidates'].items()]}
|
|
|
|
|
|
def finished(cands):
|
|
review = blank_review(cands)
|
|
e = {x['id']: x for x in review['entries']}
|
|
e['C01'].update(decision='keep', actor='lodger', wants='more time to pay', counterpart='a landlady',
|
|
excluded=['C01.02'])
|
|
e['C02'].update(decision='same', same_as='C01')
|
|
return review, e
|
|
|
|
|
|
class ReviewChecks(unittest.TestCase):
|
|
def test_a_blank_review_asks_only_for_the_required_groups(self):
|
|
cands = fixture()
|
|
errors, _, result = rf.validate(cands, blank_review(cands))
|
|
self.assertIsNone(result)
|
|
self.assertTrue(any(err.startswith('Group 1: not decided yet') for err in errors), errors)
|
|
self.assertTrue(any(err.startswith('Group 2: not decided yet') for err in errors), errors)
|
|
self.assertFalse(any('Extra group' in err for err in errors))
|
|
|
|
def test_a_finished_review_merges_and_leaves_out_unticked_scenes(self):
|
|
cands = fixture()
|
|
review, _ = finished(cands)
|
|
errors, _, result = rf.validate(cands, review)
|
|
self.assertEqual(errors, [])
|
|
self.assertEqual(len(result['situations']), 1)
|
|
s = result['situations'][0]
|
|
self.assertEqual(s['situation'], 'A lodger wants more time to pay from a landlady.')
|
|
self.assertEqual(s['from'], ['C01', 'C02'])
|
|
self.assertEqual(s['size'], 8 - 1 + 6)
|
|
self.assertNotIn('C01.02', [m['member'] for m in s['members']])
|
|
self.assertEqual(result['left_out'], ['B01'])
|
|
|
|
def test_a_kept_group_must_be_named(self):
|
|
cands = fixture()
|
|
review, e = finished(cands)
|
|
e['C01']['wants'] = ' '
|
|
errors, _, _ = rf.validate(cands, review)
|
|
self.assertTrue(any('Group 1: kept, but its name is not finished' in err and 'what they want' in err
|
|
for err in errors), errors)
|
|
|
|
def test_same_as_a_group_that_is_not_kept(self):
|
|
cands = fixture()
|
|
review, e = finished(cands)
|
|
e['C01']['decision'] = 'drop'
|
|
errors, _, _ = rf.validate(cands, review)
|
|
self.assertTrue(any(err.startswith('Group 2: marked the same as Group 1, but Group 1 is not kept')
|
|
for err in errors), errors)
|
|
|
|
def test_too_few_ticked_scenes_are_refused_with_what_to_do(self):
|
|
cands = fixture()
|
|
review, e = finished(cands)
|
|
e['C02'].update(decision='keep', same_as=None, actor='clerk', wants='a raise', counterpart='an employer',
|
|
excluded=['C02.01', 'C02.02', 'C02.03'])
|
|
errors, _, result = rf.validate(cands, review)
|
|
self.assertIsNone(result)
|
|
self.assertTrue(any(err.startswith('Group 2: 3 ticked scenes from 3 stories') and 'Tick more scenes' in err
|
|
for err in errors), errors)
|
|
|
|
def test_an_older_review_file_is_explained(self):
|
|
cands = fixture()
|
|
review, _ = finished(cands)
|
|
review['schema_version'] = 1
|
|
errors, _, _ = rf.validate(cands, review)
|
|
self.assertEqual(len(errors), 1)
|
|
self.assertIn('different version of the review tool', errors[0])
|
|
|
|
def test_a_review_for_different_groups_is_refused(self):
|
|
cands = fixture()
|
|
review, _ = finished(cands)
|
|
review['fingerprint'] = 'zzz999'
|
|
errors, _, _ = rf.validate(cands, review)
|
|
self.assertTrue(any('different set of groups' in err for err in errors), errors)
|
|
|
|
def test_an_extra_group_cannot_be_dropped_only_left_out(self):
|
|
cands = fixture()
|
|
review, e = finished(cands)
|
|
e['B01']['decision'] = 'drop'
|
|
errors, _, _ = rf.validate(cands, review)
|
|
self.assertTrue(any(err.startswith('Extra group 1') for err in errors), errors)
|
|
|
|
|
|
class PageRendering(unittest.TestCase):
|
|
def test_the_page_embeds_everything_and_nothing_can_close_its_scripts_early(self):
|
|
cands = fixture()
|
|
cands['candidates']['C01']['members'][0]['summary'] = 'A trap </script><b>bold</b>'
|
|
page = review_page.render_page(cands)
|
|
self.assertNotIn('/*__DATA__*/', page)
|
|
self.assertNotIn('/*__LOGIC__*/', page)
|
|
self.assertNotIn('__RUN_ID__', page)
|
|
self.assertEqual(page.count('</script>'), 2, 'only the two real script tags may close')
|
|
self.assertIn('Situation review · k60-s99', page)
|
|
|
|
def test_role_words_are_suggested_from_the_scenes(self):
|
|
members = [{'summary': 'A lodger begs the landlady for time.'},
|
|
{'summary': 'The landlady turns a lodger out.'},
|
|
{'summary': 'Someone hides from the landlady.'}]
|
|
self.assertEqual(review_page.role_hints(members)[:2], ['landlady', 'lodger'])
|
|
|
|
|
|
class ApplyReview(unittest.TestCase):
|
|
def run_apply(self, run_dir, *extra, downloads):
|
|
env = dict(os.environ, STP_DOWNLOADS=str(downloads))
|
|
return subprocess.run([sys.executable, 'apply_review.py', str(run_dir), *extra], cwd=HERE,
|
|
capture_output=True, text=True, env=env)
|
|
|
|
def test_it_finds_the_saved_review_in_downloads_and_writes_nothing_until_it_passes(self):
|
|
cands = fixture()
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
run_dir, downloads = pathlib.Path(tmp, 'run'), pathlib.Path(tmp, 'Downloads')
|
|
run_dir.mkdir(); downloads.mkdir()
|
|
(run_dir / 'candidates.json').write_text(json.dumps(cands))
|
|
|
|
nothing = self.run_apply(run_dir, downloads=downloads)
|
|
self.assertEqual(nothing.returncode, 1)
|
|
self.assertIn('No saved review found', nothing.stdout)
|
|
|
|
(downloads / 'review-k60-s99.json').write_text(json.dumps(blank_review(cands)))
|
|
unfinished = self.run_apply(run_dir, downloads=downloads)
|
|
self.assertEqual(unfinished.returncode, 1)
|
|
self.assertIn('The review is not finished', unfinished.stdout)
|
|
self.assertFalse((run_dir / 'situations.json').exists())
|
|
|
|
review, _ = finished(cands)
|
|
(downloads / 'review-k60-s99 (1).json').write_text(json.dumps(review))
|
|
done = self.run_apply(run_dir, downloads=downloads)
|
|
self.assertEqual(done.returncode, 0, done.stdout + done.stderr)
|
|
self.assertIn('review-k60-s99 (1).json', done.stdout, 'the newest save is the one read')
|
|
written = json.loads((run_dir / 'situations.json').read_text())
|
|
self.assertEqual(written['situations'][0]['from'], ['C01', 'C02'])
|
|
|
|
def test_a_damaged_file_is_explained(self):
|
|
cands = fixture()
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
run_dir = pathlib.Path(tmp)
|
|
(run_dir / 'candidates.json').write_text(json.dumps(cands))
|
|
(run_dir / 'review-k60-s99.json').write_text('{"entries": [ {"id": "C01",} ]}')
|
|
r = self.run_apply(run_dir, downloads=pathlib.Path(tmp, 'none'))
|
|
self.assertEqual(r.returncode, 1)
|
|
self.assertIn('damaged', r.stdout)
|
|
|
|
|
|
if __name__ == '__main__':
|
|
unittest.main()
|