Fourteen corpora split and classified: aesop, bierce, chekhov, holmes, keefe, lawson, lorimer, maupassant, nobody, plaintales, poe, torchy, wallingford and winesburg, each with its splitter and the hand-written groups and pages; catalogue v2 and v3; and the shared splitters gutenberg_chunks.py, se_split.py and se_build.py. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019rwKTmug58sEsJ72AuWsEi
43 lines
2.6 KiB
Python
43 lines
2.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Assemble records.json and groups.json from batches/ and groups_src.py, and check them."""
|
|
import collections, glob, json, pathlib
|
|
from groups_src import G, STANDING
|
|
here = pathlib.Path(__file__).parent
|
|
chunks = json.loads((here.parent / 'chunks.json').read_text())
|
|
stories = json.loads((here.parent / 'stories.json').read_text())
|
|
recs = [r for f in sorted(glob.glob(str(here / 'batches/s*.json'))) for r in json.loads(open(f).read())]
|
|
assert [r['chunk'] for r in recs] == list(range(len(chunks))), 'every scene, once, in order'
|
|
assert all(chunks[r['chunk']]['story'] == r['story'] for r in recs)
|
|
|
|
seen = collections.Counter(m for g in G for m in g['members'])
|
|
dup = [m for m, n in seen.items() if n > 1]
|
|
assert not dup, f'scenes in two groups: {dup}'
|
|
by = {r['chunk']: r for r in recs}
|
|
out = []
|
|
for g in G:
|
|
assert 3 <= len(g['options']) <= 5, g['id']
|
|
st = sorted({by[m]['story'] for m in g['members']})
|
|
stand = collections.Counter(STANDING[by[m]['story']] for m in g['members'])
|
|
out.append({
|
|
'id': g['id'], 'name': g['name'], 'predicament': g['predicament'],
|
|
'sentence': f"A {g['actor']} wants {g['wants']} from {g['counterpart']}.",
|
|
'actor': g['actor'], 'wants': g['wants'], 'counterpart': g['counterpart'],
|
|
'stories': st, 'story_titles': [stories[s]['title'] for s in st],
|
|
'standing': dict(stand), 'stage': stand.most_common(1)[0][0],
|
|
'members': [{'chunk': m, 'story': by[m]['story'], 'predicament': by[m]['predicament']} for m in g['members']],
|
|
'options': [{'text': t, 'source': s, 'moves': mv,
|
|
'in_group': (int(s.split(':')[1]) in g['members']) if s.startswith('scene:') else None}
|
|
for t, s, mv in g['options']],
|
|
})
|
|
unassigned = [r['chunk'] for r in recs if r['chunk'] not in seen]
|
|
(here / 'records.json').write_text(json.dumps(recs, indent=1, ensure_ascii=False))
|
|
(here / 'groups.json').write_text(json.dumps({'groups': out, 'unassigned': unassigned}, indent=1, ensure_ascii=False))
|
|
multi = sum(len(g['stories']) >= 2 for g in out)
|
|
print(f'{len(recs)} scenes, {len(out)} groups ({multi} span 2+ stories), '
|
|
f'{sum(seen.values())} assigned, {len(unassigned)} unassigned: {unassigned}')
|
|
for g in out:
|
|
print(f"{g['id']} {len(g['members']):2} scenes {len(g['stories'])} stories stage {g['stage']:6} {g['name']}")
|
|
src = collections.Counter(o['source'].split(':')[0] for g in out for o in g['options'])
|
|
print('option sources:', dict(src), '| scene options citing a scene outside their group:',
|
|
sum(o['in_group'] is False for g in out for o in g['options']))
|