Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
27 lines
1.1 KiB
Python
27 lines
1.1 KiB
Python
"""Cut stories into scene-sized chunks on paragraph boundaries."""
|
|
import json, pathlib, re
|
|
|
|
TARGET = 320 # words; a scene is a situation with people in it
|
|
|
|
stories = json.loads(pathlib.Path('stories.json').read_text(encoding='utf-8'))
|
|
chunks = []
|
|
for si, s in enumerate(stories):
|
|
paras = [p.strip() for p in re.split(r'\n\s*\n', s['text']) if p.strip()]
|
|
buf, n = [], 0
|
|
for p in paras:
|
|
w = len(p.split())
|
|
if n + w > TARGET and buf:
|
|
chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'],
|
|
'text': ' '.join(buf), 'words': n})
|
|
buf, n = [], 0
|
|
buf.append(p); n += w
|
|
if buf and n >= 80:
|
|
chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'],
|
|
'text': ' '.join(buf), 'words': n})
|
|
|
|
pathlib.Path('chunks.json').write_text(json.dumps(chunks), encoding='utf-8')
|
|
ws = sorted(c['words'] for c in chunks)
|
|
print(f'chunks: {len(chunks)} from {len(stories)} stories')
|
|
print(f'words/chunk min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]}')
|
|
print(f'total words: {sum(ws):,}')
|