Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
55 lines
2.2 KiB
Python
55 lines
2.2 KiB
Python
"""Split Gutenberg O. Henry volumes into individual stories.
|
|
|
|
Deterministic: each volume lists its stories in a CONTENTS block as
|
|
ALL-CAPS lines, then repeats each title as a heading in the body. No model
|
|
involved -- this is the segmentation step we get for free.
|
|
"""
|
|
import re, json, sys, pathlib
|
|
|
|
CAPS = re.compile(r'^[ \t]*([A-Z0-9][A-Z0-9 ,.;:\'\"\-’‘“”!?&()À-ÖØ-Þ]{3,70})[ \t]*$')
|
|
|
|
def strip_boilerplate(text):
|
|
i = text.find('*** START')
|
|
j = text.find('*** END')
|
|
if i >= 0: text = text[text.find('\n', i) + 1:] if j < 0 else text[text.find('\n', i) + 1:j]
|
|
return text
|
|
|
|
def split_volume(path):
|
|
raw = pathlib.Path(path).read_text(encoding='utf-8', errors='replace')
|
|
title_m = re.search(r'^Title:\s*(.+)$', raw, re.M)
|
|
volume = title_m.group(1).strip() if title_m else pathlib.Path(path).stem
|
|
body = strip_boilerplate(raw)
|
|
|
|
lines = body.split('\n')
|
|
caps_idx = [(n, CAPS.match(l).group(1).strip()) for n, l in enumerate(lines) if CAPS.match(l)]
|
|
if not caps_idx:
|
|
return volume, []
|
|
|
|
# The contents block is the densest early run of caps lines; every title in
|
|
# it appears again later as a heading. Use the *second* occurrence.
|
|
counts = {}
|
|
for n, t in caps_idx:
|
|
counts.setdefault(t, []).append(n)
|
|
headings = sorted((ns[-1], t) for t, ns in counts.items() if len(ns) >= 2 and len(t.split()) <= 12)
|
|
|
|
stories = []
|
|
for k, (n, t) in enumerate(headings):
|
|
end = headings[k + 1][0] if k + 1 < len(headings) else len(lines)
|
|
text = '\n'.join(lines[n + 1:end]).strip()
|
|
if len(text.split()) >= 400:
|
|
stories.append({'volume': volume, 'title': t, 'words': len(text.split()), 'text': text})
|
|
return volume, stories
|
|
|
|
out = []
|
|
for p in sorted(pathlib.Path('corpus').glob('*.txt')):
|
|
vol, st = split_volume(p)
|
|
print(f'{p.name:14} {vol[:44]:46} {len(st):3} stories')
|
|
out.extend(st)
|
|
|
|
print(f'\ntotal stories: {len(out)}')
|
|
if out:
|
|
ws = sorted(s['words'] for s in out)
|
|
print(f'words/story min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]}')
|
|
print(f'median pages ~{ws[len(ws)//2]//250}')
|
|
pathlib.Path('stories.json').write_text(json.dumps(out, indent=1), encoding='utf-8')
|