Files
TheLadder/tools/story-to-pack/recovered/split.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

55 lines
2.2 KiB
Python

"""Split Gutenberg O. Henry volumes into individual stories.
Deterministic: each volume lists its stories in a CONTENTS block as
ALL-CAPS lines, then repeats each title as a heading in the body. No model
involved -- this is the segmentation step we get for free.
"""
import re, json, sys, pathlib
CAPS = re.compile(r'^[ \t]*([A-Z0-9][A-Z0-9 ,.;:\'\"\-’‘“”!?&()À-ÖØ-Þ]{3,70})[ \t]*$')
def strip_boilerplate(text):
i = text.find('*** START')
j = text.find('*** END')
if i >= 0: text = text[text.find('\n', i) + 1:] if j < 0 else text[text.find('\n', i) + 1:j]
return text
def split_volume(path):
raw = pathlib.Path(path).read_text(encoding='utf-8', errors='replace')
title_m = re.search(r'^Title:\s*(.+)$', raw, re.M)
volume = title_m.group(1).strip() if title_m else pathlib.Path(path).stem
body = strip_boilerplate(raw)
lines = body.split('\n')
caps_idx = [(n, CAPS.match(l).group(1).strip()) for n, l in enumerate(lines) if CAPS.match(l)]
if not caps_idx:
return volume, []
# The contents block is the densest early run of caps lines; every title in
# it appears again later as a heading. Use the *second* occurrence.
counts = {}
for n, t in caps_idx:
counts.setdefault(t, []).append(n)
headings = sorted((ns[-1], t) for t, ns in counts.items() if len(ns) >= 2 and len(t.split()) <= 12)
stories = []
for k, (n, t) in enumerate(headings):
end = headings[k + 1][0] if k + 1 < len(headings) else len(lines)
text = '\n'.join(lines[n + 1:end]).strip()
if len(text.split()) >= 400:
stories.append({'volume': volume, 'title': t, 'words': len(text.split()), 'text': text})
return volume, stories
out = []
for p in sorted(pathlib.Path('corpus').glob('*.txt')):
vol, st = split_volume(p)
print(f'{p.name:14} {vol[:44]:46} {len(st):3} stories')
out.extend(st)
print(f'\ntotal stories: {len(out)}')
if out:
ws = sorted(s['words'] for s in out)
print(f'words/story min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]}')
print(f'median pages ~{ws[len(ws)//2]//250}')
pathlib.Path('stories.json').write_text(json.dumps(out, indent=1), encoding='utf-8')