"""Split Gutenberg O. Henry volumes into individual stories. Deterministic: each volume lists its stories in a CONTENTS block as ALL-CAPS lines, then repeats each title as a heading in the body. No model involved -- this is the segmentation step we get for free. """ import re, json, sys, pathlib CAPS = re.compile(r'^[ \t]*([A-Z0-9][A-Z0-9 ,.;:\'\"\-’‘“”!?&()À-ÖØ-Þ]{3,70})[ \t]*$') def strip_boilerplate(text): i = text.find('*** START') j = text.find('*** END') if i >= 0: text = text[text.find('\n', i) + 1:] if j < 0 else text[text.find('\n', i) + 1:j] return text def split_volume(path): raw = pathlib.Path(path).read_text(encoding='utf-8', errors='replace') title_m = re.search(r'^Title:\s*(.+)$', raw, re.M) volume = title_m.group(1).strip() if title_m else pathlib.Path(path).stem body = strip_boilerplate(raw) lines = body.split('\n') caps_idx = [(n, CAPS.match(l).group(1).strip()) for n, l in enumerate(lines) if CAPS.match(l)] if not caps_idx: return volume, [] # The contents block is the densest early run of caps lines; every title in # it appears again later as a heading. Use the *second* occurrence. counts = {} for n, t in caps_idx: counts.setdefault(t, []).append(n) headings = sorted((ns[-1], t) for t, ns in counts.items() if len(ns) >= 2 and len(t.split()) <= 12) stories = [] for k, (n, t) in enumerate(headings): end = headings[k + 1][0] if k + 1 < len(headings) else len(lines) text = '\n'.join(lines[n + 1:end]).strip() if len(text.split()) >= 400: stories.append({'volume': volume, 'title': t, 'words': len(text.split()), 'text': text}) return volume, stories out = [] for p in sorted(pathlib.Path('corpus').glob('*.txt')): vol, st = split_volume(p) print(f'{p.name:14} {vol[:44]:46} {len(st):3} stories') out.extend(st) print(f'\ntotal stories: {len(out)}') if out: ws = sorted(s['words'] for s in out) print(f'words/story min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]}') print(f'median pages ~{ws[len(ws)//2]//250}') pathlib.Path('stories.json').write_text(json.dumps(out, indent=1), encoding='utf-8')