"""Split Gutenberg O. Henry volumes into individual stories. Deterministic: each volume lists its stories in a CONTENTS block as ALL-CAPS lines, then repeats each title as a heading in the body. No model involved -- this is the segmentation step we get for free. """ import re, json, sys, pathlib ROMAN = re.compile(r'^\s*[IVXLC]+\.\s+') CAPS = re.compile(r'^[ \t]*([A-Z0-9"“‘][A-Z0-9 ,.;:\'\"\-—’‘“”!?&()À-ÖØ-Þ]{3,70})[ \t]*$') # A title may be quoted ("LITTLE SPECK IN GARNERED FRUIT"), and a contents entry may carry a subtitle # after a dash that the heading does not (“THE GUILTY PARTY”—AN EAST SIDE TRAGEDY). v1 handled # neither and silently merged three stories into the story before them. The title is what comes # before the dash, without quotes. QUOTES = '"“”‘’\'' def title_of(line): return CAPS.match(line).group(1).split('—')[0].strip().strip(QUOTES).strip() def strip_boilerplate(text): i = text.find('*** START') j = text.find('*** END') if i >= 0: text = text[text.find('\n', i) + 1:] if j < 0 else text[text.find('\n', i) + 1:j] return text def split_volume(path): raw = pathlib.Path(path).read_text(encoding='utf-8', errors='replace') title_m = re.search(r'^Title:\s*(.+)$', raw, re.M) volume = title_m.group(1).strip() if title_m else pathlib.Path(path).stem body = strip_boilerplate(raw) lines = body.split('\n') # Two volumes number their contents (" I. THE VOICE OF THE CITY") while the # body heading is the bare title, so strip the numeral before matching. lines = [ROMAN.sub('', l) for l in lines] caps_idx = [(n, title_of(l)) for n, l in enumerate(lines) if CAPS.match(l)] caps_idx = [(n, t) for n, t in caps_idx if len(t) >= 4] if not caps_idx: return volume, [] # The contents block is the densest early run of caps lines; every title in # it appears again later as a heading. Use the *second* occurrence. counts = {} for n, t in caps_idx: counts.setdefault(t, []).append(n) headings = sorted((ns[-1], t) for t, ns in counts.items() if len(ns) >= 2 and len(t.split()) <= 12) stories = [] for k, (n, t) in enumerate(headings): end = headings[k + 1][0] if k + 1 < len(headings) else len(lines) text = '\n'.join(lines[n + 1:end]).strip() if len(text.split()) >= 400: stories.append({'volume': volume, 'title': t, 'words': len(text.split()), 'text': text}) return volume, stories out = [] for p in sorted(pathlib.Path('corpus').glob('*.txt')): vol, st = split_volume(p) print(f'{p.name:14} {vol[:44]:46} {len(st):3} stories') out.extend(st) print(f'\ntotal stories: {len(out)}') if out: ws = sorted(s['words'] for s in out) print(f'words/story min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]}') print(f'median pages ~{ws[len(ws)//2]//250}') pathlib.Path('stories.json').write_text(json.dumps(out, indent=1), encoding='utf-8')