"""Shared tail for the per-corpus Gutenberg splitters: stories -> stories.json + chunks.json, with the same scene rule as chunk.py, jacobs/split_se.py and se_build.py (paragraph boundaries, ~320 words, a tail kept only if it is at least 80 words), and a per-story report.""" import json, pathlib, re TARGET = 320 def paragraphs(body, drop=()): out = [] for p in re.split(r'\n\s*\n', body): p = re.sub(r'\s+', ' ', p).strip() if p and not any(re.fullmatch(d, p) for d in drop): out.append(p) return out def write(d, stories): d = pathlib.Path(d) chunks = [] for si, s in enumerate(stories): buf, n = [], 0 for p in s['text'].split('\n\n'): w = len(p.split()) if n + w > TARGET and buf: chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'], 'text': ' '.join(buf), 'words': n}) buf, n = [], 0 buf.append(p); n += w if buf and n >= 80: chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'], 'text': ' '.join(buf), 'words': n}) (d / 'stories.json').write_text(json.dumps(stories, indent=1, ensure_ascii=False), encoding='utf-8') (d / 'chunks.json').write_text(json.dumps(chunks, ensure_ascii=False), encoding='utf-8') for si, s in enumerate(stories): print(f"{si:2} {s['title'][:44]:44} {s['words']:6} words {sum(c['story'] == si for c in chunks):3} scenes") ws = sorted(c['words'] for c in chunks) print(f'stories {len(stories)} scenes {len(chunks)} words/scene min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]} total {sum(ws):,}') def gutenberg_body(path): raw = pathlib.Path(path).read_text(encoding='utf-8').replace('\r', '') raw = raw[raw.index('*** START OF THE PROJECT GUTENBERG'):] raw = raw[raw.index('\n') + 1:] return raw[:raw.index('*** END OF THE PROJECT GUTENBERG')].split('\n')