"""Cut stories into scene-sized chunks on paragraph boundaries.""" import json, pathlib, re TARGET = 320 # words; a scene is a situation with people in it stories = json.loads(pathlib.Path('stories.json').read_text(encoding='utf-8')) chunks = [] for si, s in enumerate(stories): paras = [p.strip() for p in re.split(r'\n\s*\n', s['text']) if p.strip()] buf, n = [], 0 for p in paras: w = len(p.split()) if n + w > TARGET and buf: chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'], 'text': ' '.join(buf), 'words': n}) buf, n = [], 0 buf.append(p); n += w if buf and n >= 80: chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'], 'text': ' '.join(buf), 'words': n}) pathlib.Path('chunks.json').write_text(json.dumps(chunks), encoding='utf-8') ws = sorted(c['words'] for c in chunks) print(f'chunks: {len(chunks)} from {len(stories)} stories') print(f'words/chunk min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]}') print(f'total words: {sum(ws):,}')