Fourteen corpora split and classified: aesop, bierce, chekhov, holmes, keefe, lawson, lorimer, maupassant, nobody, plaintales, poe, torchy, wallingford and winesburg, each with its splitter and the hand-written groups and pages; catalogue v2 and v3; and the shared splitters gutenberg_chunks.py, se_split.py and se_build.py. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019rwKTmug58sEsJ72AuWsEi
31 lines
1.7 KiB
Python
31 lines
1.7 KiB
Python
"""Standard Ebooks collection -> stories.json + chunks.json, for any corpus directory.
|
|
Usage: python3 se_build.py <dir> "<Volume (Author, year)>"
|
|
Reads the single HTML file in <dir>/corpus/, splits stories with se_split.py (the `se:short-story`
|
|
markup), then cuts scenes with the same rule as chunk.py, jacobs/split_se.py and the Gutenberg
|
|
splitters: paragraph boundaries, ~320 words, a tail kept only if it is at least 80 words."""
|
|
import json, pathlib, sys
|
|
from se_split import split
|
|
|
|
TARGET = 320
|
|
d, volume = pathlib.Path(sys.argv[1]), sys.argv[2]
|
|
[src] = list((d / 'corpus').glob('*.html'))
|
|
stories = [{'volume': volume, 'title': s['title'], 'words': sum(len(p.split()) for p in s['paras']),
|
|
'text': '\n\n'.join(s['paras'])} for s in split(src)]
|
|
chunks = []
|
|
for si, s in enumerate(stories):
|
|
buf, n = [], 0
|
|
for p in s['text'].split('\n\n'):
|
|
w = len(p.split())
|
|
if n + w > TARGET and buf:
|
|
chunks.append({'story': si, 'title': s['title'], 'volume': volume, 'text': ' '.join(buf), 'words': n})
|
|
buf, n = [], 0
|
|
buf.append(p); n += w
|
|
if buf and n >= 80:
|
|
chunks.append({'story': si, 'title': s['title'], 'volume': volume, 'text': ' '.join(buf), 'words': n})
|
|
(d / 'stories.json').write_text(json.dumps(stories, indent=1, ensure_ascii=False), encoding='utf-8')
|
|
(d / 'chunks.json').write_text(json.dumps(chunks, ensure_ascii=False), encoding='utf-8')
|
|
for si, s in enumerate(stories):
|
|
print(f"{si:2} {s['title'][:40]:40} {s['words']:6} words {sum(c['story'] == si for c in chunks):3} scenes")
|
|
ws = sorted(c['words'] for c in chunks)
|
|
print(f'stories {len(stories)} scenes {len(chunks)} words/scene min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]} total {sum(ws):,}')
|