story-to-pack: add the corpora and the catalogue v2/v3
Fourteen corpora split and classified: aesop, bierce, chekhov, holmes, keefe, lawson, lorimer, maupassant, nobody, plaintales, poe, torchy, wallingford and winesburg, each with its splitter and the hand-written groups and pages; catalogue v2 and v3; and the shared splitters gutenberg_chunks.py, se_split.py and se_build.py. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019rwKTmug58sEsJ72AuWsEi
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
e5617b86ba
commit
3a4758172d
@@ -0,0 +1,39 @@
|
||||
"""Shared tail for the per-corpus Gutenberg splitters: stories -> stories.json + chunks.json, with the
|
||||
same scene rule as chunk.py, jacobs/split_se.py and se_build.py (paragraph boundaries, ~320 words,
|
||||
a tail kept only if it is at least 80 words), and a per-story report."""
|
||||
import json, pathlib, re
|
||||
|
||||
TARGET = 320
|
||||
|
||||
def paragraphs(body, drop=()):
|
||||
out = []
|
||||
for p in re.split(r'\n\s*\n', body):
|
||||
p = re.sub(r'\s+', ' ', p).strip()
|
||||
if p and not any(re.fullmatch(d, p) for d in drop): out.append(p)
|
||||
return out
|
||||
|
||||
def write(d, stories):
|
||||
d = pathlib.Path(d)
|
||||
chunks = []
|
||||
for si, s in enumerate(stories):
|
||||
buf, n = [], 0
|
||||
for p in s['text'].split('\n\n'):
|
||||
w = len(p.split())
|
||||
if n + w > TARGET and buf:
|
||||
chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'], 'text': ' '.join(buf), 'words': n})
|
||||
buf, n = [], 0
|
||||
buf.append(p); n += w
|
||||
if buf and n >= 80:
|
||||
chunks.append({'story': si, 'title': s['title'], 'volume': s['volume'], 'text': ' '.join(buf), 'words': n})
|
||||
(d / 'stories.json').write_text(json.dumps(stories, indent=1, ensure_ascii=False), encoding='utf-8')
|
||||
(d / 'chunks.json').write_text(json.dumps(chunks, ensure_ascii=False), encoding='utf-8')
|
||||
for si, s in enumerate(stories):
|
||||
print(f"{si:2} {s['title'][:44]:44} {s['words']:6} words {sum(c['story'] == si for c in chunks):3} scenes")
|
||||
ws = sorted(c['words'] for c in chunks)
|
||||
print(f'stories {len(stories)} scenes {len(chunks)} words/scene min {ws[0]} median {ws[len(ws)//2]} max {ws[-1]} total {sum(ws):,}')
|
||||
|
||||
def gutenberg_body(path):
|
||||
raw = pathlib.Path(path).read_text(encoding='utf-8').replace('\r', '')
|
||||
raw = raw[raw.index('*** START OF THE PROJECT GUTENBERG'):]
|
||||
raw = raw[raw.index('\n') + 1:]
|
||||
return raw[:raw.index('*** END OF THE PROJECT GUTENBERG')].split('\n')
|
||||
Reference in New Issue
Block a user