Fourteen corpora split and classified: aesop, bierce, chekhov, holmes, keefe, lawson, lorimer, maupassant, nobody, plaintales, poe, torchy, wallingford and winesburg, each with its splitter and the hand-written groups and pages; catalogue v2 and v3; and the shared splitters gutenberg_chunks.py, se_split.py and se_build.py. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019rwKTmug58sEsJ72AuWsEi
32 lines
2.0 KiB
Python
32 lines
2.0 KiB
Python
"""Split the Project Gutenberg text of Kipling, *Plain Tales from the Hills* (eBook #1858), into stories.
|
||
Titles are taken from the body, where each story opens with an upper-case heading line ending in a
|
||
full stop, because the contents list has typos ("LESPETH" for the body's "LISPETH"). Every body heading
|
||
is checked against the contents by closest match, and the counts must agree. Each story's opening
|
||
verse epigraph is kept: it is short, and removing it would need a per-story rule."""
|
||
import difflib, re, sys
|
||
sys.path.insert(0, '..')
|
||
from gutenberg_chunks import gutenberg_body, paragraphs, write
|
||
|
||
VOLUME = 'Plain Tales from the Hills (Rudyard Kipling, 1888)'
|
||
lines = gutenberg_body('corpus/plain-tales-from-the-hills.txt')
|
||
c0 = next(i for i, l in enumerate(lines) if l.strip() == 'CONTENTS')
|
||
b0 = next(i for i, l in enumerate(lines) if i > c0 and l.strip() == 'PLAIN TALES FROM THE HILLS')
|
||
contents = [l.strip() for l in lines[c0 + 1:b0] if l.strip()]
|
||
def is_head(l): # upper case apart from a "Mc"; the full stop is usually there but not on THE HOUSE OF SUDDHOO
|
||
l = l.strip()
|
||
return bool(re.fullmatch(r"[A-Z][A-Za-z0-9 '’,\-]+\.?", l)) and len(l) > 3 and l.replace('Mc', 'MC') == l.upper()
|
||
heads = [i for i, l in enumerate(lines) if i > b0 and is_head(l) and not lines[i - 1].strip() and not lines[i + 1].strip()]
|
||
titles = [lines[i].strip().rstrip('.').replace('Mc', 'MC') for i in heads]
|
||
assert len(titles) == len(contents), (len(titles), len(contents), set(titles) ^ set(contents))
|
||
for t, c in zip(titles, contents):
|
||
if t != c.rstrip('.'):
|
||
r = difflib.SequenceMatcher(None, t, c).ratio()
|
||
print(f' heading {t!r} vs contents {c!r} ({r:.2f})'); assert r > 0.8
|
||
stories = []
|
||
for n, a in enumerate(heads):
|
||
b = heads[n + 1] if n + 1 < len(heads) else len(lines)
|
||
paras = paragraphs('\n'.join(lines[a + 1:b]))
|
||
stories.append({'volume': VOLUME, 'title': titles[n].title(), 'words': sum(len(p.split()) for p in paras),
|
||
'text': '\n\n'.join(paras)})
|
||
write('.', stories)
|