Files
TheLadder/tools/story-to-pack/probe/se_split.py
T
JesseMarkowitzandClaude Opus 5.5 3a4758172d story-to-pack: add the corpora and the catalogue v2/v3
Fourteen corpora split and classified: aesop, bierce, chekhov, holmes,
keefe, lawson, lorimer, maupassant, nobody, plaintales, poe, torchy,
wallingford and winesburg, each with its splitter and the hand-written
groups and pages; catalogue v2 and v3; and the shared splitters
gutenberg_chunks.py, se_split.py and se_build.py.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019rwKTmug58sEsJ72AuWsEi
2026-10-07 06:05:51 -04:00

52 lines
2.5 KiB
Python

"""One splitter for any Standard Ebooks single-page collection.
Every story is an <article> or <section> whose epub:type carries `se:short-story`;
all <p> inside it (including numbered sub-parts) belong to it. No per-volume patterns.
Footnote markers (<a epub:type="noteref">) are dropped from titles and text."""
import re, sys
from html.parser import HTMLParser
HEAD = ('h2', 'h3', 'h4', 'h5', 'h6')
class P(HTMLParser):
def __init__(self):
super().__init__(convert_charrefs=True)
self.stories, self.cur, self.depth, self.head, self.inp, self.note = [], None, 0, False, False, 0
def handle_starttag(self, tag, a):
a = dict(a)
if tag == 'a' and 'noteref' in a.get('epub:type', ''): self.note += 1; return
if self.note and tag == 'a': self.note += 1; return
if tag in ('article', 'section'):
if self.cur is None and 'se:short-story' in a.get('epub:type', '').split():
self.cur, self.depth = {'title': '', 'paras': []}, 1
elif self.cur is not None:
self.depth += 1
elif self.cur is not None and tag in HEAD and not self.cur['title'] and 'title' in a.get('epub:type', ''):
self.head, self.buf = True, []
elif self.cur is not None and tag == 'p':
self.inp, self.pbuf = True, []
def handle_endtag(self, tag):
if tag == 'a' and self.note: self.note -= 1; return
if tag in ('article', 'section') and self.cur is not None:
self.depth -= 1
if self.depth == 0: self.stories.append(self.cur); self.cur = None
elif tag in HEAD and self.head:
self.cur['title'] = re.sub(r'\s+', ' ', ''.join(self.buf)).strip(); self.head = False
elif tag == 'p' and self.inp:
self.inp = False
t = re.sub(r'\s+', ' ', ''.join(self.pbuf)).strip()
if t: self.cur['paras'].append(t)
def handle_data(self, d):
if self.note: return
if self.head: self.buf.append(d)
if self.inp: self.pbuf.append(d)
def split(path):
p = P(); p.feed(open(path, encoding='utf-8').read()); return p.stories
if __name__ == '__main__':
for f in sys.argv[1:]:
st = [(s['title'], sum(len(x.split()) for x in s['paras'])) for s in split(f)]
if not st: print(f'{f}: no se:short-story sections'); continue
print(f'{f.split("/")[-1][3:-5]:36} {len(st):4} stories {sum(w for _, w in st):8,} words | '
f'first {st[0][0][:26]!r} | shortest {min(st, key=lambda x: x[1])}')