Fourteen corpora split and classified: aesop, bierce, chekhov, holmes, keefe, lawson, lorimer, maupassant, nobody, plaintales, poe, torchy, wallingford and winesburg, each with its splitter and the hand-written groups and pages; catalogue v2 and v3; and the shared splitters gutenberg_chunks.py, se_split.py and se_build.py. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019rwKTmug58sEsJ72AuWsEi
52 lines
2.5 KiB
Python
52 lines
2.5 KiB
Python
"""One splitter for any Standard Ebooks single-page collection.
|
|
Every story is an <article> or <section> whose epub:type carries `se:short-story`;
|
|
all <p> inside it (including numbered sub-parts) belong to it. No per-volume patterns.
|
|
Footnote markers (<a epub:type="noteref">) are dropped from titles and text."""
|
|
import re, sys
|
|
from html.parser import HTMLParser
|
|
|
|
HEAD = ('h2', 'h3', 'h4', 'h5', 'h6')
|
|
|
|
class P(HTMLParser):
|
|
def __init__(self):
|
|
super().__init__(convert_charrefs=True)
|
|
self.stories, self.cur, self.depth, self.head, self.inp, self.note = [], None, 0, False, False, 0
|
|
def handle_starttag(self, tag, a):
|
|
a = dict(a)
|
|
if tag == 'a' and 'noteref' in a.get('epub:type', ''): self.note += 1; return
|
|
if self.note and tag == 'a': self.note += 1; return
|
|
if tag in ('article', 'section'):
|
|
if self.cur is None and 'se:short-story' in a.get('epub:type', '').split():
|
|
self.cur, self.depth = {'title': '', 'paras': []}, 1
|
|
elif self.cur is not None:
|
|
self.depth += 1
|
|
elif self.cur is not None and tag in HEAD and not self.cur['title'] and 'title' in a.get('epub:type', ''):
|
|
self.head, self.buf = True, []
|
|
elif self.cur is not None and tag == 'p':
|
|
self.inp, self.pbuf = True, []
|
|
def handle_endtag(self, tag):
|
|
if tag == 'a' and self.note: self.note -= 1; return
|
|
if tag in ('article', 'section') and self.cur is not None:
|
|
self.depth -= 1
|
|
if self.depth == 0: self.stories.append(self.cur); self.cur = None
|
|
elif tag in HEAD and self.head:
|
|
self.cur['title'] = re.sub(r'\s+', ' ', ''.join(self.buf)).strip(); self.head = False
|
|
elif tag == 'p' and self.inp:
|
|
self.inp = False
|
|
t = re.sub(r'\s+', ' ', ''.join(self.pbuf)).strip()
|
|
if t: self.cur['paras'].append(t)
|
|
def handle_data(self, d):
|
|
if self.note: return
|
|
if self.head: self.buf.append(d)
|
|
if self.inp: self.pbuf.append(d)
|
|
|
|
def split(path):
|
|
p = P(); p.feed(open(path, encoding='utf-8').read()); return p.stories
|
|
|
|
if __name__ == '__main__':
|
|
for f in sys.argv[1:]:
|
|
st = [(s['title'], sum(len(x.split()) for x in s['paras'])) for s in split(f)]
|
|
if not st: print(f'{f}: no se:short-story sections'); continue
|
|
print(f'{f.split("/")[-1][3:-5]:36} {len(st):4} stories {sum(w for _, w in st):8,} words | '
|
|
f'first {st[0][0][:26]!r} | shortest {min(st, key=lambda x: x[1])}')
|