The pipeline's embed-and-cluster step is dead, and this commit holds both the evidence for that and the step proposed to replace it. Predicaments. Scenes are re-described as "what the person is up against", with no names, jobs or places, then embedded and clustered (redescribe.py, topic_words.py, topic_share.py). The pilot chose qwen3:14b over 3b by reading both side by side. Two defects the pilot exposed are fixed: split.py missed titles in quotes and a contents subtitle after a dash, so three stories had been merged into their neighbours, and strip_names.py read New York place names as people. The corrected corpus is probe/v2 (97 stories, 839 scenes); carry_summaries.py reuses the 829 unchanged v1 summaries. Topic share fell from 20% to 13% at k=60, short of the pre-registered 10%. Hand references. Three corpora were read scene by scene and written up by hand, under the same prompt rules the local models get, as a baseline to judge them against: O. Henry (probe/v2/claude, 839 scenes, 20 situations), Wharton's Descent of Man (probe/wharton, 262 scenes, 16 groups) and Jacobs's The Lady of the Barge (probe/jacobs, 157 scenes, 19 groups). Each has its own README and a readable page. No inference was used for any of them. Catalogue. probe/catalogue maps every hand group in the three references onto 36 situation entries, with an answer key per corpus and one recurrence rule applied to all three. classify.py assigns a scene one entry or none, leave-one-corpus- out; score.py checks it against the key, with a self-test on random labels. Why clustering is out: hand-written predicaments, embedded and clustered exactly as the model's were, agree with the hand grouping at ARI 0.05 — no better than the 14B text's 0.07. Better rewriting cannot rescue it. Embeddings cannot even shortlist: the hand label is the nearest entry 13% of the time and in the top 8 half the time. The classification runs are not here. The dev and test runs are pre-registered in probe/catalogue/README.md with the bar set beforehand, and are blocked on the inference host, whose GPU has fallen off the PCIe bus three times. The 30-scene partial output in out/ is not a result. Review page. The situation review is now a browser page rather than JSON edited by hand (review_page.py, review_page_logic.cjs with Node tests, format schema v2). It has never been rendered in a real browser. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014BygvsUXV9eU6oHkTCkKZ1
105 lines
5.1 KiB
Python
105 lines
5.1 KiB
Python
"""Replace names in summaries with "someone", deterministically.
|
||
|
||
A 3B model ignores "never use a name" more often than it obeys it: 27 of 40
|
||
pilot summaries still carried names after a reminder and a retry. Names are
|
||
exactly the story-identity signal the summaries exist to remove, so this is
|
||
done by code rather than by prompt.
|
||
|
||
The name lexicon comes from the corpus itself: a word counts as a name when it
|
||
appears capitalised mid-sentence at least twice and is almost never seen in
|
||
lowercase. Words the model introduces that never occur in lowercase anywhere in
|
||
the corpus are treated the same way.
|
||
|
||
python3 strip_names.py summaries.json summaries_clean.json
|
||
python3 strip_names.py summaries.json --show # before/after, writes nothing
|
||
"""
|
||
import collections, json, pathlib, re, sys
|
||
|
||
HONORIFICS = {'Mr', 'Mrs', 'Ms', 'Miss', 'Dr', 'Mr.', 'Mrs.', 'Ms.', 'Dr.'}
|
||
LEAD, TRAIL = '"“‘(', '"”’),.;:!?'
|
||
|
||
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
|
||
cap, low = collections.Counter(), collections.Counter()
|
||
for c in chunks:
|
||
for sentence in re.split(r'(?<=[.!?"”])\s+', c['text']):
|
||
for i, tok in enumerate(re.findall(r"[A-Za-z][A-Za-z'’]*", sentence)):
|
||
base = re.sub(r"['’]s$", '', tok)
|
||
if base[0].isupper():
|
||
if i: cap[base] += 1
|
||
else:
|
||
low[base.lower()] += 1
|
||
NAMES = {w for w, n in cap.items() if n >= 2 and low[w.lower()] <= n * 0.1 and w != 'I'}
|
||
|
||
def is_name(base, first):
|
||
if not base or not base[0].isupper() or base == 'I' or base in NOT_NAMES: return False
|
||
if base in NAMES: return True
|
||
return not first and low[base.lower()] == 0
|
||
|
||
# Every story is set in New York, and its places are capitalised and never lower-case, so the
|
||
# lexicon took them for people ("New someone"). They become "the city" before names are looked for.
|
||
PLACES = re.compile(r"\b(the )?(?:New York(?: City)?|Manhattan|Broadway|Brooklyn|Harlem|Bowery|"
|
||
r"Coney Island|New Jersey|Jersey City|Madison Square(?: Garden)?|Union Square|"
|
||
r"Central Park|Fifth Avenue|Wall Street)(ers?)?(['’]s)?\b(?=\s+[\"“‘]?([a-z][a-z-]*))?")
|
||
# After a place, these words mean it was not used as an adjective ("to New York to find", "New York is").
|
||
NOT_ADJECTIVAL = set("""to and or but is was are were has had where with for in on at as by from of that which
|
||
who while after before when than the a an his her their again itself""".split())
|
||
|
||
def _place(m):
|
||
article, dweller, possessive, following = m.group(1), m.group(2), m.group(3), m.group(4)
|
||
if dweller: # "a New Yorker", "New Yorkers"
|
||
return (article or '') + ('city dwellers' if dweller == 'ers' else 'city dweller') + (possessive or '')
|
||
if possessive: # "New York's soul"
|
||
return "the city's"
|
||
if following and following not in NOT_ADJECTIVAL:
|
||
return (article or '') + 'city' # "a New York girl", "the hidden Broadway hotel"
|
||
return 'the city' # "to New York", "the Bowery"
|
||
|
||
# Capitalised, never lower-case, and not people.
|
||
NOT_NAMES = {'Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday', 'Saturday', 'Sunday',
|
||
'Christmas', 'Thanksgiving', 'Easter'}
|
||
NAMES -= NOT_NAMES
|
||
|
||
def clean(summary):
|
||
summary = PLACES.sub(_place, summary)
|
||
out, removed, prev_name = [], [], False
|
||
for i, w in enumerate(summary.split()):
|
||
head = w[:len(w) - len(w.lstrip(LEAD))]
|
||
body = w[len(head):]
|
||
tail = body[len(body.rstrip(TRAIL)):] if body.rstrip(TRAIL) != body else ''
|
||
core = body[:len(body) - len(tail)] if tail else body
|
||
if core + '.' in HONORIFICS and tail.startswith('.'):
|
||
core, tail = core + '.', tail[1:]
|
||
possessive = re.search(r"['’]s$", core)
|
||
base = core[:possessive.start()] if possessive else core
|
||
if base in HONORIFICS or is_name(base, i == 0 or (out and out[-1].endswith(('.', '!', '?')))):
|
||
removed.append(core)
|
||
word = head + ('someone’s' if possessive else 'someone') + tail
|
||
if prev_name:
|
||
out[-1] = word # "Mr. Peters" and "Big Jim Dougherty" collapse to one someone
|
||
else:
|
||
out.append(word)
|
||
prev_name = not tail
|
||
else:
|
||
out.append(w)
|
||
prev_name = False
|
||
text = ' '.join(out)
|
||
return text[:1].upper() + text[1:], removed
|
||
|
||
src = pathlib.Path(sys.argv[1])
|
||
state = json.loads(src.read_text(encoding='utf-8'))
|
||
show = '--show' in sys.argv
|
||
changed = 0
|
||
for x in state['summaries']:
|
||
raw = x.get('raw', x['summary'])
|
||
cleaned, removed = clean(raw)
|
||
changed += bool(removed)
|
||
if show:
|
||
print(f"{x['chunk']:3} {'-' if not removed else '*'} {cleaned}")
|
||
if removed: print(f" removed: {', '.join(removed)}")
|
||
x['raw'], x['summary'], x['names_removed'] = raw, cleaned, removed
|
||
print(f'\n{len(NAMES)} names in the corpus lexicon; {changed} of {len(state["summaries"])} summaries changed')
|
||
if not show:
|
||
dst = pathlib.Path(sys.argv[2])
|
||
dst.write_text(json.dumps(state, indent=1), encoding='utf-8')
|
||
print(f'wrote {dst}')
|