Files
TheLadder/tools/story-to-pack/probe/strip_names.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

80 lines
3.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Replace names in summaries with "someone", deterministically.
A 3B model ignores "never use a name" more often than it obeys it: 27 of 40
pilot summaries still carried names after a reminder and a retry. Names are
exactly the story-identity signal the summaries exist to remove, so this is
done by code rather than by prompt.
The name lexicon comes from the corpus itself: a word counts as a name when it
appears capitalised mid-sentence at least twice and is almost never seen in
lowercase. Words the model introduces that never occur in lowercase anywhere in
the corpus are treated the same way.
python3 strip_names.py summaries.json summaries_clean.json
python3 strip_names.py summaries.json --show # before/after, writes nothing
"""
import collections, json, pathlib, re, sys
HONORIFICS = {'Mr', 'Mrs', 'Ms', 'Miss', 'Dr', 'Mr.', 'Mrs.', 'Ms.', 'Dr.'}
LEAD, TRAIL = '"“‘(', '"”’),.;:!?'
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
cap, low = collections.Counter(), collections.Counter()
for c in chunks:
for sentence in re.split(r'(?<=[.!?"”])\s+', c['text']):
for i, tok in enumerate(re.findall(r"[A-Za-z][A-Za-z'’]*", sentence)):
base = re.sub(r"['’]s$", '', tok)
if base[0].isupper():
if i: cap[base] += 1
else:
low[base.lower()] += 1
NAMES = {w for w, n in cap.items() if n >= 2 and low[w.lower()] <= n * 0.1 and w != 'I'}
def is_name(base, first):
if not base or not base[0].isupper() or base == 'I': return False
if base in NAMES: return True
return not first and low[base.lower()] == 0
def clean(summary):
out, removed, prev_name = [], [], False
for i, w in enumerate(summary.split()):
head = w[:len(w) - len(w.lstrip(LEAD))]
body = w[len(head):]
tail = body[len(body.rstrip(TRAIL)):] if body.rstrip(TRAIL) != body else ''
core = body[:len(body) - len(tail)] if tail else body
if core + '.' in HONORIFICS and tail.startswith('.'):
core, tail = core + '.', tail[1:]
possessive = re.search(r"['’]s$", core)
base = core[:possessive.start()] if possessive else core
if base in HONORIFICS or is_name(base, i == 0 or (out and out[-1].endswith(('.', '!', '?')))):
removed.append(core)
word = head + ('someone’s' if possessive else 'someone') + tail
if prev_name:
out[-1] = word # "Mr. Peters" and "Big Jim Dougherty" collapse to one someone
else:
out.append(word)
prev_name = not tail
else:
out.append(w)
prev_name = False
text = ' '.join(out)
return text[:1].upper() + text[1:], removed
src = pathlib.Path(sys.argv[1])
state = json.loads(src.read_text(encoding='utf-8'))
show = '--show' in sys.argv
changed = 0
for x in state['summaries']:
raw = x.get('raw', x['summary'])
cleaned, removed = clean(raw)
changed += bool(removed)
if show:
print(f"{x['chunk']:3} {'-' if not removed else '*'} {cleaned}")
if removed: print(f" removed: {', '.join(removed)}")
x['raw'], x['summary'], x['names_removed'] = raw, cleaned, removed
print(f'\n{len(NAMES)} names in the corpus lexicon; {changed} of {len(state["summaries"])} summaries changed')
if not show:
dst = pathlib.Path(sys.argv[2])
dst.write_text(json.dumps(state, indent=1), encoding='utf-8')
print(f'wrote {dst}')