Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
80 lines
3.4 KiB
Python
80 lines
3.4 KiB
Python
"""Replace names in summaries with "someone", deterministically.
|
||
|
||
A 3B model ignores "never use a name" more often than it obeys it: 27 of 40
|
||
pilot summaries still carried names after a reminder and a retry. Names are
|
||
exactly the story-identity signal the summaries exist to remove, so this is
|
||
done by code rather than by prompt.
|
||
|
||
The name lexicon comes from the corpus itself: a word counts as a name when it
|
||
appears capitalised mid-sentence at least twice and is almost never seen in
|
||
lowercase. Words the model introduces that never occur in lowercase anywhere in
|
||
the corpus are treated the same way.
|
||
|
||
python3 strip_names.py summaries.json summaries_clean.json
|
||
python3 strip_names.py summaries.json --show # before/after, writes nothing
|
||
"""
|
||
import collections, json, pathlib, re, sys
|
||
|
||
HONORIFICS = {'Mr', 'Mrs', 'Ms', 'Miss', 'Dr', 'Mr.', 'Mrs.', 'Ms.', 'Dr.'}
|
||
LEAD, TRAIL = '"“‘(', '"”’),.;:!?'
|
||
|
||
chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8'))
|
||
cap, low = collections.Counter(), collections.Counter()
|
||
for c in chunks:
|
||
for sentence in re.split(r'(?<=[.!?"”])\s+', c['text']):
|
||
for i, tok in enumerate(re.findall(r"[A-Za-z][A-Za-z'’]*", sentence)):
|
||
base = re.sub(r"['’]s$", '', tok)
|
||
if base[0].isupper():
|
||
if i: cap[base] += 1
|
||
else:
|
||
low[base.lower()] += 1
|
||
NAMES = {w for w, n in cap.items() if n >= 2 and low[w.lower()] <= n * 0.1 and w != 'I'}
|
||
|
||
def is_name(base, first):
|
||
if not base or not base[0].isupper() or base == 'I': return False
|
||
if base in NAMES: return True
|
||
return not first and low[base.lower()] == 0
|
||
|
||
def clean(summary):
|
||
out, removed, prev_name = [], [], False
|
||
for i, w in enumerate(summary.split()):
|
||
head = w[:len(w) - len(w.lstrip(LEAD))]
|
||
body = w[len(head):]
|
||
tail = body[len(body.rstrip(TRAIL)):] if body.rstrip(TRAIL) != body else ''
|
||
core = body[:len(body) - len(tail)] if tail else body
|
||
if core + '.' in HONORIFICS and tail.startswith('.'):
|
||
core, tail = core + '.', tail[1:]
|
||
possessive = re.search(r"['’]s$", core)
|
||
base = core[:possessive.start()] if possessive else core
|
||
if base in HONORIFICS or is_name(base, i == 0 or (out and out[-1].endswith(('.', '!', '?')))):
|
||
removed.append(core)
|
||
word = head + ('someone’s' if possessive else 'someone') + tail
|
||
if prev_name:
|
||
out[-1] = word # "Mr. Peters" and "Big Jim Dougherty" collapse to one someone
|
||
else:
|
||
out.append(word)
|
||
prev_name = not tail
|
||
else:
|
||
out.append(w)
|
||
prev_name = False
|
||
text = ' '.join(out)
|
||
return text[:1].upper() + text[1:], removed
|
||
|
||
src = pathlib.Path(sys.argv[1])
|
||
state = json.loads(src.read_text(encoding='utf-8'))
|
||
show = '--show' in sys.argv
|
||
changed = 0
|
||
for x in state['summaries']:
|
||
raw = x.get('raw', x['summary'])
|
||
cleaned, removed = clean(raw)
|
||
changed += bool(removed)
|
||
if show:
|
||
print(f"{x['chunk']:3} {'-' if not removed else '*'} {cleaned}")
|
||
if removed: print(f" removed: {', '.join(removed)}")
|
||
x['raw'], x['summary'], x['names_removed'] = raw, cleaned, removed
|
||
print(f'\n{len(NAMES)} names in the corpus lexicon; {changed} of {len(state["summaries"])} summaries changed')
|
||
if not show:
|
||
dst = pathlib.Path(sys.argv[2])
|
||
dst.write_text(json.dumps(state, indent=1), encoding='utf-8')
|
||
print(f'wrote {dst}')
|