"""Replace names in summaries with "someone", deterministically. A 3B model ignores "never use a name" more often than it obeys it: 27 of 40 pilot summaries still carried names after a reminder and a retry. Names are exactly the story-identity signal the summaries exist to remove, so this is done by code rather than by prompt. The name lexicon comes from the corpus itself: a word counts as a name when it appears capitalised mid-sentence at least twice and is almost never seen in lowercase. Words the model introduces that never occur in lowercase anywhere in the corpus are treated the same way. python3 strip_names.py summaries.json summaries_clean.json python3 strip_names.py summaries.json --show # before/after, writes nothing """ import collections, json, pathlib, re, sys HONORIFICS = {'Mr', 'Mrs', 'Ms', 'Miss', 'Dr', 'Mr.', 'Mrs.', 'Ms.', 'Dr.'} LEAD, TRAIL = '"“‘(', '"”’),.;:!?' chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8')) cap, low = collections.Counter(), collections.Counter() for c in chunks: for sentence in re.split(r'(?<=[.!?"”])\s+', c['text']): for i, tok in enumerate(re.findall(r"[A-Za-z][A-Za-z'’]*", sentence)): base = re.sub(r"['’]s$", '', tok) if base[0].isupper(): if i: cap[base] += 1 else: low[base.lower()] += 1 NAMES = {w for w, n in cap.items() if n >= 2 and low[w.lower()] <= n * 0.1 and w != 'I'} def is_name(base, first): if not base or not base[0].isupper() or base == 'I' or base in NOT_NAMES: return False if base in NAMES: return True return not first and low[base.lower()] == 0 # Every story is set in New York, and its places are capitalised and never lower-case, so the # lexicon took them for people ("New someone"). They become "the city" before names are looked for. PLACES = re.compile(r"\b(the )?(?:New York(?: City)?|Manhattan|Broadway|Brooklyn|Harlem|Bowery|" r"Coney Island|New Jersey|Jersey City|Madison Square(?: Garden)?|Union Square|" r"Central Park|Fifth Avenue|Wall Street)(ers?)?(['’]s)?\b(?=\s+[\"“‘]?([a-z][a-z-]*))?") # After a place, these words mean it was not used as an adjective ("to New York to find", "New York is"). NOT_ADJECTIVAL = set("""to and or but is was are were has had where with for in on at as by from of that which who while after before when than the a an his her their again itself""".split()) def _place(m): article, dweller, possessive, following = m.group(1), m.group(2), m.group(3), m.group(4) if dweller: # "a New Yorker", "New Yorkers" return (article or '') + ('city dwellers' if dweller == 'ers' else 'city dweller') + (possessive or '') if possessive: # "New York's soul" return "the city's" if following and following not in NOT_ADJECTIVAL: return (article or '') + 'city' # "a New York girl", "the hidden Broadway hotel" return 'the city' # "to New York", "the Bowery" # Capitalised, never lower-case, and not people. NOT_NAMES = {'Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday', 'Saturday', 'Sunday', 'Christmas', 'Thanksgiving', 'Easter'} NAMES -= NOT_NAMES def clean(summary): summary = PLACES.sub(_place, summary) out, removed, prev_name = [], [], False for i, w in enumerate(summary.split()): head = w[:len(w) - len(w.lstrip(LEAD))] body = w[len(head):] tail = body[len(body.rstrip(TRAIL)):] if body.rstrip(TRAIL) != body else '' core = body[:len(body) - len(tail)] if tail else body if core + '.' in HONORIFICS and tail.startswith('.'): core, tail = core + '.', tail[1:] possessive = re.search(r"['’]s$", core) base = core[:possessive.start()] if possessive else core if base in HONORIFICS or is_name(base, i == 0 or (out and out[-1].endswith(('.', '!', '?')))): removed.append(core) word = head + ('someone’s' if possessive else 'someone') + tail if prev_name: out[-1] = word # "Mr. Peters" and "Big Jim Dougherty" collapse to one someone else: out.append(word) prev_name = not tail else: out.append(w) prev_name = False text = ' '.join(out) return text[:1].upper() + text[1:], removed src = pathlib.Path(sys.argv[1]) state = json.loads(src.read_text(encoding='utf-8')) show = '--show' in sys.argv changed = 0 for x in state['summaries']: raw = x.get('raw', x['summary']) cleaned, removed = clean(raw) changed += bool(removed) if show: print(f"{x['chunk']:3} {'-' if not removed else '*'} {cleaned}") if removed: print(f" removed: {', '.join(removed)}") x['raw'], x['summary'], x['names_removed'] = raw, cleaned, removed print(f'\n{len(NAMES)} names in the corpus lexicon; {changed} of {len(state["summaries"])} summaries changed') if not show: dst = pathlib.Path(sys.argv[2]) dst.write_text(json.dumps(state, indent=1), encoding='utf-8') print(f'wrote {dst}')