Files
TheLadder/tools/story-to-pack/probe/judge.py
T
JesseMarkowitzandClaude Opus 5 fa3769d0fe Add story-to-pack research and a structured situation review
Research toward building a content pack from a story corpus, kept on its own
branch and independent of the game. Records the selection experiments against
blind labels, and settles selection as gate G2 followed by a human review:
review.py writes REVIEW.md and a review.json form, apply_review.py checks the
filled form and writes situations.json for the next stage.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
2026-09-15 06:59:35 -04:00

95 lines
4.4 KiB
Python

"""G4: ask the model whether a cluster's summaries share one social situation (SELECTION.md, "G4").
STP_OLLAMA=http://host:11434 python3 judge.py qwen2.5:3b-instruct --pilot # first 5 dev clusters
STP_OLLAMA=http://host:11434 python3 judge.py qwen2.5:3b-instruct # all labelled clusters
Writes judge-<model>.json, keyed by a hash of the member set, checkpointed after every call.
"""
import hashlib, json, pathlib, random, re, sys, time
import ollama
MODEL = sys.argv[1]
PILOT = '--pilot' in sys.argv
SHOW = 15
SYSTEM = """You read short summaries of scenes from different stories and decide whether they share one kind of social situation.
A social situation is a specific transaction between people: who wants what from whom, and what is at stake. Describe it in your own words.
Rules:
- Find the single situation that the largest number of summaries clearly show.
- It is normal for many summaries not to fit. Include a summary only if it clearly shows that situation.
- A shared place, mood, profession or word is not a situation.
- If no situation is clearly shown by at least three summaries, the situation is NONE.
Reply with JSON only: {"situation": "<at most 10 words, or NONE>", "fits": [<numbers of the summaries that clearly show it>]}"""
text = {x['chunk']: x['summary'] for x in
json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries']}
def clusters():
"""(set name, cluster id, members) for every labelled cluster, development first."""
out = []
for name in ('dev', 'test'):
for cid, m in json.loads(pathlib.Path(f'candidates-{name}.json').read_text()).items():
out.append((name, cid, m))
for name, mapfile in (('recover', 'recover-map.json'), ('validate', 'validate-map.json'),
('validate3', 'validate3-map.json')):
for cid, e in json.loads(pathlib.Path(mapfile).read_text()).items():
out.append((name, cid, e['members']))
return out
def key(members):
return hashlib.sha1(','.join(map(str, sorted(members))).encode()).hexdigest()[:16]
def parse(raw, shown):
raw = re.sub(r'<think>.*?</think>', '', raw, flags=re.S).strip()
m = re.search(r'\{.*\}', raw, flags=re.S)
if not m: return None
try:
d = json.loads(m.group(0))
except Exception:
return None
situation = str(d.get('situation', '')).strip()
fits = d.get('fits', [])
if not isinstance(fits, list): return None
valid = sorted({int(f) for f in fits if isinstance(f, (int, float, str)) and str(f).strip().isdigit()
and 1 <= int(f) <= shown})
none = not situation or situation.upper().startswith('NONE')
return {'situation': situation, 'fits': valid, 'fit': 0.0 if none else len(valid) / shown}
out_path = pathlib.Path(f"judge-{MODEL.replace(':', '_').replace('/', '_')}.json")
cache = json.loads(out_path.read_text()) if out_path.exists() else {}
todo, seen = [], set()
for name, cid, members in clusters():
k = key(members)
if k in seen: continue
seen.add(k)
todo.append((name, cid, members, k))
if PILOT: todo = todo[:5]
print(f'{MODEL}: {len(todo)} distinct clusters, {sum(k in cache for *_, k in todo)} already judged', flush=True)
t0, done, failures = time.time(), 0, 0
for name, cid, members, k in todo:
if k in cache: continue
sample = sorted(members)
rnd = random.Random(int(k, 16))
if len(sample) > SHOW: sample = rnd.sample(sample, SHOW)
rnd.shuffle(sample)
user = '\n'.join(f'{i + 1}. {text[c]}' for i, c in enumerate(sample))
if MODEL.startswith('qwen3'): user += '\n\n/no_think'
if MODEL.startswith('qwen3'):
raw, gen, pre = ollama.chat(MODEL, SYSTEM, user, num_ctx=2048, num_predict=300, fmt='json', think=False)
else:
raw, gen, pre = ollama.chat(MODEL, SYSTEM, user, num_ctx=2048, num_predict=200, fmt='json')
parsed = parse(raw, len(sample))
failures += parsed is None
cache[k] = {'set': name, 'id': cid, 'size': len(members), 'shown': len(sample), 'raw': raw,
'parsed': parsed is not None, **(parsed or {'situation': '', 'fits': [], 'fit': 0.0})}
out_path.write_text(json.dumps(cache, indent=1))
done += 1
if PILOT or done % 10 == 0:
print(f' {done}/{len(todo)} {time.time() - t0:.0f}s parse failures {failures} '
f'last: {cid} fit {cache[k]["fit"]:.2f} "{cache[k]["situation"][:60]}"', flush=True)
print(f'done: {done} judged this session in {time.time() - t0:.0f}s; parse failures {failures}')