Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
95 lines
4.4 KiB
Python
95 lines
4.4 KiB
Python
"""G4: ask the model whether a cluster's summaries share one social situation (SELECTION.md, "G4").
|
|
|
|
STP_OLLAMA=http://host:11434 python3 judge.py qwen2.5:3b-instruct --pilot # first 5 dev clusters
|
|
STP_OLLAMA=http://host:11434 python3 judge.py qwen2.5:3b-instruct # all labelled clusters
|
|
|
|
Writes judge-<model>.json, keyed by a hash of the member set, checkpointed after every call.
|
|
"""
|
|
import hashlib, json, pathlib, random, re, sys, time
|
|
import ollama
|
|
|
|
MODEL = sys.argv[1]
|
|
PILOT = '--pilot' in sys.argv
|
|
SHOW = 15
|
|
|
|
SYSTEM = """You read short summaries of scenes from different stories and decide whether they share one kind of social situation.
|
|
|
|
A social situation is a specific transaction between people: who wants what from whom, and what is at stake. Describe it in your own words.
|
|
|
|
Rules:
|
|
- Find the single situation that the largest number of summaries clearly show.
|
|
- It is normal for many summaries not to fit. Include a summary only if it clearly shows that situation.
|
|
- A shared place, mood, profession or word is not a situation.
|
|
- If no situation is clearly shown by at least three summaries, the situation is NONE.
|
|
|
|
Reply with JSON only: {"situation": "<at most 10 words, or NONE>", "fits": [<numbers of the summaries that clearly show it>]}"""
|
|
|
|
text = {x['chunk']: x['summary'] for x in
|
|
json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries']}
|
|
|
|
def clusters():
|
|
"""(set name, cluster id, members) for every labelled cluster, development first."""
|
|
out = []
|
|
for name in ('dev', 'test'):
|
|
for cid, m in json.loads(pathlib.Path(f'candidates-{name}.json').read_text()).items():
|
|
out.append((name, cid, m))
|
|
for name, mapfile in (('recover', 'recover-map.json'), ('validate', 'validate-map.json'),
|
|
('validate3', 'validate3-map.json')):
|
|
for cid, e in json.loads(pathlib.Path(mapfile).read_text()).items():
|
|
out.append((name, cid, e['members']))
|
|
return out
|
|
|
|
def key(members):
|
|
return hashlib.sha1(','.join(map(str, sorted(members))).encode()).hexdigest()[:16]
|
|
|
|
def parse(raw, shown):
|
|
raw = re.sub(r'<think>.*?</think>', '', raw, flags=re.S).strip()
|
|
m = re.search(r'\{.*\}', raw, flags=re.S)
|
|
if not m: return None
|
|
try:
|
|
d = json.loads(m.group(0))
|
|
except Exception:
|
|
return None
|
|
situation = str(d.get('situation', '')).strip()
|
|
fits = d.get('fits', [])
|
|
if not isinstance(fits, list): return None
|
|
valid = sorted({int(f) for f in fits if isinstance(f, (int, float, str)) and str(f).strip().isdigit()
|
|
and 1 <= int(f) <= shown})
|
|
none = not situation or situation.upper().startswith('NONE')
|
|
return {'situation': situation, 'fits': valid, 'fit': 0.0 if none else len(valid) / shown}
|
|
|
|
out_path = pathlib.Path(f"judge-{MODEL.replace(':', '_').replace('/', '_')}.json")
|
|
cache = json.loads(out_path.read_text()) if out_path.exists() else {}
|
|
todo, seen = [], set()
|
|
for name, cid, members in clusters():
|
|
k = key(members)
|
|
if k in seen: continue
|
|
seen.add(k)
|
|
todo.append((name, cid, members, k))
|
|
if PILOT: todo = todo[:5]
|
|
print(f'{MODEL}: {len(todo)} distinct clusters, {sum(k in cache for *_, k in todo)} already judged', flush=True)
|
|
|
|
t0, done, failures = time.time(), 0, 0
|
|
for name, cid, members, k in todo:
|
|
if k in cache: continue
|
|
sample = sorted(members)
|
|
rnd = random.Random(int(k, 16))
|
|
if len(sample) > SHOW: sample = rnd.sample(sample, SHOW)
|
|
rnd.shuffle(sample)
|
|
user = '\n'.join(f'{i + 1}. {text[c]}' for i, c in enumerate(sample))
|
|
if MODEL.startswith('qwen3'): user += '\n\n/no_think'
|
|
if MODEL.startswith('qwen3'):
|
|
raw, gen, pre = ollama.chat(MODEL, SYSTEM, user, num_ctx=2048, num_predict=300, fmt='json', think=False)
|
|
else:
|
|
raw, gen, pre = ollama.chat(MODEL, SYSTEM, user, num_ctx=2048, num_predict=200, fmt='json')
|
|
parsed = parse(raw, len(sample))
|
|
failures += parsed is None
|
|
cache[k] = {'set': name, 'id': cid, 'size': len(members), 'shown': len(sample), 'raw': raw,
|
|
'parsed': parsed is not None, **(parsed or {'situation': '', 'fits': [], 'fit': 0.0})}
|
|
out_path.write_text(json.dumps(cache, indent=1))
|
|
done += 1
|
|
if PILOT or done % 10 == 0:
|
|
print(f' {done}/{len(todo)} {time.time() - t0:.0f}s parse failures {failures} '
|
|
f'last: {cid} fit {cache[k]["fit"]:.2f} "{cache[k]["situation"][:60]}"', flush=True)
|
|
print(f'done: {done} judged this session in {time.time() - t0:.0f}s; parse failures {failures}')
|