Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
38 lines
1.7 KiB
Python
38 lines
1.7 KiB
Python
"""Embed texts, checkpointing after every batch so an interrupted run resumes.
|
|
|
|
python3 embed.py chunks.json embeddings.json # the scenes' prose
|
|
python3 embed.py summaries.json summary_embeddings.json # one-line situations
|
|
python3 embed.py chunks.json embeddings.json --limit 48 # pilot
|
|
|
|
Input is a JSON list of objects with a 'text' field (chunks.json), or of
|
|
strings/objects with a 'summary' field (summaries.json).
|
|
"""
|
|
import argparse, json, os, pathlib, time
|
|
import ollama
|
|
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument('src'); ap.add_argument('dst')
|
|
ap.add_argument('--limit', type=int, default=None)
|
|
ap.add_argument('--batch', type=int, default=16)
|
|
args = ap.parse_args()
|
|
MODEL = os.environ.get('STP_EMBED_MODEL', 'nomic-embed-text')
|
|
|
|
items = json.loads(pathlib.Path(args.src).read_text(encoding='utf-8'))
|
|
if isinstance(items, dict): items = items['summaries']
|
|
texts = [x if isinstance(x, str) else (x.get('summary') or x['text']) for x in items]
|
|
if args.limit: texts = texts[:args.limit]
|
|
|
|
out = pathlib.Path(args.dst)
|
|
state = json.loads(out.read_text()) if out.exists() else {'model': MODEL, 'vectors': []}
|
|
if state['model'] != MODEL: raise SystemExit(f'{out} was made with {state["model"]}, not {MODEL}')
|
|
done = state['vectors']
|
|
print(f'{len(texts)} texts, resuming at {len(done)}, model {MODEL}', flush=True)
|
|
|
|
t0, start = time.time(), len(done)
|
|
for i in range(start, len(texts), args.batch):
|
|
done.extend(ollama.embed(MODEL, texts[i:i + args.batch]))
|
|
out.write_text(json.dumps(state))
|
|
el = time.time() - t0
|
|
print(f' {len(done)}/{len(texts)} {el:.0f}s ({(len(done) - start) / max(el, 1):.2f}/s)', flush=True)
|
|
print(f'done: {len(done)} vectors, {time.time() - t0:.0f}s this session')
|