"""Embed texts, checkpointing after every batch so an interrupted run resumes. python3 embed.py chunks.json embeddings.json # the scenes' prose python3 embed.py summaries.json summary_embeddings.json # one-line situations python3 embed.py chunks.json embeddings.json --limit 48 # pilot Input is a JSON list of objects with a 'text' field (chunks.json), or of strings/objects with a 'summary' field (summaries.json). """ import argparse, json, os, pathlib, time import ollama ap = argparse.ArgumentParser() ap.add_argument('src'); ap.add_argument('dst') ap.add_argument('--limit', type=int, default=None) ap.add_argument('--batch', type=int, default=16) args = ap.parse_args() MODEL = os.environ.get('STP_EMBED_MODEL', 'nomic-embed-text') items = json.loads(pathlib.Path(args.src).read_text(encoding='utf-8')) if isinstance(items, dict): items = items['summaries'] texts = [x if isinstance(x, str) else (x.get('summary') or x['text']) for x in items] if args.limit: texts = texts[:args.limit] out = pathlib.Path(args.dst) state = json.loads(out.read_text()) if out.exists() else {'model': MODEL, 'vectors': []} if state['model'] != MODEL: raise SystemExit(f'{out} was made with {state["model"]}, not {MODEL}') done = state['vectors'] print(f'{len(texts)} texts, resuming at {len(done)}, model {MODEL}', flush=True) t0, start = time.time(), len(done) for i in range(start, len(texts), args.batch): done.extend(ollama.embed(MODEL, texts[i:i + args.batch])) out.write_text(json.dumps(state)) el = time.time() - t0 print(f' {len(done)}/{len(texts)} {el:.0f}s ({(len(done) - start) / max(el, 1):.2f}/s)', flush=True) print(f'done: {len(done)} vectors, {time.time() - t0:.0f}s this session')