"""Embed chunks on the local endpoint, checkpointing as we go.""" import json, pathlib, ssl, time, urllib.request URL = 'https://inference.lan:8443/v1/embeddings' # placeholder: the real host is never committed MODEL = 'nomic-embed-text' BATCH = 24 ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE chunks = json.loads(pathlib.Path('chunks.json').read_text(encoding='utf-8')) out = pathlib.Path('embeddings.json') done = json.loads(out.read_text()) if out.exists() else [] start = len(done) print(f'{len(chunks)} chunks, resuming at {start}', flush=True) t0 = time.time() for i in range(start, len(chunks), BATCH): batch = [c['text'] for c in chunks[i:i + BATCH]] body = json.dumps({'model': MODEL, 'input': batch}).encode() req = urllib.request.Request(URL, data=body, headers={'Content-Type': 'application/json'}) for attempt in range(4): try: with urllib.request.urlopen(req, timeout=180, context=ctx) as r: d = json.load(r) done.extend(e['embedding'] for e in d['data']) break except Exception as e: if attempt == 3: raise print(f' retry {attempt+1} at {i}: {e}', flush=True) time.sleep(3 * (attempt + 1)) out.write_text(json.dumps(done)) el = time.time() - t0 n = len(done) print(f' {n}/{len(chunks)} {el:.0f}s ({n/max(el,1):.1f}/s)', flush=True) print(f'done: {len(done)} embeddings in {time.time()-t0:.0f}s')