Research toward building a content pack from a story corpus, kept on its own branch and independent of the game. Records the selection experiments against blind labels, and settles selection as gate G2 followed by a human review: review.py writes REVIEW.md and a review.json form, apply_review.py checks the filled form and writes situations.json for the next stage. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C6UDQ9o6L6Ey173U7XVou6
50 lines
2.2 KiB
Python
50 lines
2.2 KiB
Python
"""The one place that talks to an inference host.
|
|
|
|
The host is shared with other development work, so this is deliberately
|
|
gentle: one request at a time, a fixed pause between requests, and a short
|
|
keep_alive so the model is not left occupying the GPU after a run.
|
|
|
|
Configuration comes from the environment, never from a file in this tree:
|
|
STP_OLLAMA base URL of an Ollama server, e.g. http://inference.lan:11434
|
|
STP_PAUSE seconds between requests (default 0.5)
|
|
"""
|
|
import json, os, sys, time, urllib.request
|
|
|
|
BASE = os.environ.get('STP_OLLAMA', '').rstrip('/')
|
|
PAUSE = float(os.environ.get('STP_PAUSE', '0.5'))
|
|
KEEP_ALIVE = '2m'
|
|
|
|
if not BASE:
|
|
sys.exit('STP_OLLAMA is not set -- refusing to guess an inference host')
|
|
|
|
def post(route, body, timeout=300):
|
|
req = urllib.request.Request(BASE + route, data=json.dumps(body).encode(),
|
|
headers={'Content-Type': 'application/json'})
|
|
for attempt in range(4):
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
result = json.load(r)
|
|
time.sleep(PAUSE)
|
|
return result
|
|
except Exception as e:
|
|
if attempt == 3: raise
|
|
print(f' retry {attempt + 1}: {e}', flush=True)
|
|
time.sleep(5 * (attempt + 1))
|
|
|
|
def embed(model, texts):
|
|
return post('/api/embed', {'model': model, 'input': texts, 'keep_alive': KEEP_ALIVE})['embeddings']
|
|
|
|
def chat(model, system, user, num_ctx=2048, num_predict=80, fmt=None, think=None):
|
|
# num_ctx is pinned small on purpose: the 2026-09-10 hang was a 16k-context
|
|
# variant prefilling on CPU. A scene is ~450 tokens with the prompt.
|
|
body = {'model': model, 'stream': False, 'keep_alive': KEEP_ALIVE,
|
|
'messages': [{'role': 'system', 'content': system},
|
|
{'role': 'user', 'content': user}],
|
|
'options': {'num_ctx': num_ctx, 'num_predict': num_predict, 'temperature': 0}}
|
|
if fmt: body['format'] = fmt
|
|
# Thinking models (qwen3) otherwise spend the token budget in a separate thinking
|
|
# channel and return empty content.
|
|
if think is not None: body['think'] = think
|
|
d = post('/api/chat', body)
|
|
return d['message']['content'].strip(), d.get('eval_duration', 0), d.get('prompt_eval_duration', 0)
|