"""v1.1 WP-A1 corrective: a cold model is loaded, not guessed about. A1's accounting caught a real cold-model turn: `/api/ps` knew nothing because the model was not resident, `/api/show` found no `num_ctx`, the window was therefore unverified, and the prompt was built to the configured 16,384. Ollama loaded the model at its own 4,096 default, read 2,050 of the 13,875 tokens and answered 200. Detection was right. The case is also preventable: once the model is loaded its window is readable. So before an unverified turn is assembled, the application asks the configured server, once, to load the model (`POST /api/generate` with a model and no prompt, which Ollama answers with `"done_reason": "load"` and no text), probes again, and builds the turn to whatever that probe says. A window still unverified afterwards changes nothing: the configured budget stands and the post-response accounting still watches for a cut prompt. python -m pytest tests/test_v11_cold_window.py -v """ import asyncio import httpx import pytest from fastapi import Depends from fastapi.testclient import TestClient from sqlalchemy import text from sqlalchemy.orm import undefer from app import auth, contextwindow, limits, models from app.context import builder from app.database import Base, SessionLocal, engine, get_db from app.main import app from app.providers.base import ProviderError from app.routers import adventures from fakes import ScriptedProvider ENDPOINT = "http://127.0.0.1:11434/v1" MODEL = "qwen2.5:3b-instruct" @pytest.fixture(autouse=True) def _clear_window_cache(): contextwindow.cache_clear() yield contextwindow.cache_clear() class ColdOllama: """The shapes a real Ollama 0.33 returned, with a model that starts cold. `/api/ps` lists only loaded models. `/api/show` carries no `num_ctx`. `/api/generate` with no prompt loads the model at `load_window`, exactly as the real server answered: HTTP 200, `"response": ""`, `"done_reason": "load"`. """ def __init__(self, *, loaded=None, load_window=4096, generate_status=200, report_after_load=True): self.loaded = dict(loaded or {}) self.load_window = load_window self.generate_status = generate_status self.report_after_load = report_after_load self.requests: list[tuple[str, str, dict | None]] = [] def handler(self, request: httpx.Request) -> httpx.Response: body = None if request.content: import json body = json.loads(request.content) self.requests.append((request.method, str(request.url), body)) path = request.url.path if path == "/api/ps": return httpx.Response(200, json={"models": [ {"name": name, "model": name, "context_length": tokens} for name, tokens in self.loaded.items() ]}) if path == "/api/show": return httpx.Response(200, json={ "model_info": {"qwen2.context_length": 32768}, "parameters": ""}) if path == "/api/generate": if self.generate_status != 200: return httpx.Response(self.generate_status, json={"error": "model not found"}) if self.report_after_load: self.loaded[body["model"]] = self.load_window return httpx.Response(200, json={ "model": body["model"], "response": "", "done": True, "done_reason": "load"}) return httpx.Response(404) def paths(self): return [httpx.URL(url).path for _method, url, _body in self.requests] @pytest.fixture() def server(monkeypatch): def install(fake: ColdOllama): original = httpx.AsyncClient def build(*args, **kwargs): kwargs.pop("verify", None) return original(*args, transport=httpx.MockTransport(fake.handler), **kwargs) monkeypatch.setattr(contextwindow.httpx, "AsyncClient", build) return fake return install # ------------------------------------------------------------ ensure_window def test_a_cold_model_is_loaded_once_and_its_window_verified(server): fake = server(ColdOllama(loaded={}, load_window=4096)) window, preflight = asyncio.run(contextwindow.ensure_window(ENDPOINT, MODEL)) assert (window.tokens, window.source, window.verified) == (4096, contextwindow.LOADED, True) assert preflight == {"attempted": True, "loaded": True, "verified_before": False, "verified_after": True, "detail": "the server loaded the model (load)"} assert fake.paths() == ["/api/ps", "/api/show", "/api/generate", "/api/ps"] # One load request, naming the model and nothing else: no prompt, so no text. warms = [body for _m, url, body in fake.requests if url.endswith("/api/generate")] assert warms == [{"model": MODEL}] def test_an_already_loaded_model_is_not_warmed(server): fake = server(ColdOllama(loaded={MODEL: 16384})) window, preflight = asyncio.run(contextwindow.ensure_window(ENDPOINT, MODEL)) assert window.verified and window.tokens == 16384 assert preflight["attempted"] is False assert "/api/generate" not in fake.paths() def test_a_model_that_loads_but_still_cannot_be_read_stays_unverified(server): """A server that loads the model but whose `/api/ps` still cannot say. The existing unknown path stands: no guessed window, the configured budget kept.""" fake = server(ColdOllama(loaded={}, report_after_load=False)) window, preflight = asyncio.run(contextwindow.ensure_window(ENDPOINT, MODEL)) assert not window.verified and window.tokens is None assert preflight["attempted"] is True and preflight["loaded"] is True assert preflight["verified_after"] is False assert fake.paths().count("/api/generate") == 1 assert contextwindow.effective_budget(16384, window) == 16384 @pytest.mark.parametrize("status", [404, 500]) def test_a_failed_load_is_recorded_and_leaves_the_window_unverified(server, status): fake = server(ColdOllama(loaded={}, generate_status=status)) window, preflight = asyncio.run(contextwindow.ensure_window(ENDPOINT, MODEL)) assert not window.verified assert preflight["attempted"] is True and preflight["loaded"] is False assert f"HTTP {status}" in preflight["detail"] # Bounded: one attempt, and no second probe after a failed load. assert fake.paths() == ["/api/ps", "/api/show", "/api/generate"] def test_a_declared_window_does_not_stop_the_server_being_asked(server): """A declaration fills a hole the server leaves. Loading the model can close the hole, and a verified answer always wins over a declaration.""" server(ColdOllama(loaded={}, load_window=4096)) window, _preflight = asyncio.run( contextwindow.ensure_window(ENDPOINT, MODEL, declared=8192)) assert (window.tokens, window.source) == (4096, contextwindow.LOADED) def test_an_unreachable_server_is_not_asked_to_load_anything(): """Nothing listens here. No load is attempted against a server that did not answer the probe, so an offline turn costs no second timeout.""" window, preflight = asyncio.run( contextwindow.ensure_window("http://127.0.0.1:1/v1", MODEL)) assert not window.verified assert preflight["attempted"] is False assert "did not answer" in preflight["detail"] def test_the_load_request_obeys_the_endpoint_policy(): """ADR 011. No transport is installed: a load that ignored the policy would try to reach a public address for real.""" for url in ("https://api.openai.com/v1", "http://8.8.8.8:11434/v1"): loaded, detail = asyncio.run(contextwindow.warm(url, MODEL, timeout=2)) assert loaded is False assert "not allowed" in detail def test_the_load_request_goes_only_to_the_configured_host(server): fake = server(ColdOllama(loaded={})) asyncio.run(contextwindow.ensure_window("http://192.168.0.50:11434/v1", MODEL)) hosts = {httpx.URL(url).host for _m, url, _b in fake.requests} ports = {httpx.URL(url).port for _m, url, _b in fake.requests} assert hosts == {"192.168.0.50"} and ports == {11434} # ------------------------------------------------------------- end to end @pytest.fixture() def client(monkeypatch): Base.metadata.create_all(bind=engine) setup = SessionLocal() user = models.User(is_guest=False, email="v11cold@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, model=MODEL, endpoint_url=ENDPOINT, embedding_model="", context_token_budget=16384, max_output_tokens=500, )) adventure = models.Adventure( user_id=user.id, title="Cold", campaign_canon={"rules": ["The sealed crypt is named CANON-SENTINEL-COLD-2050."]}, ) setup.add(adventure) setup.flush() setup.add(models.Action(adventure_id=adventure.id, type="start", text="Rain.")) setup.commit() adv_id, user_id = adventure.id, user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", ScriptedProvider) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.adv_id = adv_id try: yield test_client finally: app.dependency_overrides.clear() adventures.turns._active_turns.clear() Base.metadata.drop_all(bind=engine) def _long_story(adv_id, turns=120): from app import tree with SessionLocal() as db: adventure = db.get(models.Adventure, adv_id) for i in range(turns): for kind, body in ( ("do", f"I search the {i}th chamber of the undercroft."), ("ai", "The lantern gutters. " + ("Cold stone, and older dust. " * 40)), ): action = models.Action(adventure_id=adv_id, type=kind, text=body) db.add(action) db.flush() tree.place_action(db, adventure, action) db.commit() def _counts(): with SessionLocal() as db: return {table: db.execute(text(f"SELECT COUNT(*) FROM {table}")).scalar() for table in ("actions", "state_events", "state_proposals", "memories", "summaries")} def _latest_ai(adv_id): with SessionLocal() as db: return (db.query(models.Action) .filter(models.Action.adventure_id == adv_id, models.Action.type == "ai") .options(undefer(models.Action.context_snapshot)) .order_by(models.Action.id.desc()).first()) def test_a_cold_turn_is_built_to_the_window_the_loaded_model_reports(client, server): """The observed failure, prevented. Without the load this turn would be built to the configured 16,384 against a 4,096 server.""" _long_story(client.adv_id) fake = server(ColdOllama(loaded={}, load_window=4096)) ScriptedProvider.replies = ["The seal holds."] response = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "look at the seal"}) assert response.status_code == 200, response.text[:300] snapshot = _latest_ai(client.adv_id).context_snapshot assert snapshot["window"]["verified"] is True assert snapshot["tokens"]["budget"] == 4096 assert snapshot["window"]["preflight"]["attempted"] is True assert snapshot["window"]["preflight"]["verified_after"] is True system, story = ScriptedProvider.prompts[-1] sent = builder.count_tokens(system) + builder.count_tokens(story) assert sent + snapshot["tokens"]["transport"] + 500 + 256 <= 4096 assert "CANON-SENTINEL-COLD-2050" in system assert fake.paths().count("/api/generate") == 1 def test_without_the_load_the_same_cold_turn_would_have_been_built_too_large(client, server, monkeypatch): """The negative control: v1.0.0 and the first A1 tree probed only.""" _long_story(client.adv_id) server(ColdOllama(loaded={}, load_window=4096)) async def probe_only(endpoint_url, model, *, declared=None, warm_timeout=300.0): window = await contextwindow.probe(endpoint_url, model, declared=declared) return window, {"attempted": False} monkeypatch.setattr(adventures.turns.contextwindow, "ensure_window", probe_only) ScriptedProvider.replies = ["The seal holds."] client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "look at the seal"}) snapshot = _latest_ai(client.adv_id).context_snapshot assert snapshot["window"]["verified"] is False assert snapshot["tokens"]["budget"] == 16384 system, story = ScriptedProvider.prompts[-1] assert builder.count_tokens(system) + builder.count_tokens(story) > 4096 * 2 def test_the_load_itself_writes_nothing(client, server): """No action, narration, state event, proposal, memory or summary comes from the preflight: it is a request to the server and nothing else.""" fake = server(ColdOllama(loaded={})) before = _counts() asyncio.run(contextwindow.ensure_window(ENDPOINT, MODEL)) assert _counts() == before assert fake.paths().count("/api/generate") == 1 def test_a_failed_load_then_a_failed_model_call_leaves_the_story_safe(client, server): """The ordinary failure semantics: the error is reported, no narration is accepted, and nothing about the state changes.""" server(ColdOllama(loaded={}, generate_status=404)) before = _counts() ScriptedProvider.replies = [ProviderError("Endpoint or model not found (HTTP 404).")] response = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "open the door"}) assert response.status_code == 200 assert '"type": "error"' in response.text or '"error"' in response.text after = _counts() assert after["state_events"] == before["state_events"] assert after["state_proposals"] == before["state_proposals"] with SessionLocal() as db: assert db.query(models.Action).filter_by(adventure_id=client.adv_id, type="ai").count() == 0 def test_a_failed_load_does_not_stop_a_turn_the_model_can_still_answer(client, server): server(ColdOllama(loaded={}, generate_status=500)) ScriptedProvider.replies = ["The door opens."] response = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "open the door"}) assert response.status_code == 200, response.text[:300] snapshot = _latest_ai(client.adv_id).context_snapshot assert snapshot["window"]["verified"] is False assert snapshot["window"]["preflight"]["loaded"] is False assert snapshot["tokens"]["budget"] == 16384 assert snapshot["accounting"]["status"] == contextwindow.UNKNOWN def test_the_context_dry_run_never_loads_a_model(client, server): fake = server(ColdOllama(loaded={})) response = client.get(f"/api/adventures/{client.adv_id}/context") assert response.status_code == 200 assert "/api/generate" not in fake.paths()