"""M11: the application must not silently budget more input than the server accepts. This is the milestone's release blocker, and the failure it prevents is the quiet kind. M8 measured a reference deployment enforcing a **4,096**-token window while the application budgeted **16,384**. Every request returned HTTP 200. What the server did with the excess is the problem: `llama.cpp` drops the *oldest* tokens, and the oldest tokens here are the system block — the narrator's rules and the campaign canon. A 100-turn certification run against that server would have looked perfect and proved nothing. So the tests below are in two halves. **The probe** must find the real window, must refuse to guess when it cannot, and must be held to the same endpoint policy as inference — a window probe that could reach an address a turn may not would be a hole in ADR 011. **The enforcement** is the half that matters: a verified window is a *ceiling*, and the prompt that comes out of the builder must physically fit inside it. The sentinel test is the one to read — a campaign whose canon sits at the front of the prompt, a history far too long to fit, and a small verified window. The canon must still be there afterwards. That is the difference between the application choosing what to drop and the server choosing. python -m pytest tests/test_m11_context_window.py -v """ import asyncio import httpx import pytest from fastapi import Depends from fastapi.testclient import TestClient from app import auth, contextwindow, limits, models from app.context import builder from app.database import Base, SessionLocal, engine, get_db from app.main import app from app.routers import adventures from fakes import ScriptedProvider ENDPOINT = "http://127.0.0.1:11434/v1" @pytest.fixture(autouse=True) def _clear_window_cache(): contextwindow.cache_clear() yield contextwindow.cache_clear() # ------------------------------------------------------------ the arithmetic def test_the_native_api_sits_beside_the_openai_one(): assert contextwindow.native_base("http://127.0.0.1:11434/v1") == "http://127.0.0.1:11434" assert contextwindow.native_base("https://box.local:59394/v1/") == "https://box.local:59394" # Not shaped like Ollama's endpoint: used as given rather than guessed at. assert contextwindow.native_base("http://127.0.0.1:8000") == "http://127.0.0.1:8000" def test_a_verified_window_is_a_ceiling(): small = contextwindow.Window(4096, contextwindow.LOADED) assert contextwindow.effective_budget(16384, small) == 4096 def test_a_smaller_configured_budget_still_wins(): """The reader asked for a shorter prompt. The ceiling does not lengthen it.""" big = contextwindow.Window(32768, contextwindow.LOADED) assert contextwindow.effective_budget(8000, big) == 8000 def test_an_unverified_window_changes_nothing(): assert contextwindow.effective_budget(16384, contextwindow.UNVERIFIED) == 16384 assert contextwindow.effective_budget(16384, None) == 16384 # ----------------------------------------------------------------- the probe class FakeOllama: """Answers `/api/ps` and `/api/show` the way the real server does. Built from the shapes a real Ollama 0.33 returned, recorded in the M11 report: `/api/ps` carries `context_length` for a resident model, and `/api/show` carries a plain-text parameter block plus `model_info`. """ def __init__(self, *, loaded=None, parameters=None, arch_ctx=32768, show_status=200, ps_status=200): self.loaded = loaded or {} self.parameters = parameters self.arch_ctx = arch_ctx self.show_status = show_status self.ps_status = ps_status self.seen: list[str] = [] def handler(self, request: httpx.Request) -> httpx.Response: self.seen.append(str(request.url)) if request.url.path == "/api/ps": if self.ps_status != 200: return httpx.Response(self.ps_status) return httpx.Response(200, json={"models": [ {"name": name, "model": name, "context_length": tokens} for name, tokens in self.loaded.items() ]}) if request.url.path == "/api/show": if self.show_status != 200: return httpx.Response(self.show_status, json={}) body = {"model_info": {"qwen2.context_length": self.arch_ctx}} if self.parameters is not None: body["parameters"] = self.parameters return httpx.Response(200, json=body) return httpx.Response(404) @pytest.fixture() def server(monkeypatch): """Installs a fake Ollama behind httpx, and hands the test the recorder.""" holder = {} def install(fake: FakeOllama): holder["fake"] = fake original = httpx.AsyncClient def build(*args, **kwargs): kwargs.pop("verify", None) return original(*args, transport=httpx.MockTransport(fake.handler), **kwargs) monkeypatch.setattr(contextwindow.httpx, "AsyncClient", build) return fake return install def test_a_loaded_model_reports_the_window_it_is_being_served_with(server): fake = server(FakeOllama(loaded={"qwen2.5:3b-instruct": 4096})) window = asyncio.run(contextwindow.probe(ENDPOINT, "qwen2.5:3b-instruct")) assert window.tokens == 4096 assert window.source == contextwindow.LOADED assert window.verified # Asked the running server first, because a resident model has already # settled the question. assert fake.seen[0].endswith("/api/ps") def test_an_unloaded_model_falls_back_to_what_it_will_load_with(server): server(FakeOllama(loaded={}, parameters="num_ctx 16384\n")) window = asyncio.run(contextwindow.probe(ENDPOINT, "qwen2.5:3b-instruct-16k")) assert (window.tokens, window.source) == (16384, contextwindow.PARAMETERS) assert window.model_max == 32768 def test_a_model_with_no_num_ctx_is_unknown_rather_than_assumed(server): """The case that caused the bug, and it must not be papered over. The server will load this at *its* default — 4,096 with no VRAM — but the default is the server's business and is not in any answer it gave us. Reporting 4,096 here would be a guess that happens to be right on one machine, so this reports unknown and says why. """ server(FakeOllama(loaded={}, parameters=None)) window = asyncio.run(contextwindow.probe(ENDPOINT, "qwen2.5:3b-instruct")) assert not window.verified assert "num_ctx" in window.detail assert window.model_max == 32768 # still useful: raising it is possible def test_the_declared_window_cannot_exceed_the_architecture(server): server(FakeOllama(loaded={}, parameters="num_ctx 999999\n", arch_ctx=32768)) assert asyncio.run(contextwindow.probe(ENDPOINT, "m")).tokens == 32768 def test_a_probe_obeys_the_same_endpoint_policy_as_inference(): """ADR 011 / H12. A probe is a request, and requests go where turns may go. No transport is installed, so a probe that ignored the policy would attempt a real connection to a cloud host. It is refused before that. """ for url in ("https://api.openai.com/v1", "http://8.8.8.8:11434/v1", "https://replicate.com/v1"): window = asyncio.run(contextwindow.probe(url, "gpt-4")) assert not window.verified assert "not allowed" in window.detail def test_an_unreachable_server_is_unknown_not_an_exception(): """Offline is the ordinary case, and it must not cost a turn.""" window = asyncio.run(contextwindow.probe("http://127.0.0.1:1/v1", "any")) assert not window.verified assert window.tokens is None def test_a_server_that_does_not_speak_ollama_is_unknown(server): server(FakeOllama(loaded={}, show_status=404, ps_status=404)) assert not asyncio.run(contextwindow.probe(ENDPOINT, "m")).verified def test_the_answer_is_cached_so_it_costs_one_request_a_session(server): fake = server(FakeOllama(loaded={"m": 8192})) for _ in range(5): assert asyncio.run(contextwindow.probe(ENDPOINT, "m")).tokens == 8192 assert len([u for u in fake.seen if u.endswith("/api/ps")]) == 1 def test_changing_the_model_or_endpoint_forgets_what_was_learned(server): fake = server(FakeOllama(loaded={"m": 8192, "other": 2048})) assert asyncio.run(contextwindow.probe(ENDPOINT, "m")).tokens == 8192 assert asyncio.run(contextwindow.probe(ENDPOINT, "other")).tokens == 2048 contextwindow.cache_clear() assert asyncio.run(contextwindow.probe(ENDPOINT, "m")).tokens == 8192 assert len([u for u in fake.seen if u.endswith("/api/ps")]) == 3 # ----------------------------------------------------------- the enforcement @pytest.fixture() def client(monkeypatch): Base.metadata.create_all(bind=engine) setup = SessionLocal() user = models.User(is_guest=False, email="m11cw@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, model="qwen2.5:3b-instruct", endpoint_url=ENDPOINT, embedding_model="", context_token_budget=16384, max_output_tokens=800, )) adventure = models.Adventure( user_id=user.id, title="Windowed", # The real canon shape — a dict of rules — not a string. The first # version of this fixture passed a string, `_canon_section` correctly # ignored it, and the sentinel test failed against a product that was # behaving properly. Recorded in the M11 report as a harness defect. campaign_canon={"rules": [ "The abbey seal has never been broken.", "The sealed crypt is named CANON-SENTINEL-VERITAS-4417.", ]}, ) setup.add(adventure) setup.flush() setup.add(models.Action( adventure_id=adventure.id, type="start", text="Rain over Westhaven.")) setup.commit() adv_id, user_id = adventure.id, user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", ScriptedProvider) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.adv_id = adv_id try: yield test_client finally: app.dependency_overrides.clear() adventures.turns._active_turns.clear() Base.metadata.drop_all(bind=engine) def _long_story(adv_id, turns=120): """A history far larger than any small window, written straight to the tree. Written through the ORM rather than played, because what is under test is the builder's arithmetic against a big story, not the turn engine. """ from app import tree with SessionLocal() as db: adventure = db.get(models.Adventure, adv_id) for i in range(turns): for kind, text in ( ("do", f"I search the {i}th chamber of the undercroft."), ("ai", "The lantern gutters. " + ("Cold stone, and older dust. " * 40)), ): action = models.Action(adventure_id=adv_id, type=kind, text=text) db.add(action) db.flush() tree.place_action(db, adventure, action) db.commit() def _report(client, window): """Builds the prompt the way a turn would, with `window` as the server's.""" with SessionLocal() as db: adventure = db.get(models.Adventure, client.adv_id) settings = db.query(models.Settings).first() return builder.build_context(adventure, settings, window=window) def test_a_small_verified_window_caps_the_budget(client): _long_story(client.adv_id, turns=60) _, _, report = _report(client, contextwindow.Window(4096, contextwindow.LOADED)) assert report["tokens"]["budget"] == 4096 assert report["tokens"]["configured_budget"] == 16384 assert report["window"]["capped"] is True assert report["window"]["verified"] is True def test_the_prompt_physically_fits_inside_the_verified_window(client): """The invariant, measured on the assembled text rather than on intent.""" _long_story(client.adv_id, turns=60) system, story, report = _report( client, contextwindow.Window(4096, contextwindow.LOADED)) total = builder.count_tokens(system) + builder.count_tokens(story) reserve = report["tokens"]["output_reserve"] assert total + reserve <= 4096, (total, reserve) assert report["tokens"]["total"] == total def test_the_canon_at_the_front_survives_a_window_far_too_small(client): """The sentinel test: the application drops history, the server never gets to. `llama.cpp` truncates from the *front*, so if the app over-budgets, the canon is what disappears. Here the story is 120 turns long and the window is 4,096 tokens — an enormous overflow — and the canon sentinel must still be in the prompt, with the history cut instead. """ _long_story(client.adv_id, turns=120) system, story, report = _report( client, contextwindow.Window(4096, contextwindow.LOADED)) assert "CANON-SENTINEL-VERITAS-4417" in system assert builder.count_tokens(system) + builder.count_tokens(story) <= 4096 # And it is the history that gave way — the oldest of it, keeping the # newest, which is the choice the application is supposed to be making. assert report["history"]["included"] < report["history"]["total"] / 10 assert "119th chamber" in story # the most recent turn survived assert "0th chamber" not in story # the oldest did not def test_without_the_cap_the_same_prompt_would_have_overflowed(client): """Proof the test above is testing something: the defect, reproduced. The same campaign, the same builder, no verified window — which is exactly what every build before M11 did — produces a prompt several times larger than the server would read. That is the prompt whose front the server would have silently eaten. """ _long_story(client.adv_id, turns=120) system, story, _ = _report(client, None) unbounded = builder.count_tokens(system) + builder.count_tokens(story) assert unbounded > 4096 * 2, unbounded def test_an_unverified_window_is_recorded_as_unverified(client): _, _, report = _report(client, contextwindow.UNVERIFIED) assert report["window"]["verified"] is False assert report["window"]["capped"] is False assert report["tokens"]["budget"] == 16384 def test_a_window_too_small_for_the_protected_context_fails_with_advice(client): """§32's graceful failure, with the M11 sentence added. A 1,024-token server cannot hold the reply reserve plus the canon, and the honest answer is a refusal that says raising the *setting* will not help, because the setting is no longer what is binding. """ with pytest.raises(builder.ContextOverflow) as caught: _report(client, contextwindow.Window(1024, contextwindow.LOADED)) message = str(caught.value) assert "1024" in message assert "load the model with a larger window" in message def test_a_turn_records_the_window_it_was_built_against(client, monkeypatch): """End to end: the stored snapshot of a real turn carries the verdict. This is what makes an old turn auditable — a reviewer can ask of any turn in the campaign whether it was built against a checked window, rather than inferring it from what the settings say today. """ async def verified(endpoint, model, declared=None, use_cache=True): return contextwindow.Window(4096, contextwindow.LOADED, 32768, "fake") monkeypatch.setattr(adventures.turns.contextwindow, "probe", verified) ScriptedProvider.replies = ["The crypt is still sealed."] response = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "look at the seal"}) assert response.status_code == 200, response.text[:300] with SessionLocal() as db: from sqlalchemy.orm import undefer action = ( db.query(models.Action) .filter(models.Action.adventure_id == client.adv_id, models.Action.type == "ai") .options(undefer(models.Action.context_snapshot)) .order_by(models.Action.id.desc()).first() ) snapshot = action.context_snapshot assert snapshot["window"]["verified"] is True assert snapshot["window"]["tokens"] == 4096 assert snapshot["tokens"]["budget"] == 4096