"""M11 §E: the context-window fix, against a real Ollama rather than a fake one. `test_m11_context_window.py` proves the arithmetic and the enforcement with a mocked server, which is the right place for those. This file answers the question that a mock cannot: **does the probe read a real Ollama correctly?** The shapes it parses — `/api/ps`'s `context_length`, `/api/show`'s plain-text parameter block — are Ollama's, not ours, and a mock built from a misreading of them would agree with itself forever. It also demonstrates the sequence a reader actually experiences on a server whose model has no `num_ctx` baked in: turn 1 the model is not resident; the window cannot be verified; the turn proceeds and is recorded as unverified turn 2 the model is resident, `/api/ps` reports the real window, and the budget is capped to it from here on Skipped unless an endpoint is configured, so the ordinary suite stays local, deterministic and offline. The endpoint is read from the environment and never written down here. AIDND_TEST_ENDPOINT=https://:/v1 \\ AIDND_TEST_MODEL=qwen2.5:3b-instruct \\ python -m pytest tests/test_m11_real_window.py -v -s """ import asyncio import os import pytest from fastapi import Depends from fastapi.testclient import TestClient from app import auth, contextwindow, limits, models from app.database import Base, SessionLocal, engine, get_db from app.main import app ENDPOINT = os.environ.get("AIDND_TEST_ENDPOINT", "") MODEL = os.environ.get("AIDND_TEST_MODEL", "") #: A second model, with a larger window baked in, when the server has one. The #: contrast between the two is the whole point of the M8 finding. WIDE_MODEL = os.environ.get("AIDND_TEST_WIDE_MODEL", "") pytestmark = pytest.mark.skipif( not (ENDPOINT and MODEL), reason="set AIDND_TEST_ENDPOINT and AIDND_TEST_MODEL to run against a real server", ) @pytest.fixture(autouse=True) def _clear(): contextwindow.cache_clear() yield contextwindow.cache_clear() @pytest.fixture() def client(monkeypatch): Base.metadata.create_all(bind=engine) setup = SessionLocal() user = models.User(is_guest=False, email="m11real@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, model=MODEL, endpoint_url=ENDPOINT, embedding_model="", context_token_budget=16384, max_output_tokens=400, model_timeout_seconds=300, )) adventure = models.Adventure( user_id=user.id, title="Real window", campaign_canon={"rules": ["The abbey seal has never been broken."]}, ) setup.add(adventure) setup.flush() setup.add(models.Action( adventure_id=adventure.id, type="start", text="Rain over Westhaven, and the abbey bell tolling.")) setup.commit() adv_id, user_id = adventure.id, user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.adv_id = adv_id try: yield test_client finally: app.dependency_overrides.clear() Base.metadata.drop_all(bind=engine) def _snapshot(adv_id) -> dict: from sqlalchemy.orm import undefer with SessionLocal() as db: action = ( db.query(models.Action) .filter(models.Action.adventure_id == adv_id, models.Action.type == "ai") .options(undefer(models.Action.context_snapshot)) .order_by(models.Action.id.desc()).first() ) return action.context_snapshot if action else {} def _play(client, text) -> None: response = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": text}) assert response.status_code == 200, response.text[:400] def test_the_probe_reads_this_server(capsys): """Records what this deployment actually reports. Evidence, not a threshold.""" window = asyncio.run(contextwindow.probe(ENDPOINT, MODEL, use_cache=False)) with capsys.disabled(): print(f"\n model {MODEL}") print(f" tokens {window.tokens}") print(f" source {window.source}") print(f" model max {window.model_max}") print(f" detail {window.detail}") # Either answer is legitimate — what is not legitimate is a crash, a guess, # or a claim that cannot be traced to something the server said. assert window.source in (contextwindow.LOADED, contextwindow.PARAMETERS, contextwindow.UNKNOWN) if window.verified: assert window.tokens >= 512 if window.model_max: assert window.tokens <= window.model_max def test_a_real_turn_is_capped_to_what_this_server_gives(client, capsys): """The sequence a reader sees, and the cap arriving with residency.""" _play(client, "I climb the abbey steps and look back at the town.") first = _snapshot(client.adv_id)["window"] # The model is resident now, so the second turn's probe can read /api/ps. contextwindow.cache_clear() _play(client, "I try the crypt door.") second = _snapshot(client.adv_id) window, tokens = second["window"], second["tokens"] with capsys.disabled(): print(f"\n turn 1 window verified={first['verified']} " f"tokens={first['tokens']} source={first['source']}") print(f" turn 2 window verified={window['verified']} " f"tokens={window['tokens']} source={window['source']}") print(f" budget configured={tokens['configured_budget']} " f"effective={tokens['budget']}") print(f" prompt {tokens['total']} tokens " f"+ {tokens['output_reserve']} reserved") assert window["verified"], ( "the model has been served a turn, so /api/ps should now report its " f"window: {window['detail']}" ) # The invariant, on a real server: what was assembled fits what it accepts. assert tokens["budget"] == min(tokens["configured_budget"], window["tokens"]) assert tokens["total"] + tokens["output_reserve"] <= window["tokens"] @pytest.mark.skipif(not WIDE_MODEL, reason="set AIDND_TEST_WIDE_MODEL") def test_a_model_with_a_baked_window_reports_the_larger_one(capsys): """The operator's fix, seen from the application. A model created with `num_ctx` baked in reports the larger window through the same path, so the difference between a deployment that has applied DEVELOPMENT.md's fix and one that has not is visible to the application rather than only to whoever reads the server logs. """ narrow = asyncio.run(contextwindow.probe(ENDPOINT, MODEL, use_cache=False)) wide = asyncio.run(contextwindow.probe(ENDPOINT, WIDE_MODEL, use_cache=False)) with capsys.disabled(): print(f"\n {MODEL:28} {narrow.tokens} ({narrow.source})") print(f" {WIDE_MODEL:28} {wide.tokens} ({wide.source})") assert wide.verified and wide.tokens >= 8192 if narrow.verified: assert wide.tokens > narrow.tokens