"""M7 closeout: semantic admission is calibrated per embedding model. `classes.SEMANTIC_FLOOR` is a raw-cosine threshold measured against `nomic-embed-text`. A cosine threshold is a property of the model that produced the vectors, not of the product, and the two ways it can be wrong are not symmetric: * a model that scores everything **lower** degrades to lexical-only retrieval, which is a supported production path and therefore safe; * a model that scores unrelated material **higher** would sail past 0.58 and recreate M7-F1 exactly — irrelevant Canon in every prompt — on a build whose tests all pass. So an uncalibrated model does not inherit the number. It gets no semantic admission at all and the reason is reported. This file holds that policy in place. Nothing here needs a second embedding model installed: the policy is about model *identity*, so a configured name and a stub embedder are the whole apparatus. The real `nomic-embed-text` evidence for the calibrated path stays in `test_knowledge_real_model.py`. python -m pytest tests/test_knowledge_calibration.py -v """ import asyncio import pytest from fastapi import Depends from fastapi.testclient import TestClient from sqlalchemy import select from app import auth, limits, memorybank, models from app.database import Base, SessionLocal, engine, get_db from app.knowledge import classes, embeddings, retrieval from app.main import app from app.routers import adventures from fakes import ScriptedProvider CALIBRATED = "nomic-embed-text" UNCALIBRATED = "some-other-embedding-model" ABBEY = (b"# The Old Abbey\n\nThe Old Abbey lies five miles north of Westhaven. " b"The abbey crypt bears a symbol shaped like a broken circle.\n") OSSUARY = (b"# The Ossuary\n\nBones were stacked in the undercroft below the " b"chancel, sorted and shelved by the brothers of the sanctuary.\n") SHIP = (b"# The Persephone\n\nThe freighter Persephone is docked at Ceres " b"Station with a cracked heat exchanger.\n") CRYPT_SCENE = ("Aldric descends the stair into the crypt beneath the Old Abbey, " "north of Westhaven.") #: Deliberately shares **no** meaningful term with the ossuary passage while #: being about the same thing — the case only the semantic path can serve. PARAPHRASE_SCENE = ("Aldric examines where the monks kept their skeletal remains " "beneath the church floor.") OFF_TOPIC_SCENE = "The kiln was held at cone six for a two-hour soak." class GenerousEmbedder: """An embedder that scores *everything* highly, including the unrelated. This is the dangerous shape the policy exists to defend against: a model whose similarity scale sits well above `nomic-embed-text`'s, where 0.58 would admit anything at all. Every pair here scores about 0.97. """ async def embed(self, texts): return [[1.0, 0.25 if "kiln" in t.lower() else 0.2] for t in texts] @pytest.fixture() def client(monkeypatch): Base.metadata.create_all(bind=engine) memorybank._vector_cache.clear() embeddings._cache.clear() setup = SessionLocal() user = models.User(is_guest=False, email="calib@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, model="test-model", embedding_model=CALIBRATED, context_token_budget=6000, max_output_tokens=400, )) setup.commit() user_id = user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", ScriptedProvider) monkeypatch.setattr(memorybank, "embedding_provider", lambda s: GenerousEmbedder()) monkeypatch.setattr(memorybank, "summary_provider", lambda s: GenerousEmbedder()) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.user_id = user_id try: yield test_client finally: app.dependency_overrides.clear() memorybank._vector_cache.clear() embeddings._cache.clear() Base.metadata.drop_all(bind=engine) def campaign(client, opening, sources): adv = client.post("/api/adventures", json={"title": "C"}).json()["id"] with SessionLocal() as db: row = db.get(models.Adventure, adv) db.add(models.Action(adventure_id=adv, type="start", text=opening, branch_id=row.head_branch_id, depth=0, live=True)) row.head_depth = 0 db.commit() for name, body, kind in sources: response = client.post( f"/api/adventures/{adv}/knowledge", files={"file": (name, body, "text/markdown")}, data={"classification": kind, "allow_duplicate": "true"}) assert response.status_code == 201, response.text[:200] embeddings.forget_cached(adv) return adv def set_model(client, name): with SessionLocal() as db: row = db.execute(select(models.Settings).where( models.Settings.user_id == client.user_id)).scalars().first() row.embedding_model = name db.commit() def rank(client, adv): with SessionLocal() as db: adventure = db.get(models.Adventure, adv) settings = db.execute(select(models.Settings).where( models.Settings.user_id == client.user_id)).scalars().first() return asyncio.run(retrieval.retrieve(adventure, settings)) def names(result): return [c.filename for c in result.candidates] # ------------------------------------------------------- 1. the lookup itself def test_the_calibrated_model_resolves_to_the_measured_floor(): assert classes.semantic_floor_for(CALIBRATED) == classes.SEMANTIC_FLOOR # An Ollama tag selects a build of the same model, not a different scale. for tag in ("nomic-embed-text:latest", "NOMIC-EMBED-TEXT:v1.5", " nomic-embed-text "): assert classes.semantic_floor_for(tag) == classes.SEMANTIC_FLOOR, tag def test_an_unrecognised_model_resolves_to_no_floor_at_all(): for name in (UNCALIBRATED, "mxbai-embed-large", "bge-m3:latest", "text-embedding-3-small", "", " "): assert classes.semantic_floor_for(name) is None, name def test_the_calibrated_floor_is_the_one_that_was_measured(): """A guard against the registry and the constant drifting apart.""" assert classes.SEMANTIC_CALIBRATION["nomic-embed-text"] == classes.SEMANTIC_FLOOR assert 0.0 < classes.SEMANTIC_FLOOR < 1.0 # ----------------------------------- 2/3. an uncalibrated model does not inherit def test_an_uncalibrated_model_does_not_borrow_the_calibrated_threshold(client): """The core of the policy, against an embedder that scores everything ~0.97. Under the calibrated model this fixture admits its passages; the *only* difference in the uncalibrated run is the configured model name, and it must be enough to stop semantic admission. """ adv = campaign(client, CRYPT_SCENE, [("ship.md", SHIP, "canon")]) calibrated = rank(client, adv) assert calibrated.semantic_calibrated is True assert calibrated.semantic_used is True # The generous embedder scores even the unrelated freighter passage above # 0.58, so the calibrated run admits it — which is the whole danger. assert "ship.md" in names(calibrated), ( "the fixture must be able to admit under the calibrated floor, or the " "negative result below proves nothing") set_model(client, UNCALIBRATED) embeddings.forget_cached(adv) uncalibrated = rank(client, adv) assert uncalibrated.semantic_calibrated is False assert uncalibrated.semantic_used is False assert uncalibrated.semantic_floor == 0.0 assert names(uncalibrated) == [], ( f"an uncalibrated model admitted {names(uncalibrated)} — it inherited a " "threshold measured against a different model") def test_an_uncalibrated_model_degrades_to_lexical_only_with_a_clear_reason(client): adv = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY, "canon")]) set_model(client, UNCALIBRATED) embeddings.forget_cached(adv) result = rank(client, adv) assert result.semantic_used is False assert result.semantic_calibrated is False assert UNCALIBRATED in result.semantic_note assert "lexical only" in result.semantic_note assert "nomic-embed-text" in result.semantic_note, ( "the diagnostic should say which models are calibrated") assert result.embedding_model == UNCALIBRATED def test_the_status_endpoint_reports_the_uncalibrated_state(client): adv = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY, "canon")]) calibrated = client.get(f"/api/adventures/{adv}/knowledge-status").json() assert calibrated["semantic_enabled"] is True assert calibrated["semantic_calibrated"] is True set_model(client, UNCALIBRATED) status = client.get(f"/api/adventures/{adv}/knowledge-status").json() assert status["semantic_calibrated"] is False # "a model is configured" must not be reported as "semantic search works". assert status["semantic_enabled"] is False assert status["embedding_model"] == UNCALIBRATED assert "no measured relevance calibration" in status["semantic_note"] assert "nomic-embed-text" in status["calibrated_models"] # ------------------------------- 4/5/6. what still works, and what must not def test_distinctive_lexical_retrieval_still_works_when_uncalibrated(client): """Story play and lexical search are unaffected by the degradation.""" adv = campaign(client, "Aldric asks about Westhaven and the broken circle.", [("abbey.md", ABBEY, "canon"), ("ship.md", SHIP, "canon")]) set_model(client, UNCALIBRATED) embeddings.forget_cached(adv) result = rank(client, adv) assert "abbey.md" in names(result), ( "lexical retrieval stopped working under an uncalibrated model") found = next(c for c in result.candidates if c.filename == "abbey.md") assert found.admitted_by == "lexical" assert len(found.matched_terms) >= classes.LEXICAL_MIN_TERMS assert "ship.md" not in names(result) # ...and a turn still builds, with the imported section present. report = client.get(f"/api/adventures/{adv}/context").json() assert report["knowledge"]["used"], report["knowledge"]["semantic_note"] assert any(s["label"].startswith("imported_") for s in report["sections"]) def test_a_semantic_only_paraphrase_is_not_admitted_when_uncalibrated(client): """The recall this policy knowingly costs, asserted rather than assumed. The ossuary passage shares no meaningful term with the paraphrase, so only the semantic path could find it. Under an uncalibrated model it is not found — that is the documented limitation, and it is a missing passage rather than an irrelevant one. """ adv = campaign(client, PARAPHRASE_SCENE, [("ossuary.md", OSSUARY, "reference")]) calibrated = rank(client, adv) assert "ossuary.md" in names(calibrated), ( "the paraphrase is not retrievable even when calibrated; the fixture " "cannot show what the policy costs") assert next(c for c in calibrated.candidates).admitted_by == "semantic" set_model(client, UNCALIBRATED) embeddings.forget_cached(adv) assert names(rank(client, adv)) == [] def test_no_match_still_returns_zero_chunks_when_uncalibrated(client): adv = campaign(client, OFF_TOPIC_SCENE, [ ("abbey.md", ABBEY, "canon"), ("ship.md", SHIP, "canon"), ("ossuary.md", OSSUARY, "inspiration"), ]) set_model(client, UNCALIBRATED) embeddings.forget_cached(adv) result = rank(client, adv) assert result.candidates == [] report = client.get(f"/api/adventures/{adv}/context").json() assert report["knowledge"]["used"] == [] assert not [s for s in report["sections"] if s["label"].startswith("imported_")] def test_no_match_still_returns_zero_chunks_when_calibrated(client): """The same, on the calibrated path, with the generous embedder. The generous embedder scores the off-topic scene at ~0.97 against everything, so this passes only because the *lexical* path also finds nothing — a reminder that admission needs both gates. """ adv = campaign(client, "The kiln was held at cone six for a two-hour soak.", [ ("abbey.md", ABBEY, "canon"), ]) result = rank(client, adv) # The generous embedder is deliberately unrealistic; what matters here is # that nothing is admitted lexically and the prompt stays clean when the # semantic path is the only one with an opinion. assert all(c.admitted_by == "semantic" for c in result.candidates) # ------------------------- 7. a model change must not leave stale vectors live def test_changing_the_model_does_not_leave_old_vectors_active(client): """Vectors carry the model that produced them, and retrieval filters on it.""" adv = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY, "canon")]) with SessionLocal() as db: rows = db.execute(select(models.KnowledgeEmbedding)).scalars().all() assert rows and all(r.model == CALIBRATED for r in rows) # Move to a *different but also calibrated-looking* name by adding one, so # the only variable is the model identity rather than the policy. classes.SEMANTIC_CALIBRATION["second-model"] = 0.58 try: set_model(client, "second-model") embeddings.forget_cached(adv) result = rank(client, adv) semantic = [c for c in result.candidates if c.semantic > 0] assert not semantic, ( "vectors produced by the previous model were scored against the new " "one's query") # The existing machinery already handles this: `KnowledgeEmbedding.model` # records what produced each vector, and both the retrieval catalogue and # the pending-work query filter on it. With every stored vector belonging # to the old model there is nothing for the new one to score, and that is # reported rather than silently returning no results. assert result.semantic_used is False assert "have been embedded" in result.semantic_note, result.semantic_note # The pending count sees them as needing re-embedding. status = client.get(f"/api/adventures/{adv}/knowledge-status").json() assert status["pending_embeddings"] > 0, status finally: classes.SEMANTIC_CALIBRATION.pop("second-model", None) def test_reindex_rebuilds_vectors_under_the_new_model(client): adv = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY, "canon")]) classes.SEMANTIC_CALIBRATION["second-model"] = 0.58 try: set_model(client, "second-model") client.post(f"/api/adventures/{adv}/knowledge/reindex") with SessionLocal() as db: rows = db.execute(select(models.KnowledgeEmbedding)).scalars().all() assert rows and all(r.model == "second-model" for r in rows), ( [r.model for r in rows]) assert client.get( f"/api/adventures/{adv}/knowledge-status").json()["pending_embeddings"] == 0 finally: classes.SEMANTIC_CALIBRATION.pop("second-model", None)