"""M7: what retrieval admits and how it ranks — the mechanism, not the fixture. This is a **purpose-built retrieval-mechanism** suite. It uses invented sources chosen to isolate one behaviour each, not the standard campaign fixture; the acceptance-fixture tests live in `test_imported_knowledge.py`. The two are kept apart deliberately: an acceptance test says the product meets its contract, and this says the machinery underneath behaves the way the contract needs it to. ## The two stages, and why they are tested separately candidate generation -> ADMISSION -> ranking -> class weighting -> budget **Admission** decides whether a passage matched at all, from signals that mean something on their own. **Ranking** orders what survived. M7's first implementation had only the second: it normalized every score against the best of its own path and cut at a share of that best, which the best clears by construction. Something was therefore admitted on every turn, whatever the reader was doing (review finding M7-F1). ## Why the stub embedder looks the way it does The suite that shipped with M7 asserted "irrelevant Canon does not win" and passed, while the product injected five irrelevant sources into every prompt. Its stub gave unrelated text a cosine of 0.06-0.20 and its own docstring said it had *deliberately* removed the constant component that "would put a similarity floor under every pair" — which is exactly the property real embedding models have. Measured on identical texts, `nomic-embed-text` scored those same unrelated pairs 0.435-0.437. The stub was an order of magnitude more discriminative than reality, so the broken gate sailed through (finding M7-F2). `RealisticEmbedder` below therefore has a deliberate similarity floor. Unrelated passages score a substantial, nontrivial similarity, as they do in life. That is not decoration: `test_the_stub_models_the_real_problem` fails if the floor ever goes away, and `test_a_relative_only_floor_would_admit_the_irrelevant_set` demonstrates on this very fixture that the *old* rule would still be fooled by it. The stub models the shape of the problem; it does not encode the answer. python -m pytest tests/test_knowledge_retrieval_quality.py -v """ import asyncio import math import pytest from fastapi import Depends from fastapi.testclient import TestClient from sqlalchemy import select from app import auth, limits, memorybank, models from app.database import Base, SessionLocal, engine, get_db from app.knowledge import classes, embeddings, retrieval from app.main import app from app.routers import adventures from fakes import ScriptedProvider # --------------------------------------------------------------- the library ABBEY_CANON = (b"# The Old Abbey\n\nThe Old Abbey lies five miles north of " b"Westhaven. The abbey crypt bears a symbol shaped like a broken " b"circle, cut into the keystone above the stair.\n") CRYPT_REFERENCE = (b"# Crypt Construction\n\nAn abbey crypt was vaulted in stone, " b"entered by a stair descending from the nave, with burial " b"niches cut into the side walls.\n") CRYPT_MOOD = (b"# Below\n\nThe air in the crypt was older than the abbey above it, " b"and the dark pressed close around the lantern on the stair.\n") OSSUARY = (b"# The Ossuary\n\nBones were stacked in the undercroft below the " b"chancel, sorted and shelved by the brothers of the sanctuary.\n") ABBEY_COPY = (b"# The Abbey\n\nFive miles north of Westhaven stands the Old Abbey. " b"Above the crypt stair a broken circle is cut into the keystone.\n") SHIP_CANON = (b"# The Persephone\n\nThe freighter Persephone is docked at Ceres " b"Station with a cracked heat exchanger and no licence to carry " b"passengers.\n") SURGERY_REFERENCE = (b"# Cannulation\n\nThe surgeon sterilised the cannula and " b"checked the infusion pump pressure before the procedure.\n") COMPILER_INSPIRATION = (b"# Diagnostics\n\nThe compiler emits a diagnostic when the " b"lifetime of the borrow outlives the referent.\n") CRYPT_SCENE = ("Aldric descends the stair into the crypt beneath the Old Abbey, " "north of Westhaven, lantern raised.") #: A scene with no connection to any source in the library at all. OFF_TOPIC_SCENE = ("The kiln was held at cone six for a two-hour soak while the " "glaze matured.") class RealisticEmbedder: """A deterministic embedder with the two properties the real one has. * **A similarity floor.** Every pair of texts shares a constant component, so unrelated passages score a substantial similarity rather than nearly zero. This is what a real embedding model does and what the M7 stub left out; without it no fixture can detect an admission gate that cannot say "no match". * **Topical structure above the floor.** Disjoint topic axes, so a passage about the same subject scores clearly higher — including when it shares almost no vocabulary, which is the case the hybrid's semantic half exists to serve. A hashed bag of words at low weight sits underneath, so two passages on one topic in different words are close without being identical and the redundancy suppressor is not handed a fixture of clones. """ #: Deliberately disjoint: no word appears on two axes, or a query about one #: subject scores as though it were about another and the fixture stops #: meaning what it says. AXES = ( ("crypt", "abbey", "vault", "undercroft", "ossuary", "chancel", "bones", "stair", "keystone", "niches", "nave", "burial", "monastery", "emblem", "circle", "broken", "symbol", "sanctuary", "brothers", "shelved"), ("westhaven", "north", "miles", "road", "town", "stands"), ("lantern", "dark", "air", "older", "pressed", "close"), ("freighter", "persephone", "ceres", "docked", "exchanger", "licence", "passengers", "station", "cracked"), ("surgeon", "cannula", "infusion", "pump", "sterilised", "pressure", "procedure"), ("compiler", "diagnostic", "borrow", "lifetime", "referent", "emits"), ("kiln", "cone", "soak", "glaze", "matured"), ) #: The constant every vector carries. Tuned so unrelated pairs land in a #: realistic band rather than near zero — see the module docstring. BASE = 0.9 TOPIC_WEIGHT = 2.0 WORD_WEIGHT = 0.25 BUCKETS = 64 @staticmethod def _words(text): return set("".join(c.lower() if c.isalnum() or c == "-" else " " for c in text).split()) def vector(self, text): unique = self._words(text) topic = [self.TOPIC_WEIGHT * len(unique & set(axis)) / len(axis) for axis in self.AXES] buckets = [0.0] * self.BUCKETS for word in unique: index = sum((i + 1) * ord(c) for i, c in enumerate(word)) % self.BUCKETS buckets[index] += self.WORD_WEIGHT scale = math.sqrt(len(unique)) or 1.0 return [self.BASE] + topic + [b / scale for b in buckets] async def embed(self, texts): return [self.vector(t) for t in texts] @pytest.fixture() def client(monkeypatch): Base.metadata.create_all(bind=engine) memorybank._vector_cache.clear() embeddings._cache.clear() setup = SessionLocal() user = models.User(is_guest=False, email="quality@example.com") setup.add(user) setup.flush() # A *calibrated* model name, deliberately. Semantic admission is # per-model (`classes.SEMANTIC_CALIBRATION`), and the stub below # is built to model this model's similarity distribution, so the # fixture must name it or the suite would silently exercise the # uncalibrated lexical-only path instead. setup.add(models.Settings( user_id=user.id, model="test-model", embedding_model="nomic-embed-text", context_token_budget=6000, max_output_tokens=400, )) setup.commit() user_id = user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", ScriptedProvider) monkeypatch.setattr(memorybank, "embedding_provider", lambda s: RealisticEmbedder()) monkeypatch.setattr(memorybank, "summary_provider", lambda s: RealisticEmbedder()) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.user_id = user_id try: yield test_client finally: app.dependency_overrides.clear() memorybank._vector_cache.clear() embeddings._cache.clear() Base.metadata.drop_all(bind=engine) # ----------------------------------------------------------------- helpers def campaign(client, opening, sources): """A campaign with `opening` as its only turn and `sources` imported.""" adventure = client.post("/api/adventures", json={"title": "Q"}).json() adv = adventure["id"] with SessionLocal() as db: row = db.get(models.Adventure, adv) db.add(models.Action(adventure_id=adv, type="start", text=opening, branch_id=row.head_branch_id, depth=0, live=True)) row.head_depth = 0 db.commit() ids = {} for name, body, kind in sources: response = client.post( f"/api/adventures/{adv}/knowledge", files={"file": (name, body, "text/markdown")}, data={"classification": kind, "allow_duplicate": "true"}) assert response.status_code == 201, response.text[:200] ids[name] = response.json()["id"] embeddings.forget_cached(adv) return adv, ids def rank(client, adv): with SessionLocal() as db: adventure = db.get(models.Adventure, adv) settings = db.execute(select(models.Settings).where( models.Settings.user_id == client.user_id)).scalars().first() return asyncio.run(retrieval.retrieve(adventure, settings)) def table(result): rows = [f" {c.filename:22} {c.classification:12} by={c.admitted_by or 'always':9} " f"lex={c.lexical:.3f} sem={c.semantic:.3f} cos={c.cosine:.3f} " f"score={c.score:.3f} terms={c.matched_terms}" for c in result.candidates] rows += [f" {c.filename:22} SUPPRESSED (duplicate of {c.duplicate_of})" for c in result.suppressed] return (f"generated={result.generated} rejected={result.rejected} " f"floor={result.semantic_floor}\n" + "\n".join(rows) or " (nothing)") def names(result): return [c.filename for c in result.candidates] # =================================================== the stub is realistic def test_the_stub_models_the_real_problem(client): """M7-F2's guard: the stub must not be more discriminative than reality. If this ever fails because unrelated pairs score near zero, the fixture has drifted back to the one that hid the defect, and every no-match test in this file has quietly stopped proving anything. """ embedder = RealisticEmbedder() query = embedder.vector(CRYPT_SCENE) unrelated = [embedder.vector(t.decode()) for t in (SURGERY_REFERENCE, COMPILER_INSPIRATION, SHIP_CANON)] targeted = embedder.vector(ABBEY_CANON.decode()) from app.vectors import cosine floor = [cosine(query, v) for v in unrelated] hit = cosine(query, targeted) assert min(floor) > 0.10, ( f"unrelated pairs score {floor} — the stub has no similarity floor and " "cannot model the real model's behaviour") assert hit > max(floor), f"targeted {hit} vs unrelated {floor}" # Real `nomic-embed-text` puts unrelated pairs around 0.36-0.56 and targeted # matches around 0.55-0.85. The stub need not match those numbers, but it # must have the same shape: a floor well clear of zero, under a clear hit. assert hit - max(floor) < 0.9, "the stub separates far more cleanly than reality" def test_a_relative_only_floor_would_admit_the_irrelevant_set(client): """The old rule, run against this fixture, still fails — as it must. This is what makes the suite able to detect M7-F1. It reproduces the superseded admission rule (a share of the best candidate) on the same vectors the corrected code sees, and shows it admitting the whole irrelevant library. """ embedder = RealisticEmbedder() from app.vectors import cosine query = embedder.vector(OFF_TOPIC_SCENE) raw = {name: cosine(query, embedder.vector(body.decode())) for name, body in ( ("abbey", ABBEY_CANON), ("crypt-ref", CRYPT_REFERENCE), ("mood", CRYPT_MOOD), ("ship", SHIP_CANON))} best = max(raw.values()) old_floor = max(0.02, best * 0.25) # the superseded rule admitted_by_old_rule = [n for n, c in raw.items() if c / best >= old_floor / best] assert len(admitted_by_old_rule) == len(raw), ( f"the old relative-only rule admitted {admitted_by_old_rule} of {raw} — " "this fixture must be able to fool it, or it cannot prove the fix") # ...and every one of them is below the absolute floor the fix uses. assert all(c < classes.SEMANTIC_FLOOR for c in raw.values()), raw # ======================================= the four hybrid cases, A B C D def test_case_a_strong_semantic_weak_lexical_still_retrieves(client): """A conceptual match with almost no shared vocabulary must survive.""" adv, _ = campaign(client, CRYPT_SCENE, [ ("ossuary.md", OSSUARY, "reference"), ("ship.md", SHIP_CANON, "canon"), ]) result = rank(client, adv) found = next((c for c in result.candidates if c.filename == "ossuary.md"), None) assert found is not None, table(result) assert found.admitted_by == "semantic", table(result) assert found.cosine >= classes.SEMANTIC_FLOOR, table(result) assert not found.matched_terms, table(result) assert "ship.md" not in names(result), table(result) def test_case_b_strong_lexical_weak_semantic_still_retrieves(client): """A distinctive exact term must retrieve even with embeddings unavailable.""" adv, _ = campaign(client, "Aldric asks about Westhaven and the broken circle.", [ ("abbey.md", ABBEY_CANON, "canon"), ("surgery.md", SURGERY_REFERENCE, "reference"), ]) with SessionLocal() as db: row = db.execute(select(models.Settings).where( models.Settings.user_id == client.user_id)).scalars().first() row.embedding_model = "" db.commit() result = rank(client, adv) assert result.semantic_used is False assert "abbey.md" in names(result), table(result) found = next(c for c in result.candidates if c.filename == "abbey.md") assert found.admitted_by == "lexical", table(result) assert len(found.matched_terms) >= classes.LEXICAL_MIN_TERMS, table(result) assert "surgery.md" not in names(result), table(result) def test_case_c_both_strong_ranks_once_and_is_not_duplicated(client): adv, _ = campaign(client, CRYPT_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("crypt-ref.md", CRYPT_REFERENCE, "reference"), ]) result = rank(client, adv) hybrid = [c for c in result.candidates if c.admitted_by == "both"] assert hybrid, table(result) ids = [c.chunk_id for c in result.candidates] assert len(ids) == len(set(ids)), table(result) assert all(c.lexical > 0 and c.semantic > 0 for c in hybrid), table(result) def test_case_d_neither_strong_retrieves_nothing(client): """**The mandatory case.** No match on either path means no chunks at all.""" adv, _ = campaign(client, OFF_TOPIC_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("crypt-ref.md", CRYPT_REFERENCE, "reference"), ("mood.md", CRYPT_MOOD, "inspiration"), ("ship.md", SHIP_CANON, "canon"), ]) result = rank(client, adv) assert result.candidates == [], table(result) assert result.suppressed == [], table(result) assert result.generated > 0, ( "nothing was even generated — the test would pass for the wrong reason") assert result.rejected == result.generated, table(result) # ...and the assembled prompt carries no imported section at all. report = client.get(f"/api/adventures/{adv}/context").json() assert report["knowledge"]["used"] == [] assert not [s for s in report["sections"] if s["label"].startswith("imported_")] assert not [s for s in report["sections"] if s["label"] == classes.SECTION_RULE] def test_case_d_holds_on_the_lexical_only_path_too(client): adv, _ = campaign(client, OFF_TOPIC_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("ship.md", SHIP_CANON, "canon"), ]) with SessionLocal() as db: row = db.execute(select(models.Settings).where( models.Settings.user_id == client.user_id)).scalars().first() row.embedding_model = "" db.commit() result = rank(client, adv) assert result.candidates == [], table(result) # ============================== authority must not rescue irrelevance @pytest.mark.parametrize("classification", ["canon", "reference", "inspiration"]) def test_irrelevant_material_is_excluded_whatever_its_class(client, classification): """Each class, alone in the library, with nothing else to compete with. The old rule admitted whatever was best; with one source there is nothing else, so "best" and "only" coincide and the failure is unmissable. """ adv, _ = campaign(client, OFF_TOPIC_SCENE, [ ("lore.md", ABBEY_CANON, classification), ]) result = rank(client, adv) assert result.candidates == [], table(result) assert result.generated >= 1, "nothing generated; the test proves nothing" def test_canon_is_excluded_even_though_it_is_the_best_candidate(client): """Explicitly the shape of M7-F1: best of a bad set is still not relevant.""" adv, _ = campaign(client, OFF_TOPIC_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("ship.md", SHIP_CANON, "canon"), ("surgery.md", SURGERY_REFERENCE, "reference"), ]) result = rank(client, adv) assert names(result) == [], table(result) def test_once_relevant_canon_outranks_relevant_reference_and_inspiration(client): """Authority still orders what did match — the other half of §30.""" adv, _ = campaign(client, CRYPT_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("crypt-ref.md", CRYPT_REFERENCE, "reference"), ("mood.md", CRYPT_MOOD, "inspiration"), ]) result = rank(client, adv) by = {c.filename: c for c in result.candidates} assert "abbey.md" in by, table(result) for lower in ("crypt-ref.md", "mood.md"): if lower in by: assert by["abbey.md"].score > by[lower].score, table(result) # and the class is what did it, at comparable relevance equal = 0.5 assert (equal * classes.CLASS_WEIGHTS[classes.CANON] > equal * classes.CLASS_WEIGHTS[classes.REFERENCE] > equal * classes.CLASS_WEIGHTS[classes.INSPIRATION]) def test_relevant_reference_outranks_irrelevant_canon(client): adv, _ = campaign(client, CRYPT_SCENE, [ ("crypt-ref.md", CRYPT_REFERENCE, "reference"), ("ship.md", SHIP_CANON, "canon"), ]) result = rank(client, adv) assert "crypt-ref.md" in names(result), table(result) assert "ship.md" not in names(result), table(result) # ================================================ the surviving mechanics def test_near_duplicates_are_suppressed_before_the_cut(client): adv, _ = campaign(client, CRYPT_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("abbey-copy.md", ABBEY_COPY, "canon"), ]) result = rank(client, adv) kept = [c for c in result.candidates if c.filename.startswith("abbey")] assert kept, table(result) assert len(kept) == 1, table(result) assert result.suppressed, table(result) assert all(c.duplicate_of is not None for c in result.suppressed) def test_suppression_never_crosses_a_class(client): adv, _ = campaign(client, CRYPT_SCENE, [ ("abbey.md", ABBEY_CANON, "canon"), ("abbey-copy.md", ABBEY_COPY, "reference"), ]) result = rank(client, adv) by_id = {c.chunk_id: c for c in result.candidates} for suppressed in result.suppressed: keeper = by_id.get(suppressed.duplicate_of) assert keeper is not None assert keeper.classification == suppressed.classification, table(result) def test_a_disabled_source_is_excluded_before_admission(client): adv, ids = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY_CANON, "canon")]) assert "abbey.md" in names(rank(client, adv)) client.patch(f"/api/adventures/{adv}/knowledge/{ids['abbey.md']}", json={"enabled": False}) embeddings.forget_cached(adv) after = rank(client, adv) assert after.candidates == [] assert after.generated == 0, "a disabled source still reached candidate generation" def test_a_source_in_another_campaign_cannot_win(client): adv_a, _ = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY_CANON, "canon")]) adv_b, _ = campaign(client, CRYPT_SCENE, []) result = rank(client, adv_b) assert result.candidates == [] and result.generated == 0 assert "abbey.md" in names(rank(client, adv_a)) def test_every_score_and_reason_is_recorded(client): adv, _ = campaign(client, CRYPT_SCENE, [("abbey.md", ABBEY_CANON, "canon")]) result = rank(client, adv) assert result.candidates, table(result) for candidate in result.candidates: record = candidate.as_record() for field in ("chunk_id", "source_id", "filename", "classification", "mode", "lexical", "semantic", "cosine", "score", "admitted_by", "matched_terms"): assert field in record, field assert record["mode"] in ("lexical", "semantic", "hybrid", "always") assert record["admitted_by"] in ("lexical", "semantic", "both") assert result.semantic_floor == classes.SEMANTIC_FLOOR assert result.generated >= len(result.candidates)