"""M7: the semantic path, end to end, against a real local embedding model. M2 shipped with the memory bank dead and the suite green, because every test stubbed the provider factories out. M6 answered that with `test_provider_wiring.py` and the rule that at least one real provider-construction path must be exercised per milestone. This is M7's. **Nothing here is mocked.** A real `Settings` row is read back out of the database, the real factory builds the provider from it, a real request reaches the configured local Ollama, the vectors it returns are stored in `knowledge_embeddings`, and the real hybrid retrieval ranks against them and inserts the winner into a prompt built by the real context builder. It is skipped without an endpoint, and it is reported separately from the deterministic suite, because it needs a machine with a model on it: AIDND_TEST_ENDPOINT=https://inference.lan:8443/v1 \\ AIDND_TEST_EMBED_MODEL=nomic-embed-text \\ backend/.venv/bin/python -m pytest backend/tests/test_knowledge_real_model.py -v -s The endpoint goes through the ordinary policy: no allowlist bypass, no TLS weakening. A public endpoint is refused here exactly as it is in production, and the test asserts that rather than assuming it. """ import asyncio import os import pytest from fastapi import Depends from fastapi.testclient import TestClient from sqlalchemy import select from app import auth, endpoints, limits, memorybank, models from app.database import Base, SessionLocal, engine, get_db from app.knowledge import classes, embeddings, retrieval from app.main import app from app.routers import adventures from fakes import ScriptedProvider, state_block pytestmark = pytest.mark.skipif( not os.environ.get("AIDND_TEST_ENDPOINT"), reason="set AIDND_TEST_ENDPOINT (and AIDND_TEST_EMBED_MODEL) to run this", ) ENDPOINT = os.environ.get("AIDND_TEST_ENDPOINT", "") EMBED_MODEL = os.environ.get("AIDND_TEST_EMBED_MODEL", "nomic-embed-text") CANON_MD = """# The Old Abbey The Old Abbey lies five miles north of Westhaven. The abbey crypt bears a symbol shaped like a broken circle. """ REFERENCE_MD = """# Medieval Taverns Medieval taverns commonly used timber framing, stone hearths, benches, shared tables, candles, and oil lamps. """ # The conceptual case: about the crypt, sharing almost none of its words. If the # stored vectors were nonsense, this is the source that would not be found. OSSUARY_MD = """# The Ossuary Bones were stacked in the undercroft below the chancel, sorted and shelved by the brothers who kept the sanctuary. """ @pytest.fixture() def client(monkeypatch): """A campaign wired to the real endpoint. Only the *narrator* is scripted. The narrator is scripted because this file is about embeddings and a real narration would make it slow and non-deterministic for no gain. The embedding path — factory, request, storage, retrieval — is entirely real. """ assert endpoints.rejection_reason(ENDPOINT) is None, ( f"the configured test endpoint {ENDPOINT} is refused by the policy" ) Base.metadata.create_all(bind=engine) memorybank._vector_cache.clear() embeddings._cache.clear() setup = SessionLocal() user = models.User(is_guest=False, email="m7real@example.com") setup.add(user) setup.flush() setup.add(models.Settings( user_id=user.id, endpoint_url=ENDPOINT, model=os.environ.get("AIDND_TEST_MODEL", "test-model"), embedding_model=EMBED_MODEL, context_token_budget=6000, max_output_tokens=300, )) adventure = models.Adventure(user_id=user.id, title="Real Model") setup.add(adventure) setup.flush() setup.add(models.Action( adventure_id=adventure.id, type="start", text="Aldric stands in the crypt beneath the Old Abbey, north of Westhaven.", )) setup.commit() adv_id, user_id = adventure.id, user.id setup.close() monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None) monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", ScriptedProvider) app.dependency_overrides[auth.get_current_user] = ( lambda db=Depends(get_db): db.get(models.User, user_id) ) test_client = TestClient(app) test_client.adv_id = adv_id test_client.user_id = user_id try: yield test_client finally: app.dependency_overrides.clear() memorybank._vector_cache.clear() embeddings._cache.clear() Base.metadata.drop_all(bind=engine) def upload(client, name, body, classification): response = client.post( f"/api/adventures/{client.adv_id}/knowledge", files={"file": (name, body.encode(), "text/markdown")}, data={"classification": classification}, ) assert response.status_code == 201, response.text[:400] return response.json() def settings_row(client, db): return db.execute(select(models.Settings).where( models.Settings.user_id == client.user_id)).scalars().first() def test_a_real_local_model_embeds_stores_retrieves_and_reaches_the_prompt(client): """The whole semantic path, with nothing stubbed between here and Ollama.""" canon = upload(client, "canon.md", CANON_MD, "canon") upload(client, "reference.md", REFERENCE_MD, "reference") ossuary = upload(client, "ossuary.md", OSSUARY_MD, "reference") # 1. Real vectors were stored, by the import path, through the real factory. # Import embeds inline, so this is already true before anything else runs. with SessionLocal() as db: rows = db.execute(select(models.KnowledgeEmbedding).where( models.KnowledgeEmbedding.adventure_id == client.adv_id )).scalars().all() assert rows, "no vectors were stored" for row in rows: assert row.model == EMBED_MODEL assert row.dimensions > 64, row.dimensions assert len(row.vector) == row.dimensions * 4 # packed float32 dimensions = rows[0].dimensions assert all(row.dimensions == dimensions for row in rows) listing = {row["original_filename"]: row for row in client.get(f"/api/adventures/{client.adv_id}/knowledge").json()} for name, row in listing.items(): assert row["embed_state"] == "ok", (name, row["embed_detail"]) assert row["embedded_count"] == row["chunk_count"] status = client.get(f"/api/adventures/{client.adv_id}/knowledge-status").json() assert status["semantic_enabled"] is True assert status["embedding_model"] == EMBED_MODEL assert status["pending_embeddings"] == 0 assert status["failed_embedding"] == [] # 2. Real semantic retrieval, against those stored vectors. with SessionLocal() as db: adventure = db.get(models.Adventure, client.adv_id) result = asyncio.run(retrieval.retrieve(adventure, settings_row(client, db))) assert result.semantic_used, result.semantic_note scored = {c.filename: c for c in result.candidates} print("\n real-model ranking:") for candidate in result.candidates: print(f" {candidate.filename:16} {candidate.classification:12} " f"lex={candidate.lexical:.3f} sem={candidate.semantic:.3f} " f"cos={candidate.cosine:.3f} score={candidate.score:.3f}") for candidate in result.suppressed: print(f" {candidate.filename:16} SUPPRESSED") assert scored, "the real model retrieved nothing" assert any(c.cosine > 0 for c in result.candidates) # The conceptual match is the thing only a real embedding can do here: # `ossuary.md` shares almost no words with the scene and is about it. if "ossuary.md" in scored: assert scored["ossuary.md"].semantic > 0 print(f" conceptual match found: ossuary.md at cosine " f"{scored['ossuary.md'].cosine:.3f}") # 3. It reaches a prompt built by the real context builder. ScriptedProvider.replies = [f"The crypt is cold and still.\n{state_block([])}"] turn = client.post(f"/api/adventures/{client.adv_id}/actions", json={"type": "do", "text": "Aldric studies the crypt walls."}) assert turn.status_code == 200, turn.text[:300] report = client.get(f"/api/adventures/{client.adv_id}/context").json() assert report["knowledge"]["semantic_used"] is True assert report["knowledge"]["used"], report["knowledge"]["semantic_note"] used = {u["filename"]: u for u in report["knowledge"]["used"]} assert any(u["mode"] in ("semantic", "hybrid") for u in used.values()), used assert any(s["label"].startswith("imported_") for s in report["sections"]) print(f" prompt sections: " f"{[s['label'] for s in report['sections'] if s['label'].startswith('imported_')]}") assert canon and ossuary # ============================ the M7 corrective regression: admission ======== # # The failure class this exists to prevent: a deterministic stub that is more # discriminative than the real model, hiding an admission gate that cannot say # "no match" (review findings M7-F1 and M7-F2). The deterministic suite is the # normal required path; this is the reality check, and it prints the measured # separation so a model change surfaces as data rather than as a mystery. #: Passages that share almost no vocabulary with their query but are about the #: same thing — the case the semantic half of the hybrid exists to serve. PARAPHRASE_QUERY = ("What emblem is carved in the burial vault beneath the " "ruined monastery up the road from town?") #: Scenes with no connection to a fantasy campaign at all. OFF_TOPIC = [ "The kiln was held at cone six for a two-hour soak while the glaze matured.", "The compiler emits a diagnostic when the lifetime of the borrow outlives " "the referent.", "The surgeon sterilised the cannula and checked the infusion pump pressure.", "He reconciled the ledger against the quarterly depreciation schedule.", "She practised the fugue slowly, counting the subject's entries.", ] def _cosines(client, adv, texts): """Raw cosine of each text against every stored vector, as retrieval sees it.""" from app.vectors import cosine, unpack with SessionLocal() as db: settings = settings_row(client, db) rows = db.execute( select(models.KnowledgeEmbedding.vector, models.KnowledgeSource.original_filename) .join(models.KnowledgeChunk, models.KnowledgeChunk.id == models.KnowledgeEmbedding.chunk_id) .join(models.KnowledgeSource, models.KnowledgeSource.id == models.KnowledgeChunk.source_id) .where(models.KnowledgeSource.adventure_id == adv)).all() vectors = [(name, unpack(blob)) for blob, name in rows] embedded = asyncio.run( memorybank.embedding_provider(settings).embed(list(texts))) return {text: {name: cosine(vector, stored) for name, stored in vectors} for text, vector in zip(texts, embedded)} def test_the_real_model_separates_relevant_from_unrelated(client): """The measurement the admission floor rests on, re-taken every run. Fails if the configured model's scale moves far enough that `classes.SEMANTIC_FLOOR` stops sitting between the two populations — which is the one way this build could silently go back to admitting everything or start admitting nothing. """ upload(client, "canon.md", CANON_MD, "canon") upload(client, "reference.md", REFERENCE_MD, "reference") targeted = { "Aldric asks about the Old Abbey north of Westhaven and its " "broken-circle symbol.": "canon.md", PARAPHRASE_QUERY: "canon.md", "Aldric looks around the tavern at the stone hearth and the timber " "beams.": "reference.md", } scores = _cosines(client, client.adv_id, list(targeted) + OFF_TOPIC) hits = [scores[q][want] for q, want in targeted.items()] misses = [c for q in OFF_TOPIC for c in scores[q].values()] print(f"\n real-model separation ({EMBED_MODEL}):") for q, want in targeted.items(): print(f" targeted {scores[q][want]:.4f} {q[:52]}") for q in OFF_TOPIC: for name, c in scores[q].items(): print(f" off-topic {c:.4f} {q[:40]:40} -> {name}") print(f" floor = {classes.SEMANTIC_FLOOR}") assert min(hits) > classes.SEMANTIC_FLOOR, ( f"targeted matches {sorted(hits)} fall below the floor " f"{classes.SEMANTIC_FLOOR}; relevant material would be dropped") assert max(misses) < classes.SEMANTIC_FLOOR, ( f"off-topic pairs reach {max(misses):.4f}, at or above the floor " f"{classes.SEMANTIC_FLOOR}; irrelevant material would be admitted") def test_a_completely_unrelated_query_retrieves_nothing_from_a_real_model(client): """**The no-match case, end to end, with nothing mocked.** A mixed library of Canon, Reference and Inspiration, all embedded by the real model, and a scene about none of them. The prompt must carry no imported section at all. """ upload(client, "canon.md", CANON_MD, "canon") upload(client, "reference.md", REFERENCE_MD, "reference") upload(client, "ossuary.md", OSSUARY_MD, "inspiration") # The retrieval query is built from the recent story window, so the whole # window has to move off-topic — one off-topic line after a crypt opening # still leaves the crypt in the query, which is correct behaviour and would # make this test prove nothing. with SessionLocal() as db: adventure = db.get(models.Adventure, client.adv_id) adventure.narrative_state = None for depth, text in enumerate(OFF_TOPIC[:4], start=1): db.add(models.Action( adventure_id=client.adv_id, type="do", text=text, branch_id=adventure.head_branch_id, depth=depth, live=True)) adventure.head_depth = 4 db.commit() report = client.get(f"/api/adventures/{client.adv_id}/context").json() knowledge = report["knowledge"] print(f"\n generated={knowledge['generated']} " f"rejected={knowledge['rejected']} used={len(knowledge['used'])}") assert knowledge["generated"] > 0, "nothing was generated; this proves nothing" assert knowledge["used"] == [], [u["filename"] for u in knowledge["used"]] assert not [s for s in report["sections"] if s["label"].startswith("imported_")] def test_a_relevant_query_still_retrieves_from_a_real_model(client): """The positive control for the test above, on the same library.""" upload(client, "canon.md", CANON_MD, "canon") upload(client, "reference.md", REFERENCE_MD, "reference") upload(client, "ossuary.md", OSSUARY_MD, "inspiration") with SessionLocal() as db: adventure = db.get(models.Adventure, client.adv_id) db.add(models.Action( adventure_id=client.adv_id, type="do", text="Aldric asks Mara about the Old Abbey north of Westhaven and " "the broken-circle symbol in its crypt.", branch_id=adventure.head_branch_id, depth=1, live=True)) adventure.head_depth = 1 db.commit() knowledge = client.get( f"/api/adventures/{client.adv_id}/context").json()["knowledge"] used = [u["filename"] for u in knowledge["used"]] print(f"\n retrieved: {used}") assert "canon.md" in used, used for record in knowledge["used"]: assert record["admitted_by"] in ("lexical", "semantic", "both") def test_a_paraphrase_still_retrieves_from_a_real_model(client): """Strong semantic, weak lexical, against the real model.""" upload(client, "canon.md", CANON_MD, "canon") scores = _cosines(client, client.adv_id, [PARAPHRASE_QUERY]) cosine_value = scores[PARAPHRASE_QUERY]["canon.md"] print(f"\n paraphrase cosine: {cosine_value:.4f} " f"(floor {classes.SEMANTIC_FLOOR})") assert cosine_value >= classes.SEMANTIC_FLOOR, ( "a genuine paraphrase falls below the admission floor") def test_a_reindex_rebuilds_real_vectors(client): """Reindex against the real endpoint: vectors go and come back.""" upload(client, "canon.md", CANON_MD, "canon") with SessionLocal() as db: before = len(db.execute(select(models.KnowledgeEmbedding)).scalars().all()) assert before > 0 out = client.post(f"/api/adventures/{client.adv_id}/knowledge/reindex").json() assert out["semantic"] is True assert out["embedded"] == before with SessionLocal() as db: rows = db.execute(select(models.KnowledgeEmbedding)).scalars().all() assert len(rows) == before assert all(row.model == EMBED_MODEL for row in rows) def test_the_real_embedding_path_still_obeys_the_endpoint_policy(client): """The policy is checked before every request, on this path too.""" from app.providers import ProviderError with SessionLocal() as db: row = settings_row(client, db) row.endpoint_url = "https://api.openai.com/v1" db.commit() upload_body = {"classification": "canon"} # The import itself succeeds — lexical indexing needs no network — and the # embedding attempt behind it is refused by the policy rather than sent. response = client.post( f"/api/adventures/{client.adv_id}/knowledge", files={"file": ("blocked.md", CANON_MD.encode(), "text/markdown")}, data=upload_body, ) assert response.status_code == 201 assert response.json()["index_state"] == "ready" with SessionLocal() as db: adventure = db.get(models.Adventure, client.adv_id) provider = memorybank.embedding_provider(settings_row(client, db)) with pytest.raises(ProviderError) as exc: asyncio.run(provider.embed(["a line of someone's story"])) assert "can't be used" in str(exc.value) assert adventure is not None