"""M7: the shapes a retrieval produces, with no dependencies of their own. `retrieval.py` fills these in and `inject.py` prices them; `context/builder.py` needs to name the result type in its signature. Putting the two dataclasses in their own module is what lets all three refer to them without the builder having to import the retrieval machinery — which reaches the database, the provider and `context` itself, and would close the import graph into a cycle. Nothing here decides anything. The scoring rules live in `retrieval.py`, the budget rules in `inject.py`, and the class weights in `classes.py`. """ from __future__ import annotations from dataclasses import dataclass, field @dataclass class Candidate: """One passage, with everything that decided its place.""" chunk_id: int source_id: int title: str filename: str classification: str visibility: str chunk_index: int heading_path: str text: str token_count: int always_include: bool = False #: Both normalized against the best of their own path for this query, so #: that they can be compared with each other. See `retrieval.py`. lexical: float = 0.0 semantic: float = 0.0 #: The raw cosine behind `semantic`. This is the value **admission** uses, #: because a normalized score cannot tell "everything matched well" from #: "nothing did" — which is the defect the M7 corrective pass fixed. cosine: float = 0.0 relevance: float = 0.0 #: Which path admitted this passage: "lexical", "semantic" or "both". #: Empty for an always-included passage, which is asserted rather than #: matched and is not subject to admission at all. admitted_by: str = "" #: The distinct query terms this passage actually contains, when the #: lexical path admitted it. This is the evidence, shown in the inspector. matched_terms: list = field(default_factory=list) score: float = 0.0 #: Set when this passage was set aside as repeating one already chosen. duplicate_of: int | None = None @property def mode(self) -> str: if self.always_include: return "always" if self.admitted_by == "both": return "hybrid" return self.admitted_by or "lexical" def as_record(self) -> dict: """The provenance the inspector and the tests read (F05, F06).""" return { "chunk_id": self.chunk_id, "source_id": self.source_id, "title": self.title, "filename": self.filename, "classification": self.classification, "visibility": self.visibility, "chunk_index": self.chunk_index, "heading_path": self.heading_path, "tokens": self.token_count, "always_include": self.always_include, "mode": self.mode, "lexical": round(self.lexical, 4), "semantic": round(self.semantic, 4), "cosine": round(self.cosine, 4), "admitted_by": self.admitted_by, "matched_terms": list(self.matched_terms), "score": round(self.score, 4), } @dataclass class Result: """What one retrieval produced, before the budget is applied.""" candidates: list[Candidate] = field(default_factory=list) suppressed: list[Candidate] = field(default_factory=list) terms: list[str] = field(default_factory=list) considered: int = 0 #: How many distinct passages either path produced as candidates, before #: admission, and how many of them admission then rejected. Together these #: are what makes "the library was searched and nothing matched" legible #: rather than indistinguishable from "the library was never searched". generated: int = 0 rejected: int = 0 #: The raw cosine a passage had to reach to be admitted semantically. Zero #: when the configured embedding model has no calibration in this build, in #: which case no semantic admission happened at all. semantic_floor: float = 0.0 #: Whether this build has a measured relevance calibration for the #: configured embedding model. False means semantic retrieval was skipped #: rather than attempted and failed — a different thing, and the reason is #: in `semantic_note`. semantic_calibrated: bool = False embedding_model: str = "" semantic_used: bool = False #: A human-readable reason the semantic half did not run or did not finish. #: Never a failure of the retrieval as a whole: lexical results stand. semantic_note: str = "" scan_truncated: bool = False