"""M7: accepting a local file into a campaign's knowledge library. One function does the whole job — validate, hash, store, chunk, index — and it does it inside one transaction, because the alternative is the state `IMPORTED-KNOWLEDGE-DESIGN.md` §57 forbids: a source presented as usable while only half its passages exist. ## The transactional boundary validate -> no row is written at all; the caller gets a 4xx and the reader's file is untouched build -> source row, every chunk row, every FTS row, and index_state='ready' all commit together, or none of them do `index_state` is the belt to that braces. Retrieval reads only sources marked `ready`, so even a hypothetical partial commit could not be retrieved from — it would be a stored source that never answers a query, which is inert rather than wrong. A failure after validation leaves `failed` with the reason on the row. Embeddings are deliberately *outside* that boundary. They need a network call to Ollama, and a knowledge library that cannot be imported while the inference host is down would be a worse product than one whose semantic index lags. So the import commits lexically complete and the vectors are filled in afterwards, by `embeddings.py`, at import time and again after any later turn. ## Path safety There is none to get wrong, and that is the design. The only import surface is an HTTP upload: the router takes `UploadFile`, and this module takes bytes and a filename *string*. No caller anywhere accepts a server-side pathname, so there is no path to canonicalize, no root to compare against, and no symlink to resolve. `H08` is satisfied by the absence of the mechanism rather than by a check that could later be bypassed — and `safe_filename` below still strips every separator and traversal segment, because the name is displayed and stored and a `../../etc/passwd` in a title is at best confusing. """ from __future__ import annotations import unicodedata from sqlalchemy import select from sqlalchemy.orm import Session from .. import models from . import chunking, classes, fts # ---------------------------------------------------------------- the limits # # Every one of these is enforced here, on the server, and each raises a message # that says what to do. Nothing is silently truncated: a source is accepted # whole or refused with a reason (`SECURITY-THREAT-MODEL.md` §20-21, # `IMPORTED-KNOWLEDGE-DESIGN.md` §59-60). #: The largest file accepted, in bytes. One mebibyte of prose is roughly a #: 150,000-word book — far past any setting bible — and it sits comfortably #: under `limits.MAX_BODY_BYTES` (2 MiB), which the multipart request as a whole #: still has to fit inside. Raising this past that ceiling would produce a #: confusing 413 from the middleware instead of the message below. MAX_SOURCE_BYTES = 1024 * 1024 #: The most passages one source may produce. At the chunker's floor of 60 tokens #: a megabyte cannot reach this, so in practice it is a guard against a future #: chunker change rather than against a user, and it fails loudly if one is ever #: made that fragments badly. MAX_CHUNKS_PER_SOURCE = 4000 #: The most sources one campaign may hold. Bounds the retrieval scan and the #: export bundle. MAX_SOURCES_PER_ADVENTURE = 200 ALLOWED_EXTENSIONS = (".txt", ".md") MEDIA_TYPES = {".txt": "text/plain", ".md": "text/markdown"} #: Control characters that no text file legitimately contains. Tab, newline and #: carriage return are excluded because they plainly do. A file carrying any of #: these is binary that happened to decode, and it is refused. _BINARY_CONTROLS = frozenset( chr(c) for c in list(range(0, 9)) + [11, 12] + list(range(14, 32)) + [127] ) class ImportError_(ValueError): """A file that cannot be accepted, with the reason a reader needs. Named with a trailing underscore so it cannot be confused with the builtin of the same name, which means something else entirely. """ def __init__(self, message: str, *, conflict: dict | None = None): super().__init__(message) #: Set when the refusal is a duplicate rather than a fault, so the #: router can answer 409 and name the source already holding the #: content instead of a flat "rejected". self.conflict = conflict # ------------------------------------------------------------- validation DEFAULT_FILENAME = "imported.txt" def safe_filename(name: str) -> str: """The displayable basename of an uploaded filename. A *metadata* cleaner, not a path check — nothing downstream opens anything, so there is no path here for a check to protect. What this protects is the stored string: a name that reads as a path, carries a traversal segment, or smuggles a NUL or a newline into a list screen would be confusing at best and misleading at worst. The rule is "take the basename", because that is what an uploaded filename *is*. Everything before the last separator described a directory on the sender's machine, which this one does not have and will never look for, so `../../../../etc/passwd.md` stores as `passwd.md`. Leading dots then go, so a stored name can never be `..`, `.` or a hidden file. """ name = unicodedata.normalize("NFC", name or "").replace("\x00", "") for separator in ("\\", "/"): name = name.rsplit(separator, 1)[-1] # Drop Unicode format characters (category Cf), which are invisible and # include the bidirectional overrides. `U+202E` before "exe.dm.md" renders # as "dm.exe" in most UIs, so a name could otherwise lie about its own # extension on the screen it is displayed on (review finding M7-F5). They # carry no information in a filename, so removing them costs nothing. name = "".join(c for c in name if unicodedata.category(c) != "Cf") name = " ".join(name.split()).lstrip(". ") return (name or DEFAULT_FILENAME)[:255] def extension_of(filename: str) -> str: lowered = safe_filename(filename).lower() for extension in ALLOWED_EXTENSIONS: if lowered.endswith(extension): return extension return "" def decode(raw: bytes, filename: str) -> str: """Bytes to text, or a refusal that says which rule was broken. Three checks, in the order a wrong file is most likely to fail them: * **Size**, first, so a huge file is refused before it is decoded. * **Encoding**, strictly UTF-8. `SECURITY-THREAT-MODEL.md` §21 and `IMPORTED-KNOWLEDGE-DESIGN.md` §60 both ask for a clear rejection over a silent mangling, so there is no `errors="replace"` here and no charset guessing. A UTF-8 BOM is accepted and stripped, because Windows editors write one and it is not a different encoding. * **Content**, because an extension is not evidence. §21: "do not trust file extensions alone... verify readable text content, reject obvious binary data." A NUL byte or a scattering of C0 controls is what a `.txt`-renamed binary looks like after it fails to be anything else. """ if len(raw) > MAX_SOURCE_BYTES: raise ImportError_( f"“{safe_filename(filename)}” is " f"{len(raw) / 1024 / 1024:.1f} MB. The limit for one knowledge " f"source is {MAX_SOURCE_BYTES // 1024 // 1024} MB — split the file " "and import the parts, so nothing is silently left out." ) if not raw.strip(): raise ImportError_(f"“{safe_filename(filename)}” is empty.") if raw.startswith(b"\xef\xbb\xbf"): raw = raw[3:] try: text = raw.decode("utf-8") except UnicodeDecodeError as exc: raise ImportError_( f"“{safe_filename(filename)}” is not valid UTF-8 text (byte " f"{exc.start} is not part of a valid character). Save it as UTF-8 " "and import it again — the file has not been changed." ) from None controls = sum(1 for character in text if character in _BINARY_CONTROLS) if controls: raise ImportError_( f"“{safe_filename(filename)}” contains {controls} control " "character(s) that do not belong in a text file. It looks like " "binary data rather than text, and only .txt and .md are supported." ) return text def validate( raw: bytes, filename: str, classification: str, visibility: str = classes.NORMAL, ) -> tuple[str, str, str]: """Everything checked before a row is written. Returns (text, extension, title).""" extension = extension_of(filename) if not extension: raise ImportError_( f"“{safe_filename(filename)}” is not a supported file type. This " "version imports .txt and .md files." ) if not classes.is_class(classification): raise ImportError_( f"“{classification}” is not a knowledge class. Choose Canon, " "Reference or Inspiration." ) if not classes.is_visibility(visibility): raise ImportError_(f"“{visibility}” is not a visibility.") text = decode(raw, filename) clean = safe_filename(filename) return text, extension, clean[: -len(extension)] or clean # ----------------------------------------------------------------- importing def import_source( db: Session, adventure: models.Adventure, *, raw: bytes, filename: str, classification: str, title: str = "", visibility: str = classes.NORMAL, always_include: bool = False, allow_duplicate: bool = False, ) -> models.KnowledgeSource: """Validates, stores, chunks and indexes one file. All of it, or none of it. The caller commits. Nothing here commits or rolls back, so an exception leaves the session dirty and the router's error path discards it — which is what makes "no active partial source, no half-built FTS rows, no half-valid chunk set" true by construction rather than by cleanup. """ text, extension, derived_title = validate(raw, filename, classification, visibility) clean_name = safe_filename(filename) existing = db.execute( select(models.KnowledgeSource).where( models.KnowledgeSource.adventure_id == adventure.id ).limit(MAX_SOURCES_PER_ADVENTURE + 1) ).scalars().all() if len(existing) >= MAX_SOURCES_PER_ADVENTURE: raise ImportError_( f"This campaign already holds {len(existing)} knowledge sources, " f"which is the limit of {MAX_SOURCES_PER_ADVENTURE}. Delete one to " "make room." ) # Duplicate detection, over the normalized text, within this campaign only. # §13 forbids silently creating a second copy and indexing it twice; it does # not forbid the reader deciding they want one anyway, which is what # `allow_duplicate` is. A deliberately simple v1 model: no versioning UI, no # supersession chain, and the refusal names the source that already holds # the content so the choice is an informed one. content_hash = chunking.digest(text) if not allow_duplicate: twin = next((s for s in existing if s.content_hash == content_hash), None) if twin is not None: raise ImportError_( f"This campaign already holds identical content, imported as " f"“{twin.title}”. Import it again only if you want a second " "copy with its own classification.", conflict={ "source_id": twin.id, "title": twin.title, "classification": twin.classification, "content_hash": content_hash, }, ) if classification != classes.CANON: # Always-include is a Canon-only mechanism (`IMPORTED-KNOWLEDGE-DESIGN.md` # §32, `CONTEXT-AND-MEMORY.md` §41-42). The reason is that the flag # bypasses relevance entirely: asserting unranked Reference on every # turn would spend a protected budget on material that establishes # nothing. always_include = False source = models.KnowledgeSource( adventure_id=adventure.id, title=(title.strip() or derived_title)[:200], original_filename=clean_name, classification=classification, visibility=visibility, always_include=always_include, enabled=True, content=text, content_hash=content_hash, byte_size=len(raw), media_type=MEDIA_TYPES[extension], parser_version=chunking.PARSER_VERSION, chunking_version=chunking.CHUNKING_VERSION, index_state="pending", ) db.add(source) db.flush() # the chunks need the source's id build_index(db, source, markdown=extension == ".md") return source def build_index( db: Session, source: models.KnowledgeSource, *, markdown: bool | None = None ) -> int: """(Re)builds one source's passages and its lexical index. Returns the count. This is both half of an import and the whole of a lexical reindex, which is the point: there is one code path that turns content into passages, so a reindexed source is byte-identical to a freshly imported one. It leaves the source `ready` or raises, and it does not touch the source's content, classification, visibility or enabled state. """ if markdown is None: markdown = source.media_type == "text/markdown" clear_index(db, source) passages = chunking.chunk(source.content, markdown=markdown) if len(passages) > MAX_CHUNKS_PER_SOURCE: raise ImportError_( f"“{source.original_filename}” splits into {len(passages)} " f"passages, past the limit of {MAX_CHUNKS_PER_SOURCE}." ) for passage in passages: chunk_row = models.KnowledgeChunk( source_id=source.id, adventure_id=source.adventure_id, chunk_index=passage.index, heading_path=passage.heading_path, text=passage.text, token_count=passage.token_count, content_hash=passage.content_hash, ) db.add(chunk_row) db.flush() # the FTS rowid is the chunk's primary key fts.add(db, chunk_row.id, passage.heading_path, passage.text) source.parser_version = chunking.PARSER_VERSION source.chunking_version = chunking.CHUNKING_VERSION source.index_state = "ready" source.index_detail = "" return len(passages) def clear_index(db: Session, source: models.KnowledgeSource) -> None: """Removes a source's passages, its FTS rows and its vectors. The FTS rows go first, by id, while the ids still exist. Deleting the chunk rows first would leave the index holding rowids that point at nothing, and a search would then return chunk ids that no longer resolve. """ chunk_ids = list( db.execute( select(models.KnowledgeChunk.id).where( models.KnowledgeChunk.source_id == source.id ) ).scalars().all() ) if not chunk_ids: return fts.remove_chunks(db, chunk_ids) db.query(models.KnowledgeEmbedding).filter( models.KnowledgeEmbedding.chunk_id.in_(chunk_ids) ).delete(synchronize_session=False) db.query(models.KnowledgeChunk).filter( models.KnowledgeChunk.source_id == source.id ).delete(synchronize_session=False) db.expire(source, ["chunks"]) def delete_source(db: Session, source: models.KnowledgeSource) -> None: """Removes a source and everything derived from it. What it does **not** remove is the evidence of what old narrator turns were given. That lives in each turn's own context snapshot as rendered text, not as a reference to a live chunk row, so deleting a source cannot turn a historical prompt into a set of dangling ids (`IMPORTED-KNOWLEDGE-DESIGN.md` §49-50, `DATA-MODEL.md` §25). Story history, the head and the authoritative state are untouched. """ clear_index(db, source) db.delete(source)