"""M7: the chunker, on its own. Chunking is derived data that three other things assume is reproducible: an export carries only the source text, an import rebuilds the passages from it, and a reindex throws them away and rebuilds them again. All three are wrong if the same bytes can produce different passages, so determinism is asserted here directly rather than inferred from those features working once. The cases cover what `IMPORTED-KNOWLEDGE-DESIGN.md` §15-18, §59 and §61 ask of chunking — a small file, multi-heading Markdown, a long paragraph, Unicode text, and a file near the import limit — plus the two failure shapes the sizing rules exist to prevent. python -m pytest tests/test_knowledge_chunking.py -v """ import pytest from app.knowledge import chunking, fts, importer def hashes(passages): return [p.content_hash for p in passages] def test_the_same_source_always_produces_the_same_passages(): """Determinism, over a document with every structure in it at once.""" source = ( "# Setting\n\nA world of rain and stone.\n\n" "## Westhaven\n\nA town on the north road, five miles south of the abbey.\n\n" "### The Abbey\n\nThe crypt bears a broken circle.\n\n" "```\ncode = 'not a # heading'\n```\n\n" "## Rules\n\nResurrection is impossible.\n" ) first = chunking.chunk(source) for _ in range(5): again = chunking.chunk(source) assert hashes(again) == hashes(first) assert [p.text for p in again] == [p.text for p in first] assert [p.heading_path for p in again] == [p.heading_path for p in first] assert [p.index for p in again] == list(range(len(first))) def test_a_small_file_is_one_passage(): passages = chunking.chunk("The Old Abbey lies five miles north of Westhaven.\n") assert len(passages) == 1 assert passages[0].index == 0 assert passages[0].token_count > 0 assert passages[0].heading_path == "" def test_markdown_headings_become_the_passage_trail(): source = "\n\n".join( ["# Setting"] + ["A paragraph about the setting. " * 20] + ["## Westhaven"] + ["A paragraph about the town. " * 20] + ["### The Old Abbey"] + ["A paragraph about the abbey and its crypt. " * 20] ) passages = chunking.chunk(source) trails = [p.heading_path for p in passages] assert "Setting" in trails assert "Setting > Westhaven" in trails assert "Setting > Westhaven > The Old Abbey" in trails # A trail is context, so it goes into the index as well as onto the row. line = fts.index_line(passages[-1].heading_path, passages[-1].text) assert "The Old Abbey" in line def test_a_run_of_tiny_sections_does_not_become_a_run_of_fragments(): """The failure the packing rule exists to prevent.""" source = "\n\n".join( f"## Section {n}\n\nOne short line about section {n}." for n in range(40) ) passages = chunking.chunk(source) assert len(passages) < 40, "every heading became its own fragment" assert all(p.token_count >= chunking.MIN_TOKENS for p in passages[:-1]) # Nothing was lost: every section's body is still findable, and so is its # heading — as the passage's own trail for whichever section opened it, and # written into the text for every section packed in after that. joined = "\n".join(p.text for p in passages) trails = {p.heading_path for p in passages} for n in range(40): assert f"section {n}." in joined assert f"Section {n}" in joined or f"Section {n}" in trails def test_a_long_paragraph_is_split_and_a_long_section_does_not_become_one_giant(): long_paragraph = "The abbey stands above the salt flats. " * 400 passages = chunking.chunk(f"# Abbey\n\n{long_paragraph}") assert len(passages) > 1 assert all(p.token_count <= chunking.TARGET_MAX for p in passages) assert all(p.heading_path == "Abbey" for p in passages) # And the text survives the split. assert "The abbey stands above the salt flats." in passages[0].text assert "The abbey stands above the salt flats." in passages[-1].text def test_a_single_unbroken_run_of_text_still_terminates(): """A wall of characters with no sentence, no word break and no heading. The point is that it terminates and stays inside the ceiling. This is the last-resort cut, which joins its slices with whitespace — so the characters are all still there, and the boundaries between slices are not exactly where they were. That is a documented consequence for a pathological input (a base64 blob, or an unsegmented script) rather than something that happens to prose, and it is asserted here so a change to it is deliberate. """ passages = chunking.chunk("x" * 60_000) assert len(passages) > 1 assert all(p.token_count <= chunking.TARGET_MAX for p in passages) recovered = "".join(p.text for p in passages) assert "".join(recovered.split()) == "x" * 60_000 def test_unicode_text_is_chunked_and_hashed_stably(): source = ( "# Café de la Résistance\n\n" "Le vieux marin regardait la pluie tomber sur les volets sombres. " * 20 + "\n\n## Ελληνικά\n\n" + "Ο ταξιδιώτης μπήκε σε μια σιωπηλή αίθουσα. " * 20 + "\n\n## 日本語\n\n" + "旅人は静かな広間に入った。雨が暗い雨戸を叩いていた。" * 20 ) passages = chunking.chunk(source) assert passages assert hashes(chunking.chunk(source)) == hashes(passages) joined = "\n".join(p.text for p in passages) assert "Résistance" in "\n".join(p.heading_path for p in passages) or "Résistance" in joined assert "ταξιδιώτης" in joined assert "旅人" in joined def test_normalization_is_stable_across_line_endings_and_unicode_forms(): """§61: one normalization for hashing, duplicate detection and search.""" # The same accented character, composed and decomposed. composed = "Café de la Résistance\n" decomposed = "Café de la Résistance\n" assert chunking.digest(composed) == chunking.digest(decomposed) # ...and the same file through Windows. assert chunking.digest("a\nb\n") == chunking.digest("a\r\nb\r\n") # Trailing whitespace is invisible and must not make two files differ. assert chunking.digest("a\nb\n") == chunking.digest("a \nb\t\n") # But real differences still differ. assert chunking.digest("a\nb\n") != chunking.digest("a\nc\n") def test_a_file_at_the_import_limit_chunks_within_bounds(): """The largest source the importer accepts, chunked end to end.""" paragraph = "The crypt beneath the abbey is cold and the walls are damp. " body = "\n\n".join(paragraph * 12 for _ in range(1400)) body = body[: importer.MAX_SOURCE_BYTES - 100] assert len(body.encode("utf-8")) <= importer.MAX_SOURCE_BYTES passages = chunking.chunk(body) assert len(passages) <= importer.MAX_CHUNKS_PER_SOURCE assert all(p.token_count <= chunking.TARGET_MAX for p in passages) assert len({p.index for p in passages}) == len(passages) def test_a_fenced_code_block_is_not_read_as_headings(): source = ( "# Real Heading\n\nProse about the setting.\n\n" "```python\n# not a heading\n## also not a heading\n```\n\n" "More prose about the setting.\n" ) passages = chunking.chunk(source) assert all(p.heading_path in ("", "Real Heading") for p in passages) joined = "\n".join(p.text for p in passages) assert "# not a heading" in joined def test_plain_text_takes_the_same_packing_with_no_headings(): source = "\n\n".join(f"Paragraph {n} of the notes. " * 12 for n in range(20)) passages = chunking.chunk(source, markdown=False) assert len(passages) > 1 assert all(p.heading_path == "" for p in passages) assert all(p.token_count <= chunking.TARGET_MAX for p in passages) # A `#` in plain text is a character, not a heading. hashy = chunking.chunk("# not a heading\n\nsome text\n", markdown=False) assert "# not a heading" in hashy[0].text @pytest.mark.parametrize("source", ["", " \n\n \n", "\n"]) def test_an_empty_source_produces_no_passages(source): assert chunking.chunk(source) == []