"""v1.1 WP-A2: replay real stored narration through the v1.0.0 and current extractors. The A2 extractor changes remove more text from a narrator's reply than v1.0.0 did. Removing story is worse than leaving protocol (`TECHNICAL-DESIGN.md` §15.4), so every change is shown to a person rather than summarised. This tool takes every real reply the evidence kept, runs it through both extractors, and writes each turn whose prose differs with: - the v1.0.0 prose and the current prose; - every line removed, and the rule that explains it; - any removal no rule explains, which fails the replay. **Input.** A reply is read from the turn's stored `raw_output` where the evidence database kept one: that is exactly what the narrator sent. A bundle carries no raw reply, so a bundle's turns are replayed from their stored text, which is v1.0.0's output already. For those the old prose is the input itself, and the comparison is still exact. **The v1.0.0 extractor** is read from the release tag with `git show`, not copied, so this tool compares against what shipped. It shares `events` and `render` with the current tree. A2 does not change `events.SPECS` or `render.SECTION_HEADINGS`, and the report verifies that with `git diff`. Evidence stays outside the repository. Replayed text is fiction from the acceptance fixtures, but it is still somebody's run. .venv/bin/python -m tools.v11_replay_extractor \\ --db "$HOME/m11-evidence/**/*.db" \\ --bundle "$HOME/m11-evidence/closeout-3652dc6/identity/turn-99/bundle.json" \\ --out "$HOME/v11-evidence/a2-replay" """ from __future__ import annotations import argparse import difflib import glob import hashlib import importlib.util import json import sqlite3 import subprocess import sys import types import zlib from pathlib import Path RELEASE = "v1.0.0" EXTRACT_PATH = "backend/app/narrative/extract.py" #: A removal larger than this share of the v1.0.0 prose is flagged for review #: even when every line is explained, because a rule that eats most of a reply #: is the shape a false positive takes. LARGE_REMOVAL_SHARE = 0.25 def load_release_extractor(repo: Path) -> types.ModuleType: """`app.narrative.extract` as it was at the release tag.""" source = subprocess.run( ["git", "-C", str(repo), "show", f"{RELEASE}:{EXTRACT_PATH}"], check=True, capture_output=True, text=True, ).stdout import app.narrative # noqa: F401 the package the relative import needs spec = importlib.util.spec_from_loader("app.narrative._extract_release", loader=None) module = importlib.util.module_from_spec(spec) module.__package__ = "app.narrative" exec(compile(source, f"{RELEASE}:{EXTRACT_PATH}", "exec"), module.__dict__) return module def _unpack(blob): if blob is None: return None try: return json.loads(zlib.decompress(bytes(blob)).decode("utf-8")) except (zlib.error, ValueError, UnicodeDecodeError): return None def turns_from_db(path: str): connection = sqlite3.connect(f"file:{path}?mode=ro", uri=True) try: rows = connection.execute( "SELECT id, depth, text, context_snapshot FROM actions WHERE type = 'ai'" ).fetchall() finally: connection.close() for action_id, depth, text, blob in rows: snapshot = _unpack(blob) or {} raw = snapshot.get("raw_output") if isinstance(raw, str) and raw.strip(): yield {"source": path, "id": action_id, "depth": depth, "input": raw, "input_kind": "raw_output"} elif text: yield {"source": path, "id": action_id, "depth": depth, "input": text, "input_kind": "stored_text"} def turns_from_bundle(path: str): bundle = json.loads(Path(path).read_text()) for action in bundle.get("actions") or []: if action.get("type") == "ai" and action.get("text"): yield {"source": path, "id": action.get("id"), "depth": action.get("depth"), "input": action["text"], "input_kind": "stored_text"} def removed_lines(before: str, after: str) -> list[str]: """Lines present in `before` and gone from `after`, in order. Compared with trailing whitespace ignored. The extractor strips the end of every reply it cuts, so a story line that becomes the last line loses a trailing space. A first version of this tool reported that space as a rewritten line of story, which it is not. """ old = [line.rstrip() for line in before.split("\n")] new = [line.rstrip() for line in after.split("\n")] matcher = difflib.SequenceMatcher(a=old, b=new, autojunk=False) gone: list[str] = [] for tag, a0, a1, _b0, _b1 in matcher.get_opcodes(): if tag in ("delete", "replace"): gone.extend(old[a0:a1]) return gone def replay(inputs, old, new) -> dict: seen: set[str] = set() unchanged = 0 changed: list[dict] = [] duplicates = 0 for turn in inputs: digest = hashlib.sha256(turn["input"].encode()).hexdigest() if digest in seen: duplicates += 1 continue seen.add(digest) old_prose, _old_parsed, _old_raw = old.split(turn["input"]) new_prose, _new_parsed, _new_raw = new.split(turn["input"]) if old_prose == new_prose: unchanged += 1 continue lines = [] unexplained = 0 for line in removed_lines(old_prose, new_prose): if not line.strip(): continue rule = new.explain_removed_line(line) if rule is None: unexplained += 1 lines.append({"line": line, "rule": rule}) added = [line for line in removed_lines(new_prose, old_prose) if line.strip()] share = 1 - len(new_prose) / max(1, len(old_prose)) flags = [] if unexplained: flags.append("unexplained_removal") if added: flags.append("text_added_or_rewritten") if share > LARGE_REMOVAL_SHARE: flags.append("large_removal") changed.append({ **{k: turn[k] for k in ("source", "id", "depth", "input_kind")}, "sha256": digest, "old_prose": old_prose, "new_prose": new_prose, "removed": lines, "added_or_rewritten": added, "removed_chars": len(old_prose) - len(new_prose), "removed_share": round(share, 4), "flags": flags, }) return { "replayed": unchanged + len(changed), "duplicates_skipped": duplicates, "unchanged": unchanged, "changed": len(changed), "flagged": sum(1 for c in changed if c["flags"]), "turns": changed, } def write_markdown(result: dict, path: Path) -> None: out = [ "# A2 extractor replay", "", f"- replayed (unique replies): **{result['replayed']}**", f"- duplicates skipped: {result['duplicates_skipped']}", f"- unchanged: {result['unchanged']}", f"- changed: **{result['changed']}**", f"- flagged: **{result['flagged']}**", "", ] for index, turn in enumerate(result["turns"], 1): out += [ f"## {index}. {Path(turn['source']).parent.name}/{Path(turn['source']).name}" f" action {turn['id']} depth {turn['depth']} ({turn['input_kind']})", "", f"- removed chars: {turn['removed_chars']} ({turn['removed_share']:.1%})", f"- flags: {', '.join(turn['flags']) or 'none'}", "", "Removed lines:", "", ] for item in turn["removed"]: out.append(f"- `{item['rule'] or 'UNEXPLAINED'}` — {item['line']!r}") out += ["", "
v1.0.0 prose", "", "```text", turn["old_prose"], "```", "
", "", "
current prose", "", "```text", turn["new_prose"], "```", "
", ""] path.write_text("\n".join(out)) def main(argv=None) -> int: parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) parser.add_argument("--db", action="append", default=[], help="an evidence database, or a glob of them") parser.add_argument("--bundle", action="append", default=[], help="an exported bundle whose campaign has no database here") parser.add_argument("--out", required=True) args = parser.parse_args(argv) repo = Path(__file__).resolve().parents[2] from app.narrative import extract as current old = load_release_extractor(repo) dbs = sorted({p for pattern in args.db for p in glob.glob(pattern, recursive=True)}) def inputs(): for path in dbs: yield from turns_from_db(path) for path in args.bundle: yield from turns_from_bundle(path) result = replay(inputs(), old, current) result["databases"] = dbs result["bundles"] = args.bundle out = Path(args.out) out.mkdir(parents=True, exist_ok=True) (out / "replay.json").write_text(json.dumps(result, indent=2, ensure_ascii=False)) write_markdown(result, out / "replay.md") print(json.dumps({k: result[k] for k in ("replayed", "duplicates_skipped", "unchanged", "changed", "flagged")})) return 1 if result["flagged"] else 0 if __name__ == "__main__": sys.exit(main())