"""v1.1 WP-A2: replay real stored narration through the v1.0.0 and current extractors.
The A2 extractor changes remove more text from a narrator's reply than v1.0.0
did. Removing story is worse than leaving protocol (`TECHNICAL-DESIGN.md`
§15.4), so every change is shown to a person rather than summarised. This tool
takes every real reply the evidence kept, runs it through both extractors, and
writes each turn whose prose differs with:
- the v1.0.0 prose and the current prose;
- every line removed, and the rule that explains it;
- any removal no rule explains, which fails the replay.
**Input.** A reply is read from the turn's stored `raw_output` where the
evidence database kept one: that is exactly what the narrator sent. A bundle
carries no raw reply, so a bundle's turns are replayed from their stored text,
which is v1.0.0's output already. For those the old prose is the input itself,
and the comparison is still exact.
**The v1.0.0 extractor** is read from the release tag with `git show`, not
copied, so this tool compares against what shipped. It shares `events` and
`render` with the current tree. A2 does not change `events.SPECS` or
`render.SECTION_HEADINGS`, and the report verifies that with `git diff`.
Evidence stays outside the repository. Replayed text is fiction from the
acceptance fixtures, but it is still somebody's run.
.venv/bin/python -m tools.v11_replay_extractor \\
--db "$HOME/m11-evidence/**/*.db" \\
--bundle "$HOME/m11-evidence/closeout-3652dc6/identity/turn-99/bundle.json" \\
--out "$HOME/v11-evidence/a2-replay"
"""
from __future__ import annotations
import argparse
import difflib
import glob
import hashlib
import importlib.util
import json
import sqlite3
import subprocess
import sys
import types
import zlib
from pathlib import Path
RELEASE = "v1.0.0"
EXTRACT_PATH = "backend/app/narrative/extract.py"
#: A removal larger than this share of the v1.0.0 prose is flagged for review
#: even when every line is explained, because a rule that eats most of a reply
#: is the shape a false positive takes.
LARGE_REMOVAL_SHARE = 0.25
def load_release_extractor(repo: Path) -> types.ModuleType:
"""`app.narrative.extract` as it was at the release tag."""
source = subprocess.run(
["git", "-C", str(repo), "show", f"{RELEASE}:{EXTRACT_PATH}"],
check=True, capture_output=True, text=True,
).stdout
import app.narrative # noqa: F401 the package the relative import needs
spec = importlib.util.spec_from_loader("app.narrative._extract_release", loader=None)
module = importlib.util.module_from_spec(spec)
module.__package__ = "app.narrative"
exec(compile(source, f"{RELEASE}:{EXTRACT_PATH}", "exec"), module.__dict__)
return module
def _unpack(blob):
if blob is None:
return None
try:
return json.loads(zlib.decompress(bytes(blob)).decode("utf-8"))
except (zlib.error, ValueError, UnicodeDecodeError):
return None
def turns_from_db(path: str):
connection = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
try:
rows = connection.execute(
"SELECT id, depth, text, context_snapshot FROM actions WHERE type = 'ai'"
).fetchall()
finally:
connection.close()
for action_id, depth, text, blob in rows:
snapshot = _unpack(blob) or {}
raw = snapshot.get("raw_output")
if isinstance(raw, str) and raw.strip():
yield {"source": path, "id": action_id, "depth": depth,
"input": raw, "input_kind": "raw_output"}
elif text:
yield {"source": path, "id": action_id, "depth": depth,
"input": text, "input_kind": "stored_text"}
def turns_from_bundle(path: str):
bundle = json.loads(Path(path).read_text())
for action in bundle.get("actions") or []:
if action.get("type") == "ai" and action.get("text"):
yield {"source": path, "id": action.get("id"), "depth": action.get("depth"),
"input": action["text"], "input_kind": "stored_text"}
def removed_lines(before: str, after: str) -> list[str]:
"""Lines present in `before` and gone from `after`, in order.
Compared with trailing whitespace ignored. The extractor strips the end of
every reply it cuts, so a story line that becomes the last line loses a
trailing space. A first version of this tool reported that space as a
rewritten line of story, which it is not.
"""
old = [line.rstrip() for line in before.split("\n")]
new = [line.rstrip() for line in after.split("\n")]
matcher = difflib.SequenceMatcher(a=old, b=new, autojunk=False)
gone: list[str] = []
for tag, a0, a1, _b0, _b1 in matcher.get_opcodes():
if tag in ("delete", "replace"):
gone.extend(old[a0:a1])
return gone
def replay(inputs, old, new) -> dict:
seen: set[str] = set()
unchanged = 0
changed: list[dict] = []
duplicates = 0
for turn in inputs:
digest = hashlib.sha256(turn["input"].encode()).hexdigest()
if digest in seen:
duplicates += 1
continue
seen.add(digest)
old_prose, _old_parsed, _old_raw = old.split(turn["input"])
new_prose, _new_parsed, _new_raw = new.split(turn["input"])
if old_prose == new_prose:
unchanged += 1
continue
lines = []
unexplained = 0
for line in removed_lines(old_prose, new_prose):
if not line.strip():
continue
rule = new.explain_removed_line(line)
if rule is None:
unexplained += 1
lines.append({"line": line, "rule": rule})
added = [line for line in removed_lines(new_prose, old_prose) if line.strip()]
share = 1 - len(new_prose) / max(1, len(old_prose))
flags = []
if unexplained:
flags.append("unexplained_removal")
if added:
flags.append("text_added_or_rewritten")
if share > LARGE_REMOVAL_SHARE:
flags.append("large_removal")
changed.append({
**{k: turn[k] for k in ("source", "id", "depth", "input_kind")},
"sha256": digest,
"old_prose": old_prose,
"new_prose": new_prose,
"removed": lines,
"added_or_rewritten": added,
"removed_chars": len(old_prose) - len(new_prose),
"removed_share": round(share, 4),
"flags": flags,
})
return {
"replayed": unchanged + len(changed),
"duplicates_skipped": duplicates,
"unchanged": unchanged,
"changed": len(changed),
"flagged": sum(1 for c in changed if c["flags"]),
"turns": changed,
}
def write_markdown(result: dict, path: Path) -> None:
out = [
"# A2 extractor replay",
"",
f"- replayed (unique replies): **{result['replayed']}**",
f"- duplicates skipped: {result['duplicates_skipped']}",
f"- unchanged: {result['unchanged']}",
f"- changed: **{result['changed']}**",
f"- flagged: **{result['flagged']}**",
"",
]
for index, turn in enumerate(result["turns"], 1):
out += [
f"## {index}. {Path(turn['source']).parent.name}/{Path(turn['source']).name}"
f" action {turn['id']} depth {turn['depth']} ({turn['input_kind']})",
"",
f"- removed chars: {turn['removed_chars']} ({turn['removed_share']:.1%})",
f"- flags: {', '.join(turn['flags']) or 'none'}",
"",
"Removed lines:",
"",
]
for item in turn["removed"]:
out.append(f"- `{item['rule'] or 'UNEXPLAINED'}` — {item['line']!r}")
out += ["", "v1.0.0 prose
", "", "```text",
turn["old_prose"], "```", " ", "",
"current prose
", "", "```text",
turn["new_prose"], "```", " ", ""]
path.write_text("\n".join(out))
def main(argv=None) -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--db", action="append", default=[],
help="an evidence database, or a glob of them")
parser.add_argument("--bundle", action="append", default=[],
help="an exported bundle whose campaign has no database here")
parser.add_argument("--out", required=True)
args = parser.parse_args(argv)
repo = Path(__file__).resolve().parents[2]
from app.narrative import extract as current
old = load_release_extractor(repo)
dbs = sorted({p for pattern in args.db for p in glob.glob(pattern, recursive=True)})
def inputs():
for path in dbs:
yield from turns_from_db(path)
for path in args.bundle:
yield from turns_from_bundle(path)
result = replay(inputs(), old, current)
result["databases"] = dbs
result["bundles"] = args.bundle
out = Path(args.out)
out.mkdir(parents=True, exist_ok=True)
(out / "replay.json").write_text(json.dumps(result, indent=2, ensure_ascii=False))
write_markdown(result, out / "replay.md")
print(json.dumps({k: result[k] for k in
("replayed", "duplicates_skipped", "unchanged", "changed", "flagged")}))
return 1 if result["flagged"] else 0
if __name__ == "__main__":
sys.exit(main())