v1.1 WP-B.1: diagnose independent long-term memory retention

Diagnostic only; no memory behaviour changes.

- tools/memory_diagnostic.py: planted-fact isolation checks, the four-stage
  diagnosis (created / retained / ranked / injected) with a verdict, a
  production-ranking replica, deterministic summariser/embedder/narrator
  stubs and seven scenarios (default, past capacity, pinned, low top_k,
  long-block early/late, lineage control)
- tools/v11_b1_memory.py: CLI for the scenarios and for diagnosing a copy of
  a finished real campaign
- tools/m11_long_run.py: opt-in --independent-fact mode with per-turn
  isolation tracking and the recovered_through_memory_independent verdict;
  M04 verdicts unchanged
- tests: diagnostic stages, eviction, creation window, ranking, lineage and
  authority controls; two strict xfails record the diagnosed retention and
  creation defects for WP-B.2 to flip
- planning/reports/v1.1/V1.1-WP-B1-REPORT.md

First failing stage: ranking (real model); retention past capacity and
creation for early facts in long blocks (deterministic, same on v1.0.0).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
This commit is contained in:
JesseMarkowitz
2026-09-14 20:50:05 -04:00
co-authored by Claude Opus 5
parent d63804f22e
commit beb17ada10
6 changed files with 2184 additions and 0 deletions
+122
View File
@@ -0,0 +1,122 @@
"""v1.1 WP-B.1: run the deterministic memory-retention scenarios, or diagnose a real campaign.
# the deterministic scenarios, against an isolated database in --out
.venv/bin/python -m tools.v11_b1_memory scenarios --out "$HOME/v11-evidence/b1/<label>"
# the four stages for a finished real campaign (reads its database; embeds
# the recall query with the campaign's own configured embedding model)
AIDND_TEST_ENDPOINT=... AIDND_TEST_EMBED_MODEL=nomic-embed-text:latest \\
.venv/bin/python -m tools.v11_b1_memory diagnose --db <campaign.db> \\
--plant-depth 3 --out "$HOME/v11-evidence/b1/<label>"
Run from `backend/`. Nothing here changes memory behaviour; see
`tools/memory_diagnostic.py` for what is measured and what the stubs model.
"""
from __future__ import annotations
import argparse
import json
import os
import shutil
import sys
from pathlib import Path
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
sub = parser.add_subparsers(dest="command", required=True)
scen = sub.add_parser("scenarios")
scen.add_argument("--out", required=True)
scen.add_argument("--only", action="append", default=[])
diag = sub.add_parser("diagnose")
diag.add_argument("--db", required=True)
diag.add_argument("--plant-depth", type=int, required=True)
diag.add_argument("--out", required=True)
args = parser.parse_args()
out = Path(args.out)
out.mkdir(parents=True, exist_ok=True)
if args.command == "scenarios":
db_path = out / "scenarios.db"
if db_path.exists():
db_path.unlink()
os.environ["AIDND_DB_PATH"] = str(db_path)
else:
# A copy, so diagnosis never writes to the evidence database.
copy = out / "diagnosed-copy.db"
shutil.copy2(args.db, copy)
os.environ["AIDND_DB_PATH"] = str(copy)
os.environ.pop("AIDND_DATABASE_URL", None)
os.environ.pop("DATABASE_URL", None)
from tools import memory_diagnostic as md # after the database is chosen
if args.command == "scenarios":
names = args.only or list(md.SCENARIOS)
summary = {}
for name in names:
result = md.run_scenario(md.SCENARIOS[name])
(out / f"{name}.json").write_text(json.dumps(result, indent=2, default=str))
d = result.get("diagnosis") or {}
summary[name] = {
"verdict": d.get("verdict"),
"isolation_ok": (result.get("isolation") or {}).get("ok"),
"plant_depth": result.get("plant_depth"),
"recall_depth": result.get("recall_depth"),
"f_evicted_at_turn": (result.get("eviction") or {}).get("f_evicted_at_turn"),
}
print(f"{name:26} verdict={d.get('verdict')!s:26} "
f"isolation_ok={summary[name]['isolation_ok']} "
f"plant={result.get('plant_depth')} recall={result.get('recall_depth')}")
(out / "summary.json").write_text(json.dumps(summary, indent=2))
return 0
import asyncio
from sqlalchemy.orm import undefer
from app import memorybank, models
from app.database import SessionLocal
endpoint = os.environ.get("AIDND_TEST_ENDPOINT", "")
embed_model = os.environ.get("AIDND_TEST_EMBED_MODEL", "")
with SessionLocal() as db:
adventure = db.query(models.Adventure).order_by(models.Adventure.id).first()
settings = db.query(models.Settings).filter_by(user_id=adventure.user_id).first()
if endpoint:
settings.endpoint_url = endpoint
if embed_model:
settings.embedding_model = embed_model
recall_action = (db.query(models.Action)
.filter(models.Action.adventure_id == adventure.id,
models.Action.type == "ai")
.options(undefer(models.Action.context_snapshot))
.order_by(models.Action.id.desc()).first())
embed = memorybank.embedding_provider(settings).embed
iso = md.isolation(db, adventure, md.FACT_F, args.plant_depth,
recall_snapshot=recall_action.context_snapshot,
recall_depth=recall_action.depth)
diagnosis = asyncio.run(md.diagnose(db, adventure, settings, md.FACT_F, args.plant_depth,
recall_action=recall_action, embed=embed))
variants = {}
memory_id = diagnosis["created"]["memory_id"]
if memory_id is not None and not diagnosis.get("retained", {}).get("forgotten"):
for label, query in (("recall_turn", diagnosis.get("ranked", {}).get("query", "")),
("paraphrase", md.PARAPHRASE_QUERY),
("unrelated", md.UNRELATED_QUERY)):
ranking = asyncio.run(md.rank_bank(db, adventure, settings, query, embed))
row = next((r for r in ranking["scored"] if r["memory_id"] == memory_id), None)
variants[label] = {"rank": row and row["rank"], "of": len(ranking["scored"]),
"similarity": row and row["similarity"],
"selected": bool(row and row["selected"])}
db.rollback()
report = {"isolation": iso, "diagnosis": diagnosis, "ranking_variants": variants}
(out / "diagnosis.json").write_text(json.dumps(report, indent=2, default=str))
print(json.dumps({"isolation_ok": iso["ok"], "verdict": diagnosis["verdict"]}, indent=2))
return 0
if __name__ == "__main__":
sys.exit(main())