v1.1 WP-B.1: diagnose independent long-term memory retention
Diagnostic only; no memory behaviour changes. - tools/memory_diagnostic.py: planted-fact isolation checks, the four-stage diagnosis (created / retained / ranked / injected) with a verdict, a production-ranking replica, deterministic summariser/embedder/narrator stubs and seven scenarios (default, past capacity, pinned, low top_k, long-block early/late, lineage control) - tools/v11_b1_memory.py: CLI for the scenarios and for diagnosing a copy of a finished real campaign - tools/m11_long_run.py: opt-in --independent-fact mode with per-turn isolation tracking and the recovered_through_memory_independent verdict; M04 verdicts unchanged - tests: diagnostic stages, eviction, creation window, ranking, lineage and authority controls; two strict xfails record the diagnosed retention and creation defects for WP-B.2 to flip - planning/reports/v1.1/V1.1-WP-B1-REPORT.md First failing stage: ranking (real model); retention past capacity and creation for early facts in long blocks (deterministic, same on v1.0.0). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
This commit is contained in:
co-authored by
Claude Opus 5
parent
d63804f22e
commit
beb17ada10
@@ -0,0 +1,122 @@
|
||||
"""v1.1 WP-B.1: run the deterministic memory-retention scenarios, or diagnose a real campaign.
|
||||
|
||||
# the deterministic scenarios, against an isolated database in --out
|
||||
.venv/bin/python -m tools.v11_b1_memory scenarios --out "$HOME/v11-evidence/b1/<label>"
|
||||
|
||||
# the four stages for a finished real campaign (reads its database; embeds
|
||||
# the recall query with the campaign's own configured embedding model)
|
||||
AIDND_TEST_ENDPOINT=... AIDND_TEST_EMBED_MODEL=nomic-embed-text:latest \\
|
||||
.venv/bin/python -m tools.v11_b1_memory diagnose --db <campaign.db> \\
|
||||
--plant-depth 3 --out "$HOME/v11-evidence/b1/<label>"
|
||||
|
||||
Run from `backend/`. Nothing here changes memory behaviour; see
|
||||
`tools/memory_diagnostic.py` for what is measured and what the stubs model.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
sub = parser.add_subparsers(dest="command", required=True)
|
||||
scen = sub.add_parser("scenarios")
|
||||
scen.add_argument("--out", required=True)
|
||||
scen.add_argument("--only", action="append", default=[])
|
||||
diag = sub.add_parser("diagnose")
|
||||
diag.add_argument("--db", required=True)
|
||||
diag.add_argument("--plant-depth", type=int, required=True)
|
||||
diag.add_argument("--out", required=True)
|
||||
args = parser.parse_args()
|
||||
|
||||
out = Path(args.out)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
if args.command == "scenarios":
|
||||
db_path = out / "scenarios.db"
|
||||
if db_path.exists():
|
||||
db_path.unlink()
|
||||
os.environ["AIDND_DB_PATH"] = str(db_path)
|
||||
else:
|
||||
# A copy, so diagnosis never writes to the evidence database.
|
||||
copy = out / "diagnosed-copy.db"
|
||||
shutil.copy2(args.db, copy)
|
||||
os.environ["AIDND_DB_PATH"] = str(copy)
|
||||
os.environ.pop("AIDND_DATABASE_URL", None)
|
||||
os.environ.pop("DATABASE_URL", None)
|
||||
|
||||
from tools import memory_diagnostic as md # after the database is chosen
|
||||
|
||||
if args.command == "scenarios":
|
||||
names = args.only or list(md.SCENARIOS)
|
||||
summary = {}
|
||||
for name in names:
|
||||
result = md.run_scenario(md.SCENARIOS[name])
|
||||
(out / f"{name}.json").write_text(json.dumps(result, indent=2, default=str))
|
||||
d = result.get("diagnosis") or {}
|
||||
summary[name] = {
|
||||
"verdict": d.get("verdict"),
|
||||
"isolation_ok": (result.get("isolation") or {}).get("ok"),
|
||||
"plant_depth": result.get("plant_depth"),
|
||||
"recall_depth": result.get("recall_depth"),
|
||||
"f_evicted_at_turn": (result.get("eviction") or {}).get("f_evicted_at_turn"),
|
||||
}
|
||||
print(f"{name:26} verdict={d.get('verdict')!s:26} "
|
||||
f"isolation_ok={summary[name]['isolation_ok']} "
|
||||
f"plant={result.get('plant_depth')} recall={result.get('recall_depth')}")
|
||||
(out / "summary.json").write_text(json.dumps(summary, indent=2))
|
||||
return 0
|
||||
|
||||
import asyncio
|
||||
|
||||
from sqlalchemy.orm import undefer
|
||||
|
||||
from app import memorybank, models
|
||||
from app.database import SessionLocal
|
||||
|
||||
endpoint = os.environ.get("AIDND_TEST_ENDPOINT", "")
|
||||
embed_model = os.environ.get("AIDND_TEST_EMBED_MODEL", "")
|
||||
with SessionLocal() as db:
|
||||
adventure = db.query(models.Adventure).order_by(models.Adventure.id).first()
|
||||
settings = db.query(models.Settings).filter_by(user_id=adventure.user_id).first()
|
||||
if endpoint:
|
||||
settings.endpoint_url = endpoint
|
||||
if embed_model:
|
||||
settings.embedding_model = embed_model
|
||||
recall_action = (db.query(models.Action)
|
||||
.filter(models.Action.adventure_id == adventure.id,
|
||||
models.Action.type == "ai")
|
||||
.options(undefer(models.Action.context_snapshot))
|
||||
.order_by(models.Action.id.desc()).first())
|
||||
embed = memorybank.embedding_provider(settings).embed
|
||||
iso = md.isolation(db, adventure, md.FACT_F, args.plant_depth,
|
||||
recall_snapshot=recall_action.context_snapshot,
|
||||
recall_depth=recall_action.depth)
|
||||
diagnosis = asyncio.run(md.diagnose(db, adventure, settings, md.FACT_F, args.plant_depth,
|
||||
recall_action=recall_action, embed=embed))
|
||||
variants = {}
|
||||
memory_id = diagnosis["created"]["memory_id"]
|
||||
if memory_id is not None and not diagnosis.get("retained", {}).get("forgotten"):
|
||||
for label, query in (("recall_turn", diagnosis.get("ranked", {}).get("query", "")),
|
||||
("paraphrase", md.PARAPHRASE_QUERY),
|
||||
("unrelated", md.UNRELATED_QUERY)):
|
||||
ranking = asyncio.run(md.rank_bank(db, adventure, settings, query, embed))
|
||||
row = next((r for r in ranking["scored"] if r["memory_id"] == memory_id), None)
|
||||
variants[label] = {"rank": row and row["rank"], "of": len(ranking["scored"]),
|
||||
"similarity": row and row["similarity"],
|
||||
"selected": bool(row and row["selected"])}
|
||||
db.rollback()
|
||||
report = {"isolation": iso, "diagnosis": diagnosis, "ranking_variants": variants}
|
||||
(out / "diagnosis.json").write_text(json.dumps(report, indent=2, default=str))
|
||||
print(json.dumps({"isolation_ok": iso["ok"], "verdict": diagnosis["verdict"]}, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user