Files
interactive-story/backend/tools/v11_b1_memory.py
T
JesseMarkowitzandClaude Opus 5 0c1ba836ba v1.1 WP-B.2: independent long-term memory retention
Corrects the memory mechanisms WP-B.1 diagnosed, one at a time, each
verified before the next. Accepted by the owner with a documented
reference-model limitation. No schema, bundle format, setting default,
lineage, authority or protocol-cleanup change.

- B2.1 ranking: the retrieval query is the player's input plus a bounded
  scene context (state scene + end of the newest narration), embedded in
  one call. final = semantic (0.6 input / 0.4 context) + 0.15 x lexical,
  where lexical is a rarity-weighted share of the input's words, computed
  per turn over the candidates with no index. Scores and the query are
  recorded per used memory; pins and redundancy suppression unchanged.
- B2.2 coverage-aware eviction (memorybank.eviction_order): the earliest
  and newest memories are kept, the smallest coverage hole goes first,
  least-recently-used breaks ties and remains the fallback. Bounded; pins
  never evicted; frozen-bank protection kept; reads no text or vectors.
- B2.3 bounded memory creation: a block longer than 2,000 tokens is shown
  to the summariser as head + tail with an omission marker, inside the
  same budget; shorter blocks unchanged; the marker is never stored.
- The memory summariser prompt is unchanged from v1.0.0. A B2.4 prompt
  experiment was measured on the reference model, showed no reliable
  improvement for the target failure (0/5 under both prompts, with new
  "Memory:"-prefix, second-person and length regressions), and was
  reverted. memorybank.memory_user_prompt is kept as a behaviour-neutral
  helper.
- tools/memory_fidelity.py (diagnostic only): genre-neutral fixtures plus
  the failed block, a deterministic fidelity checker, and a real-model
  shipped-vs-experiment measurement.
- tools/memory_diagnostic.py: ranking replica uses production scoring;
  ranking_crowded, ranking_context_dependent and independent_full
  fixtures; per-turn isolation and provenance.
- tests: B.1's two strict xfails are now ordinary passes; ranking,
  eviction and excerpt tests; summariser acceptance tests kept apart from
  diagnostic-measurement tests.
- DEVELOPMENT.md: the GPU-host kernel/Ollama watch used `-k -u ollama`,
  which matches nothing; now the OR form.
- docs: CONTEXT-AND-MEMORY 15/18/20/21 as shipped, V1.1-PLAN (status and
  release criteria 12-13), planning README, VERSION v4.3,
  reports/v1.1/V1.1-WP-B2-REPORT.md.

Deterministic independent-memory recovery: PASS (independent_full fails
on v1.0.0 at creation and returns recovered_through_memory_independent
here). Reference-model independent recovery: FAILED on the
precondition-valid attempt, at memory creation: the summariser omitted a
player-established fact from a block it received whole. Accepted as a
documented v1.1 residual and carried into the release gate.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
2026-09-15 11:21:53 -04:00

126 lines
5.7 KiB
Python

"""v1.1 WP-B.1: run the deterministic memory-retention scenarios, or diagnose a real campaign.
# the deterministic scenarios, against an isolated database in --out
.venv/bin/python -m tools.v11_b1_memory scenarios --out "$HOME/v11-evidence/b1/<label>"
# the four stages for a finished real campaign (reads its database; embeds
# the recall query with the campaign's own configured embedding model)
AIDND_TEST_ENDPOINT=... AIDND_TEST_EMBED_MODEL=nomic-embed-text:latest \\
.venv/bin/python -m tools.v11_b1_memory diagnose --db <campaign.db> \\
--plant-depth 3 --out "$HOME/v11-evidence/b1/<label>"
Run from `backend/`. Nothing here changes memory behaviour; see
`tools/memory_diagnostic.py` for what is measured and what the stubs model.
"""
from __future__ import annotations
import argparse
import json
import os
import shutil
import sys
from pathlib import Path
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
sub = parser.add_subparsers(dest="command", required=True)
scen = sub.add_parser("scenarios")
scen.add_argument("--out", required=True)
scen.add_argument("--only", action="append", default=[])
diag = sub.add_parser("diagnose")
diag.add_argument("--db", required=True)
diag.add_argument("--plant-depth", type=int, required=True)
diag.add_argument("--out", required=True)
args = parser.parse_args()
out = Path(args.out)
out.mkdir(parents=True, exist_ok=True)
if args.command == "scenarios":
db_path = out / "scenarios.db"
if db_path.exists():
db_path.unlink()
os.environ["AIDND_DB_PATH"] = str(db_path)
else:
# A copy, so diagnosis never writes to the evidence database.
copy = out / "diagnosed-copy.db"
shutil.copy2(args.db, copy)
os.environ["AIDND_DB_PATH"] = str(copy)
os.environ.pop("AIDND_DATABASE_URL", None)
os.environ.pop("DATABASE_URL", None)
from tools import memory_diagnostic as md # after the database is chosen
if args.command == "scenarios":
names = args.only or list(md.SCENARIOS)
summary = {}
for name in names:
result = md.run_scenario(md.SCENARIOS[name])
(out / f"{name}.json").write_text(json.dumps(result, indent=2, default=str))
d = result.get("diagnosis") or {}
summary[name] = {
"verdict": d.get("verdict"),
"isolation_ok": (result.get("isolation") or {}).get("ok"),
"plant_depth": result.get("plant_depth"),
"recall_depth": result.get("recall_depth"),
"f_evicted_at_turn": (result.get("eviction") or {}).get("f_evicted_at_turn"),
}
print(f"{name:26} verdict={d.get('verdict')!s:26} "
f"isolation_ok={summary[name]['isolation_ok']} "
f"plant={result.get('plant_depth')} recall={result.get('recall_depth')}")
(out / "summary.json").write_text(json.dumps(summary, indent=2))
return 0
import asyncio
from sqlalchemy.orm import undefer
from app import memorybank, models
from app.database import SessionLocal
endpoint = os.environ.get("AIDND_TEST_ENDPOINT", "")
embed_model = os.environ.get("AIDND_TEST_EMBED_MODEL", "")
with SessionLocal() as db:
adventure = db.query(models.Adventure).order_by(models.Adventure.id).first()
settings = db.query(models.Settings).filter_by(user_id=adventure.user_id).first()
if endpoint:
settings.endpoint_url = endpoint
if embed_model:
settings.embedding_model = embed_model
recall_action = (db.query(models.Action)
.filter(models.Action.adventure_id == adventure.id,
models.Action.type == "ai")
.options(undefer(models.Action.context_snapshot))
.order_by(models.Action.id.desc()).first())
embed = memorybank.embedding_provider(settings).embed
iso = md.isolation(db, adventure, md.FACT_F, args.plant_depth,
recall_snapshot=recall_action.context_snapshot,
recall_depth=recall_action.depth)
diagnosis = asyncio.run(md.diagnose(db, adventure, settings, md.FACT_F, args.plant_depth,
recall_action=recall_action, embed=embed))
variants = {}
memory_id = diagnosis["created"]["memory_id"]
if memory_id is not None and not diagnosis.get("retained", {}).get("forgotten"):
base = md.production_query(adventure, recall_action.id)
for label, query in (("recall_turn", base),
("paraphrase", md.variant_query(base, md.PARAPHRASE_QUERY)),
("unrelated", md.variant_query(base, md.UNRELATED_QUERY))):
ranking = asyncio.run(md.rank_bank(db, adventure, settings, query, embed))
row = next((r for r in ranking["scored"] if r["memory_id"] == memory_id), None)
variants[label] = {"rank": row and row["rank"], "of": len(ranking["scored"]),
"similarity": row and row["similarity"],
"lexical_score": row and row["lexical_score"],
"final_score": row and row["final_score"],
"selected": bool(row and row["selected"])}
db.rollback()
report = {"isolation": iso, "diagnosis": diagnosis, "ranking_variants": variants}
(out / "diagnosis.json").write_text(json.dumps(report, indent=2, default=str))
print(json.dumps({"isolation_ok": iso["ok"], "verdict": diagnosis["verdict"]}, indent=2))
return 0
if __name__ == "__main__":
sys.exit(main())