Corrects the memory mechanisms WP-B.1 diagnosed, one at a time, each verified before the next. Accepted by the owner with a documented reference-model limitation. No schema, bundle format, setting default, lineage, authority or protocol-cleanup change. - B2.1 ranking: the retrieval query is the player's input plus a bounded scene context (state scene + end of the newest narration), embedded in one call. final = semantic (0.6 input / 0.4 context) + 0.15 x lexical, where lexical is a rarity-weighted share of the input's words, computed per turn over the candidates with no index. Scores and the query are recorded per used memory; pins and redundancy suppression unchanged. - B2.2 coverage-aware eviction (memorybank.eviction_order): the earliest and newest memories are kept, the smallest coverage hole goes first, least-recently-used breaks ties and remains the fallback. Bounded; pins never evicted; frozen-bank protection kept; reads no text or vectors. - B2.3 bounded memory creation: a block longer than 2,000 tokens is shown to the summariser as head + tail with an omission marker, inside the same budget; shorter blocks unchanged; the marker is never stored. - The memory summariser prompt is unchanged from v1.0.0. A B2.4 prompt experiment was measured on the reference model, showed no reliable improvement for the target failure (0/5 under both prompts, with new "Memory:"-prefix, second-person and length regressions), and was reverted. memorybank.memory_user_prompt is kept as a behaviour-neutral helper. - tools/memory_fidelity.py (diagnostic only): genre-neutral fixtures plus the failed block, a deterministic fidelity checker, and a real-model shipped-vs-experiment measurement. - tools/memory_diagnostic.py: ranking replica uses production scoring; ranking_crowded, ranking_context_dependent and independent_full fixtures; per-turn isolation and provenance. - tests: B.1's two strict xfails are now ordinary passes; ranking, eviction and excerpt tests; summariser acceptance tests kept apart from diagnostic-measurement tests. - DEVELOPMENT.md: the GPU-host kernel/Ollama watch used `-k -u ollama`, which matches nothing; now the OR form. - docs: CONTEXT-AND-MEMORY 15/18/20/21 as shipped, V1.1-PLAN (status and release criteria 12-13), planning README, VERSION v4.3, reports/v1.1/V1.1-WP-B2-REPORT.md. Deterministic independent-memory recovery: PASS (independent_full fails on v1.0.0 at creation and returns recovered_through_memory_independent here). Reference-model independent recovery: FAILED on the precondition-valid attempt, at memory creation: the summariser omitted a player-established fact from a block it received whole. Accepted as a documented v1.1 residual and carried into the release gate. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
126 lines
5.7 KiB
Python
126 lines
5.7 KiB
Python
"""v1.1 WP-B.1: run the deterministic memory-retention scenarios, or diagnose a real campaign.
|
|
|
|
# the deterministic scenarios, against an isolated database in --out
|
|
.venv/bin/python -m tools.v11_b1_memory scenarios --out "$HOME/v11-evidence/b1/<label>"
|
|
|
|
# the four stages for a finished real campaign (reads its database; embeds
|
|
# the recall query with the campaign's own configured embedding model)
|
|
AIDND_TEST_ENDPOINT=... AIDND_TEST_EMBED_MODEL=nomic-embed-text:latest \\
|
|
.venv/bin/python -m tools.v11_b1_memory diagnose --db <campaign.db> \\
|
|
--plant-depth 3 --out "$HOME/v11-evidence/b1/<label>"
|
|
|
|
Run from `backend/`. Nothing here changes memory behaviour; see
|
|
`tools/memory_diagnostic.py` for what is measured and what the stubs model.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import shutil
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
scen = sub.add_parser("scenarios")
|
|
scen.add_argument("--out", required=True)
|
|
scen.add_argument("--only", action="append", default=[])
|
|
diag = sub.add_parser("diagnose")
|
|
diag.add_argument("--db", required=True)
|
|
diag.add_argument("--plant-depth", type=int, required=True)
|
|
diag.add_argument("--out", required=True)
|
|
args = parser.parse_args()
|
|
|
|
out = Path(args.out)
|
|
out.mkdir(parents=True, exist_ok=True)
|
|
|
|
if args.command == "scenarios":
|
|
db_path = out / "scenarios.db"
|
|
if db_path.exists():
|
|
db_path.unlink()
|
|
os.environ["AIDND_DB_PATH"] = str(db_path)
|
|
else:
|
|
# A copy, so diagnosis never writes to the evidence database.
|
|
copy = out / "diagnosed-copy.db"
|
|
shutil.copy2(args.db, copy)
|
|
os.environ["AIDND_DB_PATH"] = str(copy)
|
|
os.environ.pop("AIDND_DATABASE_URL", None)
|
|
os.environ.pop("DATABASE_URL", None)
|
|
|
|
from tools import memory_diagnostic as md # after the database is chosen
|
|
|
|
if args.command == "scenarios":
|
|
names = args.only or list(md.SCENARIOS)
|
|
summary = {}
|
|
for name in names:
|
|
result = md.run_scenario(md.SCENARIOS[name])
|
|
(out / f"{name}.json").write_text(json.dumps(result, indent=2, default=str))
|
|
d = result.get("diagnosis") or {}
|
|
summary[name] = {
|
|
"verdict": d.get("verdict"),
|
|
"isolation_ok": (result.get("isolation") or {}).get("ok"),
|
|
"plant_depth": result.get("plant_depth"),
|
|
"recall_depth": result.get("recall_depth"),
|
|
"f_evicted_at_turn": (result.get("eviction") or {}).get("f_evicted_at_turn"),
|
|
}
|
|
print(f"{name:26} verdict={d.get('verdict')!s:26} "
|
|
f"isolation_ok={summary[name]['isolation_ok']} "
|
|
f"plant={result.get('plant_depth')} recall={result.get('recall_depth')}")
|
|
(out / "summary.json").write_text(json.dumps(summary, indent=2))
|
|
return 0
|
|
|
|
import asyncio
|
|
|
|
from sqlalchemy.orm import undefer
|
|
|
|
from app import memorybank, models
|
|
from app.database import SessionLocal
|
|
|
|
endpoint = os.environ.get("AIDND_TEST_ENDPOINT", "")
|
|
embed_model = os.environ.get("AIDND_TEST_EMBED_MODEL", "")
|
|
with SessionLocal() as db:
|
|
adventure = db.query(models.Adventure).order_by(models.Adventure.id).first()
|
|
settings = db.query(models.Settings).filter_by(user_id=adventure.user_id).first()
|
|
if endpoint:
|
|
settings.endpoint_url = endpoint
|
|
if embed_model:
|
|
settings.embedding_model = embed_model
|
|
recall_action = (db.query(models.Action)
|
|
.filter(models.Action.adventure_id == adventure.id,
|
|
models.Action.type == "ai")
|
|
.options(undefer(models.Action.context_snapshot))
|
|
.order_by(models.Action.id.desc()).first())
|
|
embed = memorybank.embedding_provider(settings).embed
|
|
iso = md.isolation(db, adventure, md.FACT_F, args.plant_depth,
|
|
recall_snapshot=recall_action.context_snapshot,
|
|
recall_depth=recall_action.depth)
|
|
diagnosis = asyncio.run(md.diagnose(db, adventure, settings, md.FACT_F, args.plant_depth,
|
|
recall_action=recall_action, embed=embed))
|
|
variants = {}
|
|
memory_id = diagnosis["created"]["memory_id"]
|
|
if memory_id is not None and not diagnosis.get("retained", {}).get("forgotten"):
|
|
base = md.production_query(adventure, recall_action.id)
|
|
for label, query in (("recall_turn", base),
|
|
("paraphrase", md.variant_query(base, md.PARAPHRASE_QUERY)),
|
|
("unrelated", md.variant_query(base, md.UNRELATED_QUERY))):
|
|
ranking = asyncio.run(md.rank_bank(db, adventure, settings, query, embed))
|
|
row = next((r for r in ranking["scored"] if r["memory_id"] == memory_id), None)
|
|
variants[label] = {"rank": row and row["rank"], "of": len(ranking["scored"]),
|
|
"similarity": row and row["similarity"],
|
|
"lexical_score": row and row["lexical_score"],
|
|
"final_score": row and row["final_score"],
|
|
"selected": bool(row and row["selected"])}
|
|
db.rollback()
|
|
report = {"isolation": iso, "diagnosis": diagnosis, "ranking_variants": variants}
|
|
(out / "diagnosis.json").write_text(json.dumps(report, indent=2, default=str))
|
|
print(json.dumps({"isolation_ok": iso["ok"], "verdict": diagnosis["verdict"]}, indent=2))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|