Corrects the memory mechanisms WP-B.1 diagnosed, one at a time, each verified before the next. Accepted by the owner with a documented reference-model limitation. No schema, bundle format, setting default, lineage, authority or protocol-cleanup change. - B2.1 ranking: the retrieval query is the player's input plus a bounded scene context (state scene + end of the newest narration), embedded in one call. final = semantic (0.6 input / 0.4 context) + 0.15 x lexical, where lexical is a rarity-weighted share of the input's words, computed per turn over the candidates with no index. Scores and the query are recorded per used memory; pins and redundancy suppression unchanged. - B2.2 coverage-aware eviction (memorybank.eviction_order): the earliest and newest memories are kept, the smallest coverage hole goes first, least-recently-used breaks ties and remains the fallback. Bounded; pins never evicted; frozen-bank protection kept; reads no text or vectors. - B2.3 bounded memory creation: a block longer than 2,000 tokens is shown to the summariser as head + tail with an omission marker, inside the same budget; shorter blocks unchanged; the marker is never stored. - The memory summariser prompt is unchanged from v1.0.0. A B2.4 prompt experiment was measured on the reference model, showed no reliable improvement for the target failure (0/5 under both prompts, with new "Memory:"-prefix, second-person and length regressions), and was reverted. memorybank.memory_user_prompt is kept as a behaviour-neutral helper. - tools/memory_fidelity.py (diagnostic only): genre-neutral fixtures plus the failed block, a deterministic fidelity checker, and a real-model shipped-vs-experiment measurement. - tools/memory_diagnostic.py: ranking replica uses production scoring; ranking_crowded, ranking_context_dependent and independent_full fixtures; per-turn isolation and provenance. - tests: B.1's two strict xfails are now ordinary passes; ranking, eviction and excerpt tests; summariser acceptance tests kept apart from diagnostic-measurement tests. - DEVELOPMENT.md: the GPU-host kernel/Ollama watch used `-k -u ollama`, which matches nothing; now the OR form. - docs: CONTEXT-AND-MEMORY 15/18/20/21 as shipped, V1.1-PLAN (status and release criteria 12-13), planning README, VERSION v4.3, reports/v1.1/V1.1-WP-B2-REPORT.md. Deterministic independent-memory recovery: PASS (independent_full fails on v1.0.0 at creation and returns recovered_through_memory_independent here). Reference-model independent recovery: FAILED on the precondition-valid attempt, at memory creation: the summariser omitted a player-established fact from a block it received whole. Accepted as a documented v1.1 residual and carried into the release gate. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VvegagkhuCZoFPdv4M1egY
435 lines
21 KiB
Python
435 lines
21 KiB
Python
"""v1.1 WP-B.1: the memory-retention diagnostic, deterministically.
|
|
|
|
B.1 changes no memory behaviour. These tests prove two things about the
|
|
diagnostic in `tools/memory_diagnostic.py`:
|
|
|
|
1. **It measures what it claims.**
|
|
- The fixture keeps the planted fact out of every layer except memory.
|
|
- Each stage (created, retained, ranked, injected) is reported from the rows
|
|
and the recall turn's own stored context.
|
|
- Its ranking agrees with the selection production stored.
|
|
2. **What it finds on this tree.** The scenarios run with a best-case summariser,
|
|
one that keeps a fact if and only if the fact reached it. Any failure is
|
|
therefore the application's mechanism, not a model's writing.
|
|
- WP-B.1 marked the criteria v1.0.0 did not meet `xfail(strict=True)`. WP-B.2
|
|
fixed ranking, eviction and the creation excerpt, and those tests are now
|
|
ordinary passes; the v1.0.0 results are recorded in the WP-B.2 report.
|
|
- The same file is run unchanged against v1.0.0 for the baseline.
|
|
|
|
python -m pytest tests/test_v11_b1_memory_diagnostic.py -v
|
|
"""
|
|
|
|
import asyncio
|
|
|
|
import pytest
|
|
from fastapi import Depends
|
|
from fastapi.testclient import TestClient
|
|
from sqlalchemy.orm import undefer
|
|
|
|
from app import auth, limits, memorybank, models
|
|
from app.database import Base, SessionLocal, engine, get_db
|
|
from app.main import app
|
|
from app.routers import adventures
|
|
from tools import memory_diagnostic as md
|
|
|
|
_results: dict = {}
|
|
|
|
|
|
def scenario(name: str) -> dict:
|
|
"""Runs a named scenario once per session and keeps the result."""
|
|
if name not in _results:
|
|
_results[name] = md.run_scenario(md.SCENARIOS[name])
|
|
return _results[name]
|
|
|
|
|
|
# ------------------------------------------------------- fixture preconditions
|
|
|
|
def test_the_fact_is_planted_early_and_recalled_past_depth_one_hundred():
|
|
result = scenario("independent_default")
|
|
assert result["plant_depth"] is not None and result["plant_depth"] <= 3
|
|
assert result["recall_depth"] >= 100
|
|
|
|
|
|
@pytest.mark.parametrize("check", ["state_document", "state_snapshots", "later_narration",
|
|
"summary", "knowledge", "recent_history", "state_section"])
|
|
def test_no_layer_but_memory_carries_the_fact(check):
|
|
"""A test where another layer carries F is not evidence about memory."""
|
|
isolation = scenario("independent_default")["isolation"]
|
|
assert isolation["checks"][check]["ok"], isolation["checks"][check]
|
|
assert isolation["ok"]
|
|
|
|
|
|
def test_the_isolation_check_fails_when_another_layer_carries_the_fact():
|
|
"""The negative control for the precondition itself: a state fact naming F."""
|
|
fact = md.FACT_F
|
|
with SessionLocal() as db:
|
|
Base.metadata.create_all(bind=engine)
|
|
try:
|
|
user = models.User(is_guest=False, email="b1-iso@example.com")
|
|
db.add(user)
|
|
db.flush()
|
|
adventure = models.Adventure(user_id=user.id, title="iso")
|
|
adventure.narrative_state = {"facts": [{"id": "x", "predicate": "hidden",
|
|
"value": "the amber sundial is in the teapot"}]}
|
|
db.add(adventure)
|
|
db.commit()
|
|
result = md.isolation(db, adventure, fact, 1)
|
|
assert result["ok"] is False
|
|
assert result["checks"]["state_document"]["ok"] is False
|
|
finally:
|
|
db.close()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
# ------------------------------------------------------------------- stages
|
|
|
|
def test_creation_is_reported_with_the_covering_memory_and_what_the_summariser_saw():
|
|
created = scenario("independent_default")["diagnosis"]["created"]
|
|
assert created["yes"] is True
|
|
assert created["source_start"] <= scenario("independent_default")["plant_depth"] <= created["source_end"]
|
|
assert md.FACT_F.carried_by(created["memory_text"])
|
|
covering = [c for c in created["covering_memories"] if c["memory_id"] == created["memory_id"]]
|
|
assert covering and covering[0]["fact_in_block"] and covering[0]["fact_in_summariser_excerpt"]
|
|
|
|
|
|
def test_retention_is_reported_with_the_bank_and_its_eviction_order():
|
|
retained = scenario("independent_default")["diagnosis"]["retained"]
|
|
assert retained["yes"] is True and retained["forgotten"] is False
|
|
assert retained["on_active_lineage"] is True
|
|
assert retained["active_memories"] <= retained["memory_bank_capacity"]
|
|
assert retained["eviction_position"] is not None
|
|
|
|
|
|
def test_ranking_is_production_ranking_and_agrees_with_the_stored_selection():
|
|
ranked = scenario("independent_default")["diagnosis"]["ranked"]
|
|
assert ranked["replica_matches_stored_selection"] is True
|
|
assert ranked["top_k_cutoff"] == 5
|
|
assert ranked["yes"] is True and ranked["selected"] is True
|
|
assert 1 <= ranked["rank"] <= ranked["top_k_cutoff"]
|
|
# v1.1 WP-B.2: every part of the score is reported, and they add up.
|
|
assert 0.0 <= ranked["lexical_score"] <= 1.0
|
|
assert ranked["final_score"] == pytest.approx(
|
|
ranked["semantic_score"] + memorybank.LEXICAL_WEIGHT * ranked["lexical_score"], abs=2e-4)
|
|
assert ranked["query"]["input"].endswith(md.SCENARIOS["independent_default"].recall_text)
|
|
|
|
|
|
def test_injection_is_read_from_the_recall_turns_own_context():
|
|
diagnosis = scenario("independent_default")["diagnosis"]
|
|
assert diagnosis["injected"]["yes"] is True
|
|
assert diagnosis["injected"]["context_component"] == md.MEMORIES_LABEL
|
|
assert diagnosis["injected"]["token_count"] > 0
|
|
assert diagnosis["verdict"] == "injected"
|
|
|
|
|
|
def test_ranking_variants_direct_paraphrase_and_unrelated():
|
|
variants = scenario("independent_default")["ranking_variants"]
|
|
assert variants["direct"]["rank"] == 1 and variants["direct"]["selected"]
|
|
assert variants["paraphrase"]["rank"] == 1 and variants["paraphrase"]["selected"]
|
|
assert (variants["direct"]["final_score"] > variants["paraphrase"]["final_score"]
|
|
> variants["unrelated"]["final_score"])
|
|
|
|
|
|
def test_retrieval_still_fills_top_k_whatever_the_similarity():
|
|
"""There is still no relevance floor: an unrelated question selects a full
|
|
`memory_top_k`. B.2 changed which memories those are, not how many — the
|
|
early fact is no longer carried along by an unrelated question."""
|
|
variants = scenario("independent_default")["ranking_variants"]
|
|
assert variants["unrelated"]["selected_count"] == 5
|
|
assert variants["unrelated"]["selected"] is False
|
|
assert variants["unrelated"]["rank"] > 5
|
|
|
|
|
|
# ------------------------------------------------- WP-B.2 ranking acceptance
|
|
|
|
@pytest.mark.parametrize("name", ["ranking_crowded", "ranking_context_dependent"])
|
|
def test_acceptance_the_early_memory_is_ranked_and_injected_below_capacity(name):
|
|
"""The B.1 ranking failure, made deterministic. On v1.0.0 both fixtures are
|
|
`retained_but_not_ranked` (ranks 7 and 6 of 17 against a top-k of 4)."""
|
|
result = scenario(name)
|
|
assert result["isolation"]["ok"], result["isolation"]
|
|
diagnosis = result["diagnosis"]
|
|
assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"]
|
|
assert diagnosis["retained"]["active_memories"] <= diagnosis["retained"]["memory_bank_capacity"]
|
|
assert diagnosis["ranked"]["yes"] and diagnosis["ranked"]["rank"] <= 4
|
|
assert diagnosis["ranked"]["replica_matches_stored_selection"] is True
|
|
assert diagnosis["injected"]["yes"] is True
|
|
assert diagnosis["verdict"] == "injected"
|
|
|
|
|
|
def test_the_crowded_fixture_is_won_by_the_players_question():
|
|
ranked = scenario("ranking_crowded")["diagnosis"]["ranked"]
|
|
assert ranked["rank"] == 1
|
|
assert ranked["lexical_score"] > 0 # "sundial" and "amber" are in the question
|
|
|
|
|
|
def test_a_paraphrase_is_found_by_meaning_not_by_shared_words():
|
|
"""Lexical matching must not replace semantic retrieval. The paraphrase
|
|
shares none of F's distinctive words, yet ranks first."""
|
|
for name in ("ranking_crowded", "independent_default"):
|
|
paraphrase = scenario(name)["ranking_variants"]["paraphrase"]
|
|
assert paraphrase["rank"] == 1 and paraphrase["selected"]
|
|
# Only "Mara" is shared, which is far less than the direct question holds.
|
|
direct = scenario(name)["ranking_variants"]["direct"]
|
|
assert paraphrase["lexical_score"] < direct["lexical_score"] / 2
|
|
|
|
|
|
def test_a_context_dependent_question_needs_the_scene():
|
|
""""I ask her what she keeps up there" names nothing F's memory holds. The
|
|
scene the last narration set up (Mara, the top shelf, a kettle) is what
|
|
finds it; without that context it ranks last."""
|
|
result = scenario("ranking_context_dependent")
|
|
ranked = result["diagnosis"]["ranked"]
|
|
assert ranked["lexical_score"] == 0.0
|
|
assert ranked["rank"] <= 4
|
|
assert "top shelf" in ranked["query"]["context"]
|
|
assert result["ranking_variants"]["input_only"]["rank"] > 4
|
|
|
|
|
|
def test_an_unrelated_rare_word_does_not_outrank_the_relevant_memory():
|
|
"""Negative control: the paraphrase plus a place only one other memory
|
|
holds. The decoy gains lexical score, and still ranks below F."""
|
|
for name in ("ranking_crowded", "independent_default"):
|
|
control = scenario(name)["ranking_variants"]["rare_word_with_paraphrase"]
|
|
assert control["decoy_lexical_score"] > control["lexical_score"]
|
|
assert control["rank"] == 1
|
|
assert control["decoy_rank"] > control["rank"]
|
|
|
|
|
|
def test_common_words_contribute_nothing():
|
|
common = scenario("independent_default")["ranking_variants"]["common_words"]
|
|
assert common["lexical_score"] == 0.0
|
|
|
|
|
|
# ---------------------------------------------------------- capacity/eviction
|
|
|
|
@pytest.mark.parametrize("name", ["past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"])
|
|
def test_past_capacity_the_early_memory_is_retained(name):
|
|
"""v1.1 WP-B.2. On v1.0.0 all three are `created_but_evicted`: F was the
|
|
least recently used row once recent narration stopped retrieving it, and
|
|
went first (turns 21, 21 and 36). Coverage-first eviction keeps the only
|
|
memory of the opening, so it stays active and is recalled at depth 106."""
|
|
result = scenario(name)
|
|
assert result["isolation"]["ok"], result["isolation"]
|
|
diagnosis = result["diagnosis"]
|
|
assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"]
|
|
assert result["eviction"]["f_evicted_at_turn"] is None
|
|
# The bank really was past capacity, and stayed bounded.
|
|
assert result["eviction"]["first_eviction_turn"] is not None
|
|
assert all(t["active"] <= result["scenario"]["capacity"] + (1 if result["scenario"]["pin_first_memory"] else 0)
|
|
for t in result["trace"])
|
|
assert result["trace"][-1]["total"] > result["scenario"]["capacity"]
|
|
|
|
|
|
def test_past_capacity_the_bank_still_describes_the_whole_story():
|
|
"""What the rule buys in general, not only for F: the active bank reaches
|
|
from the opening to the newest block, and no stretch between them goes
|
|
undescribed for more than twice the average spacing a bank of this capacity
|
|
can afford (story span / capacity). On v1.0.0 these banks began at depths 36
|
|
and 18: the opening was simply gone."""
|
|
for name in ("past_capacity", "past_capacity_low_top_k"):
|
|
result = scenario(name)
|
|
cover = result["trace"][-1]["coverage"]
|
|
assert cover["first_start"] == 0
|
|
assert cover["last_end"] >= result["recall_depth"] - 2 * memorybank.MEMORY_INTERVAL
|
|
assert cover["largest_gap"] <= 2 * (cover["last_end"] + 1) / result["scenario"]["capacity"]
|
|
|
|
|
|
def test_no_memory_is_evicted_by_the_same_pass_that_created_it():
|
|
"""The frozen-bank regression the v1.0.0 rule fixed, still holding."""
|
|
for name in ("past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"):
|
|
assert scenario(name)["eviction"]["created_and_evicted_same_turn"] == []
|
|
|
|
|
|
def test_a_pinned_memory_survives_capacity():
|
|
eviction = scenario("past_capacity_pinned")["eviction"]
|
|
assert eviction["pinned_memory_id"] is not None
|
|
assert eviction["pinned_memory_forgotten"] is False
|
|
|
|
|
|
def test_acceptance_an_early_fact_is_recalled_from_memory_past_capacity():
|
|
"""Was `xfail(strict=True)` in WP-B.1; B.2 fixed the eviction rule."""
|
|
for name in ("past_capacity", "past_capacity_pinned", "past_capacity_low_top_k"):
|
|
assert scenario(name)["diagnosis"]["verdict"] == "injected"
|
|
|
|
|
|
# ---------------------------------------------------------- creation window
|
|
|
|
def test_a_fact_early_in_a_long_block_now_reaches_the_summariser():
|
|
"""v1.1 WP-B.2. On v1.0.0 this block (2,079 tokens) was cut to its last
|
|
2,000, the fact at its start was never seen, and the stage was
|
|
`not_created`. The excerpt is now the block's opening and end."""
|
|
result = scenario("long_block_fact_early")
|
|
created = result["diagnosis"]["created"]
|
|
covering = created["covering_memories"]
|
|
assert covering, "the long block must have been summarised"
|
|
assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS
|
|
assert covering[0]["fact_in_block"] is True
|
|
assert covering[0]["fact_in_summariser_excerpt"] is True
|
|
assert created["yes"] is True
|
|
assert created["source_start"] <= result["plant_depth"] <= created["source_end"]
|
|
assert memorybank.EXCERPT_OMISSION_MARKER not in created["memory_text"]
|
|
|
|
|
|
def test_the_same_fact_late_in_the_same_sized_block_does():
|
|
result = scenario("long_block_fact_late")
|
|
covering = result["diagnosis"]["created"]["covering_memories"]
|
|
assert covering[0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS
|
|
assert covering[0]["fact_in_summariser_excerpt"] is True
|
|
assert result["diagnosis"]["created"]["yes"] is True
|
|
|
|
|
|
def test_acceptance_a_fact_early_in_a_long_block_is_remembered():
|
|
"""Was `xfail(strict=True)` in WP-B.1; B.2 changed the excerpt."""
|
|
assert scenario("long_block_fact_early")["diagnosis"]["created"]["yes"] is True
|
|
assert scenario("long_block_fact_early")["diagnosis"]["verdict"] == "injected"
|
|
|
|
|
|
# ------------------------------------------ WP-B.2 full deterministic acceptance
|
|
|
|
def test_acceptance_full_isolation_holds_on_every_turn():
|
|
"""`independent_full`: long blocks, a crowded query, a bank past capacity.
|
|
F must be carried by memory alone for the whole run, not only at recall."""
|
|
result = scenario("independent_full")
|
|
assert result["plant_depth"] <= 3 and result["recall_depth"] >= 100
|
|
assert not any(t["f_in_state"] for t in result["trace"])
|
|
assert not any(t["f_in_summary"] for t in result["trace"])
|
|
assert result["isolation"]["ok"], result["isolation"]
|
|
for check in ("state_document", "state_snapshots", "later_narration", "summary",
|
|
"knowledge", "recent_history", "state_section"):
|
|
assert result["isolation"]["checks"][check]["ok"], check
|
|
|
|
|
|
def test_acceptance_full_every_stage_passes_past_capacity_with_long_blocks():
|
|
"""On v1.0.0 this fixture fails at creation: every block is over 2,000
|
|
tokens, and the fact at the start of the first one is never summarised."""
|
|
result = scenario("independent_full")
|
|
diagnosis = result["diagnosis"]
|
|
assert result["trace"][-1]["total"] > result["scenario"]["capacity"]
|
|
assert diagnosis["created"]["covering_memories"][0]["block_tokens"] > memorybank.MEMORY_EXCERPT_TOKENS
|
|
assert diagnosis["created"]["yes"] and diagnosis["retained"]["yes"]
|
|
assert diagnosis["ranked"]["yes"] and diagnosis["ranked"]["replica_matches_stored_selection"]
|
|
assert diagnosis["injected"]["yes"]
|
|
assert diagnosis["verdict"] == "injected"
|
|
|
|
|
|
def test_acceptance_full_provenance_resolves_to_the_planting_turn():
|
|
provenance = scenario("independent_full")["provenance"]
|
|
assert provenance["recorded"] is not None
|
|
assert provenance["range_covers_plant"] and provenance["matches_row"]
|
|
assert provenance["source_block_holds_planting"] is True
|
|
assert provenance["recorded"]["authority"] == memorybank.ACCEPTED_STORY
|
|
|
|
|
|
def test_acceptance_full_is_the_long_run_independent_memory_verdict():
|
|
"""The same measurements, judged by the long-run tool's own verdict."""
|
|
from tools import m11_long_run as lr
|
|
|
|
result = scenario("independent_full")
|
|
checks = result["isolation"]["checks"]
|
|
diagnosis = result["diagnosis"]
|
|
verdict = lr._independent_memory_verdict({
|
|
"independent_planted_depth": result["plant_depth"],
|
|
"planted_turn_outside_history": checks["recent_history"]["ok"],
|
|
"absent_from_state": checks["state_document"]["ok"] and checks["state_snapshots"]["ok"]
|
|
and not any(t["f_in_state"] for t in result["trace"]),
|
|
"absent_from_summary": checks["summary"]["ok"]
|
|
and not any(t["f_in_summary"] for t in result["trace"]),
|
|
"absent_from_knowledge": checks["knowledge"]["ok"],
|
|
"absent_from_later_narration": checks["later_narration"]["ok"],
|
|
"memory_covering_planting_carries_fact": diagnosis["created"]["yes"],
|
|
"memory_forgotten": not diagnosis["retained"]["yes"],
|
|
"memory_injected": diagnosis["injected"]["yes"],
|
|
})
|
|
assert verdict == "recovered_through_memory_independent"
|
|
|
|
|
|
# ------------------------------------------------------- lineage control (G)
|
|
|
|
def test_an_abandoned_lines_memory_is_stored_but_never_eligible_or_injected():
|
|
g = scenario("lineage_control")["lineage_control"]
|
|
assert g["memory_ids"], "G's memory must exist on line A before it is abandoned"
|
|
assert sorted(g["stored"]) == sorted(g["memory_ids"])
|
|
assert g["eligible_on_active_line"] == []
|
|
assert g["g_text_ever_in_used_memories"] is False
|
|
# Any turn that did name G's memory was on line A, before the divergence.
|
|
assert g["eligible_after_returning_to_line_a"] == g["memory_ids"]
|
|
|
|
|
|
def test_the_lineage_scenario_still_diagnoses_f_on_the_active_line():
|
|
result = scenario("lineage_control")
|
|
assert result["isolation"]["ok"], result["isolation"]
|
|
assert result["diagnosis"]["verdict"] == "injected"
|
|
|
|
|
|
# ----------------------------------------------------- authority control
|
|
|
|
@pytest.fixture()
|
|
def authority_client(monkeypatch):
|
|
embedder = md.ConceptEmbedder()
|
|
Base.metadata.create_all(bind=engine)
|
|
memorybank._vector_cache.clear()
|
|
with SessionLocal() as db:
|
|
user = models.User(is_guest=False, email="b1-auth@example.com")
|
|
db.add(user)
|
|
db.flush()
|
|
db.add(models.Settings(user_id=user.id, model="script",
|
|
endpoint_url="http://127.0.0.1:9/v1",
|
|
embedding_model="concept-embed", memory_top_k=5))
|
|
adventure = models.Adventure(user_id=user.id, title="auth", memory_bank_enabled=True,
|
|
auto_summarize=True)
|
|
db.add(adventure)
|
|
db.flush()
|
|
db.add(models.Action(adventure_id=adventure.id, type="start", text="The tavern at dusk."))
|
|
db.commit()
|
|
adv, user_id = adventure.id, user.id
|
|
monkeypatch.setattr(limits, "check_row_cap", lambda *a, **k: None)
|
|
monkeypatch.setattr(adventures.turns, "OpenAICompatibleProvider", md.ScriptNarrator)
|
|
monkeypatch.setattr(memorybank, "embedding_provider", lambda s: embedder)
|
|
monkeypatch.setattr(memorybank, "summary_provider", lambda s: md.BestCaseSummariser())
|
|
monkeypatch.setattr(memorybank, "schedule_post_turn", lambda a: None)
|
|
app.dependency_overrides[auth.get_current_user] = (
|
|
lambda db=Depends(get_db): db.get(models.User, user_id))
|
|
client = TestClient(app)
|
|
client.adv = adv
|
|
try:
|
|
yield client
|
|
finally:
|
|
app.dependency_overrides.clear()
|
|
adventures.turns._active_turns.clear()
|
|
memorybank._vector_cache.clear()
|
|
Base.metadata.drop_all(bind=engine)
|
|
|
|
|
|
def test_a_memory_that_contradicts_state_loses_and_changes_nothing(authority_client):
|
|
client, adv = authority_client, authority_client.adv
|
|
corrected = client.post(f"/api/adventures/{adv}/state/corrections", json={"events": [
|
|
{"type": "add_fact", "predicate": "the tavern lamp is lit", "fact_id": "lamp-lit"}]})
|
|
assert corrected.status_code in (200, 201), corrected.text[:300]
|
|
made = client.post(f"/api/adventures/{adv}/memories",
|
|
json={"text": "The tavern lamp was never lit that night."})
|
|
assert made.status_code == 201, made.text[:300]
|
|
client.patch(f"/api/adventures/{adv}/memories/{made.json()['id']}", json={"pinned": True})
|
|
asyncio.run(memorybank.run_post_turn(adv)) # embed it
|
|
|
|
before = client.get(f"/api/adventures/{adv}/state").json()["document"]
|
|
md.ScriptNarrator.next_reply = 'The fire crackles.\n```state\n{"events": []}\n```'
|
|
played = client.post(f"/api/adventures/{adv}/actions",
|
|
json={"type": "do", "text": "I look at the lamp."})
|
|
assert played.status_code == 200 and '"type": "error"' not in played.text
|
|
after = client.get(f"/api/adventures/{adv}/state").json()["document"]
|
|
assert after == before # retrieval mutated no state
|
|
|
|
with SessionLocal() as db:
|
|
action = (db.query(models.Action).filter_by(adventure_id=adv, type="ai")
|
|
.options(undefer(models.Action.context_snapshot))
|
|
.order_by(models.Action.id.desc()).first())
|
|
snapshot = action.context_snapshot
|
|
state_text = md._section(snapshot, md.STATE_LABEL)
|
|
memory_text = md._section(snapshot, md.MEMORIES_LABEL)
|
|
assert "the tavern lamp is lit" in state_text
|
|
assert "never lit" in memory_text
|
|
assert memory_text.startswith("Memories from earlier in the story")
|
|
labels = [s["label"] for s in snapshot["sections"]]
|
|
# State is read last of the live sections: it settles the conflict.
|
|
assert labels.index(md.STATE_LABEL) > labels.index(md.MEMORIES_LABEL)
|