Files
interactive-story/backend/tests/test_worldstate.py
T
Claude 91907a30bd Name every stat in the guide by the path the AI has to write
The referee refused `npc.<id>.status` on a character that has a status stat
under another name, and the refusal was right: the model was reaching for a
path the guide never gave it.

The guide is the fixed list of everything a scenario tracks, and it named
stats in prose. A player stat and a world stat both read as a bare name, and
an NPC's stats read as the display name plus the stat — "Trainer Milo
active_status". The model had to turn that back into `npc.milo.active_status`
itself, and in the Pokemon demo five other characters carry a stat called
`status`, so the path it built was `npc.milo.status`. The live values do state
the paths, but only for the NPCs a scene has mentioned, so an NPC off screen
was addressable only by guesswork.

Each line now leads with the path: `player.potions`, `npc.milo.active_status`,
`flags.sandstorm_active`. An NPC's header is written even when it has no
description, because it is the one line that ties a display name to its id.
`EMIT_RULE` points at the guide for paths rather than at the live values
alone.

A free-text stat is marked `(free text)` beside its path, which is the wording
`EMIT_RULE` already used to describe it — the guide had been writing "free
text" at the end of the line instead. Such a stat now also gets a line when it
has no description, where before it was dropped and the model was left to send
a number for a stat that holds a string.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017imUKVPhqophVUZZwJK7ST
2026-08-30 06:46:12 +00:00

306 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Unit tests for the RPG world-state engine (Phase 12): delta extraction and
the code that enforces clamp, cooldown, and milestone rules.
python -m pytest tests/test_worldstate.py -v
"""
from app import worldstate as w
SCHEMA = {
"world": {"day": {"type": "counter", "min": 1, "initial": 1}},
"player": {
"hp": {"min": 0, "max": 100, "initial": 100, "max_delta_per_turn": 30,
"bands": [[0, 20, "very weak"], [20, 40, "hurt"],
[40, 60, "minor damage"], [60, 90, "healthy"],
[90, 100, "full health"]]},
"outfit": {"type": "text", "initial": "traveling clothes", "desc": "What the player is wearing"},
},
"npcs": {
"gwen": {
"name": "Gwen",
"keys": "Gwen, ranger",
"desc": "A loyal ranger",
"stats": {"trust": {"min": -100, "max": 100, "initial": 0, "cooldown": 2}},
},
"drake": {
"name": "The Drake",
"stats": {"ferocity": {"min": 0, "max": 100, "initial": 50}},
},
},
"flags": {
"has_key": {"desc": "Holds the key", "initial": False},
"disguised": {"desc": "In disguise"},
},
"milestones": {"rescue_gwen": {"desc": "Rescue Gwen"}},
}
def fresh():
return w.instantiate(SCHEMA)
def test_instantiate_uses_initials():
ws = fresh()
assert ws["world"] == {"day": 1}
assert ws["player"] == {"hp": 100, "outfit": "traveling clothes"}
# Each defined NPC is instantiated up front with its own stats.
assert ws["npc"] == {"gwen": {"trust": 0}, "drake": {"ferocity": 50}}
assert ws["milestones"] == {}
def test_text_stat_replaces_value():
ws, report = w.apply_delta(fresh(), SCHEMA, {"player.outfit": "muddy cloak"}, 1)
assert ws["player"]["outfit"] == "muddy cloak"
assert report["applied"][0] == {"path": "player.outfit", "old": "traveling clothes", "new": "muddy cloak"}
def test_text_stat_rejects_non_string():
ws, report = w.apply_delta(fresh(), SCHEMA, {"player.outfit": 5}, 1)
assert ws["player"]["outfit"] == "traveling clothes"
assert report["rejected"][0]["reason"] == "not a string"
def test_text_stat_noop_when_unchanged():
ws, report = w.apply_delta(fresh(), SCHEMA, {"player.outfit": "traveling clothes"}, 1)
assert report["applied"] == []
def test_per_npc_distinct_stats():
ws, _ = w.apply_delta(fresh(), SCHEMA, {"npc.drake.ferocity": 20}, 1)
assert ws["npc"]["drake"]["ferocity"] == 70
# gwen has no `ferocity` stat, and drake has no `trust` stat. Cross paths
# are rejected.
ws, report = w.apply_delta(ws, SCHEMA, {"npc.gwen.ferocity": 5, "npc.bogus.trust": 5}, 2)
reasons = {r["reason"] for r in report["rejected"]}
assert reasons == {"unknown npc stat", "unknown npc"}
def test_has_schema():
assert w.has_schema(SCHEMA)
assert not w.has_schema(None)
assert not w.has_schema({})
assert not w.has_schema({"npc_card_types": ["npc"]}) # config only, no stats
def test_max_delta_per_turn_clamps():
ws, report = w.apply_delta(fresh(), SCHEMA, {"player.hp": -50}, 5)
assert ws["player"]["hp"] == 70 # -50 capped to -30
assert report["clamped"]
def test_clamp_to_min():
ws, _ = w.apply_delta(fresh(), SCHEMA, {"player.hp": -30}, 1)
ws, _ = w.apply_delta(ws, SCHEMA, {"player.hp": -30}, 3)
ws, _ = w.apply_delta(ws, SCHEMA, {"player.hp": -30}, 5)
ws, report = w.apply_delta(ws, SCHEMA, {"player.hp": -30}, 7)
assert ws["player"]["hp"] == 0 # 100-30-30-30-30 clamps at min 0
assert report["clamped"]
def test_counter_rejects_negative():
ws, report = w.apply_delta(fresh(), SCHEMA, {"world.day": -1}, 3)
assert ws["world"]["day"] == 1
assert report["rejected"][0]["reason"] == "counter can't decrease"
ws, _ = w.apply_delta(ws, SCHEMA, {"world.day": 1}, 4)
assert ws["world"]["day"] == 2
def test_npc_cooldown():
ws, _ = w.apply_delta(fresh(), SCHEMA, {"npc.gwen.trust": 10}, 7)
assert ws["npc"]["gwen"]["trust"] == 10
# cooldown 2: another change at index 8 is too soon.
ws, report = w.apply_delta(ws, SCHEMA, {"npc.gwen.trust": 10}, 8)
assert ws["npc"]["gwen"]["trust"] == 10
assert report["rejected"][0]["reason"] == "cooldown"
# far enough later, it applies.
ws, _ = w.apply_delta(ws, SCHEMA, {"npc.gwen.trust": 10}, 10)
assert ws["npc"]["gwen"]["trust"] == 20
def test_milestone_sticky():
ws, report = w.apply_delta(fresh(), SCHEMA, {"milestones.rescue_gwen": True}, 9)
assert ws["milestones"]["rescue_gwen"] == {"reached": True, "at": 9}
assert report["applied"]
# second set is a silent no-op.
ws, report = w.apply_delta(ws, SCHEMA, {"milestones.rescue_gwen": True}, 11)
assert ws["milestones"]["rescue_gwen"]["at"] == 9
assert not report["applied"]
# false is ignored.
ws, report = w.apply_delta(ws, SCHEMA, {"milestones.rescue_gwen": False}, 13)
assert ws["milestones"]["rescue_gwen"]["reached"] is True
def test_flags_toggle_both_ways():
ws = fresh()
assert ws["flags"] == {"has_key": False, "disguised": False} # initials
ws, report = w.apply_delta(ws, SCHEMA, {"flags.has_key": True}, 1)
assert ws["flags"]["has_key"] is True
assert report["applied"]
# Setting it back to false works, because flags are two-way, unlike
# sticky milestones.
ws, _ = w.apply_delta(ws, SCHEMA, {"flags.has_key": False}, 2)
assert ws["flags"]["has_key"] is False
# setting to the same value is a no-op.
ws, report = w.apply_delta(ws, SCHEMA, {"flags.has_key": False}, 3)
assert not report["applied"]
def test_flag_rejects_non_bool_and_unknown():
ws, report = w.apply_delta(fresh(), SCHEMA, {"flags.has_key": 1, "flags.nope": True}, 1)
reasons = {r["reason"] for r in report["rejected"]}
assert reasons == {"not a boolean", "unknown flag"}
assert ws["flags"]["has_key"] is False
def test_override_sets_absolute_value_bypassing_cap():
# A manual override does not obey max_delta_per_turn the way a turn
# does. It sets the value directly, though the value is still clamped
# to min/max.
ws, report = w.apply_override(fresh(), SCHEMA, {"player.hp": 10})
assert ws["player"]["hp"] == 10
assert report["applied"] == [{"path": "player.hp", "old": 100, "new": 10}]
ws, report = w.apply_override(ws, SCHEMA, {"player.hp": 999})
assert ws["player"]["hp"] == 100 # still clamps to max
def test_override_bypasses_cooldown_and_counter_rule():
ws, _ = w.apply_override(fresh(), SCHEMA, {"npc.gwen.trust": 10})
# a second override immediately after would be blocked by cooldown under
# apply_delta, but override ignores cooldown entirely.
ws, report = w.apply_override(ws, SCHEMA, {"npc.gwen.trust": -5})
assert ws["npc"]["gwen"]["trust"] == -5
assert report["applied"]
# counters can be set down directly too (a correction, not a turn).
ws, report = w.apply_override(ws, SCHEMA, {"world.day": 1})
assert ws["world"]["day"] == 1
assert report["applied"]
def test_override_text_stat_replaces():
ws, report = w.apply_override(fresh(), SCHEMA, {"player.outfit": "knight's plate"})
assert ws["player"]["outfit"] == "knight's plate"
assert report["applied"]
def test_override_milestone_toggles_both_ways():
ws, _ = w.apply_override(fresh(), SCHEMA, {"milestones.rescue_gwen": True})
assert ws["milestones"]["rescue_gwen"]["reached"] is True
# unlike apply_delta, override can un-set a milestone.
ws, report = w.apply_override(ws, SCHEMA, {"milestones.rescue_gwen": False})
assert "rescue_gwen" not in ws["milestones"]
assert report["applied"]
def test_override_rejects_unknown_and_bad_type():
ws, report = w.apply_override(fresh(), SCHEMA, {
"player.hp": "not a number",
"npc.bogus.trust": 5,
"flags.nope": True,
})
reasons = {r["path"]: r["reason"] for r in report["rejected"]}
assert reasons["player.hp"] == "not a number"
assert reasons["npc.bogus.trust"] == "unknown npc"
assert reasons["flags.nope"] == "unknown flag"
assert ws["player"]["hp"] == 100 # untouched
def test_reference_includes_desc_and_bands_independently():
guide = w.render_reference(SCHEMA)
# hp has both a description and a set of value bands.
assert "very weak" in guide and "range 0–100" in guide
# A counter like day has no desc or bands, so it contributes nothing
# here. Flags do show their desc.
assert "flags.has_key — Holds the key." in guide
# NPCs contribute their own description and per-NPC stat lines.
assert "NPC Gwen (npc.gwen) — A loyal ranger." in guide
assert "npc.gwen.trust" in guide and "npc.drake.ferocity" in guide
def test_reference_names_every_stat_by_its_path():
"""The guide is the model's only complete list of what exists, so it has
to name each stat the way a state block must name it. Naming a stat after
its owner's display name ("Trainer Milo active_status") left the model to
build the path itself, and the paths it built were refused as stats and
characters that do not exist."""
guide = w.render_reference(SCHEMA)
assert "player.hp" in guide
assert "player.outfit" in guide
# The old wording, which read as prose rather than as an address.
assert "Gwen trust" not in guide
assert "The Drake ferocity" not in guide
def test_reference_names_an_npc_that_has_no_description():
"""The live values name an NPC only while a scene mentions them, so the
guide is the only place an off-screen NPC's id is stated. The Drake has no
`desc`, and used to reach the model as "The Drake ferocity" alone."""
guide = w.render_reference(SCHEMA)
assert "NPC The Drake (npc.drake)." in guide
def test_reference_marks_free_text_stats_the_way_the_emit_rule_names_them():
"""`EMIT_RULE` tells the model that a stat "marked (free text) in the stat
guide" takes a whole value rather than a delta, so the guide has to carry
that marker literally."""
assert "(free text)" in w.EMIT_RULE
guide = w.render_reference(SCHEMA)
assert "player.outfit (free text) — What the player is wearing." in guide
def test_reference_lists_a_free_text_stat_with_no_description():
"""The marker alone is the entry. Dropping the line would leave the model
sending a number for a stat that holds a string."""
schema = {"player": {"holding": {"type": "text", "initial": ""}}}
assert "player.holding (free text)." in w.render_reference(schema)
def test_unknown_paths_rejected_not_fatal():
ws, report = w.apply_delta(fresh(), SCHEMA, {"player.stamina": -5, "bogus": 1}, 2)
reasons = {r["reason"] for r in report["rejected"]}
assert reasons == {"unknown stat", "unknown path"}
assert ws["player"]["hp"] == 100 # untouched
def test_extract_fenced_delta_tolerates_mess():
text = 'You strike.\n\n```state\n{"player.hp": -15, "npc.12.trust": +5,}\n```'
clean, delta = w.extract_delta(text)
assert clean == "You strike."
assert delta == {"player.hp": -15, "npc.12.trust": 5}
def test_extract_no_block():
clean, delta = w.extract_delta("Just prose that ends normally.")
assert delta == {}
assert clean == "Just prose that ends normally."
def test_extract_prose_ending_in_brace_not_eaten():
# A bare object with no dotted keys is not a delta, so the text stays
# as is.
clean, delta = w.extract_delta('He said {this}')
assert delta == {}
assert clean == "He said {this}"
def test_band_label():
d = SCHEMA["player"]["hp"]
assert w.band_label(d, 10) == "very weak"
assert w.band_label(d, 55) == "minor damage"
assert w.band_label(d, 100) == "full health" # inclusive top edge
def test_render_delta_block_roundtrips_through_extract():
# A rendered block re-injected into history must parse back to the same delta,
# so the model sees valid examples of its own emit format.
delta = {"player.hp": -15, "milestones.escaped": True}
block = w.render_delta_block(delta)
assert block.startswith("```state")
clean, parsed = w.extract_delta(f"You flee into the night.\n{block}")
assert clean == "You flee into the night."
assert parsed == delta
def test_render_delta_block_empty_is_blank():
# No change this turn -> nothing appended to history.
assert w.render_delta_block({}) == ""
assert w.render_delta_block(None) == ""