Files

411 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Unit tests for the Codex CLI session provider.
Fixtures mirror the real 0.147.0 rollout shape observed on 2026-08-18: dialogue
carried twice (typed ``item_completed`` items plus raw ``response_item``s),
code-mode ``exec`` calls whose input is JavaScript, encrypted reasoning, and
harness-injected user messages that exist only in the raw layer.
"""
import json
import pytest
from src.blocks import (
BLOCK_TYPE_COLLAPSED,
BLOCK_TYPE_TEXT,
BLOCK_TYPE_TOOL_RESULT,
BLOCK_TYPE_TOOL_USE,
COLLAPSED_KIND_HIDDEN_CONTEXT,
)
from src.loss_report import LossReport
from src.providers.codex import CodexProvider, resolve_roots
SESSION_ID = "01a00e3f-a309-74a3-bf32-06c2cd87faa3"
FILENAME = f"rollout-2026-08-17T01-44-06-{SESSION_ID}.jsonl"
def _write_session(tmp_path, records, name=FILENAME, day="2026/08/17"):
day_dir = tmp_path / day
day_dir.mkdir(parents=True, exist_ok=True)
f = day_dir / name
# ensure_ascii=False: Codex (Rust) writes non-ASCII literally, so real
# rollouts contain raw U+0085/U+2028 inside JSON strings. Escaping them here
# would hide exactly the hazard TestExoticLineBreaks exists to catch.
f.write_text(
"\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8"
)
return f
def _item(item_type, ts="2026-08-17T05:44:10.000Z", **fields):
# `item_type`, not `kind` — Extension items carry their own `kind` field.
return {
"timestamp": ts,
"type": "event_msg",
"payload": {"type": "item_completed", "item": {"type": item_type, **fields}},
}
def _exec_call(cmd, fn="exec_command"):
"""A raw code-mode custom_tool_call — its input is JavaScript, not JSON."""
return {
"timestamp": "2026-08-17T05:44:11.000Z",
"type": "response_item",
"payload": {
"type": "custom_tool_call",
"name": "exec",
"call_id": "call_1",
"input": (
f'const r = await tools.{fn}({{"cmd":{json.dumps(cmd)},'
f'"workdir":"/home/jesse/ws","yield_time_ms":30000}});\ntext(r.output);'
),
},
}
def _records(cwd="/home/jesse/ws"):
"""A representative session: meta, harness noise, dialogue, tool traffic."""
return [
{
"timestamp": "2026-08-17T05:49:13.629Z",
"ordinal": 0,
"type": "session_meta",
"payload": {
"session_id": SESSION_ID,
"timestamp": "2026-08-17T05:44:06.666Z",
"cwd": cwd,
"cli_version": "0.147.0",
"source": "cli",
},
},
# Harness plumbing: present in the raw layer only, exactly as 0.147.0
# writes it. Must not become a message.
{
"timestamp": "2026-08-17T05:44:07.000Z",
"type": "response_item",
"payload": {
"type": "message",
"role": "user",
"content": [{"type": "input_text", "text": "# AGENTS.md instructions for /x"}],
},
},
{
"timestamp": "2026-08-17T05:44:08.000Z",
"type": "response_item",
"payload": {
"type": "message",
"role": "developer",
"content": [{"type": "input_text", "text": "<skills_instructions>…"}],
},
},
_item(
"UserMessage",
ts="2026-08-17T05:44:09.000Z",
id="u1",
content=[{"type": "text", "text": "Write the backup guide.", "text_elements": []}],
),
# Encrypted reasoning — nothing recoverable in either layer.
{
"timestamp": "2026-08-17T05:44:09.500Z",
"type": "response_item",
"payload": {"type": "reasoning", "summary": [], "encrypted_content": "gAAAA…"},
},
_item("Reasoning", id="r1", summary_text=[], raw_content=[]),
# AgentMessage content blocks are "Text" (capital T), unlike UserMessage.
_item(
"AgentMessage",
id="a1",
phase="commentary",
content=[{"type": "Text", "text": "I'll inspect the repo first."}],
),
_exec_call("ls -la"),
_item(
"CommandExecution",
id="exec-1",
process_id="123",
command=["/bin/bash", "-lc", "ls -la"],
cwd="file:///home/jesse/ws",
source="unified_exec_startup",
status="completed",
exit_code=0,
stdout="total 4\n",
aggregated_output="total 4\n",
formatted_output="total 4\n",
),
_exec_call("cat missing"),
_item(
"CommandExecution",
id="exec-2",
command=["/bin/bash", "-lc", "cat missing"],
cwd="file:///home/jesse/ws",
status="completed",
exit_code=1,
stderr="No such file\n",
aggregated_output="No such file\n",
),
# An attempt that never produced an item (sandbox failure / abort).
_exec_call("npm test"),
# A `wait` poll: not an attempt, must not inflate the shortfall.
{
"timestamp": "2026-08-17T05:44:12.000Z",
"type": "response_item",
"payload": {"type": "function_call", "name": "wait", "call_id": "call_w"},
},
_item(
"AgentMessage",
ts="2026-08-17T05:45:00.000Z",
id="a2",
phase="final_answer",
content=[{"type": "Text", "text": "Done — the guide is written."}],
),
]
class TestCodexProvider:
def test_scan_lists_session_with_metadata(self, tmp_path):
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path)
convs = prov.list_conversations()
assert len(convs) == 1
assert convs[0]["id"] == SESSION_ID
assert convs[0]["title"] == "Write the backup guide."
assert convs[0]["project"] == "ws"
# session_meta payload timestamp, not the (later) flush timestamp.
assert convs[0]["created_at"] == "2026-08-17T05:44:06.666Z"
def test_ignores_non_rollout_files(self, tmp_path):
_write_session(tmp_path, _records())
(tmp_path / "2026/08/17/notes.jsonl").write_text("{}", encoding="utf-8")
(tmp_path / "2026/08/17/rollout-garbage.jsonl").write_text("{}", encoding="utf-8")
assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1
def test_empty_file_skipped(self, tmp_path):
_write_session(tmp_path, _records())
(tmp_path / "2026/08/16").mkdir(parents=True)
(tmp_path / "2026/08/16" / FILENAME.replace("17T01", "16T01")).write_text("")
assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1
def test_missing_root_returns_empty(self, tmp_path):
prov = CodexProvider(sessions_dir=tmp_path / "nope")
assert prov.list_conversations() == []
def test_get_conversation_unknown_id_raises(self, tmp_path):
from src.providers.base import ProviderError
prov = CodexProvider(sessions_dir=tmp_path)
with pytest.raises(ProviderError):
prov.get_conversation("does-not-exist")
class TestNormalize:
def _normalized(self, tmp_path, policy="placeholder", records=None):
_write_session(tmp_path, records if records is not None else _records())
prov = CodexProvider(sessions_dir=tmp_path, hidden_content=policy)
prov.list_conversations()
report = LossReport()
return prov.normalize_conversation(prov.get_conversation(SESSION_ID), report), report
def test_dialogue_only_by_default(self, tmp_path):
conv, _ = self._normalized(tmp_path)
roles = [m["role"] for m in conv["messages"]]
texts = [
b["text"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TEXT
]
assert roles[0] == "user"
assert texts == [
"Write the backup guide.",
"I'll inspect the repo first.",
"Done — the guide is written.",
]
def test_harness_injections_never_become_messages(self, tmp_path):
conv, _ = self._normalized(tmp_path)
blob = json.dumps(conv)
assert "AGENTS.md instructions" not in blob
assert "skills_instructions" not in blob
def test_reasoning_is_dropped_and_counted(self, tmp_path):
conv, report = self._normalized(tmp_path)
assert "thinking" not in json.dumps(conv)
assert "reasoning" in report.format_summary()
def test_reasoning_stays_dropped_under_full(self, tmp_path):
# Unlike Claude Code, `full` cannot surface it — it is encrypted at rest.
conv, _ = self._normalized(tmp_path, policy="full")
assert "encrypted" not in json.dumps(conv)
def test_tool_traffic_collapses_with_shortfall(self, tmp_path):
conv, _ = self._normalized(tmp_path)
collapsed = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_COLLAPSED
]
assert len(collapsed) == 1
origin = collapsed[0]["origin"]
# 2 completed of 3 attempted; the `wait` poll is not an attempt.
assert "2 calls: exec_command ×2" in origin
assert "+1 did not complete" in origin
def test_full_policy_emits_decoded_tool_blocks(self, tmp_path):
conv, _ = self._normalized(tmp_path, policy="full")
uses = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TOOL_USE
]
results = [
b for m in conv["messages"] for b in m["blocks"]
# The "incomplete" note is asserted by TestFullPolicyShortfall.
if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") != "incomplete"
]
assert len(uses) == 2 and len(results) == 2
# The command is read from the typed item, not parsed out of the JS.
assert uses[0]["input"]["command"] == "ls -la"
assert uses[0]["input"]["cwd"] == "/home/jesse/ws" # file:// stripped
assert results[0]["is_error"] is False
assert results[1]["is_error"] is True # exit_code 1
def test_message_count_matches(self, tmp_path):
conv, _ = self._normalized(tmp_path)
assert conv["message_count"] == len(conv["messages"])
def test_updated_at_matches_listing(self, tmp_path):
"""Cache staleness compares these two; a mismatch re-exports every run."""
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path)
listed = prov.list_conversations()[0]
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
assert conv["updated_at"] == listed["updated_at"]
def test_context_compaction_is_visible(self, tmp_path):
records = _records() + [_item("ContextCompaction", id="c1")]
conv, report = self._normalized(tmp_path, records=records)
markers = [
b for m in conv["messages"] for b in m["blocks"]
if b.get("kind") == COLLAPSED_KIND_HIDDEN_CONTEXT
]
assert len(markers) == 1
assert "compacted" in markers[0]["origin"]
assert "context_compaction" in report.format_summary()
def test_unknown_item_type_is_reported(self, tmp_path):
records = _records() + [_item("QuantumMessage", id="q1", mystery=True)]
conv, report = self._normalized(tmp_path, records=records)
unknowns = [
b for m in conv["messages"] for b in m["blocks"] if b["type"] == "unknown"
]
assert len(unknowns) == 1
assert unknowns[0]["raw_type"] == "codex.QuantumMessage"
assert "codex.QuantumMessage" in report.format_summary()
def test_extension_labelled_by_kind(self, tmp_path):
records = _records() + [
_item("Extension", id="e1", kind="web.search", query="hsts", results=[]),
]
conv, _ = self._normalized(tmp_path, records=records)
origins = " ".join(
b["origin"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_COLLAPSED
)
assert "web.search ×1" in origins
class TestRepoTags:
def test_title_tagged_with_repos_touched(self, tmp_path):
repo = tmp_path / "ws" / "myrepo"
(repo / ".git").mkdir(parents=True)
records = _records(cwd=str(tmp_path / "ws")) + [
_item(
"FileChange",
id="fc1",
status="completed",
changes={str(repo / "README.md"): {"type": "add", "content": "x"}},
)
]
_write_session(tmp_path, records)
prov = CodexProvider(sessions_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
assert conv["title"].endswith("[myrepo]")
def test_no_tag_when_nothing_touched(self, tmp_path):
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
assert conv["title"] == "Write the backup guide."
class TestResolveRoots:
def test_default_root(self, monkeypatch):
monkeypatch.delenv("CODEX_DIR", raising=False)
monkeypatch.delenv("CODEX_HOME", raising=False)
assert resolve_roots()[0].name == "sessions"
def test_codex_dir_splits_on_pathsep(self, monkeypatch):
monkeypatch.setenv("CODEX_DIR", "/a/sessions:/b/sessions")
monkeypatch.delenv("CODEX_HOME", raising=False)
assert [str(p) for p in resolve_roots()] == ["/a/sessions", "/b/sessions"]
def test_codex_home_appended_and_deduped(self, monkeypatch):
monkeypatch.setenv("CODEX_DIR", "/a/sessions")
monkeypatch.setenv("CODEX_HOME", "/a")
# /a/sessions is already listed — must not appear twice.
assert [str(p) for p in resolve_roots()] == ["/a/sessions"]
class TestExoticLineBreaks:
"""U+0085 (NEL) and friends are legal inside a JSON string.
``str.splitlines()`` breaks on them, shredding one record into unparseable
fragments and losing it silently. Observed 2026-08-18 in a real rollout,
where captured command output contained two NELs.
"""
# C0 controls (\x0b, \x0c) are excluded: JSON requires them escaped, so they
# never reach the splitter literally. These three do.
@pytest.mark.parametrize("sep", ["\x85", "
", "
"])
def test_record_with_exotic_break_survives(self, tmp_path, sep):
records = _records()
records.append(
_item(
"AgentMessage",
ts="2026-08-17T05:46:00.000Z",
id="a3",
phase="final_answer",
content=[{"type": "Text", "text": f"before{sep}after"}],
)
)
_write_session(tmp_path, records)
prov = CodexProvider(sessions_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
texts = [
b["text"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TEXT
]
assert f"before{sep}after" in texts
class TestFullPolicyShortfall:
def test_shortfall_note_is_not_a_collapsed_block(self, tmp_path):
"""Under `full`, a collapsed block would advise setting the policy that
is already in force. The shortfall is stated plainly instead."""
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path, hidden_content="full")
prov.list_conversations()
report = LossReport()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID), report)
collapsed = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_COLLAPSED
]
assert collapsed == []
notes = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") == "incomplete"
]
assert len(notes) == 1
assert "1 tool call(s) produced no result" in notes[0]["output"]
assert "tool_call_incomplete" in report.format_summary()