411 lines
16 KiB
Python
411 lines
16 KiB
Python
"""Unit tests for the Codex CLI session provider.
|
||
|
||
Fixtures mirror the real 0.147.0 rollout shape observed on 2026-08-18: dialogue
|
||
carried twice (typed ``item_completed`` items plus raw ``response_item``s),
|
||
code-mode ``exec`` calls whose input is JavaScript, encrypted reasoning, and
|
||
harness-injected user messages that exist only in the raw layer.
|
||
"""
|
||
|
||
import json
|
||
|
||
import pytest
|
||
|
||
from src.blocks import (
|
||
BLOCK_TYPE_COLLAPSED,
|
||
BLOCK_TYPE_TEXT,
|
||
BLOCK_TYPE_TOOL_RESULT,
|
||
BLOCK_TYPE_TOOL_USE,
|
||
COLLAPSED_KIND_HIDDEN_CONTEXT,
|
||
)
|
||
from src.loss_report import LossReport
|
||
from src.providers.codex import CodexProvider, resolve_roots
|
||
|
||
SESSION_ID = "01a00e3f-a309-74a3-bf32-06c2cd87faa3"
|
||
FILENAME = f"rollout-2026-08-17T01-44-06-{SESSION_ID}.jsonl"
|
||
|
||
|
||
def _write_session(tmp_path, records, name=FILENAME, day="2026/08/17"):
|
||
day_dir = tmp_path / day
|
||
day_dir.mkdir(parents=True, exist_ok=True)
|
||
f = day_dir / name
|
||
# ensure_ascii=False: Codex (Rust) writes non-ASCII literally, so real
|
||
# rollouts contain raw U+0085/U+2028 inside JSON strings. Escaping them here
|
||
# would hide exactly the hazard TestExoticLineBreaks exists to catch.
|
||
f.write_text(
|
||
"\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8"
|
||
)
|
||
return f
|
||
|
||
|
||
def _item(item_type, ts="2026-08-17T05:44:10.000Z", **fields):
|
||
# `item_type`, not `kind` — Extension items carry their own `kind` field.
|
||
return {
|
||
"timestamp": ts,
|
||
"type": "event_msg",
|
||
"payload": {"type": "item_completed", "item": {"type": item_type, **fields}},
|
||
}
|
||
|
||
|
||
def _exec_call(cmd, fn="exec_command"):
|
||
"""A raw code-mode custom_tool_call — its input is JavaScript, not JSON."""
|
||
return {
|
||
"timestamp": "2026-08-17T05:44:11.000Z",
|
||
"type": "response_item",
|
||
"payload": {
|
||
"type": "custom_tool_call",
|
||
"name": "exec",
|
||
"call_id": "call_1",
|
||
"input": (
|
||
f'const r = await tools.{fn}({{"cmd":{json.dumps(cmd)},'
|
||
f'"workdir":"/home/jesse/ws","yield_time_ms":30000}});\ntext(r.output);'
|
||
),
|
||
},
|
||
}
|
||
|
||
|
||
def _records(cwd="/home/jesse/ws"):
|
||
"""A representative session: meta, harness noise, dialogue, tool traffic."""
|
||
return [
|
||
{
|
||
"timestamp": "2026-08-17T05:49:13.629Z",
|
||
"ordinal": 0,
|
||
"type": "session_meta",
|
||
"payload": {
|
||
"session_id": SESSION_ID,
|
||
"timestamp": "2026-08-17T05:44:06.666Z",
|
||
"cwd": cwd,
|
||
"cli_version": "0.147.0",
|
||
"source": "cli",
|
||
},
|
||
},
|
||
# Harness plumbing: present in the raw layer only, exactly as 0.147.0
|
||
# writes it. Must not become a message.
|
||
{
|
||
"timestamp": "2026-08-17T05:44:07.000Z",
|
||
"type": "response_item",
|
||
"payload": {
|
||
"type": "message",
|
||
"role": "user",
|
||
"content": [{"type": "input_text", "text": "# AGENTS.md instructions for /x"}],
|
||
},
|
||
},
|
||
{
|
||
"timestamp": "2026-08-17T05:44:08.000Z",
|
||
"type": "response_item",
|
||
"payload": {
|
||
"type": "message",
|
||
"role": "developer",
|
||
"content": [{"type": "input_text", "text": "<skills_instructions>…"}],
|
||
},
|
||
},
|
||
_item(
|
||
"UserMessage",
|
||
ts="2026-08-17T05:44:09.000Z",
|
||
id="u1",
|
||
content=[{"type": "text", "text": "Write the backup guide.", "text_elements": []}],
|
||
),
|
||
# Encrypted reasoning — nothing recoverable in either layer.
|
||
{
|
||
"timestamp": "2026-08-17T05:44:09.500Z",
|
||
"type": "response_item",
|
||
"payload": {"type": "reasoning", "summary": [], "encrypted_content": "gAAAA…"},
|
||
},
|
||
_item("Reasoning", id="r1", summary_text=[], raw_content=[]),
|
||
# AgentMessage content blocks are "Text" (capital T), unlike UserMessage.
|
||
_item(
|
||
"AgentMessage",
|
||
id="a1",
|
||
phase="commentary",
|
||
content=[{"type": "Text", "text": "I'll inspect the repo first."}],
|
||
),
|
||
_exec_call("ls -la"),
|
||
_item(
|
||
"CommandExecution",
|
||
id="exec-1",
|
||
process_id="123",
|
||
command=["/bin/bash", "-lc", "ls -la"],
|
||
cwd="file:///home/jesse/ws",
|
||
source="unified_exec_startup",
|
||
status="completed",
|
||
exit_code=0,
|
||
stdout="total 4\n",
|
||
aggregated_output="total 4\n",
|
||
formatted_output="total 4\n",
|
||
),
|
||
_exec_call("cat missing"),
|
||
_item(
|
||
"CommandExecution",
|
||
id="exec-2",
|
||
command=["/bin/bash", "-lc", "cat missing"],
|
||
cwd="file:///home/jesse/ws",
|
||
status="completed",
|
||
exit_code=1,
|
||
stderr="No such file\n",
|
||
aggregated_output="No such file\n",
|
||
),
|
||
# An attempt that never produced an item (sandbox failure / abort).
|
||
_exec_call("npm test"),
|
||
# A `wait` poll: not an attempt, must not inflate the shortfall.
|
||
{
|
||
"timestamp": "2026-08-17T05:44:12.000Z",
|
||
"type": "response_item",
|
||
"payload": {"type": "function_call", "name": "wait", "call_id": "call_w"},
|
||
},
|
||
_item(
|
||
"AgentMessage",
|
||
ts="2026-08-17T05:45:00.000Z",
|
||
id="a2",
|
||
phase="final_answer",
|
||
content=[{"type": "Text", "text": "Done — the guide is written."}],
|
||
),
|
||
]
|
||
|
||
|
||
class TestCodexProvider:
|
||
def test_scan_lists_session_with_metadata(self, tmp_path):
|
||
_write_session(tmp_path, _records())
|
||
prov = CodexProvider(sessions_dir=tmp_path)
|
||
convs = prov.list_conversations()
|
||
assert len(convs) == 1
|
||
assert convs[0]["id"] == SESSION_ID
|
||
assert convs[0]["title"] == "Write the backup guide."
|
||
assert convs[0]["project"] == "ws"
|
||
# session_meta payload timestamp, not the (later) flush timestamp.
|
||
assert convs[0]["created_at"] == "2026-08-17T05:44:06.666Z"
|
||
|
||
def test_ignores_non_rollout_files(self, tmp_path):
|
||
_write_session(tmp_path, _records())
|
||
(tmp_path / "2026/08/17/notes.jsonl").write_text("{}", encoding="utf-8")
|
||
(tmp_path / "2026/08/17/rollout-garbage.jsonl").write_text("{}", encoding="utf-8")
|
||
assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1
|
||
|
||
def test_empty_file_skipped(self, tmp_path):
|
||
_write_session(tmp_path, _records())
|
||
(tmp_path / "2026/08/16").mkdir(parents=True)
|
||
(tmp_path / "2026/08/16" / FILENAME.replace("17T01", "16T01")).write_text("")
|
||
assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1
|
||
|
||
def test_missing_root_returns_empty(self, tmp_path):
|
||
prov = CodexProvider(sessions_dir=tmp_path / "nope")
|
||
assert prov.list_conversations() == []
|
||
|
||
def test_get_conversation_unknown_id_raises(self, tmp_path):
|
||
from src.providers.base import ProviderError
|
||
|
||
prov = CodexProvider(sessions_dir=tmp_path)
|
||
with pytest.raises(ProviderError):
|
||
prov.get_conversation("does-not-exist")
|
||
|
||
|
||
class TestNormalize:
|
||
def _normalized(self, tmp_path, policy="placeholder", records=None):
|
||
_write_session(tmp_path, records if records is not None else _records())
|
||
prov = CodexProvider(sessions_dir=tmp_path, hidden_content=policy)
|
||
prov.list_conversations()
|
||
report = LossReport()
|
||
return prov.normalize_conversation(prov.get_conversation(SESSION_ID), report), report
|
||
|
||
def test_dialogue_only_by_default(self, tmp_path):
|
||
conv, _ = self._normalized(tmp_path)
|
||
roles = [m["role"] for m in conv["messages"]]
|
||
texts = [
|
||
b["text"] for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_TEXT
|
||
]
|
||
assert roles[0] == "user"
|
||
assert texts == [
|
||
"Write the backup guide.",
|
||
"I'll inspect the repo first.",
|
||
"Done — the guide is written.",
|
||
]
|
||
|
||
def test_harness_injections_never_become_messages(self, tmp_path):
|
||
conv, _ = self._normalized(tmp_path)
|
||
blob = json.dumps(conv)
|
||
assert "AGENTS.md instructions" not in blob
|
||
assert "skills_instructions" not in blob
|
||
|
||
def test_reasoning_is_dropped_and_counted(self, tmp_path):
|
||
conv, report = self._normalized(tmp_path)
|
||
assert "thinking" not in json.dumps(conv)
|
||
assert "reasoning" in report.format_summary()
|
||
|
||
def test_reasoning_stays_dropped_under_full(self, tmp_path):
|
||
# Unlike Claude Code, `full` cannot surface it — it is encrypted at rest.
|
||
conv, _ = self._normalized(tmp_path, policy="full")
|
||
assert "encrypted" not in json.dumps(conv)
|
||
|
||
def test_tool_traffic_collapses_with_shortfall(self, tmp_path):
|
||
conv, _ = self._normalized(tmp_path)
|
||
collapsed = [
|
||
b for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_COLLAPSED
|
||
]
|
||
assert len(collapsed) == 1
|
||
origin = collapsed[0]["origin"]
|
||
# 2 completed of 3 attempted; the `wait` poll is not an attempt.
|
||
assert "2 calls: exec_command ×2" in origin
|
||
assert "+1 did not complete" in origin
|
||
|
||
def test_full_policy_emits_decoded_tool_blocks(self, tmp_path):
|
||
conv, _ = self._normalized(tmp_path, policy="full")
|
||
uses = [
|
||
b for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_TOOL_USE
|
||
]
|
||
results = [
|
||
b for m in conv["messages"] for b in m["blocks"]
|
||
# The "incomplete" note is asserted by TestFullPolicyShortfall.
|
||
if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") != "incomplete"
|
||
]
|
||
assert len(uses) == 2 and len(results) == 2
|
||
# The command is read from the typed item, not parsed out of the JS.
|
||
assert uses[0]["input"]["command"] == "ls -la"
|
||
assert uses[0]["input"]["cwd"] == "/home/jesse/ws" # file:// stripped
|
||
assert results[0]["is_error"] is False
|
||
assert results[1]["is_error"] is True # exit_code 1
|
||
|
||
def test_message_count_matches(self, tmp_path):
|
||
conv, _ = self._normalized(tmp_path)
|
||
assert conv["message_count"] == len(conv["messages"])
|
||
|
||
def test_updated_at_matches_listing(self, tmp_path):
|
||
"""Cache staleness compares these two; a mismatch re-exports every run."""
|
||
_write_session(tmp_path, _records())
|
||
prov = CodexProvider(sessions_dir=tmp_path)
|
||
listed = prov.list_conversations()[0]
|
||
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
|
||
assert conv["updated_at"] == listed["updated_at"]
|
||
|
||
def test_context_compaction_is_visible(self, tmp_path):
|
||
records = _records() + [_item("ContextCompaction", id="c1")]
|
||
conv, report = self._normalized(tmp_path, records=records)
|
||
markers = [
|
||
b for m in conv["messages"] for b in m["blocks"]
|
||
if b.get("kind") == COLLAPSED_KIND_HIDDEN_CONTEXT
|
||
]
|
||
assert len(markers) == 1
|
||
assert "compacted" in markers[0]["origin"]
|
||
assert "context_compaction" in report.format_summary()
|
||
|
||
def test_unknown_item_type_is_reported(self, tmp_path):
|
||
records = _records() + [_item("QuantumMessage", id="q1", mystery=True)]
|
||
conv, report = self._normalized(tmp_path, records=records)
|
||
unknowns = [
|
||
b for m in conv["messages"] for b in m["blocks"] if b["type"] == "unknown"
|
||
]
|
||
assert len(unknowns) == 1
|
||
assert unknowns[0]["raw_type"] == "codex.QuantumMessage"
|
||
assert "codex.QuantumMessage" in report.format_summary()
|
||
|
||
def test_extension_labelled_by_kind(self, tmp_path):
|
||
records = _records() + [
|
||
_item("Extension", id="e1", kind="web.search", query="hsts", results=[]),
|
||
]
|
||
conv, _ = self._normalized(tmp_path, records=records)
|
||
origins = " ".join(
|
||
b["origin"] for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_COLLAPSED
|
||
)
|
||
assert "web.search ×1" in origins
|
||
|
||
|
||
class TestRepoTags:
|
||
def test_title_tagged_with_repos_touched(self, tmp_path):
|
||
repo = tmp_path / "ws" / "myrepo"
|
||
(repo / ".git").mkdir(parents=True)
|
||
records = _records(cwd=str(tmp_path / "ws")) + [
|
||
_item(
|
||
"FileChange",
|
||
id="fc1",
|
||
status="completed",
|
||
changes={str(repo / "README.md"): {"type": "add", "content": "x"}},
|
||
)
|
||
]
|
||
_write_session(tmp_path, records)
|
||
prov = CodexProvider(sessions_dir=tmp_path)
|
||
prov.list_conversations()
|
||
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
|
||
assert conv["title"].endswith("[myrepo]")
|
||
|
||
def test_no_tag_when_nothing_touched(self, tmp_path):
|
||
_write_session(tmp_path, _records())
|
||
prov = CodexProvider(sessions_dir=tmp_path)
|
||
prov.list_conversations()
|
||
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
|
||
assert conv["title"] == "Write the backup guide."
|
||
|
||
|
||
class TestResolveRoots:
|
||
def test_default_root(self, monkeypatch):
|
||
monkeypatch.delenv("CODEX_DIR", raising=False)
|
||
monkeypatch.delenv("CODEX_HOME", raising=False)
|
||
assert resolve_roots()[0].name == "sessions"
|
||
|
||
def test_codex_dir_splits_on_pathsep(self, monkeypatch):
|
||
monkeypatch.setenv("CODEX_DIR", "/a/sessions:/b/sessions")
|
||
monkeypatch.delenv("CODEX_HOME", raising=False)
|
||
assert [str(p) for p in resolve_roots()] == ["/a/sessions", "/b/sessions"]
|
||
|
||
def test_codex_home_appended_and_deduped(self, monkeypatch):
|
||
monkeypatch.setenv("CODEX_DIR", "/a/sessions")
|
||
monkeypatch.setenv("CODEX_HOME", "/a")
|
||
# /a/sessions is already listed — must not appear twice.
|
||
assert [str(p) for p in resolve_roots()] == ["/a/sessions"]
|
||
|
||
|
||
class TestExoticLineBreaks:
|
||
"""U+0085 (NEL) and friends are legal inside a JSON string.
|
||
|
||
``str.splitlines()`` breaks on them, shredding one record into unparseable
|
||
fragments and losing it silently. Observed 2026-08-18 in a real rollout,
|
||
where captured command output contained two NELs.
|
||
"""
|
||
|
||
# C0 controls (\x0b, \x0c) are excluded: JSON requires them escaped, so they
|
||
# never reach the splitter literally. These three do.
|
||
@pytest.mark.parametrize("sep", ["\x85", "
", "
"])
|
||
def test_record_with_exotic_break_survives(self, tmp_path, sep):
|
||
records = _records()
|
||
records.append(
|
||
_item(
|
||
"AgentMessage",
|
||
ts="2026-08-17T05:46:00.000Z",
|
||
id="a3",
|
||
phase="final_answer",
|
||
content=[{"type": "Text", "text": f"before{sep}after"}],
|
||
)
|
||
)
|
||
_write_session(tmp_path, records)
|
||
prov = CodexProvider(sessions_dir=tmp_path)
|
||
prov.list_conversations()
|
||
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
|
||
texts = [
|
||
b["text"] for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_TEXT
|
||
]
|
||
assert f"before{sep}after" in texts
|
||
|
||
|
||
class TestFullPolicyShortfall:
|
||
def test_shortfall_note_is_not_a_collapsed_block(self, tmp_path):
|
||
"""Under `full`, a collapsed block would advise setting the policy that
|
||
is already in force. The shortfall is stated plainly instead."""
|
||
_write_session(tmp_path, _records())
|
||
prov = CodexProvider(sessions_dir=tmp_path, hidden_content="full")
|
||
prov.list_conversations()
|
||
report = LossReport()
|
||
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID), report)
|
||
collapsed = [
|
||
b for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_COLLAPSED
|
||
]
|
||
assert collapsed == []
|
||
notes = [
|
||
b for m in conv["messages"] for b in m["blocks"]
|
||
if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") == "incomplete"
|
||
]
|
||
assert len(notes) == 1
|
||
assert "1 tool call(s) produced no result" in notes[0]["output"]
|
||
assert "tool_call_incomplete" in report.format_summary()
|