"""Unit tests for the Codex CLI session provider. Fixtures mirror the real 0.147.0 rollout shape observed on 2026-08-18: dialogue carried twice (typed ``item_completed`` items plus raw ``response_item``s), code-mode ``exec`` calls whose input is JavaScript, encrypted reasoning, and harness-injected user messages that exist only in the raw layer. """ import json import pytest from src.blocks import ( BLOCK_TYPE_COLLAPSED, BLOCK_TYPE_TEXT, BLOCK_TYPE_TOOL_RESULT, BLOCK_TYPE_TOOL_USE, COLLAPSED_KIND_HIDDEN_CONTEXT, ) from src.loss_report import LossReport from src.providers.codex import CodexProvider, resolve_roots SESSION_ID = "01a00e3f-a309-74a3-bf32-06c2cd87faa3" FILENAME = f"rollout-2026-08-17T01-44-06-{SESSION_ID}.jsonl" def _write_session(tmp_path, records, name=FILENAME, day="2026/08/17"): day_dir = tmp_path / day day_dir.mkdir(parents=True, exist_ok=True) f = day_dir / name # ensure_ascii=False: Codex (Rust) writes non-ASCII literally, so real # rollouts contain raw U+0085/U+2028 inside JSON strings. Escaping them here # would hide exactly the hazard TestExoticLineBreaks exists to catch. f.write_text( "\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8" ) return f def _item(item_type, ts="2026-08-17T05:44:10.000Z", **fields): # `item_type`, not `kind` — Extension items carry their own `kind` field. return { "timestamp": ts, "type": "event_msg", "payload": {"type": "item_completed", "item": {"type": item_type, **fields}}, } def _exec_call(cmd, fn="exec_command"): """A raw code-mode custom_tool_call — its input is JavaScript, not JSON.""" return { "timestamp": "2026-08-17T05:44:11.000Z", "type": "response_item", "payload": { "type": "custom_tool_call", "name": "exec", "call_id": "call_1", "input": ( f'const r = await tools.{fn}({{"cmd":{json.dumps(cmd)},' f'"workdir":"/home/jesse/ws","yield_time_ms":30000}});\ntext(r.output);' ), }, } def _records(cwd="/home/jesse/ws"): """A representative session: meta, harness noise, dialogue, tool traffic.""" return [ { "timestamp": "2026-08-17T05:49:13.629Z", "ordinal": 0, "type": "session_meta", "payload": { "session_id": SESSION_ID, "timestamp": "2026-08-17T05:44:06.666Z", "cwd": cwd, "cli_version": "0.147.0", "source": "cli", }, }, # Harness plumbing: present in the raw layer only, exactly as 0.147.0 # writes it. Must not become a message. { "timestamp": "2026-08-17T05:44:07.000Z", "type": "response_item", "payload": { "type": "message", "role": "user", "content": [{"type": "input_text", "text": "# AGENTS.md instructions for /x"}], }, }, { "timestamp": "2026-08-17T05:44:08.000Z", "type": "response_item", "payload": { "type": "message", "role": "developer", "content": [{"type": "input_text", "text": "…"}], }, }, _item( "UserMessage", ts="2026-08-17T05:44:09.000Z", id="u1", content=[{"type": "text", "text": "Write the backup guide.", "text_elements": []}], ), # Encrypted reasoning — nothing recoverable in either layer. { "timestamp": "2026-08-17T05:44:09.500Z", "type": "response_item", "payload": {"type": "reasoning", "summary": [], "encrypted_content": "gAAAA…"}, }, _item("Reasoning", id="r1", summary_text=[], raw_content=[]), # AgentMessage content blocks are "Text" (capital T), unlike UserMessage. _item( "AgentMessage", id="a1", phase="commentary", content=[{"type": "Text", "text": "I'll inspect the repo first."}], ), _exec_call("ls -la"), _item( "CommandExecution", id="exec-1", process_id="123", command=["/bin/bash", "-lc", "ls -la"], cwd="file:///home/jesse/ws", source="unified_exec_startup", status="completed", exit_code=0, stdout="total 4\n", aggregated_output="total 4\n", formatted_output="total 4\n", ), _exec_call("cat missing"), _item( "CommandExecution", id="exec-2", command=["/bin/bash", "-lc", "cat missing"], cwd="file:///home/jesse/ws", status="completed", exit_code=1, stderr="No such file\n", aggregated_output="No such file\n", ), # An attempt that never produced an item (sandbox failure / abort). _exec_call("npm test"), # A `wait` poll: not an attempt, must not inflate the shortfall. { "timestamp": "2026-08-17T05:44:12.000Z", "type": "response_item", "payload": {"type": "function_call", "name": "wait", "call_id": "call_w"}, }, _item( "AgentMessage", ts="2026-08-17T05:45:00.000Z", id="a2", phase="final_answer", content=[{"type": "Text", "text": "Done — the guide is written."}], ), ] class TestCodexProvider: def test_scan_lists_session_with_metadata(self, tmp_path): _write_session(tmp_path, _records()) prov = CodexProvider(sessions_dir=tmp_path) convs = prov.list_conversations() assert len(convs) == 1 assert convs[0]["id"] == SESSION_ID assert convs[0]["title"] == "Write the backup guide." assert convs[0]["project"] == "ws" # session_meta payload timestamp, not the (later) flush timestamp. assert convs[0]["created_at"] == "2026-08-17T05:44:06.666Z" def test_ignores_non_rollout_files(self, tmp_path): _write_session(tmp_path, _records()) (tmp_path / "2026/08/17/notes.jsonl").write_text("{}", encoding="utf-8") (tmp_path / "2026/08/17/rollout-garbage.jsonl").write_text("{}", encoding="utf-8") assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1 def test_empty_file_skipped(self, tmp_path): _write_session(tmp_path, _records()) (tmp_path / "2026/08/16").mkdir(parents=True) (tmp_path / "2026/08/16" / FILENAME.replace("17T01", "16T01")).write_text("") assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1 def test_missing_root_returns_empty(self, tmp_path): prov = CodexProvider(sessions_dir=tmp_path / "nope") assert prov.list_conversations() == [] def test_get_conversation_unknown_id_raises(self, tmp_path): from src.providers.base import ProviderError prov = CodexProvider(sessions_dir=tmp_path) with pytest.raises(ProviderError): prov.get_conversation("does-not-exist") class TestNormalize: def _normalized(self, tmp_path, policy="placeholder", records=None): _write_session(tmp_path, records if records is not None else _records()) prov = CodexProvider(sessions_dir=tmp_path, hidden_content=policy) prov.list_conversations() report = LossReport() return prov.normalize_conversation(prov.get_conversation(SESSION_ID), report), report def test_dialogue_only_by_default(self, tmp_path): conv, _ = self._normalized(tmp_path) roles = [m["role"] for m in conv["messages"]] texts = [ b["text"] for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_TEXT ] assert roles[0] == "user" assert texts == [ "Write the backup guide.", "I'll inspect the repo first.", "Done — the guide is written.", ] def test_harness_injections_never_become_messages(self, tmp_path): conv, _ = self._normalized(tmp_path) blob = json.dumps(conv) assert "AGENTS.md instructions" not in blob assert "skills_instructions" not in blob def test_reasoning_is_dropped_and_counted(self, tmp_path): conv, report = self._normalized(tmp_path) assert "thinking" not in json.dumps(conv) assert "reasoning" in report.format_summary() def test_reasoning_stays_dropped_under_full(self, tmp_path): # Unlike Claude Code, `full` cannot surface it — it is encrypted at rest. conv, _ = self._normalized(tmp_path, policy="full") assert "encrypted" not in json.dumps(conv) def test_tool_traffic_collapses_with_shortfall(self, tmp_path): conv, _ = self._normalized(tmp_path) collapsed = [ b for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_COLLAPSED ] assert len(collapsed) == 1 origin = collapsed[0]["origin"] # 2 completed of 3 attempted; the `wait` poll is not an attempt. assert "2 calls: exec_command ×2" in origin assert "+1 did not complete" in origin def test_full_policy_emits_decoded_tool_blocks(self, tmp_path): conv, _ = self._normalized(tmp_path, policy="full") uses = [ b for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_TOOL_USE ] results = [ b for m in conv["messages"] for b in m["blocks"] # The "incomplete" note is asserted by TestFullPolicyShortfall. if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") != "incomplete" ] assert len(uses) == 2 and len(results) == 2 # The command is read from the typed item, not parsed out of the JS. assert uses[0]["input"]["command"] == "ls -la" assert uses[0]["input"]["cwd"] == "/home/jesse/ws" # file:// stripped assert results[0]["is_error"] is False assert results[1]["is_error"] is True # exit_code 1 def test_message_count_matches(self, tmp_path): conv, _ = self._normalized(tmp_path) assert conv["message_count"] == len(conv["messages"]) def test_updated_at_matches_listing(self, tmp_path): """Cache staleness compares these two; a mismatch re-exports every run.""" _write_session(tmp_path, _records()) prov = CodexProvider(sessions_dir=tmp_path) listed = prov.list_conversations()[0] conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID)) assert conv["updated_at"] == listed["updated_at"] def test_context_compaction_is_visible(self, tmp_path): records = _records() + [_item("ContextCompaction", id="c1")] conv, report = self._normalized(tmp_path, records=records) markers = [ b for m in conv["messages"] for b in m["blocks"] if b.get("kind") == COLLAPSED_KIND_HIDDEN_CONTEXT ] assert len(markers) == 1 assert "compacted" in markers[0]["origin"] assert "context_compaction" in report.format_summary() def test_unknown_item_type_is_reported(self, tmp_path): records = _records() + [_item("QuantumMessage", id="q1", mystery=True)] conv, report = self._normalized(tmp_path, records=records) unknowns = [ b for m in conv["messages"] for b in m["blocks"] if b["type"] == "unknown" ] assert len(unknowns) == 1 assert unknowns[0]["raw_type"] == "codex.QuantumMessage" assert "codex.QuantumMessage" in report.format_summary() def test_extension_labelled_by_kind(self, tmp_path): records = _records() + [ _item("Extension", id="e1", kind="web.search", query="hsts", results=[]), ] conv, _ = self._normalized(tmp_path, records=records) origins = " ".join( b["origin"] for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_COLLAPSED ) assert "web.search ×1" in origins class TestRepoTags: def test_title_tagged_with_repos_touched(self, tmp_path): repo = tmp_path / "ws" / "myrepo" (repo / ".git").mkdir(parents=True) records = _records(cwd=str(tmp_path / "ws")) + [ _item( "FileChange", id="fc1", status="completed", changes={str(repo / "README.md"): {"type": "add", "content": "x"}}, ) ] _write_session(tmp_path, records) prov = CodexProvider(sessions_dir=tmp_path) prov.list_conversations() conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID)) assert conv["title"].endswith("[myrepo]") def test_no_tag_when_nothing_touched(self, tmp_path): _write_session(tmp_path, _records()) prov = CodexProvider(sessions_dir=tmp_path) prov.list_conversations() conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID)) assert conv["title"] == "Write the backup guide." class TestResolveRoots: def test_default_root(self, monkeypatch): monkeypatch.delenv("CODEX_DIR", raising=False) monkeypatch.delenv("CODEX_HOME", raising=False) assert resolve_roots()[0].name == "sessions" def test_codex_dir_splits_on_pathsep(self, monkeypatch): monkeypatch.setenv("CODEX_DIR", "/a/sessions:/b/sessions") monkeypatch.delenv("CODEX_HOME", raising=False) assert [str(p) for p in resolve_roots()] == ["/a/sessions", "/b/sessions"] def test_codex_home_appended_and_deduped(self, monkeypatch): monkeypatch.setenv("CODEX_DIR", "/a/sessions") monkeypatch.setenv("CODEX_HOME", "/a") # /a/sessions is already listed — must not appear twice. assert [str(p) for p in resolve_roots()] == ["/a/sessions"] class TestExoticLineBreaks: """U+0085 (NEL) and friends are legal inside a JSON string. ``str.splitlines()`` breaks on them, shredding one record into unparseable fragments and losing it silently. Observed 2026-08-18 in a real rollout, where captured command output contained two NELs. """ # C0 controls (\x0b, \x0c) are excluded: JSON requires them escaped, so they # never reach the splitter literally. These three do. @pytest.mark.parametrize("sep", ["\x85", "
", "
"]) def test_record_with_exotic_break_survives(self, tmp_path, sep): records = _records() records.append( _item( "AgentMessage", ts="2026-08-17T05:46:00.000Z", id="a3", phase="final_answer", content=[{"type": "Text", "text": f"before{sep}after"}], ) ) _write_session(tmp_path, records) prov = CodexProvider(sessions_dir=tmp_path) prov.list_conversations() conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID)) texts = [ b["text"] for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_TEXT ] assert f"before{sep}after" in texts class TestFullPolicyShortfall: def test_shortfall_note_is_not_a_collapsed_block(self, tmp_path): """Under `full`, a collapsed block would advise setting the policy that is already in force. The shortfall is stated plainly instead.""" _write_session(tmp_path, _records()) prov = CodexProvider(sessions_dir=tmp_path, hidden_content="full") prov.list_conversations() report = LossReport() conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID), report) collapsed = [ b for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_COLLAPSED ] assert collapsed == [] notes = [ b for m in conv["messages"] for b in m["blocks"] if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") == "incomplete" ] assert len(notes) == 1 assert "1 tool call(s) produced no result" in notes[0]["output"] assert "tool_call_incomplete" in report.format_summary()