added support for codex provider. fixed bug in claude code export (lines split when shouldn't be)

This commit is contained in:
JesseMarkowitz
2026-08-18 08:01:34 -04:00
parent d083aea135
commit fe5ed341ad
10 changed files with 1435 additions and 40 deletions
+31 -1
View File
@@ -20,7 +20,11 @@ def _write_session(tmp_path, records, project_dir="-home-jesse-myproj", name="ab
proj = tmp_path / project_dir
proj.mkdir(parents=True, exist_ok=True)
f = proj / f"{name}.jsonl"
f.write_text("\n".join(json.dumps(r) for r in records), encoding="utf-8")
# ensure_ascii=False so raw U+0085/U+2028 reach the parser — see
# TestExoticLineBreaks.
f.write_text(
"\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8"
)
return f
@@ -501,3 +505,29 @@ class TestMultiRoot:
monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "cfg"))
roots = resolve_roots()
assert (tmp_path / "cfg" / "projects") in roots
class TestExoticLineBreaks:
"""Same U+0085 hazard as the Codex provider — see its TestExoticLineBreaks."""
# C0 controls (\x0b, \x0c) are excluded: JSON requires them escaped, so they
# never reach the splitter literally. These three do.
@pytest.mark.parametrize("sep", ["\x85", "
", "
"])
def test_record_with_exotic_break_survives(self, tmp_path, sep):
records = [
{
"type": "user",
"cwd": "/home/jesse/myproj",
"timestamp": "2026-05-01T10:00:00.000Z",
"message": {"role": "user", "content": f"before{sep}after"},
},
]
_write_session(tmp_path, records)
prov = ClaudeCodeProvider(projects_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation("abc-123"))
texts = [
b["text"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TEXT
]
assert f"before{sep}after" in texts
+410
View File
@@ -0,0 +1,410 @@
"""Unit tests for the Codex CLI session provider.
Fixtures mirror the real 0.147.0 rollout shape observed on 2026-08-18: dialogue
carried twice (typed ``item_completed`` items plus raw ``response_item``s),
code-mode ``exec`` calls whose input is JavaScript, encrypted reasoning, and
harness-injected user messages that exist only in the raw layer.
"""
import json
import pytest
from src.blocks import (
BLOCK_TYPE_COLLAPSED,
BLOCK_TYPE_TEXT,
BLOCK_TYPE_TOOL_RESULT,
BLOCK_TYPE_TOOL_USE,
COLLAPSED_KIND_HIDDEN_CONTEXT,
)
from src.loss_report import LossReport
from src.providers.codex import CodexProvider, resolve_roots
SESSION_ID = "01a00e3f-a309-74a3-bf32-06c2cd87faa3"
FILENAME = f"rollout-2026-08-17T01-44-06-{SESSION_ID}.jsonl"
def _write_session(tmp_path, records, name=FILENAME, day="2026/08/17"):
day_dir = tmp_path / day
day_dir.mkdir(parents=True, exist_ok=True)
f = day_dir / name
# ensure_ascii=False: Codex (Rust) writes non-ASCII literally, so real
# rollouts contain raw U+0085/U+2028 inside JSON strings. Escaping them here
# would hide exactly the hazard TestExoticLineBreaks exists to catch.
f.write_text(
"\n".join(json.dumps(r, ensure_ascii=False) for r in records), encoding="utf-8"
)
return f
def _item(item_type, ts="2026-08-17T05:44:10.000Z", **fields):
# `item_type`, not `kind` — Extension items carry their own `kind` field.
return {
"timestamp": ts,
"type": "event_msg",
"payload": {"type": "item_completed", "item": {"type": item_type, **fields}},
}
def _exec_call(cmd, fn="exec_command"):
"""A raw code-mode custom_tool_call — its input is JavaScript, not JSON."""
return {
"timestamp": "2026-08-17T05:44:11.000Z",
"type": "response_item",
"payload": {
"type": "custom_tool_call",
"name": "exec",
"call_id": "call_1",
"input": (
f'const r = await tools.{fn}({{"cmd":{json.dumps(cmd)},'
f'"workdir":"/home/jesse/ws","yield_time_ms":30000}});\ntext(r.output);'
),
},
}
def _records(cwd="/home/jesse/ws"):
"""A representative session: meta, harness noise, dialogue, tool traffic."""
return [
{
"timestamp": "2026-08-17T05:49:13.629Z",
"ordinal": 0,
"type": "session_meta",
"payload": {
"session_id": SESSION_ID,
"timestamp": "2026-08-17T05:44:06.666Z",
"cwd": cwd,
"cli_version": "0.147.0",
"source": "cli",
},
},
# Harness plumbing: present in the raw layer only, exactly as 0.147.0
# writes it. Must not become a message.
{
"timestamp": "2026-08-17T05:44:07.000Z",
"type": "response_item",
"payload": {
"type": "message",
"role": "user",
"content": [{"type": "input_text", "text": "# AGENTS.md instructions for /x"}],
},
},
{
"timestamp": "2026-08-17T05:44:08.000Z",
"type": "response_item",
"payload": {
"type": "message",
"role": "developer",
"content": [{"type": "input_text", "text": "<skills_instructions>…"}],
},
},
_item(
"UserMessage",
ts="2026-08-17T05:44:09.000Z",
id="u1",
content=[{"type": "text", "text": "Write the backup guide.", "text_elements": []}],
),
# Encrypted reasoning — nothing recoverable in either layer.
{
"timestamp": "2026-08-17T05:44:09.500Z",
"type": "response_item",
"payload": {"type": "reasoning", "summary": [], "encrypted_content": "gAAAA…"},
},
_item("Reasoning", id="r1", summary_text=[], raw_content=[]),
# AgentMessage content blocks are "Text" (capital T), unlike UserMessage.
_item(
"AgentMessage",
id="a1",
phase="commentary",
content=[{"type": "Text", "text": "I'll inspect the repo first."}],
),
_exec_call("ls -la"),
_item(
"CommandExecution",
id="exec-1",
process_id="123",
command=["/bin/bash", "-lc", "ls -la"],
cwd="file:///home/jesse/ws",
source="unified_exec_startup",
status="completed",
exit_code=0,
stdout="total 4\n",
aggregated_output="total 4\n",
formatted_output="total 4\n",
),
_exec_call("cat missing"),
_item(
"CommandExecution",
id="exec-2",
command=["/bin/bash", "-lc", "cat missing"],
cwd="file:///home/jesse/ws",
status="completed",
exit_code=1,
stderr="No such file\n",
aggregated_output="No such file\n",
),
# An attempt that never produced an item (sandbox failure / abort).
_exec_call("npm test"),
# A `wait` poll: not an attempt, must not inflate the shortfall.
{
"timestamp": "2026-08-17T05:44:12.000Z",
"type": "response_item",
"payload": {"type": "function_call", "name": "wait", "call_id": "call_w"},
},
_item(
"AgentMessage",
ts="2026-08-17T05:45:00.000Z",
id="a2",
phase="final_answer",
content=[{"type": "Text", "text": "Done — the guide is written."}],
),
]
class TestCodexProvider:
def test_scan_lists_session_with_metadata(self, tmp_path):
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path)
convs = prov.list_conversations()
assert len(convs) == 1
assert convs[0]["id"] == SESSION_ID
assert convs[0]["title"] == "Write the backup guide."
assert convs[0]["project"] == "ws"
# session_meta payload timestamp, not the (later) flush timestamp.
assert convs[0]["created_at"] == "2026-08-17T05:44:06.666Z"
def test_ignores_non_rollout_files(self, tmp_path):
_write_session(tmp_path, _records())
(tmp_path / "2026/08/17/notes.jsonl").write_text("{}", encoding="utf-8")
(tmp_path / "2026/08/17/rollout-garbage.jsonl").write_text("{}", encoding="utf-8")
assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1
def test_empty_file_skipped(self, tmp_path):
_write_session(tmp_path, _records())
(tmp_path / "2026/08/16").mkdir(parents=True)
(tmp_path / "2026/08/16" / FILENAME.replace("17T01", "16T01")).write_text("")
assert len(CodexProvider(sessions_dir=tmp_path).list_conversations()) == 1
def test_missing_root_returns_empty(self, tmp_path):
prov = CodexProvider(sessions_dir=tmp_path / "nope")
assert prov.list_conversations() == []
def test_get_conversation_unknown_id_raises(self, tmp_path):
from src.providers.base import ProviderError
prov = CodexProvider(sessions_dir=tmp_path)
with pytest.raises(ProviderError):
prov.get_conversation("does-not-exist")
class TestNormalize:
def _normalized(self, tmp_path, policy="placeholder", records=None):
_write_session(tmp_path, records if records is not None else _records())
prov = CodexProvider(sessions_dir=tmp_path, hidden_content=policy)
prov.list_conversations()
report = LossReport()
return prov.normalize_conversation(prov.get_conversation(SESSION_ID), report), report
def test_dialogue_only_by_default(self, tmp_path):
conv, _ = self._normalized(tmp_path)
roles = [m["role"] for m in conv["messages"]]
texts = [
b["text"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TEXT
]
assert roles[0] == "user"
assert texts == [
"Write the backup guide.",
"I'll inspect the repo first.",
"Done — the guide is written.",
]
def test_harness_injections_never_become_messages(self, tmp_path):
conv, _ = self._normalized(tmp_path)
blob = json.dumps(conv)
assert "AGENTS.md instructions" not in blob
assert "skills_instructions" not in blob
def test_reasoning_is_dropped_and_counted(self, tmp_path):
conv, report = self._normalized(tmp_path)
assert "thinking" not in json.dumps(conv)
assert "reasoning" in report.format_summary()
def test_reasoning_stays_dropped_under_full(self, tmp_path):
# Unlike Claude Code, `full` cannot surface it — it is encrypted at rest.
conv, _ = self._normalized(tmp_path, policy="full")
assert "encrypted" not in json.dumps(conv)
def test_tool_traffic_collapses_with_shortfall(self, tmp_path):
conv, _ = self._normalized(tmp_path)
collapsed = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_COLLAPSED
]
assert len(collapsed) == 1
origin = collapsed[0]["origin"]
# 2 completed of 3 attempted; the `wait` poll is not an attempt.
assert "2 calls: exec_command ×2" in origin
assert "+1 did not complete" in origin
def test_full_policy_emits_decoded_tool_blocks(self, tmp_path):
conv, _ = self._normalized(tmp_path, policy="full")
uses = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TOOL_USE
]
results = [
b for m in conv["messages"] for b in m["blocks"]
# The "incomplete" note is asserted by TestFullPolicyShortfall.
if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") != "incomplete"
]
assert len(uses) == 2 and len(results) == 2
# The command is read from the typed item, not parsed out of the JS.
assert uses[0]["input"]["command"] == "ls -la"
assert uses[0]["input"]["cwd"] == "/home/jesse/ws" # file:// stripped
assert results[0]["is_error"] is False
assert results[1]["is_error"] is True # exit_code 1
def test_message_count_matches(self, tmp_path):
conv, _ = self._normalized(tmp_path)
assert conv["message_count"] == len(conv["messages"])
def test_updated_at_matches_listing(self, tmp_path):
"""Cache staleness compares these two; a mismatch re-exports every run."""
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path)
listed = prov.list_conversations()[0]
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
assert conv["updated_at"] == listed["updated_at"]
def test_context_compaction_is_visible(self, tmp_path):
records = _records() + [_item("ContextCompaction", id="c1")]
conv, report = self._normalized(tmp_path, records=records)
markers = [
b for m in conv["messages"] for b in m["blocks"]
if b.get("kind") == COLLAPSED_KIND_HIDDEN_CONTEXT
]
assert len(markers) == 1
assert "compacted" in markers[0]["origin"]
assert "context_compaction" in report.format_summary()
def test_unknown_item_type_is_reported(self, tmp_path):
records = _records() + [_item("QuantumMessage", id="q1", mystery=True)]
conv, report = self._normalized(tmp_path, records=records)
unknowns = [
b for m in conv["messages"] for b in m["blocks"] if b["type"] == "unknown"
]
assert len(unknowns) == 1
assert unknowns[0]["raw_type"] == "codex.QuantumMessage"
assert "codex.QuantumMessage" in report.format_summary()
def test_extension_labelled_by_kind(self, tmp_path):
records = _records() + [
_item("Extension", id="e1", kind="web.search", query="hsts", results=[]),
]
conv, _ = self._normalized(tmp_path, records=records)
origins = " ".join(
b["origin"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_COLLAPSED
)
assert "web.search ×1" in origins
class TestRepoTags:
def test_title_tagged_with_repos_touched(self, tmp_path):
repo = tmp_path / "ws" / "myrepo"
(repo / ".git").mkdir(parents=True)
records = _records(cwd=str(tmp_path / "ws")) + [
_item(
"FileChange",
id="fc1",
status="completed",
changes={str(repo / "README.md"): {"type": "add", "content": "x"}},
)
]
_write_session(tmp_path, records)
prov = CodexProvider(sessions_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
assert conv["title"].endswith("[myrepo]")
def test_no_tag_when_nothing_touched(self, tmp_path):
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
assert conv["title"] == "Write the backup guide."
class TestResolveRoots:
def test_default_root(self, monkeypatch):
monkeypatch.delenv("CODEX_DIR", raising=False)
monkeypatch.delenv("CODEX_HOME", raising=False)
assert resolve_roots()[0].name == "sessions"
def test_codex_dir_splits_on_pathsep(self, monkeypatch):
monkeypatch.setenv("CODEX_DIR", "/a/sessions:/b/sessions")
monkeypatch.delenv("CODEX_HOME", raising=False)
assert [str(p) for p in resolve_roots()] == ["/a/sessions", "/b/sessions"]
def test_codex_home_appended_and_deduped(self, monkeypatch):
monkeypatch.setenv("CODEX_DIR", "/a/sessions")
monkeypatch.setenv("CODEX_HOME", "/a")
# /a/sessions is already listed — must not appear twice.
assert [str(p) for p in resolve_roots()] == ["/a/sessions"]
class TestExoticLineBreaks:
"""U+0085 (NEL) and friends are legal inside a JSON string.
``str.splitlines()`` breaks on them, shredding one record into unparseable
fragments and losing it silently. Observed 2026-08-18 in a real rollout,
where captured command output contained two NELs.
"""
# C0 controls (\x0b, \x0c) are excluded: JSON requires them escaped, so they
# never reach the splitter literally. These three do.
@pytest.mark.parametrize("sep", ["\x85", "
", "
"])
def test_record_with_exotic_break_survives(self, tmp_path, sep):
records = _records()
records.append(
_item(
"AgentMessage",
ts="2026-08-17T05:46:00.000Z",
id="a3",
phase="final_answer",
content=[{"type": "Text", "text": f"before{sep}after"}],
)
)
_write_session(tmp_path, records)
prov = CodexProvider(sessions_dir=tmp_path)
prov.list_conversations()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID))
texts = [
b["text"] for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TEXT
]
assert f"before{sep}after" in texts
class TestFullPolicyShortfall:
def test_shortfall_note_is_not_a_collapsed_block(self, tmp_path):
"""Under `full`, a collapsed block would advise setting the policy that
is already in force. The shortfall is stated plainly instead."""
_write_session(tmp_path, _records())
prov = CodexProvider(sessions_dir=tmp_path, hidden_content="full")
prov.list_conversations()
report = LossReport()
conv = prov.normalize_conversation(prov.get_conversation(SESSION_ID), report)
collapsed = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_COLLAPSED
]
assert collapsed == []
notes = [
b for m in conv["messages"] for b in m["blocks"]
if b["type"] == BLOCK_TYPE_TOOL_RESULT and b.get("tool_name") == "incomplete"
]
assert len(notes) == 1
assert "1 tool call(s) produced no result" in notes[0]["output"]
assert "tool_call_incomplete" in report.format_summary()