From fe5ed341ad1814fcefa6ef41837f9d8fc1c2dce9 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Tue, 18 Aug 2026 08:01:34 -0400 Subject: [PATCH] added support for codex provider. fixed bug in claude code export (lines split when shouldn't be) --- .env.example | 10 + CHANGELOG.md | 11 + README.md | 23 +- src/joplin.py | 4 + src/main.py | 25 +- src/providers/claude_code.py | 40 +- src/providers/codex.py | 881 +++++++++++++++++++++++++++++++++++ src/utils.py | 39 ++ tests/test_claude_code.py | 32 +- tests/test_codex.py | 410 ++++++++++++++++ 10 files changed, 1435 insertions(+), 40 deletions(-) create mode 100644 src/providers/codex.py create mode 100644 tests/test_codex.py diff --git a/.env.example b/.env.example index 8a11b24..300d2c9 100644 --- a/.env.example +++ b/.env.example @@ -38,6 +38,16 @@ CLAUDE_SESSION_KEY= # list their names here (comma-separated). #CLAUDE_CODE_REPO_TAG_IGNORE=some-repo,another-repo +# --- Codex (local agent sessions) --- +# The codex provider reads local Codex CLI rollout files. By default it scans +# ~/.codex/sessions/ (plus $CODEX_HOME/sessions when CODEX_HOME is set). +# To scan additional roots, set a ':'-separated list of sessions roots. +#CODEX_DIR=~/.codex/sessions:/mnt/backup/laptop/.codex/sessions +# +# As with Claude Code, session titles are tagged with the git repos they +# touched. To never tag specific repos, list their names here (comma-separated). +#CODEX_REPO_TAG_IGNORE=some-repo,another-repo + # --- Output --- # Where exported Markdown files are written (default: ./exports) EXPORT_DIR=./exports diff --git a/CHANGELOG.md b/CHANGELOG.md index 45d560a..f2cfef4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,12 +6,23 @@ Format follows [Keep a Changelog](https://keepachangelog.com/en/1.0.0/). ## [Unreleased] ### Fixed +- **A single U+0085 in a transcript silently dropped a whole record.** Both local providers split session files with `str.splitlines()`, which breaks not just on `\n` but on U+0085 (NEL), U+2028 and U+2029 — all of which are legal *inside* a JSON string and are written literally by Codex (Rust does not escape non-ASCII). One NEL in captured command output shredded one record into unparseable fragments; the parser logged "skipped 3 unparseable line(s)" and lost the record. Found while exporting a real rollout. Both providers now split on `\n` only, and both have regression tests that write their fixtures with `ensure_ascii=False` — with `json.dumps`' default the hazardous characters are escaped and the bug cannot reproduce. - **Deleted uploads are no longer reported as permission errors.** ChatGPT's `/backend-api/files/{id}/download` answers a *missing* asset with `403 {"detail":"Forbidden"}`, which reads like an auth failure and isn't one. Measured live 2026-08-17 across 18 such assets: every one returned `404 {"detail":"File not found"}` on `/files/{id}`, while assets that downloaded fine returned 200 on both in the same session, and `ChatGPT-Account-Id` made no difference. A 403 is now confirmed against the metadata endpoint before being reported (one extra request on the failure path only, none on success) and a confirmed-missing asset is logged as gone and counted as `expired-or-missing`. A 403 on an asset that *does* still exist is left alone as `forbidden` — that one would be a real problem. - **4xx errors now report why.** `_make_request` ended non-retryable statuses with `raise_for_status()`, whose curl_cffi message is `HTTP Error {code}: {reason}` — and HTTP/2 carries no reason phrase, so a refused request logged as bare `HTTP Error 403:` and the response body (the only explanation the provider gives) was discarded. The body's `detail`/`error`/`message` is now carried into the `ProviderError`, redacted and truncated. This is what made the media 403s on `GET /backend-api/files/{id}/download` undiagnosable. - **`redact_secrets` missed compound key names.** It matched keys exactly, so `access_token`, `api_key`, and `session-token` passed through un-redacted into debug-logged response bodies; matching now applies per word ("keywords", "monkey", "tokenizer" stay intact). - **`tests/test_config.py::TestSessionLimiterConfig::test_defaults` depended on the developer's `.env`.** `load_config()` calls `load_dotenv(override=False)`, which re-populated the variable the test had just deleted — so it passed only on a machine with no `.env`. The test now stubs dotenv discovery. ### Added +- **Codex CLI provider (`--provider codex`).** Archives local Codex agent transcripts from `~/.codex/sessions/**/rollout-*.jsonl` — local-only, like `claude-code`: no tokens, no rate limits, no ToS exposure. Sessions land in their own top-level `AI-Codex` Joplin notebook, with the same prose-only default, repo tags (`CODEX_REPO_TAG_IGNORE`) and multi-root scanning (`CODEX_DIR`, plus `$CODEX_HOME/sessions`). + + Codex writes each session twice in one file and the choice between the two layers is the whole design. `response_item` records are the model-facing wire format, where a tool call arrives as *JavaScript* (`tools.exec_command({...})`) because Codex's `exec` tool is code-mode; `event_msg`/`item_completed` records are Codex's own typed items, already decoded into `CommandExecution`/`FileChange`/`Extension` with argv, cwd, exit code and output as fields. Measured over 7 sessions on 0.147.0 (2026-08-18), the typed layer is 1:1 with the raw layer for prose (91 `AgentMessage` ↔ 91 assistant messages, sharing ids) and additionally omits every piece of harness plumbing — all 51 `developer`-role messages plus the 7 `# AGENTS.md instructions…` and 1 `` injections — which the Claude Code provider has to strip by regex. So the typed layer is parsed for content. + + Its one gap is that it records what *ran*, not what was *attempted*: 26 of 180 `exec_command` calls produced no item (14 sandbox launch failures, 6 user aborts, ~5 still running at turn end, 1 failure). The raw layer is therefore read for a call count only, and placeholders report the shortfall — `3 calls: exec_command ×3 (+2 did not complete)` — instead of silently under-reporting. `wait` calls are process polls, not attempts, and are excluded. + + **Reasoning is not exportable from Codex.** All 345 reasoning records carry `encrypted_content`, with `summary` empty in the raw layer and `summary_text`/`raw_content` empty in the typed layer, in every session. It is always dropped and counted; unlike Claude Code, `EXPORTER_HIDDEN_CONTENT=full` cannot surface it. The sidecar SQLite databases (`state_5.sqlite`, `thread_history_1.sqlite`) are deliberately not read: `thread_history_projection_state` tracks a byte offset into the rollout file, so the JSONL is canonical and the DB derived, and its `title` column is just the first user message truncated. + + **Codex Cloud is out of scope, verified rather than assumed.** Cloud tasks are reachable at `chatgpt.com/backend-api/api/codex/tasks{,/list}` — the same host and `/backend-api` root the ChatGPT provider already uses — but local CLI sessions are never uploaded there, so it is not an alternative source for these transcripts. `codex cloud list` confirmed the account holds no cloud tasks. The provider makes no network calls. + - **`projects` command — discover the project IDs your config is missing.** `CHATGPT_PROJECT_IDS` is maintained by hand, and a project missing from it is invisible to the listing pass, so its conversations are never fetched. The command reports every project your conversations belong to, marks which are absent from `.env`, and prints a paste-ready line (`--write` applies it). It reads project ids from the conversation listing when they are there and falls back to `--deep`, one detail request per conversation, when they are not. - **Project attribution now reads the conversation's own `gizmo_id`.** Previously the project name came only from `CHATGPT_PROJECT_IDS`, so a conversation in a project you had not listed exported into `no-project/` even though its payload names its project. The detail response carries `gizmo_id`, so it is used as a fallback after the listing annotation and the project map — attribution stays correct without maintaining a list, and moving a chat into a new project no longer silently misfiles it. Only `g-p-` ids count: a custom GPT is not a project and must not become a folder. Each unconfigured project is reported once per run, naming the id to add, because the *listing* pass still needs `CHATGPT_PROJECT_IDS` — conversations that live only inside a project never appear in the default listing. diff --git a/README.md b/README.md index 587bb61..c1d4b01 100644 --- a/README.md +++ b/README.md @@ -263,6 +263,27 @@ Sessions from all roots are merged by folder (no per-machine label); if the same --- +## Codex Sessions + +The `codex` provider archives your local [Codex CLI](https://chatgpt.com/codex) agent transcripts — same deal as Claude Code: no tokens, no API, no ToS exposure. Rollout files are read from `~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl`. + +```bash +ai-chat-exporter export --provider codex +ai-chat-exporter joplin --provider codex +``` + +Exports are prose-only by default, with the same placeholder format as Claude Code, into their own top-level **`AI-Codex`** Joplin notebook. Repo tags work the same way (`CODEX_REPO_TAG_IGNORE` to suppress), and additional roots can be scanned with `CODEX_DIR` (`:`-separated); `CODEX_HOME`'s `sessions/` is picked up automatically when that variable is set. + +Three things differ from Claude Code, all forced by how Codex stores its data: + +**Reasoning cannot be exported.** Codex encrypts it at rest — every reasoning record carries `encrypted_content` with no plaintext summary in any layer of the file. It is always dropped and counted; `EXPORTER_HIDDEN_CONTENT=full` cannot bring it back. + +**Incomplete tool calls are reported.** Codex writes each session twice in one file: its own typed items (what actually ran) and the raw model-facing wire format (everything attempted). The provider reads the typed layer — it is already decoded, and it omits harness plumbing that would otherwise need stripping — but cross-checks the raw layer for calls that never produced a result, so placeholders read `3 calls: exec_command ×3 (+2 did not complete)`. Those are commands that failed to launch, that you aborted, or that were still running when the turn ended. + +**Cloud tasks are out of scope.** `codex cloud` tasks run server-side and are reachable at `chatgpt.com/backend-api/api/codex/tasks`, but local CLI sessions are never uploaded there, so the cloud API is not an alternative source for these transcripts and this provider stays entirely offline. If you start using `codex cloud exec`, those transcripts would be cloud-only and would need separate work. + +--- + ## Output Structure All exported files go under `EXPORT_DIR`. The folder structure maps directly to Joplin notebooks. @@ -371,7 +392,7 @@ ai-chat-exporter export --output /path/to/my/notes ai-chat-exporter export --dry-run ``` -Options: `--provider [chatgpt|claude|claude-code|all]`, `--format [markdown|json|both]`, `--output PATH`, `--since YYYY-MM-DD`, `--project NAME`, `--hidden-content [full|placeholder|omit]`, `--download-media [images|all|off]`, `--max-conversations N`, `--force`, `--dry-run` +Options: `--provider [chatgpt|claude|claude-code|codex|all]`, `--format [markdown|json|both]`, `--output PATH`, `--since YYYY-MM-DD`, `--project NAME`, `--hidden-content [full|placeholder|omit]`, `--download-media [images|all|off]`, `--max-conversations N`, `--force`, `--dry-run` **Re-rendering the whole archive after an upgrade.** New formatting or features (collapse policy, media downloads) only change conversations as they're re-exported. To re-render everything you already have, use `--force` — it re-exports every conversation even if unchanged, **without** `cache --clear`, so your Joplin note links are preserved (a later `joplin` run updates the existing notes instead of duplicating them). diff --git a/src/joplin.py b/src/joplin.py index 32f97d3..69654e7 100644 --- a/src/joplin.py +++ b/src/joplin.py @@ -409,6 +409,10 @@ _PROVIDER_DISPLAY = { # intermixed with Claude web projects and named by dev folder). Existing # notes self-heal into here on the next sync — update_note moves them. "claude-code": "AI-ClaudeCode", + # Codex CLI coding sessions get their own top-level notebook for the same + # reason Claude Code does — they are a distinct surface, not a ChatGPT + # project, and burying them under AI-ChatGPT makes both harder to browse. + "codex": "AI-Codex", } diff --git a/src/main.py b/src/main.py index d31604e..1ec28c6 100644 --- a/src/main.py +++ b/src/main.py @@ -684,7 +684,7 @@ def _print_doctor_table(checks: list[dict]) -> None: @cli.command() @click.option( "--provider", - type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False), + type=click.Choice(["chatgpt", "claude", "claude-code", "codex", "all"], case_sensitive=False), default="all", show_default=True, help="Which provider to export.", @@ -1076,6 +1076,21 @@ def _resolve_providers(provider: str, cfg) -> list[tuple[str, object]]: ", ".join(str(r) for r in cc_roots), ) + if provider in ("codex", "all"): + from src.providers.codex import CodexProvider, resolve_roots as codex_roots_fn + cx_roots = codex_roots_fn() + if any(r.is_dir() for r in cx_roots): + result.append(( + "codex", + CodexProvider(hidden_content=cfg.hidden_content), + )) + elif provider == "codex": + logging.getLogger(__name__).warning( + "[codex] Skipping — none of these roots found (set " + "CODEX_DIR, ':'-separated for multiple): %s", + ", ".join(str(r) for r in cx_roots), + ) + return result @@ -1172,7 +1187,7 @@ def _print_export_summary(summary: dict[str, dict[str, int]]) -> None: @cli.command(name="list") @click.option( "--provider", - type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False), + type=click.Choice(["chatgpt", "claude", "claude-code", "codex", "all"], case_sensitive=False), default="all", show_default=True, ) @@ -1238,7 +1253,7 @@ def list_conversations(ctx: click.Context, provider: str, project_filter: str | @click.option("--clear", is_flag=True, help="Clear cached entries.") @click.option( "--provider", - type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False), + type=click.Choice(["chatgpt", "claude", "claude-code", "codex", "all"], case_sensitive=False), default="all", help="Provider to target (used with --clear).", ) @@ -1376,7 +1391,7 @@ def prune(ctx: click.Context, dry_run: bool, yes: bool) -> None: @cli.command() @click.option( "--provider", - type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False), + type=click.Choice(["chatgpt", "claude", "claude-code", "codex", "all"], case_sensitive=False), default="all", show_default=True, help="Which provider's conversations to sync to Joplin.", @@ -1449,7 +1464,7 @@ def joplin(ctx: click.Context, provider: str, project_filter: str | None, dry_ru # Determine which providers to process providers_to_sync: list[str] = [] - for prov in ("chatgpt", "claude", "claude-code"): + for prov in ("chatgpt", "claude", "claude-code", "codex"): if provider in (prov, "all"): providers_to_sync.append(prov) diff --git a/src/providers/claude_code.py b/src/providers/claude_code.py index ef0636d..36fe315 100644 --- a/src/providers/claude_code.py +++ b/src/providers/claude_code.py @@ -54,6 +54,7 @@ from src.blocks import ( make_unknown_block, ) from src.loss_report import LossReport +from src.utils import git_root_name from src.providers.base import ( BaseProvider, HIDDEN_CONTENT_FULL, @@ -395,7 +396,11 @@ def _parse_jsonl(path: Path) -> list[dict]: except OSError as e: logger.warning("[claude-code] Could not read %s: %s", path, e) return records - for line in text.splitlines(): + # split("\n"), not splitlines(): splitlines() also breaks on U+0085, + # U+2028/9 and friends, which are legal *inside* a JSON string. A NEL in + # captured command output shreds one record into unparseable fragments and + # loses it silently (observed 2026-08-18 in a real Codex rollout). + for line in text.split("\n"): line = line.strip() if not line: continue @@ -458,37 +463,6 @@ def _ignored_repos() -> set[str]: return {s.strip() for s in env.split(",") if s.strip()} -# dir Path → git-repo name it belongs to (or None). Process-wide; the working -# tree doesn't change under us mid-run, so caching walked dirs is safe. -_GIT_ROOT_CACHE: dict[Path, str | None] = {} -_CACHE_MISS = object() - - -def _git_root_name(path: Path, max_steps: int = 25) -> str | None: - """Name of the git repo ``path`` lives in — nearest ancestor with ``.git``. - - Walks up from ``path`` until a ``.git`` entry is found (returns that dir's - basename) or the filesystem root is reached (returns ``None``). Disk-based: - a path in no git repo, or a repo no longer on disk, yields ``None``. - """ - cur = path - for _ in range(max_steps): - cached = _GIT_ROOT_CACHE.get(cur, _CACHE_MISS) - if cached is not _CACHE_MISS: - return cached - try: - if (cur / ".git").exists(): - _GIT_ROOT_CACHE[cur] = cur.name - return cur.name - except OSError: - break - if cur.parent == cur: # filesystem root - break - cur = cur.parent - _GIT_ROOT_CACHE[path] = None - return None - - # Absolute path-like tokens inside Bash command strings (file_path/path keys are # matched directly). git-root resolution short-circuits on non-repo paths. _ABS_PATH_RE = re.compile(r"/(?:[\w.\-]+/)*[\w.\-]+") @@ -521,7 +495,7 @@ def _repos_touched( if p in seen: return seen.add(p) - name = _git_root_name(Path(p)) + name = git_root_name(Path(p)) if name and not name.startswith(".") and name not in ignore: counts[name] += 1 diff --git a/src/providers/codex.py b/src/providers/codex.py new file mode 100644 index 0000000..2a6c2d2 --- /dev/null +++ b/src/providers/codex.py @@ -0,0 +1,881 @@ +"""Codex CLI session provider — archives local agent transcripts. + +Reads JSONL rollout files from ``~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl`` +(override with ``CODEX_DIR``). Like Claude Code: no tokens, no rate limits, no +ToS risk — the data is local and Codex may prune it. + +Local-only, deliberately +------------------------ +Codex Cloud tasks (``codex cloud``) live server-side at +``https://chatgpt.com/backend-api/api/codex/tasks{,/list}`` — the same host and +``/backend-api`` root the ChatGPT provider already speaks. **CLI sessions are +never uploaded there**, so the cloud API is not an alternative source for these +transcripts and this provider does not talk to the network. Verified 2026-08-18 +against Codex 0.147.0. If ``codex cloud exec`` ever enters regular use, those +transcripts *would* be cloud-only and would need a separate provider. + +Two representations, one file (measured 2026-08-18 over 7 sessions / 0.147.0) +---------------------------------------------------------------------------- +Every rollout line is ``{timestamp, ordinal, type, payload}``. Dialogue appears +twice, in two different shapes, and we parse the **typed** one: + +* ``response_item`` — the model-facing wire format (mirrors the OpenAI Responses + API). Tool calls arrive as *JavaScript source* because Codex's ``exec`` tool is + code-mode:: + + const r = await tools.exec_command({"cmd":"git status","workdir":"/x", …}); + text(r.output); + + Exactly one ``tools.*`` call per invocation; three functions observed: + ``exec_command`` (180), ``web__run`` (15), ``apply_patch`` (13). + +* ``event_msg`` / ``item_completed`` — Codex's own typed items, already decoded: + ``UserMessage``, ``AgentMessage``, ``Reasoning``, ``CommandExecution``, + ``FileChange``, ``Extension``, ``ContextCompaction``. + +The typed layer wins on every axis that matters here. It is 1:1 with the raw +layer for prose (91 ``AgentMessage`` ↔ 91 assistant messages, same ids; 345 +``Reasoning`` ↔ 345), it hands us structured command/exit-code/output fields +instead of JS we would have to regex, and it pre-filters harness plumbing for +free: all 51 ``developer``-role messages (skills manifests, ````, +"Approved command prefix saved") plus the 7 ``# AGENTS.md instructions…`` +injections and 1 ```` have no typed item. That is the same +noise ``claude_code._HARNESS_TAG_RE`` strips by hand. + +Its one weakness: it records what *ran*, not what was *attempted*. 26 of 180 +``exec_command`` calls produced no ``CommandExecution`` item — 14 sandbox launch +failures (``bwrap: loopback: Failed RTM_NEWADDR``), 6 user aborts ("aborted by +user after 504.6s"), ~5 still running at turn end, 1 "Script failed". So we read +the raw layer *only* to count attempts, and the collapsed placeholder reports the +shortfall ("3 did not complete") rather than silently under-reporting. ``wait`` +function calls (50) are process polls, not attempts, and are not counted. + +Reasoning is unrecoverable +-------------------------- +All 345 reasoning items carry ``encrypted_content``; ``summary`` is ``[]`` in the +raw layer and ``summary_text``/``raw_content`` are empty in the typed layer, in +every session. Codex does not persist readable reasoning locally. Thinking is +therefore always dropped and counted — the same end state as the Claude Code +policy (decision 2026-06-12), but by necessity rather than by choice, so even +``full`` cannot surface it. + +Why the sidecar SQLite is not read +---------------------------------- +``~/.codex/state_5.sqlite`` carries a ``threads`` table (title, cwd, model, +tokens_used, rollout_path) and ``thread_history_1.sqlite`` a projection of the +items — but ``thread_history_projection_state`` tracks a byte offset *into the +rollout file*, i.e. the JSONL is canonical and SQLite is derived. Its ``title`` +is just the first user message truncated (identical to ``first_user_message`` and +``preview``), so it offers nothing the JSONL lacks, and its filename carries a +schema version that will churn. We read the files. + +Rendering follows the EXPORTER_HIDDEN_CONTENT policy: prose-only by default +(dialogue kept, tool traffic collapsed to one grouped placeholder per activity +run, reasoning dropped); ``full`` keeps the decoded tool calls and their output. + +Subagents: Codex 0.147.0's ``thread_spawn_edges`` table exists but is empty and no +sub-transcripts were observed, so there is no subagent folding here (contrast +``claude_code._load_subagents``). If spawned agents start appearing they will +arrive as new item types and land in the loss report as unknowns. +""" + +import json +import logging +import os +import re +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from src.blocks import ( + COLLAPSED_KIND_HIDDEN_CONTEXT, + COLLAPSED_KIND_TOOL_DUMP, + UNKNOWN_REASON_UNKNOWN_TYPE, + make_collapsed_block, + make_text_block, + make_tool_result_block, + make_tool_use_block, + make_unknown_block, +) +from src.loss_report import LossReport +from src.providers.base import ( + BaseProvider, + HIDDEN_CONTENT_FULL, + ProviderError, + VALID_HIDDEN_CONTENT_POLICIES, + resolve_hidden_content_policy, +) +from src.utils import git_root_name + +logger = logging.getLogger(__name__) + +DEFAULT_SESSIONS_DIR = "~/.codex/sessions" + +# rollout--.jsonl — the trailing UUID is the thread id. +_ROLLOUT_RE = re.compile( + r"^rollout-\d{4}-\d{2}-\d{2}T[\d-]+-" + r"([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})$" +) + +# The single tools.* call inside an `exec` custom_tool_call's JavaScript body. +_TOOLS_CALL_RE = re.compile(r"tools\.([A-Za-z_][\w.]*)\s*\(") + +# Typed items that represent tool traffic, mapped to the label used in the +# collapsed placeholder. These labels must match the ones derived from the raw +# layer in _attempt_label, or the attempted-vs-completed delta is meaningless. +_TOOL_ITEM_LABELS = { + "CommandExecution": "exec_command", + "FileChange": "apply_patch", + # Extension is labelled by its `kind` (e.g. "web.search"); see _tool_label. +} + +# function_call names that poll an already-running process rather than starting +# new work. Counting them as attempts would inflate the shortfall. +_POLLING_FUNCTIONS = {"wait"} + +# How many distinct tool names to list in a collapsed-activity placeholder. +_TOOL_NAMES_SHOWN = 4 + +# Harness-injected user text. The typed layer already omits these (they have no +# UserMessage item), so this is a belt-and-braces guard for other Codex versions. +_HARNESS_USER_RE = re.compile( + r"^\s*(?:#\s*AGENTS\.md instructions for\b|)", +) + + +def resolve_roots(sessions_dir=None) -> list[Path]: + """Resolve the ordered list of Codex ``sessions/`` roots to scan. + + Precedence: + 1. Explicit ``sessions_dir`` (a single path or a list) — used by tests. + 2. ``CODEX_DIR`` env, split on ``os.pathsep`` (``:``) so multiple roots can + be given; a single path (no separator) stays backward compatible. Falls + back to the default root when unset. + 3. Additionally, if ``CODEX_HOME`` is set (Codex's own name for its state + directory), its ``sessions/`` subdir is appended. + + Roots are expanded and de-duplicated (order preserved); existence is checked + by the caller / ``_scan``. + """ + raw: list[str] + if sessions_dir is not None: + raw = ( + [str(p) for p in sessions_dir] + if isinstance(sessions_dir, (list, tuple)) + else [str(sessions_dir)] + ) + else: + env = os.getenv("CODEX_DIR") + raw = env.split(os.pathsep) if env else [DEFAULT_SESSIONS_DIR] + codex_home = os.getenv("CODEX_HOME") + if codex_home: + raw.append(str(Path(codex_home) / "sessions")) + + roots: list[Path] = [] + for p in raw: + p = p.strip() + if not p: + continue + path = Path(p).expanduser() + if path not in roots: + roots.append(path) + return roots + + +class CodexProvider(BaseProvider): + """Local-file provider over Codex CLI rollout transcripts.""" + + provider_name = "codex" + + def __init__( + self, + sessions_dir: str | Path | None = None, + hidden_content: str | None = None, + ) -> None: + super().__init__() + self._sessions_dirs = resolve_roots(sessions_dir) + self._hidden_content = ( + hidden_content + if hidden_content in VALID_HIDDEN_CONTENT_POLICIES + else resolve_hidden_content_policy() + ) + # conv_id → rollout file path, populated by _scan() + self._path_map: dict[str, Path] = {} + + # ------------------------------------------------------------------ + # BaseProvider interface + # ------------------------------------------------------------------ + + def list_conversations(self, offset: int = 0, limit: int = 100) -> list[dict]: + full = self._scan() + return full[offset : offset + limit] + + def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]: + convs = self._scan() + if since is not None: + since_aware = since if since.tzinfo else since.replace(tzinfo=timezone.utc) + convs = [ + c for c in convs + if datetime.fromisoformat(c["updated_at"]) >= since_aware + ] + logger.info( + "[codex] Found %d session(s) under %s", + len(convs), + ", ".join(str(d) for d in self._sessions_dirs), + ) + return convs + + def get_conversation(self, conv_id: str) -> dict: + path = self._path_map.get(conv_id) + if path is None: + # Direct call without a prior listing (e.g. tests) — scan first. + self._scan() + path = self._path_map.get(conv_id) + if path is None or not path.exists(): + raise ProviderError( + self.provider_name, + f"get_conversation({conv_id[:8]})", + FileNotFoundError(f"No rollout file for id {conv_id}"), + ) + + records = _parse_jsonl(path) + mtime = datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc) + return { + "id": conv_id, + "_path": str(path), + "_records": records, + # Listing and normalized updated_at must match, or the cache + # staleness comparison would re-export every session every run. + "_mtime_iso": mtime.isoformat(), + } + + def normalize_conversation(self, raw: dict, loss_report: LossReport | None = None) -> dict: + report = loss_report if loss_report is not None else LossReport() + policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy() + conv_id = raw.get("id") or "" + records: list[dict] = raw.get("_records") or [] + + title = _extract_title(records) + # Append the repos this session touched, e.g. "… [repo-a, repo-b]". + # Codex sessions are commonly all launched from one workspace root, so + # without this every session lands in the same notebook with no way to + # tell them apart. + launch_cwd = _extract_launch_cwd(records) + if launch_cwd: + repos = _repos_touched(records, launch_cwd) + if repos: + title = f"{title} [{', '.join(repos)}]" + project = _extract_project(records) + created_at = _extract_created_at(records) + updated_at = raw.get("_mtime_iso") or next( + (r.get("timestamp") for r in reversed(records) if r.get("timestamp")), "" + ) + + messages = _extract_messages(records, conv_id, report, policy) + for _ in messages: + report.record_message() + report.record_conversation() + + return { + "id": conv_id, + "title": title, + "provider": self.provider_name, + "project": project, + "created_at": created_at or "", + "updated_at": updated_at or "", + "message_count": len(messages), + "messages": messages, + } + + # ------------------------------------------------------------------ + # Scanning + # ------------------------------------------------------------------ + + def _scan(self) -> list[dict]: + existing = [d for d in self._sessions_dirs if d.is_dir()] + if not existing: + logger.warning( + "[codex] No sessions directory exists: %s", + ", ".join(str(d) for d in self._sessions_dirs), + ) + return [] + + # conv_id → (mtime, conv dict). If the same thread id appears under two + # roots (e.g. a live dir and a backup), the newer-mtime copy wins. + by_id: dict[str, tuple[float, dict]] = {} + for root in existing: + # Rollouts are filed under YYYY/MM/DD; rglob keeps us agnostic to + # that layout in case Codex reorganises it. + for session_file in sorted(root.rglob("rollout-*.jsonl")): + match = _ROLLOUT_RE.match(session_file.stem) + if not match: + logger.debug("[codex] Skipping unrecognised filename %s", session_file.name) + continue + try: + stat = session_file.stat() + except OSError: + continue + if stat.st_size == 0: + continue + conv_id = match.group(1) + prev = by_id.get(conv_id) + if prev is not None and prev[0] >= stat.st_mtime: + logger.debug( + "[codex] Duplicate session %s in %s; keeping newer copy", + conv_id[:8], root, + ) + continue + self._path_map[conv_id] = session_file + title, project, created = _read_session_meta(session_file) + by_id[conv_id] = ( + stat.st_mtime, + { + "id": conv_id, + "title": title, + "project": project, + # The --project filter and dry-run table read the + # listing dict, not the normalized conversation. + "_project_name": project, + "created_at": created, + "updated_at": datetime.fromtimestamp( + stat.st_mtime, tz=timezone.utc + ).isoformat(), + "_path": str(session_file), + }, + ) + # Deterministic order: by creation date bucket, then thread id. + return sorted( + (conv for _, conv in by_id.values()), + key=lambda c: (c["created_at"], c["id"]), + ) + + +# --------------------------------------------------------------------------- +# Internal helpers +# --------------------------------------------------------------------------- + + +def _parse_jsonl(path: Path) -> list[dict]: + """Read a JSONL file into records, tolerating (and logging) bad lines.""" + records: list[dict] = [] + bad_lines = 0 + try: + text = path.read_text(encoding="utf-8") + except OSError as e: + logger.warning("[codex] Could not read %s: %s", path, e) + return records + # split("\n"), not splitlines(): splitlines() also breaks on U+0085, + # U+2028/9 and friends, which are legal *inside* a JSON string. A NEL in + # captured command output shreds one record into unparseable fragments and + # loses it silently (observed 2026-08-18 in a real Codex rollout). + for line in text.split("\n"): + line = line.strip() + if not line: + continue + try: + records.append(json.loads(line)) + except json.JSONDecodeError: + bad_lines += 1 + if bad_lines: + logger.warning("[codex] %s: skipped %d unparseable line(s)", path.name, bad_lines) + return records + + +def _item(rec: dict) -> dict | None: + """Return the typed item from an ``event_msg``/``item_completed`` record.""" + if rec.get("type") != "event_msg": + return None + payload = rec.get("payload") or {} + if payload.get("type") != "item_completed": + return None + item = payload.get("item") + return item if isinstance(item, dict) else None + + +def _item_kind(item: dict) -> str: + """Typed item discriminator. 0.147.0 uses ``type``; ``item_type`` is a hedge.""" + return str(item.get("item_type") or item.get("type") or "") + + +def _item_text(item: dict) -> str: + """Concatenate the text of a UserMessage / AgentMessage item. + + The two disagree on case — ``UserMessage`` blocks are ``"text"`` and + ``AgentMessage`` blocks are ``"Text"`` — so the comparison is case-folded. + """ + content = item.get("content") + if isinstance(content, str): + return content.strip() + if not isinstance(content, list): + return "" + parts = [] + for block in content: + if isinstance(block, dict) and str(block.get("type", "")).lower() == "text": + text = block.get("text") + if isinstance(text, str): + parts.append(text) + return "".join(parts).strip() + + +def _read_session_meta(path: Path) -> tuple[str, str | None, str]: + """Light single-pass scan for listing metadata: (title, project, created_at). + + Substring guards keep this cheap — only candidate lines are JSON-parsed, and + the scan stops as soon as the title is found (the first UserMessage is + usually within the first few dozen lines of a multi-megabyte file). + """ + title = "" + project: str | None = None + created = "" + try: + with path.open(encoding="utf-8") as fh: + for line in fh: + if not created and '"session_meta"' in line: + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + payload = rec.get("payload") or {} + created = str(payload.get("timestamp") or rec.get("timestamp") or "") + cwd = payload.get("cwd") + if cwd: + project = Path(cwd).name or None + continue + if not title and '"UserMessage"' in line: + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + item = _item(rec) + if item is None or _item_kind(item) != "UserMessage": + continue + text = _item_text(item) + if text and not _HARNESS_USER_RE.match(text): + title = text[:80] + break + except OSError as e: + logger.warning("[codex] Could not read %s: %s", path, e) + return title or "Untitled session", project, created + + +def _extract_title(records: list[dict]) -> str: + """Codex has no AI-generated title — the first real user prompt is the title. + + (``state_5.sqlite``'s ``title`` column is this same string truncated; see the + module docstring on why the sidecar DB is not consulted.) + """ + for rec in records: + item = _item(rec) + if item is None or _item_kind(item) != "UserMessage": + continue + text = _item_text(item) + if text and not _HARNESS_USER_RE.match(text): + return text[:80] + return "Untitled session" + + +def _session_meta_payload(records: list[dict]) -> dict: + for rec in records: + if rec.get("type") == "session_meta": + payload = rec.get("payload") + if isinstance(payload, dict): + return payload + return {} + + +def _extract_launch_cwd(records: list[dict]) -> str | None: + """The session's working directory (constant per session).""" + cwd = _session_meta_payload(records).get("cwd") + return cwd if isinstance(cwd, str) and cwd else None + + +def _extract_project(records: list[dict]) -> str | None: + """Project = basename of the session's working directory.""" + cwd = _extract_launch_cwd(records) + if cwd: + return Path(cwd).name or None + return None + + +def _extract_created_at(records: list[dict]) -> str: + """Session start time. + + Prefers ``session_meta.payload.timestamp`` — the outer line ``timestamp`` is + when the record was *flushed*, which can trail the true start by minutes. + """ + payload = _session_meta_payload(records) + ts = payload.get("timestamp") + if isinstance(ts, str) and ts: + return ts + return next((r.get("timestamp") for r in records if r.get("timestamp")), "") or "" + + +def _ignored_repos() -> set[str]: + """Optional ignore-list: git repos to never tag (comma-separated names).""" + env = os.getenv("CODEX_REPO_TAG_IGNORE", "") + return {s.strip() for s in env.split(",") if s.strip()} + + +def _strip_file_uri(path: str) -> str: + """``CommandExecution.cwd`` is a ``file://`` URI; every other path is plain.""" + return path[7:] if path.startswith("file://") else path + + +# Absolute path-like tokens inside command strings. +_ABS_PATH_RE = re.compile(r"/(?:[\w.\-]+/)*[\w.\-]+") + + +def _repos_touched( + records: list[dict], launch_cwd: str, cap: int = 3, ignore: set[str] | None = None +) -> list[str]: + """Repos a session touched, for the title's ``[repo-a, repo-b]`` tag. + + A file's repo is the git repository it lives in (nearest ancestor with a + ``.git``). Paths come from the typed items: ``FileChange.changes`` keys + (absolute), and ``CommandExecution``'s argv plus its ``cwd``. Ordered by + touch frequency, capped with a trailing ``…``. + """ + ignore = _ignored_repos() if ignore is None else ignore + counts: Counter = Counter() + seen: set[str] = set() + + def note(p) -> None: + if not isinstance(p, str) or not p: + return + p = _strip_file_uri(p) + if not p.startswith("/"): # resolve relative paths against the launch cwd + if not launch_cwd: + return + p = str(Path(launch_cwd) / p) + if p in seen: + return + seen.add(p) + name = git_root_name(Path(p)) + if name and not name.startswith(".") and name not in ignore: + counts[name] += 1 + + for rec in records: + item = _item(rec) + if item is None: + continue + kind = _item_kind(item) + if kind == "FileChange": + changes = item.get("changes") + if isinstance(changes, dict): + for file_path in changes: + note(file_path) + elif kind == "CommandExecution": + note(item.get("cwd")) + command = item.get("command") + if isinstance(command, list) and command: + tail = command[-1] + if isinstance(tail, str): + for m in _ABS_PATH_RE.finditer(tail): + note(m.group(0)) + + ordered = [name for name, _ in counts.most_common()] + if len(ordered) > cap: + return ordered[:cap] + ["…"] + return ordered + + +def _tool_label(item: dict, kind: str) -> str: + """Placeholder label for a completed tool item. + + ``Extension`` covers everything routed through the model's own extensions + (``web.search`` so far), so its ``kind`` field is the useful name. + """ + if kind == "Extension": + return str(item.get("kind") or "extension") + return _TOOL_ITEM_LABELS.get(kind, kind) + + +def _attempt_label(payload: dict) -> str | None: + """Label for a raw tool call, or None if it should not count as an attempt. + + ``exec`` custom_tool_calls carry JavaScript; the inner ``tools.`` name is + what lines up with the typed items' labels. ``wait`` polls an already-running + process and starts no new work. + """ + ptype = payload.get("type") + name = payload.get("name") or "" + if ptype == "custom_tool_call": + if name == "exec": + match = _TOOLS_CALL_RE.search(payload.get("input") or "") + if match: + # tools.web__run → the Extension item calls itself "web.search"; + # both are "one web call", which is all the count claims. + return match.group(1) + return "exec" + return name or "tool" + if ptype == "function_call": + if name in _POLLING_FUNCTIONS: + return None + return name or "function" + return None + + +def _command_string(item: dict) -> str: + """Human-readable command from a ``CommandExecution`` argv list. + + The argv is ``["/bin/bash", "-lc", "