"""Claude Code session provider — archives local agent transcripts. Reads JSONL session files from ``~/.claude/projects//.jsonl`` (override with ``CLAUDE_CODE_DIR``). No tokens, no rate limits, no ToS risk — but the data lives in single files Claude Code may clean up, and it contains deliverables (reviews, plans, analyses) that exist nowhere else. Rendering follows the EXPORTER_HIDDEN_CONTENT policy (decided 2026-06-12): prose-only by default. User prompts and assistant text are kept; tool_use / tool_result traffic is collapsed to one grouped placeholder per activity run (measured: dialogue prose is ~4% of session bytes); thinking blocks are dropped (counted in the run summary, no placeholder). ``full`` keeps everything. Record types in a session file: ``user`` / ``assistant`` carry the dialogue (Anthropic-style ``message.content`` block arrays); ``ai-title`` carries the evolving session title (last one wins); ``last-prompt``, ``file-history-snapshot``, ``attachment``, ``permission-mode``, ``system`` are harness records and are skipped. ``isMeta`` records are harness-generated user records and are skipped. Subagents (Task tool): Claude Code stores each subagent's transcript as a separate ``/subagents/agent-*.jsonl`` file with an ``agent-*.meta.json`` sidecar (``agentType``, ``description``, ``toolUseId``). ``_load_subagents`` loads them keyed by ``toolUseId``; at the ``Task``/``Agent`` tool call that spawned it, the subagent is folded inline as a collapsible ``
`` block (the subagent's own records are flagged ``isSidechain`` and only processed on this recursive pass — the top-level pass still skips sidechain records). Scan scope: multiple ``projects/`` roots are supported — ``CLAUDE_CODE_DIR`` may be an ``os.pathsep``-separated list and ``CLAUDE_CONFIG_DIR``'s ``projects/`` is included when set. Sessions from all roots are merged by launch-folder; a session UUID present in two roots keeps the newer-mtime copy. See ``resolve_roots``. """ import json import logging import os import re from collections import Counter from datetime import datetime, timezone from pathlib import Path from src.blocks import ( COLLAPSED_KIND_TOOL_DUMP, UNKNOWN_REASON_UNKNOWN_TYPE, make_collapsed_block, make_image_placeholder, make_subagent_block, make_text_block, make_thinking_block, make_tool_result_block, make_tool_use_block, make_unknown_block, ) from src.loss_report import LossReport from src.utils import git_root_name from src.providers.base import ( BaseProvider, HIDDEN_CONTENT_FULL, ProviderError, VALID_HIDDEN_CONTENT_POLICIES, resolve_hidden_content_policy, ) logger = logging.getLogger(__name__) DEFAULT_PROJECTS_DIR = "~/.claude/projects" # Tool names whose call spawns a subagent (Task tool) — its separate transcript # is folded inline. Both spellings have appeared across Claude Code versions. _SUBAGENT_TOOL_NAMES = {"Task", "Agent"} def resolve_roots(projects_dir=None) -> list[Path]: """Resolve the ordered list of Claude Code ``projects/`` roots to scan. Precedence: 1. Explicit ``projects_dir`` (a single path or a list) — used by tests. 2. ``CLAUDE_CODE_DIR`` env, split on ``os.pathsep`` (``:``) so multiple roots can be given; a single path (no separator) stays backward compatible. Falls back to the default root when unset. 3. Additionally, if ``CLAUDE_CONFIG_DIR`` is set, its ``projects/`` subdir is appended — this is the common "non-default config dir" case. Roots are expanded, de-duplicated (order preserved), and returned as-is (existence is checked by the caller / ``_scan``). Sessions from all roots are merged by launch-folder; there is no per-source label. """ raw: list[str] if projects_dir is not None: raw = [str(p) for p in projects_dir] if isinstance(projects_dir, (list, tuple)) \ else [str(projects_dir)] else: env = os.getenv("CLAUDE_CODE_DIR") raw = env.split(os.pathsep) if env else [DEFAULT_PROJECTS_DIR] config_dir = os.getenv("CLAUDE_CONFIG_DIR") if config_dir: raw.append(str(Path(config_dir) / "projects")) roots: list[Path] = [] for p in raw: p = p.strip() if not p: continue path = Path(p).expanduser() if path not in roots: roots.append(path) return roots # Harness-injected tags inside user message text. Stripped so exports contain # the dialogue, not the CLI plumbing. A record that is nothing but tags # (e.g. a /model invocation) ends up empty and is skipped. _HARNESS_TAG_RE = re.compile( r".*?" r"|.*?" r"|.*?" r"|.*?" r"|.*?" r"|.*?" r"|.*?" # A fork subagent's first user turn: generic worker rules ahead of the # fork's actual "Your directive: …", which is kept. r"|.*?", re.DOTALL, ) # How many distinct tool names to list in a collapsed-activity placeholder. _TOOL_NAMES_SHOWN = 4 def _strip_harness_noise(text: str) -> str: if not isinstance(text, str): return "" return _HARNESS_TAG_RE.sub("", text).strip() class ClaudeCodeProvider(BaseProvider): """Local-file provider over Claude Code session transcripts.""" provider_name = "claude-code" def __init__( self, projects_dir: str | Path | None = None, hidden_content: str | None = None, ) -> None: super().__init__() self._projects_dirs = resolve_roots(projects_dir) self._hidden_content = ( hidden_content if hidden_content in VALID_HIDDEN_CONTENT_POLICIES else resolve_hidden_content_policy() ) # conv_id → session file path, populated by _scan() self._path_map: dict[str, Path] = {} # ------------------------------------------------------------------ # BaseProvider interface # ------------------------------------------------------------------ def list_conversations(self, offset: int = 0, limit: int = 100) -> list[dict]: full = self._scan() return full[offset : offset + limit] def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]: convs = self._scan() if since is not None: since_aware = since if since.tzinfo else since.replace(tzinfo=timezone.utc) convs = [ c for c in convs if datetime.fromisoformat(c["updated_at"]) >= since_aware ] logger.info( "[claude-code] Found %d session(s) under %s", len(convs), ", ".join(str(d) for d in self._projects_dirs), ) return convs def get_conversation(self, conv_id: str) -> dict: path = self._path_map.get(conv_id) if path is None: # Direct call without a prior listing (e.g. tests) — scan first. self._scan() path = self._path_map.get(conv_id) if path is None or not path.exists(): raise ProviderError( self.provider_name, f"get_conversation({conv_id[:8]})", FileNotFoundError(f"No session file for id {conv_id}"), ) records = _parse_jsonl(path) mtime = datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc) return { "id": conv_id, "_path": str(path), "_records": records, # Subagent (Task-tool) transcripts, keyed by the parent tool_use id # that spawned them, folded inline during normalization. "_subagents": _load_subagents(path), # Listing and normalized updated_at must match, or the cache # staleness comparison would re-export every session every run. "_mtime_iso": mtime.isoformat(), } def normalize_conversation(self, raw: dict, loss_report: LossReport | None = None) -> dict: report = loss_report if loss_report is not None else LossReport() policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy() conv_id = raw.get("id") or "" records: list[dict] = raw.get("_records") or [] subagents: dict = raw.get("_subagents") or {} title = _extract_title(records) # Append the repos this session touched, e.g. "… [repo-a, repo-b]", so # sessions launched from a workspace root (which all land in one # folder-named notebook) stay scannable and searchable. launch_cwd = _extract_launch_cwd(records) if launch_cwd: repos = _repos_touched(records, launch_cwd) if repos: title = f"{title} [{', '.join(repos)}]" project = _extract_project(records, raw.get("_path")) created_at = next( (r.get("timestamp") for r in records if r.get("timestamp")), "" ) updated_at = raw.get("_mtime_iso") or next( (r.get("timestamp") for r in reversed(records) if r.get("timestamp")), "" ) messages = _extract_messages(records, conv_id, report, policy, subagents) for _ in messages: report.record_message() report.record_conversation() return { "id": conv_id, "title": title, "provider": self.provider_name, "project": project, "created_at": created_at or "", "updated_at": updated_at or "", "message_count": len(messages), "messages": messages, } # ------------------------------------------------------------------ # Scanning # ------------------------------------------------------------------ def _scan(self) -> list[dict]: existing = [d for d in self._projects_dirs if d.is_dir()] if not existing: logger.warning( "[claude-code] No projects directory exists: %s", ", ".join(str(d) for d in self._projects_dirs), ) return [] # conv_id → (mtime, conv dict). Roots are merged by launch-folder; if the # same session UUID appears in two roots (e.g. a live dir and a backup), # the newer-mtime copy wins so we never emit two entries for one session. by_id: dict[str, tuple[float, dict]] = {} for root in existing: for proj_dir in sorted(p for p in root.iterdir() if p.is_dir()): for session_file in sorted(proj_dir.glob("*.jsonl")): try: stat = session_file.stat() except OSError: continue if stat.st_size == 0: continue conv_id = session_file.stem prev = by_id.get(conv_id) if prev is not None and prev[0] >= stat.st_mtime: logger.debug( "[claude-code] Duplicate session %s in %s; keeping newer copy", conv_id[:8], root, ) continue self._path_map[conv_id] = session_file title, project, created = _read_session_meta(session_file) by_id[conv_id] = ( stat.st_mtime, { "id": conv_id, "title": title, "project": project, # The --project filter and dry-run table read the # listing dict, not the normalized conversation. "_project_name": project, "created_at": created, "updated_at": datetime.fromtimestamp( stat.st_mtime, tz=timezone.utc ).isoformat(), "_path": str(session_file), "_project_dir": proj_dir.name, }, ) # Deterministic order: by bucket dir, then session id. return sorted( (conv for _, conv in by_id.values()), key=lambda c: (c["_project_dir"], c["id"]), ) # --------------------------------------------------------------------------- # Internal helpers # --------------------------------------------------------------------------- def _read_session_meta(path: Path) -> tuple[str, str | None, str]: """Light single-pass scan for listing metadata: (title, project, created_at). Substring guards keep this cheap — only candidate lines are JSON-parsed. The full-fidelity extraction happens later in normalize_conversation. """ title = "" project: str | None = None created = "" try: with path.open(encoding="utf-8") as fh: for line in fh: if '"ai-title"' in line: try: rec = json.loads(line) except json.JSONDecodeError: continue if rec.get("type") == "ai-title" and rec.get("aiTitle"): title = str(rec["aiTitle"]) # last one wins continue if (not created or project is None) and ( '"timestamp"' in line or '"cwd"' in line ): try: rec = json.loads(line) except json.JSONDecodeError: continue if not created and rec.get("timestamp"): created = str(rec["timestamp"]) if project is None and rec.get("cwd"): project = Path(rec["cwd"]).name or None except OSError as e: logger.warning("[claude-code] Could not read %s: %s", path, e) return title, project, created def _extract_title(records: list[dict]) -> str: """Last ai-title record wins; fall back to the first real user prompt.""" title = "" for rec in records: if rec.get("type") == "ai-title" and rec.get("aiTitle"): title = str(rec["aiTitle"]) if title: return title for rec in records: if rec.get("type") != "user" or rec.get("isMeta") or rec.get("isSidechain"): continue content = (rec.get("message") or {}).get("content") if isinstance(content, str): text = _strip_harness_noise(content) elif isinstance(content, list): text = " ".join( _strip_harness_noise(item.get("text", "")) for item in content if isinstance(item, dict) and item.get("type") == "text" ).strip() else: text = "" if text: return text[:80] return "Untitled session" def _extract_project(records: list[dict], path: str | None) -> str | None: """Project = basename of the session's working directory.""" for rec in records: cwd = rec.get("cwd") if cwd: name = Path(cwd).name if name: return name # Fallback: the munged directory name (cannot be reliably de-munged # because '-' is both the path separator and a legal name character). if path: return Path(path).parent.name.lstrip("-") or None return None def _parse_jsonl(path: Path) -> list[dict]: """Read a JSONL file into records, tolerating (and logging) bad lines.""" records: list[dict] = [] bad_lines = 0 try: text = path.read_text(encoding="utf-8") except OSError as e: logger.warning("[claude-code] Could not read %s: %s", path, e) return records # split("\n"), not splitlines(): splitlines() also breaks on U+0085, # U+2028/9 and friends, which are legal *inside* a JSON string. A NEL in # captured command output shreds one record into unparseable fragments and # loses it silently (observed 2026-08-18 in a real Codex rollout). for line in text.split("\n"): line = line.strip() if not line: continue try: records.append(json.loads(line)) except json.JSONDecodeError: bad_lines += 1 if bad_lines: logger.warning( "[claude-code] %s: skipped %d unparseable line(s)", path.name, bad_lines ) return records def _load_subagents(session_path: Path) -> dict: """Map spawning tool_use id → ``{"meta": {...}, "records": [...]}``. Claude Code writes subagent transcripts to ``/subagents/agent-*.jsonl`` with an ``agent-*.meta.json`` sidecar carrying ``toolUseId`` (which parent ``Task``/``Agent`` call spawned it). Files without a readable meta (no id to position them) are skipped. """ submap: dict[str, dict] = {} subdir = session_path.parent / session_path.stem / "subagents" if not subdir.is_dir(): return submap for jf in sorted(subdir.glob("*.jsonl")): meta_path = jf.with_suffix(".meta.json") try: meta = json.loads(meta_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError): logger.warning( "[claude-code] Subagent %s has no readable meta; skipping", jf.name ) continue tool_id = meta.get("toolUseId") if not tool_id: continue submap[tool_id] = {"meta": meta, "records": _parse_jsonl(jf)} return submap def _extract_launch_cwd(records: list[dict]) -> str | None: """The session's working directory (constant per session; first cwd wins).""" for rec in records: cwd = rec.get("cwd") if cwd: return cwd return None # tool_use input keys that carry a file path. _TOOL_PATH_KEYS = ("file_path", "path", "notebook_path") # Optional ignore-list: git repos to never tag (comma-separated names). Rarely # needed with git-root detection (config/reference dirs are already excluded # because they aren't git repos), but kept as an escape hatch. def _ignored_repos() -> set[str]: env = os.getenv("CLAUDE_CODE_REPO_TAG_IGNORE", "") return {s.strip() for s in env.split(",") if s.strip()} # Absolute path-like tokens inside Bash command strings (file_path/path keys are # matched directly). git-root resolution short-circuits on non-repo paths. _ABS_PATH_RE = re.compile(r"/(?:[\w.\-]+/)*[\w.\-]+") def _repos_touched( records: list[dict], launch_cwd: str, cap: int = 3, ignore: set[str] | None = None ) -> list[str]: """Repos a session touched, for the title's ``[repo-a, repo-b]`` tag. A file's repo is the **git repository it lives in** (nearest ancestor with a ``.git``), resolved for every absolute path seen in a tool_use input (``file_path``/``path``/``notebook_path`` and absolute home paths inside Bash ``command`` strings). This works anywhere in the filesystem — not just under the launch directory — so cross-workspace work is captured, and non-repo noise (config dirs, one-off files, reference dirs) is excluded because it isn't a git repo. Ordered by touch frequency, capped with a trailing ``…``. """ ignore = _ignored_repos() if ignore is None else ignore counts: Counter = Counter() seen: set[str] = set() def note(p) -> None: if not isinstance(p, str) or not p: return if not p.startswith("/"): # resolve relative paths against the launch cwd if not launch_cwd: return p = str(Path(launch_cwd) / p) if p in seen: return seen.add(p) name = git_root_name(Path(p)) if name and not name.startswith(".") and name not in ignore: counts[name] += 1 for rec in records: msg = rec.get("message") or {} content = msg.get("content") if not isinstance(content, list): continue for item in content: if not isinstance(item, dict) or item.get("type") != "tool_use": continue inp = item.get("input") if not isinstance(inp, dict): continue for key in _TOOL_PATH_KEYS: note(inp.get(key)) cmd = inp.get("command") if isinstance(cmd, str): for m in _ABS_PATH_RE.finditer(cmd): note(m.group(0)) ordered = [name for name, _ in counts.most_common()] if len(ordered) > cap: return ordered[:cap] + ["…"] return ordered def _extract_messages( records: list[dict], conv_id: str, report: LossReport, policy: str, subagents: dict | None = None, include_sidechain: bool = False, expanding: frozenset[str] = frozenset(), ) -> list[dict]: """Normalize Claude Code records into messages. ``subagents`` maps a spawning ``Task``/``Agent`` tool_use id to its separate transcript ``{"meta": ..., "records": ...}``; when a matching tool_use is seen its subagent is folded inline as a subagent block (extracted recursively under the same policy). ``include_sidechain`` is set True for those recursive subagent passes (subagent records are flagged ``isSidechain``); the top-level pass keeps skipping sidechain records so a subagent is never also emitted as a stray top-level turn. ``expanding`` holds the spawn ids of the subagents being folded around this pass. A ``fork`` subagent's transcript opens with a copy of the parent turn that spawned it, its own spawn call included; that copy is dropped, since the enclosing subagent block already stands for it. Expanding it again recursed without end (RecursionError, every daily sync from 2026-09-24). """ subagents = subagents or {} messages: list[dict] = [] # Pending collapsed tool activity: name → call count, plus total bytes. pending_tools: Counter = Counter() pending_bytes = 0 def flush_pending() -> None: nonlocal pending_bytes if not pending_tools: return shown = ", ".join( f"{name} ×{count}" for name, count in pending_tools.most_common(_TOOL_NAMES_SHOWN) ) if len(pending_tools) > _TOOL_NAMES_SHOWN: shown += ", …" calls = sum(pending_tools.values()) block = make_collapsed_block( origin=f"{calls} calls: {shown}", content_type="tool_activity", size_bytes=pending_bytes, kind=COLLAPSED_KIND_TOOL_DUMP, ) if messages and messages[-1]["role"] == "assistant": messages[-1]["blocks"].append(block) else: messages.append( {"role": "tool", "content_type": "text", "timestamp": None, "blocks": [block]} ) pending_tools.clear() pending_bytes = 0 for rec in records: if rec.get("type") not in ("user", "assistant"): continue if rec.get("isMeta"): continue if rec.get("isSidechain") and not include_sidechain: continue msg = rec.get("message") or {} role = msg.get("role") or rec["type"] content = msg.get("content") blocks: list[dict] = [] # This record's own tool traffic — merged into pending only after the # record's message is appended, so the placeholder lands on it (not on # the previous message). local_tools: Counter = Counter() local_bytes = 0 if isinstance(content, str): text = _strip_harness_noise(content) block = make_text_block(text) if block: blocks.append(block) elif isinstance(content, list): for item in content: if not isinstance(item, dict): continue item_type = item.get("type", "") if item_type == "text": block = make_text_block(_strip_harness_noise(item.get("text", ""))) if block: blocks.append(block) elif item_type in ("thinking", "redacted_thinking"): if policy == HIDDEN_CONTENT_FULL: block = make_thinking_block( item.get("thinking") or item.get("text") or "" ) if block: blocks.append(block) else: # Decision 2026-06-12: thinking is dropped without a # placeholder, but stays visible in the run summary. report.record_collapsed( "thinking", len(json.dumps(item, default=str)) ) elif item_type == "tool_use": name = item.get("name") or "tool" tool_id = item.get("id") if tool_id in expanding: continue if name in _SUBAGENT_TOOL_NAMES and tool_id in subagents: # A Task/Agent spawn: fold its separate transcript inline # instead of collapsing it. Its own tool traffic is # collapsed by the recursive pass under the same policy. sub = subagents[tool_id] meta = sub.get("meta") or {} sub_msgs = _extract_messages( sub.get("records") or [], conv_id, report, policy, subagents, include_sidechain=True, expanding=expanding | {tool_id}, ) blocks.append( make_subagent_block( agent_type=meta.get("agentType") or name, description=meta.get("description") or "", messages=sub_msgs, ) ) elif policy == HIDDEN_CONTENT_FULL: blocks.append( make_tool_use_block( item.get("name", ""), item.get("input"), item.get("id") ) ) else: size = len(json.dumps(item, default=str)) local_tools[name] += 1 local_bytes += size report.record_collapsed(name, size) elif item_type == "tool_result": if policy == HIDDEN_CONTENT_FULL: blocks.append( make_tool_result_block( _stringify_tool_result(item.get("content")), is_error=bool(item.get("is_error")), ) ) else: size = len(json.dumps(item, default=str)) local_bytes += size report.record_collapsed("tool_result", size) elif item_type == "image": blocks.append( make_image_placeholder(ref="embedded image", source="user_upload") ) else: logger.warning( "[claude-code] Unknown content block type %r in session %s", item_type, conv_id[:8], ) report.record_unknown(f"claude-code.{item_type or '?'}") blocks.append( make_unknown_block( raw_type=f"claude-code.{item_type or '?'}", observed_keys=list(item.keys()), reason=UNKNOWN_REASON_UNKNOWN_TYPE, ) ) if not blocks: # Tool-result-only records (and similar): traffic accumulates and # is attached to the message that initiated it. pending_tools.update(local_tools) pending_bytes += local_bytes continue # Attach previous records' tool activity to the previous message # before starting a new one, so the placeholder lands between the # dialogue turns it actually occurred between. flush_pending() messages.append( { "role": role, "content_type": "text", "timestamp": rec.get("timestamp"), "blocks": blocks, } ) pending_tools.update(local_tools) pending_bytes += local_bytes flush_pending() return messages def _stringify_tool_result(content) -> str: """tool_result content may be a string or a list of text blocks.""" if isinstance(content, str): return content if isinstance(content, list): parts = [] for item in content: if isinstance(item, dict) and item.get("type") == "text": parts.append(item.get("text", "")) else: parts.append(json.dumps(item, default=str)) return "\n".join(parts) return json.dumps(content, default=str) if content is not None else ""