Files
AIChatExporter/src/providers/claude_code.py
T
JesseMarkowitzandClaude Opus 5.5 64068bb19b fix: fork subagents recursed forever in the Claude Code export
A fork subagent's transcript opens with a copy of the parent turn that spawned it, its own Agent call included; folding that call re-entered the same fork until RecursionError, failing every daily sync since 2026-09-24. Skip a spawn call while its own subagent is being expanded, and strip the <fork-boilerplate> preamble.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LbnmGHnFqDjyhPcCg1SEfF
2026-10-05 07:14:16 -04:00

739 lines
29 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Claude Code session provider — archives local agent transcripts.
Reads JSONL session files from ``~/.claude/projects/<munged-cwd>/<uuid>.jsonl``
(override with ``CLAUDE_CODE_DIR``). No tokens, no rate limits, no ToS risk —
but the data lives in single files Claude Code may clean up, and it contains
deliverables (reviews, plans, analyses) that exist nowhere else.
Rendering follows the EXPORTER_HIDDEN_CONTENT policy (decided 2026-06-12):
prose-only by default. User prompts and assistant text are kept; tool_use /
tool_result traffic is collapsed to one grouped placeholder per activity run
(measured: dialogue prose is ~4% of session bytes); thinking blocks are
dropped (counted in the run summary, no placeholder). ``full`` keeps
everything.
Record types in a session file: ``user`` / ``assistant`` carry the dialogue
(Anthropic-style ``message.content`` block arrays); ``ai-title`` carries the
evolving session title (last one wins); ``last-prompt``,
``file-history-snapshot``, ``attachment``, ``permission-mode``, ``system``
are harness records and are skipped. ``isMeta`` records are harness-generated
user records and are skipped.
Subagents (Task tool): Claude Code stores each subagent's transcript as a
separate ``<session>/subagents/agent-*.jsonl`` file with an ``agent-*.meta.json``
sidecar (``agentType``, ``description``, ``toolUseId``). ``_load_subagents``
loads them keyed by ``toolUseId``; at the ``Task``/``Agent`` tool call that
spawned it, the subagent is folded inline as a collapsible ``<details>`` block
(the subagent's own records are flagged ``isSidechain`` and only processed on
this recursive pass — the top-level pass still skips sidechain records).
Scan scope: multiple ``projects/`` roots are supported — ``CLAUDE_CODE_DIR`` may
be an ``os.pathsep``-separated list and ``CLAUDE_CONFIG_DIR``'s ``projects/`` is
included when set. Sessions from all roots are merged by launch-folder; a session
UUID present in two roots keeps the newer-mtime copy. See ``resolve_roots``.
"""
import json
import logging
import os
import re
from collections import Counter
from datetime import datetime, timezone
from pathlib import Path
from src.blocks import (
COLLAPSED_KIND_TOOL_DUMP,
UNKNOWN_REASON_UNKNOWN_TYPE,
make_collapsed_block,
make_image_placeholder,
make_subagent_block,
make_text_block,
make_thinking_block,
make_tool_result_block,
make_tool_use_block,
make_unknown_block,
)
from src.loss_report import LossReport
from src.utils import git_root_name
from src.providers.base import (
BaseProvider,
HIDDEN_CONTENT_FULL,
ProviderError,
VALID_HIDDEN_CONTENT_POLICIES,
resolve_hidden_content_policy,
)
logger = logging.getLogger(__name__)
DEFAULT_PROJECTS_DIR = "~/.claude/projects"
# Tool names whose call spawns a subagent (Task tool) — its separate transcript
# is folded inline. Both spellings have appeared across Claude Code versions.
_SUBAGENT_TOOL_NAMES = {"Task", "Agent"}
def resolve_roots(projects_dir=None) -> list[Path]:
"""Resolve the ordered list of Claude Code ``projects/`` roots to scan.
Precedence:
1. Explicit ``projects_dir`` (a single path or a list) — used by tests.
2. ``CLAUDE_CODE_DIR`` env, split on ``os.pathsep`` (``:``) so multiple
roots can be given; a single path (no separator) stays backward
compatible. Falls back to the default root when unset.
3. Additionally, if ``CLAUDE_CONFIG_DIR`` is set, its ``projects/`` subdir
is appended — this is the common "non-default config dir" case.
Roots are expanded, de-duplicated (order preserved), and returned as-is
(existence is checked by the caller / ``_scan``). Sessions from all roots are
merged by launch-folder; there is no per-source label.
"""
raw: list[str]
if projects_dir is not None:
raw = [str(p) for p in projects_dir] if isinstance(projects_dir, (list, tuple)) \
else [str(projects_dir)]
else:
env = os.getenv("CLAUDE_CODE_DIR")
raw = env.split(os.pathsep) if env else [DEFAULT_PROJECTS_DIR]
config_dir = os.getenv("CLAUDE_CONFIG_DIR")
if config_dir:
raw.append(str(Path(config_dir) / "projects"))
roots: list[Path] = []
for p in raw:
p = p.strip()
if not p:
continue
path = Path(p).expanduser()
if path not in roots:
roots.append(path)
return roots
# Harness-injected tags inside user message text. Stripped so exports contain
# the dialogue, not the CLI plumbing. A record that is nothing but tags
# (e.g. a /model invocation) ends up empty and is skipped.
_HARNESS_TAG_RE = re.compile(
r"<local-command-caveat>.*?</local-command-caveat>"
r"|<command-name>.*?</command-name>"
r"|<command-message>.*?</command-message>"
r"|<command-args>.*?</command-args>"
r"|<command-contents>.*?</command-contents>"
r"|<local-command-stdout>.*?</local-command-stdout>"
r"|<system-reminder>.*?</system-reminder>"
# A fork subagent's first user turn: generic worker rules ahead of the
# fork's actual "Your directive: …", which is kept.
r"|<fork-boilerplate>.*?</fork-boilerplate>",
re.DOTALL,
)
# How many distinct tool names to list in a collapsed-activity placeholder.
_TOOL_NAMES_SHOWN = 4
def _strip_harness_noise(text: str) -> str:
if not isinstance(text, str):
return ""
return _HARNESS_TAG_RE.sub("", text).strip()
class ClaudeCodeProvider(BaseProvider):
"""Local-file provider over Claude Code session transcripts."""
provider_name = "claude-code"
def __init__(
self,
projects_dir: str | Path | None = None,
hidden_content: str | None = None,
) -> None:
super().__init__()
self._projects_dirs = resolve_roots(projects_dir)
self._hidden_content = (
hidden_content
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
else resolve_hidden_content_policy()
)
# conv_id → session file path, populated by _scan()
self._path_map: dict[str, Path] = {}
# ------------------------------------------------------------------
# BaseProvider interface
# ------------------------------------------------------------------
def list_conversations(self, offset: int = 0, limit: int = 100) -> list[dict]:
full = self._scan()
return full[offset : offset + limit]
def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]:
convs = self._scan()
if since is not None:
since_aware = since if since.tzinfo else since.replace(tzinfo=timezone.utc)
convs = [
c for c in convs
if datetime.fromisoformat(c["updated_at"]) >= since_aware
]
logger.info(
"[claude-code] Found %d session(s) under %s",
len(convs),
", ".join(str(d) for d in self._projects_dirs),
)
return convs
def get_conversation(self, conv_id: str) -> dict:
path = self._path_map.get(conv_id)
if path is None:
# Direct call without a prior listing (e.g. tests) — scan first.
self._scan()
path = self._path_map.get(conv_id)
if path is None or not path.exists():
raise ProviderError(
self.provider_name,
f"get_conversation({conv_id[:8]})",
FileNotFoundError(f"No session file for id {conv_id}"),
)
records = _parse_jsonl(path)
mtime = datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc)
return {
"id": conv_id,
"_path": str(path),
"_records": records,
# Subagent (Task-tool) transcripts, keyed by the parent tool_use id
# that spawned them, folded inline during normalization.
"_subagents": _load_subagents(path),
# Listing and normalized updated_at must match, or the cache
# staleness comparison would re-export every session every run.
"_mtime_iso": mtime.isoformat(),
}
def normalize_conversation(self, raw: dict, loss_report: LossReport | None = None) -> dict:
report = loss_report if loss_report is not None else LossReport()
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
conv_id = raw.get("id") or ""
records: list[dict] = raw.get("_records") or []
subagents: dict = raw.get("_subagents") or {}
title = _extract_title(records)
# Append the repos this session touched, e.g. "… [repo-a, repo-b]", so
# sessions launched from a workspace root (which all land in one
# folder-named notebook) stay scannable and searchable.
launch_cwd = _extract_launch_cwd(records)
if launch_cwd:
repos = _repos_touched(records, launch_cwd)
if repos:
title = f"{title} [{', '.join(repos)}]"
project = _extract_project(records, raw.get("_path"))
created_at = next(
(r.get("timestamp") for r in records if r.get("timestamp")), ""
)
updated_at = raw.get("_mtime_iso") or next(
(r.get("timestamp") for r in reversed(records) if r.get("timestamp")), ""
)
messages = _extract_messages(records, conv_id, report, policy, subagents)
for _ in messages:
report.record_message()
report.record_conversation()
return {
"id": conv_id,
"title": title,
"provider": self.provider_name,
"project": project,
"created_at": created_at or "",
"updated_at": updated_at or "",
"message_count": len(messages),
"messages": messages,
}
# ------------------------------------------------------------------
# Scanning
# ------------------------------------------------------------------
def _scan(self) -> list[dict]:
existing = [d for d in self._projects_dirs if d.is_dir()]
if not existing:
logger.warning(
"[claude-code] No projects directory exists: %s",
", ".join(str(d) for d in self._projects_dirs),
)
return []
# conv_id → (mtime, conv dict). Roots are merged by launch-folder; if the
# same session UUID appears in two roots (e.g. a live dir and a backup),
# the newer-mtime copy wins so we never emit two entries for one session.
by_id: dict[str, tuple[float, dict]] = {}
for root in existing:
for proj_dir in sorted(p for p in root.iterdir() if p.is_dir()):
for session_file in sorted(proj_dir.glob("*.jsonl")):
try:
stat = session_file.stat()
except OSError:
continue
if stat.st_size == 0:
continue
conv_id = session_file.stem
prev = by_id.get(conv_id)
if prev is not None and prev[0] >= stat.st_mtime:
logger.debug(
"[claude-code] Duplicate session %s in %s; keeping newer copy",
conv_id[:8], root,
)
continue
self._path_map[conv_id] = session_file
title, project, created = _read_session_meta(session_file)
by_id[conv_id] = (
stat.st_mtime,
{
"id": conv_id,
"title": title,
"project": project,
# The --project filter and dry-run table read the
# listing dict, not the normalized conversation.
"_project_name": project,
"created_at": created,
"updated_at": datetime.fromtimestamp(
stat.st_mtime, tz=timezone.utc
).isoformat(),
"_path": str(session_file),
"_project_dir": proj_dir.name,
},
)
# Deterministic order: by bucket dir, then session id.
return sorted(
(conv for _, conv in by_id.values()),
key=lambda c: (c["_project_dir"], c["id"]),
)
# ---------------------------------------------------------------------------
# Internal helpers
# ---------------------------------------------------------------------------
def _read_session_meta(path: Path) -> tuple[str, str | None, str]:
"""Light single-pass scan for listing metadata: (title, project, created_at).
Substring guards keep this cheap — only candidate lines are JSON-parsed.
The full-fidelity extraction happens later in normalize_conversation.
"""
title = ""
project: str | None = None
created = ""
try:
with path.open(encoding="utf-8") as fh:
for line in fh:
if '"ai-title"' in line:
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
if rec.get("type") == "ai-title" and rec.get("aiTitle"):
title = str(rec["aiTitle"]) # last one wins
continue
if (not created or project is None) and (
'"timestamp"' in line or '"cwd"' in line
):
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
if not created and rec.get("timestamp"):
created = str(rec["timestamp"])
if project is None and rec.get("cwd"):
project = Path(rec["cwd"]).name or None
except OSError as e:
logger.warning("[claude-code] Could not read %s: %s", path, e)
return title, project, created
def _extract_title(records: list[dict]) -> str:
"""Last ai-title record wins; fall back to the first real user prompt."""
title = ""
for rec in records:
if rec.get("type") == "ai-title" and rec.get("aiTitle"):
title = str(rec["aiTitle"])
if title:
return title
for rec in records:
if rec.get("type") != "user" or rec.get("isMeta") or rec.get("isSidechain"):
continue
content = (rec.get("message") or {}).get("content")
if isinstance(content, str):
text = _strip_harness_noise(content)
elif isinstance(content, list):
text = " ".join(
_strip_harness_noise(item.get("text", ""))
for item in content
if isinstance(item, dict) and item.get("type") == "text"
).strip()
else:
text = ""
if text:
return text[:80]
return "Untitled session"
def _extract_project(records: list[dict], path: str | None) -> str | None:
"""Project = basename of the session's working directory."""
for rec in records:
cwd = rec.get("cwd")
if cwd:
name = Path(cwd).name
if name:
return name
# Fallback: the munged directory name (cannot be reliably de-munged
# because '-' is both the path separator and a legal name character).
if path:
return Path(path).parent.name.lstrip("-") or None
return None
def _parse_jsonl(path: Path) -> list[dict]:
"""Read a JSONL file into records, tolerating (and logging) bad lines."""
records: list[dict] = []
bad_lines = 0
try:
text = path.read_text(encoding="utf-8")
except OSError as e:
logger.warning("[claude-code] Could not read %s: %s", path, e)
return records
# split("\n"), not splitlines(): splitlines() also breaks on U+0085,
# U+2028/9 and friends, which are legal *inside* a JSON string. A NEL in
# captured command output shreds one record into unparseable fragments and
# loses it silently (observed 2026-08-18 in a real Codex rollout).
for line in text.split("\n"):
line = line.strip()
if not line:
continue
try:
records.append(json.loads(line))
except json.JSONDecodeError:
bad_lines += 1
if bad_lines:
logger.warning(
"[claude-code] %s: skipped %d unparseable line(s)", path.name, bad_lines
)
return records
def _load_subagents(session_path: Path) -> dict:
"""Map spawning tool_use id → ``{"meta": {...}, "records": [...]}``.
Claude Code writes subagent transcripts to
``<session>/subagents/agent-*.jsonl`` with an ``agent-*.meta.json`` sidecar
carrying ``toolUseId`` (which parent ``Task``/``Agent`` call spawned it).
Files without a readable meta (no id to position them) are skipped.
"""
submap: dict[str, dict] = {}
subdir = session_path.parent / session_path.stem / "subagents"
if not subdir.is_dir():
return submap
for jf in sorted(subdir.glob("*.jsonl")):
meta_path = jf.with_suffix(".meta.json")
try:
meta = json.loads(meta_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
logger.warning(
"[claude-code] Subagent %s has no readable meta; skipping", jf.name
)
continue
tool_id = meta.get("toolUseId")
if not tool_id:
continue
submap[tool_id] = {"meta": meta, "records": _parse_jsonl(jf)}
return submap
def _extract_launch_cwd(records: list[dict]) -> str | None:
"""The session's working directory (constant per session; first cwd wins)."""
for rec in records:
cwd = rec.get("cwd")
if cwd:
return cwd
return None
# tool_use input keys that carry a file path.
_TOOL_PATH_KEYS = ("file_path", "path", "notebook_path")
# Optional ignore-list: git repos to never tag (comma-separated names). Rarely
# needed with git-root detection (config/reference dirs are already excluded
# because they aren't git repos), but kept as an escape hatch.
def _ignored_repos() -> set[str]:
env = os.getenv("CLAUDE_CODE_REPO_TAG_IGNORE", "")
return {s.strip() for s in env.split(",") if s.strip()}
# Absolute path-like tokens inside Bash command strings (file_path/path keys are
# matched directly). git-root resolution short-circuits on non-repo paths.
_ABS_PATH_RE = re.compile(r"/(?:[\w.\-]+/)*[\w.\-]+")
def _repos_touched(
records: list[dict], launch_cwd: str, cap: int = 3, ignore: set[str] | None = None
) -> list[str]:
"""Repos a session touched, for the title's ``[repo-a, repo-b]`` tag.
A file's repo is the **git repository it lives in** (nearest ancestor with a
``.git``), resolved for every absolute path seen in a tool_use input
(``file_path``/``path``/``notebook_path`` and absolute home paths inside Bash
``command`` strings). This works anywhere in the filesystem — not just under
the launch directory — so cross-workspace work is captured, and non-repo
noise (config dirs, one-off files, reference dirs) is excluded because it
isn't a git repo. Ordered by touch frequency, capped with a trailing ``…``.
"""
ignore = _ignored_repos() if ignore is None else ignore
counts: Counter = Counter()
seen: set[str] = set()
def note(p) -> None:
if not isinstance(p, str) or not p:
return
if not p.startswith("/"): # resolve relative paths against the launch cwd
if not launch_cwd:
return
p = str(Path(launch_cwd) / p)
if p in seen:
return
seen.add(p)
name = git_root_name(Path(p))
if name and not name.startswith(".") and name not in ignore:
counts[name] += 1
for rec in records:
msg = rec.get("message") or {}
content = msg.get("content")
if not isinstance(content, list):
continue
for item in content:
if not isinstance(item, dict) or item.get("type") != "tool_use":
continue
inp = item.get("input")
if not isinstance(inp, dict):
continue
for key in _TOOL_PATH_KEYS:
note(inp.get(key))
cmd = inp.get("command")
if isinstance(cmd, str):
for m in _ABS_PATH_RE.finditer(cmd):
note(m.group(0))
ordered = [name for name, _ in counts.most_common()]
if len(ordered) > cap:
return ordered[:cap] + ["…"]
return ordered
def _extract_messages(
records: list[dict],
conv_id: str,
report: LossReport,
policy: str,
subagents: dict | None = None,
include_sidechain: bool = False,
expanding: frozenset[str] = frozenset(),
) -> list[dict]:
"""Normalize Claude Code records into messages.
``subagents`` maps a spawning ``Task``/``Agent`` tool_use id to its separate
transcript ``{"meta": ..., "records": ...}``; when a matching tool_use is
seen its subagent is folded inline as a subagent block (extracted
recursively under the same policy). ``include_sidechain`` is set True for
those recursive subagent passes (subagent records are flagged
``isSidechain``); the top-level pass keeps skipping sidechain records so a
subagent is never also emitted as a stray top-level turn.
``expanding`` holds the spawn ids of the subagents being folded around
this pass. A ``fork`` subagent's transcript opens with a copy of the parent
turn that spawned it, its own spawn call included; that copy is dropped,
since the enclosing subagent block already stands for it. Expanding it
again recursed without end (RecursionError, every daily sync from
2026-09-24).
"""
subagents = subagents or {}
messages: list[dict] = []
# Pending collapsed tool activity: name → call count, plus total bytes.
pending_tools: Counter = Counter()
pending_bytes = 0
def flush_pending() -> None:
nonlocal pending_bytes
if not pending_tools:
return
shown = ", ".join(
f"{name} ×{count}" for name, count in pending_tools.most_common(_TOOL_NAMES_SHOWN)
)
if len(pending_tools) > _TOOL_NAMES_SHOWN:
shown += ", …"
calls = sum(pending_tools.values())
block = make_collapsed_block(
origin=f"{calls} calls: {shown}",
content_type="tool_activity",
size_bytes=pending_bytes,
kind=COLLAPSED_KIND_TOOL_DUMP,
)
if messages and messages[-1]["role"] == "assistant":
messages[-1]["blocks"].append(block)
else:
messages.append(
{"role": "tool", "content_type": "text", "timestamp": None, "blocks": [block]}
)
pending_tools.clear()
pending_bytes = 0
for rec in records:
if rec.get("type") not in ("user", "assistant"):
continue
if rec.get("isMeta"):
continue
if rec.get("isSidechain") and not include_sidechain:
continue
msg = rec.get("message") or {}
role = msg.get("role") or rec["type"]
content = msg.get("content")
blocks: list[dict] = []
# This record's own tool traffic — merged into pending only after the
# record's message is appended, so the placeholder lands on it (not on
# the previous message).
local_tools: Counter = Counter()
local_bytes = 0
if isinstance(content, str):
text = _strip_harness_noise(content)
block = make_text_block(text)
if block:
blocks.append(block)
elif isinstance(content, list):
for item in content:
if not isinstance(item, dict):
continue
item_type = item.get("type", "")
if item_type == "text":
block = make_text_block(_strip_harness_noise(item.get("text", "")))
if block:
blocks.append(block)
elif item_type in ("thinking", "redacted_thinking"):
if policy == HIDDEN_CONTENT_FULL:
block = make_thinking_block(
item.get("thinking") or item.get("text") or ""
)
if block:
blocks.append(block)
else:
# Decision 2026-06-12: thinking is dropped without a
# placeholder, but stays visible in the run summary.
report.record_collapsed(
"thinking", len(json.dumps(item, default=str))
)
elif item_type == "tool_use":
name = item.get("name") or "tool"
tool_id = item.get("id")
if tool_id in expanding:
continue
if name in _SUBAGENT_TOOL_NAMES and tool_id in subagents:
# A Task/Agent spawn: fold its separate transcript inline
# instead of collapsing it. Its own tool traffic is
# collapsed by the recursive pass under the same policy.
sub = subagents[tool_id]
meta = sub.get("meta") or {}
sub_msgs = _extract_messages(
sub.get("records") or [],
conv_id,
report,
policy,
subagents,
include_sidechain=True,
expanding=expanding | {tool_id},
)
blocks.append(
make_subagent_block(
agent_type=meta.get("agentType") or name,
description=meta.get("description") or "",
messages=sub_msgs,
)
)
elif policy == HIDDEN_CONTENT_FULL:
blocks.append(
make_tool_use_block(
item.get("name", ""), item.get("input"), item.get("id")
)
)
else:
size = len(json.dumps(item, default=str))
local_tools[name] += 1
local_bytes += size
report.record_collapsed(name, size)
elif item_type == "tool_result":
if policy == HIDDEN_CONTENT_FULL:
blocks.append(
make_tool_result_block(
_stringify_tool_result(item.get("content")),
is_error=bool(item.get("is_error")),
)
)
else:
size = len(json.dumps(item, default=str))
local_bytes += size
report.record_collapsed("tool_result", size)
elif item_type == "image":
blocks.append(
make_image_placeholder(ref="embedded image", source="user_upload")
)
else:
logger.warning(
"[claude-code] Unknown content block type %r in session %s",
item_type,
conv_id[:8],
)
report.record_unknown(f"claude-code.{item_type or '?'}")
blocks.append(
make_unknown_block(
raw_type=f"claude-code.{item_type or '?'}",
observed_keys=list(item.keys()),
reason=UNKNOWN_REASON_UNKNOWN_TYPE,
)
)
if not blocks:
# Tool-result-only records (and similar): traffic accumulates and
# is attached to the message that initiated it.
pending_tools.update(local_tools)
pending_bytes += local_bytes
continue
# Attach previous records' tool activity to the previous message
# before starting a new one, so the placeholder lands between the
# dialogue turns it actually occurred between.
flush_pending()
messages.append(
{
"role": role,
"content_type": "text",
"timestamp": rec.get("timestamp"),
"blocks": blocks,
}
)
pending_tools.update(local_tools)
pending_bytes += local_bytes
flush_pending()
return messages
def _stringify_tool_result(content) -> str:
"""tool_result content may be a string or a list of text blocks."""
if isinstance(content, str):
return content
if isinstance(content, list):
parts = []
for item in content:
if isinstance(item, dict) and item.get("type") == "text":
parts.append(item.get("text", ""))
else:
parts.append(json.dumps(item, default=str))
return "\n".join(parts)
return json.dumps(content, default=str) if content is not None else ""