feat: v0.6.0 — collapse policy, session limiter, Claude Code provider, prune, browser auth, media downloads

This commit is contained in:
JesseMarkowitz
2026-06-12 18:26:14 -04:00
parent 557994f7d9
commit 9e1a8ab7cb
23 changed files with 3133 additions and 197 deletions
+63 -3
View File
@@ -11,6 +11,7 @@ at the call site — see plan §Data-loss visibility.
"""
import json
from pathlib import Path
from typing import Any
BLOCK_TYPE_TEXT = "text"
@@ -23,6 +24,10 @@ BLOCK_TYPE_IMAGE_PLACEHOLDER = "image_placeholder"
BLOCK_TYPE_FILE_PLACEHOLDER = "file_placeholder"
BLOCK_TYPE_UNKNOWN = "unknown"
BLOCK_TYPE_HIDDEN_CONTEXT_MARKER = "hidden_context_marker"
BLOCK_TYPE_COLLAPSED = "collapsed"
COLLAPSED_KIND_TOOL_DUMP = "tool_dump"
COLLAPSED_KIND_HIDDEN_CONTEXT = "hidden_context"
UNKNOWN_REASON_UNKNOWN_TYPE = "unknown_type"
UNKNOWN_REASON_EXTRACTION_FAILED = "extraction_failed"
@@ -164,6 +169,30 @@ def make_unknown_block(
}
def make_collapsed_block(
origin: str,
content_type: str,
size_bytes: int,
kind: str,
) -> dict:
"""One-line stand-in for a message omitted by the EXPORTER_HIDDEN_CONTENT policy.
``kind`` ∈ {COLLAPSED_KIND_TOOL_DUMP, COLLAPSED_KIND_HIDDEN_CONTEXT}.
Unlike ``unknown`` blocks (unexpected loss), a collapsed block records an
intentional policy decision — the content was invisible in the provider's
web UI and the user chose not to keep it. Every call site MUST tally via
``LossReport.record_collapsed`` so the omission stays visible in the
post-export summary.
"""
return {
"type": BLOCK_TYPE_COLLAPSED,
"origin": origin or "",
"content_type": content_type or "",
"size_bytes": int(size_bytes or 0),
"kind": kind,
}
def make_hidden_context_marker(content_type: str) -> dict:
"""A short prepend block that flags the surrounding message as hidden context.
@@ -246,6 +275,10 @@ def _render_one(block: dict) -> str:
ref = block.get("ref", "")
source = block.get("source", "unknown")
mime = block.get("mime")
local_path = block.get("local_path")
if local_path:
# Downloaded — render as a real inline image.
return f"![{source or 'image'}]({local_path})"
meta_parts = [source] if source else []
if mime:
meta_parts.append(mime)
@@ -259,14 +292,20 @@ def _render_one(block: dict) -> str:
mime = block.get("mime")
size_bytes = block.get("size_bytes")
duration = block.get("duration_seconds")
local_path = block.get("local_path")
meta_parts: list[str] = []
if mime:
meta_parts.append(mime)
if isinstance(size_bytes, int) and size_bytes > 0:
kb = size_bytes / 1024
meta_parts.append(f"{kb:.1f} KB" if kb < 1024 else f"{kb / 1024:.2f} MB")
size_label = _format_size(size_bytes)
if size_label:
meta_parts.append(size_label)
if isinstance(duration, (int, float)) and duration > 0:
meta_parts.append(f"{duration:.2f}s")
if local_path:
# Downloaded — render as a link to the local copy.
meta = ", ".join(meta_parts) if meta_parts else ""
meta_str = f" ({meta})" if meta else ""
return f"> 📎 **File attached** — [{Path(local_path).name}]({local_path}){meta_str}"
meta_parts.append("content not preserved in this export")
meta = ", ".join(meta_parts)
return f"> 📎 **File attached** — `{label}` ({meta})"
@@ -286,6 +325,19 @@ def _render_one(block: dict) -> str:
if btype == BLOCK_TYPE_HIDDEN_CONTEXT_MARKER:
ctype = block.get("content_type", "")
return f"> ℹ️ **Hidden context** — `{ctype}`"
if btype == BLOCK_TYPE_COLLAPSED:
origin = block.get("origin", "")
kind = block.get("kind", "")
if kind == COLLAPSED_KIND_HIDDEN_CONTEXT:
icon, label = "ℹ️", "Hidden context"
else:
icon, label = "🔧", "Tool output"
size_label = _format_size(block.get("size_bytes"))
size_part = f" ({size_label})" if size_label else ""
return (
f"> {icon} **{label}** — `{origin}`{size_part} — omitted "
"(EXPORTER_HIDDEN_CONTENT=full to keep)"
)
# Defensive: a block of unrecognised local type (shouldn't happen if
# constructors are used). Render as visible warning rather than dropping.
@@ -297,6 +349,14 @@ def _render_one(block: dict) -> str:
# ---------------------------------------------------------------------------
def _format_size(size_bytes: Any) -> str:
"""Human-readable size ('24.1 KB', '1.05 MB'), or '' when absent/zero."""
if not isinstance(size_bytes, int) or size_bytes <= 0:
return ""
kb = size_bytes / 1024
return f"{kb:.1f} KB" if kb < 1024 else f"{kb / 1024:.2f} MB"
def _safe_fence(text: str) -> str:
"""Return a backtick fence longer than the longest run of backticks in ``text``.
+101
View File
@@ -0,0 +1,101 @@
"""Extract session tokens directly from a local browser's cookie store.
Default browser is Brave. Uses browser-cookie3, which handles the
platform-specific cookie decryption (Linux: AES key derived from the OS
keyring "Safe Storage" secret; Windows: DPAPI; macOS: Keychain). The browser
must be the one the user is actually logged in with, on the same machine.
Failure modes worth knowing:
- Browser not installed / profile not found → BrowserTokenError
- Cookie DB locked (browser running with exclusive lock; rare on Linux,
common on Windows) → BrowserTokenError suggesting the browser be closed
- Logged out / cookie expired → BrowserTokenError naming the missing cookie
"""
import logging
logger = logging.getLogger(__name__)
DEFAULT_BROWSER = "brave"
SUPPORTED_BROWSERS = ("brave", "chrome", "chromium", "edge", "firefox")
# ChatGPT splits large session tokens across two cookies to stay under the
# browser's 4KB cookie limit; small tokens use the unsuffixed name.
_CHATGPT_COOKIE_0 = "__Secure-next-auth.session-token.0"
_CHATGPT_COOKIE_1 = "__Secure-next-auth.session-token.1"
_CHATGPT_COOKIE_SINGLE = "__Secure-next-auth.session-token"
_CLAUDE_COOKIE = "sessionKey"
class BrowserTokenError(Exception):
"""Raised when tokens cannot be extracted from the browser profile."""
def _load_cookies(browser: str, domain: str) -> dict[str, str]:
"""Return {cookie_name: value} for ``domain`` from the given browser."""
if browser not in SUPPORTED_BROWSERS:
raise BrowserTokenError(
f"Unsupported browser {browser!r}. "
f"Supported: {', '.join(SUPPORTED_BROWSERS)}"
)
import browser_cookie3
loader = getattr(browser_cookie3, browser)
try:
jar = loader(domain_name=domain)
except Exception as e:
# browser_cookie3 raises BrowserCookieError plus assorted OS/keyring
# errors. All mean the same thing to the caller: fall back to manual.
hint = ""
text = str(e).lower()
if "lock" in text or "database is locked" in text:
hint = " Close the browser and try again."
elif "not find" in text or "no such file" in text:
hint = " Is the browser installed on this machine?"
raise BrowserTokenError(
f"Could not read {browser} cookies for {domain}: {e}.{hint}"
) from e
return {cookie.name: cookie.value or "" for cookie in jar}
def extract_chatgpt_tokens(browser: str = DEFAULT_BROWSER) -> tuple[str, str | None]:
"""Return (session_token_0, session_token_1_or_None) from the browser.
Raises BrowserTokenError when no session cookie is present (not logged in,
or logged out since the last visit).
"""
cookies = _load_cookies(browser, "chatgpt.com")
token_0 = cookies.get(_CHATGPT_COOKIE_0, "").strip()
token_1 = cookies.get(_CHATGPT_COOKIE_1, "").strip() or None
if token_0:
logger.info(
"[browser-tokens] ChatGPT session cookies found in %s (chunked: %s)",
browser,
"yes" if token_1 else "no",
)
return token_0, token_1
single = cookies.get(_CHATGPT_COOKIE_SINGLE, "").strip()
if single:
logger.info("[browser-tokens] ChatGPT session cookie found in %s (single)", browser)
return single, None
raise BrowserTokenError(
f"No ChatGPT session cookie in {browser} — open https://chatgpt.com "
"in that browser and log in, then retry."
)
def extract_claude_session(browser: str = DEFAULT_BROWSER) -> str:
"""Return the Claude ``sessionKey`` cookie value from the browser."""
cookies = _load_cookies(browser, "claude.ai")
session_key = cookies.get(_CLAUDE_COOKIE, "").strip()
if session_key:
logger.info("[browser-tokens] Claude sessionKey found in %s", browser)
return session_key
raise BrowserTokenError(
f"No Claude sessionKey cookie in {browser} — open https://claude.ai "
"in that browser and log in, then retry."
)
+35
View File
@@ -173,6 +173,22 @@ class Cache:
entry["joplin_synced_at"] = datetime.now(tz=timezone.utc).isoformat()
self._save()
def get_joplin_resources(self, provider: str, conv_id: str) -> dict[str, str]:
"""Return the {relative_media_path: joplin_resource_id} map for a conversation."""
entry = self._data.get(provider, {}).get(conv_id)
if not isinstance(entry, dict):
return {}
resources = entry.get("joplin_resources")
return dict(resources) if isinstance(resources, dict) else {}
def set_joplin_resources(self, provider: str, conv_id: str, resources: dict[str, str]) -> None:
"""Persist the {relative_media_path: joplin_resource_id} map for a conversation."""
entry = self._data.get(provider, {}).get(conv_id)
if entry is None:
return
entry["joplin_resources"] = dict(resources)
self._save()
def get_joplin_pending(self, provider: str) -> list[tuple[str, dict]]:
"""Return (conv_id, entry) pairs that need to be synced to Joplin.
@@ -210,6 +226,25 @@ class Cache:
return pending
def all_file_paths(self) -> set[str]:
"""Return every non-empty ``file_path`` in the manifest, across providers.
Used by ``prune`` (files on disk but not here are stale) and by the
doctor integrity check (files here but not on disk are missing).
"""
paths: set[str] = set()
for key, entries in self._data.items():
if not isinstance(entries, dict) or key in (
"version",
"last_run",
"tos_acknowledged_at",
):
continue
for entry in entries.values():
if isinstance(entry, dict) and entry.get("file_path"):
paths.add(entry["file_path"])
return paths
def last_run(self) -> str | None:
"""Return the ISO8601 timestamp of the last export run, or None."""
return self._data.get("last_run")
+75 -1
View File
@@ -20,6 +20,12 @@ _CLAUDE_PLACEHOLDER = ""
# Valid OUTPUT_STRUCTURE values
VALID_STRUCTURES = {"provider/project/year", "provider/project", "provider/year"}
# Valid EXPORTER_HIDDEN_CONTENT values (see src/providers/base.py)
VALID_HIDDEN_CONTENT = {"full", "placeholder", "omit"}
# Valid EXPORTER_DOWNLOAD_MEDIA values (see src/media.py)
VALID_DOWNLOAD_MEDIA = {"images", "all", "off"}
class ConfigError(Exception):
"""Raised when required configuration is missing or invalid."""
@@ -43,6 +49,16 @@ class Config:
# Joplin local REST API settings (Web Clipper service)
joplin_api_token: str | None = None
joplin_api_url: str = "http://localhost:41184"
# Policy for content invisible in the provider web UI (retrieval dumps,
# hidden context): full | placeholder | omit
hidden_content: str = "placeholder"
# Session cap: max conversations downloaded per export run (None = unlimited).
# Every run is resumable, so a capped run just continues next time.
max_conversations: int | None = None
# Seconds between consecutive API requests (politeness pacing; 0 disables)
request_delay: float = 1.0
# Asset download policy: images | all | off
download_media: str = "images"
def load_config() -> Config:
@@ -67,6 +83,12 @@ def load_config() -> Config:
joplin_token = os.getenv("JOPLIN_API_TOKEN", "").strip() or None
joplin_url = os.getenv("JOPLIN_API_URL", "http://localhost:41184").strip()
hidden_content = os.getenv("EXPORTER_HIDDEN_CONTENT", "").strip().lower() or "placeholder"
max_conversations_raw = os.getenv("MAX_CONVERSATIONS_PER_RUN", "").strip()
request_delay_raw = os.getenv("REQUEST_DELAY", "").strip()
download_media = os.getenv("EXPORTER_DOWNLOAD_MEDIA", "").strip().lower() or "images"
# Parse CHATGPT_PROJECT_IDS — comma-separated list of gizmo IDs (g-p-xxx)
_project_ids_raw = os.getenv("CHATGPT_PROJECT_IDS", "").strip()
chatgpt_project_ids = [
@@ -90,6 +112,46 @@ def load_config() -> Config:
f"Must be one of: {', '.join(sorted(VALID_STRUCTURES))}"
)
# Validate hidden-content policy
if hidden_content not in VALID_HIDDEN_CONTENT:
errors.append(
f"EXPORTER_HIDDEN_CONTENT '{hidden_content}' is invalid. "
f"Must be one of: {', '.join(sorted(VALID_HIDDEN_CONTENT))}"
)
# Validate media download policy
if download_media not in VALID_DOWNLOAD_MEDIA:
errors.append(
f"EXPORTER_DOWNLOAD_MEDIA '{download_media}' is invalid. "
f"Must be one of: {', '.join(sorted(VALID_DOWNLOAD_MEDIA))}"
)
# Validate session cap
max_conversations: int | None = None
if max_conversations_raw:
try:
max_conversations = int(max_conversations_raw)
except ValueError:
errors.append(
f"MAX_CONVERSATIONS_PER_RUN '{max_conversations_raw}' is not an integer."
)
else:
if max_conversations < 1:
errors.append(
f"MAX_CONVERSATIONS_PER_RUN must be at least 1 (got {max_conversations})."
)
# Validate request pacing
request_delay = 1.0
if request_delay_raw:
try:
request_delay = float(request_delay_raw)
except ValueError:
errors.append(f"REQUEST_DELAY '{request_delay_raw}' is not a number.")
else:
if request_delay < 0:
errors.append(f"REQUEST_DELAY must be >= 0 (got {request_delay}).")
# Validate and decode ChatGPT JWT
chatgpt_expiry: datetime | None = None
if chatgpt_token:
@@ -139,6 +201,10 @@ def load_config() -> Config:
chatgpt_project_ids=chatgpt_project_ids,
joplin_api_token=joplin_token,
joplin_api_url=joplin_url,
hidden_content=hidden_content,
max_conversations=max_conversations,
request_delay=request_delay,
download_media=download_media,
)
_log_startup_summary(config)
@@ -223,7 +289,11 @@ def _log_startup_summary(cfg: Config) -> None:
"Joplin: %s | "
"export_dir=%s | "
"structure=%s | "
"cache_dir=%s",
"cache_dir=%s | "
"hidden_content=%s | "
"max_conversations=%s | "
"request_delay=%.1fs | "
"download_media=%s",
chatgpt_status,
claude_status,
len(cfg.chatgpt_project_ids),
@@ -231,4 +301,8 @@ def _log_startup_summary(cfg: Config) -> None:
cfg.export_dir,
cfg.output_structure,
cfg.cache_dir,
cfg.hidden_content,
cfg.max_conversations if cfg.max_conversations is not None else "unlimited",
cfg.request_delay,
cfg.download_media,
)
+86
View File
@@ -1,7 +1,10 @@
"""Joplin Data API client for importing notes into Joplin desktop."""
import json
import logging
import os
import re
from pathlib import Path
from typing import Any
import requests
@@ -190,6 +193,46 @@ class JoplinClient:
self._put(f"/notes/{note_id}", {"title": title, "body": body})
logger.info("[joplin] Note updated: %r (%s)", title, note_id)
# ------------------------------------------------------------------
# Resources (embedded media)
# ------------------------------------------------------------------
def create_resource(self, file_path: "Path", title: str | None = None) -> str:
"""Upload a local file as a Joplin resource and return its ID.
Joplin's ``POST /resources`` is multipart: a ``data`` file part plus a
``props`` JSON part. The returned ID is referenced from note bodies as
``:/<id>``.
"""
url = f"{self._base_url}/resources"
props = json.dumps({"title": title or file_path.name})
logger.debug("[joplin] POST /resources (%s)", file_path.name)
try:
with file_path.open("rb") as fh:
resp = requests.post(
url,
params={"token": self._token},
files={
"data": (file_path.name, fh),
"props": (None, props),
},
timeout=_REQUEST_TIMEOUT,
)
resp.raise_for_status()
resource_id = resp.json()["id"]
logger.info("[joplin] Resource created: %s → %s", file_path.name, resource_id)
return resource_id
except requests.exceptions.ConnectionError as e:
raise JoplinError(
"Cannot connect to Joplin. Is Joplin desktop running with Web Clipper enabled?"
) from e
except requests.exceptions.Timeout as e:
raise JoplinError(_timeout_message("POST", "/resources")) from e
except requests.exceptions.HTTPError as e:
raise JoplinError(_http_error_message("POST", "/resources", e)) from e
except (requests.exceptions.RequestException, KeyError, OSError) as e:
raise JoplinError(f"Joplin resource upload failed for {file_path.name}: {e}") from e
# ------------------------------------------------------------------
# HTTP helpers
# ------------------------------------------------------------------
@@ -304,6 +347,46 @@ def _http_error_message(method: str, path: str, e: requests.exceptions.HTTPError
return f"Joplin {method} {path} failed: HTTP {status}{body_snippet}"
# ------------------------------------------------------------------
# Embedded-media rewriting
# ------------------------------------------------------------------
# Markdown image/link targets pointing at the local media/ sibling directory,
# e.g. ![alt](media/file_x.png) or [name](media/clip.wav).
_MEDIA_LINK_RE = re.compile(r"(!?\[[^\]]*\])\((media/[^)\s]+)\)")
def upload_media_and_rewrite(
body: str,
note_dir: Path,
client: "JoplinClient",
resource_map: dict[str, str],
) -> tuple[str, dict[str, str]]:
"""Rewrite local ``media/…`` links to Joplin ``:/resourceId`` references.
Uploads each referenced file as a Joplin resource (once), rewriting both
image embeds and file links. ``resource_map`` (relative-path → resource_id)
is the idempotency record from the cache: known paths are reused, new ones
are added and returned so the caller can persist them. Missing files are
left as-is so the link still points at the on-disk copy.
"""
new_map = dict(resource_map)
def replace(match: re.Match) -> str:
label, rel_path = match.group(1), match.group(2)
resource_id = new_map.get(rel_path)
if resource_id is None:
file_path = (note_dir / rel_path).resolve()
if not file_path.is_file():
logger.warning("[joplin] Media file missing, leaving link: %s", rel_path)
return match.group(0)
resource_id = client.create_resource(file_path)
new_map[rel_path] = resource_id
return f"{label}(:/{resource_id})"
return _MEDIA_LINK_RE.sub(replace, body), new_map
# ------------------------------------------------------------------
# Notebook naming helper
# ------------------------------------------------------------------
@@ -312,6 +395,9 @@ def _http_error_message(method: str, path: str, e: requests.exceptions.HTTPError
_PROVIDER_DISPLAY = {
"chatgpt": "AI-ChatGPT",
"claude": "AI-Claude",
# Decision 2026-06-12: Claude Code coding projects nest under the same
# AI-Claude parent, alongside Claude web projects.
"claude-code": "AI-Claude",
}
+38
View File
@@ -23,11 +23,20 @@ class LossReport:
unknown_blocks: Counter = field(default_factory=Counter)
extraction_failures: Counter = field(default_factory=Counter)
filtered_roles: Counter = field(default_factory=Counter)
# Messages collapsed/omitted by the EXPORTER_HIDDEN_CONTENT policy —
# intentional, but still surfaced so the omission is never invisible.
collapsed: Counter = field(default_factory=Counter)
collapsed_bytes: int = 0
# Aggregate counters
messages_rendered: int = 0
conversations: int = 0
# Media downloads (EXPORTER_DOWNLOAD_MEDIA): successes / skips / failures
media_downloaded: int = 0
media_downloaded_bytes: int = 0
media_failed: Counter = field(default_factory=Counter)
# Recording -------------------------------------------------------------
def record_unknown(self, raw_type: str) -> None:
@@ -39,6 +48,19 @@ class LossReport:
def record_filtered_role(self, role: str) -> None:
self.filtered_roles[role or "?"] += 1
def record_collapsed(self, origin: str, size_bytes: int = 0) -> None:
self.collapsed[origin or "?"] += 1
if isinstance(size_bytes, int) and size_bytes > 0:
self.collapsed_bytes += size_bytes
def record_media_downloaded(self, size_bytes: int = 0) -> None:
self.media_downloaded += 1
if isinstance(size_bytes, int) and size_bytes > 0:
self.media_downloaded_bytes += size_bytes
def record_media_failed(self, reason: str) -> None:
self.media_failed[reason or "?"] += 1
def record_message(self) -> None:
self.messages_rendered += 1
@@ -58,6 +80,22 @@ class LossReport:
lines.append(f" messages rendered: {self.messages_rendered}")
lines.extend(_format_section("unknown blocks: ", self.unknown_blocks))
lines.extend(_format_section("extraction failures: ", self.extraction_failures))
lines.extend(_format_section("collapsed by policy: ", self.collapsed))
if self.collapsed:
lines.append(
f" ≈{self.collapsed_bytes / 1024:.0f} KB omitted "
"(intentional — EXPORTER_HIDDEN_CONTENT=full to keep)"
)
if self.media_downloaded or self.media_failed:
lines.append(
f" media downloaded: {self.media_downloaded} "
f"(≈{self.media_downloaded_bytes / 1024:.0f} KB)"
)
if self.media_failed:
total_failed = sum(self.media_failed.values())
lines.append(f" media failed: {total_failed}")
for reason, count in self.media_failed.most_common(_TOP_N_BREAKDOWN):
lines.append(f" {reason}={count}")
lines.append(
" filtered roles: "
"(filter lifted in v0.4.0 — counter retained for future use, expected 0)"
+363 -41
View File
@@ -2,6 +2,7 @@
import importlib.metadata
import logging
import os
import platform
import shutil
import sys
@@ -109,17 +110,37 @@ def cli(ctx: click.Context, verbose: bool, quiet: bool, debug: bool, no_log_file
@cli.command()
@click.option(
"--from-browser",
"from_browser",
is_flag=False,
flag_value="brave",
default=None,
metavar="[BROWSER]",
help=(
"Extract tokens straight from a local browser's cookie store instead "
"of the manual DevTools flow. Defaults to brave when no browser is "
"named; also supports chrome, chromium, edge, firefox. The browser "
"must be installed and logged in on this machine."
),
)
@click.pass_context
def auth(ctx: click.Context) -> None:
def auth(ctx: click.Context, from_browser: str | None) -> None:
"""Interactive setup wizard for session tokens.
Guides you through finding and saving your ChatGPT and Claude session
tokens. Tokens are never echoed to the terminal.
tokens. Tokens are never echoed to the terminal. With --from-browser,
tokens are pulled from the browser's cookie store and written to .env
without any copy-pasting.
Token lifetimes:
ChatGPT (__Secure-next-auth.session-token): ~7 days (JWT)
Claude (sessionKey): ~30 days (opaque string)
"""
if from_browser:
_auth_from_browser(from_browser.lower())
return
os_name = platform.system()
console.print("\n[bold cyan]AI Chat Exporter — Token Setup Wizard[/bold cyan]\n")
@@ -279,38 +300,117 @@ def _auth_claude(os_name: str) -> None:
_write_token_to_env("CLAUDE_SESSION_KEY", key)
def _auth_from_browser(browser: str) -> None:
"""Extract ChatGPT + Claude tokens from a local browser and write .env.
Each provider is validated against its live API before anything is
written, so a stale cookie (logged out since last visit) never replaces
a working token. Exits 1 only when nothing could be configured.
"""
from src.browser_tokens import (
BrowserTokenError,
SUPPORTED_BROWSERS,
extract_chatgpt_tokens,
extract_claude_session,
)
if browser not in SUPPORTED_BROWSERS:
err_console.print(
f"[red]Unsupported browser '{browser}'. "
f"Supported: {', '.join(SUPPORTED_BROWSERS)}[/red]"
)
sys.exit(1)
console.print(f"\n[bold cyan]Extracting session tokens from {browser}…[/bold cyan]\n")
configured = 0
# ── ChatGPT ──────────────────────────────────────────────────────────
try:
token, token_1 = extract_chatgpt_tokens(browser)
with console.status("[dim]Validating ChatGPT token…[/dim]"):
from src.providers.chatgpt import ChatGPTProvider
prov = ChatGPTProvider(session_token=token, session_token_1=token_1)
prov._fetch_access_token()
_set_env_key("CHATGPT_SESSION_TOKEN", token)
_set_env_key("CHATGPT_SESSION_TOKEN_1", token_1 or "")
console.print("[green]ChatGPT: token extracted, validated, and saved.[/green]")
configured += 1
except BrowserTokenError as e:
console.print(f"[yellow]ChatGPT: {e}[/yellow]")
except ProviderError as e:
console.print(
f"[yellow]ChatGPT: extracted cookie failed live validation "
f"({e.original}) — .env not changed. Log in to chatgpt.com in "
f"{browser} and retry.[/yellow]"
)
# ── Claude ───────────────────────────────────────────────────────────
try:
session_key = extract_claude_session(browser)
with console.status("[dim]Validating Claude session key…[/dim]"):
from src.providers.claude import ClaudeProvider
prov = ClaudeProvider(session_key)
prov.list_conversations(offset=0, limit=1)
_set_env_key("CLAUDE_SESSION_KEY", session_key)
console.print("[green]Claude: session key extracted, validated, and saved.[/green]")
configured += 1
except BrowserTokenError as e:
console.print(f"[yellow]Claude: {e}[/yellow]")
except ProviderError as e:
console.print(
f"[yellow]Claude: extracted cookie failed live validation "
f"({e.original}) — .env not changed. Log in to claude.ai in "
f"{browser} and retry.[/yellow]"
)
if configured:
console.print(
f"\n[green]Done — {configured} provider(s) configured. "
"Run 'ai-chat-exporter doctor' to verify.[/green]"
)
else:
err_console.print(
"\n[red]No tokens could be extracted. Use the manual wizard "
"instead: ai-chat-exporter auth[/red]"
)
sys.exit(1)
def _write_token_to_env(key: str, value: str) -> None:
"""Write or update a key in .env, offering to create the file if it doesn't exist."""
env_path = Path(".env")
if click.confirm(f"Write {key} to .env?", default=True):
if not env_path.exists():
# Create from example if available
example = Path(".env.example")
if example.exists():
import shutil as _shutil
_shutil.copy2(example, env_path)
console.print("[dim]Created .env from .env.example[/dim]")
else:
env_path.touch()
_set_env_key(key, value)
lines = env_path.read_text(encoding="utf-8").splitlines(keepends=True)
updated = False
new_lines = []
for line in lines:
if line.startswith(f"{key}=") or line.startswith(f"{key} ="):
new_lines.append(f"{key}={value}\n")
updated = True
else:
new_lines.append(line)
if not updated:
new_lines.append(f"\n{key}={value}\n")
def _set_env_key(key: str, value: str) -> None:
"""Write or update a key in .env without prompting (creates .env if needed)."""
env_path = Path(".env")
if not env_path.exists():
# Create from example if available
example = Path(".env.example")
if example.exists():
import shutil as _shutil
_shutil.copy2(example, env_path)
console.print("[dim]Created .env from .env.example[/dim]")
else:
env_path.touch()
env_path.write_text("".join(new_lines), encoding="utf-8")
import os
os.chmod(env_path, 0o600)
console.print(f"[green]{key} written to .env (permissions: 600)[/green]")
lines = env_path.read_text(encoding="utf-8").splitlines(keepends=True)
updated = False
new_lines = []
for line in lines:
if line.startswith(f"{key}=") or line.startswith(f"{key} ="):
new_lines.append(f"{key}={value}\n")
updated = True
else:
new_lines.append(line)
if not updated:
new_lines.append(f"\n{key}={value}\n")
env_path.write_text("".join(new_lines), encoding="utf-8")
os.chmod(env_path, 0o600)
console.print(f"[green]{key} written to .env (permissions: 600)[/green]")
# ──────────────────────────────────────────────────────────────────────────────
@@ -326,15 +426,19 @@ def doctor(ctx: click.Context) -> None:
Checks token presence, format, expiry, directory permissions, disk space,
and live API reachability. Exits with code 1 if any checks fail.
"""
checks = _run_doctor_checks()
checks = _run_doctor_checks(cache=ctx.obj["cache"])
_print_doctor_table(checks)
if any(not c["pass"] for c in checks):
sys.exit(1)
def _run_doctor_checks() -> list[dict]:
"""Run all doctor checks and return results."""
def _run_doctor_checks(cache: Cache | None = None) -> list[dict]:
"""Run all doctor checks and return results.
``cache``: when provided, also verify manifest ↔ disk integrity (every
``file_path`` recorded in the manifest exists on disk).
"""
import os
import jwt as pyjwt
from datetime import timezone
@@ -405,6 +509,18 @@ def _run_doctor_checks() -> list[dict]:
except OSError as e:
add("Disk space check", False, str(e))
# Manifest ↔ disk integrity
if cache is not None:
referenced = cache.all_file_paths()
if referenced:
missing = sorted(
p for p in referenced if not Path(p).expanduser().exists()
)
detail = f"{len(referenced) - len(missing)}/{len(referenced)} manifest files present"
if missing:
detail += f"; first missing: {missing[0]} — re-export to fix"
add("Manifest files on disk", not missing, detail)
# API reachability
if chatgpt_token:
try:
@@ -453,7 +569,7 @@ def _print_doctor_table(checks: list[dict]) -> None:
@cli.command()
@click.option(
"--provider",
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
default="all",
show_default=True,
help="Which provider to export.",
@@ -487,6 +603,37 @@ def _print_doctor_table(checks: list[dict]) -> None:
"Use 'none' for conversations outside any project."
),
)
@click.option(
"--hidden-content",
type=click.Choice(["full", "placeholder", "omit"], case_sensitive=False),
default=None,
help=(
"What to do with content that was invisible in the provider's web UI "
"(file-retrieval tool dumps, hidden context): keep in full, collapse to "
"a one-line placeholder, or omit. Overrides EXPORTER_HIDDEN_CONTENT "
"(default: placeholder)."
),
)
@click.option(
"--max-conversations",
type=click.IntRange(min=1),
default=None,
help=(
"Cap how many conversations are downloaded this run (per provider). "
"Runs are resumable, so re-running continues where the cap stopped. "
"Overrides MAX_CONVERSATIONS_PER_RUN (default: unlimited)."
),
)
@click.option(
"--download-media",
type=click.Choice(["images", "all", "off"], case_sensitive=False),
default=None,
help=(
"Download conversation assets next to the exports: images only, "
"all (also audio/files), or off. Overrides EXPORTER_DOWNLOAD_MEDIA "
"(default: images)."
),
)
@click.option("--dry-run", is_flag=True, help="Show what would be exported without writing anything.")
@click.pass_context
def export(
@@ -496,6 +643,9 @@ def export(
output_dir: str | None,
since: str | None,
project_filter: str | None,
hidden_content: str | None,
max_conversations: int | None,
download_media: str | None,
dry_run: bool,
) -> None:
"""Export new and updated conversations to Markdown or JSON.
@@ -507,6 +657,13 @@ def export(
debug = ctx.obj.get("debug", False)
cache: Cache = ctx.obj["cache"]
# CLI flags win over .env: set before load_config (load_dotenv uses
# override=False, so an existing env var is not clobbered by .env).
if hidden_content:
os.environ["EXPORTER_HIDDEN_CONTENT"] = hidden_content.lower()
if download_media:
os.environ["EXPORTER_DOWNLOAD_MEDIA"] = download_media.lower()
# Load config (may raise ConfigError)
try:
from src.config import load_config
@@ -517,7 +674,7 @@ def export(
# First-run: auto-doctor
if not cache.last_run():
console.print("[dim]First run — checking configuration…[/dim]")
checks = _run_doctor_checks()
checks = _run_doctor_checks(cache=cache)
_print_doctor_table(checks)
if any(not c["pass"] for c in checks):
err_console.print(
@@ -537,6 +694,9 @@ def export(
err_console.print(f"[red]Invalid --since date: '{since}'. Use YYYY-MM-DD.[/red]")
sys.exit(1)
# Session cap: CLI flag wins over MAX_CONVERSATIONS_PER_RUN env default.
session_cap = max_conversations if max_conversations is not None else cfg.max_conversations
# Determine which providers to run
providers_to_run = _resolve_providers(provider, cfg)
if not providers_to_run:
@@ -580,15 +740,34 @@ def export(
skipped = len(all_convs) - len(to_export)
summary[prov_name]["skipped"] = skipped
# Apply the session cap. Resumability makes this safe: the manifest
# records each conversation immediately, so the next run picks up
# exactly where the cap stopped.
deferred = 0
if session_cap is not None and len(to_export) > session_cap:
deferred = len(to_export) - session_cap
to_export = to_export[:session_cap]
summary[prov_name]["deferred"] = deferred
if dry_run:
_print_dry_run_table(prov_name, to_export, prov_instance, export_base, structure, skipped)
if deferred:
console.print(
f" [yellow]Session cap {session_cap}: {deferred} more pending beyond this run.[/yellow]"
)
continue
if not to_export:
console.print(f" [dim]{skipped} conversations already up to date.[/dim]")
continue
console.print(f" [dim]{len(to_export)} to export, {skipped} already up to date.[/dim]")
if deferred:
console.print(
f" [dim]{len(to_export)} to export (session cap; {deferred} deferred), "
f"{skipped} already up to date.[/dim]"
)
else:
console.print(f" [dim]{len(to_export)} to export, {skipped} already up to date.[/dim]")
from rich.progress import Progress, SpinnerColumn, TextColumn, BarColumn, TaskProgressColumn
@@ -617,6 +796,19 @@ def export(
full_raw[key] = val
normalized = prov_instance.normalize_conversation(full_raw, loss_report)
# Download assets (images by default) so the renderers
# can inline them. Failures keep placeholders; never fatal.
if cfg.download_media != "off":
from src.media import resolve_media
resolve_media(
normalized,
prov_instance,
export_base,
structure,
cfg.download_media,
loss_report,
)
exported_path: Path | None = None
if md_exporter:
exported_path = md_exporter.export(normalized)
@@ -647,6 +839,12 @@ def export(
if not dry_run:
_print_export_summary(summary)
total_deferred = sum(s.get("deferred", 0) for s in summary.values())
if total_deferred:
console.print(
f"[yellow]{total_deferred} conversation(s) deferred by the session cap "
f"({session_cap} per provider per run). Re-run the same command to continue.[/yellow]"
)
# Emit the data-loss summary at INFO level so it lands in the log file
# AND the operator's console (default level is INFO).
for line in loss_report.format_summary().split("\n"):
@@ -683,6 +881,7 @@ def _resolve_providers(provider: str, cfg) -> list[tuple[str, object]]:
session_token=cfg.chatgpt_session_token,
session_token_1=cfg.chatgpt_session_token_1,
project_ids=cfg.chatgpt_project_ids,
hidden_content=cfg.hidden_content,
),
))
except ProviderError as e:
@@ -694,6 +893,19 @@ def _resolve_providers(provider: str, cfg) -> list[tuple[str, object]]:
if provider in ("claude", "all"):
try_add("claude", cfg.claude_session_key, ClaudeProvider)
if provider in ("claude-code", "all"):
from src.providers.claude_code import ClaudeCodeProvider, DEFAULT_PROJECTS_DIR
cc_dir = Path(os.getenv("CLAUDE_CODE_DIR", DEFAULT_PROJECTS_DIR)).expanduser()
if cc_dir.is_dir():
result.append((
"claude-code",
ClaudeCodeProvider(projects_dir=cc_dir, hidden_content=cfg.hidden_content),
))
elif provider == "claude-code":
logging.getLogger(__name__).warning(
"[claude-code] Skipping — %s not found (set CLAUDE_CODE_DIR).", cc_dir
)
return result
@@ -790,7 +1002,7 @@ def _print_export_summary(summary: dict[str, dict[str, int]]) -> None:
@cli.command(name="list")
@click.option(
"--provider",
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
default="all",
show_default=True,
)
@@ -856,7 +1068,7 @@ def list_conversations(ctx: click.Context, provider: str, project_filter: str |
@click.option("--clear", is_flag=True, help="Clear cached entries.")
@click.option(
"--provider",
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
default="all",
help="Provider to target (used with --clear).",
)
@@ -886,6 +1098,106 @@ def cache(ctx: click.Context, show: bool, clear: bool, provider: str) -> None:
console.print("Specify --show or --clear. Use --help for options.")
# ──────────────────────────────────────────────────────────────────────────────
# prune command
# ──────────────────────────────────────────────────────────────────────────────
@cli.command()
@click.option("--dry-run", is_flag=True, help="List stale files without deleting anything.")
@click.option("--yes", "-y", is_flag=True, help="Delete without a confirmation prompt.")
@click.pass_context
def prune(ctx: click.Context, dry_run: bool, yes: bool) -> None:
"""Delete export files no longer referenced by the cache manifest.
Removes leftovers from old folder layouts and orphaned files from fixed
bugs (e.g. pre-v0.5.0 nested year folders, files with a missing
conversation ID). Only .md and .json files under EXPORT_DIR are
considered; the manifest is the source of truth.
"""
cache_obj: Cache = ctx.obj["cache"]
from dotenv import load_dotenv
load_dotenv(override=False)
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
if not export_dir.exists():
console.print(f"[yellow]Export directory {export_dir} does not exist — nothing to prune.[/yellow]")
return
referenced_raw = cache_obj.all_file_paths()
if not referenced_raw:
# Footgun guard: with an empty manifest (e.g. right after `cache
# --clear`) every export file would look stale and prune would wipe
# the entire archive.
err_console.print(
"[red]The manifest references no export files (did you just run "
"'cache --clear'?). Refusing to prune — run an export first.[/red]"
)
sys.exit(1)
referenced = {Path(p).expanduser().resolve() for p in referenced_raw}
on_disk = sorted(
f for f in export_dir.rglob("*")
if f.is_file() and f.suffix in (".md", ".json")
)
stale = [f for f in on_disk if f.resolve() not in referenced]
if not stale:
console.print(
f"[green]Nothing to prune — all {len(on_disk)} export files are referenced "
"by the manifest.[/green]"
)
return
total_bytes = sum(f.stat().st_size for f in stale)
console.print(
f"[bold]{len(stale)} stale file(s)[/bold] "
f"({total_bytes / (1024 * 1024):.1f} MB) not referenced by the manifest "
f"(of {len(on_disk)} total):\n"
)
for f in stale:
try:
shown = f.relative_to(export_dir.resolve()) if f.is_absolute() else f
except ValueError:
shown = f
console.print(f" [dim]{shown}[/dim]")
if dry_run:
console.print("\n[yellow]Dry run — nothing deleted.[/yellow]")
return
if not yes and not click.confirm(f"\nDelete these {len(stale)} files?", default=False):
console.print("Aborted — nothing deleted.")
return
deleted = 0
for f in stale:
try:
f.unlink()
deleted += 1
except OSError as e:
logger.error("[prune] Could not delete %s: %s", f, e)
# Sweep now-empty directories (deepest first).
removed_dirs = 0
for d in sorted(
(d for d in export_dir.rglob("*") if d.is_dir()),
key=lambda p: len(p.parts),
reverse=True,
):
try:
d.rmdir() # only succeeds when empty
removed_dirs += 1
except OSError:
pass
console.print(
f"[green]Deleted {deleted} file(s) ({total_bytes / (1024 * 1024):.1f} MB) "
f"and {removed_dirs} empty director(ies).[/green]"
)
# ──────────────────────────────────────────────────────────────────────────────
# joplin command
# ──────────────────────────────────────────────────────────────────────────────
@@ -894,7 +1206,7 @@ def cache(ctx: click.Context, show: bool, clear: bool, provider: str) -> None:
@cli.command()
@click.option(
"--provider",
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
default="all",
show_default=True,
help="Which provider's conversations to sync to Joplin.",
@@ -967,10 +1279,9 @@ def joplin(ctx: click.Context, provider: str, project_filter: str | None, dry_ru
# Determine which providers to process
providers_to_sync: list[str] = []
if provider in ("chatgpt", "all"):
providers_to_sync.append("chatgpt")
if provider in ("claude", "all"):
providers_to_sync.append("claude")
for prov in ("chatgpt", "claude", "claude-code"):
if provider in (prov, "all"):
providers_to_sync.append(prov)
summary: dict[str, dict[str, int]] = {}
@@ -1042,6 +1353,17 @@ def joplin(ctx: click.Context, provider: str, project_filter: str | None, dry_ru
body = Path(file_path).read_text(encoding="utf-8")
logger.debug("[joplin] Read %d chars from %s", len(body), file_path)
# Upload any downloaded media as Joplin resources and
# rewrite local media/ links to :/resourceId references.
# The cache stores the resource map so re-syncs reuse IDs.
from src.joplin import upload_media_and_rewrite
res_map = cache_obj.get_joplin_resources(prov_name, conv_id)
body, new_res_map = upload_media_and_rewrite(
body, Path(file_path).parent, client, res_map
)
if new_res_map != res_map:
cache_obj.set_joplin_resources(prov_name, conv_id, new_res_map)
# Get or create the nested notebook
nb_path = notebook_path(prov_name, project)
notebook_id = client.get_or_create_notebook_path(list(nb_path))
+202
View File
@@ -0,0 +1,202 @@
"""Media resolution — download conversation assets next to the export files.
Walks a normalized conversation's placeholder blocks, downloads each asset via
the provider, saves it under a ``media/`` directory beside the conversation's
Markdown file, and annotates the block with a relative ``local_path`` so the
renderer emits a real inline image / file link instead of a placeholder.
Policy (``EXPORTER_DOWNLOAD_MEDIA``):
images (default) — image_placeholder blocks only
all — also file_placeholder blocks (voice-mode audio etc.;
measured 2026-06-12: 556 clips ≈ 162MB whose transcripts
are already in the exports as text)
off — leave placeholders untouched
Failures (expired assets, network) keep the existing placeholder and are
counted in the run summary — never fatal to the export.
"""
import logging
import os
import tempfile
from pathlib import Path
from src.blocks import BLOCK_TYPE_FILE_PLACEHOLDER, BLOCK_TYPE_IMAGE_PLACEHOLDER
from src.loss_report import LossReport
from src.providers.base import ProviderError
from src.utils import build_export_path, generate_filename
logger = logging.getLogger(__name__)
MEDIA_IMAGES = "images"
MEDIA_ALL = "all"
MEDIA_OFF = "off"
VALID_MEDIA_POLICIES = {MEDIA_IMAGES, MEDIA_ALL, MEDIA_OFF}
_EXT_BY_MIME = {
"image/png": ".png",
"image/jpeg": ".jpg",
"image/gif": ".gif",
"image/webp": ".webp",
"audio/wav": ".wav",
"audio/x-wav": ".wav",
"audio/mpeg": ".mp3",
"audio/mp4": ".m4a",
"audio/aac": ".aac",
"application/pdf": ".pdf",
}
def resolve_media_policy() -> str:
"""Read EXPORTER_DOWNLOAD_MEDIA from the environment, defaulting to images."""
value = os.getenv("EXPORTER_DOWNLOAD_MEDIA", "").strip().lower()
if not value:
return MEDIA_IMAGES
if value not in VALID_MEDIA_POLICIES:
logger.warning(
"EXPORTER_DOWNLOAD_MEDIA=%r is invalid (expected images|all|off) "
"— using 'images'.",
value,
)
return MEDIA_IMAGES
return value
def resolve_media(
normalized: dict,
provider,
export_base: Path,
structure: str,
policy: str,
report: LossReport,
) -> int:
"""Download assets for one normalized conversation. Returns download count.
Mutates placeholder blocks in place (adds ``local_path``). Idempotent:
an asset already on disk is annotated without an API call.
"""
if policy == MEDIA_OFF:
return 0
download = getattr(provider, "download_asset", None)
if download is None:
# Provider has no remote assets (e.g. claude-code local transcripts).
return 0
wanted_types = {BLOCK_TYPE_IMAGE_PLACEHOLDER}
if policy == MEDIA_ALL:
wanted_types.add(BLOCK_TYPE_FILE_PLACEHOLDER)
targets = [
block
for message in normalized.get("messages", [])
for block in message.get("blocks", [])
if block.get("type") in wanted_types and block.get("ref")
]
if not targets:
return 0
# Same path computation the exporter uses, so media/ lands beside the .md.
filename = generate_filename(
normalized.get("title", "Untitled"),
normalized.get("id", ""),
normalized.get("created_at") or "2000-01-01",
)
conv_dir = build_export_path(
export_base,
normalized.get("provider", ""),
normalized.get("project"),
normalized.get("created_at") or "2000-01-01",
filename,
structure,
).parent
media_dir = conv_dir / "media"
downloaded = 0
for block in targets:
ref = block["ref"]
file_id = _safe_asset_name(provider, ref)
if not file_id:
report.record_media_failed("unparseable-ref")
continue
existing = _find_existing(media_dir, file_id)
if existing is not None:
block["local_path"] = f"media/{existing.name}"
continue
try:
content, mime, file_name = download(ref)
except ProviderError as e:
logger.warning(
"[media] Could not download %s: %s", ref[:60], e.original
)
reason = "expired-or-missing" if "404" in str(e.original) or "not found" in str(
e.original
).lower() else "download-error"
report.record_media_failed(reason)
continue
ext = _pick_extension(mime, file_name)
target = media_dir / f"{file_id}{ext}"
try:
_write_atomic(target, content)
except OSError as e:
logger.error("[media] Could not write %s: %s", target, e)
report.record_media_failed("write-error")
continue
block["local_path"] = f"media/{target.name}"
report.record_media_downloaded(len(content))
downloaded += 1
return downloaded
def _safe_asset_name(provider, ref: str) -> str | None:
"""A stable, filesystem-safe name for the asset (the provider file ID)."""
parser = getattr(provider, "parse_asset_file_id", None)
if parser is None:
from src.providers.chatgpt import parse_asset_file_id as parser
file_id = parser(ref)
if not file_id:
return None
return "".join(c if c.isalnum() or c in "-_." else "_" for c in file_id)
def _find_existing(media_dir: Path, file_id: str) -> Path | None:
"""Return an already-downloaded file for this asset, if present."""
if not media_dir.is_dir():
return None
for candidate in media_dir.glob(f"{file_id}.*"):
if candidate.is_file() and candidate.stat().st_size > 0:
return candidate
return None
def _pick_extension(mime: str | None, file_name: str | None) -> str:
if mime:
base_mime = mime.split(";")[0].strip().lower()
if base_mime in _EXT_BY_MIME:
return _EXT_BY_MIME[base_mime]
if file_name:
suffix = Path(file_name).suffix
if suffix and len(suffix) <= 8:
return suffix.lower()
return ".bin"
def _write_atomic(target: Path, content: bytes) -> None:
target.parent.mkdir(parents=True, exist_ok=True)
fd, tmp_name = tempfile.mkstemp(dir=target.parent, suffix=".tmp")
try:
with os.fdopen(fd, "wb") as fh:
fh.write(content)
os.chmod(tmp_name, 0o600)
os.replace(tmp_name, target)
except OSError:
try:
os.unlink(tmp_name)
except OSError:
pass
raise
+78
View File
@@ -1,6 +1,7 @@
"""Abstract base class for AI chat providers."""
import logging
import os
import random
import time
from abc import ABC, abstractmethod
@@ -37,6 +38,62 @@ MAX_RETRIES = 3
BACKOFF_BASE = 2.0
BACKOFF_MAX = 60.0
# EXPORTER_HIDDEN_CONTENT policy — what to do with content that was invisible
# in the provider's UI (retrieval dumps, hidden-flagged context, agent tool
# traffic). Shared by all providers.
HIDDEN_CONTENT_FULL = "full"
HIDDEN_CONTENT_PLACEHOLDER = "placeholder"
HIDDEN_CONTENT_OMIT = "omit"
VALID_HIDDEN_CONTENT_POLICIES = {
HIDDEN_CONTENT_FULL,
HIDDEN_CONTENT_PLACEHOLDER,
HIDDEN_CONTENT_OMIT,
}
def resolve_hidden_content_policy() -> str:
"""Read EXPORTER_HIDDEN_CONTENT from the environment, defaulting to placeholder."""
value = os.getenv("EXPORTER_HIDDEN_CONTENT", "").strip().lower()
if not value:
return HIDDEN_CONTENT_PLACEHOLDER
if value not in VALID_HIDDEN_CONTENT_POLICIES:
logger.warning(
"EXPORTER_HIDDEN_CONTENT=%r is invalid (expected full|placeholder|omit) "
"— using 'placeholder'.",
value,
)
return HIDDEN_CONTENT_PLACEHOLDER
return value
# Default polite pacing between consecutive API requests (seconds). Keeps
# traffic human-paced instead of bursty. REQUEST_DELAY=0 disables.
DEFAULT_REQUEST_DELAY = 1.0
def resolve_request_delay() -> float:
"""Read REQUEST_DELAY from the environment, defaulting to DEFAULT_REQUEST_DELAY."""
raw = os.getenv("REQUEST_DELAY", "").strip()
if not raw:
return DEFAULT_REQUEST_DELAY
try:
value = float(raw)
except ValueError:
logger.warning(
"REQUEST_DELAY=%r is not a number — using default %.1fs.",
raw,
DEFAULT_REQUEST_DELAY,
)
return DEFAULT_REQUEST_DELAY
if value < 0:
logger.warning(
"REQUEST_DELAY=%r is negative — using default %.1fs.",
raw,
DEFAULT_REQUEST_DELAY,
)
return DEFAULT_REQUEST_DELAY
return value
# Realistic Chrome User-Agent
USER_AGENT = (
"Mozilla/5.0 (X11; Linux x86_64) "
@@ -76,6 +133,9 @@ class BaseProvider(ABC):
"Accept-Language": "en-US,en;q=0.9",
}
)
# Polite pacing between consecutive requests (REQUEST_DELAY env).
self._request_delay = resolve_request_delay()
self._last_request_at: float | None = None
# ------------------------------------------------------------------
# Abstract interface — subclasses must implement these
@@ -103,6 +163,23 @@ class BaseProvider(ABC):
# Concrete helpers
# ------------------------------------------------------------------
def _pace(self) -> None:
"""Sleep so consecutive requests are at least REQUEST_DELAY apart (±25% jitter).
getattr fallback: tests construct providers via ``__new__``, skipping
``__init__`` — those run unpaced.
"""
delay = getattr(self, "_request_delay", 0)
if delay <= 0:
return
target = delay * random.uniform(0.75, 1.25)
last = getattr(self, "_last_request_at", None)
if last is not None:
wait = target - (time.monotonic() - last)
if wait > 0:
time.sleep(wait)
self._last_request_at = time.monotonic()
def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]:
"""Fetch every conversation, handling pagination automatically.
@@ -208,6 +285,7 @@ class BaseProvider(ABC):
while attempt <= MAX_RETRIES:
attempt += 1
self._pace()
start = time.monotonic()
try:
+158 -10
View File
@@ -19,6 +19,7 @@ Response: {"items": [...], "cursor": "<opaque_base64_or_null>"}
Pagination ends when cursor is null or an empty string.
"""
import json
import logging
import os
from typing import Any
@@ -26,10 +27,13 @@ from typing import Any
from curl_cffi import requests as curl_requests
from src.blocks import (
COLLAPSED_KIND_HIDDEN_CONTEXT,
COLLAPSED_KIND_TOOL_DUMP,
UNKNOWN_REASON_EXTRACTION_FAILED,
UNKNOWN_REASON_UNKNOWN_FIELD_IN_KNOWN_TYPE,
UNKNOWN_REASON_UNKNOWN_TYPE,
make_code_block,
make_collapsed_block,
make_file_placeholder,
make_hidden_context_marker,
make_image_placeholder,
@@ -39,7 +43,16 @@ from src.blocks import (
make_unknown_block,
)
from src.loss_report import LossReport
from src.providers.base import BaseProvider, ProviderError, REQUEST_TIMEOUT
from src.providers.base import (
BaseProvider,
HIDDEN_CONTENT_FULL,
HIDDEN_CONTENT_OMIT,
HIDDEN_CONTENT_PLACEHOLDER,
ProviderError,
REQUEST_TIMEOUT,
VALID_HIDDEN_CONTENT_POLICIES,
resolve_hidden_content_policy,
)
logger = logging.getLogger(__name__)
@@ -50,6 +63,34 @@ AUTH_SESSION_URL = "https://chatgpt.com/api/auth/session"
# Run: python -c "from curl_cffi.requests import BrowserType; print(list(BrowserType))"
IMPERSONATE = "chrome120"
def parse_asset_file_id(ref: str) -> str | None:
"""Extract the file ID from a ChatGPT asset pointer.
Observed forms (2026-06-12):
sediment://file_00000000245c71fda50034d7f1647791 → file_…
sediment://8456107fc383a53#file_…979c…#p_6.png (generated) → file_…
file-service://file-AbCdEf → file-AbCdEf
"""
if not isinstance(ref, str) or not ref:
return None
if ref.startswith("file-service://"):
return ref.removeprefix("file-service://") or None
if ref.startswith("sediment://"):
body = ref.removeprefix("sediment://")
for part in body.split("#"):
if part.startswith("file_") or part.startswith("file-"):
return part
return body or None
return None
# Tool-role authors whose messages are retrieval dumps — the full text of
# attached/project files re-injected by ChatGPT on every tool run. Measured
# 2026-06-12 across this user's three largest conversations: 86% of all
# content bytes, not flagged is_visually_hidden_from_conversation.
# myfiles_browser is the legacy name for the same retrieval tool.
_COLLAPSE_TOOL_AUTHORS = {"file_search", "myfiles_browser"}
class ChatGPTProvider(BaseProvider):
"""Provider for ChatGPT conversations via the internal web API.
@@ -72,6 +113,7 @@ class ChatGPTProvider(BaseProvider):
session_token: str | None = None,
session_token_1: str | None = None,
project_ids: list[str] | None = None,
hidden_content: str | None = None,
) -> None:
# Pass a curl_cffi session to the base class instead of a requests.Session.
# curl_cffi.requests.Session is API-compatible with requests.Session.
@@ -112,6 +154,14 @@ class ChatGPTProvider(BaseProvider):
# Cache of project_id → display name (avoids re-fetching gizmo details)
self._project_name_cache: dict[str, str] = {}
# Policy for messages invisible in the web UI (retrieval dumps,
# hidden-flagged context): full | placeholder | omit.
self._hidden_content = (
hidden_content
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
else resolve_hidden_content_policy()
)
# ChatGPT now splits large session cookies into .0 / .1 chunks.
# Always send both named chunks; the server reassembles them.
self._session.cookies.set(
@@ -561,6 +611,59 @@ class ChatGPTProvider(BaseProvider):
)
return data
# ------------------------------------------------------------------
# Asset downloads
# ------------------------------------------------------------------
def download_asset(self, ref: str) -> tuple[bytes, str | None, str | None]:
"""Download a sediment:// / file-service:// asset.
Two hops (verified live 2026-06-12): the files endpoint returns a
signed ``download_url``; fetching that yields the bytes. Expired
assets (e.g. old generated images) 404 on the first hop.
Returns:
(content_bytes, mime_type_or_None, file_name_or_None)
Raises:
ProviderError: unparseable ref, expired/missing asset, or
download failure.
"""
file_id = parse_asset_file_id(ref)
if not file_id:
raise ProviderError(
self.provider_name,
f"download_asset({ref[:40]})",
ValueError(f"Unrecognised asset reference: {ref[:80]}"),
)
meta = self._make_request("GET", f"{BASE_URL}/files/{file_id}/download")
download_url = meta.get("download_url")
if not download_url:
raise ProviderError(
self.provider_name,
f"download_asset({file_id})",
RuntimeError(f"No download_url in response: {meta.get('detail') or meta}"),
)
self._pace()
resp = self._session.request("GET", download_url, timeout=REQUEST_TIMEOUT)
if resp.status_code != 200:
raise ProviderError(
self.provider_name,
f"download_asset({file_id})",
RuntimeError(f"Signed URL returned HTTP {resp.status_code}"),
)
mime = resp.headers.get("content-type") or None
file_name = meta.get("file_name") or None
logger.debug(
"[chatgpt] Downloaded asset %s: %d bytes (%s)",
file_id,
len(resp.content),
mime,
)
return resp.content, mime, file_name
# ------------------------------------------------------------------
# Normalization
# ------------------------------------------------------------------
@@ -598,7 +701,9 @@ class ChatGPTProvider(BaseProvider):
)
mapping: dict = raw.get("mapping", {})
messages = _extract_messages(mapping, raw, conv_id, report)
# getattr fallback: tests construct providers via __new__, skipping __init__.
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
messages = _extract_messages(mapping, raw, conv_id, report, policy)
for _ in messages:
report.record_message()
report.record_conversation()
@@ -631,7 +736,11 @@ def _ts_to_iso(ts: float | int | str | None) -> str:
def _extract_messages(
mapping: dict[str, Any], raw: dict, conv_id: str, report: LossReport
mapping: dict[str, Any],
raw: dict,
conv_id: str,
report: LossReport,
policy: str = HIDDEN_CONTENT_FULL,
) -> list[dict]:
"""Walk the ChatGPT conversation mapping tree to produce an ordered message list.
@@ -661,7 +770,7 @@ def _extract_messages(
node = mapping.get(node_id, {})
msg_data = node.get("message")
if msg_data:
built = _build_message(msg_data, conv_id, node_id, report)
built = _build_message(msg_data, conv_id, node_id, report, policy)
if built is not None:
messages.append(built)
@@ -688,12 +797,17 @@ def _find_root(mapping: dict[str, Any]) -> str | None:
def _build_message(
msg_data: dict, conv_id: str, node_id: str, report: LossReport
msg_data: dict,
conv_id: str,
node_id: str,
report: LossReport,
policy: str = HIDDEN_CONTENT_FULL,
) -> dict | None:
"""Construct a normalized message dict (with ``blocks``) for one ChatGPT node.
Returns None for messages that should be skipped (truly empty). Otherwise
returns a dict with ``role``, ``content_type``, ``timestamp``, ``blocks``.
Returns None for messages that should be skipped (truly empty, or omitted
by the EXPORTER_HIDDEN_CONTENT policy). Otherwise returns a dict with
``role``, ``content_type``, ``timestamp``, ``blocks``.
"""
author = msg_data.get("author") or {}
role = author.get("role", "") or ""
@@ -727,9 +841,43 @@ def _build_message(
)
return None
if is_hidden:
# Prepend a marker so the reader knows this message is hidden in the
# source UI. The marker is content-type-agnostic.
# Messages invisible in the web UI: tool retrieval dumps (identified by
# author.name — they are NOT hidden-flagged) and hidden-flagged context.
collapse_kind: str | None = None
if role == "tool" and author_name in _COLLAPSE_TOOL_AUTHORS:
collapse_kind = COLLAPSED_KIND_TOOL_DUMP
elif is_hidden:
collapse_kind = COLLAPSED_KIND_HIDDEN_CONTEXT
if collapse_kind is not None and policy != HIDDEN_CONTENT_FULL:
origin = (
author_name if collapse_kind == COLLAPSED_KIND_TOOL_DUMP else content_type
) or "?"
size_bytes = len(json.dumps(content_obj, ensure_ascii=False, default=str))
report.record_collapsed(origin, size_bytes)
logger.debug(
"[chatgpt] %s %s message (%s, %dB) in conversation %s "
"per EXPORTER_HIDDEN_CONTENT=%s",
"Omitted" if policy == HIDDEN_CONTENT_OMIT else "Collapsed",
collapse_kind,
origin,
size_bytes,
conv_id[:8],
policy,
)
if policy == HIDDEN_CONTENT_OMIT:
return None
blocks = [
make_collapsed_block(
origin=origin,
content_type=content_type,
size_bytes=size_bytes,
kind=collapse_kind,
)
]
elif is_hidden:
# Policy 'full': prepend a marker so the reader knows this message is
# hidden in the source UI. The marker is content-type-agnostic.
blocks = [make_hidden_context_marker(content_type)] + blocks
# Vestigial content_type: "code" for code-only messages, otherwise "text"
+476
View File
@@ -0,0 +1,476 @@
"""Claude Code session provider — archives local agent transcripts.
Reads JSONL session files from ``~/.claude/projects/<munged-cwd>/<uuid>.jsonl``
(override with ``CLAUDE_CODE_DIR``). No tokens, no rate limits, no ToS risk —
but the data lives in single files Claude Code may clean up, and it contains
deliverables (reviews, plans, analyses) that exist nowhere else.
Rendering follows the EXPORTER_HIDDEN_CONTENT policy (decided 2026-06-12):
prose-only by default. User prompts and assistant text are kept; tool_use /
tool_result traffic is collapsed to one grouped placeholder per activity run
(measured: dialogue prose is ~4% of session bytes); thinking blocks are
dropped (counted in the run summary, no placeholder). ``full`` keeps
everything.
Record types in a session file: ``user`` / ``assistant`` carry the dialogue
(Anthropic-style ``message.content`` block arrays); ``ai-title`` carries the
evolving session title (last one wins); ``last-prompt``,
``file-history-snapshot``, ``attachment``, ``permission-mode``, ``system``
are harness records and are skipped. Records flagged ``isSidechain`` are
subagent transcripts; ``isMeta`` are harness-generated user records — both
skipped.
"""
import json
import logging
import os
import re
from collections import Counter
from datetime import datetime, timezone
from pathlib import Path
from src.blocks import (
COLLAPSED_KIND_TOOL_DUMP,
UNKNOWN_REASON_UNKNOWN_TYPE,
make_collapsed_block,
make_image_placeholder,
make_text_block,
make_thinking_block,
make_tool_result_block,
make_tool_use_block,
make_unknown_block,
)
from src.loss_report import LossReport
from src.providers.base import (
BaseProvider,
HIDDEN_CONTENT_FULL,
ProviderError,
VALID_HIDDEN_CONTENT_POLICIES,
resolve_hidden_content_policy,
)
logger = logging.getLogger(__name__)
DEFAULT_PROJECTS_DIR = "~/.claude/projects"
# Harness-injected tags inside user message text. Stripped so exports contain
# the dialogue, not the CLI plumbing. A record that is nothing but tags
# (e.g. a /model invocation) ends up empty and is skipped.
_HARNESS_TAG_RE = re.compile(
r"<local-command-caveat>.*?</local-command-caveat>"
r"|<command-name>.*?</command-name>"
r"|<command-message>.*?</command-message>"
r"|<command-args>.*?</command-args>"
r"|<command-contents>.*?</command-contents>"
r"|<local-command-stdout>.*?</local-command-stdout>"
r"|<system-reminder>.*?</system-reminder>",
re.DOTALL,
)
# How many distinct tool names to list in a collapsed-activity placeholder.
_TOOL_NAMES_SHOWN = 4
def _strip_harness_noise(text: str) -> str:
if not isinstance(text, str):
return ""
return _HARNESS_TAG_RE.sub("", text).strip()
class ClaudeCodeProvider(BaseProvider):
"""Local-file provider over Claude Code session transcripts."""
provider_name = "claude-code"
def __init__(
self,
projects_dir: str | Path | None = None,
hidden_content: str | None = None,
) -> None:
super().__init__()
self._projects_dir = Path(
projects_dir or os.getenv("CLAUDE_CODE_DIR", DEFAULT_PROJECTS_DIR)
).expanduser()
self._hidden_content = (
hidden_content
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
else resolve_hidden_content_policy()
)
# conv_id → session file path, populated by _scan()
self._path_map: dict[str, Path] = {}
# ------------------------------------------------------------------
# BaseProvider interface
# ------------------------------------------------------------------
def list_conversations(self, offset: int = 0, limit: int = 100) -> list[dict]:
full = self._scan()
return full[offset : offset + limit]
def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]:
convs = self._scan()
if since is not None:
since_aware = since if since.tzinfo else since.replace(tzinfo=timezone.utc)
convs = [
c for c in convs
if datetime.fromisoformat(c["updated_at"]) >= since_aware
]
logger.info(
"[claude-code] Found %d session(s) under %s", len(convs), self._projects_dir
)
return convs
def get_conversation(self, conv_id: str) -> dict:
path = self._path_map.get(conv_id)
if path is None:
# Direct call without a prior listing (e.g. tests) — scan first.
self._scan()
path = self._path_map.get(conv_id)
if path is None or not path.exists():
raise ProviderError(
self.provider_name,
f"get_conversation({conv_id[:8]})",
FileNotFoundError(f"No session file for id {conv_id}"),
)
records: list[dict] = []
bad_lines = 0
for line in path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line:
continue
try:
records.append(json.loads(line))
except json.JSONDecodeError:
bad_lines += 1
if bad_lines:
logger.warning(
"[claude-code] %s: skipped %d unparseable line(s)", path.name, bad_lines
)
mtime = datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc)
return {
"id": conv_id,
"_path": str(path),
"_records": records,
# Listing and normalized updated_at must match, or the cache
# staleness comparison would re-export every session every run.
"_mtime_iso": mtime.isoformat(),
}
def normalize_conversation(self, raw: dict, loss_report: LossReport | None = None) -> dict:
report = loss_report if loss_report is not None else LossReport()
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
conv_id = raw.get("id") or ""
records: list[dict] = raw.get("_records") or []
title = _extract_title(records)
project = _extract_project(records, raw.get("_path"))
created_at = next(
(r.get("timestamp") for r in records if r.get("timestamp")), ""
)
updated_at = raw.get("_mtime_iso") or next(
(r.get("timestamp") for r in reversed(records) if r.get("timestamp")), ""
)
messages = _extract_messages(records, conv_id, report, policy)
for _ in messages:
report.record_message()
report.record_conversation()
return {
"id": conv_id,
"title": title,
"provider": self.provider_name,
"project": project,
"created_at": created_at or "",
"updated_at": updated_at or "",
"message_count": len(messages),
"messages": messages,
}
# ------------------------------------------------------------------
# Scanning
# ------------------------------------------------------------------
def _scan(self) -> list[dict]:
if not self._projects_dir.is_dir():
logger.warning(
"[claude-code] Projects directory %s does not exist", self._projects_dir
)
return []
convs: list[dict] = []
for proj_dir in sorted(p for p in self._projects_dir.iterdir() if p.is_dir()):
for session_file in sorted(proj_dir.glob("*.jsonl")):
try:
stat = session_file.stat()
except OSError:
continue
if stat.st_size == 0:
continue
conv_id = session_file.stem
self._path_map[conv_id] = session_file
title, project, created = _read_session_meta(session_file)
convs.append(
{
"id": conv_id,
"title": title,
"project": project,
# The --project filter and dry-run table read the
# listing dict, not the normalized conversation.
"_project_name": project,
"created_at": created,
"updated_at": datetime.fromtimestamp(
stat.st_mtime, tz=timezone.utc
).isoformat(),
"_path": str(session_file),
"_project_dir": proj_dir.name,
}
)
return convs
# ---------------------------------------------------------------------------
# Internal helpers
# ---------------------------------------------------------------------------
def _read_session_meta(path: Path) -> tuple[str, str | None, str]:
"""Light single-pass scan for listing metadata: (title, project, created_at).
Substring guards keep this cheap — only candidate lines are JSON-parsed.
The full-fidelity extraction happens later in normalize_conversation.
"""
title = ""
project: str | None = None
created = ""
try:
with path.open(encoding="utf-8") as fh:
for line in fh:
if '"ai-title"' in line:
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
if rec.get("type") == "ai-title" and rec.get("aiTitle"):
title = str(rec["aiTitle"]) # last one wins
continue
if (not created or project is None) and (
'"timestamp"' in line or '"cwd"' in line
):
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
if not created and rec.get("timestamp"):
created = str(rec["timestamp"])
if project is None and rec.get("cwd"):
project = Path(rec["cwd"]).name or None
except OSError as e:
logger.warning("[claude-code] Could not read %s: %s", path, e)
return title, project, created
def _extract_title(records: list[dict]) -> str:
"""Last ai-title record wins; fall back to the first real user prompt."""
title = ""
for rec in records:
if rec.get("type") == "ai-title" and rec.get("aiTitle"):
title = str(rec["aiTitle"])
if title:
return title
for rec in records:
if rec.get("type") != "user" or rec.get("isMeta") or rec.get("isSidechain"):
continue
content = (rec.get("message") or {}).get("content")
if isinstance(content, str):
text = _strip_harness_noise(content)
elif isinstance(content, list):
text = " ".join(
_strip_harness_noise(item.get("text", ""))
for item in content
if isinstance(item, dict) and item.get("type") == "text"
).strip()
else:
text = ""
if text:
return text[:80]
return "Untitled session"
def _extract_project(records: list[dict], path: str | None) -> str | None:
"""Project = basename of the session's working directory."""
for rec in records:
cwd = rec.get("cwd")
if cwd:
name = Path(cwd).name
if name:
return name
# Fallback: the munged directory name (cannot be reliably de-munged
# because '-' is both the path separator and a legal name character).
if path:
return Path(path).parent.name.lstrip("-") or None
return None
def _extract_messages(
records: list[dict], conv_id: str, report: LossReport, policy: str
) -> list[dict]:
messages: list[dict] = []
# Pending collapsed tool activity: name → call count, plus total bytes.
pending_tools: Counter = Counter()
pending_bytes = 0
def flush_pending() -> None:
nonlocal pending_bytes
if not pending_tools:
return
shown = ", ".join(
f"{name} ×{count}" for name, count in pending_tools.most_common(_TOOL_NAMES_SHOWN)
)
if len(pending_tools) > _TOOL_NAMES_SHOWN:
shown += ", …"
calls = sum(pending_tools.values())
block = make_collapsed_block(
origin=f"{calls} calls: {shown}",
content_type="tool_activity",
size_bytes=pending_bytes,
kind=COLLAPSED_KIND_TOOL_DUMP,
)
if messages and messages[-1]["role"] == "assistant":
messages[-1]["blocks"].append(block)
else:
messages.append(
{"role": "tool", "content_type": "text", "timestamp": None, "blocks": [block]}
)
pending_tools.clear()
pending_bytes = 0
for rec in records:
if rec.get("type") not in ("user", "assistant"):
continue
if rec.get("isSidechain") or rec.get("isMeta"):
continue
msg = rec.get("message") or {}
role = msg.get("role") or rec["type"]
content = msg.get("content")
blocks: list[dict] = []
# This record's own tool traffic — merged into pending only after the
# record's message is appended, so the placeholder lands on it (not on
# the previous message).
local_tools: Counter = Counter()
local_bytes = 0
if isinstance(content, str):
text = _strip_harness_noise(content)
block = make_text_block(text)
if block:
blocks.append(block)
elif isinstance(content, list):
for item in content:
if not isinstance(item, dict):
continue
item_type = item.get("type", "")
if item_type == "text":
block = make_text_block(_strip_harness_noise(item.get("text", "")))
if block:
blocks.append(block)
elif item_type in ("thinking", "redacted_thinking"):
if policy == HIDDEN_CONTENT_FULL:
block = make_thinking_block(
item.get("thinking") or item.get("text") or ""
)
if block:
blocks.append(block)
else:
# Decision 2026-06-12: thinking is dropped without a
# placeholder, but stays visible in the run summary.
report.record_collapsed(
"thinking", len(json.dumps(item, default=str))
)
elif item_type == "tool_use":
if policy == HIDDEN_CONTENT_FULL:
blocks.append(
make_tool_use_block(
item.get("name", ""), item.get("input"), item.get("id")
)
)
else:
name = item.get("name") or "tool"
size = len(json.dumps(item, default=str))
local_tools[name] += 1
local_bytes += size
report.record_collapsed(name, size)
elif item_type == "tool_result":
if policy == HIDDEN_CONTENT_FULL:
blocks.append(
make_tool_result_block(
_stringify_tool_result(item.get("content")),
is_error=bool(item.get("is_error")),
)
)
else:
size = len(json.dumps(item, default=str))
local_bytes += size
report.record_collapsed("tool_result", size)
elif item_type == "image":
blocks.append(
make_image_placeholder(ref="embedded image", source="user_upload")
)
else:
logger.warning(
"[claude-code] Unknown content block type %r in session %s",
item_type,
conv_id[:8],
)
report.record_unknown(f"claude-code.{item_type or '?'}")
blocks.append(
make_unknown_block(
raw_type=f"claude-code.{item_type or '?'}",
observed_keys=list(item.keys()),
reason=UNKNOWN_REASON_UNKNOWN_TYPE,
)
)
if not blocks:
# Tool-result-only records (and similar): traffic accumulates and
# is attached to the message that initiated it.
pending_tools.update(local_tools)
pending_bytes += local_bytes
continue
# Attach previous records' tool activity to the previous message
# before starting a new one, so the placeholder lands between the
# dialogue turns it actually occurred between.
flush_pending()
messages.append(
{
"role": role,
"content_type": "text",
"timestamp": rec.get("timestamp"),
"blocks": blocks,
}
)
pending_tools.update(local_tools)
pending_bytes += local_bytes
flush_pending()
return messages
def _stringify_tool_result(content) -> str:
"""tool_result content may be a string or a list of text blocks."""
if isinstance(content, str):
return content
if isinstance(content, list):
parts = []
for item in content:
if isinstance(item, dict) and item.get("type") == "text":
parts.append(item.get("text", ""))
else:
parts.append(json.dumps(item, default=str))
return "\n".join(parts)
return json.dumps(content, default=str) if content is not None else ""