feat: v0.6.0 — collapse policy, session limiter, Claude Code provider, prune, browser auth, media downloads
This commit is contained in:
+63
-3
@@ -11,6 +11,7 @@ at the call site — see plan §Data-loss visibility.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
BLOCK_TYPE_TEXT = "text"
|
||||
@@ -23,6 +24,10 @@ BLOCK_TYPE_IMAGE_PLACEHOLDER = "image_placeholder"
|
||||
BLOCK_TYPE_FILE_PLACEHOLDER = "file_placeholder"
|
||||
BLOCK_TYPE_UNKNOWN = "unknown"
|
||||
BLOCK_TYPE_HIDDEN_CONTEXT_MARKER = "hidden_context_marker"
|
||||
BLOCK_TYPE_COLLAPSED = "collapsed"
|
||||
|
||||
COLLAPSED_KIND_TOOL_DUMP = "tool_dump"
|
||||
COLLAPSED_KIND_HIDDEN_CONTEXT = "hidden_context"
|
||||
|
||||
UNKNOWN_REASON_UNKNOWN_TYPE = "unknown_type"
|
||||
UNKNOWN_REASON_EXTRACTION_FAILED = "extraction_failed"
|
||||
@@ -164,6 +169,30 @@ def make_unknown_block(
|
||||
}
|
||||
|
||||
|
||||
def make_collapsed_block(
|
||||
origin: str,
|
||||
content_type: str,
|
||||
size_bytes: int,
|
||||
kind: str,
|
||||
) -> dict:
|
||||
"""One-line stand-in for a message omitted by the EXPORTER_HIDDEN_CONTENT policy.
|
||||
|
||||
``kind`` ∈ {COLLAPSED_KIND_TOOL_DUMP, COLLAPSED_KIND_HIDDEN_CONTEXT}.
|
||||
Unlike ``unknown`` blocks (unexpected loss), a collapsed block records an
|
||||
intentional policy decision — the content was invisible in the provider's
|
||||
web UI and the user chose not to keep it. Every call site MUST tally via
|
||||
``LossReport.record_collapsed`` so the omission stays visible in the
|
||||
post-export summary.
|
||||
"""
|
||||
return {
|
||||
"type": BLOCK_TYPE_COLLAPSED,
|
||||
"origin": origin or "",
|
||||
"content_type": content_type or "",
|
||||
"size_bytes": int(size_bytes or 0),
|
||||
"kind": kind,
|
||||
}
|
||||
|
||||
|
||||
def make_hidden_context_marker(content_type: str) -> dict:
|
||||
"""A short prepend block that flags the surrounding message as hidden context.
|
||||
|
||||
@@ -246,6 +275,10 @@ def _render_one(block: dict) -> str:
|
||||
ref = block.get("ref", "")
|
||||
source = block.get("source", "unknown")
|
||||
mime = block.get("mime")
|
||||
local_path = block.get("local_path")
|
||||
if local_path:
|
||||
# Downloaded — render as a real inline image.
|
||||
return f""
|
||||
meta_parts = [source] if source else []
|
||||
if mime:
|
||||
meta_parts.append(mime)
|
||||
@@ -259,14 +292,20 @@ def _render_one(block: dict) -> str:
|
||||
mime = block.get("mime")
|
||||
size_bytes = block.get("size_bytes")
|
||||
duration = block.get("duration_seconds")
|
||||
local_path = block.get("local_path")
|
||||
meta_parts: list[str] = []
|
||||
if mime:
|
||||
meta_parts.append(mime)
|
||||
if isinstance(size_bytes, int) and size_bytes > 0:
|
||||
kb = size_bytes / 1024
|
||||
meta_parts.append(f"{kb:.1f} KB" if kb < 1024 else f"{kb / 1024:.2f} MB")
|
||||
size_label = _format_size(size_bytes)
|
||||
if size_label:
|
||||
meta_parts.append(size_label)
|
||||
if isinstance(duration, (int, float)) and duration > 0:
|
||||
meta_parts.append(f"{duration:.2f}s")
|
||||
if local_path:
|
||||
# Downloaded — render as a link to the local copy.
|
||||
meta = ", ".join(meta_parts) if meta_parts else ""
|
||||
meta_str = f" ({meta})" if meta else ""
|
||||
return f"> 📎 **File attached** — [{Path(local_path).name}]({local_path}){meta_str}"
|
||||
meta_parts.append("content not preserved in this export")
|
||||
meta = ", ".join(meta_parts)
|
||||
return f"> 📎 **File attached** — `{label}` ({meta})"
|
||||
@@ -286,6 +325,19 @@ def _render_one(block: dict) -> str:
|
||||
if btype == BLOCK_TYPE_HIDDEN_CONTEXT_MARKER:
|
||||
ctype = block.get("content_type", "")
|
||||
return f"> ℹ️ **Hidden context** — `{ctype}`"
|
||||
if btype == BLOCK_TYPE_COLLAPSED:
|
||||
origin = block.get("origin", "")
|
||||
kind = block.get("kind", "")
|
||||
if kind == COLLAPSED_KIND_HIDDEN_CONTEXT:
|
||||
icon, label = "ℹ️", "Hidden context"
|
||||
else:
|
||||
icon, label = "🔧", "Tool output"
|
||||
size_label = _format_size(block.get("size_bytes"))
|
||||
size_part = f" ({size_label})" if size_label else ""
|
||||
return (
|
||||
f"> {icon} **{label}** — `{origin}`{size_part} — omitted "
|
||||
"(EXPORTER_HIDDEN_CONTENT=full to keep)"
|
||||
)
|
||||
|
||||
# Defensive: a block of unrecognised local type (shouldn't happen if
|
||||
# constructors are used). Render as visible warning rather than dropping.
|
||||
@@ -297,6 +349,14 @@ def _render_one(block: dict) -> str:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _format_size(size_bytes: Any) -> str:
|
||||
"""Human-readable size ('24.1 KB', '1.05 MB'), or '' when absent/zero."""
|
||||
if not isinstance(size_bytes, int) or size_bytes <= 0:
|
||||
return ""
|
||||
kb = size_bytes / 1024
|
||||
return f"{kb:.1f} KB" if kb < 1024 else f"{kb / 1024:.2f} MB"
|
||||
|
||||
|
||||
def _safe_fence(text: str) -> str:
|
||||
"""Return a backtick fence longer than the longest run of backticks in ``text``.
|
||||
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
"""Extract session tokens directly from a local browser's cookie store.
|
||||
|
||||
Default browser is Brave. Uses browser-cookie3, which handles the
|
||||
platform-specific cookie decryption (Linux: AES key derived from the OS
|
||||
keyring "Safe Storage" secret; Windows: DPAPI; macOS: Keychain). The browser
|
||||
must be the one the user is actually logged in with, on the same machine.
|
||||
|
||||
Failure modes worth knowing:
|
||||
- Browser not installed / profile not found → BrowserTokenError
|
||||
- Cookie DB locked (browser running with exclusive lock; rare on Linux,
|
||||
common on Windows) → BrowserTokenError suggesting the browser be closed
|
||||
- Logged out / cookie expired → BrowserTokenError naming the missing cookie
|
||||
"""
|
||||
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_BROWSER = "brave"
|
||||
SUPPORTED_BROWSERS = ("brave", "chrome", "chromium", "edge", "firefox")
|
||||
|
||||
# ChatGPT splits large session tokens across two cookies to stay under the
|
||||
# browser's 4KB cookie limit; small tokens use the unsuffixed name.
|
||||
_CHATGPT_COOKIE_0 = "__Secure-next-auth.session-token.0"
|
||||
_CHATGPT_COOKIE_1 = "__Secure-next-auth.session-token.1"
|
||||
_CHATGPT_COOKIE_SINGLE = "__Secure-next-auth.session-token"
|
||||
_CLAUDE_COOKIE = "sessionKey"
|
||||
|
||||
|
||||
class BrowserTokenError(Exception):
|
||||
"""Raised when tokens cannot be extracted from the browser profile."""
|
||||
|
||||
|
||||
def _load_cookies(browser: str, domain: str) -> dict[str, str]:
|
||||
"""Return {cookie_name: value} for ``domain`` from the given browser."""
|
||||
if browser not in SUPPORTED_BROWSERS:
|
||||
raise BrowserTokenError(
|
||||
f"Unsupported browser {browser!r}. "
|
||||
f"Supported: {', '.join(SUPPORTED_BROWSERS)}"
|
||||
)
|
||||
|
||||
import browser_cookie3
|
||||
|
||||
loader = getattr(browser_cookie3, browser)
|
||||
try:
|
||||
jar = loader(domain_name=domain)
|
||||
except Exception as e:
|
||||
# browser_cookie3 raises BrowserCookieError plus assorted OS/keyring
|
||||
# errors. All mean the same thing to the caller: fall back to manual.
|
||||
hint = ""
|
||||
text = str(e).lower()
|
||||
if "lock" in text or "database is locked" in text:
|
||||
hint = " Close the browser and try again."
|
||||
elif "not find" in text or "no such file" in text:
|
||||
hint = " Is the browser installed on this machine?"
|
||||
raise BrowserTokenError(
|
||||
f"Could not read {browser} cookies for {domain}: {e}.{hint}"
|
||||
) from e
|
||||
|
||||
return {cookie.name: cookie.value or "" for cookie in jar}
|
||||
|
||||
|
||||
def extract_chatgpt_tokens(browser: str = DEFAULT_BROWSER) -> tuple[str, str | None]:
|
||||
"""Return (session_token_0, session_token_1_or_None) from the browser.
|
||||
|
||||
Raises BrowserTokenError when no session cookie is present (not logged in,
|
||||
or logged out since the last visit).
|
||||
"""
|
||||
cookies = _load_cookies(browser, "chatgpt.com")
|
||||
token_0 = cookies.get(_CHATGPT_COOKIE_0, "").strip()
|
||||
token_1 = cookies.get(_CHATGPT_COOKIE_1, "").strip() or None
|
||||
if token_0:
|
||||
logger.info(
|
||||
"[browser-tokens] ChatGPT session cookies found in %s (chunked: %s)",
|
||||
browser,
|
||||
"yes" if token_1 else "no",
|
||||
)
|
||||
return token_0, token_1
|
||||
|
||||
single = cookies.get(_CHATGPT_COOKIE_SINGLE, "").strip()
|
||||
if single:
|
||||
logger.info("[browser-tokens] ChatGPT session cookie found in %s (single)", browser)
|
||||
return single, None
|
||||
|
||||
raise BrowserTokenError(
|
||||
f"No ChatGPT session cookie in {browser} — open https://chatgpt.com "
|
||||
"in that browser and log in, then retry."
|
||||
)
|
||||
|
||||
|
||||
def extract_claude_session(browser: str = DEFAULT_BROWSER) -> str:
|
||||
"""Return the Claude ``sessionKey`` cookie value from the browser."""
|
||||
cookies = _load_cookies(browser, "claude.ai")
|
||||
session_key = cookies.get(_CLAUDE_COOKIE, "").strip()
|
||||
if session_key:
|
||||
logger.info("[browser-tokens] Claude sessionKey found in %s", browser)
|
||||
return session_key
|
||||
raise BrowserTokenError(
|
||||
f"No Claude sessionKey cookie in {browser} — open https://claude.ai "
|
||||
"in that browser and log in, then retry."
|
||||
)
|
||||
@@ -173,6 +173,22 @@ class Cache:
|
||||
entry["joplin_synced_at"] = datetime.now(tz=timezone.utc).isoformat()
|
||||
self._save()
|
||||
|
||||
def get_joplin_resources(self, provider: str, conv_id: str) -> dict[str, str]:
|
||||
"""Return the {relative_media_path: joplin_resource_id} map for a conversation."""
|
||||
entry = self._data.get(provider, {}).get(conv_id)
|
||||
if not isinstance(entry, dict):
|
||||
return {}
|
||||
resources = entry.get("joplin_resources")
|
||||
return dict(resources) if isinstance(resources, dict) else {}
|
||||
|
||||
def set_joplin_resources(self, provider: str, conv_id: str, resources: dict[str, str]) -> None:
|
||||
"""Persist the {relative_media_path: joplin_resource_id} map for a conversation."""
|
||||
entry = self._data.get(provider, {}).get(conv_id)
|
||||
if entry is None:
|
||||
return
|
||||
entry["joplin_resources"] = dict(resources)
|
||||
self._save()
|
||||
|
||||
def get_joplin_pending(self, provider: str) -> list[tuple[str, dict]]:
|
||||
"""Return (conv_id, entry) pairs that need to be synced to Joplin.
|
||||
|
||||
@@ -210,6 +226,25 @@ class Cache:
|
||||
|
||||
return pending
|
||||
|
||||
def all_file_paths(self) -> set[str]:
|
||||
"""Return every non-empty ``file_path`` in the manifest, across providers.
|
||||
|
||||
Used by ``prune`` (files on disk but not here are stale) and by the
|
||||
doctor integrity check (files here but not on disk are missing).
|
||||
"""
|
||||
paths: set[str] = set()
|
||||
for key, entries in self._data.items():
|
||||
if not isinstance(entries, dict) or key in (
|
||||
"version",
|
||||
"last_run",
|
||||
"tos_acknowledged_at",
|
||||
):
|
||||
continue
|
||||
for entry in entries.values():
|
||||
if isinstance(entry, dict) and entry.get("file_path"):
|
||||
paths.add(entry["file_path"])
|
||||
return paths
|
||||
|
||||
def last_run(self) -> str | None:
|
||||
"""Return the ISO8601 timestamp of the last export run, or None."""
|
||||
return self._data.get("last_run")
|
||||
|
||||
+75
-1
@@ -20,6 +20,12 @@ _CLAUDE_PLACEHOLDER = ""
|
||||
# Valid OUTPUT_STRUCTURE values
|
||||
VALID_STRUCTURES = {"provider/project/year", "provider/project", "provider/year"}
|
||||
|
||||
# Valid EXPORTER_HIDDEN_CONTENT values (see src/providers/base.py)
|
||||
VALID_HIDDEN_CONTENT = {"full", "placeholder", "omit"}
|
||||
|
||||
# Valid EXPORTER_DOWNLOAD_MEDIA values (see src/media.py)
|
||||
VALID_DOWNLOAD_MEDIA = {"images", "all", "off"}
|
||||
|
||||
|
||||
class ConfigError(Exception):
|
||||
"""Raised when required configuration is missing or invalid."""
|
||||
@@ -43,6 +49,16 @@ class Config:
|
||||
# Joplin local REST API settings (Web Clipper service)
|
||||
joplin_api_token: str | None = None
|
||||
joplin_api_url: str = "http://localhost:41184"
|
||||
# Policy for content invisible in the provider web UI (retrieval dumps,
|
||||
# hidden context): full | placeholder | omit
|
||||
hidden_content: str = "placeholder"
|
||||
# Session cap: max conversations downloaded per export run (None = unlimited).
|
||||
# Every run is resumable, so a capped run just continues next time.
|
||||
max_conversations: int | None = None
|
||||
# Seconds between consecutive API requests (politeness pacing; 0 disables)
|
||||
request_delay: float = 1.0
|
||||
# Asset download policy: images | all | off
|
||||
download_media: str = "images"
|
||||
|
||||
|
||||
def load_config() -> Config:
|
||||
@@ -67,6 +83,12 @@ def load_config() -> Config:
|
||||
joplin_token = os.getenv("JOPLIN_API_TOKEN", "").strip() or None
|
||||
joplin_url = os.getenv("JOPLIN_API_URL", "http://localhost:41184").strip()
|
||||
|
||||
hidden_content = os.getenv("EXPORTER_HIDDEN_CONTENT", "").strip().lower() or "placeholder"
|
||||
|
||||
max_conversations_raw = os.getenv("MAX_CONVERSATIONS_PER_RUN", "").strip()
|
||||
request_delay_raw = os.getenv("REQUEST_DELAY", "").strip()
|
||||
download_media = os.getenv("EXPORTER_DOWNLOAD_MEDIA", "").strip().lower() or "images"
|
||||
|
||||
# Parse CHATGPT_PROJECT_IDS — comma-separated list of gizmo IDs (g-p-xxx)
|
||||
_project_ids_raw = os.getenv("CHATGPT_PROJECT_IDS", "").strip()
|
||||
chatgpt_project_ids = [
|
||||
@@ -90,6 +112,46 @@ def load_config() -> Config:
|
||||
f"Must be one of: {', '.join(sorted(VALID_STRUCTURES))}"
|
||||
)
|
||||
|
||||
# Validate hidden-content policy
|
||||
if hidden_content not in VALID_HIDDEN_CONTENT:
|
||||
errors.append(
|
||||
f"EXPORTER_HIDDEN_CONTENT '{hidden_content}' is invalid. "
|
||||
f"Must be one of: {', '.join(sorted(VALID_HIDDEN_CONTENT))}"
|
||||
)
|
||||
|
||||
# Validate media download policy
|
||||
if download_media not in VALID_DOWNLOAD_MEDIA:
|
||||
errors.append(
|
||||
f"EXPORTER_DOWNLOAD_MEDIA '{download_media}' is invalid. "
|
||||
f"Must be one of: {', '.join(sorted(VALID_DOWNLOAD_MEDIA))}"
|
||||
)
|
||||
|
||||
# Validate session cap
|
||||
max_conversations: int | None = None
|
||||
if max_conversations_raw:
|
||||
try:
|
||||
max_conversations = int(max_conversations_raw)
|
||||
except ValueError:
|
||||
errors.append(
|
||||
f"MAX_CONVERSATIONS_PER_RUN '{max_conversations_raw}' is not an integer."
|
||||
)
|
||||
else:
|
||||
if max_conversations < 1:
|
||||
errors.append(
|
||||
f"MAX_CONVERSATIONS_PER_RUN must be at least 1 (got {max_conversations})."
|
||||
)
|
||||
|
||||
# Validate request pacing
|
||||
request_delay = 1.0
|
||||
if request_delay_raw:
|
||||
try:
|
||||
request_delay = float(request_delay_raw)
|
||||
except ValueError:
|
||||
errors.append(f"REQUEST_DELAY '{request_delay_raw}' is not a number.")
|
||||
else:
|
||||
if request_delay < 0:
|
||||
errors.append(f"REQUEST_DELAY must be >= 0 (got {request_delay}).")
|
||||
|
||||
# Validate and decode ChatGPT JWT
|
||||
chatgpt_expiry: datetime | None = None
|
||||
if chatgpt_token:
|
||||
@@ -139,6 +201,10 @@ def load_config() -> Config:
|
||||
chatgpt_project_ids=chatgpt_project_ids,
|
||||
joplin_api_token=joplin_token,
|
||||
joplin_api_url=joplin_url,
|
||||
hidden_content=hidden_content,
|
||||
max_conversations=max_conversations,
|
||||
request_delay=request_delay,
|
||||
download_media=download_media,
|
||||
)
|
||||
|
||||
_log_startup_summary(config)
|
||||
@@ -223,7 +289,11 @@ def _log_startup_summary(cfg: Config) -> None:
|
||||
"Joplin: %s | "
|
||||
"export_dir=%s | "
|
||||
"structure=%s | "
|
||||
"cache_dir=%s",
|
||||
"cache_dir=%s | "
|
||||
"hidden_content=%s | "
|
||||
"max_conversations=%s | "
|
||||
"request_delay=%.1fs | "
|
||||
"download_media=%s",
|
||||
chatgpt_status,
|
||||
claude_status,
|
||||
len(cfg.chatgpt_project_ids),
|
||||
@@ -231,4 +301,8 @@ def _log_startup_summary(cfg: Config) -> None:
|
||||
cfg.export_dir,
|
||||
cfg.output_structure,
|
||||
cfg.cache_dir,
|
||||
cfg.hidden_content,
|
||||
cfg.max_conversations if cfg.max_conversations is not None else "unlimited",
|
||||
cfg.request_delay,
|
||||
cfg.download_media,
|
||||
)
|
||||
|
||||
@@ -1,7 +1,10 @@
|
||||
"""Joplin Data API client for importing notes into Joplin desktop."""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
@@ -190,6 +193,46 @@ class JoplinClient:
|
||||
self._put(f"/notes/{note_id}", {"title": title, "body": body})
|
||||
logger.info("[joplin] Note updated: %r (%s)", title, note_id)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Resources (embedded media)
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def create_resource(self, file_path: "Path", title: str | None = None) -> str:
|
||||
"""Upload a local file as a Joplin resource and return its ID.
|
||||
|
||||
Joplin's ``POST /resources`` is multipart: a ``data`` file part plus a
|
||||
``props`` JSON part. The returned ID is referenced from note bodies as
|
||||
``:/<id>``.
|
||||
"""
|
||||
url = f"{self._base_url}/resources"
|
||||
props = json.dumps({"title": title or file_path.name})
|
||||
logger.debug("[joplin] POST /resources (%s)", file_path.name)
|
||||
try:
|
||||
with file_path.open("rb") as fh:
|
||||
resp = requests.post(
|
||||
url,
|
||||
params={"token": self._token},
|
||||
files={
|
||||
"data": (file_path.name, fh),
|
||||
"props": (None, props),
|
||||
},
|
||||
timeout=_REQUEST_TIMEOUT,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
resource_id = resp.json()["id"]
|
||||
logger.info("[joplin] Resource created: %s → %s", file_path.name, resource_id)
|
||||
return resource_id
|
||||
except requests.exceptions.ConnectionError as e:
|
||||
raise JoplinError(
|
||||
"Cannot connect to Joplin. Is Joplin desktop running with Web Clipper enabled?"
|
||||
) from e
|
||||
except requests.exceptions.Timeout as e:
|
||||
raise JoplinError(_timeout_message("POST", "/resources")) from e
|
||||
except requests.exceptions.HTTPError as e:
|
||||
raise JoplinError(_http_error_message("POST", "/resources", e)) from e
|
||||
except (requests.exceptions.RequestException, KeyError, OSError) as e:
|
||||
raise JoplinError(f"Joplin resource upload failed for {file_path.name}: {e}") from e
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# HTTP helpers
|
||||
# ------------------------------------------------------------------
|
||||
@@ -304,6 +347,46 @@ def _http_error_message(method: str, path: str, e: requests.exceptions.HTTPError
|
||||
return f"Joplin {method} {path} failed: HTTP {status}{body_snippet}"
|
||||
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Embedded-media rewriting
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
# Markdown image/link targets pointing at the local media/ sibling directory,
|
||||
# e.g.  or [name](media/clip.wav).
|
||||
_MEDIA_LINK_RE = re.compile(r"(!?\[[^\]]*\])\((media/[^)\s]+)\)")
|
||||
|
||||
|
||||
def upload_media_and_rewrite(
|
||||
body: str,
|
||||
note_dir: Path,
|
||||
client: "JoplinClient",
|
||||
resource_map: dict[str, str],
|
||||
) -> tuple[str, dict[str, str]]:
|
||||
"""Rewrite local ``media/…`` links to Joplin ``:/resourceId`` references.
|
||||
|
||||
Uploads each referenced file as a Joplin resource (once), rewriting both
|
||||
image embeds and file links. ``resource_map`` (relative-path → resource_id)
|
||||
is the idempotency record from the cache: known paths are reused, new ones
|
||||
are added and returned so the caller can persist them. Missing files are
|
||||
left as-is so the link still points at the on-disk copy.
|
||||
"""
|
||||
new_map = dict(resource_map)
|
||||
|
||||
def replace(match: re.Match) -> str:
|
||||
label, rel_path = match.group(1), match.group(2)
|
||||
resource_id = new_map.get(rel_path)
|
||||
if resource_id is None:
|
||||
file_path = (note_dir / rel_path).resolve()
|
||||
if not file_path.is_file():
|
||||
logger.warning("[joplin] Media file missing, leaving link: %s", rel_path)
|
||||
return match.group(0)
|
||||
resource_id = client.create_resource(file_path)
|
||||
new_map[rel_path] = resource_id
|
||||
return f"{label}(:/{resource_id})"
|
||||
|
||||
return _MEDIA_LINK_RE.sub(replace, body), new_map
|
||||
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Notebook naming helper
|
||||
# ------------------------------------------------------------------
|
||||
@@ -312,6 +395,9 @@ def _http_error_message(method: str, path: str, e: requests.exceptions.HTTPError
|
||||
_PROVIDER_DISPLAY = {
|
||||
"chatgpt": "AI-ChatGPT",
|
||||
"claude": "AI-Claude",
|
||||
# Decision 2026-06-12: Claude Code coding projects nest under the same
|
||||
# AI-Claude parent, alongside Claude web projects.
|
||||
"claude-code": "AI-Claude",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -23,11 +23,20 @@ class LossReport:
|
||||
unknown_blocks: Counter = field(default_factory=Counter)
|
||||
extraction_failures: Counter = field(default_factory=Counter)
|
||||
filtered_roles: Counter = field(default_factory=Counter)
|
||||
# Messages collapsed/omitted by the EXPORTER_HIDDEN_CONTENT policy —
|
||||
# intentional, but still surfaced so the omission is never invisible.
|
||||
collapsed: Counter = field(default_factory=Counter)
|
||||
collapsed_bytes: int = 0
|
||||
|
||||
# Aggregate counters
|
||||
messages_rendered: int = 0
|
||||
conversations: int = 0
|
||||
|
||||
# Media downloads (EXPORTER_DOWNLOAD_MEDIA): successes / skips / failures
|
||||
media_downloaded: int = 0
|
||||
media_downloaded_bytes: int = 0
|
||||
media_failed: Counter = field(default_factory=Counter)
|
||||
|
||||
# Recording -------------------------------------------------------------
|
||||
|
||||
def record_unknown(self, raw_type: str) -> None:
|
||||
@@ -39,6 +48,19 @@ class LossReport:
|
||||
def record_filtered_role(self, role: str) -> None:
|
||||
self.filtered_roles[role or "?"] += 1
|
||||
|
||||
def record_collapsed(self, origin: str, size_bytes: int = 0) -> None:
|
||||
self.collapsed[origin or "?"] += 1
|
||||
if isinstance(size_bytes, int) and size_bytes > 0:
|
||||
self.collapsed_bytes += size_bytes
|
||||
|
||||
def record_media_downloaded(self, size_bytes: int = 0) -> None:
|
||||
self.media_downloaded += 1
|
||||
if isinstance(size_bytes, int) and size_bytes > 0:
|
||||
self.media_downloaded_bytes += size_bytes
|
||||
|
||||
def record_media_failed(self, reason: str) -> None:
|
||||
self.media_failed[reason or "?"] += 1
|
||||
|
||||
def record_message(self) -> None:
|
||||
self.messages_rendered += 1
|
||||
|
||||
@@ -58,6 +80,22 @@ class LossReport:
|
||||
lines.append(f" messages rendered: {self.messages_rendered}")
|
||||
lines.extend(_format_section("unknown blocks: ", self.unknown_blocks))
|
||||
lines.extend(_format_section("extraction failures: ", self.extraction_failures))
|
||||
lines.extend(_format_section("collapsed by policy: ", self.collapsed))
|
||||
if self.collapsed:
|
||||
lines.append(
|
||||
f" ≈{self.collapsed_bytes / 1024:.0f} KB omitted "
|
||||
"(intentional — EXPORTER_HIDDEN_CONTENT=full to keep)"
|
||||
)
|
||||
if self.media_downloaded or self.media_failed:
|
||||
lines.append(
|
||||
f" media downloaded: {self.media_downloaded} "
|
||||
f"(≈{self.media_downloaded_bytes / 1024:.0f} KB)"
|
||||
)
|
||||
if self.media_failed:
|
||||
total_failed = sum(self.media_failed.values())
|
||||
lines.append(f" media failed: {total_failed}")
|
||||
for reason, count in self.media_failed.most_common(_TOP_N_BREAKDOWN):
|
||||
lines.append(f" {reason}={count}")
|
||||
lines.append(
|
||||
" filtered roles: "
|
||||
"(filter lifted in v0.4.0 — counter retained for future use, expected 0)"
|
||||
|
||||
+363
-41
@@ -2,6 +2,7 @@
|
||||
|
||||
import importlib.metadata
|
||||
import logging
|
||||
import os
|
||||
import platform
|
||||
import shutil
|
||||
import sys
|
||||
@@ -109,17 +110,37 @@ def cli(ctx: click.Context, verbose: bool, quiet: bool, debug: bool, no_log_file
|
||||
|
||||
|
||||
@cli.command()
|
||||
@click.option(
|
||||
"--from-browser",
|
||||
"from_browser",
|
||||
is_flag=False,
|
||||
flag_value="brave",
|
||||
default=None,
|
||||
metavar="[BROWSER]",
|
||||
help=(
|
||||
"Extract tokens straight from a local browser's cookie store instead "
|
||||
"of the manual DevTools flow. Defaults to brave when no browser is "
|
||||
"named; also supports chrome, chromium, edge, firefox. The browser "
|
||||
"must be installed and logged in on this machine."
|
||||
),
|
||||
)
|
||||
@click.pass_context
|
||||
def auth(ctx: click.Context) -> None:
|
||||
def auth(ctx: click.Context, from_browser: str | None) -> None:
|
||||
"""Interactive setup wizard for session tokens.
|
||||
|
||||
Guides you through finding and saving your ChatGPT and Claude session
|
||||
tokens. Tokens are never echoed to the terminal.
|
||||
tokens. Tokens are never echoed to the terminal. With --from-browser,
|
||||
tokens are pulled from the browser's cookie store and written to .env
|
||||
without any copy-pasting.
|
||||
|
||||
Token lifetimes:
|
||||
ChatGPT (__Secure-next-auth.session-token): ~7 days (JWT)
|
||||
Claude (sessionKey): ~30 days (opaque string)
|
||||
"""
|
||||
if from_browser:
|
||||
_auth_from_browser(from_browser.lower())
|
||||
return
|
||||
|
||||
os_name = platform.system()
|
||||
|
||||
console.print("\n[bold cyan]AI Chat Exporter — Token Setup Wizard[/bold cyan]\n")
|
||||
@@ -279,38 +300,117 @@ def _auth_claude(os_name: str) -> None:
|
||||
_write_token_to_env("CLAUDE_SESSION_KEY", key)
|
||||
|
||||
|
||||
def _auth_from_browser(browser: str) -> None:
|
||||
"""Extract ChatGPT + Claude tokens from a local browser and write .env.
|
||||
|
||||
Each provider is validated against its live API before anything is
|
||||
written, so a stale cookie (logged out since last visit) never replaces
|
||||
a working token. Exits 1 only when nothing could be configured.
|
||||
"""
|
||||
from src.browser_tokens import (
|
||||
BrowserTokenError,
|
||||
SUPPORTED_BROWSERS,
|
||||
extract_chatgpt_tokens,
|
||||
extract_claude_session,
|
||||
)
|
||||
|
||||
if browser not in SUPPORTED_BROWSERS:
|
||||
err_console.print(
|
||||
f"[red]Unsupported browser '{browser}'. "
|
||||
f"Supported: {', '.join(SUPPORTED_BROWSERS)}[/red]"
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
console.print(f"\n[bold cyan]Extracting session tokens from {browser}…[/bold cyan]\n")
|
||||
configured = 0
|
||||
|
||||
# ── ChatGPT ──────────────────────────────────────────────────────────
|
||||
try:
|
||||
token, token_1 = extract_chatgpt_tokens(browser)
|
||||
with console.status("[dim]Validating ChatGPT token…[/dim]"):
|
||||
from src.providers.chatgpt import ChatGPTProvider
|
||||
prov = ChatGPTProvider(session_token=token, session_token_1=token_1)
|
||||
prov._fetch_access_token()
|
||||
_set_env_key("CHATGPT_SESSION_TOKEN", token)
|
||||
_set_env_key("CHATGPT_SESSION_TOKEN_1", token_1 or "")
|
||||
console.print("[green]ChatGPT: token extracted, validated, and saved.[/green]")
|
||||
configured += 1
|
||||
except BrowserTokenError as e:
|
||||
console.print(f"[yellow]ChatGPT: {e}[/yellow]")
|
||||
except ProviderError as e:
|
||||
console.print(
|
||||
f"[yellow]ChatGPT: extracted cookie failed live validation "
|
||||
f"({e.original}) — .env not changed. Log in to chatgpt.com in "
|
||||
f"{browser} and retry.[/yellow]"
|
||||
)
|
||||
|
||||
# ── Claude ───────────────────────────────────────────────────────────
|
||||
try:
|
||||
session_key = extract_claude_session(browser)
|
||||
with console.status("[dim]Validating Claude session key…[/dim]"):
|
||||
from src.providers.claude import ClaudeProvider
|
||||
prov = ClaudeProvider(session_key)
|
||||
prov.list_conversations(offset=0, limit=1)
|
||||
_set_env_key("CLAUDE_SESSION_KEY", session_key)
|
||||
console.print("[green]Claude: session key extracted, validated, and saved.[/green]")
|
||||
configured += 1
|
||||
except BrowserTokenError as e:
|
||||
console.print(f"[yellow]Claude: {e}[/yellow]")
|
||||
except ProviderError as e:
|
||||
console.print(
|
||||
f"[yellow]Claude: extracted cookie failed live validation "
|
||||
f"({e.original}) — .env not changed. Log in to claude.ai in "
|
||||
f"{browser} and retry.[/yellow]"
|
||||
)
|
||||
|
||||
if configured:
|
||||
console.print(
|
||||
f"\n[green]Done — {configured} provider(s) configured. "
|
||||
"Run 'ai-chat-exporter doctor' to verify.[/green]"
|
||||
)
|
||||
else:
|
||||
err_console.print(
|
||||
"\n[red]No tokens could be extracted. Use the manual wizard "
|
||||
"instead: ai-chat-exporter auth[/red]"
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _write_token_to_env(key: str, value: str) -> None:
|
||||
"""Write or update a key in .env, offering to create the file if it doesn't exist."""
|
||||
env_path = Path(".env")
|
||||
|
||||
if click.confirm(f"Write {key} to .env?", default=True):
|
||||
if not env_path.exists():
|
||||
# Create from example if available
|
||||
example = Path(".env.example")
|
||||
if example.exists():
|
||||
import shutil as _shutil
|
||||
_shutil.copy2(example, env_path)
|
||||
console.print("[dim]Created .env from .env.example[/dim]")
|
||||
else:
|
||||
env_path.touch()
|
||||
_set_env_key(key, value)
|
||||
|
||||
lines = env_path.read_text(encoding="utf-8").splitlines(keepends=True)
|
||||
updated = False
|
||||
new_lines = []
|
||||
for line in lines:
|
||||
if line.startswith(f"{key}=") or line.startswith(f"{key} ="):
|
||||
new_lines.append(f"{key}={value}\n")
|
||||
updated = True
|
||||
else:
|
||||
new_lines.append(line)
|
||||
|
||||
if not updated:
|
||||
new_lines.append(f"\n{key}={value}\n")
|
||||
def _set_env_key(key: str, value: str) -> None:
|
||||
"""Write or update a key in .env without prompting (creates .env if needed)."""
|
||||
env_path = Path(".env")
|
||||
if not env_path.exists():
|
||||
# Create from example if available
|
||||
example = Path(".env.example")
|
||||
if example.exists():
|
||||
import shutil as _shutil
|
||||
_shutil.copy2(example, env_path)
|
||||
console.print("[dim]Created .env from .env.example[/dim]")
|
||||
else:
|
||||
env_path.touch()
|
||||
|
||||
env_path.write_text("".join(new_lines), encoding="utf-8")
|
||||
import os
|
||||
os.chmod(env_path, 0o600)
|
||||
console.print(f"[green]{key} written to .env (permissions: 600)[/green]")
|
||||
lines = env_path.read_text(encoding="utf-8").splitlines(keepends=True)
|
||||
updated = False
|
||||
new_lines = []
|
||||
for line in lines:
|
||||
if line.startswith(f"{key}=") or line.startswith(f"{key} ="):
|
||||
new_lines.append(f"{key}={value}\n")
|
||||
updated = True
|
||||
else:
|
||||
new_lines.append(line)
|
||||
|
||||
if not updated:
|
||||
new_lines.append(f"\n{key}={value}\n")
|
||||
|
||||
env_path.write_text("".join(new_lines), encoding="utf-8")
|
||||
os.chmod(env_path, 0o600)
|
||||
console.print(f"[green]{key} written to .env (permissions: 600)[/green]")
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
@@ -326,15 +426,19 @@ def doctor(ctx: click.Context) -> None:
|
||||
Checks token presence, format, expiry, directory permissions, disk space,
|
||||
and live API reachability. Exits with code 1 if any checks fail.
|
||||
"""
|
||||
checks = _run_doctor_checks()
|
||||
checks = _run_doctor_checks(cache=ctx.obj["cache"])
|
||||
_print_doctor_table(checks)
|
||||
|
||||
if any(not c["pass"] for c in checks):
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _run_doctor_checks() -> list[dict]:
|
||||
"""Run all doctor checks and return results."""
|
||||
def _run_doctor_checks(cache: Cache | None = None) -> list[dict]:
|
||||
"""Run all doctor checks and return results.
|
||||
|
||||
``cache``: when provided, also verify manifest ↔ disk integrity (every
|
||||
``file_path`` recorded in the manifest exists on disk).
|
||||
"""
|
||||
import os
|
||||
import jwt as pyjwt
|
||||
from datetime import timezone
|
||||
@@ -405,6 +509,18 @@ def _run_doctor_checks() -> list[dict]:
|
||||
except OSError as e:
|
||||
add("Disk space check", False, str(e))
|
||||
|
||||
# Manifest ↔ disk integrity
|
||||
if cache is not None:
|
||||
referenced = cache.all_file_paths()
|
||||
if referenced:
|
||||
missing = sorted(
|
||||
p for p in referenced if not Path(p).expanduser().exists()
|
||||
)
|
||||
detail = f"{len(referenced) - len(missing)}/{len(referenced)} manifest files present"
|
||||
if missing:
|
||||
detail += f"; first missing: {missing[0]} — re-export to fix"
|
||||
add("Manifest files on disk", not missing, detail)
|
||||
|
||||
# API reachability
|
||||
if chatgpt_token:
|
||||
try:
|
||||
@@ -453,7 +569,7 @@ def _print_doctor_table(checks: list[dict]) -> None:
|
||||
@cli.command()
|
||||
@click.option(
|
||||
"--provider",
|
||||
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
|
||||
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
|
||||
default="all",
|
||||
show_default=True,
|
||||
help="Which provider to export.",
|
||||
@@ -487,6 +603,37 @@ def _print_doctor_table(checks: list[dict]) -> None:
|
||||
"Use 'none' for conversations outside any project."
|
||||
),
|
||||
)
|
||||
@click.option(
|
||||
"--hidden-content",
|
||||
type=click.Choice(["full", "placeholder", "omit"], case_sensitive=False),
|
||||
default=None,
|
||||
help=(
|
||||
"What to do with content that was invisible in the provider's web UI "
|
||||
"(file-retrieval tool dumps, hidden context): keep in full, collapse to "
|
||||
"a one-line placeholder, or omit. Overrides EXPORTER_HIDDEN_CONTENT "
|
||||
"(default: placeholder)."
|
||||
),
|
||||
)
|
||||
@click.option(
|
||||
"--max-conversations",
|
||||
type=click.IntRange(min=1),
|
||||
default=None,
|
||||
help=(
|
||||
"Cap how many conversations are downloaded this run (per provider). "
|
||||
"Runs are resumable, so re-running continues where the cap stopped. "
|
||||
"Overrides MAX_CONVERSATIONS_PER_RUN (default: unlimited)."
|
||||
),
|
||||
)
|
||||
@click.option(
|
||||
"--download-media",
|
||||
type=click.Choice(["images", "all", "off"], case_sensitive=False),
|
||||
default=None,
|
||||
help=(
|
||||
"Download conversation assets next to the exports: images only, "
|
||||
"all (also audio/files), or off. Overrides EXPORTER_DOWNLOAD_MEDIA "
|
||||
"(default: images)."
|
||||
),
|
||||
)
|
||||
@click.option("--dry-run", is_flag=True, help="Show what would be exported without writing anything.")
|
||||
@click.pass_context
|
||||
def export(
|
||||
@@ -496,6 +643,9 @@ def export(
|
||||
output_dir: str | None,
|
||||
since: str | None,
|
||||
project_filter: str | None,
|
||||
hidden_content: str | None,
|
||||
max_conversations: int | None,
|
||||
download_media: str | None,
|
||||
dry_run: bool,
|
||||
) -> None:
|
||||
"""Export new and updated conversations to Markdown or JSON.
|
||||
@@ -507,6 +657,13 @@ def export(
|
||||
debug = ctx.obj.get("debug", False)
|
||||
cache: Cache = ctx.obj["cache"]
|
||||
|
||||
# CLI flags win over .env: set before load_config (load_dotenv uses
|
||||
# override=False, so an existing env var is not clobbered by .env).
|
||||
if hidden_content:
|
||||
os.environ["EXPORTER_HIDDEN_CONTENT"] = hidden_content.lower()
|
||||
if download_media:
|
||||
os.environ["EXPORTER_DOWNLOAD_MEDIA"] = download_media.lower()
|
||||
|
||||
# Load config (may raise ConfigError)
|
||||
try:
|
||||
from src.config import load_config
|
||||
@@ -517,7 +674,7 @@ def export(
|
||||
# First-run: auto-doctor
|
||||
if not cache.last_run():
|
||||
console.print("[dim]First run — checking configuration…[/dim]")
|
||||
checks = _run_doctor_checks()
|
||||
checks = _run_doctor_checks(cache=cache)
|
||||
_print_doctor_table(checks)
|
||||
if any(not c["pass"] for c in checks):
|
||||
err_console.print(
|
||||
@@ -537,6 +694,9 @@ def export(
|
||||
err_console.print(f"[red]Invalid --since date: '{since}'. Use YYYY-MM-DD.[/red]")
|
||||
sys.exit(1)
|
||||
|
||||
# Session cap: CLI flag wins over MAX_CONVERSATIONS_PER_RUN env default.
|
||||
session_cap = max_conversations if max_conversations is not None else cfg.max_conversations
|
||||
|
||||
# Determine which providers to run
|
||||
providers_to_run = _resolve_providers(provider, cfg)
|
||||
if not providers_to_run:
|
||||
@@ -580,15 +740,34 @@ def export(
|
||||
skipped = len(all_convs) - len(to_export)
|
||||
summary[prov_name]["skipped"] = skipped
|
||||
|
||||
# Apply the session cap. Resumability makes this safe: the manifest
|
||||
# records each conversation immediately, so the next run picks up
|
||||
# exactly where the cap stopped.
|
||||
deferred = 0
|
||||
if session_cap is not None and len(to_export) > session_cap:
|
||||
deferred = len(to_export) - session_cap
|
||||
to_export = to_export[:session_cap]
|
||||
summary[prov_name]["deferred"] = deferred
|
||||
|
||||
if dry_run:
|
||||
_print_dry_run_table(prov_name, to_export, prov_instance, export_base, structure, skipped)
|
||||
if deferred:
|
||||
console.print(
|
||||
f" [yellow]Session cap {session_cap}: {deferred} more pending beyond this run.[/yellow]"
|
||||
)
|
||||
continue
|
||||
|
||||
if not to_export:
|
||||
console.print(f" [dim]{skipped} conversations already up to date.[/dim]")
|
||||
continue
|
||||
|
||||
console.print(f" [dim]{len(to_export)} to export, {skipped} already up to date.[/dim]")
|
||||
if deferred:
|
||||
console.print(
|
||||
f" [dim]{len(to_export)} to export (session cap; {deferred} deferred), "
|
||||
f"{skipped} already up to date.[/dim]"
|
||||
)
|
||||
else:
|
||||
console.print(f" [dim]{len(to_export)} to export, {skipped} already up to date.[/dim]")
|
||||
|
||||
from rich.progress import Progress, SpinnerColumn, TextColumn, BarColumn, TaskProgressColumn
|
||||
|
||||
@@ -617,6 +796,19 @@ def export(
|
||||
full_raw[key] = val
|
||||
normalized = prov_instance.normalize_conversation(full_raw, loss_report)
|
||||
|
||||
# Download assets (images by default) so the renderers
|
||||
# can inline them. Failures keep placeholders; never fatal.
|
||||
if cfg.download_media != "off":
|
||||
from src.media import resolve_media
|
||||
resolve_media(
|
||||
normalized,
|
||||
prov_instance,
|
||||
export_base,
|
||||
structure,
|
||||
cfg.download_media,
|
||||
loss_report,
|
||||
)
|
||||
|
||||
exported_path: Path | None = None
|
||||
if md_exporter:
|
||||
exported_path = md_exporter.export(normalized)
|
||||
@@ -647,6 +839,12 @@ def export(
|
||||
|
||||
if not dry_run:
|
||||
_print_export_summary(summary)
|
||||
total_deferred = sum(s.get("deferred", 0) for s in summary.values())
|
||||
if total_deferred:
|
||||
console.print(
|
||||
f"[yellow]{total_deferred} conversation(s) deferred by the session cap "
|
||||
f"({session_cap} per provider per run). Re-run the same command to continue.[/yellow]"
|
||||
)
|
||||
# Emit the data-loss summary at INFO level so it lands in the log file
|
||||
# AND the operator's console (default level is INFO).
|
||||
for line in loss_report.format_summary().split("\n"):
|
||||
@@ -683,6 +881,7 @@ def _resolve_providers(provider: str, cfg) -> list[tuple[str, object]]:
|
||||
session_token=cfg.chatgpt_session_token,
|
||||
session_token_1=cfg.chatgpt_session_token_1,
|
||||
project_ids=cfg.chatgpt_project_ids,
|
||||
hidden_content=cfg.hidden_content,
|
||||
),
|
||||
))
|
||||
except ProviderError as e:
|
||||
@@ -694,6 +893,19 @@ def _resolve_providers(provider: str, cfg) -> list[tuple[str, object]]:
|
||||
if provider in ("claude", "all"):
|
||||
try_add("claude", cfg.claude_session_key, ClaudeProvider)
|
||||
|
||||
if provider in ("claude-code", "all"):
|
||||
from src.providers.claude_code import ClaudeCodeProvider, DEFAULT_PROJECTS_DIR
|
||||
cc_dir = Path(os.getenv("CLAUDE_CODE_DIR", DEFAULT_PROJECTS_DIR)).expanduser()
|
||||
if cc_dir.is_dir():
|
||||
result.append((
|
||||
"claude-code",
|
||||
ClaudeCodeProvider(projects_dir=cc_dir, hidden_content=cfg.hidden_content),
|
||||
))
|
||||
elif provider == "claude-code":
|
||||
logging.getLogger(__name__).warning(
|
||||
"[claude-code] Skipping — %s not found (set CLAUDE_CODE_DIR).", cc_dir
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
@@ -790,7 +1002,7 @@ def _print_export_summary(summary: dict[str, dict[str, int]]) -> None:
|
||||
@cli.command(name="list")
|
||||
@click.option(
|
||||
"--provider",
|
||||
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
|
||||
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
|
||||
default="all",
|
||||
show_default=True,
|
||||
)
|
||||
@@ -856,7 +1068,7 @@ def list_conversations(ctx: click.Context, provider: str, project_filter: str |
|
||||
@click.option("--clear", is_flag=True, help="Clear cached entries.")
|
||||
@click.option(
|
||||
"--provider",
|
||||
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
|
||||
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
|
||||
default="all",
|
||||
help="Provider to target (used with --clear).",
|
||||
)
|
||||
@@ -886,6 +1098,106 @@ def cache(ctx: click.Context, show: bool, clear: bool, provider: str) -> None:
|
||||
console.print("Specify --show or --clear. Use --help for options.")
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
# prune command
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@cli.command()
|
||||
@click.option("--dry-run", is_flag=True, help="List stale files without deleting anything.")
|
||||
@click.option("--yes", "-y", is_flag=True, help="Delete without a confirmation prompt.")
|
||||
@click.pass_context
|
||||
def prune(ctx: click.Context, dry_run: bool, yes: bool) -> None:
|
||||
"""Delete export files no longer referenced by the cache manifest.
|
||||
|
||||
Removes leftovers from old folder layouts and orphaned files from fixed
|
||||
bugs (e.g. pre-v0.5.0 nested year folders, files with a missing
|
||||
conversation ID). Only .md and .json files under EXPORT_DIR are
|
||||
considered; the manifest is the source of truth.
|
||||
"""
|
||||
cache_obj: Cache = ctx.obj["cache"]
|
||||
|
||||
from dotenv import load_dotenv
|
||||
load_dotenv(override=False)
|
||||
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||
|
||||
if not export_dir.exists():
|
||||
console.print(f"[yellow]Export directory {export_dir} does not exist — nothing to prune.[/yellow]")
|
||||
return
|
||||
|
||||
referenced_raw = cache_obj.all_file_paths()
|
||||
if not referenced_raw:
|
||||
# Footgun guard: with an empty manifest (e.g. right after `cache
|
||||
# --clear`) every export file would look stale and prune would wipe
|
||||
# the entire archive.
|
||||
err_console.print(
|
||||
"[red]The manifest references no export files (did you just run "
|
||||
"'cache --clear'?). Refusing to prune — run an export first.[/red]"
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
referenced = {Path(p).expanduser().resolve() for p in referenced_raw}
|
||||
on_disk = sorted(
|
||||
f for f in export_dir.rglob("*")
|
||||
if f.is_file() and f.suffix in (".md", ".json")
|
||||
)
|
||||
stale = [f for f in on_disk if f.resolve() not in referenced]
|
||||
|
||||
if not stale:
|
||||
console.print(
|
||||
f"[green]Nothing to prune — all {len(on_disk)} export files are referenced "
|
||||
"by the manifest.[/green]"
|
||||
)
|
||||
return
|
||||
|
||||
total_bytes = sum(f.stat().st_size for f in stale)
|
||||
console.print(
|
||||
f"[bold]{len(stale)} stale file(s)[/bold] "
|
||||
f"({total_bytes / (1024 * 1024):.1f} MB) not referenced by the manifest "
|
||||
f"(of {len(on_disk)} total):\n"
|
||||
)
|
||||
for f in stale:
|
||||
try:
|
||||
shown = f.relative_to(export_dir.resolve()) if f.is_absolute() else f
|
||||
except ValueError:
|
||||
shown = f
|
||||
console.print(f" [dim]{shown}[/dim]")
|
||||
|
||||
if dry_run:
|
||||
console.print("\n[yellow]Dry run — nothing deleted.[/yellow]")
|
||||
return
|
||||
|
||||
if not yes and not click.confirm(f"\nDelete these {len(stale)} files?", default=False):
|
||||
console.print("Aborted — nothing deleted.")
|
||||
return
|
||||
|
||||
deleted = 0
|
||||
for f in stale:
|
||||
try:
|
||||
f.unlink()
|
||||
deleted += 1
|
||||
except OSError as e:
|
||||
logger.error("[prune] Could not delete %s: %s", f, e)
|
||||
|
||||
# Sweep now-empty directories (deepest first).
|
||||
removed_dirs = 0
|
||||
for d in sorted(
|
||||
(d for d in export_dir.rglob("*") if d.is_dir()),
|
||||
key=lambda p: len(p.parts),
|
||||
reverse=True,
|
||||
):
|
||||
try:
|
||||
d.rmdir() # only succeeds when empty
|
||||
removed_dirs += 1
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
console.print(
|
||||
f"[green]Deleted {deleted} file(s) ({total_bytes / (1024 * 1024):.1f} MB) "
|
||||
f"and {removed_dirs} empty director(ies).[/green]"
|
||||
)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
# joplin command
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
@@ -894,7 +1206,7 @@ def cache(ctx: click.Context, show: bool, clear: bool, provider: str) -> None:
|
||||
@cli.command()
|
||||
@click.option(
|
||||
"--provider",
|
||||
type=click.Choice(["chatgpt", "claude", "all"], case_sensitive=False),
|
||||
type=click.Choice(["chatgpt", "claude", "claude-code", "all"], case_sensitive=False),
|
||||
default="all",
|
||||
show_default=True,
|
||||
help="Which provider's conversations to sync to Joplin.",
|
||||
@@ -967,10 +1279,9 @@ def joplin(ctx: click.Context, provider: str, project_filter: str | None, dry_ru
|
||||
|
||||
# Determine which providers to process
|
||||
providers_to_sync: list[str] = []
|
||||
if provider in ("chatgpt", "all"):
|
||||
providers_to_sync.append("chatgpt")
|
||||
if provider in ("claude", "all"):
|
||||
providers_to_sync.append("claude")
|
||||
for prov in ("chatgpt", "claude", "claude-code"):
|
||||
if provider in (prov, "all"):
|
||||
providers_to_sync.append(prov)
|
||||
|
||||
summary: dict[str, dict[str, int]] = {}
|
||||
|
||||
@@ -1042,6 +1353,17 @@ def joplin(ctx: click.Context, provider: str, project_filter: str | None, dry_ru
|
||||
body = Path(file_path).read_text(encoding="utf-8")
|
||||
logger.debug("[joplin] Read %d chars from %s", len(body), file_path)
|
||||
|
||||
# Upload any downloaded media as Joplin resources and
|
||||
# rewrite local media/ links to :/resourceId references.
|
||||
# The cache stores the resource map so re-syncs reuse IDs.
|
||||
from src.joplin import upload_media_and_rewrite
|
||||
res_map = cache_obj.get_joplin_resources(prov_name, conv_id)
|
||||
body, new_res_map = upload_media_and_rewrite(
|
||||
body, Path(file_path).parent, client, res_map
|
||||
)
|
||||
if new_res_map != res_map:
|
||||
cache_obj.set_joplin_resources(prov_name, conv_id, new_res_map)
|
||||
|
||||
# Get or create the nested notebook
|
||||
nb_path = notebook_path(prov_name, project)
|
||||
notebook_id = client.get_or_create_notebook_path(list(nb_path))
|
||||
|
||||
+202
@@ -0,0 +1,202 @@
|
||||
"""Media resolution — download conversation assets next to the export files.
|
||||
|
||||
Walks a normalized conversation's placeholder blocks, downloads each asset via
|
||||
the provider, saves it under a ``media/`` directory beside the conversation's
|
||||
Markdown file, and annotates the block with a relative ``local_path`` so the
|
||||
renderer emits a real inline image / file link instead of a placeholder.
|
||||
|
||||
Policy (``EXPORTER_DOWNLOAD_MEDIA``):
|
||||
images (default) — image_placeholder blocks only
|
||||
all — also file_placeholder blocks (voice-mode audio etc.;
|
||||
measured 2026-06-12: 556 clips ≈ 162MB whose transcripts
|
||||
are already in the exports as text)
|
||||
off — leave placeholders untouched
|
||||
|
||||
Failures (expired assets, network) keep the existing placeholder and are
|
||||
counted in the run summary — never fatal to the export.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
from src.blocks import BLOCK_TYPE_FILE_PLACEHOLDER, BLOCK_TYPE_IMAGE_PLACEHOLDER
|
||||
from src.loss_report import LossReport
|
||||
from src.providers.base import ProviderError
|
||||
from src.utils import build_export_path, generate_filename
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MEDIA_IMAGES = "images"
|
||||
MEDIA_ALL = "all"
|
||||
MEDIA_OFF = "off"
|
||||
VALID_MEDIA_POLICIES = {MEDIA_IMAGES, MEDIA_ALL, MEDIA_OFF}
|
||||
|
||||
_EXT_BY_MIME = {
|
||||
"image/png": ".png",
|
||||
"image/jpeg": ".jpg",
|
||||
"image/gif": ".gif",
|
||||
"image/webp": ".webp",
|
||||
"audio/wav": ".wav",
|
||||
"audio/x-wav": ".wav",
|
||||
"audio/mpeg": ".mp3",
|
||||
"audio/mp4": ".m4a",
|
||||
"audio/aac": ".aac",
|
||||
"application/pdf": ".pdf",
|
||||
}
|
||||
|
||||
|
||||
def resolve_media_policy() -> str:
|
||||
"""Read EXPORTER_DOWNLOAD_MEDIA from the environment, defaulting to images."""
|
||||
value = os.getenv("EXPORTER_DOWNLOAD_MEDIA", "").strip().lower()
|
||||
if not value:
|
||||
return MEDIA_IMAGES
|
||||
if value not in VALID_MEDIA_POLICIES:
|
||||
logger.warning(
|
||||
"EXPORTER_DOWNLOAD_MEDIA=%r is invalid (expected images|all|off) "
|
||||
"— using 'images'.",
|
||||
value,
|
||||
)
|
||||
return MEDIA_IMAGES
|
||||
return value
|
||||
|
||||
|
||||
def resolve_media(
|
||||
normalized: dict,
|
||||
provider,
|
||||
export_base: Path,
|
||||
structure: str,
|
||||
policy: str,
|
||||
report: LossReport,
|
||||
) -> int:
|
||||
"""Download assets for one normalized conversation. Returns download count.
|
||||
|
||||
Mutates placeholder blocks in place (adds ``local_path``). Idempotent:
|
||||
an asset already on disk is annotated without an API call.
|
||||
"""
|
||||
if policy == MEDIA_OFF:
|
||||
return 0
|
||||
|
||||
download = getattr(provider, "download_asset", None)
|
||||
if download is None:
|
||||
# Provider has no remote assets (e.g. claude-code local transcripts).
|
||||
return 0
|
||||
|
||||
wanted_types = {BLOCK_TYPE_IMAGE_PLACEHOLDER}
|
||||
if policy == MEDIA_ALL:
|
||||
wanted_types.add(BLOCK_TYPE_FILE_PLACEHOLDER)
|
||||
|
||||
targets = [
|
||||
block
|
||||
for message in normalized.get("messages", [])
|
||||
for block in message.get("blocks", [])
|
||||
if block.get("type") in wanted_types and block.get("ref")
|
||||
]
|
||||
if not targets:
|
||||
return 0
|
||||
|
||||
# Same path computation the exporter uses, so media/ lands beside the .md.
|
||||
filename = generate_filename(
|
||||
normalized.get("title", "Untitled"),
|
||||
normalized.get("id", ""),
|
||||
normalized.get("created_at") or "2000-01-01",
|
||||
)
|
||||
conv_dir = build_export_path(
|
||||
export_base,
|
||||
normalized.get("provider", ""),
|
||||
normalized.get("project"),
|
||||
normalized.get("created_at") or "2000-01-01",
|
||||
filename,
|
||||
structure,
|
||||
).parent
|
||||
media_dir = conv_dir / "media"
|
||||
|
||||
downloaded = 0
|
||||
for block in targets:
|
||||
ref = block["ref"]
|
||||
file_id = _safe_asset_name(provider, ref)
|
||||
if not file_id:
|
||||
report.record_media_failed("unparseable-ref")
|
||||
continue
|
||||
|
||||
existing = _find_existing(media_dir, file_id)
|
||||
if existing is not None:
|
||||
block["local_path"] = f"media/{existing.name}"
|
||||
continue
|
||||
|
||||
try:
|
||||
content, mime, file_name = download(ref)
|
||||
except ProviderError as e:
|
||||
logger.warning(
|
||||
"[media] Could not download %s: %s", ref[:60], e.original
|
||||
)
|
||||
reason = "expired-or-missing" if "404" in str(e.original) or "not found" in str(
|
||||
e.original
|
||||
).lower() else "download-error"
|
||||
report.record_media_failed(reason)
|
||||
continue
|
||||
|
||||
ext = _pick_extension(mime, file_name)
|
||||
target = media_dir / f"{file_id}{ext}"
|
||||
try:
|
||||
_write_atomic(target, content)
|
||||
except OSError as e:
|
||||
logger.error("[media] Could not write %s: %s", target, e)
|
||||
report.record_media_failed("write-error")
|
||||
continue
|
||||
|
||||
block["local_path"] = f"media/{target.name}"
|
||||
report.record_media_downloaded(len(content))
|
||||
downloaded += 1
|
||||
|
||||
return downloaded
|
||||
|
||||
|
||||
def _safe_asset_name(provider, ref: str) -> str | None:
|
||||
"""A stable, filesystem-safe name for the asset (the provider file ID)."""
|
||||
parser = getattr(provider, "parse_asset_file_id", None)
|
||||
if parser is None:
|
||||
from src.providers.chatgpt import parse_asset_file_id as parser
|
||||
file_id = parser(ref)
|
||||
if not file_id:
|
||||
return None
|
||||
return "".join(c if c.isalnum() or c in "-_." else "_" for c in file_id)
|
||||
|
||||
|
||||
def _find_existing(media_dir: Path, file_id: str) -> Path | None:
|
||||
"""Return an already-downloaded file for this asset, if present."""
|
||||
if not media_dir.is_dir():
|
||||
return None
|
||||
for candidate in media_dir.glob(f"{file_id}.*"):
|
||||
if candidate.is_file() and candidate.stat().st_size > 0:
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def _pick_extension(mime: str | None, file_name: str | None) -> str:
|
||||
if mime:
|
||||
base_mime = mime.split(";")[0].strip().lower()
|
||||
if base_mime in _EXT_BY_MIME:
|
||||
return _EXT_BY_MIME[base_mime]
|
||||
if file_name:
|
||||
suffix = Path(file_name).suffix
|
||||
if suffix and len(suffix) <= 8:
|
||||
return suffix.lower()
|
||||
return ".bin"
|
||||
|
||||
|
||||
def _write_atomic(target: Path, content: bytes) -> None:
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
fd, tmp_name = tempfile.mkstemp(dir=target.parent, suffix=".tmp")
|
||||
try:
|
||||
with os.fdopen(fd, "wb") as fh:
|
||||
fh.write(content)
|
||||
os.chmod(tmp_name, 0o600)
|
||||
os.replace(tmp_name, target)
|
||||
except OSError:
|
||||
try:
|
||||
os.unlink(tmp_name)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
@@ -1,6 +1,7 @@
|
||||
"""Abstract base class for AI chat providers."""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import random
|
||||
import time
|
||||
from abc import ABC, abstractmethod
|
||||
@@ -37,6 +38,62 @@ MAX_RETRIES = 3
|
||||
BACKOFF_BASE = 2.0
|
||||
BACKOFF_MAX = 60.0
|
||||
|
||||
# EXPORTER_HIDDEN_CONTENT policy — what to do with content that was invisible
|
||||
# in the provider's UI (retrieval dumps, hidden-flagged context, agent tool
|
||||
# traffic). Shared by all providers.
|
||||
HIDDEN_CONTENT_FULL = "full"
|
||||
HIDDEN_CONTENT_PLACEHOLDER = "placeholder"
|
||||
HIDDEN_CONTENT_OMIT = "omit"
|
||||
VALID_HIDDEN_CONTENT_POLICIES = {
|
||||
HIDDEN_CONTENT_FULL,
|
||||
HIDDEN_CONTENT_PLACEHOLDER,
|
||||
HIDDEN_CONTENT_OMIT,
|
||||
}
|
||||
|
||||
|
||||
def resolve_hidden_content_policy() -> str:
|
||||
"""Read EXPORTER_HIDDEN_CONTENT from the environment, defaulting to placeholder."""
|
||||
value = os.getenv("EXPORTER_HIDDEN_CONTENT", "").strip().lower()
|
||||
if not value:
|
||||
return HIDDEN_CONTENT_PLACEHOLDER
|
||||
if value not in VALID_HIDDEN_CONTENT_POLICIES:
|
||||
logger.warning(
|
||||
"EXPORTER_HIDDEN_CONTENT=%r is invalid (expected full|placeholder|omit) "
|
||||
"— using 'placeholder'.",
|
||||
value,
|
||||
)
|
||||
return HIDDEN_CONTENT_PLACEHOLDER
|
||||
return value
|
||||
|
||||
|
||||
# Default polite pacing between consecutive API requests (seconds). Keeps
|
||||
# traffic human-paced instead of bursty. REQUEST_DELAY=0 disables.
|
||||
DEFAULT_REQUEST_DELAY = 1.0
|
||||
|
||||
|
||||
def resolve_request_delay() -> float:
|
||||
"""Read REQUEST_DELAY from the environment, defaulting to DEFAULT_REQUEST_DELAY."""
|
||||
raw = os.getenv("REQUEST_DELAY", "").strip()
|
||||
if not raw:
|
||||
return DEFAULT_REQUEST_DELAY
|
||||
try:
|
||||
value = float(raw)
|
||||
except ValueError:
|
||||
logger.warning(
|
||||
"REQUEST_DELAY=%r is not a number — using default %.1fs.",
|
||||
raw,
|
||||
DEFAULT_REQUEST_DELAY,
|
||||
)
|
||||
return DEFAULT_REQUEST_DELAY
|
||||
if value < 0:
|
||||
logger.warning(
|
||||
"REQUEST_DELAY=%r is negative — using default %.1fs.",
|
||||
raw,
|
||||
DEFAULT_REQUEST_DELAY,
|
||||
)
|
||||
return DEFAULT_REQUEST_DELAY
|
||||
return value
|
||||
|
||||
# Realistic Chrome User-Agent
|
||||
USER_AGENT = (
|
||||
"Mozilla/5.0 (X11; Linux x86_64) "
|
||||
@@ -76,6 +133,9 @@ class BaseProvider(ABC):
|
||||
"Accept-Language": "en-US,en;q=0.9",
|
||||
}
|
||||
)
|
||||
# Polite pacing between consecutive requests (REQUEST_DELAY env).
|
||||
self._request_delay = resolve_request_delay()
|
||||
self._last_request_at: float | None = None
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Abstract interface — subclasses must implement these
|
||||
@@ -103,6 +163,23 @@ class BaseProvider(ABC):
|
||||
# Concrete helpers
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _pace(self) -> None:
|
||||
"""Sleep so consecutive requests are at least REQUEST_DELAY apart (±25% jitter).
|
||||
|
||||
getattr fallback: tests construct providers via ``__new__``, skipping
|
||||
``__init__`` — those run unpaced.
|
||||
"""
|
||||
delay = getattr(self, "_request_delay", 0)
|
||||
if delay <= 0:
|
||||
return
|
||||
target = delay * random.uniform(0.75, 1.25)
|
||||
last = getattr(self, "_last_request_at", None)
|
||||
if last is not None:
|
||||
wait = target - (time.monotonic() - last)
|
||||
if wait > 0:
|
||||
time.sleep(wait)
|
||||
self._last_request_at = time.monotonic()
|
||||
|
||||
def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]:
|
||||
"""Fetch every conversation, handling pagination automatically.
|
||||
|
||||
@@ -208,6 +285,7 @@ class BaseProvider(ABC):
|
||||
|
||||
while attempt <= MAX_RETRIES:
|
||||
attempt += 1
|
||||
self._pace()
|
||||
start = time.monotonic()
|
||||
|
||||
try:
|
||||
|
||||
+158
-10
@@ -19,6 +19,7 @@ Response: {"items": [...], "cursor": "<opaque_base64_or_null>"}
|
||||
Pagination ends when cursor is null or an empty string.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from typing import Any
|
||||
@@ -26,10 +27,13 @@ from typing import Any
|
||||
from curl_cffi import requests as curl_requests
|
||||
|
||||
from src.blocks import (
|
||||
COLLAPSED_KIND_HIDDEN_CONTEXT,
|
||||
COLLAPSED_KIND_TOOL_DUMP,
|
||||
UNKNOWN_REASON_EXTRACTION_FAILED,
|
||||
UNKNOWN_REASON_UNKNOWN_FIELD_IN_KNOWN_TYPE,
|
||||
UNKNOWN_REASON_UNKNOWN_TYPE,
|
||||
make_code_block,
|
||||
make_collapsed_block,
|
||||
make_file_placeholder,
|
||||
make_hidden_context_marker,
|
||||
make_image_placeholder,
|
||||
@@ -39,7 +43,16 @@ from src.blocks import (
|
||||
make_unknown_block,
|
||||
)
|
||||
from src.loss_report import LossReport
|
||||
from src.providers.base import BaseProvider, ProviderError, REQUEST_TIMEOUT
|
||||
from src.providers.base import (
|
||||
BaseProvider,
|
||||
HIDDEN_CONTENT_FULL,
|
||||
HIDDEN_CONTENT_OMIT,
|
||||
HIDDEN_CONTENT_PLACEHOLDER,
|
||||
ProviderError,
|
||||
REQUEST_TIMEOUT,
|
||||
VALID_HIDDEN_CONTENT_POLICIES,
|
||||
resolve_hidden_content_policy,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -50,6 +63,34 @@ AUTH_SESSION_URL = "https://chatgpt.com/api/auth/session"
|
||||
# Run: python -c "from curl_cffi.requests import BrowserType; print(list(BrowserType))"
|
||||
IMPERSONATE = "chrome120"
|
||||
|
||||
def parse_asset_file_id(ref: str) -> str | None:
|
||||
"""Extract the file ID from a ChatGPT asset pointer.
|
||||
|
||||
Observed forms (2026-06-12):
|
||||
sediment://file_00000000245c71fda50034d7f1647791 → file_…
|
||||
sediment://8456107fc383a53#file_…979c…#p_6.png (generated) → file_…
|
||||
file-service://file-AbCdEf → file-AbCdEf
|
||||
"""
|
||||
if not isinstance(ref, str) or not ref:
|
||||
return None
|
||||
if ref.startswith("file-service://"):
|
||||
return ref.removeprefix("file-service://") or None
|
||||
if ref.startswith("sediment://"):
|
||||
body = ref.removeprefix("sediment://")
|
||||
for part in body.split("#"):
|
||||
if part.startswith("file_") or part.startswith("file-"):
|
||||
return part
|
||||
return body or None
|
||||
return None
|
||||
|
||||
|
||||
# Tool-role authors whose messages are retrieval dumps — the full text of
|
||||
# attached/project files re-injected by ChatGPT on every tool run. Measured
|
||||
# 2026-06-12 across this user's three largest conversations: 86% of all
|
||||
# content bytes, not flagged is_visually_hidden_from_conversation.
|
||||
# myfiles_browser is the legacy name for the same retrieval tool.
|
||||
_COLLAPSE_TOOL_AUTHORS = {"file_search", "myfiles_browser"}
|
||||
|
||||
|
||||
class ChatGPTProvider(BaseProvider):
|
||||
"""Provider for ChatGPT conversations via the internal web API.
|
||||
@@ -72,6 +113,7 @@ class ChatGPTProvider(BaseProvider):
|
||||
session_token: str | None = None,
|
||||
session_token_1: str | None = None,
|
||||
project_ids: list[str] | None = None,
|
||||
hidden_content: str | None = None,
|
||||
) -> None:
|
||||
# Pass a curl_cffi session to the base class instead of a requests.Session.
|
||||
# curl_cffi.requests.Session is API-compatible with requests.Session.
|
||||
@@ -112,6 +154,14 @@ class ChatGPTProvider(BaseProvider):
|
||||
# Cache of project_id → display name (avoids re-fetching gizmo details)
|
||||
self._project_name_cache: dict[str, str] = {}
|
||||
|
||||
# Policy for messages invisible in the web UI (retrieval dumps,
|
||||
# hidden-flagged context): full | placeholder | omit.
|
||||
self._hidden_content = (
|
||||
hidden_content
|
||||
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
|
||||
else resolve_hidden_content_policy()
|
||||
)
|
||||
|
||||
# ChatGPT now splits large session cookies into .0 / .1 chunks.
|
||||
# Always send both named chunks; the server reassembles them.
|
||||
self._session.cookies.set(
|
||||
@@ -561,6 +611,59 @@ class ChatGPTProvider(BaseProvider):
|
||||
)
|
||||
return data
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Asset downloads
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def download_asset(self, ref: str) -> tuple[bytes, str | None, str | None]:
|
||||
"""Download a sediment:// / file-service:// asset.
|
||||
|
||||
Two hops (verified live 2026-06-12): the files endpoint returns a
|
||||
signed ``download_url``; fetching that yields the bytes. Expired
|
||||
assets (e.g. old generated images) 404 on the first hop.
|
||||
|
||||
Returns:
|
||||
(content_bytes, mime_type_or_None, file_name_or_None)
|
||||
|
||||
Raises:
|
||||
ProviderError: unparseable ref, expired/missing asset, or
|
||||
download failure.
|
||||
"""
|
||||
file_id = parse_asset_file_id(ref)
|
||||
if not file_id:
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"download_asset({ref[:40]})",
|
||||
ValueError(f"Unrecognised asset reference: {ref[:80]}"),
|
||||
)
|
||||
|
||||
meta = self._make_request("GET", f"{BASE_URL}/files/{file_id}/download")
|
||||
download_url = meta.get("download_url")
|
||||
if not download_url:
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"download_asset({file_id})",
|
||||
RuntimeError(f"No download_url in response: {meta.get('detail') or meta}"),
|
||||
)
|
||||
|
||||
self._pace()
|
||||
resp = self._session.request("GET", download_url, timeout=REQUEST_TIMEOUT)
|
||||
if resp.status_code != 200:
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"download_asset({file_id})",
|
||||
RuntimeError(f"Signed URL returned HTTP {resp.status_code}"),
|
||||
)
|
||||
mime = resp.headers.get("content-type") or None
|
||||
file_name = meta.get("file_name") or None
|
||||
logger.debug(
|
||||
"[chatgpt] Downloaded asset %s: %d bytes (%s)",
|
||||
file_id,
|
||||
len(resp.content),
|
||||
mime,
|
||||
)
|
||||
return resp.content, mime, file_name
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Normalization
|
||||
# ------------------------------------------------------------------
|
||||
@@ -598,7 +701,9 @@ class ChatGPTProvider(BaseProvider):
|
||||
)
|
||||
|
||||
mapping: dict = raw.get("mapping", {})
|
||||
messages = _extract_messages(mapping, raw, conv_id, report)
|
||||
# getattr fallback: tests construct providers via __new__, skipping __init__.
|
||||
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
|
||||
messages = _extract_messages(mapping, raw, conv_id, report, policy)
|
||||
for _ in messages:
|
||||
report.record_message()
|
||||
report.record_conversation()
|
||||
@@ -631,7 +736,11 @@ def _ts_to_iso(ts: float | int | str | None) -> str:
|
||||
|
||||
|
||||
def _extract_messages(
|
||||
mapping: dict[str, Any], raw: dict, conv_id: str, report: LossReport
|
||||
mapping: dict[str, Any],
|
||||
raw: dict,
|
||||
conv_id: str,
|
||||
report: LossReport,
|
||||
policy: str = HIDDEN_CONTENT_FULL,
|
||||
) -> list[dict]:
|
||||
"""Walk the ChatGPT conversation mapping tree to produce an ordered message list.
|
||||
|
||||
@@ -661,7 +770,7 @@ def _extract_messages(
|
||||
node = mapping.get(node_id, {})
|
||||
msg_data = node.get("message")
|
||||
if msg_data:
|
||||
built = _build_message(msg_data, conv_id, node_id, report)
|
||||
built = _build_message(msg_data, conv_id, node_id, report, policy)
|
||||
if built is not None:
|
||||
messages.append(built)
|
||||
|
||||
@@ -688,12 +797,17 @@ def _find_root(mapping: dict[str, Any]) -> str | None:
|
||||
|
||||
|
||||
def _build_message(
|
||||
msg_data: dict, conv_id: str, node_id: str, report: LossReport
|
||||
msg_data: dict,
|
||||
conv_id: str,
|
||||
node_id: str,
|
||||
report: LossReport,
|
||||
policy: str = HIDDEN_CONTENT_FULL,
|
||||
) -> dict | None:
|
||||
"""Construct a normalized message dict (with ``blocks``) for one ChatGPT node.
|
||||
|
||||
Returns None for messages that should be skipped (truly empty). Otherwise
|
||||
returns a dict with ``role``, ``content_type``, ``timestamp``, ``blocks``.
|
||||
Returns None for messages that should be skipped (truly empty, or omitted
|
||||
by the EXPORTER_HIDDEN_CONTENT policy). Otherwise returns a dict with
|
||||
``role``, ``content_type``, ``timestamp``, ``blocks``.
|
||||
"""
|
||||
author = msg_data.get("author") or {}
|
||||
role = author.get("role", "") or ""
|
||||
@@ -727,9 +841,43 @@ def _build_message(
|
||||
)
|
||||
return None
|
||||
|
||||
if is_hidden:
|
||||
# Prepend a marker so the reader knows this message is hidden in the
|
||||
# source UI. The marker is content-type-agnostic.
|
||||
# Messages invisible in the web UI: tool retrieval dumps (identified by
|
||||
# author.name — they are NOT hidden-flagged) and hidden-flagged context.
|
||||
collapse_kind: str | None = None
|
||||
if role == "tool" and author_name in _COLLAPSE_TOOL_AUTHORS:
|
||||
collapse_kind = COLLAPSED_KIND_TOOL_DUMP
|
||||
elif is_hidden:
|
||||
collapse_kind = COLLAPSED_KIND_HIDDEN_CONTEXT
|
||||
|
||||
if collapse_kind is not None and policy != HIDDEN_CONTENT_FULL:
|
||||
origin = (
|
||||
author_name if collapse_kind == COLLAPSED_KIND_TOOL_DUMP else content_type
|
||||
) or "?"
|
||||
size_bytes = len(json.dumps(content_obj, ensure_ascii=False, default=str))
|
||||
report.record_collapsed(origin, size_bytes)
|
||||
logger.debug(
|
||||
"[chatgpt] %s %s message (%s, %dB) in conversation %s "
|
||||
"per EXPORTER_HIDDEN_CONTENT=%s",
|
||||
"Omitted" if policy == HIDDEN_CONTENT_OMIT else "Collapsed",
|
||||
collapse_kind,
|
||||
origin,
|
||||
size_bytes,
|
||||
conv_id[:8],
|
||||
policy,
|
||||
)
|
||||
if policy == HIDDEN_CONTENT_OMIT:
|
||||
return None
|
||||
blocks = [
|
||||
make_collapsed_block(
|
||||
origin=origin,
|
||||
content_type=content_type,
|
||||
size_bytes=size_bytes,
|
||||
kind=collapse_kind,
|
||||
)
|
||||
]
|
||||
elif is_hidden:
|
||||
# Policy 'full': prepend a marker so the reader knows this message is
|
||||
# hidden in the source UI. The marker is content-type-agnostic.
|
||||
blocks = [make_hidden_context_marker(content_type)] + blocks
|
||||
|
||||
# Vestigial content_type: "code" for code-only messages, otherwise "text"
|
||||
|
||||
@@ -0,0 +1,476 @@
|
||||
"""Claude Code session provider — archives local agent transcripts.
|
||||
|
||||
Reads JSONL session files from ``~/.claude/projects/<munged-cwd>/<uuid>.jsonl``
|
||||
(override with ``CLAUDE_CODE_DIR``). No tokens, no rate limits, no ToS risk —
|
||||
but the data lives in single files Claude Code may clean up, and it contains
|
||||
deliverables (reviews, plans, analyses) that exist nowhere else.
|
||||
|
||||
Rendering follows the EXPORTER_HIDDEN_CONTENT policy (decided 2026-06-12):
|
||||
prose-only by default. User prompts and assistant text are kept; tool_use /
|
||||
tool_result traffic is collapsed to one grouped placeholder per activity run
|
||||
(measured: dialogue prose is ~4% of session bytes); thinking blocks are
|
||||
dropped (counted in the run summary, no placeholder). ``full`` keeps
|
||||
everything.
|
||||
|
||||
Record types in a session file: ``user`` / ``assistant`` carry the dialogue
|
||||
(Anthropic-style ``message.content`` block arrays); ``ai-title`` carries the
|
||||
evolving session title (last one wins); ``last-prompt``,
|
||||
``file-history-snapshot``, ``attachment``, ``permission-mode``, ``system``
|
||||
are harness records and are skipped. Records flagged ``isSidechain`` are
|
||||
subagent transcripts; ``isMeta`` are harness-generated user records — both
|
||||
skipped.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from collections import Counter
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from src.blocks import (
|
||||
COLLAPSED_KIND_TOOL_DUMP,
|
||||
UNKNOWN_REASON_UNKNOWN_TYPE,
|
||||
make_collapsed_block,
|
||||
make_image_placeholder,
|
||||
make_text_block,
|
||||
make_thinking_block,
|
||||
make_tool_result_block,
|
||||
make_tool_use_block,
|
||||
make_unknown_block,
|
||||
)
|
||||
from src.loss_report import LossReport
|
||||
from src.providers.base import (
|
||||
BaseProvider,
|
||||
HIDDEN_CONTENT_FULL,
|
||||
ProviderError,
|
||||
VALID_HIDDEN_CONTENT_POLICIES,
|
||||
resolve_hidden_content_policy,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_PROJECTS_DIR = "~/.claude/projects"
|
||||
|
||||
# Harness-injected tags inside user message text. Stripped so exports contain
|
||||
# the dialogue, not the CLI plumbing. A record that is nothing but tags
|
||||
# (e.g. a /model invocation) ends up empty and is skipped.
|
||||
_HARNESS_TAG_RE = re.compile(
|
||||
r"<local-command-caveat>.*?</local-command-caveat>"
|
||||
r"|<command-name>.*?</command-name>"
|
||||
r"|<command-message>.*?</command-message>"
|
||||
r"|<command-args>.*?</command-args>"
|
||||
r"|<command-contents>.*?</command-contents>"
|
||||
r"|<local-command-stdout>.*?</local-command-stdout>"
|
||||
r"|<system-reminder>.*?</system-reminder>",
|
||||
re.DOTALL,
|
||||
)
|
||||
|
||||
# How many distinct tool names to list in a collapsed-activity placeholder.
|
||||
_TOOL_NAMES_SHOWN = 4
|
||||
|
||||
|
||||
def _strip_harness_noise(text: str) -> str:
|
||||
if not isinstance(text, str):
|
||||
return ""
|
||||
return _HARNESS_TAG_RE.sub("", text).strip()
|
||||
|
||||
|
||||
class ClaudeCodeProvider(BaseProvider):
|
||||
"""Local-file provider over Claude Code session transcripts."""
|
||||
|
||||
provider_name = "claude-code"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
projects_dir: str | Path | None = None,
|
||||
hidden_content: str | None = None,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self._projects_dir = Path(
|
||||
projects_dir or os.getenv("CLAUDE_CODE_DIR", DEFAULT_PROJECTS_DIR)
|
||||
).expanduser()
|
||||
self._hidden_content = (
|
||||
hidden_content
|
||||
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
|
||||
else resolve_hidden_content_policy()
|
||||
)
|
||||
# conv_id → session file path, populated by _scan()
|
||||
self._path_map: dict[str, Path] = {}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# BaseProvider interface
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def list_conversations(self, offset: int = 0, limit: int = 100) -> list[dict]:
|
||||
full = self._scan()
|
||||
return full[offset : offset + limit]
|
||||
|
||||
def fetch_all_conversations(self, since: datetime | None = None) -> list[dict]:
|
||||
convs = self._scan()
|
||||
if since is not None:
|
||||
since_aware = since if since.tzinfo else since.replace(tzinfo=timezone.utc)
|
||||
convs = [
|
||||
c for c in convs
|
||||
if datetime.fromisoformat(c["updated_at"]) >= since_aware
|
||||
]
|
||||
logger.info(
|
||||
"[claude-code] Found %d session(s) under %s", len(convs), self._projects_dir
|
||||
)
|
||||
return convs
|
||||
|
||||
def get_conversation(self, conv_id: str) -> dict:
|
||||
path = self._path_map.get(conv_id)
|
||||
if path is None:
|
||||
# Direct call without a prior listing (e.g. tests) — scan first.
|
||||
self._scan()
|
||||
path = self._path_map.get(conv_id)
|
||||
if path is None or not path.exists():
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"get_conversation({conv_id[:8]})",
|
||||
FileNotFoundError(f"No session file for id {conv_id}"),
|
||||
)
|
||||
|
||||
records: list[dict] = []
|
||||
bad_lines = 0
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
records.append(json.loads(line))
|
||||
except json.JSONDecodeError:
|
||||
bad_lines += 1
|
||||
if bad_lines:
|
||||
logger.warning(
|
||||
"[claude-code] %s: skipped %d unparseable line(s)", path.name, bad_lines
|
||||
)
|
||||
|
||||
mtime = datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc)
|
||||
return {
|
||||
"id": conv_id,
|
||||
"_path": str(path),
|
||||
"_records": records,
|
||||
# Listing and normalized updated_at must match, or the cache
|
||||
# staleness comparison would re-export every session every run.
|
||||
"_mtime_iso": mtime.isoformat(),
|
||||
}
|
||||
|
||||
def normalize_conversation(self, raw: dict, loss_report: LossReport | None = None) -> dict:
|
||||
report = loss_report if loss_report is not None else LossReport()
|
||||
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
|
||||
conv_id = raw.get("id") or ""
|
||||
records: list[dict] = raw.get("_records") or []
|
||||
|
||||
title = _extract_title(records)
|
||||
project = _extract_project(records, raw.get("_path"))
|
||||
created_at = next(
|
||||
(r.get("timestamp") for r in records if r.get("timestamp")), ""
|
||||
)
|
||||
updated_at = raw.get("_mtime_iso") or next(
|
||||
(r.get("timestamp") for r in reversed(records) if r.get("timestamp")), ""
|
||||
)
|
||||
|
||||
messages = _extract_messages(records, conv_id, report, policy)
|
||||
for _ in messages:
|
||||
report.record_message()
|
||||
report.record_conversation()
|
||||
|
||||
return {
|
||||
"id": conv_id,
|
||||
"title": title,
|
||||
"provider": self.provider_name,
|
||||
"project": project,
|
||||
"created_at": created_at or "",
|
||||
"updated_at": updated_at or "",
|
||||
"message_count": len(messages),
|
||||
"messages": messages,
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Scanning
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _scan(self) -> list[dict]:
|
||||
if not self._projects_dir.is_dir():
|
||||
logger.warning(
|
||||
"[claude-code] Projects directory %s does not exist", self._projects_dir
|
||||
)
|
||||
return []
|
||||
|
||||
convs: list[dict] = []
|
||||
for proj_dir in sorted(p for p in self._projects_dir.iterdir() if p.is_dir()):
|
||||
for session_file in sorted(proj_dir.glob("*.jsonl")):
|
||||
try:
|
||||
stat = session_file.stat()
|
||||
except OSError:
|
||||
continue
|
||||
if stat.st_size == 0:
|
||||
continue
|
||||
conv_id = session_file.stem
|
||||
self._path_map[conv_id] = session_file
|
||||
title, project, created = _read_session_meta(session_file)
|
||||
convs.append(
|
||||
{
|
||||
"id": conv_id,
|
||||
"title": title,
|
||||
"project": project,
|
||||
# The --project filter and dry-run table read the
|
||||
# listing dict, not the normalized conversation.
|
||||
"_project_name": project,
|
||||
"created_at": created,
|
||||
"updated_at": datetime.fromtimestamp(
|
||||
stat.st_mtime, tz=timezone.utc
|
||||
).isoformat(),
|
||||
"_path": str(session_file),
|
||||
"_project_dir": proj_dir.name,
|
||||
}
|
||||
)
|
||||
return convs
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Internal helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _read_session_meta(path: Path) -> tuple[str, str | None, str]:
|
||||
"""Light single-pass scan for listing metadata: (title, project, created_at).
|
||||
|
||||
Substring guards keep this cheap — only candidate lines are JSON-parsed.
|
||||
The full-fidelity extraction happens later in normalize_conversation.
|
||||
"""
|
||||
title = ""
|
||||
project: str | None = None
|
||||
created = ""
|
||||
try:
|
||||
with path.open(encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
if '"ai-title"' in line:
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if rec.get("type") == "ai-title" and rec.get("aiTitle"):
|
||||
title = str(rec["aiTitle"]) # last one wins
|
||||
continue
|
||||
if (not created or project is None) and (
|
||||
'"timestamp"' in line or '"cwd"' in line
|
||||
):
|
||||
try:
|
||||
rec = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if not created and rec.get("timestamp"):
|
||||
created = str(rec["timestamp"])
|
||||
if project is None and rec.get("cwd"):
|
||||
project = Path(rec["cwd"]).name or None
|
||||
except OSError as e:
|
||||
logger.warning("[claude-code] Could not read %s: %s", path, e)
|
||||
return title, project, created
|
||||
|
||||
|
||||
def _extract_title(records: list[dict]) -> str:
|
||||
"""Last ai-title record wins; fall back to the first real user prompt."""
|
||||
title = ""
|
||||
for rec in records:
|
||||
if rec.get("type") == "ai-title" and rec.get("aiTitle"):
|
||||
title = str(rec["aiTitle"])
|
||||
if title:
|
||||
return title
|
||||
|
||||
for rec in records:
|
||||
if rec.get("type") != "user" or rec.get("isMeta") or rec.get("isSidechain"):
|
||||
continue
|
||||
content = (rec.get("message") or {}).get("content")
|
||||
if isinstance(content, str):
|
||||
text = _strip_harness_noise(content)
|
||||
elif isinstance(content, list):
|
||||
text = " ".join(
|
||||
_strip_harness_noise(item.get("text", ""))
|
||||
for item in content
|
||||
if isinstance(item, dict) and item.get("type") == "text"
|
||||
).strip()
|
||||
else:
|
||||
text = ""
|
||||
if text:
|
||||
return text[:80]
|
||||
return "Untitled session"
|
||||
|
||||
|
||||
def _extract_project(records: list[dict], path: str | None) -> str | None:
|
||||
"""Project = basename of the session's working directory."""
|
||||
for rec in records:
|
||||
cwd = rec.get("cwd")
|
||||
if cwd:
|
||||
name = Path(cwd).name
|
||||
if name:
|
||||
return name
|
||||
# Fallback: the munged directory name (cannot be reliably de-munged
|
||||
# because '-' is both the path separator and a legal name character).
|
||||
if path:
|
||||
return Path(path).parent.name.lstrip("-") or None
|
||||
return None
|
||||
|
||||
|
||||
def _extract_messages(
|
||||
records: list[dict], conv_id: str, report: LossReport, policy: str
|
||||
) -> list[dict]:
|
||||
messages: list[dict] = []
|
||||
# Pending collapsed tool activity: name → call count, plus total bytes.
|
||||
pending_tools: Counter = Counter()
|
||||
pending_bytes = 0
|
||||
|
||||
def flush_pending() -> None:
|
||||
nonlocal pending_bytes
|
||||
if not pending_tools:
|
||||
return
|
||||
shown = ", ".join(
|
||||
f"{name} ×{count}" for name, count in pending_tools.most_common(_TOOL_NAMES_SHOWN)
|
||||
)
|
||||
if len(pending_tools) > _TOOL_NAMES_SHOWN:
|
||||
shown += ", …"
|
||||
calls = sum(pending_tools.values())
|
||||
block = make_collapsed_block(
|
||||
origin=f"{calls} calls: {shown}",
|
||||
content_type="tool_activity",
|
||||
size_bytes=pending_bytes,
|
||||
kind=COLLAPSED_KIND_TOOL_DUMP,
|
||||
)
|
||||
if messages and messages[-1]["role"] == "assistant":
|
||||
messages[-1]["blocks"].append(block)
|
||||
else:
|
||||
messages.append(
|
||||
{"role": "tool", "content_type": "text", "timestamp": None, "blocks": [block]}
|
||||
)
|
||||
pending_tools.clear()
|
||||
pending_bytes = 0
|
||||
|
||||
for rec in records:
|
||||
if rec.get("type") not in ("user", "assistant"):
|
||||
continue
|
||||
if rec.get("isSidechain") or rec.get("isMeta"):
|
||||
continue
|
||||
|
||||
msg = rec.get("message") or {}
|
||||
role = msg.get("role") or rec["type"]
|
||||
content = msg.get("content")
|
||||
blocks: list[dict] = []
|
||||
# This record's own tool traffic — merged into pending only after the
|
||||
# record's message is appended, so the placeholder lands on it (not on
|
||||
# the previous message).
|
||||
local_tools: Counter = Counter()
|
||||
local_bytes = 0
|
||||
|
||||
if isinstance(content, str):
|
||||
text = _strip_harness_noise(content)
|
||||
block = make_text_block(text)
|
||||
if block:
|
||||
blocks.append(block)
|
||||
elif isinstance(content, list):
|
||||
for item in content:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
item_type = item.get("type", "")
|
||||
if item_type == "text":
|
||||
block = make_text_block(_strip_harness_noise(item.get("text", "")))
|
||||
if block:
|
||||
blocks.append(block)
|
||||
elif item_type in ("thinking", "redacted_thinking"):
|
||||
if policy == HIDDEN_CONTENT_FULL:
|
||||
block = make_thinking_block(
|
||||
item.get("thinking") or item.get("text") or ""
|
||||
)
|
||||
if block:
|
||||
blocks.append(block)
|
||||
else:
|
||||
# Decision 2026-06-12: thinking is dropped without a
|
||||
# placeholder, but stays visible in the run summary.
|
||||
report.record_collapsed(
|
||||
"thinking", len(json.dumps(item, default=str))
|
||||
)
|
||||
elif item_type == "tool_use":
|
||||
if policy == HIDDEN_CONTENT_FULL:
|
||||
blocks.append(
|
||||
make_tool_use_block(
|
||||
item.get("name", ""), item.get("input"), item.get("id")
|
||||
)
|
||||
)
|
||||
else:
|
||||
name = item.get("name") or "tool"
|
||||
size = len(json.dumps(item, default=str))
|
||||
local_tools[name] += 1
|
||||
local_bytes += size
|
||||
report.record_collapsed(name, size)
|
||||
elif item_type == "tool_result":
|
||||
if policy == HIDDEN_CONTENT_FULL:
|
||||
blocks.append(
|
||||
make_tool_result_block(
|
||||
_stringify_tool_result(item.get("content")),
|
||||
is_error=bool(item.get("is_error")),
|
||||
)
|
||||
)
|
||||
else:
|
||||
size = len(json.dumps(item, default=str))
|
||||
local_bytes += size
|
||||
report.record_collapsed("tool_result", size)
|
||||
elif item_type == "image":
|
||||
blocks.append(
|
||||
make_image_placeholder(ref="embedded image", source="user_upload")
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
"[claude-code] Unknown content block type %r in session %s",
|
||||
item_type,
|
||||
conv_id[:8],
|
||||
)
|
||||
report.record_unknown(f"claude-code.{item_type or '?'}")
|
||||
blocks.append(
|
||||
make_unknown_block(
|
||||
raw_type=f"claude-code.{item_type or '?'}",
|
||||
observed_keys=list(item.keys()),
|
||||
reason=UNKNOWN_REASON_UNKNOWN_TYPE,
|
||||
)
|
||||
)
|
||||
|
||||
if not blocks:
|
||||
# Tool-result-only records (and similar): traffic accumulates and
|
||||
# is attached to the message that initiated it.
|
||||
pending_tools.update(local_tools)
|
||||
pending_bytes += local_bytes
|
||||
continue
|
||||
|
||||
# Attach previous records' tool activity to the previous message
|
||||
# before starting a new one, so the placeholder lands between the
|
||||
# dialogue turns it actually occurred between.
|
||||
flush_pending()
|
||||
messages.append(
|
||||
{
|
||||
"role": role,
|
||||
"content_type": "text",
|
||||
"timestamp": rec.get("timestamp"),
|
||||
"blocks": blocks,
|
||||
}
|
||||
)
|
||||
pending_tools.update(local_tools)
|
||||
pending_bytes += local_bytes
|
||||
|
||||
flush_pending()
|
||||
return messages
|
||||
|
||||
|
||||
def _stringify_tool_result(content) -> str:
|
||||
"""tool_result content may be a string or a list of text blocks."""
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
if isinstance(content, list):
|
||||
parts = []
|
||||
for item in content:
|
||||
if isinstance(item, dict) and item.get("type") == "text":
|
||||
parts.append(item.get("text", ""))
|
||||
else:
|
||||
parts.append(json.dumps(item, default=str))
|
||||
return "\n".join(parts)
|
||||
return json.dumps(content, default=str) if content is not None else ""
|
||||
Reference in New Issue
Block a user