"Do I have to export within N days?" is answerable from the exports already on disk — the renderer records every image's outcome inline ( when saved, a placeholder when not) and the conversation date is in the filename. Group by month and source and the hypotheses separate: a clean old/new cutoff means expiry, user_upload dying at an age model_generated survives means the source matters, and losses scattered through months that otherwise downloaded fine means neither. Offline, no token, no API calls.
1550 lines
59 KiB
Python
1550 lines
59 KiB
Python
"""ChatGPT provider — accesses chat.openai.com internal web API.
|
|
|
|
ChatGPT Projects discovery
|
|
--------------------------
|
|
ChatGPT Projects are internally implemented as "snorlax"-type gizmos with IDs
|
|
starting with "g-p-". They are *not* returned by any gizmo listing endpoint
|
|
(/gizmos/mine, /gizmos/pinned, /gizmos/discovery, /gizmos/search). The
|
|
frontend appears to load project IDs from page-level state, not a dedicated
|
|
listing API.
|
|
|
|
Therefore, project IDs must be supplied by the user via CHATGPT_PROJECT_IDS.
|
|
Each project gizmo ID looks like "g-p-68c2b2b3037c8191890036fb4ae3ed9f" and
|
|
can be read from the browser URL when viewing a project:
|
|
https://chatgpt.com/g/{project-gizmo-id}-{slug}/project
|
|
|
|
Project conversations are fetched via cursor-based pagination at:
|
|
GET /backend-api/gizmos/{project_gizmo_id}/conversations?cursor=0
|
|
Response: {"items": [...], "cursor": "<opaque_base64_or_null>"}
|
|
Pagination ends when cursor is null or an empty string.
|
|
"""
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
from typing import Any
|
|
|
|
from curl_cffi import requests as curl_requests
|
|
|
|
from src.blocks import (
|
|
COLLAPSED_KIND_HIDDEN_CONTEXT,
|
|
COLLAPSED_KIND_TOOL_DUMP,
|
|
UNKNOWN_REASON_EXTRACTION_FAILED,
|
|
UNKNOWN_REASON_UNKNOWN_FIELD_IN_KNOWN_TYPE,
|
|
UNKNOWN_REASON_UNKNOWN_TYPE,
|
|
make_code_block,
|
|
make_collapsed_block,
|
|
make_file_placeholder,
|
|
make_hidden_context_marker,
|
|
make_image_placeholder,
|
|
make_text_block,
|
|
make_thinking_block,
|
|
make_tool_result_block,
|
|
make_unknown_block,
|
|
)
|
|
from src.loss_report import LossReport
|
|
from src.providers.base import (
|
|
BaseProvider,
|
|
DRIFT_ERROR,
|
|
DRIFT_OK,
|
|
DRIFT_WARN,
|
|
HIDDEN_CONTENT_FULL,
|
|
HIDDEN_CONTENT_OMIT,
|
|
HIDDEN_CONTENT_PLACEHOLDER,
|
|
ProviderError,
|
|
REQUEST_TIMEOUT,
|
|
VALID_HIDDEN_CONTENT_POLICIES,
|
|
drift_finding,
|
|
resolve_hidden_content_policy,
|
|
)
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
BASE_URL = "https://chatgpt.com/backend-api"
|
|
AUTH_SESSION_URL = "https://chatgpt.com/api/auth/session"
|
|
|
|
# Chrome version to impersonate — must match a version curl_cffi supports.
|
|
# Run: python -c "from curl_cffi.requests import BrowserType; print(list(BrowserType))"
|
|
IMPERSONATE = "chrome120"
|
|
|
|
def parse_asset_file_id(ref: str) -> str | None:
|
|
"""Extract the file ID from a ChatGPT asset pointer.
|
|
|
|
Observed forms (2026-06-12):
|
|
sediment://file_00000000245c71fda50034d7f1647791 → file_…
|
|
sediment://8456107fc383a53#file_…979c…#p_6.png (generated) → file_…
|
|
file-service://file-AbCdEf → file-AbCdEf
|
|
"""
|
|
if not isinstance(ref, str) or not ref:
|
|
return None
|
|
if ref.startswith("file-service://"):
|
|
return ref.removeprefix("file-service://") or None
|
|
if ref.startswith("sediment://"):
|
|
body = ref.removeprefix("sediment://")
|
|
for part in body.split("#"):
|
|
if part.startswith("file_") or part.startswith("file-"):
|
|
return part
|
|
return body or None
|
|
return None
|
|
|
|
|
|
# Tool-role authors whose messages are retrieval dumps — the full text of
|
|
# attached/project files re-injected by ChatGPT on every tool run. Measured
|
|
# 2026-06-12 across this user's three largest conversations: 86% of all
|
|
# content bytes, not flagged is_visually_hidden_from_conversation.
|
|
# myfiles_browser is the legacy name for the same retrieval tool.
|
|
_COLLAPSE_TOOL_AUTHORS = {"file_search", "myfiles_browser"}
|
|
|
|
# ── API-drift canary vocabularies (FUTURE.md §10) ──────────────────────────
|
|
# Tool authors seen live 2026-06-28. The collapse keys off author.name, so a
|
|
# RENAMED retrieval tool would silently bypass the collapse and re-bloat the
|
|
# archive — the canary flags any unfamiliar (role="tool", author.name) pair.
|
|
_KNOWN_TOOL_AUTHORS = _COLLAPSE_TOOL_AUTHORS | {"web.run", "python"}
|
|
# content_types the normalizer dispatches on (keep in sync with
|
|
# _extract_blocks_for_content). A new value already degrades to an `unknown`
|
|
# block + LossReport at export time; the canary flags it proactively.
|
|
_HANDLED_CONTENT_TYPES = frozenset({
|
|
"text", "multimodal_text", "execution_output", "system_error",
|
|
"tether_browsing_display", "code", "thoughts", "reasoning_recap",
|
|
"user_editable_context", "model_editable_context", "image_asset_pointer",
|
|
})
|
|
|
|
|
|
class ChatGPTProvider(BaseProvider):
|
|
"""Provider for ChatGPT conversations via the internal web API.
|
|
|
|
Uses curl_cffi to impersonate Chrome's TLS fingerprint, bypassing
|
|
Cloudflare's bot detection which blocks standard Python requests.
|
|
|
|
Authentication is a two-step process:
|
|
1. Send __Secure-next-auth.session-token as a Cookie to /api/auth/session
|
|
to obtain a short-lived accessToken.
|
|
2. Use that accessToken as the Bearer token for all backend-api calls.
|
|
|
|
Token: __Secure-next-auth.session-token cookie (~7 day lifetime).
|
|
"""
|
|
|
|
provider_name = "chatgpt"
|
|
|
|
def __init__(
|
|
self,
|
|
session_token: str | None = None,
|
|
session_token_1: str | None = None,
|
|
project_ids: list[str] | None = None,
|
|
hidden_content: str | None = None,
|
|
) -> None:
|
|
# Pass a curl_cffi session to the base class instead of a requests.Session.
|
|
# curl_cffi.requests.Session is API-compatible with requests.Session.
|
|
cf_session = curl_requests.Session(impersonate=IMPERSONATE)
|
|
super().__init__(session=cf_session) # type: ignore[arg-type]
|
|
|
|
# Remove headers that curl_cffi manages as part of its Chrome fingerprint.
|
|
# Overriding User-Agent, Accept, or Accept-Language with non-Chrome values
|
|
# creates header/TLS inconsistencies that Cloudflare's bot detection flags.
|
|
self._session.headers.pop("User-Agent", None)
|
|
self._session.headers.pop("Accept", None)
|
|
self._session.headers.pop("Accept-Language", None)
|
|
|
|
token = session_token or os.getenv("CHATGPT_SESSION_TOKEN", "").strip()
|
|
if not token:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
"init",
|
|
RuntimeError(
|
|
"CHATGPT_SESSION_TOKEN is not set. "
|
|
"Run 'ai-chat-exporter auth' to configure it."
|
|
),
|
|
)
|
|
self._session_token = token
|
|
|
|
# Second chunk of the session token (ChatGPT splits large cookies into
|
|
# __Secure-next-auth.session-token.0 and .1 to stay under the 4KB limit).
|
|
token_1 = session_token_1 or os.getenv("CHATGPT_SESSION_TOKEN_1", "").strip() or None
|
|
|
|
# Project gizmo IDs (g-p-xxx) whose conversations we'll fetch.
|
|
# ChatGPT project conversations do not appear in the default
|
|
# /conversations listing — they require explicit project IDs.
|
|
self._project_ids: list[str] = project_ids or []
|
|
|
|
# Maps conv_id → project_name; populated by fetch_all_conversations()
|
|
self._project_map: dict[str, str] = {}
|
|
|
|
# Cache of project_id → display name (avoids re-fetching gizmo details)
|
|
self._project_name_cache: dict[str, str] = {}
|
|
|
|
# Policy for messages invisible in the web UI (retrieval dumps,
|
|
# hidden-flagged context): full | placeholder | omit.
|
|
self._hidden_content = (
|
|
hidden_content
|
|
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
|
|
else resolve_hidden_content_policy()
|
|
)
|
|
|
|
# ChatGPT now splits large session cookies into .0 / .1 chunks.
|
|
# Always send both named chunks; the server reassembles them.
|
|
self._session.cookies.set(
|
|
"__Secure-next-auth.session-token.0",
|
|
token,
|
|
domain="chatgpt.com",
|
|
path="/",
|
|
)
|
|
if token_1:
|
|
self._session.cookies.set(
|
|
"__Secure-next-auth.session-token.1",
|
|
token_1,
|
|
domain="chatgpt.com",
|
|
path="/",
|
|
)
|
|
logger.debug("[chatgpt] Set both session cookie chunks (.0 and .1)")
|
|
else:
|
|
logger.debug("[chatgpt] Set session cookie chunk .0 only (no .1 configured)")
|
|
|
|
# Set only Referer and sec-fetch-* headers for the auth exchange.
|
|
# Origin is intentionally omitted: Chrome does not send Origin on
|
|
# same-origin GET requests, and its presence alongside
|
|
# sec-fetch-site: same-origin contradicts the browser fingerprint.
|
|
self._session.headers.update(
|
|
{
|
|
"Referer": "https://chatgpt.com/",
|
|
"sec-fetch-dest": "empty",
|
|
"sec-fetch-mode": "cors",
|
|
"sec-fetch-site": "same-origin",
|
|
}
|
|
)
|
|
|
|
# Exchange the session cookie for an access token
|
|
self._access_token: str = self._fetch_access_token()
|
|
|
|
# Now set backend-api headers (after auth, so they don't interfere with
|
|
# the auth exchange which expects a browser-style request).
|
|
self._session.headers["Authorization"] = f"Bearer {self._access_token}"
|
|
self._session.headers["Accept"] = "application/json"
|
|
self._session.headers["Origin"] = "https://chatgpt.com"
|
|
logger.debug(
|
|
"[chatgpt] Session initialised (Chrome TLS impersonation, %d project ID(s) configured)",
|
|
len(self._project_ids),
|
|
)
|
|
|
|
def _get_session_payload(self) -> dict:
|
|
"""GET /api/auth/session and return the parsed NextAuth JSON.
|
|
|
|
Health signal (verified live 2026-06-28): a healthy session returns
|
|
``accessToken`` with ``error`` absent/null. A *dead* session-token
|
|
returns HTTP 200 with ``error == "RefreshAccessTokenError"`` and a
|
|
STALE ``accessToken`` — so ``error``, not token presence, is the
|
|
reliable validity signal. ``expires`` is a rolling window (it advances
|
|
on every call even for a dead token) and must NOT be used to judge
|
|
validity.
|
|
"""
|
|
resp = self._session.get(AUTH_SESSION_URL, timeout=REQUEST_TIMEOUT)
|
|
resp.raise_for_status()
|
|
return resp.json()
|
|
|
|
def session_health(self) -> tuple[bool, str]:
|
|
"""Return ``(ok, detail)`` describing session-token validity.
|
|
|
|
Used by ``doctor`` for a cheap, honest token-health check. Returns
|
|
``ok=False`` with a refresh hint when NextAuth reports an auth
|
|
``error`` or omits the access token.
|
|
"""
|
|
try:
|
|
data = self._get_session_payload()
|
|
except Exception as e: # network error / non-JSON / HTTP 4xx-5xx
|
|
return False, f"/api/auth/session unreachable: {str(e)[:60]}"
|
|
error = data.get("error")
|
|
if error:
|
|
return False, (
|
|
f"Session expired (error={error}) — refresh "
|
|
"__Secure-next-auth.session-token via DevTools / 'auth'"
|
|
)
|
|
if not data.get("accessToken"):
|
|
return False, "No accessToken — refresh token via DevTools / 'auth'"
|
|
return True, "Session active"
|
|
|
|
def _fetch_access_token(self) -> str:
|
|
"""Exchange the session cookie for a Bearer access token.
|
|
|
|
Calls GET /api/auth/session — the cookie jar already contains the
|
|
session token, so no manual Cookie header is needed.
|
|
"""
|
|
logger.debug("[chatgpt] Fetching access token from %s", AUTH_SESSION_URL)
|
|
try:
|
|
data = self._get_session_payload()
|
|
except Exception as e:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
"fetch_access_token",
|
|
RuntimeError(
|
|
f"Could not exchange session token for access token: {e}. "
|
|
"Check that your CHATGPT_SESSION_TOKEN is current and not expired."
|
|
),
|
|
) from e
|
|
|
|
# A dead session-token returns HTTP 200 with error=RefreshAccessTokenError
|
|
# and a stale accessToken. Fail fast here rather than letting the stale
|
|
# token produce a confusing downstream 401.
|
|
error = data.get("error")
|
|
if error:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
"fetch_access_token",
|
|
RuntimeError(
|
|
f"Session token expired (/api/auth/session error={error!r}). "
|
|
"Refresh __Secure-next-auth.session-token via DevTools, then run "
|
|
"'ai-chat-exporter auth' or update CHATGPT_SESSION_TOKEN in .env."
|
|
),
|
|
)
|
|
|
|
access_token = data.get("accessToken")
|
|
if not access_token:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
"fetch_access_token",
|
|
RuntimeError(
|
|
"No accessToken in /api/auth/session response. "
|
|
"Your session token may be expired — run 'ai-chat-exporter auth' to refresh."
|
|
),
|
|
)
|
|
return access_token
|
|
|
|
def _handle_401(self) -> None:
|
|
msg = (
|
|
"[chatgpt] Authentication failed (401 Unauthorized). "
|
|
"Your __Secure-next-auth.session-token has likely expired (~7 day lifetime). "
|
|
"The session token is used to obtain a short-lived access token via /api/auth/session. "
|
|
"To refresh: open chatgpt.com in Chrome → F12 → Application → Cookies "
|
|
"→ find '__Secure-next-auth.session-token' → copy the value. "
|
|
"Then run 'ai-chat-exporter auth' or update CHATGPT_SESSION_TOKEN in .env."
|
|
)
|
|
logger.error(msg)
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
"authentication",
|
|
RuntimeError("401 Unauthorized — ChatGPT token expired"),
|
|
)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Default workspace conversations (offset-based pagination)
|
|
# ------------------------------------------------------------------
|
|
|
|
def list_conversations(self, offset: int = 0, limit: int = 100) -> list[dict]:
|
|
"""Fetch one page of conversations from the default workspace.
|
|
|
|
Note: Project conversations are NOT included here. They require
|
|
separate fetching via list_project_conversations().
|
|
|
|
Returns:
|
|
List of conversation summary dicts.
|
|
"""
|
|
url = f"{BASE_URL}/conversations"
|
|
params = {"offset": offset, "limit": limit, "order": "updated"}
|
|
logger.debug("[chatgpt] list_conversations: GET %s params=%s", url, params)
|
|
try:
|
|
data = self._make_request("GET", url, params=params)
|
|
except ProviderError:
|
|
raise
|
|
except Exception as e:
|
|
raise ProviderError(self.provider_name, "list_conversations", e) from e
|
|
|
|
if not isinstance(data, dict):
|
|
self._warn_unexpected_schema("list_conversations", "root")
|
|
logger.debug("[chatgpt] list_conversations: unexpected root type %s", type(data))
|
|
return []
|
|
|
|
items = data.get("items")
|
|
if items is None:
|
|
self._warn_unexpected_schema("list_conversations", "items")
|
|
logger.debug("[chatgpt] list_conversations: response keys = %s", list(data.keys()))
|
|
return []
|
|
|
|
logger.debug("[chatgpt] list_conversations: got %d items (offset=%d)", len(items), offset)
|
|
return items
|
|
|
|
# ------------------------------------------------------------------
|
|
# Project conversations (cursor-based pagination)
|
|
# ------------------------------------------------------------------
|
|
|
|
def _fetch_project_name(self, project_id: str) -> str:
|
|
"""Fetch the display name for a project gizmo.
|
|
|
|
Calls GET /backend-api/gizmos/{project_id} and returns the display
|
|
name from gizmo.display.name. Falls back to the project_id itself
|
|
if the fetch fails or the name is missing.
|
|
|
|
Result is cached in self._project_name_cache.
|
|
"""
|
|
if project_id in self._project_name_cache:
|
|
return self._project_name_cache[project_id]
|
|
|
|
url = f"{BASE_URL}/gizmos/{project_id}"
|
|
logger.debug("[chatgpt] _fetch_project_name: GET %s", url)
|
|
try:
|
|
data = self._make_request("GET", url)
|
|
gizmo = data.get("gizmo", {}) if isinstance(data, dict) else {}
|
|
name = (gizmo.get("display") or {}).get("name") or gizmo.get("name") or ""
|
|
name = name.strip() or project_id
|
|
gizmo_type = gizmo.get("gizmo_type", "?")
|
|
logger.debug(
|
|
"[chatgpt] _fetch_project_name[%s]: name=%r gizmo_type=%r",
|
|
project_id[:12],
|
|
name,
|
|
gizmo_type,
|
|
)
|
|
except ProviderError as e:
|
|
logger.warning(
|
|
"[chatgpt] Could not fetch project name for %s: %s — using ID as name",
|
|
project_id,
|
|
e,
|
|
)
|
|
name = project_id
|
|
|
|
self._project_name_cache[project_id] = name
|
|
return name
|
|
|
|
def list_project_conversations(
|
|
self, project_id: str, cursor: str = "0"
|
|
) -> tuple[list[dict], str | None]:
|
|
"""Fetch one page of conversations for a project gizmo.
|
|
|
|
Uses cursor-based pagination (not offset). The initial cursor is "0".
|
|
Subsequent cursors come from the response's "cursor" field.
|
|
|
|
Endpoint: GET /backend-api/gizmos/{project_id}/conversations?cursor=<cursor>
|
|
|
|
Returns:
|
|
(items, next_cursor) — next_cursor is None or "" when exhausted.
|
|
"""
|
|
url = f"{BASE_URL}/gizmos/{project_id}/conversations"
|
|
params = {"cursor": cursor}
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: GET %s cursor=%r",
|
|
project_id[:12],
|
|
url,
|
|
cursor,
|
|
)
|
|
|
|
try:
|
|
data = self._make_request("GET", url, params=params)
|
|
except ProviderError:
|
|
raise
|
|
except Exception as e:
|
|
raise ProviderError(self.provider_name, "list_project_conversations", e) from e
|
|
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: response type=%s",
|
|
project_id[:12],
|
|
type(data).__name__,
|
|
)
|
|
|
|
if isinstance(data, list):
|
|
# Bare list — no next cursor available
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: bare list with %d items",
|
|
project_id[:12],
|
|
len(data),
|
|
)
|
|
return data, None
|
|
|
|
if not isinstance(data, dict):
|
|
self._warn_unexpected_schema("list_project_conversations", "root")
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: unexpected type %s value=%r",
|
|
project_id[:12],
|
|
type(data),
|
|
data,
|
|
)
|
|
return [], None
|
|
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: response keys=%s",
|
|
project_id[:12],
|
|
list(data.keys()),
|
|
)
|
|
|
|
items = data.get("items") or data.get("conversations") or []
|
|
next_cursor = data.get("cursor") or None # empty string → treat as None
|
|
|
|
if not items and data:
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: no items found; full response=%r",
|
|
project_id[:12],
|
|
data,
|
|
)
|
|
|
|
logger.debug(
|
|
"[chatgpt] list_project_conversations[%s]: %d items, next_cursor=%r",
|
|
project_id[:12],
|
|
len(items),
|
|
next_cursor[:20] + "…" if next_cursor and len(next_cursor) > 20 else next_cursor,
|
|
)
|
|
return items, next_cursor
|
|
|
|
# ------------------------------------------------------------------
|
|
# Combined fetch (default workspace + all configured projects)
|
|
# ------------------------------------------------------------------
|
|
|
|
def fetch_all_conversations(self, since=None) -> list[dict]:
|
|
"""Fetch all conversations: default workspace + every configured project.
|
|
|
|
ChatGPT project conversations are not included in the default
|
|
/conversations listing. They must be fetched separately via the
|
|
gizmos conversations endpoint using project IDs from CHATGPT_PROJECT_IDS.
|
|
|
|
Builds self._project_map (conv_id → project_name) as a side effect so
|
|
that normalize_conversation() can attach the project name without an
|
|
additional API call.
|
|
|
|
Args:
|
|
since: Optional datetime — only return conversations updated at or
|
|
after this time (client-side filter, same as base class).
|
|
|
|
Returns:
|
|
Combined list of raw conversation summary dicts.
|
|
"""
|
|
# Reset maps so a fresh fetch always rebuilds them cleanly
|
|
self._project_map = {}
|
|
|
|
# --- Default workspace (base class handles offset-based pagination) ---
|
|
logger.info("[chatgpt] Fetching default workspace conversations…")
|
|
default_convs = super().fetch_all_conversations(since=None)
|
|
logger.info("[chatgpt] Default workspace: %d conversations", len(default_convs))
|
|
|
|
# --- Project conversations ---
|
|
if not self._project_ids:
|
|
logger.info(
|
|
"[chatgpt] No project IDs configured — skipping project conversations. "
|
|
"To include projects, set CHATGPT_PROJECT_IDS in .env "
|
|
"(see 'ai-chat-exporter auth' for instructions)."
|
|
)
|
|
return self._apply_since_filter(default_convs, since)
|
|
|
|
logger.info(
|
|
"[chatgpt] Fetching conversations for %d project(s): %s",
|
|
len(self._project_ids),
|
|
self._project_ids,
|
|
)
|
|
|
|
project_convs: list[dict] = []
|
|
for project_id in self._project_ids:
|
|
project_name = self._fetch_project_name(project_id)
|
|
logger.info(
|
|
"[chatgpt] Project '%s' (%s): fetching conversations…",
|
|
project_name,
|
|
project_id,
|
|
)
|
|
|
|
cursor: str = "0"
|
|
page = 0
|
|
project_total = 0
|
|
|
|
while True:
|
|
page += 1
|
|
logger.debug(
|
|
"[chatgpt] Project '%s': page %d cursor=%r",
|
|
project_name,
|
|
page,
|
|
cursor[:20] + "…" if len(cursor) > 20 else cursor,
|
|
)
|
|
|
|
try:
|
|
batch, next_cursor = self.list_project_conversations(
|
|
project_id, cursor=cursor
|
|
)
|
|
except ProviderError as e:
|
|
logger.warning(
|
|
"[chatgpt] Project '%s': failed to fetch page %d: %s — stopping pagination",
|
|
project_name,
|
|
page,
|
|
e,
|
|
)
|
|
break
|
|
|
|
if not batch:
|
|
logger.debug(
|
|
"[chatgpt] Project '%s': empty batch on page %d — done",
|
|
project_name,
|
|
page,
|
|
)
|
|
break
|
|
|
|
for conv in batch:
|
|
conv_id = conv.get("id")
|
|
if conv_id:
|
|
self._project_map[conv_id] = project_name
|
|
else:
|
|
logger.debug(
|
|
"[chatgpt] Project '%s': conversation with no id: %r",
|
|
project_name,
|
|
conv,
|
|
)
|
|
# Annotate so callers can filter by project without the map
|
|
conv["_project_name"] = project_name
|
|
|
|
project_convs.extend(batch)
|
|
project_total += len(batch)
|
|
|
|
logger.debug(
|
|
"[chatgpt] Project '%s': page %d → %d items (project total: %d)",
|
|
project_name,
|
|
page,
|
|
len(batch),
|
|
project_total,
|
|
)
|
|
|
|
if not next_cursor:
|
|
logger.debug(
|
|
"[chatgpt] Project '%s': no next cursor — pagination complete",
|
|
project_name,
|
|
)
|
|
break
|
|
|
|
cursor = next_cursor
|
|
|
|
logger.info(
|
|
"[chatgpt] Project '%s': %d conversations fetched",
|
|
project_name,
|
|
project_total,
|
|
)
|
|
|
|
all_convs = default_convs + project_convs
|
|
logger.info(
|
|
"[chatgpt] Total: %d conversations (%d default + %d from %d project(s))",
|
|
len(all_convs),
|
|
len(default_convs),
|
|
len(project_convs),
|
|
len(self._project_ids),
|
|
)
|
|
logger.debug(
|
|
"[chatgpt] _project_map: %d entries → %s",
|
|
len(self._project_map),
|
|
{k[:8]: v for k, v in self._project_map.items()},
|
|
)
|
|
|
|
return self._apply_since_filter(all_convs, since)
|
|
|
|
def _apply_since_filter(self, convs: list[dict], since) -> list[dict]:
|
|
"""Filter conversations to those updated at or after `since`."""
|
|
if since is None:
|
|
return convs
|
|
|
|
since_naive = since.replace(tzinfo=None)
|
|
filtered = []
|
|
for c in convs:
|
|
raw_ts = c.get("updated_at") or c.get("update_time") or ""
|
|
if raw_ts:
|
|
try:
|
|
from src.utils import _parse_dt
|
|
updated = _parse_dt(str(raw_ts)).replace(tzinfo=None)
|
|
if updated >= since_naive:
|
|
filtered.append(c)
|
|
except Exception:
|
|
filtered.append(c) # include if date unparseable
|
|
else:
|
|
filtered.append(c)
|
|
|
|
logger.info(
|
|
"[chatgpt] After --since filter: %d/%d conversations",
|
|
len(filtered),
|
|
len(convs),
|
|
)
|
|
return filtered
|
|
|
|
# ------------------------------------------------------------------
|
|
# Single conversation detail
|
|
# ------------------------------------------------------------------
|
|
|
|
def get_conversation(self, conv_id: str) -> dict:
|
|
"""Fetch full conversation detail for a single ID."""
|
|
url = f"{BASE_URL}/conversation/{conv_id}"
|
|
logger.debug("[chatgpt] get_conversation: GET %s", url)
|
|
try:
|
|
data = self._make_request("GET", url)
|
|
except ProviderError:
|
|
raise
|
|
except Exception as e:
|
|
raise ProviderError(self.provider_name, "get_conversation", e) from e
|
|
|
|
if not isinstance(data, dict):
|
|
self._warn_unexpected_schema("get_conversation", "root")
|
|
return {}
|
|
|
|
logger.debug(
|
|
"[chatgpt] get_conversation[%s]: keys=%s mapping_size=%d",
|
|
conv_id[:8],
|
|
list(data.keys()),
|
|
len(data.get("mapping", {})),
|
|
)
|
|
return data
|
|
|
|
# ------------------------------------------------------------------
|
|
# Asset downloads
|
|
# ------------------------------------------------------------------
|
|
|
|
def download_asset(self, ref: str) -> tuple[bytes, str | None, str | None]:
|
|
"""Download a sediment:// / file-service:// asset.
|
|
|
|
Two hops (verified live 2026-06-12): the files endpoint returns a
|
|
signed ``download_url``; fetching that yields the bytes. Expired
|
|
assets (e.g. old generated images) 404 on the first hop.
|
|
|
|
A *deleted* asset does not 404 on the first hop — it answers
|
|
``403 {"detail":"Forbidden"}``, which reads like a permissions
|
|
problem and isn't one. Measured 2026-08-17 across 18 such assets:
|
|
every one returned ``404 {"detail":"File not found"}`` on
|
|
``/files/{id}`` while assets that downloaded fine returned 200 on
|
|
both, in the same session. So a 403 here is confirmed against the
|
|
metadata endpoint before it is reported, and a missing record is
|
|
called what it is rather than a refusal.
|
|
|
|
Returns:
|
|
(content_bytes, mime_type_or_None, file_name_or_None)
|
|
|
|
Raises:
|
|
ProviderError: unparseable ref, expired/missing asset, or
|
|
download failure.
|
|
"""
|
|
file_id = parse_asset_file_id(ref)
|
|
if not file_id:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
f"download_asset({ref[:40]})",
|
|
ValueError(f"Unrecognised asset reference: {ref[:80]}"),
|
|
)
|
|
|
|
try:
|
|
meta = self._make_request("GET", f"{BASE_URL}/files/{file_id}/download")
|
|
except ProviderError as e:
|
|
if "HTTP 403" in str(e.original) and self._asset_record_missing(file_id):
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
f"download_asset({file_id})",
|
|
RuntimeError(
|
|
"Asset no longer exists — HTTP 404 'File not found' on "
|
|
"/files/{id}. The upload was deleted or expired "
|
|
"server-side; it is not recoverable from ChatGPT."
|
|
),
|
|
) from e
|
|
raise
|
|
|
|
download_url = meta.get("download_url")
|
|
if not download_url:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
f"download_asset({file_id})",
|
|
RuntimeError(f"No download_url in response: {meta.get('detail') or meta}"),
|
|
)
|
|
|
|
self._pace()
|
|
resp = self._session.request("GET", download_url, timeout=REQUEST_TIMEOUT)
|
|
if resp.status_code != 200:
|
|
raise ProviderError(
|
|
self.provider_name,
|
|
f"download_asset({file_id})",
|
|
RuntimeError(f"Signed URL returned HTTP {resp.status_code}"),
|
|
)
|
|
mime = resp.headers.get("content-type") or None
|
|
file_name = meta.get("file_name") or None
|
|
logger.debug(
|
|
"[chatgpt] Downloaded asset %s: %d bytes (%s)",
|
|
file_id,
|
|
len(resp.content),
|
|
mime,
|
|
)
|
|
return resp.content, mime, file_name
|
|
|
|
def _asset_record_missing(self, file_id: str) -> bool:
|
|
"""True if ``/files/{id}`` reports the asset gone.
|
|
|
|
Runs only on the 403 path, so it costs one extra request per failed
|
|
asset and none per successful one. Any other outcome — 200, a network
|
|
error, an unparseable body — returns False, leaving the original 403
|
|
to be reported as-is rather than guessing that it means deletion.
|
|
"""
|
|
try:
|
|
self._pace()
|
|
resp = self._session.request(
|
|
"GET", f"{BASE_URL}/files/{file_id}", timeout=REQUEST_TIMEOUT
|
|
)
|
|
except Exception as e: # noqa: BLE001 - a failed probe must not mask the 403
|
|
logger.debug(
|
|
"[chatgpt] Existence probe for %s failed: %s", file_id, e
|
|
)
|
|
return False
|
|
return resp.status_code == 404
|
|
|
|
# ------------------------------------------------------------------
|
|
# Normalization
|
|
# ------------------------------------------------------------------
|
|
|
|
def normalize_conversation(self, raw: dict, loss_report: LossReport | None = None) -> dict:
|
|
"""Transform ChatGPT raw schema to the common normalized schema.
|
|
|
|
ChatGPT stores messages in a nested ``mapping`` dict where each node
|
|
has an ``id``, ``message``, and ``children`` list. We walk the tree
|
|
from the root node to build a flat ordered message list.
|
|
|
|
Project name is looked up from self._project_map (populated by
|
|
fetch_all_conversations). The conversation detail endpoint does not
|
|
include project information.
|
|
"""
|
|
report = loss_report if loss_report is not None else LossReport()
|
|
# ChatGPT's /backend-api/conversation/<id> response uses ``conversation_id``
|
|
# at the top level (not ``id``); fixtures and listing summaries use ``id``.
|
|
# Read both so both code paths populate the normalized ``id`` correctly.
|
|
conv_id = raw.get("id") or raw.get("conversation_id") or ""
|
|
title = raw.get("title") or "Untitled"
|
|
created_at = _ts_to_iso(raw.get("create_time"))
|
|
updated_at = _ts_to_iso(raw.get("update_time"))
|
|
|
|
# Prefer _project_name annotation injected from the listing summary
|
|
# (propagated by the export loop). Fall back to _project_map lookup.
|
|
project = raw.get("_project_name") or (
|
|
self._project_map.get(conv_id) if conv_id else None
|
|
)
|
|
logger.debug(
|
|
"[chatgpt] normalize_conversation[%s]: project=%r (source=%s)",
|
|
conv_id[:8] if conv_id else "?",
|
|
project,
|
|
"_project_name" if raw.get("_project_name") else "_project_map",
|
|
)
|
|
|
|
mapping: dict = raw.get("mapping", {})
|
|
# getattr fallback: tests construct providers via __new__, skipping __init__.
|
|
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
|
|
messages = _extract_messages(mapping, raw, conv_id, report, policy)
|
|
for _ in messages:
|
|
report.record_message()
|
|
report.record_conversation()
|
|
|
|
return {
|
|
"id": conv_id,
|
|
"title": title,
|
|
"provider": "chatgpt",
|
|
"project": project,
|
|
"created_at": created_at,
|
|
"updated_at": updated_at,
|
|
"message_count": len(messages),
|
|
"messages": messages,
|
|
}
|
|
|
|
def check_drift(self) -> list[dict]:
|
|
"""Probe one listing page + one conversation; assert load-bearing fields.
|
|
|
|
See FUTURE.md §10. Asserts only the fields the normalizer depends on —
|
|
not the full response shape (which churns harmlessly). The top silent
|
|
risk is a renamed retrieval tool author bypassing the collapse.
|
|
"""
|
|
findings: list[dict] = []
|
|
page = self.list_conversations(offset=0, limit=5)
|
|
if not page:
|
|
return [drift_finding(DRIFT_WARN, "listing",
|
|
"empty listing — cannot verify shape")]
|
|
|
|
item = page[0]
|
|
for f in ("id", "title"):
|
|
if f not in item:
|
|
findings.append(drift_finding(
|
|
DRIFT_ERROR, "listing", f"summary item missing '{f}'"))
|
|
if not (item.get("update_time") or item.get("create_time")):
|
|
findings.append(drift_finding(
|
|
DRIFT_WARN, "listing", "no update_time/create_time on summary"))
|
|
|
|
conv_id = item.get("id")
|
|
if not conv_id:
|
|
return findings or [drift_finding(DRIFT_ERROR, "listing", "no id to fetch")]
|
|
raw = self.get_conversation(conv_id)
|
|
if not (raw.get("conversation_id") or raw.get("id")):
|
|
findings.append(drift_finding(
|
|
DRIFT_ERROR, "detail", "no conversation_id/id on detail"))
|
|
mapping = raw.get("mapping")
|
|
if not isinstance(mapping, dict) or not mapping:
|
|
findings.append(drift_finding(
|
|
DRIFT_ERROR, "detail", "mapping missing/empty — tree walk will yield 0 messages"))
|
|
return findings
|
|
|
|
saw_text_with_parts = False
|
|
seen_new_ct: set[str] = set()
|
|
seen_new_tool: set[str] = set()
|
|
for node in mapping.values():
|
|
if "children" not in node:
|
|
findings.append(drift_finding(
|
|
DRIFT_ERROR, "mapping", "node missing 'children' — tree walk breaks"))
|
|
break
|
|
msg = node.get("message")
|
|
if not msg:
|
|
continue
|
|
author = msg.get("author") or {}
|
|
role, name = author.get("role"), author.get("name")
|
|
content = msg.get("content") or {}
|
|
ct = content.get("content_type")
|
|
if ct and ct not in _HANDLED_CONTENT_TYPES:
|
|
seen_new_ct.add(ct)
|
|
if role == "tool" and name and name not in _KNOWN_TOOL_AUTHORS:
|
|
seen_new_tool.add(name)
|
|
if ct == "text":
|
|
parts = content.get("parts") or []
|
|
if any(isinstance(p, str) and p.strip() for p in parts):
|
|
saw_text_with_parts = True
|
|
|
|
for name in sorted(seen_new_tool):
|
|
findings.append(drift_finding(
|
|
DRIFT_WARN, "collapse",
|
|
f"unfamiliar tool author {name!r} — if it's a retrieval dump it "
|
|
"is NOT being collapsed (silent archive bloat); update "
|
|
"_COLLAPSE_TOOL_AUTHORS / _KNOWN_TOOL_AUTHORS"))
|
|
for ct in sorted(seen_new_ct):
|
|
findings.append(drift_finding(
|
|
DRIFT_WARN, "content_type",
|
|
f"new content_type {ct!r} — exported as an unknown block; add a handler"))
|
|
if not saw_text_with_parts:
|
|
findings.append(drift_finding(
|
|
DRIFT_WARN, "parts",
|
|
"no text message yielded non-empty parts — content.parts may have drifted"))
|
|
|
|
if not findings:
|
|
findings.append(drift_finding(DRIFT_OK, "schema", "all load-bearing fields present"))
|
|
return findings
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Internal helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _ts_to_iso(ts: float | int | str | None) -> str:
|
|
"""Convert a Unix timestamp (float) or ISO string to ISO8601."""
|
|
if ts is None:
|
|
return ""
|
|
if isinstance(ts, (int, float)):
|
|
from datetime import datetime, timezone
|
|
return datetime.fromtimestamp(float(ts), tz=timezone.utc).isoformat()
|
|
return str(ts)
|
|
|
|
|
|
def _extract_messages(
|
|
mapping: dict[str, Any],
|
|
raw: dict,
|
|
conv_id: str,
|
|
report: LossReport,
|
|
policy: str = HIDDEN_CONTENT_FULL,
|
|
) -> list[dict]:
|
|
"""Walk the ChatGPT conversation mapping tree to produce an ordered message list.
|
|
|
|
All roles (user/assistant/system/tool) are processed; the prior filter that
|
|
dropped non-user/assistant messages is lifted in v0.4.0 — truly empty
|
|
messages skip via the empty-content guard, anything with content renders.
|
|
"""
|
|
if not mapping:
|
|
logger.warning("[chatgpt] Conversation %s has empty mapping", conv_id[:8])
|
|
return []
|
|
|
|
root_id = _find_root(mapping)
|
|
if root_id is None:
|
|
logger.warning(
|
|
"[chatgpt] Could not determine root node for conversation %s", conv_id[:8]
|
|
)
|
|
return []
|
|
|
|
messages: list[dict] = []
|
|
visited: set[str] = set()
|
|
|
|
def walk(node_id: str) -> None:
|
|
if node_id in visited:
|
|
return
|
|
visited.add(node_id)
|
|
|
|
node = mapping.get(node_id, {})
|
|
msg_data = node.get("message")
|
|
if msg_data:
|
|
built = _build_message(msg_data, conv_id, node_id, report, policy)
|
|
if built is not None:
|
|
messages.append(built)
|
|
|
|
# Walk children in order (linear in typical conversations)
|
|
for child_id in node.get("children", []):
|
|
walk(child_id)
|
|
|
|
walk(root_id)
|
|
return messages
|
|
|
|
|
|
def _find_root(mapping: dict[str, Any]) -> str | None:
|
|
"""Find the root node ID — the node whose parent is absent or None."""
|
|
child_ids: set[str] = set()
|
|
for node in mapping.values():
|
|
for child in node.get("children", []):
|
|
child_ids.add(child)
|
|
|
|
for node_id in mapping:
|
|
if node_id not in child_ids:
|
|
return node_id
|
|
|
|
return None
|
|
|
|
|
|
def _build_message(
|
|
msg_data: dict,
|
|
conv_id: str,
|
|
node_id: str,
|
|
report: LossReport,
|
|
policy: str = HIDDEN_CONTENT_FULL,
|
|
) -> dict | None:
|
|
"""Construct a normalized message dict (with ``blocks``) for one ChatGPT node.
|
|
|
|
Returns None for messages that should be skipped (truly empty, or omitted
|
|
by the EXPORTER_HIDDEN_CONTENT policy). Otherwise returns a dict with
|
|
``role``, ``content_type``, ``timestamp``, ``blocks``.
|
|
"""
|
|
author = msg_data.get("author") or {}
|
|
role = author.get("role", "") or ""
|
|
if role not in ("user", "assistant", "system", "tool"):
|
|
# Unrecognised role — log and surface, but pass through so role metadata
|
|
# is preserved for the reader.
|
|
logger.debug(
|
|
"[chatgpt] Unrecognised role %r in conversation %s message %s",
|
|
role,
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
|
|
content_obj = msg_data.get("content") or {}
|
|
content_type = content_obj.get("content_type", "text")
|
|
ts = msg_data.get("create_time")
|
|
metadata = msg_data.get("metadata") or {}
|
|
is_hidden = bool(metadata.get("is_visually_hidden_from_conversation"))
|
|
author_name = author.get("name") or None
|
|
|
|
blocks = _extract_blocks_for_content(
|
|
content_type, content_obj, role, conv_id, node_id, report,
|
|
author_name=author_name, msg_metadata=metadata,
|
|
)
|
|
|
|
if not blocks:
|
|
logger.debug(
|
|
"[chatgpt] Skipping empty %s message in conversation %s",
|
|
content_type,
|
|
conv_id[:8],
|
|
)
|
|
return None
|
|
|
|
# Messages invisible in the web UI: tool retrieval dumps (identified by
|
|
# author.name — they are NOT hidden-flagged) and hidden-flagged context.
|
|
collapse_kind: str | None = None
|
|
if role == "tool" and author_name in _COLLAPSE_TOOL_AUTHORS:
|
|
collapse_kind = COLLAPSED_KIND_TOOL_DUMP
|
|
elif is_hidden:
|
|
collapse_kind = COLLAPSED_KIND_HIDDEN_CONTEXT
|
|
|
|
if collapse_kind is not None and policy != HIDDEN_CONTENT_FULL:
|
|
origin = (
|
|
author_name if collapse_kind == COLLAPSED_KIND_TOOL_DUMP else content_type
|
|
) or "?"
|
|
size_bytes = len(json.dumps(content_obj, ensure_ascii=False, default=str))
|
|
report.record_collapsed(origin, size_bytes)
|
|
logger.debug(
|
|
"[chatgpt] %s %s message (%s, %dB) in conversation %s "
|
|
"per EXPORTER_HIDDEN_CONTENT=%s",
|
|
"Omitted" if policy == HIDDEN_CONTENT_OMIT else "Collapsed",
|
|
collapse_kind,
|
|
origin,
|
|
size_bytes,
|
|
conv_id[:8],
|
|
policy,
|
|
)
|
|
if policy == HIDDEN_CONTENT_OMIT:
|
|
return None
|
|
blocks = [
|
|
make_collapsed_block(
|
|
origin=origin,
|
|
content_type=content_type,
|
|
size_bytes=size_bytes,
|
|
kind=collapse_kind,
|
|
)
|
|
]
|
|
elif is_hidden:
|
|
# Policy 'full': prepend a marker so the reader knows this message is
|
|
# hidden in the source UI. The marker is content-type-agnostic.
|
|
blocks = [make_hidden_context_marker(content_type)] + blocks
|
|
|
|
# Vestigial content_type: "code" for code-only messages, otherwise "text"
|
|
msg_content_type = "code" if (
|
|
len(blocks) == 1 and blocks[0].get("type") == "code"
|
|
) else "text"
|
|
|
|
return {
|
|
"role": role or "user",
|
|
"content_type": msg_content_type,
|
|
"timestamp": _ts_to_iso(ts) if ts else None,
|
|
"blocks": blocks,
|
|
}
|
|
|
|
|
|
# Content types whose ``parts`` are plain text strings.
|
|
_PLAIN_TEXT_PARTS_TYPES = {"text"}
|
|
# Content types that carry inline reasoning/thoughts.
|
|
_THINKING_TYPES = {"thoughts", "reasoning_recap"}
|
|
# Custom-Instructions / model-context types — direct fields, NOT parts.
|
|
_DIRECT_FIELD_CONTEXT_TYPES = {
|
|
"user_editable_context",
|
|
"model_editable_context",
|
|
}
|
|
# Known direct fields per context type. Anything not listed but non-null
|
|
# becomes an `unknown` block per the no-silent-drop-of-non-null-fields rule.
|
|
_USER_EDITABLE_CONTEXT_KNOWN_FIELDS = ("user_profile", "user_instructions")
|
|
_MODEL_EDITABLE_CONTEXT_KNOWN_FIELDS = (
|
|
"model_set_context",
|
|
"repository",
|
|
"repo_summary",
|
|
"structured_context",
|
|
)
|
|
|
|
|
|
def _extract_blocks_for_content(
|
|
content_type: str,
|
|
content_obj: dict,
|
|
role: str,
|
|
conv_id: str,
|
|
node_id: str,
|
|
report: LossReport,
|
|
author_name: str | None = None,
|
|
msg_metadata: dict | None = None,
|
|
) -> list[dict]:
|
|
"""Dispatch on content_type and return a list of blocks for one message."""
|
|
|
|
if content_type in _PLAIN_TEXT_PARTS_TYPES:
|
|
return _extract_text_content_type_blocks(content_obj, conv_id, node_id, report)
|
|
|
|
if content_type == "multimodal_text":
|
|
return _extract_multimodal_blocks(content_obj, role, conv_id, node_id, report)
|
|
|
|
if content_type == "execution_output":
|
|
return _extract_execution_output_blocks(
|
|
content_obj, author_name, msg_metadata or {}, conv_id, node_id
|
|
)
|
|
|
|
if content_type == "system_error":
|
|
return _extract_system_error_blocks(content_obj, author_name)
|
|
|
|
if content_type == "tether_browsing_display":
|
|
return _extract_tether_browsing_display_blocks(
|
|
content_obj, author_name, conv_id, node_id
|
|
)
|
|
|
|
if content_type == "code":
|
|
code_text = content_obj.get("text") or "\n".join(
|
|
p for p in content_obj.get("parts", []) if isinstance(p, str)
|
|
)
|
|
language = content_obj.get("language", "") or ""
|
|
block = make_code_block(code_text, language)
|
|
return [block] if block else []
|
|
|
|
if content_type in _THINKING_TYPES:
|
|
text = _join_string_parts(content_obj)
|
|
block = make_thinking_block(text)
|
|
return [block] if block else []
|
|
|
|
if content_type in _DIRECT_FIELD_CONTEXT_TYPES:
|
|
return _extract_editable_context_blocks(
|
|
content_type, content_obj, conv_id, node_id, report
|
|
)
|
|
|
|
if content_type == "image_asset_pointer":
|
|
# Top-level image (rare — usually nested inside multimodal_text).
|
|
ref = content_obj.get("asset_pointer", "")
|
|
source = "user_upload" if role == "user" else "model_generated"
|
|
return [make_image_placeholder(ref=ref, source=source)]
|
|
|
|
# Unknown content_type → visible unknown block + WARNING + tally
|
|
keys = list(content_obj.keys())
|
|
logger.warning(
|
|
"[chatgpt] Unknown content_type %r in conversation %s message %s "
|
|
"— see plan §Data-loss visibility (rendering as unknown block)",
|
|
content_type,
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
report.record_unknown(content_type or "?")
|
|
return [
|
|
make_unknown_block(
|
|
raw_type=content_type or "?",
|
|
observed_keys=keys,
|
|
reason=UNKNOWN_REASON_UNKNOWN_TYPE,
|
|
)
|
|
]
|
|
|
|
|
|
def _extract_text_content_type_blocks(
|
|
content_obj: dict, conv_id: str, node_id: str, report: LossReport
|
|
) -> list[dict]:
|
|
"""Extract blocks for ``content_type == "text"``.
|
|
|
|
Plural-parts rule: emit ONE text block per message with all string parts
|
|
joined by ``\\n``. Don't emit one block per part.
|
|
|
|
Dict parts inside a text content_type message (the suspected o1/o3 reasoning
|
|
subpart shape ``{"summary": ..., "content": ...}``) are preserved as text
|
|
today — defensive behavior pending real-data capture in v0.4.1.
|
|
"""
|
|
parts = content_obj.get("parts", []) or []
|
|
string_chunks: list[str] = []
|
|
|
|
for part in parts:
|
|
if isinstance(part, str):
|
|
string_chunks.append(part)
|
|
elif isinstance(part, dict):
|
|
part_type = part.get("content_type", "")
|
|
if part_type == "text":
|
|
txt = part.get("text", "") or ""
|
|
if txt:
|
|
string_chunks.append(txt)
|
|
elif "content" in part:
|
|
# Suspected o1/o3 reasoning subpart. Defensive: preserve as text
|
|
# block (matches current behavior). v0.4.1 reclassifies once
|
|
# the real shape is captured live.
|
|
content_val = part.get("content", "") or ""
|
|
if content_val:
|
|
string_chunks.append(content_val)
|
|
elif part_type:
|
|
# Non-text dict part inside a text content_type — surface it.
|
|
logger.warning(
|
|
"[chatgpt] Unexpected %s part inside text content_type "
|
|
"in conversation %s message %s — rendering as unknown block",
|
|
part_type,
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
report.record_unknown(part_type)
|
|
# Inline mark in the joined text so order is preserved.
|
|
string_chunks.append(
|
|
f"\n[Unknown part: type={part_type}; "
|
|
f"keys={list(part.keys())[:10]}]\n"
|
|
)
|
|
|
|
joined = "\n".join(c for c in string_chunks if c)
|
|
block = make_text_block(joined)
|
|
return [block] if block else []
|
|
|
|
|
|
def _join_string_parts(content_obj: dict) -> str:
|
|
"""Helper: join all string parts in ``parts`` with newlines."""
|
|
parts = content_obj.get("parts", []) or []
|
|
return "\n".join(p for p in parts if isinstance(p, str) and p)
|
|
|
|
|
|
def _extract_multimodal_blocks(
|
|
content_obj: dict, role: str, conv_id: str, node_id: str, report: LossReport
|
|
) -> list[dict]:
|
|
"""Extract blocks from a ``multimodal_text`` content object.
|
|
|
|
Walks ``parts`` in array order — order varies between user and assistant
|
|
turns, and the extractor preserves source ordering. Emits text +
|
|
image_placeholder + file_placeholder blocks per part.
|
|
"""
|
|
parts = content_obj.get("parts", []) or []
|
|
blocks: list[dict] = []
|
|
|
|
for part in parts:
|
|
if isinstance(part, str):
|
|
block = make_text_block(part)
|
|
if block:
|
|
blocks.append(block)
|
|
continue
|
|
|
|
if not isinstance(part, dict):
|
|
continue
|
|
|
|
part_type = part.get("content_type", "")
|
|
|
|
if part_type == "audio_transcription":
|
|
txt = part.get("text", "") or ""
|
|
block = make_text_block(txt)
|
|
if block:
|
|
blocks.append(block)
|
|
elif "text" not in part:
|
|
logger.warning(
|
|
"[chatgpt] audio_transcription part missing 'text' key "
|
|
"in conversation %s message %s",
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
report.record_extraction_failure("audio_transcription")
|
|
blocks.append(
|
|
make_unknown_block(
|
|
raw_type="audio_transcription",
|
|
observed_keys=list(part.keys()),
|
|
reason=UNKNOWN_REASON_EXTRACTION_FAILED,
|
|
summary="expected key 'text' not found",
|
|
)
|
|
)
|
|
continue
|
|
|
|
if part_type == "image_asset_pointer":
|
|
ref = part.get("asset_pointer", "")
|
|
source = "user_upload" if role == "user" else "model_generated"
|
|
mime = None
|
|
blocks.append(make_image_placeholder(ref=ref, source=source, mime=mime))
|
|
continue
|
|
|
|
if part_type == "audio_asset_pointer":
|
|
blocks.append(_audio_asset_placeholder(part))
|
|
continue
|
|
|
|
if part_type == "real_time_user_audio_video_asset_pointer":
|
|
# Wrapper carrying a nested audio_asset_pointer + optional video frames.
|
|
nested_audio = part.get("audio_asset_pointer")
|
|
if isinstance(nested_audio, dict):
|
|
blocks.append(_audio_asset_placeholder(nested_audio))
|
|
else:
|
|
logger.warning(
|
|
"[chatgpt] real_time_user_audio_video_asset_pointer missing "
|
|
"nested audio_asset_pointer in conversation %s message %s",
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
report.record_extraction_failure(
|
|
"real_time_user_audio_video_asset_pointer"
|
|
)
|
|
blocks.append(
|
|
make_unknown_block(
|
|
raw_type="real_time_user_audio_video_asset_pointer",
|
|
observed_keys=list(part.keys()),
|
|
reason=UNKNOWN_REASON_EXTRACTION_FAILED,
|
|
summary="expected nested 'audio_asset_pointer' not found",
|
|
)
|
|
)
|
|
|
|
frames = part.get("frames_asset_pointers") or []
|
|
if frames:
|
|
# Defensive: empty in all observed cases, but if non-empty
|
|
# surface as a separate file placeholder.
|
|
video_ref = part.get("video_container_asset_pointer") or "(video frames)"
|
|
blocks.append(
|
|
make_file_placeholder(
|
|
ref=str(video_ref),
|
|
mime="video/unknown",
|
|
)
|
|
)
|
|
continue
|
|
|
|
# Anything else inside multimodal_text — visible unknown block
|
|
logger.warning(
|
|
"[chatgpt] Unknown multimodal_text part type %r in conversation %s message %s",
|
|
part_type,
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
report.record_unknown(part_type or "?")
|
|
blocks.append(
|
|
make_unknown_block(
|
|
raw_type=part_type or "?",
|
|
observed_keys=list(part.keys()),
|
|
reason=UNKNOWN_REASON_UNKNOWN_TYPE,
|
|
)
|
|
)
|
|
|
|
return blocks
|
|
|
|
|
|
def _audio_asset_placeholder(audio_part: dict) -> dict:
|
|
"""Build a file_placeholder for an audio_asset_pointer dict.
|
|
|
|
Handles missing/zero metadata defensively.
|
|
"""
|
|
ref = audio_part.get("asset_pointer", "") or ""
|
|
fmt = audio_part.get("format") or "unknown"
|
|
size_bytes = audio_part.get("size_bytes")
|
|
if not isinstance(size_bytes, int) or size_bytes <= 0:
|
|
size_bytes = None
|
|
metadata = audio_part.get("metadata") or {}
|
|
start = metadata.get("start") if isinstance(metadata, dict) else None
|
|
end = metadata.get("end") if isinstance(metadata, dict) else None
|
|
duration: float | None = None
|
|
if isinstance(start, (int, float)) and isinstance(end, (int, float)):
|
|
diff = float(end) - float(start)
|
|
if diff > 0:
|
|
duration = diff
|
|
return make_file_placeholder(
|
|
ref=ref,
|
|
mime=f"audio/{fmt}" if fmt else "audio/unknown",
|
|
size_bytes=size_bytes,
|
|
duration_seconds=duration,
|
|
)
|
|
|
|
|
|
def _extract_editable_context_blocks(
|
|
content_type: str, content_obj: dict, conv_id: str, node_id: str, report: LossReport
|
|
) -> list[dict]:
|
|
"""Extract blocks from user_editable_context / model_editable_context messages.
|
|
|
|
These have no ``parts`` field — they carry direct keys. Read all known
|
|
fields, emit one labeled fenced block per non-null known field, and emit an
|
|
``unknown`` block for any unrecognised non-null direct field (no-silent-drop
|
|
rule).
|
|
"""
|
|
if content_type == "user_editable_context":
|
|
known_fields: tuple[str, ...] = _USER_EDITABLE_CONTEXT_KNOWN_FIELDS
|
|
elif content_type == "model_editable_context":
|
|
known_fields = _MODEL_EDITABLE_CONTEXT_KNOWN_FIELDS
|
|
else:
|
|
known_fields = ()
|
|
|
|
blocks: list[dict] = []
|
|
label_kind = "Custom Instructions" if content_type == "user_editable_context" else "Model Context"
|
|
|
|
for field in known_fields:
|
|
value = content_obj.get(field)
|
|
if value is None or (isinstance(value, str) and not value.strip()):
|
|
continue
|
|
if isinstance(value, (dict, list)):
|
|
# Render as a JSON-rendered text block. _safe_fence will wrap it.
|
|
import json as _json
|
|
rendered = _json.dumps(value, indent=2, default=str, ensure_ascii=False)
|
|
else:
|
|
rendered = str(value)
|
|
label = f"**{label_kind} — {field}:**"
|
|
# Emit as text block; the renderer's _safe_fence wraps the raw value.
|
|
# We use a "labeled fenced block" pattern: header line + raw content
|
|
# joined inside one text block, where the renderer will leave it alone.
|
|
# To get the safe-fence wrap we use a code block (which calls _safe_fence
|
|
# internally and renders without language-hint corruption risk).
|
|
blocks.append(make_text_block(label))
|
|
code_block = make_code_block(rendered, language="")
|
|
if code_block:
|
|
blocks.append(code_block)
|
|
|
|
# Catch unknown non-null direct fields (no-silent-drop rule).
|
|
structural_keys = {"content_type", "parts"}
|
|
for key, value in content_obj.items():
|
|
if key in structural_keys or key in known_fields:
|
|
continue
|
|
if value is None:
|
|
continue
|
|
# Reject null/empty containers.
|
|
if isinstance(value, (str, list, dict)) and not value:
|
|
continue
|
|
logger.warning(
|
|
"[chatgpt] Unknown non-null field %r in %s message %s/%s",
|
|
key,
|
|
content_type,
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
report.record_unknown(f"{content_type}.{key}")
|
|
blocks.append(
|
|
make_unknown_block(
|
|
raw_type=f"{content_type}.{key}",
|
|
observed_keys=list(content_obj.keys()),
|
|
reason=UNKNOWN_REASON_UNKNOWN_FIELD_IN_KNOWN_TYPE,
|
|
summary=f"unknown non-null field '{key}' in {content_type}",
|
|
)
|
|
)
|
|
|
|
return blocks
|
|
|
|
|
|
def _extract_execution_output_blocks(
|
|
content_obj: dict,
|
|
author_name: str | None,
|
|
msg_metadata: dict,
|
|
conv_id: str,
|
|
node_id: str,
|
|
) -> list[dict]:
|
|
"""Map a ChatGPT ``execution_output`` content (Code Interpreter / container.exec
|
|
/ python tool output) onto a ``tool_result`` block.
|
|
|
|
Locked shape (captured live during planning v0.4.1):
|
|
content.text → output
|
|
author.name → tool_name
|
|
metadata.aggregate_result.status → "error" → is_error=True
|
|
metadata.reasoning_title → summary
|
|
|
|
Empty ``content.text`` → skip (DEBUG log) — a tool that emits no output is
|
|
a transient artifact, not archival content.
|
|
"""
|
|
text = content_obj.get("text") or ""
|
|
if not text.strip():
|
|
logger.debug(
|
|
"[chatgpt] Skipping empty execution_output in conversation %s message %s",
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
return []
|
|
|
|
aggregate = msg_metadata.get("aggregate_result") or {}
|
|
status = aggregate.get("status") if isinstance(aggregate, dict) else None
|
|
is_error = isinstance(status, str) and status.lower() == "error"
|
|
summary = msg_metadata.get("reasoning_title") or None
|
|
|
|
return [
|
|
make_tool_result_block(
|
|
output=text,
|
|
tool_name=author_name,
|
|
is_error=is_error,
|
|
summary=summary,
|
|
)
|
|
]
|
|
|
|
|
|
def _extract_system_error_blocks(
|
|
content_obj: dict,
|
|
author_name: str | None,
|
|
) -> list[dict]:
|
|
"""Map a ChatGPT ``system_error`` content onto an error ``tool_result`` block.
|
|
|
|
Captured shape: ``{content_type, name, text}`` where ``text`` is the error
|
|
message (e.g. ``"Error: Error from browse service: 503"``). ``author.name``
|
|
identifies the failing tool (e.g. ``"web"``).
|
|
"""
|
|
text = content_obj.get("text") or ""
|
|
if not text:
|
|
text = "(error with no message)"
|
|
return [
|
|
make_tool_result_block(
|
|
output=text,
|
|
tool_name=author_name,
|
|
is_error=True,
|
|
)
|
|
]
|
|
|
|
|
|
def _extract_tether_browsing_display_blocks(
|
|
content_obj: dict,
|
|
author_name: str | None,
|
|
conv_id: str,
|
|
node_id: str,
|
|
) -> list[dict]:
|
|
"""Handle ChatGPT's ``tether_browsing_display`` content.
|
|
|
|
Captured live: most instances are **spinner placeholders** (transient UI
|
|
state — empty fields, ``metadata.command == "spinner"``). The actual
|
|
retrieval content arrives as a sibling/child ``multimodal_text`` message
|
|
that already extracts cleanly via the existing handler.
|
|
|
|
Locked behavior:
|
|
- If ``result`` AND ``summary`` are both empty → skip silently (DEBUG).
|
|
These are spinners; the real content is elsewhere.
|
|
- Otherwise (defensive: never observed populated in real data) → render
|
|
as a ``tool_result`` block carrying ``result`` as output and
|
|
``summary`` as the optional summary line.
|
|
"""
|
|
result = content_obj.get("result") or ""
|
|
summary = content_obj.get("summary") or ""
|
|
|
|
if not result.strip() and not summary.strip():
|
|
logger.debug(
|
|
"[chatgpt] Skipping tether_browsing_display spinner in "
|
|
"conversation %s message %s (empty result/summary)",
|
|
conv_id[:8],
|
|
node_id[:8],
|
|
)
|
|
return []
|
|
|
|
return [
|
|
make_tool_result_block(
|
|
output=result or summary,
|
|
tool_name=author_name,
|
|
is_error=False,
|
|
summary=summary if result and summary else None,
|
|
)
|
|
]
|