feat: v0.6.0 — collapse policy, session limiter, Claude Code provider, prune, browser auth, media downloads
This commit is contained in:
+158
-10
@@ -19,6 +19,7 @@ Response: {"items": [...], "cursor": "<opaque_base64_or_null>"}
|
||||
Pagination ends when cursor is null or an empty string.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from typing import Any
|
||||
@@ -26,10 +27,13 @@ from typing import Any
|
||||
from curl_cffi import requests as curl_requests
|
||||
|
||||
from src.blocks import (
|
||||
COLLAPSED_KIND_HIDDEN_CONTEXT,
|
||||
COLLAPSED_KIND_TOOL_DUMP,
|
||||
UNKNOWN_REASON_EXTRACTION_FAILED,
|
||||
UNKNOWN_REASON_UNKNOWN_FIELD_IN_KNOWN_TYPE,
|
||||
UNKNOWN_REASON_UNKNOWN_TYPE,
|
||||
make_code_block,
|
||||
make_collapsed_block,
|
||||
make_file_placeholder,
|
||||
make_hidden_context_marker,
|
||||
make_image_placeholder,
|
||||
@@ -39,7 +43,16 @@ from src.blocks import (
|
||||
make_unknown_block,
|
||||
)
|
||||
from src.loss_report import LossReport
|
||||
from src.providers.base import BaseProvider, ProviderError, REQUEST_TIMEOUT
|
||||
from src.providers.base import (
|
||||
BaseProvider,
|
||||
HIDDEN_CONTENT_FULL,
|
||||
HIDDEN_CONTENT_OMIT,
|
||||
HIDDEN_CONTENT_PLACEHOLDER,
|
||||
ProviderError,
|
||||
REQUEST_TIMEOUT,
|
||||
VALID_HIDDEN_CONTENT_POLICIES,
|
||||
resolve_hidden_content_policy,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -50,6 +63,34 @@ AUTH_SESSION_URL = "https://chatgpt.com/api/auth/session"
|
||||
# Run: python -c "from curl_cffi.requests import BrowserType; print(list(BrowserType))"
|
||||
IMPERSONATE = "chrome120"
|
||||
|
||||
def parse_asset_file_id(ref: str) -> str | None:
|
||||
"""Extract the file ID from a ChatGPT asset pointer.
|
||||
|
||||
Observed forms (2026-06-12):
|
||||
sediment://file_00000000245c71fda50034d7f1647791 → file_…
|
||||
sediment://8456107fc383a53#file_…979c…#p_6.png (generated) → file_…
|
||||
file-service://file-AbCdEf → file-AbCdEf
|
||||
"""
|
||||
if not isinstance(ref, str) or not ref:
|
||||
return None
|
||||
if ref.startswith("file-service://"):
|
||||
return ref.removeprefix("file-service://") or None
|
||||
if ref.startswith("sediment://"):
|
||||
body = ref.removeprefix("sediment://")
|
||||
for part in body.split("#"):
|
||||
if part.startswith("file_") or part.startswith("file-"):
|
||||
return part
|
||||
return body or None
|
||||
return None
|
||||
|
||||
|
||||
# Tool-role authors whose messages are retrieval dumps — the full text of
|
||||
# attached/project files re-injected by ChatGPT on every tool run. Measured
|
||||
# 2026-06-12 across this user's three largest conversations: 86% of all
|
||||
# content bytes, not flagged is_visually_hidden_from_conversation.
|
||||
# myfiles_browser is the legacy name for the same retrieval tool.
|
||||
_COLLAPSE_TOOL_AUTHORS = {"file_search", "myfiles_browser"}
|
||||
|
||||
|
||||
class ChatGPTProvider(BaseProvider):
|
||||
"""Provider for ChatGPT conversations via the internal web API.
|
||||
@@ -72,6 +113,7 @@ class ChatGPTProvider(BaseProvider):
|
||||
session_token: str | None = None,
|
||||
session_token_1: str | None = None,
|
||||
project_ids: list[str] | None = None,
|
||||
hidden_content: str | None = None,
|
||||
) -> None:
|
||||
# Pass a curl_cffi session to the base class instead of a requests.Session.
|
||||
# curl_cffi.requests.Session is API-compatible with requests.Session.
|
||||
@@ -112,6 +154,14 @@ class ChatGPTProvider(BaseProvider):
|
||||
# Cache of project_id → display name (avoids re-fetching gizmo details)
|
||||
self._project_name_cache: dict[str, str] = {}
|
||||
|
||||
# Policy for messages invisible in the web UI (retrieval dumps,
|
||||
# hidden-flagged context): full | placeholder | omit.
|
||||
self._hidden_content = (
|
||||
hidden_content
|
||||
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
|
||||
else resolve_hidden_content_policy()
|
||||
)
|
||||
|
||||
# ChatGPT now splits large session cookies into .0 / .1 chunks.
|
||||
# Always send both named chunks; the server reassembles them.
|
||||
self._session.cookies.set(
|
||||
@@ -561,6 +611,59 @@ class ChatGPTProvider(BaseProvider):
|
||||
)
|
||||
return data
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Asset downloads
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def download_asset(self, ref: str) -> tuple[bytes, str | None, str | None]:
|
||||
"""Download a sediment:// / file-service:// asset.
|
||||
|
||||
Two hops (verified live 2026-06-12): the files endpoint returns a
|
||||
signed ``download_url``; fetching that yields the bytes. Expired
|
||||
assets (e.g. old generated images) 404 on the first hop.
|
||||
|
||||
Returns:
|
||||
(content_bytes, mime_type_or_None, file_name_or_None)
|
||||
|
||||
Raises:
|
||||
ProviderError: unparseable ref, expired/missing asset, or
|
||||
download failure.
|
||||
"""
|
||||
file_id = parse_asset_file_id(ref)
|
||||
if not file_id:
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"download_asset({ref[:40]})",
|
||||
ValueError(f"Unrecognised asset reference: {ref[:80]}"),
|
||||
)
|
||||
|
||||
meta = self._make_request("GET", f"{BASE_URL}/files/{file_id}/download")
|
||||
download_url = meta.get("download_url")
|
||||
if not download_url:
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"download_asset({file_id})",
|
||||
RuntimeError(f"No download_url in response: {meta.get('detail') or meta}"),
|
||||
)
|
||||
|
||||
self._pace()
|
||||
resp = self._session.request("GET", download_url, timeout=REQUEST_TIMEOUT)
|
||||
if resp.status_code != 200:
|
||||
raise ProviderError(
|
||||
self.provider_name,
|
||||
f"download_asset({file_id})",
|
||||
RuntimeError(f"Signed URL returned HTTP {resp.status_code}"),
|
||||
)
|
||||
mime = resp.headers.get("content-type") or None
|
||||
file_name = meta.get("file_name") or None
|
||||
logger.debug(
|
||||
"[chatgpt] Downloaded asset %s: %d bytes (%s)",
|
||||
file_id,
|
||||
len(resp.content),
|
||||
mime,
|
||||
)
|
||||
return resp.content, mime, file_name
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Normalization
|
||||
# ------------------------------------------------------------------
|
||||
@@ -598,7 +701,9 @@ class ChatGPTProvider(BaseProvider):
|
||||
)
|
||||
|
||||
mapping: dict = raw.get("mapping", {})
|
||||
messages = _extract_messages(mapping, raw, conv_id, report)
|
||||
# getattr fallback: tests construct providers via __new__, skipping __init__.
|
||||
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
|
||||
messages = _extract_messages(mapping, raw, conv_id, report, policy)
|
||||
for _ in messages:
|
||||
report.record_message()
|
||||
report.record_conversation()
|
||||
@@ -631,7 +736,11 @@ def _ts_to_iso(ts: float | int | str | None) -> str:
|
||||
|
||||
|
||||
def _extract_messages(
|
||||
mapping: dict[str, Any], raw: dict, conv_id: str, report: LossReport
|
||||
mapping: dict[str, Any],
|
||||
raw: dict,
|
||||
conv_id: str,
|
||||
report: LossReport,
|
||||
policy: str = HIDDEN_CONTENT_FULL,
|
||||
) -> list[dict]:
|
||||
"""Walk the ChatGPT conversation mapping tree to produce an ordered message list.
|
||||
|
||||
@@ -661,7 +770,7 @@ def _extract_messages(
|
||||
node = mapping.get(node_id, {})
|
||||
msg_data = node.get("message")
|
||||
if msg_data:
|
||||
built = _build_message(msg_data, conv_id, node_id, report)
|
||||
built = _build_message(msg_data, conv_id, node_id, report, policy)
|
||||
if built is not None:
|
||||
messages.append(built)
|
||||
|
||||
@@ -688,12 +797,17 @@ def _find_root(mapping: dict[str, Any]) -> str | None:
|
||||
|
||||
|
||||
def _build_message(
|
||||
msg_data: dict, conv_id: str, node_id: str, report: LossReport
|
||||
msg_data: dict,
|
||||
conv_id: str,
|
||||
node_id: str,
|
||||
report: LossReport,
|
||||
policy: str = HIDDEN_CONTENT_FULL,
|
||||
) -> dict | None:
|
||||
"""Construct a normalized message dict (with ``blocks``) for one ChatGPT node.
|
||||
|
||||
Returns None for messages that should be skipped (truly empty). Otherwise
|
||||
returns a dict with ``role``, ``content_type``, ``timestamp``, ``blocks``.
|
||||
Returns None for messages that should be skipped (truly empty, or omitted
|
||||
by the EXPORTER_HIDDEN_CONTENT policy). Otherwise returns a dict with
|
||||
``role``, ``content_type``, ``timestamp``, ``blocks``.
|
||||
"""
|
||||
author = msg_data.get("author") or {}
|
||||
role = author.get("role", "") or ""
|
||||
@@ -727,9 +841,43 @@ def _build_message(
|
||||
)
|
||||
return None
|
||||
|
||||
if is_hidden:
|
||||
# Prepend a marker so the reader knows this message is hidden in the
|
||||
# source UI. The marker is content-type-agnostic.
|
||||
# Messages invisible in the web UI: tool retrieval dumps (identified by
|
||||
# author.name — they are NOT hidden-flagged) and hidden-flagged context.
|
||||
collapse_kind: str | None = None
|
||||
if role == "tool" and author_name in _COLLAPSE_TOOL_AUTHORS:
|
||||
collapse_kind = COLLAPSED_KIND_TOOL_DUMP
|
||||
elif is_hidden:
|
||||
collapse_kind = COLLAPSED_KIND_HIDDEN_CONTEXT
|
||||
|
||||
if collapse_kind is not None and policy != HIDDEN_CONTENT_FULL:
|
||||
origin = (
|
||||
author_name if collapse_kind == COLLAPSED_KIND_TOOL_DUMP else content_type
|
||||
) or "?"
|
||||
size_bytes = len(json.dumps(content_obj, ensure_ascii=False, default=str))
|
||||
report.record_collapsed(origin, size_bytes)
|
||||
logger.debug(
|
||||
"[chatgpt] %s %s message (%s, %dB) in conversation %s "
|
||||
"per EXPORTER_HIDDEN_CONTENT=%s",
|
||||
"Omitted" if policy == HIDDEN_CONTENT_OMIT else "Collapsed",
|
||||
collapse_kind,
|
||||
origin,
|
||||
size_bytes,
|
||||
conv_id[:8],
|
||||
policy,
|
||||
)
|
||||
if policy == HIDDEN_CONTENT_OMIT:
|
||||
return None
|
||||
blocks = [
|
||||
make_collapsed_block(
|
||||
origin=origin,
|
||||
content_type=content_type,
|
||||
size_bytes=size_bytes,
|
||||
kind=collapse_kind,
|
||||
)
|
||||
]
|
||||
elif is_hidden:
|
||||
# Policy 'full': prepend a marker so the reader knows this message is
|
||||
# hidden in the source UI. The marker is content-type-agnostic.
|
||||
blocks = [make_hidden_context_marker(content_type)] + blocks
|
||||
|
||||
# Vestigial content_type: "code" for code-only messages, otherwise "text"
|
||||
|
||||
Reference in New Issue
Block a user