feat: v0.6.0 — collapse policy, session limiter, Claude Code provider, prune, browser auth, media downloads

This commit is contained in:
JesseMarkowitz
2026-06-12 18:26:14 -04:00
parent 557994f7d9
commit 9e1a8ab7cb
23 changed files with 3133 additions and 197 deletions
+158 -10
View File
@@ -19,6 +19,7 @@ Response: {"items": [...], "cursor": "<opaque_base64_or_null>"}
Pagination ends when cursor is null or an empty string.
"""
import json
import logging
import os
from typing import Any
@@ -26,10 +27,13 @@ from typing import Any
from curl_cffi import requests as curl_requests
from src.blocks import (
COLLAPSED_KIND_HIDDEN_CONTEXT,
COLLAPSED_KIND_TOOL_DUMP,
UNKNOWN_REASON_EXTRACTION_FAILED,
UNKNOWN_REASON_UNKNOWN_FIELD_IN_KNOWN_TYPE,
UNKNOWN_REASON_UNKNOWN_TYPE,
make_code_block,
make_collapsed_block,
make_file_placeholder,
make_hidden_context_marker,
make_image_placeholder,
@@ -39,7 +43,16 @@ from src.blocks import (
make_unknown_block,
)
from src.loss_report import LossReport
from src.providers.base import BaseProvider, ProviderError, REQUEST_TIMEOUT
from src.providers.base import (
BaseProvider,
HIDDEN_CONTENT_FULL,
HIDDEN_CONTENT_OMIT,
HIDDEN_CONTENT_PLACEHOLDER,
ProviderError,
REQUEST_TIMEOUT,
VALID_HIDDEN_CONTENT_POLICIES,
resolve_hidden_content_policy,
)
logger = logging.getLogger(__name__)
@@ -50,6 +63,34 @@ AUTH_SESSION_URL = "https://chatgpt.com/api/auth/session"
# Run: python -c "from curl_cffi.requests import BrowserType; print(list(BrowserType))"
IMPERSONATE = "chrome120"
def parse_asset_file_id(ref: str) -> str | None:
"""Extract the file ID from a ChatGPT asset pointer.
Observed forms (2026-06-12):
sediment://file_00000000245c71fda50034d7f1647791 → file_…
sediment://8456107fc383a53#file_…979c…#p_6.png (generated) → file_…
file-service://file-AbCdEf → file-AbCdEf
"""
if not isinstance(ref, str) or not ref:
return None
if ref.startswith("file-service://"):
return ref.removeprefix("file-service://") or None
if ref.startswith("sediment://"):
body = ref.removeprefix("sediment://")
for part in body.split("#"):
if part.startswith("file_") or part.startswith("file-"):
return part
return body or None
return None
# Tool-role authors whose messages are retrieval dumps — the full text of
# attached/project files re-injected by ChatGPT on every tool run. Measured
# 2026-06-12 across this user's three largest conversations: 86% of all
# content bytes, not flagged is_visually_hidden_from_conversation.
# myfiles_browser is the legacy name for the same retrieval tool.
_COLLAPSE_TOOL_AUTHORS = {"file_search", "myfiles_browser"}
class ChatGPTProvider(BaseProvider):
"""Provider for ChatGPT conversations via the internal web API.
@@ -72,6 +113,7 @@ class ChatGPTProvider(BaseProvider):
session_token: str | None = None,
session_token_1: str | None = None,
project_ids: list[str] | None = None,
hidden_content: str | None = None,
) -> None:
# Pass a curl_cffi session to the base class instead of a requests.Session.
# curl_cffi.requests.Session is API-compatible with requests.Session.
@@ -112,6 +154,14 @@ class ChatGPTProvider(BaseProvider):
# Cache of project_id → display name (avoids re-fetching gizmo details)
self._project_name_cache: dict[str, str] = {}
# Policy for messages invisible in the web UI (retrieval dumps,
# hidden-flagged context): full | placeholder | omit.
self._hidden_content = (
hidden_content
if hidden_content in VALID_HIDDEN_CONTENT_POLICIES
else resolve_hidden_content_policy()
)
# ChatGPT now splits large session cookies into .0 / .1 chunks.
# Always send both named chunks; the server reassembles them.
self._session.cookies.set(
@@ -561,6 +611,59 @@ class ChatGPTProvider(BaseProvider):
)
return data
# ------------------------------------------------------------------
# Asset downloads
# ------------------------------------------------------------------
def download_asset(self, ref: str) -> tuple[bytes, str | None, str | None]:
"""Download a sediment:// / file-service:// asset.
Two hops (verified live 2026-06-12): the files endpoint returns a
signed ``download_url``; fetching that yields the bytes. Expired
assets (e.g. old generated images) 404 on the first hop.
Returns:
(content_bytes, mime_type_or_None, file_name_or_None)
Raises:
ProviderError: unparseable ref, expired/missing asset, or
download failure.
"""
file_id = parse_asset_file_id(ref)
if not file_id:
raise ProviderError(
self.provider_name,
f"download_asset({ref[:40]})",
ValueError(f"Unrecognised asset reference: {ref[:80]}"),
)
meta = self._make_request("GET", f"{BASE_URL}/files/{file_id}/download")
download_url = meta.get("download_url")
if not download_url:
raise ProviderError(
self.provider_name,
f"download_asset({file_id})",
RuntimeError(f"No download_url in response: {meta.get('detail') or meta}"),
)
self._pace()
resp = self._session.request("GET", download_url, timeout=REQUEST_TIMEOUT)
if resp.status_code != 200:
raise ProviderError(
self.provider_name,
f"download_asset({file_id})",
RuntimeError(f"Signed URL returned HTTP {resp.status_code}"),
)
mime = resp.headers.get("content-type") or None
file_name = meta.get("file_name") or None
logger.debug(
"[chatgpt] Downloaded asset %s: %d bytes (%s)",
file_id,
len(resp.content),
mime,
)
return resp.content, mime, file_name
# ------------------------------------------------------------------
# Normalization
# ------------------------------------------------------------------
@@ -598,7 +701,9 @@ class ChatGPTProvider(BaseProvider):
)
mapping: dict = raw.get("mapping", {})
messages = _extract_messages(mapping, raw, conv_id, report)
# getattr fallback: tests construct providers via __new__, skipping __init__.
policy = getattr(self, "_hidden_content", None) or resolve_hidden_content_policy()
messages = _extract_messages(mapping, raw, conv_id, report, policy)
for _ in messages:
report.record_message()
report.record_conversation()
@@ -631,7 +736,11 @@ def _ts_to_iso(ts: float | int | str | None) -> str:
def _extract_messages(
mapping: dict[str, Any], raw: dict, conv_id: str, report: LossReport
mapping: dict[str, Any],
raw: dict,
conv_id: str,
report: LossReport,
policy: str = HIDDEN_CONTENT_FULL,
) -> list[dict]:
"""Walk the ChatGPT conversation mapping tree to produce an ordered message list.
@@ -661,7 +770,7 @@ def _extract_messages(
node = mapping.get(node_id, {})
msg_data = node.get("message")
if msg_data:
built = _build_message(msg_data, conv_id, node_id, report)
built = _build_message(msg_data, conv_id, node_id, report, policy)
if built is not None:
messages.append(built)
@@ -688,12 +797,17 @@ def _find_root(mapping: dict[str, Any]) -> str | None:
def _build_message(
msg_data: dict, conv_id: str, node_id: str, report: LossReport
msg_data: dict,
conv_id: str,
node_id: str,
report: LossReport,
policy: str = HIDDEN_CONTENT_FULL,
) -> dict | None:
"""Construct a normalized message dict (with ``blocks``) for one ChatGPT node.
Returns None for messages that should be skipped (truly empty). Otherwise
returns a dict with ``role``, ``content_type``, ``timestamp``, ``blocks``.
Returns None for messages that should be skipped (truly empty, or omitted
by the EXPORTER_HIDDEN_CONTENT policy). Otherwise returns a dict with
``role``, ``content_type``, ``timestamp``, ``blocks``.
"""
author = msg_data.get("author") or {}
role = author.get("role", "") or ""
@@ -727,9 +841,43 @@ def _build_message(
)
return None
if is_hidden:
# Prepend a marker so the reader knows this message is hidden in the
# source UI. The marker is content-type-agnostic.
# Messages invisible in the web UI: tool retrieval dumps (identified by
# author.name — they are NOT hidden-flagged) and hidden-flagged context.
collapse_kind: str | None = None
if role == "tool" and author_name in _COLLAPSE_TOOL_AUTHORS:
collapse_kind = COLLAPSED_KIND_TOOL_DUMP
elif is_hidden:
collapse_kind = COLLAPSED_KIND_HIDDEN_CONTEXT
if collapse_kind is not None and policy != HIDDEN_CONTENT_FULL:
origin = (
author_name if collapse_kind == COLLAPSED_KIND_TOOL_DUMP else content_type
) or "?"
size_bytes = len(json.dumps(content_obj, ensure_ascii=False, default=str))
report.record_collapsed(origin, size_bytes)
logger.debug(
"[chatgpt] %s %s message (%s, %dB) in conversation %s "
"per EXPORTER_HIDDEN_CONTENT=%s",
"Omitted" if policy == HIDDEN_CONTENT_OMIT else "Collapsed",
collapse_kind,
origin,
size_bytes,
conv_id[:8],
policy,
)
if policy == HIDDEN_CONTENT_OMIT:
return None
blocks = [
make_collapsed_block(
origin=origin,
content_type=content_type,
size_bytes=size_bytes,
kind=collapse_kind,
)
]
elif is_hidden:
# Policy 'full': prepend a marker so the reader knows this message is
# hidden in the source UI. The marker is content-type-agnostic.
blocks = [make_hidden_context_marker(content_type)] + blocks
# Vestigial content_type: "code" for code-only messages, otherwise "text"