From d30a9510bb3c91fdd7a6d722d541b039b2b86c35 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 11:07:18 -0400 Subject: [PATCH] tools: find what the browser sends that we don't MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The decisive fact arrived: the images DO render in ChatGPT. So a valid signature exists for files that hand us 403, and there is a route to find. The rest is now pinned down. /files/{id}/download is the minting endpoint — a working file's download_url is the same estuary/content URL the browser uses. Signatures are per-file: swapping an id into a working URL returns "Invalid signature or expired URL", and a ts without a sig gets the same. So the difference is in the request, not the route. Our call sends Authorization: Bearer plus cookies. A browser leans on cookies and adds client headers a gated endpoint may check. This walks those variants — no Authorization, conversation Referer, oai-device-id, oai-language, sec-fetch-*, an image Accept, a conversation_id param — against a file that is currently refused, and reports which mints a URL. If none do, it prints the DevTools recipe for finding the minting call by searching the Network panel for the file id. --- tools/mint_headers_probe.py | 192 ++++++++++++++++++++++++++++++++++++ 1 file changed, 192 insertions(+) create mode 100644 tools/mint_headers_probe.py diff --git a/tools/mint_headers_probe.py b/tools/mint_headers_probe.py new file mode 100644 index 0000000..1ac63d2 --- /dev/null +++ b/tools/mint_headers_probe.py @@ -0,0 +1,192 @@ +"""The browser mints signatures for files we get 403 on. What does it send that we don't? + +Established: + * The images DO render in ChatGPT, so a valid signature exists for them. + * /files/{id}/download is the minting endpoint — a working file's + download_url is the same estuary/content URL the browser uses. + * Signatures are per-file: swapping an id into a working URL gives + "Invalid signature or expired URL". + +So the difference is in the request, not the route. Our call sends an +Authorization: Bearer header plus session cookies; a browser leans on cookies +and adds client headers (oai-device-id, oai-client-version, a conversation +Referer) that a gated endpoint may check. This walks those variants against a +file that is currently refused, and prints which combination mints a URL. + +Any 200 carrying a download_url is the answer, and the exporter adopts those +headers. + +Run from the project root: + + python tools/mint_headers_probe.py +""" + +import json +import os +import re +import sys +import uuid +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + +from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402 + +DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") +CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE) + + +def find_failed(export_dir: Path) -> list[tuple[str, str]]: + """(file_id, conversation_id) for every image that failed to download.""" + out = [] + for md in export_dir.rglob("*.md"): + try: + text = md.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + conv = CONV_ID_RE.search(text) + if not conv: + continue + for ref, _meta in DEAD_RE.findall(text): + fid = parse_asset_file_id(ref) + if fid: + out.append((fid, conv.group(1))) + return out + + +def main() -> None: + export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() + failed = find_failed(export_dir) + if not failed: + print(f"no failed images found under {export_dir}") + return + + provider = ChatGPTProvider() + session = provider._session + + # Pick one that still exists — a deleted file 403s no matter what we send. + target = conv_id = None + for fid, cid in failed: + provider._pace() + meta = session.request("GET", f"{BASE_URL}/files/{fid}", timeout=30) + if meta.status_code == 200: + target, conv_id = fid, cid + break + if not target: + print("every failure is a deleted file — nothing to test") + return + + url = f"{BASE_URL}/files/{target}/download" + print("=" * 78) + print(f"Minting attempts for {target}") + print(f" conversation {conv_id}") + print("=" * 78) + + def attempt( + label: str, + *, + add: dict | None = None, + drop: tuple[str, ...] = (), + target_url: str | None = None, + ) -> bool: + removed = {} + for header in drop: + for key in list(session.headers.keys()): + if key.lower() == header.lower(): + removed[key] = session.headers.pop(key) + try: + provider._pace() + resp = session.request( + "GET", target_url or url, headers=add or None, timeout=30 + ) + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {label:<46} error {type(e).__name__}") + return False + finally: + session.headers.update(removed) + + try: + body = resp.json() + except Exception: # noqa: BLE001 - diagnostic + body = {} + if resp.status_code == 200 and isinstance(body, dict) and body.get("download_url"): + print(f" {label:<46} 200 ← MINTED") + print(f" {str(body['download_url'])[:110]}") + return True + detail = body.get("detail") if isinstance(body, dict) else None + print(f" {label:<46} {resp.status_code} {json.dumps(detail, default=str)[:60] if detail else ''}") + return False + + device_id = str(uuid.uuid4()) + winners = [] + + checks = [ + ("baseline (bearer + cookies)", {}, ()), + ("no Authorization (cookies only)", {}, ("Authorization",)), + ("Referer: the conversation", {"Referer": f"https://chatgpt.com/c/{conv_id}"}, ()), + ("oai-device-id", {"oai-device-id": device_id}, ()), + ("oai-language", {"oai-language": "en-US"}, ()), + ( + "browser-ish set", + { + "Referer": f"https://chatgpt.com/c/{conv_id}", + "oai-device-id": device_id, + "oai-language": "en-US", + "sec-fetch-site": "same-origin", + "sec-fetch-mode": "cors", + "sec-fetch-dest": "empty", + }, + (), + ), + ( + "browser-ish, no Authorization", + { + "Referer": f"https://chatgpt.com/c/{conv_id}", + "oai-device-id": device_id, + "oai-language": "en-US", + }, + ("Authorization",), + ), + ("Accept: image/*", {"Accept": "image/avif,image/webp,*/*"}, ()), + ] + + for label, add, drop in checks: + if attempt(label, add=add, drop=drop): + winners.append(label) + + # Same call with the conversation named in the query string. + if attempt( + "?conversation_id=", + target_url=f"{url}?conversation_id={conv_id}", + ): + winners.append("?conversation_id=") + + print() + print("=" * 78) + print("Result") + print("=" * 78) + if winners: + print(" Minted by:") + for label in winners: + print(f" {label}") + print() + print(" → Adopt those headers in ChatGPTProvider and the 11 refused") + print(" images become downloadable.") + else: + print(" No header combination minted a URL.") + print() + print(" The browser is getting its signature some other way. Find it:") + print(" 1. Open the conversation in ChatGPT with DevTools → Network.") + print(" 2. Press Ctrl+F (search inside requests) and search for") + print(f" {target[5:22]}") + print(" 3. The hit that is NOT the estuary/content request is the") + print(" call that mints the signature — send me its URL, method,") + print(" and (if POST) its request body.") + + +if __name__ == "__main__": + main()