From 024bfde0306a0c1110a23b338278af175c51a15c Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 10:27:57 -0400 Subject: [PATCH] tools: test whether a browser image URL works from the exporter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The endpoint hunt is done guessing. Results: /files/{libfile}/download 200 {"error_code":"file_not_found", "error_type":"GetDownloadLinkError"} — twice, on two different Library ids, so the Library id is simply not valid there /library/* 404 across every shape POST on the download route 405 /content?asset_pointer= 422 requiring query params id, ts, p That last one is the find: /backend-api/content is a signed route, and the web UI has the values we cannot compute. So the question shifts from "which endpoint" to "is the UI's URL reusable outside the browser". try_pasted_url.py takes a URL copied from DevTools and fetches it three ways — bare, through the exporter's authenticated session, and with the Authorization header removed in case bearer and signature conflict. The pattern says whether the exporter can fetch these at all, and therefore whether it is worth hunting for what mints the signature. Before any of that: check whether the images still render in ChatGPT's own UI. If they show broken there, file_not_found is literally true, these 11 are lost like the other 7, and no route exists to find. --- tools/try_pasted_url.py | 146 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 146 insertions(+) create mode 100644 tools/try_pasted_url.py diff --git a/tools/try_pasted_url.py b/tools/try_pasted_url.py new file mode 100644 index 0000000..0afc203 --- /dev/null +++ b/tools/try_pasted_url.py @@ -0,0 +1,146 @@ +"""Can the exporter reuse an image URL copied from the browser? + +The endpoint hunt bottomed out: the Library id is rejected outright +(file_not_found from GetDownloadLinkError), and /backend-api/content wants +signed query params (id, ts, p, and in practice a signature) that cannot be +guessed. The web UI has those values, so the remaining question is whether a +URL taken from the UI works outside the browser. + +Paste the request URL DevTools shows for one of the unreachable images: + + python tools/try_pasted_url.py "https://chatgpt.com/backend-api/content?id=…&ts=…&p=…&sig=…" + +It fetches that URL three ways, and the pattern of results says what to build: + + works with our session, not bare → the signature is fine but the request + needs auth; the exporter can mint these + itself if we find what returns them. + works bare too → the URL is self-authenticating; whatever + produced it is the endpoint we need. + works in neither → the URL is bound to the browser session + (or already expired — check ts), so the + exporter cannot reuse it as-is. + +Nothing is written to disk unless --save is passed. +""" + +import sys +from pathlib import Path +from urllib.parse import parse_qs, urlparse + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + + +def describe(url: str) -> None: + parsed = urlparse(url) + print(f" host : {parsed.netloc}") + print(f" path : {parsed.path}") + params = parse_qs(parsed.query) + print(" query :") + for key, values in params.items(): + value = values[0] if values else "" + shown = value if len(value) <= 48 else f"{value[:45]}…" + print(f" {key:<12} {shown}") + + +def main() -> None: + args = [a for a in sys.argv[1:] if a != "--save"] + save = "--save" in sys.argv + if not args: + print(__doc__) + return + url = args[0] + + print("=" * 78) + print("The URL") + print("=" * 78) + describe(url) + + print() + print("=" * 78) + print("Fetch attempts") + print("=" * 78) + + def report(label: str, resp) -> bytes | None: + ctype = (resp.headers.get("content-type") or "").split(";")[0] + size = len(resp.content or b"") + verdict = "" + if resp.status_code == 200 and ctype.startswith("image/"): + verdict = " ← IMAGE BYTES" + elif resp.status_code == 200: + verdict = f" 200 but {ctype or 'unknown type'}" + print(f" {label:<34} {resp.status_code} {ctype} {size}B{verdict}") + if resp.status_code == 200 and ctype.startswith("image/"): + return resp.content + if resp.status_code != 200: + preview = (resp.text or "")[:160].replace("\n", " ") + if preview: + print(f" {preview}") + return None + + image: bytes | None = None + + # 1. Bare request, no cookies, no auth headers. + try: + from curl_cffi import requests as curl_requests + + bare = curl_requests.Session(impersonate="chrome120") + image = report("bare (no auth)", bare.get(url, timeout=30)) or image + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {'bare (no auth)':<34} error {type(e).__name__}: {e}") + + # 2. Through the exporter's authenticated session. + try: + from src.providers.chatgpt import ChatGPTProvider + + provider = ChatGPTProvider() + provider._pace() + image = report( + "exporter session", provider._session.request("GET", url, timeout=30) + ) or image + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {'exporter session':<34} error {type(e).__name__}: {e}") + provider = None + + # 3. Authenticated session without the Authorization header, in case the + # signature and the bearer token conflict. + if provider is not None: + try: + saved = provider._session.headers.pop("Authorization", None) + provider._pace() + image = report( + "session, no Authorization", + provider._session.request("GET", url, timeout=30), + ) or image + if saved: + provider._session.headers["Authorization"] = saved + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {'session, no Authorization':<34} error {type(e).__name__}") + + print() + print("=" * 78) + print("Verdict") + print("=" * 78) + if image: + print(f" Served {len(image)} bytes of image data.") + print(" → The exporter CAN fetch these. Next: find what mints the") + print(" signed params, so it can build the URL itself.") + if save: + out = Path("pasted_url_result.bin") + out.write_bytes(image) + print(f" Saved to {out}") + else: + print(" (pass --save to write the bytes out)") + else: + print(" No attempt returned image bytes.") + print(" → Either the URL is bound to the browser session, or its ts has") + print(" expired. Re-copy a fresh URL and retry once; if it still") + print(" fails, these images are not reachable programmatically.") + + +if __name__ == "__main__": + main()