"""Can the exporter reuse an image URL copied from the browser? The endpoint hunt bottomed out: the Library id is rejected outright (file_not_found from GetDownloadLinkError), and /backend-api/content wants signed query params (id, ts, p, and in practice a signature) that cannot be guessed. The web UI has those values, so the remaining question is whether a URL taken from the UI works outside the browser. Paste the request URL DevTools shows for one of the unreachable images: python tools/try_pasted_url.py "https://chatgpt.com/backend-api/content?id=…&ts=…&p=…&sig=…" It fetches that URL three ways, and the pattern of results says what to build: works with our session, not bare → the signature is fine but the request needs auth; the exporter can mint these itself if we find what returns them. works bare too → the URL is self-authenticating; whatever produced it is the endpoint we need. works in neither → the URL is bound to the browser session (or already expired — check ts), so the exporter cannot reuse it as-is. Nothing is written to disk unless --save is passed. """ import sys from pathlib import Path from urllib.parse import parse_qs, urlparse sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() def describe(url: str) -> None: parsed = urlparse(url) print(f" host : {parsed.netloc}") print(f" path : {parsed.path}") params = parse_qs(parsed.query) print(" query :") for key, values in params.items(): value = values[0] if values else "" shown = value if len(value) <= 48 else f"{value[:45]}…" print(f" {key:<12} {shown}") def main() -> None: args = [a for a in sys.argv[1:] if a != "--save"] save = "--save" in sys.argv if not args: print(__doc__) return url = args[0] print("=" * 78) print("The URL") print("=" * 78) describe(url) print() print("=" * 78) print("Fetch attempts") print("=" * 78) def report(label: str, resp) -> bytes | None: ctype = (resp.headers.get("content-type") or "").split(";")[0] size = len(resp.content or b"") verdict = "" if resp.status_code == 200 and ctype.startswith("image/"): verdict = " ← IMAGE BYTES" elif resp.status_code == 200: verdict = f" 200 but {ctype or 'unknown type'}" print(f" {label:<34} {resp.status_code} {ctype} {size}B{verdict}") if resp.status_code == 200 and ctype.startswith("image/"): return resp.content if resp.status_code != 200: preview = (resp.text or "")[:160].replace("\n", " ") if preview: print(f" {preview}") return None image: bytes | None = None # 1. Bare request, no cookies, no auth headers. try: from curl_cffi import requests as curl_requests bare = curl_requests.Session(impersonate="chrome120") image = report("bare (no auth)", bare.get(url, timeout=30)) or image except Exception as e: # noqa: BLE001 - diagnostic print(f" {'bare (no auth)':<34} error {type(e).__name__}: {e}") # 2. Through the exporter's authenticated session. try: from src.providers.chatgpt import ChatGPTProvider provider = ChatGPTProvider() provider._pace() image = report( "exporter session", provider._session.request("GET", url, timeout=30) ) or image except Exception as e: # noqa: BLE001 - diagnostic print(f" {'exporter session':<34} error {type(e).__name__}: {e}") provider = None # 3. Authenticated session without the Authorization header, in case the # signature and the bearer token conflict. if provider is not None: try: saved = provider._session.headers.pop("Authorization", None) provider._pace() image = report( "session, no Authorization", provider._session.request("GET", url, timeout=30), ) or image if saved: provider._session.headers["Authorization"] = saved except Exception as e: # noqa: BLE001 - diagnostic print(f" {'session, no Authorization':<34} error {type(e).__name__}") print() print("=" * 78) print("Verdict") print("=" * 78) if image: print(f" Served {len(image)} bytes of image data.") print(" → The exporter CAN fetch these. Next: find what mints the") print(" signed params, so it can build the URL itself.") if save: out = Path("pasted_url_result.bin") out.write_bytes(image) print(f" Saved to {out}") else: print(" (pass --save to write the bytes out)") else: print(" No attempt returned image bytes.") print(" → Either the URL is bound to the browser session, or its ts has") print(" expired. Re-copy a fresh URL and retry once; if it still") print(" fails, these images are not reachable programmatically.") if __name__ == "__main__": main()