Files
AIChatExporter/tools/try_pasted_url.py
T
JesseMarkowitz 024bfde030 tools: test whether a browser image URL works from the exporter
The endpoint hunt is done guessing. Results:

  /files/{libfile}/download    200 {"error_code":"file_not_found",
                                    "error_type":"GetDownloadLinkError"}
                               — twice, on two different Library ids, so the
                                 Library id is simply not valid there
  /library/*                   404 across every shape
  POST on the download route   405
  /content?asset_pointer=      422 requiring query params id, ts, p

That last one is the find: /backend-api/content is a signed route, and the
web UI has the values we cannot compute. So the question shifts from "which
endpoint" to "is the UI's URL reusable outside the browser".

try_pasted_url.py takes a URL copied from DevTools and fetches it three
ways — bare, through the exporter's authenticated session, and with the
Authorization header removed in case bearer and signature conflict. The
pattern says whether the exporter can fetch these at all, and therefore
whether it is worth hunting for what mints the signature.

Before any of that: check whether the images still render in ChatGPT's own
UI. If they show broken there, file_not_found is literally true, these 11
are lost like the other 7, and no route exists to find.
2026-08-17 10:27:57 -04:00

147 lines
5.3 KiB
Python

"""Can the exporter reuse an image URL copied from the browser?
The endpoint hunt bottomed out: the Library id is rejected outright
(file_not_found from GetDownloadLinkError), and /backend-api/content wants
signed query params (id, ts, p, and in practice a signature) that cannot be
guessed. The web UI has those values, so the remaining question is whether a
URL taken from the UI works outside the browser.
Paste the request URL DevTools shows for one of the unreachable images:
python tools/try_pasted_url.py "https://chatgpt.com/backend-api/content?id=…&ts=…&p=…&sig=…"
It fetches that URL three ways, and the pattern of results says what to build:
works with our session, not bare → the signature is fine but the request
needs auth; the exporter can mint these
itself if we find what returns them.
works bare too → the URL is self-authenticating; whatever
produced it is the endpoint we need.
works in neither → the URL is bound to the browser session
(or already expired — check ts), so the
exporter cannot reuse it as-is.
Nothing is written to disk unless --save is passed.
"""
import sys
from pathlib import Path
from urllib.parse import parse_qs, urlparse
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from dotenv import load_dotenv
load_dotenv()
def describe(url: str) -> None:
parsed = urlparse(url)
print(f" host : {parsed.netloc}")
print(f" path : {parsed.path}")
params = parse_qs(parsed.query)
print(" query :")
for key, values in params.items():
value = values[0] if values else ""
shown = value if len(value) <= 48 else f"{value[:45]}…"
print(f" {key:<12} {shown}")
def main() -> None:
args = [a for a in sys.argv[1:] if a != "--save"]
save = "--save" in sys.argv
if not args:
print(__doc__)
return
url = args[0]
print("=" * 78)
print("The URL")
print("=" * 78)
describe(url)
print()
print("=" * 78)
print("Fetch attempts")
print("=" * 78)
def report(label: str, resp) -> bytes | None:
ctype = (resp.headers.get("content-type") or "").split(";")[0]
size = len(resp.content or b"")
verdict = ""
if resp.status_code == 200 and ctype.startswith("image/"):
verdict = " ← IMAGE BYTES"
elif resp.status_code == 200:
verdict = f" 200 but {ctype or 'unknown type'}"
print(f" {label:<34} {resp.status_code} {ctype} {size}B{verdict}")
if resp.status_code == 200 and ctype.startswith("image/"):
return resp.content
if resp.status_code != 200:
preview = (resp.text or "")[:160].replace("\n", " ")
if preview:
print(f" {preview}")
return None
image: bytes | None = None
# 1. Bare request, no cookies, no auth headers.
try:
from curl_cffi import requests as curl_requests
bare = curl_requests.Session(impersonate="chrome120")
image = report("bare (no auth)", bare.get(url, timeout=30)) or image
except Exception as e: # noqa: BLE001 - diagnostic
print(f" {'bare (no auth)':<34} error {type(e).__name__}: {e}")
# 2. Through the exporter's authenticated session.
try:
from src.providers.chatgpt import ChatGPTProvider
provider = ChatGPTProvider()
provider._pace()
image = report(
"exporter session", provider._session.request("GET", url, timeout=30)
) or image
except Exception as e: # noqa: BLE001 - diagnostic
print(f" {'exporter session':<34} error {type(e).__name__}: {e}")
provider = None
# 3. Authenticated session without the Authorization header, in case the
# signature and the bearer token conflict.
if provider is not None:
try:
saved = provider._session.headers.pop("Authorization", None)
provider._pace()
image = report(
"session, no Authorization",
provider._session.request("GET", url, timeout=30),
) or image
if saved:
provider._session.headers["Authorization"] = saved
except Exception as e: # noqa: BLE001 - diagnostic
print(f" {'session, no Authorization':<34} error {type(e).__name__}")
print()
print("=" * 78)
print("Verdict")
print("=" * 78)
if image:
print(f" Served {len(image)} bytes of image data.")
print(" → The exporter CAN fetch these. Next: find what mints the")
print(" signed params, so it can build the URL itself.")
if save:
out = Path("pasted_url_result.bin")
out.write_bytes(image)
print(f" Saved to {out}")
else:
print(" (pass --save to write the bytes out)")
else:
print(" No attempt returned image bytes.")
print(" → Either the URL is bound to the browser session, or its ts has")
print(" expired. Re-copy a fresh URL and retry once; if it still")
print(" fails, these images are not reachable programmatically.")
if __name__ == "__main__":
main()