Confirmed by the last run: a working file's download_url is
chatgpt.com/backend-api/estuary/content?cid&id&p&sig&ts&v — the same route
the browser uses. So /files/{id}/download is the minting endpoint, and a
gizmo file's 403 is a refusal to mint the signature. That is why no amount
of scoping helped; we were turned away at the only door that issues them.
Nothing else in the estuary namespace serves files: 404 across the board.
But section B tested file_000000001b3071f5…, one of the seven DELETED
files, because it took the first failure without checking its class. Those
results say nothing about the eleven recoverable ones.
- Pick the target by probing /files/{id} and taking one that answers 200,
skipping (and naming) the deleted ones.
- Add two experiments: estuary/content with a ts but no sig, since
validation asked only for id/p/ts and never sig; and a working file's
freshly minted URL with the refused id swapped in, which shows whether
the signature is bound to the file.
204 lines
8.4 KiB
Python
204 lines
8.4 KiB
Python
"""The web UI serves images from /backend-api/estuary/content — can we mint that?
|
|
|
|
A URL copied from the browser looks like:
|
|
|
|
/backend-api/estuary/content?id=file_…&ts=496382&p=fs&cid=1&sig=…&v=0
|
|
|
|
Fetched with no cookies it returns 403 {"detail":"File stream access denied."},
|
|
so the signature is not a bypass — it still rides on the session. Two things
|
|
follow, and this probe checks both:
|
|
|
|
A. What does /files/{id}/download hand back for a file that WORKS? If its
|
|
download_url is one of these estuary URLs, then that endpoint is the
|
|
minting step, and a gizmo-scoped file failing there means we are refused
|
|
at exactly the point the signature would be issued. Then the only hope is
|
|
a different minting route.
|
|
|
|
B. Does the estuary namespace expose one? /backend-api/estuary/* is new to us
|
|
and was never probed — every earlier attempt used /backend-api/files/*.
|
|
|
|
Run from the project root:
|
|
|
|
python tools/estuary_probe.py
|
|
|
|
Optionally pass a browser URL for one of the UNREACHABLE images to check
|
|
whether the exporter's session can replay it:
|
|
|
|
python tools/estuary_probe.py "https://chatgpt.com/backend-api/estuary/content?id=…"
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
from urllib.parse import parse_qs, urlparse
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402
|
|
|
|
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
|
|
|
|
|
def sample_ids(export_dir: Path) -> tuple[str | None, list[str]]:
|
|
"""(one working file id, all failed file ids)."""
|
|
working = None
|
|
failed: list[str] = []
|
|
for md in export_dir.rglob("*.md"):
|
|
try:
|
|
text = md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
for ref, _meta in DEAD_RE.findall(text):
|
|
fid = parse_asset_file_id(ref)
|
|
if fid:
|
|
failed.append(fid)
|
|
if working is None:
|
|
for _src, rel in SAVED_RE.findall(text):
|
|
stem = Path(rel).stem
|
|
if stem.startswith("file"):
|
|
working = stem
|
|
break
|
|
return working, failed
|
|
|
|
|
|
def main() -> None:
|
|
pasted = sys.argv[1] if len(sys.argv) > 1 else None
|
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
|
working, failed = sample_ids(export_dir)
|
|
provider = ChatGPTProvider()
|
|
|
|
def call(label: str, url: str, headers: dict | None = None) -> None:
|
|
try:
|
|
provider._pace()
|
|
resp = provider._session.request("GET", url, headers=headers, timeout=30)
|
|
except Exception as e: # noqa: BLE001 - diagnostic
|
|
print(f" {label:<48} error {type(e).__name__}")
|
|
return
|
|
ctype = (resp.headers.get("content-type") or "").split(";")[0]
|
|
if resp.status_code == 200 and ctype.startswith("image/"):
|
|
print(f" {label:<48} 200 {ctype} {len(resp.content)}B ← IMAGE BYTES")
|
|
return
|
|
body = ""
|
|
try:
|
|
body = json.dumps(resp.json(), default=str)[:220]
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
body = (resp.text or "")[:160].replace("\n", " ")
|
|
print(f" {label:<48} {resp.status_code} {ctype}")
|
|
if body:
|
|
print(f" {body}")
|
|
|
|
# ── A. what a working file's download_url actually looks like ──────────
|
|
print("=" * 78)
|
|
print("A. The minting step, on a file that works")
|
|
print("=" * 78)
|
|
if working:
|
|
print(f" working file: {working}")
|
|
try:
|
|
provider._pace()
|
|
resp = provider._session.request(
|
|
"GET", f"{BASE_URL}/files/{working}/download", timeout=30
|
|
)
|
|
body = resp.json() if resp.status_code == 200 else {}
|
|
url = body.get("download_url") or ""
|
|
print(f" /files/{{id}}/download → {resp.status_code}")
|
|
if url:
|
|
parsed = urlparse(url)
|
|
print(f" host {parsed.netloc}")
|
|
print(f" path {parsed.path}")
|
|
params = parse_qs(parsed.query)
|
|
print(f" query {sorted(params)}")
|
|
if "estuary" in parsed.path:
|
|
print(" → SAME estuary route the browser uses.")
|
|
print(" So /files/{id}/download IS the minting step, and a")
|
|
print(" gizmo file's 403 is a refusal to mint. Look for")
|
|
print(" another minter below.")
|
|
else:
|
|
print(" → a different route from the browser's estuary URL;")
|
|
print(" the UI must mint its URLs somewhere else.")
|
|
except Exception as e: # noqa: BLE001 - diagnostic
|
|
print(f" could not check: {e}")
|
|
else:
|
|
print(" no working image found on disk to compare against")
|
|
|
|
# ── B. does the estuary namespace offer a route for refused files? ─────
|
|
print()
|
|
print("=" * 78)
|
|
print("B. The estuary namespace, on a refused file")
|
|
print("=" * 78)
|
|
if not failed:
|
|
print(" no failed images found")
|
|
else:
|
|
# The failures are two different classes and only one is interesting.
|
|
# A deleted file 404s on /files/{id}; a refused one answers 200. Testing
|
|
# a deleted file here proves nothing, so pick a refused one.
|
|
target = None
|
|
for candidate in failed:
|
|
try:
|
|
provider._pace()
|
|
meta = provider._session.request(
|
|
"GET", f"{BASE_URL}/files/{candidate}", timeout=30
|
|
)
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
continue
|
|
if meta.status_code == 200:
|
|
target = candidate
|
|
break
|
|
print(f" (skipping {candidate[:28]}… — deleted, {meta.status_code})")
|
|
|
|
if target is None:
|
|
print(" every failure is a deleted file; nothing in the refused class")
|
|
return
|
|
print(f"\n refused file: {target} (exists, download refused)\n")
|
|
for label, url in [
|
|
("estuary/files/{id}/download", f"{BASE_URL}/estuary/files/{target}/download"),
|
|
("estuary/files/{id}", f"{BASE_URL}/estuary/files/{target}"),
|
|
("estuary/content?id=", f"{BASE_URL}/estuary/content?id={target}"),
|
|
("estuary/content?id=&p=fs&cid=1&v=0", f"{BASE_URL}/estuary/content?id={target}&p=fs&cid=1&v=0"),
|
|
("estuary/{id}", f"{BASE_URL}/estuary/{target}"),
|
|
("estuary/download?id=", f"{BASE_URL}/estuary/download?id={target}"),
|
|
("files/{id}/download?v=0", f"{BASE_URL}/files/{target}/download?v=0"),
|
|
# Validation asked only for id, p and ts — never sig. Perhaps the
|
|
# signature is enforced elsewhere, or not at all for an owner.
|
|
("estuary/content (no sig)", f"{BASE_URL}/estuary/content?id={target}&p=fs&cid=1&v=0&ts=496382"),
|
|
]:
|
|
call(label, url)
|
|
|
|
# Does a working file's freshly minted URL serve if we swap in the
|
|
# refused id? If it does, the signature is not bound to the file.
|
|
if working:
|
|
try:
|
|
provider._pace()
|
|
resp = provider._session.request(
|
|
"GET", f"{BASE_URL}/files/{working}/download", timeout=30
|
|
)
|
|
minted = (resp.json() or {}).get("download_url") if resp.status_code == 200 else None
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
minted = None
|
|
if minted:
|
|
swapped = re.sub(r"id=file_[0-9a-f]+", f"id={target}", minted)
|
|
print()
|
|
call("working file's URL, refused id swapped in", swapped)
|
|
|
|
# ── C. replay a pasted URL through our session ─────────────────────────
|
|
if pasted:
|
|
print()
|
|
print("=" * 78)
|
|
print("C. Replaying the pasted URL through the exporter session")
|
|
print("=" * 78)
|
|
params = parse_qs(urlparse(pasted).query)
|
|
pasted_id = (params.get("id") or [""])[0]
|
|
print(f" id in URL: {pasted_id}")
|
|
print(f" that id is {'a REFUSED file' if pasted_id in failed else 'NOT one of the refused files'}")
|
|
call("pasted URL", pasted)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|