Files
AIChatExporter/tools/estuary_probe.py
T
JesseMarkowitz 726f57bdf9 tools: probe the estuary namespace
A URL copied from the browser gave us the route we never tried:

    /backend-api/estuary/content?id=…&ts=496382&p=fs&cid=1&sig=…&v=0

Fetched with no cookies it returns 403 {"detail":"File stream access
denied."}, so the signature rides on the session rather than replacing it.
Every earlier probe lived under /backend-api/files/*; estuary/* is new
ground.

estuary_probe.py checks two things:

A. What /files/{id}/download hands back for a file that works. If its
   download_url is an estuary URL, that endpoint is the minting step, and a
   gizmo file's 403 is a refusal to mint — which is why no amount of
   scoping helped.
B. Whether the estuary namespace exposes a route that serves a refused
   file directly.

It also replays a pasted URL through the exporter's session, and says
plainly whether the id in that URL is one of the refused files — the two
captured so far were working images, so they showed the shape without
telling us whether the broken ones have a URL at all.
2026-08-17 10:50:23 -04:00

166 lines
6.6 KiB
Python

"""The web UI serves images from /backend-api/estuary/content — can we mint that?
A URL copied from the browser looks like:
/backend-api/estuary/content?id=file_…&ts=496382&p=fs&cid=1&sig=…&v=0
Fetched with no cookies it returns 403 {"detail":"File stream access denied."},
so the signature is not a bypass — it still rides on the session. Two things
follow, and this probe checks both:
A. What does /files/{id}/download hand back for a file that WORKS? If its
download_url is one of these estuary URLs, then that endpoint is the
minting step, and a gizmo-scoped file failing there means we are refused
at exactly the point the signature would be issued. Then the only hope is
a different minting route.
B. Does the estuary namespace expose one? /backend-api/estuary/* is new to us
and was never probed — every earlier attempt used /backend-api/files/*.
Run from the project root:
python tools/estuary_probe.py
Optionally pass a browser URL for one of the UNREACHABLE images to check
whether the exporter's session can replay it:
python tools/estuary_probe.py "https://chatgpt.com/backend-api/estuary/content?id=…"
"""
import json
import os
import re
import sys
from pathlib import Path
from urllib.parse import parse_qs, urlparse
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from dotenv import load_dotenv
load_dotenv()
from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
def sample_ids(export_dir: Path) -> tuple[str | None, list[str]]:
"""(one working file id, all failed file ids)."""
working = None
failed: list[str] = []
for md in export_dir.rglob("*.md"):
try:
text = md.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
for ref, _meta in DEAD_RE.findall(text):
fid = parse_asset_file_id(ref)
if fid:
failed.append(fid)
if working is None:
for _src, rel in SAVED_RE.findall(text):
stem = Path(rel).stem
if stem.startswith("file"):
working = stem
break
return working, failed
def main() -> None:
pasted = sys.argv[1] if len(sys.argv) > 1 else None
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
working, failed = sample_ids(export_dir)
provider = ChatGPTProvider()
def call(label: str, url: str, headers: dict | None = None) -> None:
try:
provider._pace()
resp = provider._session.request("GET", url, headers=headers, timeout=30)
except Exception as e: # noqa: BLE001 - diagnostic
print(f" {label:<48} error {type(e).__name__}")
return
ctype = (resp.headers.get("content-type") or "").split(";")[0]
if resp.status_code == 200 and ctype.startswith("image/"):
print(f" {label:<48} 200 {ctype} {len(resp.content)}B ← IMAGE BYTES")
return
body = ""
try:
body = json.dumps(resp.json(), default=str)[:220]
except Exception: # noqa: BLE001 - diagnostic
body = (resp.text or "")[:160].replace("\n", " ")
print(f" {label:<48} {resp.status_code} {ctype}")
if body:
print(f" {body}")
# ── A. what a working file's download_url actually looks like ──────────
print("=" * 78)
print("A. The minting step, on a file that works")
print("=" * 78)
if working:
print(f" working file: {working}")
try:
provider._pace()
resp = provider._session.request(
"GET", f"{BASE_URL}/files/{working}/download", timeout=30
)
body = resp.json() if resp.status_code == 200 else {}
url = body.get("download_url") or ""
print(f" /files/{{id}}/download → {resp.status_code}")
if url:
parsed = urlparse(url)
print(f" host {parsed.netloc}")
print(f" path {parsed.path}")
params = parse_qs(parsed.query)
print(f" query {sorted(params)}")
if "estuary" in parsed.path:
print(" → SAME estuary route the browser uses.")
print(" So /files/{id}/download IS the minting step, and a")
print(" gizmo file's 403 is a refusal to mint. Look for")
print(" another minter below.")
else:
print(" → a different route from the browser's estuary URL;")
print(" the UI must mint its URLs somewhere else.")
except Exception as e: # noqa: BLE001 - diagnostic
print(f" could not check: {e}")
else:
print(" no working image found on disk to compare against")
# ── B. does the estuary namespace offer a route for refused files? ─────
print()
print("=" * 78)
print("B. The estuary namespace, on a refused file")
print("=" * 78)
if not failed:
print(" no failed images found")
else:
target = failed[0]
print(f" refused file: {target}\n")
for label, url in [
("estuary/files/{id}/download", f"{BASE_URL}/estuary/files/{target}/download"),
("estuary/files/{id}", f"{BASE_URL}/estuary/files/{target}"),
("estuary/content?id=", f"{BASE_URL}/estuary/content?id={target}"),
("estuary/content?id=&p=fs&cid=1&v=0", f"{BASE_URL}/estuary/content?id={target}&p=fs&cid=1&v=0"),
("estuary/{id}", f"{BASE_URL}/estuary/{target}"),
("estuary/download?id=", f"{BASE_URL}/estuary/download?id={target}"),
("files/{id}/download?v=0", f"{BASE_URL}/files/{target}/download?v=0"),
]:
call(label, url)
# ── C. replay a pasted URL through our session ─────────────────────────
if pasted:
print()
print("=" * 78)
print("C. Replaying the pasted URL through the exporter session")
print("=" * 78)
params = parse_qs(urlparse(pasted).query)
pasted_id = (params.get("id") or [""])[0]
print(f" id in URL: {pasted_id}")
print(f" that id is {'a REFUSED file' if pasted_id in failed else 'NOT one of the refused files'}")
call("pasted URL", pasted)
if __name__ == "__main__":
main()