From 726f57bdf9253c20c83df6ba5830818861abf091 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 10:50:23 -0400 Subject: [PATCH] tools: probe the estuary namespace MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A URL copied from the browser gave us the route we never tried: /backend-api/estuary/content?id=…&ts=496382&p=fs&cid=1&sig=…&v=0 Fetched with no cookies it returns 403 {"detail":"File stream access denied."}, so the signature rides on the session rather than replacing it. Every earlier probe lived under /backend-api/files/*; estuary/* is new ground. estuary_probe.py checks two things: A. What /files/{id}/download hands back for a file that works. If its download_url is an estuary URL, that endpoint is the minting step, and a gizmo file's 403 is a refusal to mint — which is why no amount of scoping helped. B. Whether the estuary namespace exposes a route that serves a refused file directly. It also replays a pasted URL through the exporter's session, and says plainly whether the id in that URL is one of the refused files — the two captured so far were working images, so they showed the shape without telling us whether the broken ones have a URL at all. --- tools/estuary_probe.py | 165 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 165 insertions(+) create mode 100644 tools/estuary_probe.py diff --git a/tools/estuary_probe.py b/tools/estuary_probe.py new file mode 100644 index 0000000..1c9b00a --- /dev/null +++ b/tools/estuary_probe.py @@ -0,0 +1,165 @@ +"""The web UI serves images from /backend-api/estuary/content — can we mint that? + +A URL copied from the browser looks like: + + /backend-api/estuary/content?id=file_…&ts=496382&p=fs&cid=1&sig=…&v=0 + +Fetched with no cookies it returns 403 {"detail":"File stream access denied."}, +so the signature is not a bypass — it still rides on the session. Two things +follow, and this probe checks both: + +A. What does /files/{id}/download hand back for a file that WORKS? If its + download_url is one of these estuary URLs, then that endpoint is the + minting step, and a gizmo-scoped file failing there means we are refused + at exactly the point the signature would be issued. Then the only hope is + a different minting route. + +B. Does the estuary namespace expose one? /backend-api/estuary/* is new to us + and was never probed — every earlier attempt used /backend-api/files/*. + +Run from the project root: + + python tools/estuary_probe.py + +Optionally pass a browser URL for one of the UNREACHABLE images to check +whether the exporter's session can replay it: + + python tools/estuary_probe.py "https://chatgpt.com/backend-api/estuary/content?id=…" +""" + +import json +import os +import re +import sys +from pathlib import Path +from urllib.parse import parse_qs, urlparse + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + +from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402 + +SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)") +DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") + + +def sample_ids(export_dir: Path) -> tuple[str | None, list[str]]: + """(one working file id, all failed file ids).""" + working = None + failed: list[str] = [] + for md in export_dir.rglob("*.md"): + try: + text = md.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + for ref, _meta in DEAD_RE.findall(text): + fid = parse_asset_file_id(ref) + if fid: + failed.append(fid) + if working is None: + for _src, rel in SAVED_RE.findall(text): + stem = Path(rel).stem + if stem.startswith("file"): + working = stem + break + return working, failed + + +def main() -> None: + pasted = sys.argv[1] if len(sys.argv) > 1 else None + export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() + working, failed = sample_ids(export_dir) + provider = ChatGPTProvider() + + def call(label: str, url: str, headers: dict | None = None) -> None: + try: + provider._pace() + resp = provider._session.request("GET", url, headers=headers, timeout=30) + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {label:<48} error {type(e).__name__}") + return + ctype = (resp.headers.get("content-type") or "").split(";")[0] + if resp.status_code == 200 and ctype.startswith("image/"): + print(f" {label:<48} 200 {ctype} {len(resp.content)}B ← IMAGE BYTES") + return + body = "" + try: + body = json.dumps(resp.json(), default=str)[:220] + except Exception: # noqa: BLE001 - diagnostic + body = (resp.text or "")[:160].replace("\n", " ") + print(f" {label:<48} {resp.status_code} {ctype}") + if body: + print(f" {body}") + + # ── A. what a working file's download_url actually looks like ────────── + print("=" * 78) + print("A. The minting step, on a file that works") + print("=" * 78) + if working: + print(f" working file: {working}") + try: + provider._pace() + resp = provider._session.request( + "GET", f"{BASE_URL}/files/{working}/download", timeout=30 + ) + body = resp.json() if resp.status_code == 200 else {} + url = body.get("download_url") or "" + print(f" /files/{{id}}/download → {resp.status_code}") + if url: + parsed = urlparse(url) + print(f" host {parsed.netloc}") + print(f" path {parsed.path}") + params = parse_qs(parsed.query) + print(f" query {sorted(params)}") + if "estuary" in parsed.path: + print(" → SAME estuary route the browser uses.") + print(" So /files/{id}/download IS the minting step, and a") + print(" gizmo file's 403 is a refusal to mint. Look for") + print(" another minter below.") + else: + print(" → a different route from the browser's estuary URL;") + print(" the UI must mint its URLs somewhere else.") + except Exception as e: # noqa: BLE001 - diagnostic + print(f" could not check: {e}") + else: + print(" no working image found on disk to compare against") + + # ── B. does the estuary namespace offer a route for refused files? ───── + print() + print("=" * 78) + print("B. The estuary namespace, on a refused file") + print("=" * 78) + if not failed: + print(" no failed images found") + else: + target = failed[0] + print(f" refused file: {target}\n") + for label, url in [ + ("estuary/files/{id}/download", f"{BASE_URL}/estuary/files/{target}/download"), + ("estuary/files/{id}", f"{BASE_URL}/estuary/files/{target}"), + ("estuary/content?id=", f"{BASE_URL}/estuary/content?id={target}"), + ("estuary/content?id=&p=fs&cid=1&v=0", f"{BASE_URL}/estuary/content?id={target}&p=fs&cid=1&v=0"), + ("estuary/{id}", f"{BASE_URL}/estuary/{target}"), + ("estuary/download?id=", f"{BASE_URL}/estuary/download?id={target}"), + ("files/{id}/download?v=0", f"{BASE_URL}/files/{target}/download?v=0"), + ]: + call(label, url) + + # ── C. replay a pasted URL through our session ───────────────────────── + if pasted: + print() + print("=" * 78) + print("C. Replaying the pasted URL through the exporter session") + print("=" * 78) + params = parse_qs(urlparse(pasted).query) + pasted_id = (params.get("id") or [""])[0] + print(f" id in URL: {pasted_id}") + print(f" that id is {'a REFUSED file' if pasted_id in failed else 'NOT one of the refused files'}") + call("pasted URL", pasted) + + +if __name__ == "__main__": + main()