tools: probe the estuary namespace
A URL copied from the browser gave us the route we never tried:
/backend-api/estuary/content?id=…&ts=496382&p=fs&cid=1&sig=…&v=0
Fetched with no cookies it returns 403 {"detail":"File stream access
denied."}, so the signature rides on the session rather than replacing it.
Every earlier probe lived under /backend-api/files/*; estuary/* is new
ground.
estuary_probe.py checks two things:
A. What /files/{id}/download hands back for a file that works. If its
download_url is an estuary URL, that endpoint is the minting step, and a
gizmo file's 403 is a refusal to mint — which is why no amount of
scoping helped.
B. Whether the estuary namespace exposes a route that serves a refused
file directly.
It also replays a pasted URL through the exporter's session, and says
plainly whether the id in that URL is one of the refused files — the two
captured so far were working images, so they showed the shape without
telling us whether the broken ones have a URL at all.
This commit is contained in:
@@ -0,0 +1,165 @@
|
||||
"""The web UI serves images from /backend-api/estuary/content — can we mint that?
|
||||
|
||||
A URL copied from the browser looks like:
|
||||
|
||||
/backend-api/estuary/content?id=file_…&ts=496382&p=fs&cid=1&sig=…&v=0
|
||||
|
||||
Fetched with no cookies it returns 403 {"detail":"File stream access denied."},
|
||||
so the signature is not a bypass — it still rides on the session. Two things
|
||||
follow, and this probe checks both:
|
||||
|
||||
A. What does /files/{id}/download hand back for a file that WORKS? If its
|
||||
download_url is one of these estuary URLs, then that endpoint is the
|
||||
minting step, and a gizmo-scoped file failing there means we are refused
|
||||
at exactly the point the signature would be issued. Then the only hope is
|
||||
a different minting route.
|
||||
|
||||
B. Does the estuary namespace expose one? /backend-api/estuary/* is new to us
|
||||
and was never probed — every earlier attempt used /backend-api/files/*.
|
||||
|
||||
Run from the project root:
|
||||
|
||||
python tools/estuary_probe.py
|
||||
|
||||
Optionally pass a browser URL for one of the UNREACHABLE images to check
|
||||
whether the exporter's session can replay it:
|
||||
|
||||
python tools/estuary_probe.py "https://chatgpt.com/backend-api/estuary/content?id=…"
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from urllib.parse import parse_qs, urlparse
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
||||
from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402
|
||||
|
||||
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
||||
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
||||
|
||||
|
||||
def sample_ids(export_dir: Path) -> tuple[str | None, list[str]]:
|
||||
"""(one working file id, all failed file ids)."""
|
||||
working = None
|
||||
failed: list[str] = []
|
||||
for md in export_dir.rglob("*.md"):
|
||||
try:
|
||||
text = md.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
continue
|
||||
for ref, _meta in DEAD_RE.findall(text):
|
||||
fid = parse_asset_file_id(ref)
|
||||
if fid:
|
||||
failed.append(fid)
|
||||
if working is None:
|
||||
for _src, rel in SAVED_RE.findall(text):
|
||||
stem = Path(rel).stem
|
||||
if stem.startswith("file"):
|
||||
working = stem
|
||||
break
|
||||
return working, failed
|
||||
|
||||
|
||||
def main() -> None:
|
||||
pasted = sys.argv[1] if len(sys.argv) > 1 else None
|
||||
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||
working, failed = sample_ids(export_dir)
|
||||
provider = ChatGPTProvider()
|
||||
|
||||
def call(label: str, url: str, headers: dict | None = None) -> None:
|
||||
try:
|
||||
provider._pace()
|
||||
resp = provider._session.request("GET", url, headers=headers, timeout=30)
|
||||
except Exception as e: # noqa: BLE001 - diagnostic
|
||||
print(f" {label:<48} error {type(e).__name__}")
|
||||
return
|
||||
ctype = (resp.headers.get("content-type") or "").split(";")[0]
|
||||
if resp.status_code == 200 and ctype.startswith("image/"):
|
||||
print(f" {label:<48} 200 {ctype} {len(resp.content)}B ← IMAGE BYTES")
|
||||
return
|
||||
body = ""
|
||||
try:
|
||||
body = json.dumps(resp.json(), default=str)[:220]
|
||||
except Exception: # noqa: BLE001 - diagnostic
|
||||
body = (resp.text or "")[:160].replace("\n", " ")
|
||||
print(f" {label:<48} {resp.status_code} {ctype}")
|
||||
if body:
|
||||
print(f" {body}")
|
||||
|
||||
# ── A. what a working file's download_url actually looks like ──────────
|
||||
print("=" * 78)
|
||||
print("A. The minting step, on a file that works")
|
||||
print("=" * 78)
|
||||
if working:
|
||||
print(f" working file: {working}")
|
||||
try:
|
||||
provider._pace()
|
||||
resp = provider._session.request(
|
||||
"GET", f"{BASE_URL}/files/{working}/download", timeout=30
|
||||
)
|
||||
body = resp.json() if resp.status_code == 200 else {}
|
||||
url = body.get("download_url") or ""
|
||||
print(f" /files/{{id}}/download → {resp.status_code}")
|
||||
if url:
|
||||
parsed = urlparse(url)
|
||||
print(f" host {parsed.netloc}")
|
||||
print(f" path {parsed.path}")
|
||||
params = parse_qs(parsed.query)
|
||||
print(f" query {sorted(params)}")
|
||||
if "estuary" in parsed.path:
|
||||
print(" → SAME estuary route the browser uses.")
|
||||
print(" So /files/{id}/download IS the minting step, and a")
|
||||
print(" gizmo file's 403 is a refusal to mint. Look for")
|
||||
print(" another minter below.")
|
||||
else:
|
||||
print(" → a different route from the browser's estuary URL;")
|
||||
print(" the UI must mint its URLs somewhere else.")
|
||||
except Exception as e: # noqa: BLE001 - diagnostic
|
||||
print(f" could not check: {e}")
|
||||
else:
|
||||
print(" no working image found on disk to compare against")
|
||||
|
||||
# ── B. does the estuary namespace offer a route for refused files? ─────
|
||||
print()
|
||||
print("=" * 78)
|
||||
print("B. The estuary namespace, on a refused file")
|
||||
print("=" * 78)
|
||||
if not failed:
|
||||
print(" no failed images found")
|
||||
else:
|
||||
target = failed[0]
|
||||
print(f" refused file: {target}\n")
|
||||
for label, url in [
|
||||
("estuary/files/{id}/download", f"{BASE_URL}/estuary/files/{target}/download"),
|
||||
("estuary/files/{id}", f"{BASE_URL}/estuary/files/{target}"),
|
||||
("estuary/content?id=", f"{BASE_URL}/estuary/content?id={target}"),
|
||||
("estuary/content?id=&p=fs&cid=1&v=0", f"{BASE_URL}/estuary/content?id={target}&p=fs&cid=1&v=0"),
|
||||
("estuary/{id}", f"{BASE_URL}/estuary/{target}"),
|
||||
("estuary/download?id=", f"{BASE_URL}/estuary/download?id={target}"),
|
||||
("files/{id}/download?v=0", f"{BASE_URL}/files/{target}/download?v=0"),
|
||||
]:
|
||||
call(label, url)
|
||||
|
||||
# ── C. replay a pasted URL through our session ─────────────────────────
|
||||
if pasted:
|
||||
print()
|
||||
print("=" * 78)
|
||||
print("C. Replaying the pasted URL through the exporter session")
|
||||
print("=" * 78)
|
||||
params = parse_qs(urlparse(pasted).query)
|
||||
pasted_id = (params.get("id") or [""])[0]
|
||||
print(f" id in URL: {pasted_id}")
|
||||
print(f" that id is {'a REFUSED file' if pasted_id in failed else 'NOT one of the refused files'}")
|
||||
call("pasted URL", pasted)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user