From b1da1df9864fda868845ab2a84db193911950042 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 10:57:52 -0400 Subject: [PATCH] tools: probe a refused file, not a deleted one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Confirmed by the last run: a working file's download_url is chatgpt.com/backend-api/estuary/content?cid&id&p&sig&ts&v — the same route the browser uses. So /files/{id}/download is the minting endpoint, and a gizmo file's 403 is a refusal to mint the signature. That is why no amount of scoping helped; we were turned away at the only door that issues them. Nothing else in the estuary namespace serves files: 404 across the board. But section B tested file_000000001b3071f5…, one of the seven DELETED files, because it took the first failure without checking its class. Those results say nothing about the eleven recoverable ones. - Pick the target by probing /files/{id} and taking one that answers 200, skipping (and naming) the deleted ones. - Add two experiments: estuary/content with a ts but no sig, since validation asked only for id/p/ts and never sig; and a working file's freshly minted URL with the refused id swapped in, which shows whether the signature is bound to the file. --- tools/estuary_probe.py | 42 ++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 40 insertions(+), 2 deletions(-) diff --git a/tools/estuary_probe.py b/tools/estuary_probe.py index 1c9b00a..f3e61dd 100644 --- a/tools/estuary_probe.py +++ b/tools/estuary_probe.py @@ -135,8 +135,27 @@ def main() -> None: if not failed: print(" no failed images found") else: - target = failed[0] - print(f" refused file: {target}\n") + # The failures are two different classes and only one is interesting. + # A deleted file 404s on /files/{id}; a refused one answers 200. Testing + # a deleted file here proves nothing, so pick a refused one. + target = None + for candidate in failed: + try: + provider._pace() + meta = provider._session.request( + "GET", f"{BASE_URL}/files/{candidate}", timeout=30 + ) + except Exception: # noqa: BLE001 - diagnostic + continue + if meta.status_code == 200: + target = candidate + break + print(f" (skipping {candidate[:28]}… — deleted, {meta.status_code})") + + if target is None: + print(" every failure is a deleted file; nothing in the refused class") + return + print(f"\n refused file: {target} (exists, download refused)\n") for label, url in [ ("estuary/files/{id}/download", f"{BASE_URL}/estuary/files/{target}/download"), ("estuary/files/{id}", f"{BASE_URL}/estuary/files/{target}"), @@ -145,9 +164,28 @@ def main() -> None: ("estuary/{id}", f"{BASE_URL}/estuary/{target}"), ("estuary/download?id=", f"{BASE_URL}/estuary/download?id={target}"), ("files/{id}/download?v=0", f"{BASE_URL}/files/{target}/download?v=0"), + # Validation asked only for id, p and ts — never sig. Perhaps the + # signature is enforced elsewhere, or not at all for an owner. + ("estuary/content (no sig)", f"{BASE_URL}/estuary/content?id={target}&p=fs&cid=1&v=0&ts=496382"), ]: call(label, url) + # Does a working file's freshly minted URL serve if we swap in the + # refused id? If it does, the signature is not bound to the file. + if working: + try: + provider._pace() + resp = provider._session.request( + "GET", f"{BASE_URL}/files/{working}/download", timeout=30 + ) + minted = (resp.json() or {}).get("download_url") if resp.status_code == 200 else None + except Exception: # noqa: BLE001 - diagnostic + minted = None + if minted: + swapped = re.sub(r"id=file_[0-9a-f]+", f"id={target}", minted) + print() + call("working file's URL, refused id swapped in", swapped) + # ── C. replay a pasted URL through our session ───────────────────────── if pasted: print()