tools: add one-off probe for the media 403s
The improved error reporting landed, and the answer it produced is
{"detail":"Forbidden"} — generic, no reason. The cause has to be narrowed
by experiment instead, so collect the experiments in one script:
- classify the failed assets offline from the exported placeholders
(user_upload vs model_generated) — no API call needed
- probe a known-good asset in the same session as a control, to rule the
session in or out
- retry the 403 with ChatGPT-Account-Id, which the exporter never sends
and which workspace-scoped resources can require
- compare /files/{id} against /files/{id}/download
Temporary: delete once the cause is known, or fold into `doctor` if the
check earns a permanent home.
This commit is contained in:
@@ -0,0 +1,213 @@
|
|||||||
|
"""One-off diagnostic: why do some ChatGPT media assets 403?
|
||||||
|
|
||||||
|
The API answers a refused download with a bare {"detail":"Forbidden"}, so the
|
||||||
|
cause has to be narrowed by experiment. This script runs the experiments that
|
||||||
|
distinguish the plausible causes and prints a table.
|
||||||
|
|
||||||
|
Hypothesis A — auth is fine, the asset is the problem.
|
||||||
|
Probe: fetch an asset that DID download alongside one that 403'd. If the
|
||||||
|
good one still 200s in the same session, the session is not at fault.
|
||||||
|
|
||||||
|
Hypothesis B — the asset is scoped to a workspace/account we don't name.
|
||||||
|
The exporter never sends ChatGPT-Account-Id. Resources belonging to a
|
||||||
|
workspace (vs the personal account) can require it.
|
||||||
|
Probe: retry the 403 with each account id the session reports.
|
||||||
|
|
||||||
|
Hypothesis C — only the signed-URL step is restricted.
|
||||||
|
Probe: hit /files/{id} (metadata, no /download) and see if that differs.
|
||||||
|
|
||||||
|
Also, with no API call at all: what KIND of asset are the failures? The
|
||||||
|
exported placeholders record source=user_upload | model_generated, so
|
||||||
|
scanning exports/ classifies the failures for free.
|
||||||
|
|
||||||
|
Run from the project root with the venv active:
|
||||||
|
|
||||||
|
python tools/probe_media_403.py
|
||||||
|
|
||||||
|
Delete this file once the cause is known.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from collections import Counter
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
|
load_dotenv()
|
||||||
|
|
||||||
|
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
|
||||||
|
|
||||||
|
# The IDs that 403'd in the 2026-08-17 runs.
|
||||||
|
FAILED_IDS = [
|
||||||
|
"file_00000000c18471f6a9b8f726c599b15d",
|
||||||
|
"file_00000000667071f694d8aa00b4d42c7b",
|
||||||
|
"file_00000000a53c71f6ba82c859cdfc9659",
|
||||||
|
"file_00000000dccc71f69472e8526e6c7e0c",
|
||||||
|
"file_00000000f34071f6af231eaf1c655f6d",
|
||||||
|
"file_00000000d5e471f6bbbac75623c0cc3d",
|
||||||
|
"file_000000003454722f9481506b96aed510",
|
||||||
|
"file_00000000f550722f990b7c3227f2cb59",
|
||||||
|
"file_0000000094c8722fa508cfe647750137",
|
||||||
|
"file_000000000000722f8c9757d63f45679b",
|
||||||
|
"file_000000007a4081f58297ab692fb625e5",
|
||||||
|
"file_000000002f5081f5bc459887eb6f494f",
|
||||||
|
"file_000000001b3071f59e30d3213f8fcca5",
|
||||||
|
"file_000000009478822faec2289717d9f628",
|
||||||
|
"file_00000000f0bc81f5933009bec614088c",
|
||||||
|
"file_00000000902881f59dca2333091a59cc",
|
||||||
|
"file_0000000017e0820cb269596ec8a79497",
|
||||||
|
"file_00000000762c81f5ac85d9fdcae20061",
|
||||||
|
]
|
||||||
|
|
||||||
|
ACCOUNT_ENDPOINTS = [
|
||||||
|
f"{BASE_URL}/accounts/check/v4-2023-04-27",
|
||||||
|
f"{BASE_URL}/accounts/check",
|
||||||
|
]
|
||||||
|
|
||||||
|
# "> 🖼️ **Image attached** — `sediment://file_x` (model_generated, image/png, …)"
|
||||||
|
PLACEHOLDER_RE = re.compile(
|
||||||
|
r"\*\*(?:Image|File) attached\*\* — `([^`]+)`\s*\(([^)]*)\)"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def classify_from_exports(export_dir: Path) -> None:
|
||||||
|
"""Offline: what kind of asset were the failures, and where did they live?"""
|
||||||
|
print("=" * 78)
|
||||||
|
print("A. What the exports already say (no API calls)")
|
||||||
|
print("=" * 78)
|
||||||
|
if not export_dir.is_dir():
|
||||||
|
print(f" exports dir not found: {export_dir} — skipping\n")
|
||||||
|
return
|
||||||
|
|
||||||
|
wanted = set(FAILED_IDS)
|
||||||
|
hits: dict[str, list[tuple[str, str]]] = {}
|
||||||
|
all_sources: Counter = Counter()
|
||||||
|
failed_sources: Counter = Counter()
|
||||||
|
|
||||||
|
for md in export_dir.rglob("*.md"):
|
||||||
|
try:
|
||||||
|
text = md.read_text(encoding="utf-8", errors="replace")
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
for ref, meta in PLACEHOLDER_RE.findall(text):
|
||||||
|
source = meta.split(",")[0].strip()
|
||||||
|
all_sources[source] += 1
|
||||||
|
for fid in wanted:
|
||||||
|
if fid in ref:
|
||||||
|
hits.setdefault(fid, []).append((source, str(md.relative_to(export_dir))))
|
||||||
|
failed_sources[source] += 1
|
||||||
|
|
||||||
|
print(f" placeholders still unresolved across exports: {sum(all_sources.values())}")
|
||||||
|
print(f" by source: {dict(all_sources)}")
|
||||||
|
print(f" of those, matching a known 403 ID: {sum(failed_sources.values())}")
|
||||||
|
print(f" by source: {dict(failed_sources)}")
|
||||||
|
print()
|
||||||
|
for fid, places in sorted(hits.items()):
|
||||||
|
source, path = places[0]
|
||||||
|
print(f" {fid[:28]}… source={source:<16} {path}")
|
||||||
|
if not hits:
|
||||||
|
print(" (no matches — exports may live elsewhere; set EXPORT_DIR)")
|
||||||
|
print()
|
||||||
|
|
||||||
|
|
||||||
|
def find_good_ids(export_dir: Path, limit: int = 3) -> list[str]:
|
||||||
|
"""IDs that downloaded successfully — media/ files are named by file ID."""
|
||||||
|
good = []
|
||||||
|
for media_file in export_dir.rglob("media/*"):
|
||||||
|
if media_file.is_file() and media_file.stem.startswith("file"):
|
||||||
|
good.append(media_file.stem)
|
||||||
|
if len(good) >= limit:
|
||||||
|
break
|
||||||
|
return good
|
||||||
|
|
||||||
|
|
||||||
|
def get_account_ids(provider) -> list[str]:
|
||||||
|
print("=" * 78)
|
||||||
|
print("B. Account / workspace IDs this session reports")
|
||||||
|
print("=" * 78)
|
||||||
|
ids: list[str] = []
|
||||||
|
for url in ACCOUNT_ENDPOINTS:
|
||||||
|
try:
|
||||||
|
resp = provider._session.request("GET", url, timeout=30)
|
||||||
|
except Exception as e: # noqa: BLE001 - diagnostic
|
||||||
|
print(f" {url} → error {e}")
|
||||||
|
continue
|
||||||
|
print(f" {url} → {resp.status_code}")
|
||||||
|
if resp.status_code != 200:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
data = resp.json()
|
||||||
|
except Exception: # noqa: BLE001 - diagnostic
|
||||||
|
continue
|
||||||
|
accounts = data.get("accounts") if isinstance(data, dict) else None
|
||||||
|
if isinstance(accounts, dict):
|
||||||
|
for key, value in accounts.items():
|
||||||
|
acct = (value or {}).get("account", {}) if isinstance(value, dict) else {}
|
||||||
|
acct_id = acct.get("account_id")
|
||||||
|
plan = acct.get("structure") or acct.get("plan_type")
|
||||||
|
if acct_id:
|
||||||
|
ids.append(acct_id)
|
||||||
|
print(f" key={key!r:<28} account_id={acct_id} ({plan})")
|
||||||
|
if ids:
|
||||||
|
break
|
||||||
|
if not ids:
|
||||||
|
print(" (none discovered — hypothesis B untestable)")
|
||||||
|
print()
|
||||||
|
return ids
|
||||||
|
|
||||||
|
|
||||||
|
def probe(provider, file_id: str, account_ids: list[str]) -> None:
|
||||||
|
def call(url: str, headers: dict | None = None) -> str:
|
||||||
|
try:
|
||||||
|
resp = provider._session.request("GET", url, headers=headers, timeout=30)
|
||||||
|
except Exception as e: # noqa: BLE001 - diagnostic
|
||||||
|
return f"error {type(e).__name__}"
|
||||||
|
detail = ""
|
||||||
|
if resp.status_code != 200:
|
||||||
|
try:
|
||||||
|
detail = f" {resp.json().get('detail', '')}"
|
||||||
|
except Exception: # noqa: BLE001 - diagnostic
|
||||||
|
detail = f" {resp.text[:60]}"
|
||||||
|
return f"{resp.status_code}{detail}"
|
||||||
|
|
||||||
|
dl = f"{BASE_URL}/files/{file_id}/download"
|
||||||
|
meta = f"{BASE_URL}/files/{file_id}"
|
||||||
|
|
||||||
|
print(f" {file_id}")
|
||||||
|
print(f" /download → {call(dl)}")
|
||||||
|
print(f" /files/{{id}} (metadata) → {call(meta)}")
|
||||||
|
for acct in account_ids:
|
||||||
|
header = {"ChatGPT-Account-Id": acct}
|
||||||
|
print(f" /download + Account-Id → {call(dl, header)} [{acct[:8]}…]")
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||||
|
classify_from_exports(export_dir)
|
||||||
|
|
||||||
|
provider = ChatGPTProvider()
|
||||||
|
account_ids = get_account_ids(provider)
|
||||||
|
|
||||||
|
print("=" * 78)
|
||||||
|
print("C. Live probes")
|
||||||
|
print("=" * 78)
|
||||||
|
|
||||||
|
good_ids = find_good_ids(export_dir)
|
||||||
|
print("-- assets that downloaded fine (control group) --")
|
||||||
|
if not good_ids:
|
||||||
|
print(" (none found under exports/**/media — control group unavailable)")
|
||||||
|
for fid in good_ids:
|
||||||
|
probe(provider, fid, account_ids)
|
||||||
|
|
||||||
|
print()
|
||||||
|
print("-- assets that 403'd --")
|
||||||
|
for fid in FAILED_IDS[:4]:
|
||||||
|
probe(provider, fid, account_ids)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user