"""One-off diagnostic: why do some ChatGPT media assets 403? The API answers a refused download with a bare {"detail":"Forbidden"}, so the cause has to be narrowed by experiment. This script runs the experiments that distinguish the plausible causes and prints a table. Hypothesis A — auth is fine, the asset is the problem. Probe: fetch an asset that DID download alongside one that 403'd. If the good one still 200s in the same session, the session is not at fault. Hypothesis B — the asset is scoped to a workspace/account we don't name. The exporter never sends ChatGPT-Account-Id. Resources belonging to a workspace (vs the personal account) can require it. Probe: retry the 403 with each account id the session reports. Hypothesis C — only the signed-URL step is restricted. Probe: hit /files/{id} (metadata, no /download) and see if that differs. Also, with no API call at all: what KIND of asset are the failures? The exported placeholders record source=user_upload | model_generated, so scanning exports/ classifies the failures for free. Run from the project root with the venv active: python tools/probe_media_403.py Delete this file once the cause is known. """ import os import re import sys from collections import Counter from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402 # The IDs that 403'd in the 2026-08-17 runs. FAILED_IDS = [ "file_00000000c18471f6a9b8f726c599b15d", "file_00000000667071f694d8aa00b4d42c7b", "file_00000000a53c71f6ba82c859cdfc9659", "file_00000000dccc71f69472e8526e6c7e0c", "file_00000000f34071f6af231eaf1c655f6d", "file_00000000d5e471f6bbbac75623c0cc3d", "file_000000003454722f9481506b96aed510", "file_00000000f550722f990b7c3227f2cb59", "file_0000000094c8722fa508cfe647750137", "file_000000000000722f8c9757d63f45679b", "file_000000007a4081f58297ab692fb625e5", "file_000000002f5081f5bc459887eb6f494f", "file_000000001b3071f59e30d3213f8fcca5", "file_000000009478822faec2289717d9f628", "file_00000000f0bc81f5933009bec614088c", "file_00000000902881f59dca2333091a59cc", "file_0000000017e0820cb269596ec8a79497", "file_00000000762c81f5ac85d9fdcae20061", ] ACCOUNT_ENDPOINTS = [ f"{BASE_URL}/accounts/check/v4-2023-04-27", f"{BASE_URL}/accounts/check", ] # "> 🖼️ **Image attached** — `sediment://file_x` (model_generated, image/png, …)" PLACEHOLDER_RE = re.compile( r"\*\*(?:Image|File) attached\*\* — `([^`]+)`\s*\(([^)]*)\)" ) def classify_from_exports(export_dir: Path) -> None: """Offline: what kind of asset were the failures, and where did they live?""" print("=" * 78) print("A. What the exports already say (no API calls)") print("=" * 78) if not export_dir.is_dir(): print(f" exports dir not found: {export_dir} — skipping\n") return wanted = set(FAILED_IDS) hits: dict[str, list[tuple[str, str]]] = {} all_sources: Counter = Counter() failed_sources: Counter = Counter() for md in export_dir.rglob("*.md"): try: text = md.read_text(encoding="utf-8", errors="replace") except OSError: continue for ref, meta in PLACEHOLDER_RE.findall(text): source = meta.split(",")[0].strip() all_sources[source] += 1 for fid in wanted: if fid in ref: hits.setdefault(fid, []).append((source, str(md.relative_to(export_dir)))) failed_sources[source] += 1 print(f" placeholders still unresolved across exports: {sum(all_sources.values())}") print(f" by source: {dict(all_sources)}") print(f" of those, matching a known 403 ID: {sum(failed_sources.values())}") print(f" by source: {dict(failed_sources)}") print() for fid, places in sorted(hits.items()): source, path = places[0] print(f" {fid[:28]}… source={source:<16} {path}") if not hits: print(" (no matches — exports may live elsewhere; set EXPORT_DIR)") print() def find_good_ids(export_dir: Path, limit: int = 3) -> list[str]: """IDs that downloaded successfully — media/ files are named by file ID.""" good = [] for media_file in export_dir.rglob("media/*"): if media_file.is_file() and media_file.stem.startswith("file"): good.append(media_file.stem) if len(good) >= limit: break return good def get_account_ids(provider) -> list[str]: print("=" * 78) print("B. Account / workspace IDs this session reports") print("=" * 78) ids: list[str] = [] for url in ACCOUNT_ENDPOINTS: try: resp = provider._session.request("GET", url, timeout=30) except Exception as e: # noqa: BLE001 - diagnostic print(f" {url} → error {e}") continue print(f" {url} → {resp.status_code}") if resp.status_code != 200: continue try: data = resp.json() except Exception: # noqa: BLE001 - diagnostic continue accounts = data.get("accounts") if isinstance(data, dict) else None if isinstance(accounts, dict): for key, value in accounts.items(): acct = (value or {}).get("account", {}) if isinstance(value, dict) else {} acct_id = acct.get("account_id") plan = acct.get("structure") or acct.get("plan_type") if acct_id: ids.append(acct_id) print(f" key={key!r:<28} account_id={acct_id} ({plan})") if ids: break if not ids: print(" (none discovered — hypothesis B untestable)") print() return ids def probe(provider, file_id: str, account_ids: list[str]) -> None: def call(url: str, headers: dict | None = None) -> str: try: resp = provider._session.request("GET", url, headers=headers, timeout=30) except Exception as e: # noqa: BLE001 - diagnostic return f"error {type(e).__name__}" detail = "" if resp.status_code != 200: try: detail = f" {resp.json().get('detail', '')}" except Exception: # noqa: BLE001 - diagnostic detail = f" {resp.text[:60]}" return f"{resp.status_code}{detail}" dl = f"{BASE_URL}/files/{file_id}/download" meta = f"{BASE_URL}/files/{file_id}" print(f" {file_id}") print(f" /download → {call(dl)}") print(f" /files/{{id}} (metadata) → {call(meta)}") for acct in account_ids: header = {"ChatGPT-Account-Id": acct} print(f" /download + Account-Id → {call(dl, header)} [{acct[:8]}…]") def main() -> None: export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() classify_from_exports(export_dir) provider = ChatGPTProvider() account_ids = get_account_ids(provider) print("=" * 78) print("C. Live probes") print("=" * 78) good_ids = find_good_ids(export_dir) print("-- assets that downloaded fine (control group) --") if not good_ids: print(" (none found under exports/**/media — control group unavailable)") for fid in good_ids: probe(provider, fid, account_ids) print() print("-- assets that 403'd --") for fid in FAILED_IDS[:4]: probe(provider, fid, account_ids) if __name__ == "__main__": main()