diff --git a/tools/probe_media_403.py b/tools/probe_media_403.py new file mode 100644 index 0000000..02bbc91 --- /dev/null +++ b/tools/probe_media_403.py @@ -0,0 +1,213 @@ +"""One-off diagnostic: why do some ChatGPT media assets 403? + +The API answers a refused download with a bare {"detail":"Forbidden"}, so the +cause has to be narrowed by experiment. This script runs the experiments that +distinguish the plausible causes and prints a table. + + Hypothesis A — auth is fine, the asset is the problem. + Probe: fetch an asset that DID download alongside one that 403'd. If the + good one still 200s in the same session, the session is not at fault. + + Hypothesis B — the asset is scoped to a workspace/account we don't name. + The exporter never sends ChatGPT-Account-Id. Resources belonging to a + workspace (vs the personal account) can require it. + Probe: retry the 403 with each account id the session reports. + + Hypothesis C — only the signed-URL step is restricted. + Probe: hit /files/{id} (metadata, no /download) and see if that differs. + + Also, with no API call at all: what KIND of asset are the failures? The + exported placeholders record source=user_upload | model_generated, so + scanning exports/ classifies the failures for free. + +Run from the project root with the venv active: + + python tools/probe_media_403.py + +Delete this file once the cause is known. +""" + +import os +import re +import sys +from collections import Counter +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + +from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402 + +# The IDs that 403'd in the 2026-08-17 runs. +FAILED_IDS = [ + "file_00000000c18471f6a9b8f726c599b15d", + "file_00000000667071f694d8aa00b4d42c7b", + "file_00000000a53c71f6ba82c859cdfc9659", + "file_00000000dccc71f69472e8526e6c7e0c", + "file_00000000f34071f6af231eaf1c655f6d", + "file_00000000d5e471f6bbbac75623c0cc3d", + "file_000000003454722f9481506b96aed510", + "file_00000000f550722f990b7c3227f2cb59", + "file_0000000094c8722fa508cfe647750137", + "file_000000000000722f8c9757d63f45679b", + "file_000000007a4081f58297ab692fb625e5", + "file_000000002f5081f5bc459887eb6f494f", + "file_000000001b3071f59e30d3213f8fcca5", + "file_000000009478822faec2289717d9f628", + "file_00000000f0bc81f5933009bec614088c", + "file_00000000902881f59dca2333091a59cc", + "file_0000000017e0820cb269596ec8a79497", + "file_00000000762c81f5ac85d9fdcae20061", +] + +ACCOUNT_ENDPOINTS = [ + f"{BASE_URL}/accounts/check/v4-2023-04-27", + f"{BASE_URL}/accounts/check", +] + +# "> 🖼️ **Image attached** — `sediment://file_x` (model_generated, image/png, …)" +PLACEHOLDER_RE = re.compile( + r"\*\*(?:Image|File) attached\*\* — `([^`]+)`\s*\(([^)]*)\)" +) + + +def classify_from_exports(export_dir: Path) -> None: + """Offline: what kind of asset were the failures, and where did they live?""" + print("=" * 78) + print("A. What the exports already say (no API calls)") + print("=" * 78) + if not export_dir.is_dir(): + print(f" exports dir not found: {export_dir} — skipping\n") + return + + wanted = set(FAILED_IDS) + hits: dict[str, list[tuple[str, str]]] = {} + all_sources: Counter = Counter() + failed_sources: Counter = Counter() + + for md in export_dir.rglob("*.md"): + try: + text = md.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + for ref, meta in PLACEHOLDER_RE.findall(text): + source = meta.split(",")[0].strip() + all_sources[source] += 1 + for fid in wanted: + if fid in ref: + hits.setdefault(fid, []).append((source, str(md.relative_to(export_dir)))) + failed_sources[source] += 1 + + print(f" placeholders still unresolved across exports: {sum(all_sources.values())}") + print(f" by source: {dict(all_sources)}") + print(f" of those, matching a known 403 ID: {sum(failed_sources.values())}") + print(f" by source: {dict(failed_sources)}") + print() + for fid, places in sorted(hits.items()): + source, path = places[0] + print(f" {fid[:28]}… source={source:<16} {path}") + if not hits: + print(" (no matches — exports may live elsewhere; set EXPORT_DIR)") + print() + + +def find_good_ids(export_dir: Path, limit: int = 3) -> list[str]: + """IDs that downloaded successfully — media/ files are named by file ID.""" + good = [] + for media_file in export_dir.rglob("media/*"): + if media_file.is_file() and media_file.stem.startswith("file"): + good.append(media_file.stem) + if len(good) >= limit: + break + return good + + +def get_account_ids(provider) -> list[str]: + print("=" * 78) + print("B. Account / workspace IDs this session reports") + print("=" * 78) + ids: list[str] = [] + for url in ACCOUNT_ENDPOINTS: + try: + resp = provider._session.request("GET", url, timeout=30) + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {url} → error {e}") + continue + print(f" {url} → {resp.status_code}") + if resp.status_code != 200: + continue + try: + data = resp.json() + except Exception: # noqa: BLE001 - diagnostic + continue + accounts = data.get("accounts") if isinstance(data, dict) else None + if isinstance(accounts, dict): + for key, value in accounts.items(): + acct = (value or {}).get("account", {}) if isinstance(value, dict) else {} + acct_id = acct.get("account_id") + plan = acct.get("structure") or acct.get("plan_type") + if acct_id: + ids.append(acct_id) + print(f" key={key!r:<28} account_id={acct_id} ({plan})") + if ids: + break + if not ids: + print(" (none discovered — hypothesis B untestable)") + print() + return ids + + +def probe(provider, file_id: str, account_ids: list[str]) -> None: + def call(url: str, headers: dict | None = None) -> str: + try: + resp = provider._session.request("GET", url, headers=headers, timeout=30) + except Exception as e: # noqa: BLE001 - diagnostic + return f"error {type(e).__name__}" + detail = "" + if resp.status_code != 200: + try: + detail = f" {resp.json().get('detail', '')}" + except Exception: # noqa: BLE001 - diagnostic + detail = f" {resp.text[:60]}" + return f"{resp.status_code}{detail}" + + dl = f"{BASE_URL}/files/{file_id}/download" + meta = f"{BASE_URL}/files/{file_id}" + + print(f" {file_id}") + print(f" /download → {call(dl)}") + print(f" /files/{{id}} (metadata) → {call(meta)}") + for acct in account_ids: + header = {"ChatGPT-Account-Id": acct} + print(f" /download + Account-Id → {call(dl, header)} [{acct[:8]}…]") + + +def main() -> None: + export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() + classify_from_exports(export_dir) + + provider = ChatGPTProvider() + account_ids = get_account_ids(provider) + + print("=" * 78) + print("C. Live probes") + print("=" * 78) + + good_ids = find_good_ids(export_dir) + print("-- assets that downloaded fine (control group) --") + if not good_ids: + print(" (none found under exports/**/media — control group unavailable)") + for fid in good_ids: + probe(provider, fid, account_ids) + + print() + print("-- assets that 403'd --") + for fid in FAILED_IDS[:4]: + probe(provider, fid, account_ids) + + +if __name__ == "__main__": + main()