"""How do you download a gizmo-scoped file? metadata_diff.py found the discriminator: every refused image carries ``use_case: "gizmo"`` (ChatGPT's name for Projects and custom GPTs), while everything that downloads is ``image_gen`` or ``multimodal``. The files are healthy — ``state: "ready"`` — so this is not loss, it is the plain /files/{id}/download endpoint declining to serve a project-scoped file. So find the call that works. Two parts: 1. Dump the raw conversation node carrying one of these images. The part dict may name the scope the download needs (a gizmo id, a file token, a conversation id) — cheaper than guessing. 2. Try the plausible variants and print what each returns: ?gizmo_id= scope by query param ?conversation_id= scope by conversation /gizmos/{gizmo}/files/{id}/download scope by path Referer: the project URL scope by origin Anything that returns 200 is the fix, and the exporter can adopt it. Run from the project root: python tools/gizmo_download_probe.py """ import json import os import re import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402 from src.utils import redact_secrets # noqa: E402 DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE) def find_candidates(export_dir: Path) -> list[tuple[str, str, str]]: """(file_id, conversation_id, export filename) for images that failed.""" out = [] for md in export_dir.rglob("*.md"): try: text = md.read_text(encoding="utf-8", errors="replace") except OSError: continue conv_match = CONV_ID_RE.search(text) if not conv_match: continue for ref, _meta in DEAD_RE.findall(text): file_id = parse_asset_file_id(ref) if file_id: out.append((file_id, conv_match.group(1), md.name)) return out def is_gizmo_scoped(provider, file_id: str) -> bool: provider._pace() resp = provider._session.request("GET", f"{BASE_URL}/files/{file_id}", timeout=30) if resp.status_code != 200: return False try: return resp.json().get("use_case") == "gizmo" except Exception: # noqa: BLE001 - diagnostic return False def dump_conversation_node(provider, conv_id: str, file_id: str) -> str | None: """Print the raw part carrying this asset. Returns a gizmo id if found.""" try: raw = provider.get_conversation(conv_id) except Exception as e: # noqa: BLE001 - diagnostic print(f" could not fetch conversation: {e}") return None gizmo_id = raw.get("gizmo_id") or raw.get("conversation_template_id") print(f" conversation gizmo_id : {gizmo_id}") print(f" conversation keys : {sorted(raw.keys())}") for node_id, node in (raw.get("mapping") or {}).items(): message = node.get("message") or {} content = message.get("content") or {} candidates = [] if content.get("content_type") == "image_asset_pointer": candidates.append(content) for part in content.get("parts") or []: if isinstance(part, dict): candidates.append(part) for part in candidates: pointer = part.get("asset_pointer") or "" if file_id in pointer: print(f"\n --- raw part on node {node_id} ---") print(json.dumps(redact_secrets(part), indent=4, default=str)[:1800]) meta = message.get("metadata") or {} interesting = { k: v for k, v in meta.items() if any(t in k for t in ("gizmo", "file", "attach", "source")) } if interesting: print(f"\n --- message.metadata (filtered) ---") print(json.dumps(redact_secrets(interesting), indent=4, default=str)[:1200]) return gizmo_id print(" (asset not found in the conversation mapping)") return gizmo_id def try_variants(provider, file_id: str, conv_id: str, gizmo_id: str | None) -> None: project_ids = [p for p in provider._project_ids] if gizmo_id and gizmo_id not in project_ids: project_ids.insert(0, gizmo_id) base = f"{BASE_URL}/files/{file_id}/download" def show(label: str, url: str, headers: dict | None = None) -> None: try: provider._pace() resp = provider._session.request("GET", url, headers=headers, timeout=30) except Exception as e: # noqa: BLE001 - diagnostic print(f" {label:<52} error {type(e).__name__}") return note = "" if resp.status_code == 200: try: note = " ← WORKS" if resp.json().get("download_url") else " (200, no download_url)" except Exception: # noqa: BLE001 - diagnostic note = " ← WORKS (non-JSON)" print(f" {label:<52} {resp.status_code}{note}") print("\n --- variants ---") show("baseline", base) show("?conversation_id", f"{base}?conversation_id={conv_id}") for pid in project_ids[:6]: show(f"?gizmo_id={pid[:22]}…", f"{base}?gizmo_id={pid}") for pid in project_ids[:2]: show( f"/gizmos/{pid[:16]}…/files/…/download", f"{BASE_URL}/gizmos/{pid}/files/{file_id}/download", ) if gizmo_id: show( "Referer: project URL", base, {"Referer": f"https://chatgpt.com/g/{gizmo_id}/project"}, ) def main() -> None: export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() candidates = find_candidates(export_dir) if not candidates: print(f"no failed images found under {export_dir}") return provider = ChatGPTProvider() print(f"configured project ids: {len(provider._project_ids)}") tested = 0 for file_id, conv_id, md_name in candidates: if not is_gizmo_scoped(provider, file_id): continue print() print("=" * 78) print(f"{file_id}") print(f" from {md_name}") print("=" * 78) gizmo_id = dump_conversation_node(provider, conv_id, file_id) try_variants(provider, file_id, conv_id, gizmo_id) tested += 1 if tested >= 2: break if not tested: print("no gizmo-scoped failures found — nothing to probe") if __name__ == "__main__": main()