metadata_diff found the discriminator. Every refused image carries
use_case="gizmo" — ChatGPT's name for Projects and custom GPTs — while
everything that downloads is image_gen or multimodal. All are state=ready,
so nothing is damaged: the plain /files/{id}/download endpoint just will
not serve a project-scoped file.
Second clue: all 11 were created 2026-07-14 within minutes of each other,
yet appear in conversations dated 2025-11 through 2026-08, several
predating their own creation_time. Something on July 14 re-created them as
gizmo-scoped copies and repointed the conversations — which is also why
the losses start in July. Not a policy change, an event.
gizmo_download_probe.py dumps the raw conversation part carrying one of
these assets (it may simply name the scope the download wants) and then
tries the plausible calls — gizmo_id/conversation_id query params, a
/gizmos/{id}/files path, a project Referer. A 200 is the fix.
186 lines
6.6 KiB
Python
186 lines
6.6 KiB
Python
"""How do you download a gizmo-scoped file?
|
|
|
|
metadata_diff.py found the discriminator: every refused image carries
|
|
``use_case: "gizmo"`` (ChatGPT's name for Projects and custom GPTs), while
|
|
everything that downloads is ``image_gen`` or ``multimodal``. The files are
|
|
healthy — ``state: "ready"`` — so this is not loss, it is the plain
|
|
/files/{id}/download endpoint declining to serve a project-scoped file.
|
|
|
|
So find the call that works. Two parts:
|
|
|
|
1. Dump the raw conversation node carrying one of these images. The part dict
|
|
may name the scope the download needs (a gizmo id, a file token, a
|
|
conversation id) — cheaper than guessing.
|
|
|
|
2. Try the plausible variants and print what each returns:
|
|
?gizmo_id=<each configured project> scope by query param
|
|
?conversation_id=<the conversation> scope by conversation
|
|
/gizmos/{gizmo}/files/{id}/download scope by path
|
|
Referer: the project URL scope by origin
|
|
|
|
Anything that returns 200 is the fix, and the exporter can adopt it.
|
|
|
|
Run from the project root:
|
|
|
|
python tools/gizmo_download_probe.py
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402
|
|
from src.utils import redact_secrets # noqa: E402
|
|
|
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
|
CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE)
|
|
|
|
|
|
def find_candidates(export_dir: Path) -> list[tuple[str, str, str]]:
|
|
"""(file_id, conversation_id, export filename) for images that failed."""
|
|
out = []
|
|
for md in export_dir.rglob("*.md"):
|
|
try:
|
|
text = md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
conv_match = CONV_ID_RE.search(text)
|
|
if not conv_match:
|
|
continue
|
|
for ref, _meta in DEAD_RE.findall(text):
|
|
file_id = parse_asset_file_id(ref)
|
|
if file_id:
|
|
out.append((file_id, conv_match.group(1), md.name))
|
|
return out
|
|
|
|
|
|
def is_gizmo_scoped(provider, file_id: str) -> bool:
|
|
provider._pace()
|
|
resp = provider._session.request("GET", f"{BASE_URL}/files/{file_id}", timeout=30)
|
|
if resp.status_code != 200:
|
|
return False
|
|
try:
|
|
return resp.json().get("use_case") == "gizmo"
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
return False
|
|
|
|
|
|
def dump_conversation_node(provider, conv_id: str, file_id: str) -> str | None:
|
|
"""Print the raw part carrying this asset. Returns a gizmo id if found."""
|
|
try:
|
|
raw = provider.get_conversation(conv_id)
|
|
except Exception as e: # noqa: BLE001 - diagnostic
|
|
print(f" could not fetch conversation: {e}")
|
|
return None
|
|
|
|
gizmo_id = raw.get("gizmo_id") or raw.get("conversation_template_id")
|
|
print(f" conversation gizmo_id : {gizmo_id}")
|
|
print(f" conversation keys : {sorted(raw.keys())}")
|
|
|
|
for node_id, node in (raw.get("mapping") or {}).items():
|
|
message = node.get("message") or {}
|
|
content = message.get("content") or {}
|
|
candidates = []
|
|
if content.get("content_type") == "image_asset_pointer":
|
|
candidates.append(content)
|
|
for part in content.get("parts") or []:
|
|
if isinstance(part, dict):
|
|
candidates.append(part)
|
|
for part in candidates:
|
|
pointer = part.get("asset_pointer") or ""
|
|
if file_id in pointer:
|
|
print(f"\n --- raw part on node {node_id} ---")
|
|
print(json.dumps(redact_secrets(part), indent=4, default=str)[:1800])
|
|
meta = message.get("metadata") or {}
|
|
interesting = {
|
|
k: v for k, v in meta.items()
|
|
if any(t in k for t in ("gizmo", "file", "attach", "source"))
|
|
}
|
|
if interesting:
|
|
print(f"\n --- message.metadata (filtered) ---")
|
|
print(json.dumps(redact_secrets(interesting), indent=4, default=str)[:1200])
|
|
return gizmo_id
|
|
print(" (asset not found in the conversation mapping)")
|
|
return gizmo_id
|
|
|
|
|
|
def try_variants(provider, file_id: str, conv_id: str, gizmo_id: str | None) -> None:
|
|
project_ids = [p for p in provider._project_ids]
|
|
if gizmo_id and gizmo_id not in project_ids:
|
|
project_ids.insert(0, gizmo_id)
|
|
|
|
base = f"{BASE_URL}/files/{file_id}/download"
|
|
|
|
def show(label: str, url: str, headers: dict | None = None) -> None:
|
|
try:
|
|
provider._pace()
|
|
resp = provider._session.request("GET", url, headers=headers, timeout=30)
|
|
except Exception as e: # noqa: BLE001 - diagnostic
|
|
print(f" {label:<52} error {type(e).__name__}")
|
|
return
|
|
note = ""
|
|
if resp.status_code == 200:
|
|
try:
|
|
note = " ← WORKS" if resp.json().get("download_url") else " (200, no download_url)"
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
note = " ← WORKS (non-JSON)"
|
|
print(f" {label:<52} {resp.status_code}{note}")
|
|
|
|
print("\n --- variants ---")
|
|
show("baseline", base)
|
|
show("?conversation_id", f"{base}?conversation_id={conv_id}")
|
|
for pid in project_ids[:6]:
|
|
show(f"?gizmo_id={pid[:22]}…", f"{base}?gizmo_id={pid}")
|
|
for pid in project_ids[:2]:
|
|
show(
|
|
f"/gizmos/{pid[:16]}…/files/…/download",
|
|
f"{BASE_URL}/gizmos/{pid}/files/{file_id}/download",
|
|
)
|
|
if gizmo_id:
|
|
show(
|
|
"Referer: project URL",
|
|
base,
|
|
{"Referer": f"https://chatgpt.com/g/{gizmo_id}/project"},
|
|
)
|
|
|
|
|
|
def main() -> None:
|
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
|
candidates = find_candidates(export_dir)
|
|
if not candidates:
|
|
print(f"no failed images found under {export_dir}")
|
|
return
|
|
|
|
provider = ChatGPTProvider()
|
|
print(f"configured project ids: {len(provider._project_ids)}")
|
|
|
|
tested = 0
|
|
for file_id, conv_id, md_name in candidates:
|
|
if not is_gizmo_scoped(provider, file_id):
|
|
continue
|
|
print()
|
|
print("=" * 78)
|
|
print(f"{file_id}")
|
|
print(f" from {md_name}")
|
|
print("=" * 78)
|
|
gizmo_id = dump_conversation_node(provider, conv_id, file_id)
|
|
try_variants(provider, file_id, conv_id, gizmo_id)
|
|
tested += 1
|
|
if tested >= 2:
|
|
break
|
|
|
|
if not tested:
|
|
print("no gizmo-scoped failures found — nothing to probe")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|