tools: probe how to download a gizmo-scoped file
metadata_diff found the discriminator. Every refused image carries
use_case="gizmo" — ChatGPT's name for Projects and custom GPTs — while
everything that downloads is image_gen or multimodal. All are state=ready,
so nothing is damaged: the plain /files/{id}/download endpoint just will
not serve a project-scoped file.
Second clue: all 11 were created 2026-07-14 within minutes of each other,
yet appear in conversations dated 2025-11 through 2026-08, several
predating their own creation_time. Something on July 14 re-created them as
gizmo-scoped copies and repointed the conversations — which is also why
the losses start in July. Not a policy change, an event.
gizmo_download_probe.py dumps the raw conversation part carrying one of
these assets (it may simply name the scope the download wants) and then
tries the plausible calls — gizmo_id/conversation_id query params, a
/gizmos/{id}/files path, a project Referer. A 200 is the fix.
This commit is contained in:
@@ -0,0 +1,185 @@
|
||||
"""How do you download a gizmo-scoped file?
|
||||
|
||||
metadata_diff.py found the discriminator: every refused image carries
|
||||
``use_case: "gizmo"`` (ChatGPT's name for Projects and custom GPTs), while
|
||||
everything that downloads is ``image_gen`` or ``multimodal``. The files are
|
||||
healthy — ``state: "ready"`` — so this is not loss, it is the plain
|
||||
/files/{id}/download endpoint declining to serve a project-scoped file.
|
||||
|
||||
So find the call that works. Two parts:
|
||||
|
||||
1. Dump the raw conversation node carrying one of these images. The part dict
|
||||
may name the scope the download needs (a gizmo id, a file token, a
|
||||
conversation id) — cheaper than guessing.
|
||||
|
||||
2. Try the plausible variants and print what each returns:
|
||||
?gizmo_id=<each configured project> scope by query param
|
||||
?conversation_id=<the conversation> scope by conversation
|
||||
/gizmos/{gizmo}/files/{id}/download scope by path
|
||||
Referer: the project URL scope by origin
|
||||
|
||||
Anything that returns 200 is the fix, and the exporter can adopt it.
|
||||
|
||||
Run from the project root:
|
||||
|
||||
python tools/gizmo_download_probe.py
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
||||
from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402
|
||||
from src.utils import redact_secrets # noqa: E402
|
||||
|
||||
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
||||
CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE)
|
||||
|
||||
|
||||
def find_candidates(export_dir: Path) -> list[tuple[str, str, str]]:
|
||||
"""(file_id, conversation_id, export filename) for images that failed."""
|
||||
out = []
|
||||
for md in export_dir.rglob("*.md"):
|
||||
try:
|
||||
text = md.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
continue
|
||||
conv_match = CONV_ID_RE.search(text)
|
||||
if not conv_match:
|
||||
continue
|
||||
for ref, _meta in DEAD_RE.findall(text):
|
||||
file_id = parse_asset_file_id(ref)
|
||||
if file_id:
|
||||
out.append((file_id, conv_match.group(1), md.name))
|
||||
return out
|
||||
|
||||
|
||||
def is_gizmo_scoped(provider, file_id: str) -> bool:
|
||||
provider._pace()
|
||||
resp = provider._session.request("GET", f"{BASE_URL}/files/{file_id}", timeout=30)
|
||||
if resp.status_code != 200:
|
||||
return False
|
||||
try:
|
||||
return resp.json().get("use_case") == "gizmo"
|
||||
except Exception: # noqa: BLE001 - diagnostic
|
||||
return False
|
||||
|
||||
|
||||
def dump_conversation_node(provider, conv_id: str, file_id: str) -> str | None:
|
||||
"""Print the raw part carrying this asset. Returns a gizmo id if found."""
|
||||
try:
|
||||
raw = provider.get_conversation(conv_id)
|
||||
except Exception as e: # noqa: BLE001 - diagnostic
|
||||
print(f" could not fetch conversation: {e}")
|
||||
return None
|
||||
|
||||
gizmo_id = raw.get("gizmo_id") or raw.get("conversation_template_id")
|
||||
print(f" conversation gizmo_id : {gizmo_id}")
|
||||
print(f" conversation keys : {sorted(raw.keys())}")
|
||||
|
||||
for node_id, node in (raw.get("mapping") or {}).items():
|
||||
message = node.get("message") or {}
|
||||
content = message.get("content") or {}
|
||||
candidates = []
|
||||
if content.get("content_type") == "image_asset_pointer":
|
||||
candidates.append(content)
|
||||
for part in content.get("parts") or []:
|
||||
if isinstance(part, dict):
|
||||
candidates.append(part)
|
||||
for part in candidates:
|
||||
pointer = part.get("asset_pointer") or ""
|
||||
if file_id in pointer:
|
||||
print(f"\n --- raw part on node {node_id} ---")
|
||||
print(json.dumps(redact_secrets(part), indent=4, default=str)[:1800])
|
||||
meta = message.get("metadata") or {}
|
||||
interesting = {
|
||||
k: v for k, v in meta.items()
|
||||
if any(t in k for t in ("gizmo", "file", "attach", "source"))
|
||||
}
|
||||
if interesting:
|
||||
print(f"\n --- message.metadata (filtered) ---")
|
||||
print(json.dumps(redact_secrets(interesting), indent=4, default=str)[:1200])
|
||||
return gizmo_id
|
||||
print(" (asset not found in the conversation mapping)")
|
||||
return gizmo_id
|
||||
|
||||
|
||||
def try_variants(provider, file_id: str, conv_id: str, gizmo_id: str | None) -> None:
|
||||
project_ids = [p for p in provider._project_ids]
|
||||
if gizmo_id and gizmo_id not in project_ids:
|
||||
project_ids.insert(0, gizmo_id)
|
||||
|
||||
base = f"{BASE_URL}/files/{file_id}/download"
|
||||
|
||||
def show(label: str, url: str, headers: dict | None = None) -> None:
|
||||
try:
|
||||
provider._pace()
|
||||
resp = provider._session.request("GET", url, headers=headers, timeout=30)
|
||||
except Exception as e: # noqa: BLE001 - diagnostic
|
||||
print(f" {label:<52} error {type(e).__name__}")
|
||||
return
|
||||
note = ""
|
||||
if resp.status_code == 200:
|
||||
try:
|
||||
note = " ← WORKS" if resp.json().get("download_url") else " (200, no download_url)"
|
||||
except Exception: # noqa: BLE001 - diagnostic
|
||||
note = " ← WORKS (non-JSON)"
|
||||
print(f" {label:<52} {resp.status_code}{note}")
|
||||
|
||||
print("\n --- variants ---")
|
||||
show("baseline", base)
|
||||
show("?conversation_id", f"{base}?conversation_id={conv_id}")
|
||||
for pid in project_ids[:6]:
|
||||
show(f"?gizmo_id={pid[:22]}…", f"{base}?gizmo_id={pid}")
|
||||
for pid in project_ids[:2]:
|
||||
show(
|
||||
f"/gizmos/{pid[:16]}…/files/…/download",
|
||||
f"{BASE_URL}/gizmos/{pid}/files/{file_id}/download",
|
||||
)
|
||||
if gizmo_id:
|
||||
show(
|
||||
"Referer: project URL",
|
||||
base,
|
||||
{"Referer": f"https://chatgpt.com/g/{gizmo_id}/project"},
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||
candidates = find_candidates(export_dir)
|
||||
if not candidates:
|
||||
print(f"no failed images found under {export_dir}")
|
||||
return
|
||||
|
||||
provider = ChatGPTProvider()
|
||||
print(f"configured project ids: {len(provider._project_ids)}")
|
||||
|
||||
tested = 0
|
||||
for file_id, conv_id, md_name in candidates:
|
||||
if not is_gizmo_scoped(provider, file_id):
|
||||
continue
|
||||
print()
|
||||
print("=" * 78)
|
||||
print(f"{file_id}")
|
||||
print(f" from {md_name}")
|
||||
print("=" * 78)
|
||||
gizmo_id = dump_conversation_node(provider, conv_id, file_id)
|
||||
try_variants(provider, file_id, conv_id, gizmo_id)
|
||||
tested += 1
|
||||
if tested >= 2:
|
||||
break
|
||||
|
||||
if not tested:
|
||||
print("no gizmo-scoped failures found — nothing to probe")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user