"""Can a refused image be fetched by its Library ID instead? gizmo_download_probe dumped the message metadata and found what the asset pointer never carried: each attachment has a second identity. "id": "file_000000003454722f9481506b96aed510" ← refused "library_file_id": "libfile_4eb82f478fe081919127e2eba9886e86" "source": "local" The exporter only ever knew the sediment id from the asset pointer, and asks /files/{sediment_id}/download — which 403s for these. The Library is a separate store with its own ids, so the natural reading is that we are asking for a conversation-scoped copy of a file that now lives in the Library. This walks the attachments in message.metadata to pair each refused sediment id with its library_file_id, then tries the endpoints that could serve it. Whatever returns a download_url is what the exporter should use for any attachment carrying a library_file_id. Run from the project root: python tools/library_download_probe.py """ import json import os import re import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402 DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE) def find_failed(export_dir: Path) -> dict[str, set[str]]: """conversation_id → {failed sediment file ids}.""" out: dict[str, set[str]] = {} for md in export_dir.rglob("*.md"): try: text = md.read_text(encoding="utf-8", errors="replace") except OSError: continue conv = CONV_ID_RE.search(text) if not conv: continue ids = { fid for ref, _meta in DEAD_RE.findall(text) if (fid := parse_asset_file_id(ref)) } if ids: out.setdefault(conv.group(1), set()).update(ids) return out def attachment_index(raw: dict) -> dict[str, dict]: """sediment file id → its attachment record from message.metadata.""" index: dict[str, dict] = {} for node in (raw.get("mapping") or {}).values(): message = node.get("message") or {} for att in (message.get("metadata") or {}).get("attachments") or []: if isinstance(att, dict) and att.get("id"): index[att["id"]] = att return index def main() -> None: export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() failed = find_failed(export_dir) if not failed: print(f"no failed images found under {export_dir}") return provider = ChatGPTProvider() pairs: list[tuple[str, str, str]] = [] print("=" * 78) print("Pairing refused assets with their Library IDs") print("=" * 78) for conv_id, file_ids in failed.items(): try: raw = provider.get_conversation(conv_id) except Exception as e: # noqa: BLE001 - diagnostic print(f" {conv_id[:8]}… could not fetch: {e}") continue index = attachment_index(raw) for file_id in sorted(file_ids): att = index.get(file_id) lib = (att or {}).get("library_file_id") source = (att or {}).get("source") print(f" {file_id[:34]}… library={str(lib)[:34]:<36} source={source}") if lib: pairs.append((file_id, lib, conv_id)) if not pairs: print("\n No refused asset carries a library_file_id — different cause.") return file_id, lib_id, conv_id = pairs[0] print() print("=" * 78) print(f"Endpoint hunt for {lib_id}") print(f" (sediment id {file_id})") print("=" * 78) def show(label: str, url: str, method: str = "GET") -> bool: """Print status AND body. A 200 carrying an error envelope says what the endpoint wants — printing only the keys threw that away.""" try: provider._pace() resp = provider._session.request(method, url, timeout=30) except Exception as e: # noqa: BLE001 - diagnostic print(f" {label:<50} error {type(e).__name__}") return False hit = False try: body = resp.json() except Exception: # noqa: BLE001 - diagnostic preview = (resp.text or "").strip()[:200] if resp.status_code == 200 and resp.content: print(f" {label:<50} 200 non-JSON ({len(resp.content)} bytes) ← BYTES?") return True print(f" {label:<50} {resp.status_code} {preview}") return False if isinstance(body, dict) and body.get("download_url"): print(f" {label:<50} {resp.status_code} ← DOWNLOAD URL") print(f" {json.dumps(body, default=str)[:300]}") return True print(f" {label:<50} {resp.status_code}") print(f" {json.dumps(body, default=str)[:400]}") return hit candidates = [ # This one already returns 200 with an error envelope — read it. ("/files/{lib}/download", f"{BASE_URL}/files/{lib_id}/download"), ("/files/{lib}/download?conversation_id=", f"{BASE_URL}/files/{lib_id}/download?conversation_id={conv_id}"), ("/files/{lib}/download (POST)", f"{BASE_URL}/files/{lib_id}/download"), # How does the UI enumerate the Library? A listing tends to reveal both # the id form it uses and the route that serves bytes. ("/library/files?limit=3", f"{BASE_URL}/library/files?limit=3"), ("/library?limit=3", f"{BASE_URL}/library?limit=3"), ("/files?limit=3", f"{BASE_URL}/files?limit=3"), ("/my_files?limit=3", f"{BASE_URL}/my_files?limit=3"), # Content routes that serve the asset rather than a signed URL. ("/files/{lib}/content", f"{BASE_URL}/files/{lib_id}/content"), ("/content?asset_pointer=sediment://{sediment}", f"{BASE_URL}/content?asset_pointer=sediment://{file_id}"), ] winners = [] for label, url in candidates: method = "POST" if "(POST)" in label else "GET" if show(label, url, method): winners.append((label, url)) # If a second file behaves differently, the first one was the anomaly. if len(pairs) > 1: _other_file, other_lib, _other_conv = pairs[1] print() print(f" --- second sample: {other_lib} ---") show("/files/{lib}/download", f"{BASE_URL}/files/{other_lib}/download") print() print("=" * 78) print("Result") print("=" * 78) if winners: print(" Served by:") for label, url in winners: print(f" {label}") print() print(" → The exporter can pair each asset_pointer with the") print(" library_file_id in message.metadata.attachments and fetch") print(f" {len(pairs)} otherwise-unreachable image(s) this way.") else: print(" None of these served the file. The Library ID is real but the") print(" route is elsewhere — next step is watching what chatgpt.com") print(" itself requests when it renders one of these images.") if __name__ == "__main__": main()