diff --git a/tools/library_download_probe.py b/tools/library_download_probe.py new file mode 100644 index 0000000..19f002a --- /dev/null +++ b/tools/library_download_probe.py @@ -0,0 +1,175 @@ +"""Can a refused image be fetched by its Library ID instead? + +gizmo_download_probe dumped the message metadata and found what the asset +pointer never carried: each attachment has a second identity. + + "id": "file_000000003454722f9481506b96aed510" ← refused + "library_file_id": "libfile_4eb82f478fe081919127e2eba9886e86" + "source": "local" + +The exporter only ever knew the sediment id from the asset pointer, and asks +/files/{sediment_id}/download — which 403s for these. The Library is a +separate store with its own ids, so the natural reading is that we are asking +for a conversation-scoped copy of a file that now lives in the Library. + +This walks the attachments in message.metadata to pair each refused sediment +id with its library_file_id, then tries the endpoints that could serve it. +Whatever returns a download_url is what the exporter should use for any +attachment carrying a library_file_id. + +Run from the project root: + + python tools/library_download_probe.py +""" + +import json +import os +import re +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + +from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402 + +DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") +CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE) + + +def find_failed(export_dir: Path) -> dict[str, set[str]]: + """conversation_id → {failed sediment file ids}.""" + out: dict[str, set[str]] = {} + for md in export_dir.rglob("*.md"): + try: + text = md.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + conv = CONV_ID_RE.search(text) + if not conv: + continue + ids = { + fid + for ref, _meta in DEAD_RE.findall(text) + if (fid := parse_asset_file_id(ref)) + } + if ids: + out.setdefault(conv.group(1), set()).update(ids) + return out + + +def attachment_index(raw: dict) -> dict[str, dict]: + """sediment file id → its attachment record from message.metadata.""" + index: dict[str, dict] = {} + for node in (raw.get("mapping") or {}).values(): + message = node.get("message") or {} + for att in (message.get("metadata") or {}).get("attachments") or []: + if isinstance(att, dict) and att.get("id"): + index[att["id"]] = att + return index + + +def main() -> None: + export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() + failed = find_failed(export_dir) + if not failed: + print(f"no failed images found under {export_dir}") + return + + provider = ChatGPTProvider() + pairs: list[tuple[str, str]] = [] + + print("=" * 78) + print("Pairing refused assets with their Library IDs") + print("=" * 78) + for conv_id, file_ids in failed.items(): + try: + raw = provider.get_conversation(conv_id) + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {conv_id[:8]}… could not fetch: {e}") + continue + index = attachment_index(raw) + for file_id in sorted(file_ids): + att = index.get(file_id) + lib = (att or {}).get("library_file_id") + source = (att or {}).get("source") + print(f" {file_id[:34]}… library={str(lib)[:34]:<36} source={source}") + if lib: + pairs.append((file_id, lib)) + + if not pairs: + print("\n No refused asset carries a library_file_id — different cause.") + return + + file_id, lib_id = pairs[0] + print() + print("=" * 78) + print(f"Endpoint hunt for {lib_id}") + print(f" (sediment id {file_id})") + print("=" * 78) + + def show(label: str, url: str) -> bool: + try: + provider._pace() + resp = provider._session.request("GET", url, timeout=30) + except Exception as e: # noqa: BLE001 - diagnostic + print(f" {label:<50} error {type(e).__name__}") + return False + note = "" + hit = False + if resp.status_code == 200: + try: + body = resp.json() + if isinstance(body, dict) and body.get("download_url"): + note = " ← DOWNLOAD URL" + hit = True + else: + keys = list(body)[:8] if isinstance(body, dict) else type(body).__name__ + note = f" 200 keys={keys}" + except Exception: # noqa: BLE001 - diagnostic + note = f" 200 non-JSON ({len(resp.content)} bytes)" + hit = True + print(f" {label:<50} {resp.status_code}{note}") + return hit + + candidates = [ + ("/files/{lib}/download", f"{BASE_URL}/files/{lib_id}/download"), + ("/files/{lib}", f"{BASE_URL}/files/{lib_id}"), + ("/library/files/{lib}/download", f"{BASE_URL}/library/files/{lib_id}/download"), + ("/library/files/{lib}", f"{BASE_URL}/library/files/{lib_id}"), + ("/library/{lib}", f"{BASE_URL}/library/{lib_id}"), + ( + "/files/{sediment}/download?library_file_id=", + f"{BASE_URL}/files/{file_id}/download?library_file_id={lib_id}", + ), + ("/files/{lib}/download?use_case=gizmo", f"{BASE_URL}/files/{lib_id}/download?use_case=gizmo"), + ] + + winners = [] + for label, url in candidates: + if show(label, url): + winners.append((label, url)) + + print() + print("=" * 78) + print("Result") + print("=" * 78) + if winners: + print(" Served by:") + for label, url in winners: + print(f" {label}") + print() + print(" → The exporter can pair each asset_pointer with the") + print(" library_file_id in message.metadata.attachments and fetch") + print(f" {len(pairs)} otherwise-unreachable image(s) this way.") + else: + print(" None of these served the file. The Library ID is real but the") + print(" route is elsewhere — next step is watching what chatgpt.com") + print(" itself requests when it renders one of these images.") + + +if __name__ == "__main__": + main()