tools: try fetching refused images by their Library ID
Dumping the raw message metadata found the identity the asset pointer never
carried:
"id": "file_000000003454722f9481506b96aed510" ← refused
"library_file_id": "libfile_4eb82f478fe081919127e2eba9886e86"
"source": "local"
The exporter only knows the sediment id from the asset_pointer and asks
/files/{sediment}/download, which 403s for these. The Library is a separate
store with its own ids, so we have been asking for the conversation-scoped
copy of a file that now lives in the Library. That also fits the 2026-07-14
creation-time cluster: a Library migration would mint exactly these new
records.
Neither gizmo_id, conversation_id, a /gizmos path nor a project Referer
helped (403/404), so scope was never the missing piece — identity was.
library_download_probe.py pairs each refused sediment id with its
library_file_id from message.metadata.attachments and tries the endpoints
that could serve it.
This commit is contained in:
@@ -0,0 +1,175 @@
|
|||||||
|
"""Can a refused image be fetched by its Library ID instead?
|
||||||
|
|
||||||
|
gizmo_download_probe dumped the message metadata and found what the asset
|
||||||
|
pointer never carried: each attachment has a second identity.
|
||||||
|
|
||||||
|
"id": "file_000000003454722f9481506b96aed510" ← refused
|
||||||
|
"library_file_id": "libfile_4eb82f478fe081919127e2eba9886e86"
|
||||||
|
"source": "local"
|
||||||
|
|
||||||
|
The exporter only ever knew the sediment id from the asset pointer, and asks
|
||||||
|
/files/{sediment_id}/download — which 403s for these. The Library is a
|
||||||
|
separate store with its own ids, so the natural reading is that we are asking
|
||||||
|
for a conversation-scoped copy of a file that now lives in the Library.
|
||||||
|
|
||||||
|
This walks the attachments in message.metadata to pair each refused sediment
|
||||||
|
id with its library_file_id, then tries the endpoints that could serve it.
|
||||||
|
Whatever returns a download_url is what the exporter should use for any
|
||||||
|
attachment carrying a library_file_id.
|
||||||
|
|
||||||
|
Run from the project root:
|
||||||
|
|
||||||
|
python tools/library_download_probe.py
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
|
load_dotenv()
|
||||||
|
|
||||||
|
from src.providers.chatgpt import BASE_URL, ChatGPTProvider, parse_asset_file_id # noqa: E402
|
||||||
|
|
||||||
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
||||||
|
CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE)
|
||||||
|
|
||||||
|
|
||||||
|
def find_failed(export_dir: Path) -> dict[str, set[str]]:
|
||||||
|
"""conversation_id → {failed sediment file ids}."""
|
||||||
|
out: dict[str, set[str]] = {}
|
||||||
|
for md in export_dir.rglob("*.md"):
|
||||||
|
try:
|
||||||
|
text = md.read_text(encoding="utf-8", errors="replace")
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
conv = CONV_ID_RE.search(text)
|
||||||
|
if not conv:
|
||||||
|
continue
|
||||||
|
ids = {
|
||||||
|
fid
|
||||||
|
for ref, _meta in DEAD_RE.findall(text)
|
||||||
|
if (fid := parse_asset_file_id(ref))
|
||||||
|
}
|
||||||
|
if ids:
|
||||||
|
out.setdefault(conv.group(1), set()).update(ids)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def attachment_index(raw: dict) -> dict[str, dict]:
|
||||||
|
"""sediment file id → its attachment record from message.metadata."""
|
||||||
|
index: dict[str, dict] = {}
|
||||||
|
for node in (raw.get("mapping") or {}).values():
|
||||||
|
message = node.get("message") or {}
|
||||||
|
for att in (message.get("metadata") or {}).get("attachments") or []:
|
||||||
|
if isinstance(att, dict) and att.get("id"):
|
||||||
|
index[att["id"]] = att
|
||||||
|
return index
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||||
|
failed = find_failed(export_dir)
|
||||||
|
if not failed:
|
||||||
|
print(f"no failed images found under {export_dir}")
|
||||||
|
return
|
||||||
|
|
||||||
|
provider = ChatGPTProvider()
|
||||||
|
pairs: list[tuple[str, str]] = []
|
||||||
|
|
||||||
|
print("=" * 78)
|
||||||
|
print("Pairing refused assets with their Library IDs")
|
||||||
|
print("=" * 78)
|
||||||
|
for conv_id, file_ids in failed.items():
|
||||||
|
try:
|
||||||
|
raw = provider.get_conversation(conv_id)
|
||||||
|
except Exception as e: # noqa: BLE001 - diagnostic
|
||||||
|
print(f" {conv_id[:8]}… could not fetch: {e}")
|
||||||
|
continue
|
||||||
|
index = attachment_index(raw)
|
||||||
|
for file_id in sorted(file_ids):
|
||||||
|
att = index.get(file_id)
|
||||||
|
lib = (att or {}).get("library_file_id")
|
||||||
|
source = (att or {}).get("source")
|
||||||
|
print(f" {file_id[:34]}… library={str(lib)[:34]:<36} source={source}")
|
||||||
|
if lib:
|
||||||
|
pairs.append((file_id, lib))
|
||||||
|
|
||||||
|
if not pairs:
|
||||||
|
print("\n No refused asset carries a library_file_id — different cause.")
|
||||||
|
return
|
||||||
|
|
||||||
|
file_id, lib_id = pairs[0]
|
||||||
|
print()
|
||||||
|
print("=" * 78)
|
||||||
|
print(f"Endpoint hunt for {lib_id}")
|
||||||
|
print(f" (sediment id {file_id})")
|
||||||
|
print("=" * 78)
|
||||||
|
|
||||||
|
def show(label: str, url: str) -> bool:
|
||||||
|
try:
|
||||||
|
provider._pace()
|
||||||
|
resp = provider._session.request("GET", url, timeout=30)
|
||||||
|
except Exception as e: # noqa: BLE001 - diagnostic
|
||||||
|
print(f" {label:<50} error {type(e).__name__}")
|
||||||
|
return False
|
||||||
|
note = ""
|
||||||
|
hit = False
|
||||||
|
if resp.status_code == 200:
|
||||||
|
try:
|
||||||
|
body = resp.json()
|
||||||
|
if isinstance(body, dict) and body.get("download_url"):
|
||||||
|
note = " ← DOWNLOAD URL"
|
||||||
|
hit = True
|
||||||
|
else:
|
||||||
|
keys = list(body)[:8] if isinstance(body, dict) else type(body).__name__
|
||||||
|
note = f" 200 keys={keys}"
|
||||||
|
except Exception: # noqa: BLE001 - diagnostic
|
||||||
|
note = f" 200 non-JSON ({len(resp.content)} bytes)"
|
||||||
|
hit = True
|
||||||
|
print(f" {label:<50} {resp.status_code}{note}")
|
||||||
|
return hit
|
||||||
|
|
||||||
|
candidates = [
|
||||||
|
("/files/{lib}/download", f"{BASE_URL}/files/{lib_id}/download"),
|
||||||
|
("/files/{lib}", f"{BASE_URL}/files/{lib_id}"),
|
||||||
|
("/library/files/{lib}/download", f"{BASE_URL}/library/files/{lib_id}/download"),
|
||||||
|
("/library/files/{lib}", f"{BASE_URL}/library/files/{lib_id}"),
|
||||||
|
("/library/{lib}", f"{BASE_URL}/library/{lib_id}"),
|
||||||
|
(
|
||||||
|
"/files/{sediment}/download?library_file_id=",
|
||||||
|
f"{BASE_URL}/files/{file_id}/download?library_file_id={lib_id}",
|
||||||
|
),
|
||||||
|
("/files/{lib}/download?use_case=gizmo", f"{BASE_URL}/files/{lib_id}/download?use_case=gizmo"),
|
||||||
|
]
|
||||||
|
|
||||||
|
winners = []
|
||||||
|
for label, url in candidates:
|
||||||
|
if show(label, url):
|
||||||
|
winners.append((label, url))
|
||||||
|
|
||||||
|
print()
|
||||||
|
print("=" * 78)
|
||||||
|
print("Result")
|
||||||
|
print("=" * 78)
|
||||||
|
if winners:
|
||||||
|
print(" Served by:")
|
||||||
|
for label, url in winners:
|
||||||
|
print(f" {label}")
|
||||||
|
print()
|
||||||
|
print(" → The exporter can pair each asset_pointer with the")
|
||||||
|
print(" library_file_id in message.metadata.attachments and fetch")
|
||||||
|
print(f" {len(pairs)} otherwise-unreachable image(s) this way.")
|
||||||
|
else:
|
||||||
|
print(" None of these served the file. The Library ID is real but the")
|
||||||
|
print(" route is elsewhere — next step is watching what chatgpt.com")
|
||||||
|
print(" itself requests when it renders one of these images.")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user