"""What distinguishes an image ChatGPT refuses to serve from one it serves? branch_check.py turned up something the earlier probing missed: of the images that failed to download, only some are actually gone. The rest answer /files/{id} with 200 and full metadata, and refuse only /download. Three distinct states: gone 404 on /files/{id} — deleted, unrecoverable refused 200 on /files/{id}, 403 on download — exists, will not serve working 200 on both — fine "refused" is the interesting one, because those files still exist and may be recoverable. Since the metadata comes back for them, the discriminator can be read straight off: fetch metadata for every file in each state and compare the fields. A field that is constant within "refused" and different in "working" is the cause. Also re-checks /download now. If a file that was refused during the export serves today, the failure was transient and a retry pass recovers it — a very different fix from anything permanent. Run from the project root: python tools/metadata_diff.py """ import json import os import re import sys from collections import defaultdict from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402 SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)") DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") WORKING_SAMPLE = 6 def collect(export_dir: Path) -> tuple[list[str], list[str]]: """(ids that failed to download, ids that downloaded fine).""" from src.providers.chatgpt import parse_asset_file_id failed: list[str] = [] working: list[str] = [] for md in export_dir.rglob("*.md"): try: text = md.read_text(encoding="utf-8", errors="replace") except OSError: continue for ref, _meta in DEAD_RE.findall(text): file_id = parse_asset_file_id(ref) if file_id: failed.append(file_id) for _source, rel in SAVED_RE.findall(text): stem = Path(rel).stem if stem.startswith("file"): working.append(stem) return failed, working def probe(provider, file_id: str) -> tuple[str, dict]: """(state, metadata) for one file.""" provider._pace() meta_resp = provider._session.request( "GET", f"{BASE_URL}/files/{file_id}", timeout=30 ) if meta_resp.status_code == 404: return "gone", {} if meta_resp.status_code != 200: return f"meta-{meta_resp.status_code}", {} try: meta = meta_resp.json() except Exception: # noqa: BLE001 - diagnostic meta = {} provider._pace() dl_resp = provider._session.request( "GET", f"{BASE_URL}/files/{file_id}/download", timeout=30 ) state = "working" if dl_resp.status_code == 200 else f"refused-{dl_resp.status_code}" return state, meta def main() -> None: export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() failed, working = collect(export_dir) if not failed: print(f"no failed images found under {export_dir}") return # Sample working files across the archive rather than all of them. step = max(1, len(working) // WORKING_SAMPLE) working_sample = working[::step][:WORKING_SAMPLE] provider = ChatGPTProvider() by_state: dict[str, list[tuple[str, dict]]] = defaultdict(list) print("=" * 78) print(f"Probing {len(failed)} failed + {len(working_sample)} working files") print("=" * 78) for file_id in failed + working_sample: state, meta = probe(provider, file_id) by_state[state].append((file_id, meta)) print(f" {file_id[:40]:<42} {state}") print() print("=" * 78) print("States") print("=" * 78) for state, entries in sorted(by_state.items()): print(f" {state:<16} {len(entries)}") print() if any(s.startswith("refused") for s in by_state) is False: print(" No file is in the 'refused' state right now.") print(" → Every previously-failed file that still exists now serves.") print(" The export-time 403s were TRANSIENT; a retry pass recovers them.") print() print("=" * 78) print("Metadata field comparison") print("=" * 78) states = [s for s in by_state if by_state[s] and any(m for _i, m in by_state[s])] all_fields: set[str] = set() for state in states: for _fid, meta in by_state[state]: all_fields.update(meta.keys()) for field in sorted(all_fields): line = f" {field:<24}" distinct_per_state = [] for state in sorted(states): values = { json.dumps(meta.get(field), default=str)[:28] for _fid, meta in by_state[state] if meta } shown = ", ".join(sorted(values)[:3]) if len(values) > 3: shown += f" (+{len(values) - 3} more)" distinct_per_state.append((state, values, shown)) line += f" [{state}] {shown}" # Flag fields that cleanly separate the states. value_sets = [v for _s, v, _sh in distinct_per_state] if len(value_sets) > 1 and all( not (a & b) for i, a in enumerate(value_sets) for b in value_sets[i + 1:] ): line += " ← DISCRIMINATOR" print(line) print() print("=" * 78) print("One full record per state") print("=" * 78) for state in sorted(states): fid, meta = next(((f, m) for f, m in by_state[state] if m), (None, None)) if not meta: continue print(f" --- {state} ({fid}) ---") for key, value in sorted(meta.items()): print(f" {key:<24} {json.dumps(value, default=str)[:90]}") print() if __name__ == "__main__": main()