From a3ac279e392fd11a3c9a07389deaf852dbc19703 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 09:18:32 -0400 Subject: [PATCH] tools: find what separates a refused image from a served one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit branch_check killed the abandoned-branch hypothesis — every lost image is on the live branch, and the 3 images that do sit on abandoned branches are alive. But it turned up something the earlier probing missed by sampling four IDs from one family and generalising: of the 19 failures, only 7 are actually gone. The other 12 answer /files/{id} with 200 and full metadata and refuse only /download. They exist, and may be recoverable. Three states, then: gone (404), refused (200 meta + 403 download), working (200 both). Since metadata comes back for the refused ones, the discriminator can be read straight off — fetch it for every file in each state and compare fields, flagging any field whose values never overlap between states. Also re-checks /download now: if a file refused during the export serves today, those 403s were transient and a retry pass recovers them, which is a completely different fix from anything permanent. --- tools/metadata_diff.py | 176 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 176 insertions(+) create mode 100644 tools/metadata_diff.py diff --git a/tools/metadata_diff.py b/tools/metadata_diff.py new file mode 100644 index 0000000..00b9d52 --- /dev/null +++ b/tools/metadata_diff.py @@ -0,0 +1,176 @@ +"""What distinguishes an image ChatGPT refuses to serve from one it serves? + +branch_check.py turned up something the earlier probing missed: of the images +that failed to download, only some are actually gone. The rest answer +/files/{id} with 200 and full metadata, and refuse only /download. Three +distinct states: + + gone 404 on /files/{id} — deleted, unrecoverable + refused 200 on /files/{id}, 403 on download — exists, will not serve + working 200 on both — fine + +"refused" is the interesting one, because those files still exist and may be +recoverable. Since the metadata comes back for them, the discriminator can be +read straight off: fetch metadata for every file in each state and compare the +fields. A field that is constant within "refused" and different in "working" +is the cause. + +Also re-checks /download now. If a file that was refused during the export +serves today, the failure was transient and a retry pass recovers it — a very +different fix from anything permanent. + +Run from the project root: + + python tools/metadata_diff.py +""" + +import json +import os +import re +import sys +from collections import defaultdict +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + +from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402 + +SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)") +DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)") + +WORKING_SAMPLE = 6 + + +def collect(export_dir: Path) -> tuple[list[str], list[str]]: + """(ids that failed to download, ids that downloaded fine).""" + from src.providers.chatgpt import parse_asset_file_id + + failed: list[str] = [] + working: list[str] = [] + for md in export_dir.rglob("*.md"): + try: + text = md.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + for ref, _meta in DEAD_RE.findall(text): + file_id = parse_asset_file_id(ref) + if file_id: + failed.append(file_id) + for _source, rel in SAVED_RE.findall(text): + stem = Path(rel).stem + if stem.startswith("file"): + working.append(stem) + return failed, working + + +def probe(provider, file_id: str) -> tuple[str, dict]: + """(state, metadata) for one file.""" + provider._pace() + meta_resp = provider._session.request( + "GET", f"{BASE_URL}/files/{file_id}", timeout=30 + ) + if meta_resp.status_code == 404: + return "gone", {} + if meta_resp.status_code != 200: + return f"meta-{meta_resp.status_code}", {} + + try: + meta = meta_resp.json() + except Exception: # noqa: BLE001 - diagnostic + meta = {} + + provider._pace() + dl_resp = provider._session.request( + "GET", f"{BASE_URL}/files/{file_id}/download", timeout=30 + ) + state = "working" if dl_resp.status_code == 200 else f"refused-{dl_resp.status_code}" + return state, meta + + +def main() -> None: + export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() + failed, working = collect(export_dir) + if not failed: + print(f"no failed images found under {export_dir}") + return + + # Sample working files across the archive rather than all of them. + step = max(1, len(working) // WORKING_SAMPLE) + working_sample = working[::step][:WORKING_SAMPLE] + + provider = ChatGPTProvider() + by_state: dict[str, list[tuple[str, dict]]] = defaultdict(list) + + print("=" * 78) + print(f"Probing {len(failed)} failed + {len(working_sample)} working files") + print("=" * 78) + for file_id in failed + working_sample: + state, meta = probe(provider, file_id) + by_state[state].append((file_id, meta)) + print(f" {file_id[:40]:<42} {state}") + + print() + print("=" * 78) + print("States") + print("=" * 78) + for state, entries in sorted(by_state.items()): + print(f" {state:<16} {len(entries)}") + print() + + if any(s.startswith("refused") for s in by_state) is False: + print(" No file is in the 'refused' state right now.") + print(" → Every previously-failed file that still exists now serves.") + print(" The export-time 403s were TRANSIENT; a retry pass recovers them.") + print() + + print("=" * 78) + print("Metadata field comparison") + print("=" * 78) + states = [s for s in by_state if by_state[s] and any(m for _i, m in by_state[s])] + all_fields: set[str] = set() + for state in states: + for _fid, meta in by_state[state]: + all_fields.update(meta.keys()) + + for field in sorted(all_fields): + line = f" {field:<24}" + distinct_per_state = [] + for state in sorted(states): + values = { + json.dumps(meta.get(field), default=str)[:28] + for _fid, meta in by_state[state] + if meta + } + shown = ", ".join(sorted(values)[:3]) + if len(values) > 3: + shown += f" (+{len(values) - 3} more)" + distinct_per_state.append((state, values, shown)) + line += f" [{state}] {shown}" + # Flag fields that cleanly separate the states. + value_sets = [v for _s, v, _sh in distinct_per_state] + if len(value_sets) > 1 and all( + not (a & b) for i, a in enumerate(value_sets) for b in value_sets[i + 1:] + ): + line += " ← DISCRIMINATOR" + print(line) + + print() + print("=" * 78) + print("One full record per state") + print("=" * 78) + for state in sorted(states): + fid, meta = next(((f, m) for f, m in by_state[state] if m), (None, None)) + if not meta: + continue + print(f" --- {state} ({fid}) ---") + for key, value in sorted(meta.items()): + print(f" {key:<24} {json.dumps(value, default=str)[:90]}") + print() + + +if __name__ == "__main__": + main()