Files
AIChatExporter/tools/metadata_diff.py
T
JesseMarkowitz a3ac279e39 tools: find what separates a refused image from a served one
branch_check killed the abandoned-branch hypothesis — every lost image is
on the live branch, and the 3 images that do sit on abandoned branches are
alive. But it turned up something the earlier probing missed by sampling
four IDs from one family and generalising: of the 19 failures, only 7 are
actually gone. The other 12 answer /files/{id} with 200 and full metadata
and refuse only /download. They exist, and may be recoverable.

Three states, then: gone (404), refused (200 meta + 403 download), working
(200 both). Since metadata comes back for the refused ones, the
discriminator can be read straight off — fetch it for every file in each
state and compare fields, flagging any field whose values never overlap
between states.

Also re-checks /download now: if a file refused during the export serves
today, those 403s were transient and a retry pass recovers them, which is
a completely different fix from anything permanent.
2026-08-17 09:18:32 -04:00

177 lines
5.9 KiB
Python

"""What distinguishes an image ChatGPT refuses to serve from one it serves?
branch_check.py turned up something the earlier probing missed: of the images
that failed to download, only some are actually gone. The rest answer
/files/{id} with 200 and full metadata, and refuse only /download. Three
distinct states:
gone 404 on /files/{id} — deleted, unrecoverable
refused 200 on /files/{id}, 403 on download — exists, will not serve
working 200 on both — fine
"refused" is the interesting one, because those files still exist and may be
recoverable. Since the metadata comes back for them, the discriminator can be
read straight off: fetch metadata for every file in each state and compare the
fields. A field that is constant within "refused" and different in "working"
is the cause.
Also re-checks /download now. If a file that was refused during the export
serves today, the failure was transient and a retry pass recovers it — a very
different fix from anything permanent.
Run from the project root:
python tools/metadata_diff.py
"""
import json
import os
import re
import sys
from collections import defaultdict
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from dotenv import load_dotenv
load_dotenv()
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
WORKING_SAMPLE = 6
def collect(export_dir: Path) -> tuple[list[str], list[str]]:
"""(ids that failed to download, ids that downloaded fine)."""
from src.providers.chatgpt import parse_asset_file_id
failed: list[str] = []
working: list[str] = []
for md in export_dir.rglob("*.md"):
try:
text = md.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
for ref, _meta in DEAD_RE.findall(text):
file_id = parse_asset_file_id(ref)
if file_id:
failed.append(file_id)
for _source, rel in SAVED_RE.findall(text):
stem = Path(rel).stem
if stem.startswith("file"):
working.append(stem)
return failed, working
def probe(provider, file_id: str) -> tuple[str, dict]:
"""(state, metadata) for one file."""
provider._pace()
meta_resp = provider._session.request(
"GET", f"{BASE_URL}/files/{file_id}", timeout=30
)
if meta_resp.status_code == 404:
return "gone", {}
if meta_resp.status_code != 200:
return f"meta-{meta_resp.status_code}", {}
try:
meta = meta_resp.json()
except Exception: # noqa: BLE001 - diagnostic
meta = {}
provider._pace()
dl_resp = provider._session.request(
"GET", f"{BASE_URL}/files/{file_id}/download", timeout=30
)
state = "working" if dl_resp.status_code == 200 else f"refused-{dl_resp.status_code}"
return state, meta
def main() -> None:
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
failed, working = collect(export_dir)
if not failed:
print(f"no failed images found under {export_dir}")
return
# Sample working files across the archive rather than all of them.
step = max(1, len(working) // WORKING_SAMPLE)
working_sample = working[::step][:WORKING_SAMPLE]
provider = ChatGPTProvider()
by_state: dict[str, list[tuple[str, dict]]] = defaultdict(list)
print("=" * 78)
print(f"Probing {len(failed)} failed + {len(working_sample)} working files")
print("=" * 78)
for file_id in failed + working_sample:
state, meta = probe(provider, file_id)
by_state[state].append((file_id, meta))
print(f" {file_id[:40]:<42} {state}")
print()
print("=" * 78)
print("States")
print("=" * 78)
for state, entries in sorted(by_state.items()):
print(f" {state:<16} {len(entries)}")
print()
if any(s.startswith("refused") for s in by_state) is False:
print(" No file is in the 'refused' state right now.")
print(" → Every previously-failed file that still exists now serves.")
print(" The export-time 403s were TRANSIENT; a retry pass recovers them.")
print()
print("=" * 78)
print("Metadata field comparison")
print("=" * 78)
states = [s for s in by_state if by_state[s] and any(m for _i, m in by_state[s])]
all_fields: set[str] = set()
for state in states:
for _fid, meta in by_state[state]:
all_fields.update(meta.keys())
for field in sorted(all_fields):
line = f" {field:<24}"
distinct_per_state = []
for state in sorted(states):
values = {
json.dumps(meta.get(field), default=str)[:28]
for _fid, meta in by_state[state]
if meta
}
shown = ", ".join(sorted(values)[:3])
if len(values) > 3:
shown += f" (+{len(values) - 3} more)"
distinct_per_state.append((state, values, shown))
line += f" [{state}] {shown}"
# Flag fields that cleanly separate the states.
value_sets = [v for _s, v, _sh in distinct_per_state]
if len(value_sets) > 1 and all(
not (a & b) for i, a in enumerate(value_sets) for b in value_sets[i + 1:]
):
line += " ← DISCRIMINATOR"
print(line)
print()
print("=" * 78)
print("One full record per state")
print("=" * 78)
for state in sorted(states):
fid, meta = next(((f, m) for f, m in by_state[state] if m), (None, None))
if not meta:
continue
print(f" --- {state} ({fid}) ---")
for key, value in sorted(meta.items()):
print(f" {key:<24} {json.dumps(value, default=str)[:90]}")
print()
if __name__ == "__main__":
main()