branch_check killed the abandoned-branch hypothesis — every lost image is
on the live branch, and the 3 images that do sit on abandoned branches are
alive. But it turned up something the earlier probing missed by sampling
four IDs from one family and generalising: of the 19 failures, only 7 are
actually gone. The other 12 answer /files/{id} with 200 and full metadata
and refuse only /download. They exist, and may be recoverable.
Three states, then: gone (404), refused (200 meta + 403 download), working
(200 both). Since metadata comes back for the refused ones, the
discriminator can be read straight off — fetch it for every file in each
state and compare fields, flagging any field whose values never overlap
between states.
Also re-checks /download now: if a file refused during the export serves
today, those 403s were transient and a retry pass recovers them, which is
a completely different fix from anything permanent.
177 lines
5.9 KiB
Python
177 lines
5.9 KiB
Python
"""What distinguishes an image ChatGPT refuses to serve from one it serves?
|
|
|
|
branch_check.py turned up something the earlier probing missed: of the images
|
|
that failed to download, only some are actually gone. The rest answer
|
|
/files/{id} with 200 and full metadata, and refuse only /download. Three
|
|
distinct states:
|
|
|
|
gone 404 on /files/{id} — deleted, unrecoverable
|
|
refused 200 on /files/{id}, 403 on download — exists, will not serve
|
|
working 200 on both — fine
|
|
|
|
"refused" is the interesting one, because those files still exist and may be
|
|
recoverable. Since the metadata comes back for them, the discriminator can be
|
|
read straight off: fetch metadata for every file in each state and compare the
|
|
fields. A field that is constant within "refused" and different in "working"
|
|
is the cause.
|
|
|
|
Also re-checks /download now. If a file that was refused during the export
|
|
serves today, the failure was transient and a retry pass recovers it — a very
|
|
different fix from anything permanent.
|
|
|
|
Run from the project root:
|
|
|
|
python tools/metadata_diff.py
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
|
|
|
|
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
|
|
|
WORKING_SAMPLE = 6
|
|
|
|
|
|
def collect(export_dir: Path) -> tuple[list[str], list[str]]:
|
|
"""(ids that failed to download, ids that downloaded fine)."""
|
|
from src.providers.chatgpt import parse_asset_file_id
|
|
|
|
failed: list[str] = []
|
|
working: list[str] = []
|
|
for md in export_dir.rglob("*.md"):
|
|
try:
|
|
text = md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
for ref, _meta in DEAD_RE.findall(text):
|
|
file_id = parse_asset_file_id(ref)
|
|
if file_id:
|
|
failed.append(file_id)
|
|
for _source, rel in SAVED_RE.findall(text):
|
|
stem = Path(rel).stem
|
|
if stem.startswith("file"):
|
|
working.append(stem)
|
|
return failed, working
|
|
|
|
|
|
def probe(provider, file_id: str) -> tuple[str, dict]:
|
|
"""(state, metadata) for one file."""
|
|
provider._pace()
|
|
meta_resp = provider._session.request(
|
|
"GET", f"{BASE_URL}/files/{file_id}", timeout=30
|
|
)
|
|
if meta_resp.status_code == 404:
|
|
return "gone", {}
|
|
if meta_resp.status_code != 200:
|
|
return f"meta-{meta_resp.status_code}", {}
|
|
|
|
try:
|
|
meta = meta_resp.json()
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
meta = {}
|
|
|
|
provider._pace()
|
|
dl_resp = provider._session.request(
|
|
"GET", f"{BASE_URL}/files/{file_id}/download", timeout=30
|
|
)
|
|
state = "working" if dl_resp.status_code == 200 else f"refused-{dl_resp.status_code}"
|
|
return state, meta
|
|
|
|
|
|
def main() -> None:
|
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
|
failed, working = collect(export_dir)
|
|
if not failed:
|
|
print(f"no failed images found under {export_dir}")
|
|
return
|
|
|
|
# Sample working files across the archive rather than all of them.
|
|
step = max(1, len(working) // WORKING_SAMPLE)
|
|
working_sample = working[::step][:WORKING_SAMPLE]
|
|
|
|
provider = ChatGPTProvider()
|
|
by_state: dict[str, list[tuple[str, dict]]] = defaultdict(list)
|
|
|
|
print("=" * 78)
|
|
print(f"Probing {len(failed)} failed + {len(working_sample)} working files")
|
|
print("=" * 78)
|
|
for file_id in failed + working_sample:
|
|
state, meta = probe(provider, file_id)
|
|
by_state[state].append((file_id, meta))
|
|
print(f" {file_id[:40]:<42} {state}")
|
|
|
|
print()
|
|
print("=" * 78)
|
|
print("States")
|
|
print("=" * 78)
|
|
for state, entries in sorted(by_state.items()):
|
|
print(f" {state:<16} {len(entries)}")
|
|
print()
|
|
|
|
if any(s.startswith("refused") for s in by_state) is False:
|
|
print(" No file is in the 'refused' state right now.")
|
|
print(" → Every previously-failed file that still exists now serves.")
|
|
print(" The export-time 403s were TRANSIENT; a retry pass recovers them.")
|
|
print()
|
|
|
|
print("=" * 78)
|
|
print("Metadata field comparison")
|
|
print("=" * 78)
|
|
states = [s for s in by_state if by_state[s] and any(m for _i, m in by_state[s])]
|
|
all_fields: set[str] = set()
|
|
for state in states:
|
|
for _fid, meta in by_state[state]:
|
|
all_fields.update(meta.keys())
|
|
|
|
for field in sorted(all_fields):
|
|
line = f" {field:<24}"
|
|
distinct_per_state = []
|
|
for state in sorted(states):
|
|
values = {
|
|
json.dumps(meta.get(field), default=str)[:28]
|
|
for _fid, meta in by_state[state]
|
|
if meta
|
|
}
|
|
shown = ", ".join(sorted(values)[:3])
|
|
if len(values) > 3:
|
|
shown += f" (+{len(values) - 3} more)"
|
|
distinct_per_state.append((state, values, shown))
|
|
line += f" [{state}] {shown}"
|
|
# Flag fields that cleanly separate the states.
|
|
value_sets = [v for _s, v, _sh in distinct_per_state]
|
|
if len(value_sets) > 1 and all(
|
|
not (a & b) for i, a in enumerate(value_sets) for b in value_sets[i + 1:]
|
|
):
|
|
line += " ← DISCRIMINATOR"
|
|
print(line)
|
|
|
|
print()
|
|
print("=" * 78)
|
|
print("One full record per state")
|
|
print("=" * 78)
|
|
for state in sorted(states):
|
|
fid, meta = next(((f, m) for f, m in by_state[state] if m), (None, None))
|
|
if not meta:
|
|
continue
|
|
print(f" --- {state} ({fid}) ---")
|
|
for key, value in sorted(meta.items()):
|
|
print(f" {key:<24} {json.dumps(value, default=str)[:90]}")
|
|
print()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|