tools: find what separates a refused image from a served one
branch_check killed the abandoned-branch hypothesis — every lost image is
on the live branch, and the 3 images that do sit on abandoned branches are
alive. But it turned up something the earlier probing missed by sampling
four IDs from one family and generalising: of the 19 failures, only 7 are
actually gone. The other 12 answer /files/{id} with 200 and full metadata
and refuse only /download. They exist, and may be recoverable.
Three states, then: gone (404), refused (200 meta + 403 download), working
(200 both). Since metadata comes back for the refused ones, the
discriminator can be read straight off — fetch it for every file in each
state and compare fields, flagging any field whose values never overlap
between states.
Also re-checks /download now: if a file refused during the export serves
today, those 403s were transient and a retry pass recovers them, which is
a completely different fix from anything permanent.
This commit is contained in:
@@ -0,0 +1,176 @@
|
||||
"""What distinguishes an image ChatGPT refuses to serve from one it serves?
|
||||
|
||||
branch_check.py turned up something the earlier probing missed: of the images
|
||||
that failed to download, only some are actually gone. The rest answer
|
||||
/files/{id} with 200 and full metadata, and refuse only /download. Three
|
||||
distinct states:
|
||||
|
||||
gone 404 on /files/{id} — deleted, unrecoverable
|
||||
refused 200 on /files/{id}, 403 on download — exists, will not serve
|
||||
working 200 on both — fine
|
||||
|
||||
"refused" is the interesting one, because those files still exist and may be
|
||||
recoverable. Since the metadata comes back for them, the discriminator can be
|
||||
read straight off: fetch metadata for every file in each state and compare the
|
||||
fields. A field that is constant within "refused" and different in "working"
|
||||
is the cause.
|
||||
|
||||
Also re-checks /download now. If a file that was refused during the export
|
||||
serves today, the failure was transient and a retry pass recovers it — a very
|
||||
different fix from anything permanent.
|
||||
|
||||
Run from the project root:
|
||||
|
||||
python tools/metadata_diff.py
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
||||
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
|
||||
|
||||
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
||||
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
||||
|
||||
WORKING_SAMPLE = 6
|
||||
|
||||
|
||||
def collect(export_dir: Path) -> tuple[list[str], list[str]]:
|
||||
"""(ids that failed to download, ids that downloaded fine)."""
|
||||
from src.providers.chatgpt import parse_asset_file_id
|
||||
|
||||
failed: list[str] = []
|
||||
working: list[str] = []
|
||||
for md in export_dir.rglob("*.md"):
|
||||
try:
|
||||
text = md.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
continue
|
||||
for ref, _meta in DEAD_RE.findall(text):
|
||||
file_id = parse_asset_file_id(ref)
|
||||
if file_id:
|
||||
failed.append(file_id)
|
||||
for _source, rel in SAVED_RE.findall(text):
|
||||
stem = Path(rel).stem
|
||||
if stem.startswith("file"):
|
||||
working.append(stem)
|
||||
return failed, working
|
||||
|
||||
|
||||
def probe(provider, file_id: str) -> tuple[str, dict]:
|
||||
"""(state, metadata) for one file."""
|
||||
provider._pace()
|
||||
meta_resp = provider._session.request(
|
||||
"GET", f"{BASE_URL}/files/{file_id}", timeout=30
|
||||
)
|
||||
if meta_resp.status_code == 404:
|
||||
return "gone", {}
|
||||
if meta_resp.status_code != 200:
|
||||
return f"meta-{meta_resp.status_code}", {}
|
||||
|
||||
try:
|
||||
meta = meta_resp.json()
|
||||
except Exception: # noqa: BLE001 - diagnostic
|
||||
meta = {}
|
||||
|
||||
provider._pace()
|
||||
dl_resp = provider._session.request(
|
||||
"GET", f"{BASE_URL}/files/{file_id}/download", timeout=30
|
||||
)
|
||||
state = "working" if dl_resp.status_code == 200 else f"refused-{dl_resp.status_code}"
|
||||
return state, meta
|
||||
|
||||
|
||||
def main() -> None:
|
||||
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||
failed, working = collect(export_dir)
|
||||
if not failed:
|
||||
print(f"no failed images found under {export_dir}")
|
||||
return
|
||||
|
||||
# Sample working files across the archive rather than all of them.
|
||||
step = max(1, len(working) // WORKING_SAMPLE)
|
||||
working_sample = working[::step][:WORKING_SAMPLE]
|
||||
|
||||
provider = ChatGPTProvider()
|
||||
by_state: dict[str, list[tuple[str, dict]]] = defaultdict(list)
|
||||
|
||||
print("=" * 78)
|
||||
print(f"Probing {len(failed)} failed + {len(working_sample)} working files")
|
||||
print("=" * 78)
|
||||
for file_id in failed + working_sample:
|
||||
state, meta = probe(provider, file_id)
|
||||
by_state[state].append((file_id, meta))
|
||||
print(f" {file_id[:40]:<42} {state}")
|
||||
|
||||
print()
|
||||
print("=" * 78)
|
||||
print("States")
|
||||
print("=" * 78)
|
||||
for state, entries in sorted(by_state.items()):
|
||||
print(f" {state:<16} {len(entries)}")
|
||||
print()
|
||||
|
||||
if any(s.startswith("refused") for s in by_state) is False:
|
||||
print(" No file is in the 'refused' state right now.")
|
||||
print(" → Every previously-failed file that still exists now serves.")
|
||||
print(" The export-time 403s were TRANSIENT; a retry pass recovers them.")
|
||||
print()
|
||||
|
||||
print("=" * 78)
|
||||
print("Metadata field comparison")
|
||||
print("=" * 78)
|
||||
states = [s for s in by_state if by_state[s] and any(m for _i, m in by_state[s])]
|
||||
all_fields: set[str] = set()
|
||||
for state in states:
|
||||
for _fid, meta in by_state[state]:
|
||||
all_fields.update(meta.keys())
|
||||
|
||||
for field in sorted(all_fields):
|
||||
line = f" {field:<24}"
|
||||
distinct_per_state = []
|
||||
for state in sorted(states):
|
||||
values = {
|
||||
json.dumps(meta.get(field), default=str)[:28]
|
||||
for _fid, meta in by_state[state]
|
||||
if meta
|
||||
}
|
||||
shown = ", ".join(sorted(values)[:3])
|
||||
if len(values) > 3:
|
||||
shown += f" (+{len(values) - 3} more)"
|
||||
distinct_per_state.append((state, values, shown))
|
||||
line += f" [{state}] {shown}"
|
||||
# Flag fields that cleanly separate the states.
|
||||
value_sets = [v for _s, v, _sh in distinct_per_state]
|
||||
if len(value_sets) > 1 and all(
|
||||
not (a & b) for i, a in enumerate(value_sets) for b in value_sets[i + 1:]
|
||||
):
|
||||
line += " ← DISCRIMINATOR"
|
||||
print(line)
|
||||
|
||||
print()
|
||||
print("=" * 78)
|
||||
print("One full record per state")
|
||||
print("=" * 78)
|
||||
for state in sorted(states):
|
||||
fid, meta = next(((f, m) for f, m in by_state[state] if m), (None, None))
|
||||
if not meta:
|
||||
continue
|
||||
print(f" --- {state} ({fid}) ---")
|
||||
for key, value in sorted(meta.items()):
|
||||
print(f" {key:<24} {json.dumps(value, default=str)[:90]}")
|
||||
print()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user