"""Are already-downloaded images still alive on ChatGPT, or did we just catch them in time? analyze_media_age.py cannot answer this. It reports whether an image was ever captured — and an image downloaded by a run back in June looks "saved" forever after, whether or not ChatGPT still holds it today. So a month with zero losses proves the exports were timely, not that the assets survived. This closes that gap: take images that ARE on disk, grouped by the month of their conversation, and ask the API whether each still exists right now. Old months still 200 → assets do not expire with age. Export cadence is not what saved them, and shortening it would not have saved the July losses either. Old months now 404 → assets DO die with age; the old months are on disk only because a run reached them in time. Cadence is the whole ballgame, and the July losses are the first ones we simply arrived too late for. Liveness is checked with /files/{id}, which answers a missing record with a clean 404 (verified 2026-08-17); /files/{id}/download reports the same state as a misleading 403. Run from the project root: python tools/probe_survival_by_month.py [samples_per_month] """ import os import re import sys from collections import defaultdict from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402 SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)") DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_") DEFAULT_SAMPLES = 4 def collect_saved_ids(export_dir: Path) -> dict[str, list[tuple[str, str]]]: """month → [(file_id, source)] for images already downloaded to disk.""" by_month: dict[str, list[tuple[str, str]]] = defaultdict(list) for md in export_dir.rglob("*.md"): date_match = DATE_RE.search(md.name) if not date_match: continue month = date_match.group(1)[:7] try: text = md.read_text(encoding="utf-8", errors="replace") except OSError: continue for source, rel_path in SAVED_RE.findall(text): file_id = Path(rel_path).stem if file_id.startswith("file"): by_month[month].append((file_id, source or "unknown")) return by_month def main() -> None: samples = DEFAULT_SAMPLES if len(sys.argv) > 1: try: samples = max(1, int(sys.argv[1])) except ValueError: pass export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() by_month = collect_saved_ids(export_dir) if not by_month: print(f"no downloaded images found under {export_dir}") return provider = ChatGPTProvider() print("=" * 78) print(f"Are on-disk images still live on ChatGPT? ({samples} sampled per month)") print("=" * 78) print(f"{'month':<9} {'alive':>6} {'gone':>6} {'err':>5} sampled sources") totals = {"alive": 0, "gone": 0, "err": 0} verdict_rows: list[tuple[str, int, int]] = [] for month in sorted(by_month): entries = by_month[month] # Spread the sample across the month rather than taking the first few # from one conversation. step = max(1, len(entries) // samples) picked = entries[::step][:samples] alive = gone = err = 0 sources: list[str] = [] for file_id, source in picked: sources.append(source[:4]) try: provider._pace() resp = provider._session.request( "GET", f"{BASE_URL}/files/{file_id}", timeout=30 ) except Exception: # noqa: BLE001 - diagnostic err += 1 continue if resp.status_code == 200: alive += 1 elif resp.status_code == 404: gone += 1 else: err += 1 totals["alive"] += alive totals["gone"] += gone totals["err"] += err verdict_rows.append((month, alive, gone)) print( f"{month:<9} {alive:>6} {gone:>6} {err:>5} " f"{','.join(sources)} (of {len(entries)} on disk)" ) print() print(f"total: {totals['alive']} alive, {totals['gone']} gone, {totals['err']} error") print() print("=" * 78) print("Verdict") print("=" * 78) old_gone = [m for m, _a, g in verdict_rows[:-2] if g] old_alive = [m for m, a, _g in verdict_rows[:-2] if a] if old_gone and not old_alive: print(" Every sampled older image is GONE from ChatGPT.") print(" → Uploads expire. Your old exports survive only because a run") print(" reached them in time. Export cadence directly determines") print(" what you can still save.") elif old_alive and not old_gone: print(" Every sampled older image is STILL LIVE on ChatGPT.") print(" → Uploads do not expire with age. The July losses are") print(" something else, and exporting sooner would not have") print(" prevented them.") else: print(" Mixed: some older images alive, some gone.") print(" → Not a clean expiry. Compare the 'sampled sources' column and") print(" the per-month rates above; the cause is likely per-asset.") if __name__ == "__main__": main()