diff --git a/tools/analyze_media_age.py b/tools/analyze_media_age.py index d993171..de77e9f 100644 --- a/tools/analyze_media_age.py +++ b/tools/analyze_media_age.py @@ -20,6 +20,15 @@ grouping outcomes by month and by source distinguishes the hypotheses: * Neither → deaths scattered across months, or concentrated in a few conversations while their neighbours survive. +IMPORTANT — what "saved" does and does not mean. It means the image was +downloaded by *some* run at *some* point, not that ChatGPT still holds it. An +image captured in June looks saved forever after, even if it died in July. So +a month with no losses shows the exports were timely; it cannot show the +assets survived. Use probe_survival_by_month.py to ask the API what is still +live today. The one signal here that is immune to this confound is a single +conversation that both kept and lost images: same age, same chat, opposite +outcomes means the cause is per-asset, not retention. + Run from the project root: python tools/analyze_media_age.py @@ -123,12 +132,21 @@ def main() -> None: print(f" dead range : {old_d or '—'} … {new_d or '—'}") if old_s and new_d and old_s < new_d: print( - f" → an asset from {old_s} survived while one from " - f"{new_d} did not: age alone does not explain the loss." + f" → an asset from {old_s} was captured while one from " + f"{new_d} was not." ) + print( + " This does NOT disprove expiry: 'captured' means some " + "run got it in time," + ) + print( + " not that it is still on ChatGPT today. Run " + "probe_survival_by_month.py to tell" + ) + print(" those apart.") elif old_s and old_d and old_s > old_d: print( - f" → everything dead is older than everything alive " + f" → everything lost is older than everything captured " f"(cutoff between {old_d} and {old_s}): consistent with expiry." ) print() diff --git a/tools/probe_survival_by_month.py b/tools/probe_survival_by_month.py new file mode 100644 index 0000000..c9f1042 --- /dev/null +++ b/tools/probe_survival_by_month.py @@ -0,0 +1,153 @@ +"""Are already-downloaded images still alive on ChatGPT, or did we just catch them in time? + +analyze_media_age.py cannot answer this. It reports whether an image was ever +captured — and an image downloaded by a run back in June looks "saved" forever +after, whether or not ChatGPT still holds it today. So a month with zero losses +proves the exports were timely, not that the assets survived. + +This closes that gap: take images that ARE on disk, grouped by the month of +their conversation, and ask the API whether each still exists right now. + + Old months still 200 → assets do not expire with age. Export cadence is not + what saved them, and shortening it would not have + saved the July losses either. + Old months now 404 → assets DO die with age; the old months are on disk + only because a run reached them in time. Cadence is + the whole ballgame, and the July losses are the first + ones we simply arrived too late for. + +Liveness is checked with /files/{id}, which answers a missing record with a +clean 404 (verified 2026-08-17); /files/{id}/download reports the same state +as a misleading 403. + +Run from the project root: + + python tools/probe_survival_by_month.py [samples_per_month] +""" + +import os +import re +import sys +from collections import defaultdict +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from dotenv import load_dotenv + +load_dotenv() + +from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402 + +SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)") +DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_") + +DEFAULT_SAMPLES = 4 + + +def collect_saved_ids(export_dir: Path) -> dict[str, list[tuple[str, str]]]: + """month → [(file_id, source)] for images already downloaded to disk.""" + by_month: dict[str, list[tuple[str, str]]] = defaultdict(list) + for md in export_dir.rglob("*.md"): + date_match = DATE_RE.search(md.name) + if not date_match: + continue + month = date_match.group(1)[:7] + try: + text = md.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + for source, rel_path in SAVED_RE.findall(text): + file_id = Path(rel_path).stem + if file_id.startswith("file"): + by_month[month].append((file_id, source or "unknown")) + return by_month + + +def main() -> None: + samples = DEFAULT_SAMPLES + if len(sys.argv) > 1: + try: + samples = max(1, int(sys.argv[1])) + except ValueError: + pass + + export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() + by_month = collect_saved_ids(export_dir) + if not by_month: + print(f"no downloaded images found under {export_dir}") + return + + provider = ChatGPTProvider() + + print("=" * 78) + print(f"Are on-disk images still live on ChatGPT? ({samples} sampled per month)") + print("=" * 78) + print(f"{'month':<9} {'alive':>6} {'gone':>6} {'err':>5} sampled sources") + + totals = {"alive": 0, "gone": 0, "err": 0} + verdict_rows: list[tuple[str, int, int]] = [] + + for month in sorted(by_month): + entries = by_month[month] + # Spread the sample across the month rather than taking the first few + # from one conversation. + step = max(1, len(entries) // samples) + picked = entries[::step][:samples] + + alive = gone = err = 0 + sources: list[str] = [] + for file_id, source in picked: + sources.append(source[:4]) + try: + provider._pace() + resp = provider._session.request( + "GET", f"{BASE_URL}/files/{file_id}", timeout=30 + ) + except Exception: # noqa: BLE001 - diagnostic + err += 1 + continue + if resp.status_code == 200: + alive += 1 + elif resp.status_code == 404: + gone += 1 + else: + err += 1 + + totals["alive"] += alive + totals["gone"] += gone + totals["err"] += err + verdict_rows.append((month, alive, gone)) + print( + f"{month:<9} {alive:>6} {gone:>6} {err:>5} " + f"{','.join(sources)} (of {len(entries)} on disk)" + ) + + print() + print(f"total: {totals['alive']} alive, {totals['gone']} gone, {totals['err']} error") + print() + print("=" * 78) + print("Verdict") + print("=" * 78) + + old_gone = [m for m, _a, g in verdict_rows[:-2] if g] + old_alive = [m for m, a, _g in verdict_rows[:-2] if a] + + if old_gone and not old_alive: + print(" Every sampled older image is GONE from ChatGPT.") + print(" → Uploads expire. Your old exports survive only because a run") + print(" reached them in time. Export cadence directly determines") + print(" what you can still save.") + elif old_alive and not old_gone: + print(" Every sampled older image is STILL LIVE on ChatGPT.") + print(" → Uploads do not expire with age. The July losses are") + print(" something else, and exporting sooner would not have") + print(" prevented them.") + else: + print(" Mixed: some older images alive, some gone.") + print(" → Not a clean expiry. Compare the 'sampled sources' column and") + print(" the per-month rates above; the cause is likely per-asset.") + + +if __name__ == "__main__": + main()