"""Offline: is the media loss age-based, source-based, or neither? Answers the question the 403s raised β€” do I have to export within N days? β€” from the exports already on disk. No API calls, no token, nothing to expire. It works because the renderer records the outcome of every image in the Markdown itself: ![user_upload](media/file_x.png) ← downloaded, still alive > πŸ–ΌοΈ **Image attached** β€” `sediment://file_y` (user_upload, content not preserved…) ← dead or never fetched and the conversation's date is in its filename (YYYY-MM-DD_slug_id.md). So grouping outcomes by month and by source distinguishes the hypotheses: * Age-based expiry β†’ old months all-dead, recent months all-alive, with a clean cutoff between them. * Source-based β†’ user_upload dies while model_generated survives at the same age. * Neither β†’ deaths scattered across months, or concentrated in a few conversations while their neighbours survive. IMPORTANT β€” what "saved" does and does not mean. It means the image was downloaded by *some* run at *some* point, not that ChatGPT still holds it. An image captured in June looks saved forever after, even if it died in July. So a month with no losses shows the exports were timely; it cannot show the assets survived. Use probe_survival_by_month.py to ask the API what is still live today. The one signal here that is immune to this confound is a single conversation that both kept and lost images: same age, same chat, opposite outcomes means the cause is per-asset, not retention. Run from the project root: python tools/analyze_media_age.py """ import os import re import sys from collections import defaultdict from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from dotenv import load_dotenv load_dotenv() SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)") DEAD_RE = re.compile(r"πŸ–ΌοΈ \*\*Image attached\*\* β€” `([^`]+)`\s*\(([^)]*)\)") DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_") def main() -> None: export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser() if not export_dir.is_dir(): print(f"exports dir not found: {export_dir}") return # month β†’ source β†’ {"saved": n, "dead": n} stats: dict[str, dict[str, dict[str, int]]] = defaultdict( lambda: defaultdict(lambda: {"saved": 0, "dead": 0}) ) # conversations that lost at least one image losses: dict[str, dict[str, int]] = defaultdict(lambda: {"saved": 0, "dead": 0}) oldest_saved: dict[str, str] = {} newest_dead: dict[str, str] = {} oldest_dead: dict[str, str] = {} for md in export_dir.rglob("*.md"): date_match = DATE_RE.search(md.name) if not date_match: continue date = date_match.group(1) month = date[:7] try: text = md.read_text(encoding="utf-8", errors="replace") except OSError: continue conv_key = f"{date} {md.stem}" for source, _path in SAVED_RE.findall(text): source = source or "unknown" stats[month][source]["saved"] += 1 losses[conv_key]["saved"] += 1 if source not in oldest_saved or date < oldest_saved[source]: oldest_saved[source] = date for _ref, meta in DEAD_RE.findall(text): source = meta.split(",")[0].strip() or "unknown" stats[month][source]["dead"] += 1 losses[conv_key]["dead"] += 1 if source not in newest_dead or date > newest_dead[source]: newest_dead[source] = date if source not in oldest_dead or date < oldest_dead[source]: oldest_dead[source] = date if not stats: print(f"no dated conversations with images found under {export_dir}") return sources = sorted({s for m in stats.values() for s in m}) print("=" * 78) print("Image outcomes by conversation month") print("=" * 78) header = f"{'month':<9}" for source in sources: header += f" {source[:16]:>16} (saved/dead)" print(header) for month in sorted(stats): row = f"{month:<9}" for source in sources: cell = stats[month].get(source, {"saved": 0, "dead": 0}) if cell["saved"] or cell["dead"]: row += f" {cell['saved']:>8} / {cell['dead']:<17}" else: row += f" {'β€”':>8} {'':<17}" print(row) print() print("=" * 78) print("The age question") print("=" * 78) for source in sources: old_s = oldest_saved.get(source) old_d = oldest_dead.get(source) new_d = newest_dead.get(source) print(f" {source}:") print(f" oldest still downloadable : {old_s or 'β€”'}") print(f" dead range : {old_d or 'β€”'} … {new_d or 'β€”'}") if old_s and new_d and old_s < new_d: print( f" β†’ an asset from {old_s} was captured while one from " f"{new_d} was not." ) print( " This does NOT disprove expiry: 'captured' means some " "run got it in time," ) print( " not that it is still on ChatGPT today. Run " "probe_survival_by_month.py to tell" ) print(" those apart.") elif old_s and old_d and old_s > old_d: print( f" β†’ everything lost is older than everything captured " f"(cutoff between {old_d} and {old_s}): consistent with expiry." ) print() print("=" * 78) print("Conversations that lost images (clustering check)") print("=" * 78) lossy = {k: v for k, v in losses.items() if v["dead"]} for conv, counts in sorted(lossy.items()): print(f" {conv[:66]:<66} saved={counts['saved']:<4} dead={counts['dead']}") total_dead = sum(v["dead"] for v in lossy.values()) print() print(f" {len(lossy)} conversation(s) affected, {total_dead} image(s) lost") mixed = [k for k, v in lossy.items() if v["saved"]] if mixed: print( f" {len(mixed)} of them ALSO kept images β€” same conversation, same " "age, different outcome:" ) for conv in sorted(mixed): print(f" {conv[:70]}") print(" β†’ whatever killed these is per-asset, not per-conversation.") if __name__ == "__main__": main()