analyze_media_age.py claimed age was ruled out because an image from
2025-11 was saved while one from 2026-08 was not. That conclusion does not
follow. "Saved" means some earlier run downloaded it, not that ChatGPT
still holds it — an image captured in June looks saved forever after, even
if it died in July. A month with no losses shows the exports were timely,
not that the assets survived.
- probe_survival_by_month.py: sample images already on disk, grouped by
their conversation's month, and ask /files/{id} whether each still
exists today. Old months still 200 → no expiry, and cadence did not save
them. Old months now 404 → uploads do expire and cadence is the whole
ballgame.
- analyze_media_age.py: stop asserting the unsupported verdict; say what
the number does and does not show, and point at the probe.
One signal there is immune to the confound and survives: 2026-07-09 kept
38 images and lost 12. Same conversation, same day, opposite outcomes —
no retention policy does that, so at least part of this is per-asset.
176 lines
6.5 KiB
Python
176 lines
6.5 KiB
Python
"""Offline: is the media loss age-based, source-based, or neither?
|
|
|
|
Answers the question the 403s raised — do I have to export within N days? —
|
|
from the exports already on disk. No API calls, no token, nothing to expire.
|
|
|
|
It works because the renderer records the outcome of every image in the
|
|
Markdown itself:
|
|
|
|
 ← downloaded, still alive
|
|
> 🖼️ **Image attached** — `sediment://file_y`
|
|
(user_upload, content not preserved…) ← dead or never fetched
|
|
|
|
and the conversation's date is in its filename (YYYY-MM-DD_slug_id.md). So
|
|
grouping outcomes by month and by source distinguishes the hypotheses:
|
|
|
|
* Age-based expiry → old months all-dead, recent months all-alive, with a
|
|
clean cutoff between them.
|
|
* Source-based → user_upload dies while model_generated survives at
|
|
the same age.
|
|
* Neither → deaths scattered across months, or concentrated in a
|
|
few conversations while their neighbours survive.
|
|
|
|
IMPORTANT — what "saved" does and does not mean. It means the image was
|
|
downloaded by *some* run at *some* point, not that ChatGPT still holds it. An
|
|
image captured in June looks saved forever after, even if it died in July. So
|
|
a month with no losses shows the exports were timely; it cannot show the
|
|
assets survived. Use probe_survival_by_month.py to ask the API what is still
|
|
live today. The one signal here that is immune to this confound is a single
|
|
conversation that both kept and lost images: same age, same chat, opposite
|
|
outcomes means the cause is per-asset, not retention.
|
|
|
|
Run from the project root:
|
|
|
|
python tools/analyze_media_age.py
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
|
DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_")
|
|
|
|
|
|
def main() -> None:
|
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
|
if not export_dir.is_dir():
|
|
print(f"exports dir not found: {export_dir}")
|
|
return
|
|
|
|
# month → source → {"saved": n, "dead": n}
|
|
stats: dict[str, dict[str, dict[str, int]]] = defaultdict(
|
|
lambda: defaultdict(lambda: {"saved": 0, "dead": 0})
|
|
)
|
|
# conversations that lost at least one image
|
|
losses: dict[str, dict[str, int]] = defaultdict(lambda: {"saved": 0, "dead": 0})
|
|
oldest_saved: dict[str, str] = {}
|
|
newest_dead: dict[str, str] = {}
|
|
oldest_dead: dict[str, str] = {}
|
|
|
|
for md in export_dir.rglob("*.md"):
|
|
date_match = DATE_RE.search(md.name)
|
|
if not date_match:
|
|
continue
|
|
date = date_match.group(1)
|
|
month = date[:7]
|
|
try:
|
|
text = md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
|
|
conv_key = f"{date} {md.stem}"
|
|
|
|
for source, _path in SAVED_RE.findall(text):
|
|
source = source or "unknown"
|
|
stats[month][source]["saved"] += 1
|
|
losses[conv_key]["saved"] += 1
|
|
if source not in oldest_saved or date < oldest_saved[source]:
|
|
oldest_saved[source] = date
|
|
|
|
for _ref, meta in DEAD_RE.findall(text):
|
|
source = meta.split(",")[0].strip() or "unknown"
|
|
stats[month][source]["dead"] += 1
|
|
losses[conv_key]["dead"] += 1
|
|
if source not in newest_dead or date > newest_dead[source]:
|
|
newest_dead[source] = date
|
|
if source not in oldest_dead or date < oldest_dead[source]:
|
|
oldest_dead[source] = date
|
|
|
|
if not stats:
|
|
print(f"no dated conversations with images found under {export_dir}")
|
|
return
|
|
|
|
sources = sorted({s for m in stats.values() for s in m})
|
|
|
|
print("=" * 78)
|
|
print("Image outcomes by conversation month")
|
|
print("=" * 78)
|
|
header = f"{'month':<9}"
|
|
for source in sources:
|
|
header += f" {source[:16]:>16} (saved/dead)"
|
|
print(header)
|
|
for month in sorted(stats):
|
|
row = f"{month:<9}"
|
|
for source in sources:
|
|
cell = stats[month].get(source, {"saved": 0, "dead": 0})
|
|
if cell["saved"] or cell["dead"]:
|
|
row += f" {cell['saved']:>8} / {cell['dead']:<17}"
|
|
else:
|
|
row += f" {'—':>8} {'':<17}"
|
|
print(row)
|
|
|
|
print()
|
|
print("=" * 78)
|
|
print("The age question")
|
|
print("=" * 78)
|
|
for source in sources:
|
|
old_s = oldest_saved.get(source)
|
|
old_d = oldest_dead.get(source)
|
|
new_d = newest_dead.get(source)
|
|
print(f" {source}:")
|
|
print(f" oldest still downloadable : {old_s or '—'}")
|
|
print(f" dead range : {old_d or '—'} … {new_d or '—'}")
|
|
if old_s and new_d and old_s < new_d:
|
|
print(
|
|
f" → an asset from {old_s} was captured while one from "
|
|
f"{new_d} was not."
|
|
)
|
|
print(
|
|
" This does NOT disprove expiry: 'captured' means some "
|
|
"run got it in time,"
|
|
)
|
|
print(
|
|
" not that it is still on ChatGPT today. Run "
|
|
"probe_survival_by_month.py to tell"
|
|
)
|
|
print(" those apart.")
|
|
elif old_s and old_d and old_s > old_d:
|
|
print(
|
|
f" → everything lost is older than everything captured "
|
|
f"(cutoff between {old_d} and {old_s}): consistent with expiry."
|
|
)
|
|
print()
|
|
|
|
print("=" * 78)
|
|
print("Conversations that lost images (clustering check)")
|
|
print("=" * 78)
|
|
lossy = {k: v for k, v in losses.items() if v["dead"]}
|
|
for conv, counts in sorted(lossy.items()):
|
|
print(f" {conv[:66]:<66} saved={counts['saved']:<4} dead={counts['dead']}")
|
|
total_dead = sum(v["dead"] for v in lossy.values())
|
|
print()
|
|
print(f" {len(lossy)} conversation(s) affected, {total_dead} image(s) lost")
|
|
mixed = [k for k, v in lossy.items() if v["saved"]]
|
|
if mixed:
|
|
print(
|
|
f" {len(mixed)} of them ALSO kept images — same conversation, same "
|
|
"age, different outcome:"
|
|
)
|
|
for conv in sorted(mixed):
|
|
print(f" {conv[:70]}")
|
|
print(" → whatever killed these is per-asset, not per-conversation.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|