tools: distinguish "captured in time" from "still alive"
analyze_media_age.py claimed age was ruled out because an image from
2025-11 was saved while one from 2026-08 was not. That conclusion does not
follow. "Saved" means some earlier run downloaded it, not that ChatGPT
still holds it — an image captured in June looks saved forever after, even
if it died in July. A month with no losses shows the exports were timely,
not that the assets survived.
- probe_survival_by_month.py: sample images already on disk, grouped by
their conversation's month, and ask /files/{id} whether each still
exists today. Old months still 200 → no expiry, and cadence did not save
them. Old months now 404 → uploads do expire and cadence is the whole
ballgame.
- analyze_media_age.py: stop asserting the unsupported verdict; say what
the number does and does not show, and point at the probe.
One signal there is immune to the confound and survives: 2026-07-09 kept
38 images and lost 12. Same conversation, same day, opposite outcomes —
no retention policy does that, so at least part of this is per-asset.
This commit is contained in:
@@ -20,6 +20,15 @@ grouping outcomes by month and by source distinguishes the hypotheses:
|
||||
* Neither → deaths scattered across months, or concentrated in a
|
||||
few conversations while their neighbours survive.
|
||||
|
||||
IMPORTANT — what "saved" does and does not mean. It means the image was
|
||||
downloaded by *some* run at *some* point, not that ChatGPT still holds it. An
|
||||
image captured in June looks saved forever after, even if it died in July. So
|
||||
a month with no losses shows the exports were timely; it cannot show the
|
||||
assets survived. Use probe_survival_by_month.py to ask the API what is still
|
||||
live today. The one signal here that is immune to this confound is a single
|
||||
conversation that both kept and lost images: same age, same chat, opposite
|
||||
outcomes means the cause is per-asset, not retention.
|
||||
|
||||
Run from the project root:
|
||||
|
||||
python tools/analyze_media_age.py
|
||||
@@ -123,12 +132,21 @@ def main() -> None:
|
||||
print(f" dead range : {old_d or '—'} … {new_d or '—'}")
|
||||
if old_s and new_d and old_s < new_d:
|
||||
print(
|
||||
f" → an asset from {old_s} survived while one from "
|
||||
f"{new_d} did not: age alone does not explain the loss."
|
||||
f" → an asset from {old_s} was captured while one from "
|
||||
f"{new_d} was not."
|
||||
)
|
||||
print(
|
||||
" This does NOT disprove expiry: 'captured' means some "
|
||||
"run got it in time,"
|
||||
)
|
||||
print(
|
||||
" not that it is still on ChatGPT today. Run "
|
||||
"probe_survival_by_month.py to tell"
|
||||
)
|
||||
print(" those apart.")
|
||||
elif old_s and old_d and old_s > old_d:
|
||||
print(
|
||||
f" → everything dead is older than everything alive "
|
||||
f" → everything lost is older than everything captured "
|
||||
f"(cutoff between {old_d} and {old_s}): consistent with expiry."
|
||||
)
|
||||
print()
|
||||
|
||||
@@ -0,0 +1,153 @@
|
||||
"""Are already-downloaded images still alive on ChatGPT, or did we just catch them in time?
|
||||
|
||||
analyze_media_age.py cannot answer this. It reports whether an image was ever
|
||||
captured — and an image downloaded by a run back in June looks "saved" forever
|
||||
after, whether or not ChatGPT still holds it today. So a month with zero losses
|
||||
proves the exports were timely, not that the assets survived.
|
||||
|
||||
This closes that gap: take images that ARE on disk, grouped by the month of
|
||||
their conversation, and ask the API whether each still exists right now.
|
||||
|
||||
Old months still 200 → assets do not expire with age. Export cadence is not
|
||||
what saved them, and shortening it would not have
|
||||
saved the July losses either.
|
||||
Old months now 404 → assets DO die with age; the old months are on disk
|
||||
only because a run reached them in time. Cadence is
|
||||
the whole ballgame, and the July losses are the first
|
||||
ones we simply arrived too late for.
|
||||
|
||||
Liveness is checked with /files/{id}, which answers a missing record with a
|
||||
clean 404 (verified 2026-08-17); /files/{id}/download reports the same state
|
||||
as a misleading 403.
|
||||
|
||||
Run from the project root:
|
||||
|
||||
python tools/probe_survival_by_month.py [samples_per_month]
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
||||
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
|
||||
|
||||
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
||||
DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_")
|
||||
|
||||
DEFAULT_SAMPLES = 4
|
||||
|
||||
|
||||
def collect_saved_ids(export_dir: Path) -> dict[str, list[tuple[str, str]]]:
|
||||
"""month → [(file_id, source)] for images already downloaded to disk."""
|
||||
by_month: dict[str, list[tuple[str, str]]] = defaultdict(list)
|
||||
for md in export_dir.rglob("*.md"):
|
||||
date_match = DATE_RE.search(md.name)
|
||||
if not date_match:
|
||||
continue
|
||||
month = date_match.group(1)[:7]
|
||||
try:
|
||||
text = md.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
continue
|
||||
for source, rel_path in SAVED_RE.findall(text):
|
||||
file_id = Path(rel_path).stem
|
||||
if file_id.startswith("file"):
|
||||
by_month[month].append((file_id, source or "unknown"))
|
||||
return by_month
|
||||
|
||||
|
||||
def main() -> None:
|
||||
samples = DEFAULT_SAMPLES
|
||||
if len(sys.argv) > 1:
|
||||
try:
|
||||
samples = max(1, int(sys.argv[1]))
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
||||
by_month = collect_saved_ids(export_dir)
|
||||
if not by_month:
|
||||
print(f"no downloaded images found under {export_dir}")
|
||||
return
|
||||
|
||||
provider = ChatGPTProvider()
|
||||
|
||||
print("=" * 78)
|
||||
print(f"Are on-disk images still live on ChatGPT? ({samples} sampled per month)")
|
||||
print("=" * 78)
|
||||
print(f"{'month':<9} {'alive':>6} {'gone':>6} {'err':>5} sampled sources")
|
||||
|
||||
totals = {"alive": 0, "gone": 0, "err": 0}
|
||||
verdict_rows: list[tuple[str, int, int]] = []
|
||||
|
||||
for month in sorted(by_month):
|
||||
entries = by_month[month]
|
||||
# Spread the sample across the month rather than taking the first few
|
||||
# from one conversation.
|
||||
step = max(1, len(entries) // samples)
|
||||
picked = entries[::step][:samples]
|
||||
|
||||
alive = gone = err = 0
|
||||
sources: list[str] = []
|
||||
for file_id, source in picked:
|
||||
sources.append(source[:4])
|
||||
try:
|
||||
provider._pace()
|
||||
resp = provider._session.request(
|
||||
"GET", f"{BASE_URL}/files/{file_id}", timeout=30
|
||||
)
|
||||
except Exception: # noqa: BLE001 - diagnostic
|
||||
err += 1
|
||||
continue
|
||||
if resp.status_code == 200:
|
||||
alive += 1
|
||||
elif resp.status_code == 404:
|
||||
gone += 1
|
||||
else:
|
||||
err += 1
|
||||
|
||||
totals["alive"] += alive
|
||||
totals["gone"] += gone
|
||||
totals["err"] += err
|
||||
verdict_rows.append((month, alive, gone))
|
||||
print(
|
||||
f"{month:<9} {alive:>6} {gone:>6} {err:>5} "
|
||||
f"{','.join(sources)} (of {len(entries)} on disk)"
|
||||
)
|
||||
|
||||
print()
|
||||
print(f"total: {totals['alive']} alive, {totals['gone']} gone, {totals['err']} error")
|
||||
print()
|
||||
print("=" * 78)
|
||||
print("Verdict")
|
||||
print("=" * 78)
|
||||
|
||||
old_gone = [m for m, _a, g in verdict_rows[:-2] if g]
|
||||
old_alive = [m for m, a, _g in verdict_rows[:-2] if a]
|
||||
|
||||
if old_gone and not old_alive:
|
||||
print(" Every sampled older image is GONE from ChatGPT.")
|
||||
print(" → Uploads expire. Your old exports survive only because a run")
|
||||
print(" reached them in time. Export cadence directly determines")
|
||||
print(" what you can still save.")
|
||||
elif old_alive and not old_gone:
|
||||
print(" Every sampled older image is STILL LIVE on ChatGPT.")
|
||||
print(" → Uploads do not expire with age. The July losses are")
|
||||
print(" something else, and exporting sooner would not have")
|
||||
print(" prevented them.")
|
||||
else:
|
||||
print(" Mixed: some older images alive, some gone.")
|
||||
print(" → Not a clean expiry. Compare the 'sampled sources' column and")
|
||||
print(" the per-month rates above; the cause is likely per-asset.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user