"Do I have to export within N days?" is answerable from the exports already on disk — the renderer records every image's outcome inline ( when saved, a placeholder when not) and the conversation date is in the filename. Group by month and source and the hypotheses separate: a clean old/new cutoff means expiry, user_upload dying at an age model_generated survives means the source matters, and losses scattered through months that otherwise downloaded fine means neither. Offline, no token, no API calls.
158 lines
5.6 KiB
Python
158 lines
5.6 KiB
Python
"""Offline: is the media loss age-based, source-based, or neither?
|
|
|
|
Answers the question the 403s raised — do I have to export within N days? —
|
|
from the exports already on disk. No API calls, no token, nothing to expire.
|
|
|
|
It works because the renderer records the outcome of every image in the
|
|
Markdown itself:
|
|
|
|
 ← downloaded, still alive
|
|
> 🖼️ **Image attached** — `sediment://file_y`
|
|
(user_upload, content not preserved…) ← dead or never fetched
|
|
|
|
and the conversation's date is in its filename (YYYY-MM-DD_slug_id.md). So
|
|
grouping outcomes by month and by source distinguishes the hypotheses:
|
|
|
|
* Age-based expiry → old months all-dead, recent months all-alive, with a
|
|
clean cutoff between them.
|
|
* Source-based → user_upload dies while model_generated survives at
|
|
the same age.
|
|
* Neither → deaths scattered across months, or concentrated in a
|
|
few conversations while their neighbours survive.
|
|
|
|
Run from the project root:
|
|
|
|
python tools/analyze_media_age.py
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
|
DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_")
|
|
|
|
|
|
def main() -> None:
|
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
|
if not export_dir.is_dir():
|
|
print(f"exports dir not found: {export_dir}")
|
|
return
|
|
|
|
# month → source → {"saved": n, "dead": n}
|
|
stats: dict[str, dict[str, dict[str, int]]] = defaultdict(
|
|
lambda: defaultdict(lambda: {"saved": 0, "dead": 0})
|
|
)
|
|
# conversations that lost at least one image
|
|
losses: dict[str, dict[str, int]] = defaultdict(lambda: {"saved": 0, "dead": 0})
|
|
oldest_saved: dict[str, str] = {}
|
|
newest_dead: dict[str, str] = {}
|
|
oldest_dead: dict[str, str] = {}
|
|
|
|
for md in export_dir.rglob("*.md"):
|
|
date_match = DATE_RE.search(md.name)
|
|
if not date_match:
|
|
continue
|
|
date = date_match.group(1)
|
|
month = date[:7]
|
|
try:
|
|
text = md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
|
|
conv_key = f"{date} {md.stem}"
|
|
|
|
for source, _path in SAVED_RE.findall(text):
|
|
source = source or "unknown"
|
|
stats[month][source]["saved"] += 1
|
|
losses[conv_key]["saved"] += 1
|
|
if source not in oldest_saved or date < oldest_saved[source]:
|
|
oldest_saved[source] = date
|
|
|
|
for _ref, meta in DEAD_RE.findall(text):
|
|
source = meta.split(",")[0].strip() or "unknown"
|
|
stats[month][source]["dead"] += 1
|
|
losses[conv_key]["dead"] += 1
|
|
if source not in newest_dead or date > newest_dead[source]:
|
|
newest_dead[source] = date
|
|
if source not in oldest_dead or date < oldest_dead[source]:
|
|
oldest_dead[source] = date
|
|
|
|
if not stats:
|
|
print(f"no dated conversations with images found under {export_dir}")
|
|
return
|
|
|
|
sources = sorted({s for m in stats.values() for s in m})
|
|
|
|
print("=" * 78)
|
|
print("Image outcomes by conversation month")
|
|
print("=" * 78)
|
|
header = f"{'month':<9}"
|
|
for source in sources:
|
|
header += f" {source[:16]:>16} (saved/dead)"
|
|
print(header)
|
|
for month in sorted(stats):
|
|
row = f"{month:<9}"
|
|
for source in sources:
|
|
cell = stats[month].get(source, {"saved": 0, "dead": 0})
|
|
if cell["saved"] or cell["dead"]:
|
|
row += f" {cell['saved']:>8} / {cell['dead']:<17}"
|
|
else:
|
|
row += f" {'—':>8} {'':<17}"
|
|
print(row)
|
|
|
|
print()
|
|
print("=" * 78)
|
|
print("The age question")
|
|
print("=" * 78)
|
|
for source in sources:
|
|
old_s = oldest_saved.get(source)
|
|
old_d = oldest_dead.get(source)
|
|
new_d = newest_dead.get(source)
|
|
print(f" {source}:")
|
|
print(f" oldest still downloadable : {old_s or '—'}")
|
|
print(f" dead range : {old_d or '—'} … {new_d or '—'}")
|
|
if old_s and new_d and old_s < new_d:
|
|
print(
|
|
f" → an asset from {old_s} survived while one from "
|
|
f"{new_d} did not: age alone does not explain the loss."
|
|
)
|
|
elif old_s and old_d and old_s > old_d:
|
|
print(
|
|
f" → everything dead is older than everything alive "
|
|
f"(cutoff between {old_d} and {old_s}): consistent with expiry."
|
|
)
|
|
print()
|
|
|
|
print("=" * 78)
|
|
print("Conversations that lost images (clustering check)")
|
|
print("=" * 78)
|
|
lossy = {k: v for k, v in losses.items() if v["dead"]}
|
|
for conv, counts in sorted(lossy.items()):
|
|
print(f" {conv[:66]:<66} saved={counts['saved']:<4} dead={counts['dead']}")
|
|
total_dead = sum(v["dead"] for v in lossy.values())
|
|
print()
|
|
print(f" {len(lossy)} conversation(s) affected, {total_dead} image(s) lost")
|
|
mixed = [k for k, v in lossy.items() if v["saved"]]
|
|
if mixed:
|
|
print(
|
|
f" {len(mixed)} of them ALSO kept images — same conversation, same "
|
|
"age, different outcome:"
|
|
)
|
|
for conv in sorted(mixed):
|
|
print(f" {conv[:70]}")
|
|
print(" → whatever killed these is per-asset, not per-conversation.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|