tools: analyze whether media loss is age-based

"Do I have to export within N days?" is answerable from the exports
already on disk — the renderer records every image's outcome inline
(![source](media/…) when saved, a placeholder when not) and the
conversation date is in the filename. Group by month and source and the
hypotheses separate: a clean old/new cutoff means expiry, user_upload
dying at an age model_generated survives means the source matters, and
losses scattered through months that otherwise downloaded fine means
neither.

Offline, no token, no API calls.
This commit is contained in:
JesseMarkowitz
2026-08-17 08:45:21 -04:00
parent f40b25001a
commit 04191eed8c
6 changed files with 326 additions and 215 deletions
+157
View File
@@ -0,0 +1,157 @@
"""Offline: is the media loss age-based, source-based, or neither?
Answers the question the 403s raised — do I have to export within N days? —
from the exports already on disk. No API calls, no token, nothing to expire.
It works because the renderer records the outcome of every image in the
Markdown itself:
![user_upload](media/file_x.png) ← downloaded, still alive
> 🖼️ **Image attached** — `sediment://file_y`
(user_upload, content not preserved…) ← dead or never fetched
and the conversation's date is in its filename (YYYY-MM-DD_slug_id.md). So
grouping outcomes by month and by source distinguishes the hypotheses:
* Age-based expiry → old months all-dead, recent months all-alive, with a
clean cutoff between them.
* Source-based → user_upload dies while model_generated survives at
the same age.
* Neither → deaths scattered across months, or concentrated in a
few conversations while their neighbours survive.
Run from the project root:
python tools/analyze_media_age.py
"""
import os
import re
import sys
from collections import defaultdict
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from dotenv import load_dotenv
load_dotenv()
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
DATE_RE = re.compile(r"(\d{4}-\d{2}-\d{2})_")
def main() -> None:
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
if not export_dir.is_dir():
print(f"exports dir not found: {export_dir}")
return
# month → source → {"saved": n, "dead": n}
stats: dict[str, dict[str, dict[str, int]]] = defaultdict(
lambda: defaultdict(lambda: {"saved": 0, "dead": 0})
)
# conversations that lost at least one image
losses: dict[str, dict[str, int]] = defaultdict(lambda: {"saved": 0, "dead": 0})
oldest_saved: dict[str, str] = {}
newest_dead: dict[str, str] = {}
oldest_dead: dict[str, str] = {}
for md in export_dir.rglob("*.md"):
date_match = DATE_RE.search(md.name)
if not date_match:
continue
date = date_match.group(1)
month = date[:7]
try:
text = md.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
conv_key = f"{date} {md.stem}"
for source, _path in SAVED_RE.findall(text):
source = source or "unknown"
stats[month][source]["saved"] += 1
losses[conv_key]["saved"] += 1
if source not in oldest_saved or date < oldest_saved[source]:
oldest_saved[source] = date
for _ref, meta in DEAD_RE.findall(text):
source = meta.split(",")[0].strip() or "unknown"
stats[month][source]["dead"] += 1
losses[conv_key]["dead"] += 1
if source not in newest_dead or date > newest_dead[source]:
newest_dead[source] = date
if source not in oldest_dead or date < oldest_dead[source]:
oldest_dead[source] = date
if not stats:
print(f"no dated conversations with images found under {export_dir}")
return
sources = sorted({s for m in stats.values() for s in m})
print("=" * 78)
print("Image outcomes by conversation month")
print("=" * 78)
header = f"{'month':<9}"
for source in sources:
header += f" {source[:16]:>16} (saved/dead)"
print(header)
for month in sorted(stats):
row = f"{month:<9}"
for source in sources:
cell = stats[month].get(source, {"saved": 0, "dead": 0})
if cell["saved"] or cell["dead"]:
row += f" {cell['saved']:>8} / {cell['dead']:<17}"
else:
row += f" {'—':>8} {'':<17}"
print(row)
print()
print("=" * 78)
print("The age question")
print("=" * 78)
for source in sources:
old_s = oldest_saved.get(source)
old_d = oldest_dead.get(source)
new_d = newest_dead.get(source)
print(f" {source}:")
print(f" oldest still downloadable : {old_s or '—'}")
print(f" dead range : {old_d or '—'} … {new_d or '—'}")
if old_s and new_d and old_s < new_d:
print(
f" → an asset from {old_s} survived while one from "
f"{new_d} did not: age alone does not explain the loss."
)
elif old_s and old_d and old_s > old_d:
print(
f" → everything dead is older than everything alive "
f"(cutoff between {old_d} and {old_s}): consistent with expiry."
)
print()
print("=" * 78)
print("Conversations that lost images (clustering check)")
print("=" * 78)
lossy = {k: v for k, v in losses.items() if v["dead"]}
for conv, counts in sorted(lossy.items()):
print(f" {conv[:66]:<66} saved={counts['saved']:<4} dead={counts['dead']}")
total_dead = sum(v["dead"] for v in lossy.values())
print()
print(f" {len(lossy)} conversation(s) affected, {total_dead} image(s) lost")
mixed = [k for k, v in lossy.items() if v["saved"]]
if mixed:
print(
f" {len(mixed)} of them ALSO kept images — same conversation, same "
"age, different outcome:"
)
for conv in sorted(mixed):
print(f" {conv[:70]}")
print(" → whatever killed these is per-asset, not per-conversation.")
if __name__ == "__main__":
main()
-213
View File
@@ -1,213 +0,0 @@
"""One-off diagnostic: why do some ChatGPT media assets 403?
The API answers a refused download with a bare {"detail":"Forbidden"}, so the
cause has to be narrowed by experiment. This script runs the experiments that
distinguish the plausible causes and prints a table.
Hypothesis A — auth is fine, the asset is the problem.
Probe: fetch an asset that DID download alongside one that 403'd. If the
good one still 200s in the same session, the session is not at fault.
Hypothesis B — the asset is scoped to a workspace/account we don't name.
The exporter never sends ChatGPT-Account-Id. Resources belonging to a
workspace (vs the personal account) can require it.
Probe: retry the 403 with each account id the session reports.
Hypothesis C — only the signed-URL step is restricted.
Probe: hit /files/{id} (metadata, no /download) and see if that differs.
Also, with no API call at all: what KIND of asset are the failures? The
exported placeholders record source=user_upload | model_generated, so
scanning exports/ classifies the failures for free.
Run from the project root with the venv active:
python tools/probe_media_403.py
Delete this file once the cause is known.
"""
import os
import re
import sys
from collections import Counter
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from dotenv import load_dotenv
load_dotenv()
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
# The IDs that 403'd in the 2026-08-17 runs.
FAILED_IDS = [
"file_00000000c18471f6a9b8f726c599b15d",
"file_00000000667071f694d8aa00b4d42c7b",
"file_00000000a53c71f6ba82c859cdfc9659",
"file_00000000dccc71f69472e8526e6c7e0c",
"file_00000000f34071f6af231eaf1c655f6d",
"file_00000000d5e471f6bbbac75623c0cc3d",
"file_000000003454722f9481506b96aed510",
"file_00000000f550722f990b7c3227f2cb59",
"file_0000000094c8722fa508cfe647750137",
"file_000000000000722f8c9757d63f45679b",
"file_000000007a4081f58297ab692fb625e5",
"file_000000002f5081f5bc459887eb6f494f",
"file_000000001b3071f59e30d3213f8fcca5",
"file_000000009478822faec2289717d9f628",
"file_00000000f0bc81f5933009bec614088c",
"file_00000000902881f59dca2333091a59cc",
"file_0000000017e0820cb269596ec8a79497",
"file_00000000762c81f5ac85d9fdcae20061",
]
ACCOUNT_ENDPOINTS = [
f"{BASE_URL}/accounts/check/v4-2023-04-27",
f"{BASE_URL}/accounts/check",
]
# "> 🖼️ **Image attached** — `sediment://file_x` (model_generated, image/png, …)"
PLACEHOLDER_RE = re.compile(
r"\*\*(?:Image|File) attached\*\* — `([^`]+)`\s*\(([^)]*)\)"
)
def classify_from_exports(export_dir: Path) -> None:
"""Offline: what kind of asset were the failures, and where did they live?"""
print("=" * 78)
print("A. What the exports already say (no API calls)")
print("=" * 78)
if not export_dir.is_dir():
print(f" exports dir not found: {export_dir} — skipping\n")
return
wanted = set(FAILED_IDS)
hits: dict[str, list[tuple[str, str]]] = {}
all_sources: Counter = Counter()
failed_sources: Counter = Counter()
for md in export_dir.rglob("*.md"):
try:
text = md.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
for ref, meta in PLACEHOLDER_RE.findall(text):
source = meta.split(",")[0].strip()
all_sources[source] += 1
for fid in wanted:
if fid in ref:
hits.setdefault(fid, []).append((source, str(md.relative_to(export_dir))))
failed_sources[source] += 1
print(f" placeholders still unresolved across exports: {sum(all_sources.values())}")
print(f" by source: {dict(all_sources)}")
print(f" of those, matching a known 403 ID: {sum(failed_sources.values())}")
print(f" by source: {dict(failed_sources)}")
print()
for fid, places in sorted(hits.items()):
source, path = places[0]
print(f" {fid[:28]}… source={source:<16} {path}")
if not hits:
print(" (no matches — exports may live elsewhere; set EXPORT_DIR)")
print()
def find_good_ids(export_dir: Path, limit: int = 3) -> list[str]:
"""IDs that downloaded successfully — media/ files are named by file ID."""
good = []
for media_file in export_dir.rglob("media/*"):
if media_file.is_file() and media_file.stem.startswith("file"):
good.append(media_file.stem)
if len(good) >= limit:
break
return good
def get_account_ids(provider) -> list[str]:
print("=" * 78)
print("B. Account / workspace IDs this session reports")
print("=" * 78)
ids: list[str] = []
for url in ACCOUNT_ENDPOINTS:
try:
resp = provider._session.request("GET", url, timeout=30)
except Exception as e: # noqa: BLE001 - diagnostic
print(f" {url} → error {e}")
continue
print(f" {url} → {resp.status_code}")
if resp.status_code != 200:
continue
try:
data = resp.json()
except Exception: # noqa: BLE001 - diagnostic
continue
accounts = data.get("accounts") if isinstance(data, dict) else None
if isinstance(accounts, dict):
for key, value in accounts.items():
acct = (value or {}).get("account", {}) if isinstance(value, dict) else {}
acct_id = acct.get("account_id")
plan = acct.get("structure") or acct.get("plan_type")
if acct_id:
ids.append(acct_id)
print(f" key={key!r:<28} account_id={acct_id} ({plan})")
if ids:
break
if not ids:
print(" (none discovered — hypothesis B untestable)")
print()
return ids
def probe(provider, file_id: str, account_ids: list[str]) -> None:
def call(url: str, headers: dict | None = None) -> str:
try:
resp = provider._session.request("GET", url, headers=headers, timeout=30)
except Exception as e: # noqa: BLE001 - diagnostic
return f"error {type(e).__name__}"
detail = ""
if resp.status_code != 200:
try:
detail = f" {resp.json().get('detail', '')}"
except Exception: # noqa: BLE001 - diagnostic
detail = f" {resp.text[:60]}"
return f"{resp.status_code}{detail}"
dl = f"{BASE_URL}/files/{file_id}/download"
meta = f"{BASE_URL}/files/{file_id}"
print(f" {file_id}")
print(f" /download → {call(dl)}")
print(f" /files/{{id}} (metadata) → {call(meta)}")
for acct in account_ids:
header = {"ChatGPT-Account-Id": acct}
print(f" /download + Account-Id → {call(dl, header)} [{acct[:8]}…]")
def main() -> None:
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
classify_from_exports(export_dir)
provider = ChatGPTProvider()
account_ids = get_account_ids(provider)
print("=" * 78)
print("C. Live probes")
print("=" * 78)
good_ids = find_good_ids(export_dir)
print("-- assets that downloaded fine (control group) --")
if not good_ids:
print(" (none found under exports/**/media — control group unavailable)")
for fid in good_ids:
probe(provider, fid, account_ids)
print()
print("-- assets that 403'd --")
for fid in FAILED_IDS[:4]:
probe(provider, fid, account_ids)
if __name__ == "__main__":
main()