Expiry is now ruled out: 36/36 sampled images from 2025-09 through 2026-08 are still live server-side, so nothing dies of age and export cadence is not the variable. The loss is per-asset — 2026-07-09 kept 38 images and lost 12 in one conversation. Next hypothesis: those images belong to messages that were edited or regenerated. ChatGPT stores a conversation as a tree and editing forks it, stranding the superseded messages on an abandoned branch. The exporter walks every node (chatgpt.py:963), so it exports those branches too — and an attachment unreachable from live history is a natural GC target. branch_check.py fetches the raw conversation, derives the live branch by walking parent links up from current_node, and cross-tabulates every image against (on the live branch?) x (still downloadable?). If the lost images are all off-branch and nothing on the live branch is missing, this is not data loss at all — it is attachments to messages that were replaced.
203 lines
7.5 KiB
Python
203 lines
7.5 KiB
Python
"""Do the lost images sit on abandoned conversation branches?
|
|
|
|
Established so far: uploads do not expire (36/36 sampled images from
|
|
2025-09 through 2026-08 are still live), and the loss is per-asset — one
|
|
conversation on 2026-07-09 kept 38 images and lost 12. So something
|
|
distinguishes those 12 from their neighbours in the same chat.
|
|
|
|
Hypothesis: they are attached to messages that were edited or regenerated.
|
|
ChatGPT keeps the whole conversation as a tree; editing a message forks it,
|
|
leaving the superseded messages on an abandoned branch. The exporter walks
|
|
every node (chatgpt.py:963), so it exports abandoned branches too — and an
|
|
attachment that is no longer reachable from the live conversation is exactly
|
|
the kind of thing a backend would garbage-collect.
|
|
|
|
The test: fetch the raw conversation, compute the live branch by walking
|
|
parent links up from ``current_node``, then cross-tabulate every image
|
|
against (on the live branch?) x (still downloadable?).
|
|
|
|
on-branch alive + off-branch dead → confirmed, and it is not data loss:
|
|
those images belong to messages you
|
|
replaced.
|
|
dead on the live branch → hypothesis dead; the images are
|
|
genuinely missing from live history.
|
|
|
|
Run from the project root:
|
|
|
|
python tools/branch_check.py
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
from src.providers.chatgpt import BASE_URL, ChatGPTProvider # noqa: E402
|
|
|
|
SAVED_RE = re.compile(r"!\[([^\]]*)\]\((media/[^)]+)\)")
|
|
DEAD_RE = re.compile(r"🖼️ \*\*Image attached\*\* — `([^`]+)`\s*\(([^)]*)\)")
|
|
CONV_ID_RE = re.compile(r"^conversation_id:\s*(\S+)", re.MULTILINE)
|
|
|
|
|
|
def find_mixed_conversations(export_dir: Path) -> list[tuple[Path, str, int, int]]:
|
|
"""Conversations that both kept and lost images — the informative ones."""
|
|
out = []
|
|
for md in export_dir.rglob("*.md"):
|
|
try:
|
|
text = md.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
saved = len(SAVED_RE.findall(text))
|
|
dead = len(DEAD_RE.findall(text))
|
|
if not dead:
|
|
continue
|
|
match = CONV_ID_RE.search(text)
|
|
if not match:
|
|
continue
|
|
out.append((md, match.group(1), saved, dead))
|
|
return out
|
|
|
|
|
|
def live_branch(mapping: dict, current_node: str | None) -> set[str]:
|
|
"""Node IDs reachable by walking parent links up from current_node."""
|
|
live: set[str] = set()
|
|
node_id = current_node
|
|
while node_id and node_id in mapping and node_id not in live:
|
|
live.add(node_id)
|
|
node_id = mapping[node_id].get("parent")
|
|
return live
|
|
|
|
|
|
def image_refs(node: dict) -> list[str]:
|
|
"""asset_pointer refs on this node's message, if any."""
|
|
message = node.get("message") or {}
|
|
content = message.get("content") or {}
|
|
refs = []
|
|
if content.get("content_type") == "image_asset_pointer":
|
|
ref = content.get("asset_pointer")
|
|
if ref:
|
|
refs.append(ref)
|
|
for part in content.get("parts") or []:
|
|
if isinstance(part, dict) and part.get("content_type") == "image_asset_pointer":
|
|
ref = part.get("asset_pointer")
|
|
if ref:
|
|
refs.append(ref)
|
|
return refs
|
|
|
|
|
|
def still_live(provider, file_id: str) -> bool | None:
|
|
try:
|
|
provider._pace()
|
|
resp = provider._session.request(
|
|
"GET", f"{BASE_URL}/files/{file_id}", timeout=30
|
|
)
|
|
except Exception: # noqa: BLE001 - diagnostic
|
|
return None
|
|
if resp.status_code == 200:
|
|
return True
|
|
if resp.status_code == 404:
|
|
return False
|
|
return None
|
|
|
|
|
|
def main() -> None:
|
|
from src.providers.chatgpt import parse_asset_file_id
|
|
|
|
export_dir = Path(os.getenv("EXPORT_DIR", "./exports")).expanduser()
|
|
mixed = find_mixed_conversations(export_dir)
|
|
if not mixed:
|
|
print(f"no conversations with lost images found under {export_dir}")
|
|
return
|
|
|
|
provider = ChatGPTProvider()
|
|
grand = {"on_alive": 0, "on_dead": 0, "off_alive": 0, "off_dead": 0, "unknown": 0}
|
|
|
|
for md, conv_id, saved, dead in sorted(mixed, key=lambda r: -r[3]):
|
|
print("=" * 78)
|
|
print(f"{md.name}")
|
|
print(f" {saved} kept, {dead} lost conversation_id={conv_id}")
|
|
print("=" * 78)
|
|
|
|
try:
|
|
raw = provider.get_conversation(conv_id)
|
|
except Exception as e: # noqa: BLE001 - diagnostic
|
|
print(f" could not fetch: {e}\n")
|
|
continue
|
|
|
|
mapping = raw.get("mapping") or {}
|
|
current = raw.get("current_node")
|
|
live_nodes = live_branch(mapping, current)
|
|
print(f" mapping nodes: {len(mapping)} live branch: {len(live_nodes)}")
|
|
if len(live_nodes) < len(mapping):
|
|
print(
|
|
f" → {len(mapping) - len(live_nodes)} node(s) are OFF the live "
|
|
"branch (edited or regenerated messages)"
|
|
)
|
|
else:
|
|
print(" → conversation is linear; nothing was edited or regenerated")
|
|
print()
|
|
|
|
rows = []
|
|
for node_id, node in mapping.items():
|
|
on_branch = node_id in live_nodes
|
|
for ref in image_refs(node):
|
|
file_id = parse_asset_file_id(ref) or ref
|
|
alive = still_live(provider, file_id)
|
|
rows.append((on_branch, alive, file_id))
|
|
if alive is None:
|
|
grand["unknown"] += 1
|
|
else:
|
|
key = ("on_" if on_branch else "off_") + ("alive" if alive else "dead")
|
|
grand[key] += 1
|
|
|
|
counts = {"on_alive": 0, "on_dead": 0, "off_alive": 0, "off_dead": 0, "unk": 0}
|
|
for on_branch, alive, _fid in rows:
|
|
if alive is None:
|
|
counts["unk"] += 1
|
|
else:
|
|
counts[("on_" if on_branch else "off_") + ("alive" if alive else "dead")] += 1
|
|
|
|
print(f" {'':<18}{'still live':>12}{'gone':>8}")
|
|
print(f" {'on live branch':<18}{counts['on_alive']:>12}{counts['on_dead']:>8}")
|
|
print(f" {'off (abandoned)':<18}{counts['off_alive']:>12}{counts['off_dead']:>8}")
|
|
if counts["unk"]:
|
|
print(f" ({counts['unk']} indeterminate)")
|
|
print()
|
|
|
|
for on_branch, alive, fid in sorted(rows, key=lambda r: (r[0], r[1] is not False)):
|
|
if alive is False:
|
|
where = "live branch" if on_branch else "ABANDONED branch"
|
|
print(f" LOST {fid[:40]:<42} {where}")
|
|
print()
|
|
|
|
print("=" * 78)
|
|
print("Verdict")
|
|
print("=" * 78)
|
|
print(
|
|
f" on live branch : {grand['on_alive']} live, {grand['on_dead']} gone"
|
|
)
|
|
print(
|
|
f" abandoned : {grand['off_alive']} live, {grand['off_dead']} gone"
|
|
)
|
|
if grand["off_dead"] and not grand["on_dead"]:
|
|
print()
|
|
print(" Every lost image is on an abandoned branch, and nothing on the")
|
|
print(" live conversation is missing. These are attachments to messages")
|
|
print(" you edited or regenerated — ChatGPT collects them once they are")
|
|
print(" no longer part of the conversation. Your live history is intact,")
|
|
print(" and no export schedule would have changed this.")
|
|
elif grand["on_dead"]:
|
|
print()
|
|
print(" Images are missing from the LIVE conversation — not explained by")
|
|
print(" editing. Real loss from current history; worth digging further.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|