canary: detect provider API schema drift; release v0.7.0

The export reads ChatGPT's and Claude's undocumented internal web APIs,
which can change shape without notice; the worst failure for a backup tool
is a silent one (skipped/mis-parsed content with no error). Add a `canary`
command + BaseProvider.check_drift() (overridden by ChatGPT/Claude) that
fetches one listing page + one conversation per provider and asserts only
the normalizer's load-bearing fields — not the full response shape, which
churns harmlessly. The top silent risk it guards is a renamed retrieval-tool
author bypassing the hidden-content collapse. ERROR findings exit non-zero;
WARN findings are surfaced but non-fatal so a backup run is never blocked.

Also retires the in-app watch mode and headless StartOS direction from the
roadmap (tool stays a local, manually-run CLI) and updates docs.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
JesseMarkowitz
2026-06-28 01:37:24 -04:00
co-authored by Claude Opus 4.8
parent ef603cf659
commit 1e016ea652
10 changed files with 693 additions and 80 deletions
+96
View File
@@ -45,12 +45,16 @@ from src.blocks import (
from src.loss_report import LossReport
from src.providers.base import (
BaseProvider,
DRIFT_ERROR,
DRIFT_OK,
DRIFT_WARN,
HIDDEN_CONTENT_FULL,
HIDDEN_CONTENT_OMIT,
HIDDEN_CONTENT_PLACEHOLDER,
ProviderError,
REQUEST_TIMEOUT,
VALID_HIDDEN_CONTENT_POLICIES,
drift_finding,
resolve_hidden_content_policy,
)
@@ -91,6 +95,20 @@ def parse_asset_file_id(ref: str) -> str | None:
# myfiles_browser is the legacy name for the same retrieval tool.
_COLLAPSE_TOOL_AUTHORS = {"file_search", "myfiles_browser"}
# ── API-drift canary vocabularies (FUTURE.md §10) ──────────────────────────
# Tool authors seen live 2026-06-28. The collapse keys off author.name, so a
# RENAMED retrieval tool would silently bypass the collapse and re-bloat the
# archive — the canary flags any unfamiliar (role="tool", author.name) pair.
_KNOWN_TOOL_AUTHORS = _COLLAPSE_TOOL_AUTHORS | {"web.run", "python"}
# content_types the normalizer dispatches on (keep in sync with
# _extract_blocks_for_content). A new value already degrades to an `unknown`
# block + LossReport at export time; the canary flags it proactively.
_HANDLED_CONTENT_TYPES = frozenset({
"text", "multimodal_text", "execution_output", "system_error",
"tether_browsing_display", "code", "thoughts", "reasoning_recap",
"user_editable_context", "model_editable_context", "image_asset_pointer",
})
class ChatGPTProvider(BaseProvider):
"""Provider for ChatGPT conversations via the internal web API.
@@ -766,6 +784,84 @@ class ChatGPTProvider(BaseProvider):
"messages": messages,
}
def check_drift(self) -> list[dict]:
"""Probe one listing page + one conversation; assert load-bearing fields.
See FUTURE.md §10. Asserts only the fields the normalizer depends on —
not the full response shape (which churns harmlessly). The top silent
risk is a renamed retrieval tool author bypassing the collapse.
"""
findings: list[dict] = []
page = self.list_conversations(offset=0, limit=5)
if not page:
return [drift_finding(DRIFT_WARN, "listing",
"empty listing — cannot verify shape")]
item = page[0]
for f in ("id", "title"):
if f not in item:
findings.append(drift_finding(
DRIFT_ERROR, "listing", f"summary item missing '{f}'"))
if not (item.get("update_time") or item.get("create_time")):
findings.append(drift_finding(
DRIFT_WARN, "listing", "no update_time/create_time on summary"))
conv_id = item.get("id")
if not conv_id:
return findings or [drift_finding(DRIFT_ERROR, "listing", "no id to fetch")]
raw = self.get_conversation(conv_id)
if not (raw.get("conversation_id") or raw.get("id")):
findings.append(drift_finding(
DRIFT_ERROR, "detail", "no conversation_id/id on detail"))
mapping = raw.get("mapping")
if not isinstance(mapping, dict) or not mapping:
findings.append(drift_finding(
DRIFT_ERROR, "detail", "mapping missing/empty — tree walk will yield 0 messages"))
return findings
saw_text_with_parts = False
seen_new_ct: set[str] = set()
seen_new_tool: set[str] = set()
for node in mapping.values():
if "children" not in node:
findings.append(drift_finding(
DRIFT_ERROR, "mapping", "node missing 'children' — tree walk breaks"))
break
msg = node.get("message")
if not msg:
continue
author = msg.get("author") or {}
role, name = author.get("role"), author.get("name")
content = msg.get("content") or {}
ct = content.get("content_type")
if ct and ct not in _HANDLED_CONTENT_TYPES:
seen_new_ct.add(ct)
if role == "tool" and name and name not in _KNOWN_TOOL_AUTHORS:
seen_new_tool.add(name)
if ct == "text":
parts = content.get("parts") or []
if any(isinstance(p, str) and p.strip() for p in parts):
saw_text_with_parts = True
for name in sorted(seen_new_tool):
findings.append(drift_finding(
DRIFT_WARN, "collapse",
f"unfamiliar tool author {name!r} — if it's a retrieval dump it "
"is NOT being collapsed (silent archive bloat); update "
"_COLLAPSE_TOOL_AUTHORS / _KNOWN_TOOL_AUTHORS"))
for ct in sorted(seen_new_ct):
findings.append(drift_finding(
DRIFT_WARN, "content_type",
f"new content_type {ct!r} — exported as an unknown block; add a handler"))
if not saw_text_with_parts:
findings.append(drift_finding(
DRIFT_WARN, "parts",
"no text message yielded non-empty parts — content.parts may have drifted"))
if not findings:
findings.append(drift_finding(DRIFT_OK, "schema", "all load-bearing fields present"))
return findings
# ---------------------------------------------------------------------------
# Internal helpers