Files
AIChatExporter/tests/test_cli.py
T

627 lines
23 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""CLI-level tests using Click's CliRunner — no live API calls required."""
from pathlib import Path
import pytest
from click.testing import CliRunner
from src.cache import Cache
from src.main import _filter_by_project, cli
# ---------------------------------------------------------------------------
# _filter_by_project (T-27)
# ---------------------------------------------------------------------------
class TestFilterByProject:
"""Unit tests for the project filter logic used by export/list/joplin."""
# ChatGPT conversations use the _project_name annotation key
def _chatgpt(self, conv_id, project_name):
return {"id": conv_id, "_project_name": project_name}
# Claude conversations use the project dict key
def _claude(self, conv_id, project_name):
proj = {"name": project_name} if project_name else None
return {"id": conv_id, "project": proj}
def test_none_filter_keeps_no_project_chatgpt(self):
convs = [self._chatgpt("a", None), self._chatgpt("b", "Python Course")]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_none_filter_keeps_no_project_claude(self):
convs = [self._claude("a", None), self._claude("b", "Python Course")]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_name_filter_case_insensitive(self):
convs = [
self._chatgpt("a", "Python Course"),
self._chatgpt("b", "Java Course"),
self._chatgpt("c", None),
]
result = _filter_by_project(convs, "PYTHON")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_name_filter_substring_match(self):
convs = [
self._chatgpt("a", "Python Advanced Course"),
self._chatgpt("b", "Python Basics"),
self._chatgpt("c", "JavaScript"),
]
result = _filter_by_project(convs, "python")
assert len(result) == 2
assert {c["id"] for c in result} == {"a", "b"}
def test_no_matches_returns_empty(self):
convs = [self._chatgpt("a", "Python Course"), self._chatgpt("b", None)]
result = _filter_by_project(convs, "ruby")
assert result == []
def test_none_filter_excludes_all_with_projects(self):
convs = [self._chatgpt("a", "Project A"), self._chatgpt("b", "Project B")]
result = _filter_by_project(convs, "none")
assert result == []
def test_empty_string_project_treated_as_no_project(self):
convs = [{"id": "a", "_project_name": ""}, {"id": "b", "_project_name": "Real"}]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_claude_project_string_matched(self):
# Claude can also have project as a plain string
convs = [{"id": "a", "project": "python-course"}, {"id": "b", "project": None}]
result = _filter_by_project(convs, "python")
assert len(result) == 1
assert result[0]["id"] == "a"
# ---------------------------------------------------------------------------
# export --since validation (T-25)
# ---------------------------------------------------------------------------
class TestExportSinceValidation:
"""Test that --since with an invalid date exits cleanly with an error message."""
def _pre_populated_cache(self, tmp_path) -> Cache:
"""Create a cache that passes the ToS gate and first-run doctor check."""
cache = Cache(tmp_path)
cache.acknowledge_tos()
cache.mark_exported("chatgpt", "dummy-conv", {"updated_at": "2024-01-01T00:00:00Z"})
return cache
def test_invalid_since_date_exits_with_error(self, tmp_path):
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--since", "notadate"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 1
assert "Invalid --since date" in result.output
assert "YYYY-MM-DD" in result.output
def test_valid_since_date_does_not_error(self, tmp_path):
"""A valid date should not produce the invalid-date error (may fail later on API)."""
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--since", "2024-01-01"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
# This test hits the real (failing) auth endpoint with retries;
# don't add politeness pacing on top of the backoff sleeps.
"REQUEST_DELAY": "0",
},
)
assert "Invalid --since date" not in result.output
def test_max_conversations_zero_rejected(self, tmp_path):
"""--max-conversations uses IntRange(min=1); 0 must be rejected by click."""
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--max-conversations", "0"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 2
assert "max-conversations" in result.output
# ---------------------------------------------------------------------------
# LossReport summary
# ---------------------------------------------------------------------------
class TestLossReportSummary:
"""The LossReport's format_summary() pinned format covers zero, top-5, and overflow cases."""
def test_zero_summary_uses_none_sentinel(self):
from src.loss_report import LossReport
report = LossReport()
out = report.format_summary()
assert "[export] Run summary:" in out
assert "conversations: 0" in out
assert "messages rendered: 0" in out
# All three "(none)" sentinels present — never empty parens
# (unknown blocks, extraction failures, collapsed by policy)
assert out.count("(none)") == 3
def test_top_5_breakdown(self):
from src.loss_report import LossReport
report = LossReport()
for raw_type in ("a", "b", "c", "d", "e", "f", "g"):
report.record_unknown(raw_type)
if raw_type == "a":
# Make 'a' the most common
for _ in range(4):
report.record_unknown("a")
out = report.format_summary()
# Top entry shown
assert "a=5" in out
# Overflow line present (7 types, top 5 + 2 more)
assert "+ 2 more types" in out
def test_messages_and_conversations_recorded(self):
from src.loss_report import LossReport
report = LossReport()
report.record_conversation()
report.record_message()
report.record_message()
out = report.format_summary()
assert "conversations: 1" in out
assert "messages rendered: 2" in out
# ---------------------------------------------------------------------------
# prune command
# ---------------------------------------------------------------------------
class TestPrune:
"""prune deletes export files not referenced by the manifest."""
def _setup(self, tmp_path):
"""Cache with one referenced file; one stale file + empty-dir candidate."""
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
export_dir = tmp_path / "exports"
keep = export_dir / "chatgpt" / "proj.2026" / "keep.md"
keep.parent.mkdir(parents=True)
keep.write_text("kept")
stale = export_dir / "chatgpt" / "proj" / "2026" / "old-layout.md"
stale.parent.mkdir(parents=True)
stale.write_text("stale")
cache.mark_exported("chatgpt", "conv-1", {"file_path": str(keep)})
return export_dir, keep, stale
def _invoke(self, tmp_path, *args):
runner = CliRunner(mix_stderr=True)
return runner.invoke(
cli,
["--no-log-file", "prune", *args],
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
def test_dry_run_lists_but_keeps_files(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
result = self._invoke(tmp_path, "--dry-run")
assert result.exit_code == 0
assert "old-layout.md" in result.output
assert "Dry run" in result.output
assert stale.exists() and keep.exists()
def test_yes_deletes_stale_keeps_referenced_sweeps_dirs(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
result = self._invoke(tmp_path, "--yes")
assert result.exit_code == 0
assert not stale.exists()
assert keep.exists()
# Old-layout dirs are now empty and swept
assert not (export_dir / "chatgpt" / "proj").exists()
def test_refuses_with_empty_manifest(self, tmp_path):
"""Footgun guard: after cache --clear, prune must not wipe the archive."""
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
export_dir = tmp_path / "exports"
f = export_dir / "chatgpt" / "a.md"
f.parent.mkdir(parents=True)
f.write_text("data")
result = self._invoke(tmp_path, "--yes")
assert result.exit_code == 1
assert "Refusing to prune" in result.output
assert f.exists()
def test_aborts_without_confirmation(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "prune"],
input="n\n",
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 0
assert "Aborted" in result.output
assert stale.exists()
class TestCanaryCommand:
"""`canary` wiring — no live API calls (no tokens configured)."""
def test_no_tokens_exits_nonzero(self, tmp_path):
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "canary"],
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
# Empty so .env (override=False) can't repopulate real tokens.
"CHATGPT_SESSION_TOKEN": "",
"CLAUDE_SESSION_KEY": "",
},
)
assert result.exit_code == 1
assert "No web-API provider tokens" in result.output
class TestProjectsCommand:
"""`projects` discovers project IDs missing from CHATGPT_PROJECT_IDS."""
def _patch_provider(self, monkeypatch, summaries, details=None, names=None):
import src.providers.chatgpt as chatgpt_mod
names = names or {}
details = details or {}
class FakeProvider:
def __init__(self, **kwargs):
self._project_ids = kwargs.get("project_ids") or []
def fetch_all_conversations(self, since=None):
return summaries
def get_conversation(self, conv_id):
return details.get(conv_id, {})
def _fetch_project_name(self, gizmo_id):
return names.get(gizmo_id, gizmo_id)
monkeypatch.setattr(chatgpt_mod, "ChatGPTProvider", FakeProvider)
return FakeProvider
def _env(self, tmp_path, **extra):
# Clears the ToS gate and the first-run doctor check.
cache = Cache(tmp_path)
cache.acknowledge_tos()
cache.mark_exported("chatgpt", "dummy", {"updated_at": "2024-01-01T00:00:00Z"})
env = {
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
}
env.update(extra)
return env
def test_reports_project_absent_from_config(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-missing"}],
names={"g-p-missing": "Tech Questions"},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert result.exit_code == 0
assert "Tech Questions" in result.output
assert "g-p-missing" in result.output
assert "CHATGPT_PROJECT_IDS=g-p-missing" in result.output
def test_quiet_when_everything_is_configured(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-known"}],
names={"g-p-known": "Known"},
)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "projects"],
env=self._env(tmp_path, CHATGPT_PROJECT_IDS="g-p-known"),
)
assert result.exit_code == 0
assert "already configured" in result.output
def test_custom_gpt_ids_are_ignored(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-notaproject"}],
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert "g-notaproject" not in result.output
def test_deep_reads_conversation_details(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1"}], # listing carries no gizmo_id
details={"c1": {"gizmo_id": "g-p-fromdetail"}},
names={"g-p-fromdetail": "Found Deep"},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects", "--deep"], env=self._env(tmp_path)
)
assert "Found Deep" in result.output
assert "CHATGPT_PROJECT_IDS=g-p-fromdetail" in result.output
def test_without_deep_says_so_rather_than_reporting_nothing(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1"}],
details={"c1": {"gizmo_id": "g-p-fromdetail"}},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert "--deep" in result.output
def test_write_updates_env(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-new"}],
names={"g-p-new": "New Project"},
)
runner = CliRunner(mix_stderr=True)
with runner.isolated_filesystem(temp_dir=tmp_path) as fs:
result = runner.invoke(
cli, ["--no-log-file", "projects", "--write"], env=self._env(tmp_path)
)
assert result.exit_code == 0
env_text = (Path(fs) / ".env").read_text(encoding="utf-8")
assert "CHATGPT_PROJECT_IDS=g-p-new" in env_text
# ---------------------------------------------------------------------------
# sync command + non-interactive ToS gate
# ---------------------------------------------------------------------------
class TestSyncCommand:
"""`sync` chains export → joplin for schedulers, with a real exit code."""
def _cache(self, tmp_path) -> Cache:
cache = Cache(tmp_path)
cache.acknowledge_tos()
# Non-empty last_run so the first-run doctor gate stays out of the way.
cache.mark_exported("codex", "dummy", {"updated_at": "2024-01-01T00:00:00Z"})
return cache
def _env(self, tmp_path) -> dict:
"""A real (minimal) codex session — `export` exits 1 on no providers at
all, so an empty directory would test the wrong failure."""
import json
day = tmp_path / "sessions" / "2026" / "08" / "17"
day.mkdir(parents=True, exist_ok=True)
sid = "01a00e3f-a309-74a3-bf32-06c2cd87faa3"
records = [
{
"timestamp": "2026-08-17T05:44:06.666Z",
"type": "session_meta",
"payload": {"session_id": sid, "timestamp": "2026-08-17T05:44:06.666Z",
"cwd": str(tmp_path / "ws")},
},
{
"timestamp": "2026-08-17T05:44:09.000Z",
"type": "event_msg",
"payload": {"type": "item_completed", "item": {
"type": "UserMessage", "id": "u1",
"content": [{"type": "text", "text": "hello"}]}},
},
]
(day / f"rollout-2026-08-17T01-44-06-{sid}.jsonl").write_text(
"\n".join(json.dumps(r) for r in records), encoding="utf-8"
)
return {
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
"CODEX_DIR": str(tmp_path / "sessions"),
}
def test_skip_joplin_exits_zero(self, tmp_path):
self._cache(tmp_path)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex", "--skip-joplin"],
env=self._env(tmp_path),
)
assert result.exit_code == 0
assert "Skipping Joplin sync" in result.output
assert "Sync complete" in result.output
def test_joplin_optional_survives_unreachable_joplin(self, tmp_path):
"""Joplin being closed must not fail a scheduled run — the export is done."""
self._cache(tmp_path)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex", "--joplin-optional"],
env={**self._env(tmp_path), "JOPLIN_API_URL": "http://127.0.0.1:9"},
)
assert result.exit_code == 0
assert "Joplin sync skipped" in result.output
def test_unreachable_joplin_fails_without_the_flag(self, tmp_path):
self._cache(tmp_path)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex"],
env={**self._env(tmp_path), "JOPLIN_API_URL": "http://127.0.0.1:9"},
)
assert result.exit_code == 1
def test_export_failures_set_nonzero_exit(self, tmp_path, monkeypatch):
"""A scheduler must be able to tell a real run from a silent no-op."""
self._cache(tmp_path)
import src.main as main_mod
real_export = main_mod.export.callback
def fake_export(*args, **kwargs):
import click
ctx = click.get_current_context()
ctx.obj["last_export_summary"] = {
"codex": {"exported": 0, "skipped": 0, "failed": 3}
}
monkeypatch.setattr(main_mod.export, "callback", fake_export)
try:
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex", "--skip-joplin"],
env=self._env(tmp_path),
)
finally:
monkeypatch.setattr(main_mod.export, "callback", real_export)
assert result.exit_code == 1
assert "3 conversation(s) failed to export" in result.output
class TestNonInteractiveTosGate:
"""Without a TTY the gate must fail loudly, not exit 0 having done nothing."""
def test_no_tty_exits_one_with_explanation(self, tmp_path, monkeypatch):
Cache(tmp_path) # fresh cache: ToS not acknowledged
monkeypatch.setattr("sys.stdin.isatty", lambda: False)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "doctor"],
env={"CACHE_DIR": str(tmp_path), "EXPORT_DIR": str(tmp_path / "exports")},
)
assert result.exit_code == 1
assert "no terminal to" in result.output
# ---------------------------------------------------------------------------
# ntfy notifications
# ---------------------------------------------------------------------------
class TestNotifyFormatting:
"""The payload must stay counts-only: a public ntfy topic is world-readable."""
def test_success_summary(self):
from src.notify import format_summary
title, body, tags, priority = format_summary(
{"codex": {"exported": 2, "skipped": 5, "failed": 0}}, None, []
)
assert title.startswith("AI archive OK")
assert "codex: 2 exported, 5 up to date" in body
assert tags == "white_check_mark"
assert priority == "default"
def test_quiet_run_is_low_priority(self):
from src.notify import format_summary
_, _, tags, priority = format_summary(
{"codex": {"exported": 0, "skipped": 7, "failed": 0}}, None, []
)
assert tags == "zzz"
assert priority == "low"
def test_failure_summary_is_high_priority(self):
from src.notify import format_summary
title, body, tags, priority = format_summary(
{"chatgpt": {"exported": 0, "skipped": 0, "failed": 12}},
None,
["chatgpt: 12 conversation(s) failed to export"],
)
assert "FAILED" in title
assert "12 FAILED" in body
assert tags == "rotating_light"
assert priority == "high"
def test_hostname_present(self):
"""Two machines share one topic — counts are meaningless without it."""
from src.notify import format_summary, machine_name
title, _, _, _ = format_summary({}, None, [])
assert machine_name() in title
class TestNotifyHeaderEncoding:
"""HTTP headers are latin-1; an em dash in a title loses the notification."""
def test_smart_punctuation_flattened(self):
from src.notify import _ascii_header
out = _ascii_header("AI archive — don’t “fail”…")
assert out == 'AI archive - don\'t "fail"...'
out.encode("ascii") # must not raise
def test_arbitrary_unicode_survives_as_ascii(self):
from src.notify import _ascii_header
_ascii_header("héllo — 世界").encode("ascii")
class TestNotifySend:
def test_no_topic_is_a_no_op(self, monkeypatch):
from src import notify as notify_mod
monkeypatch.delenv("NTFY_TOPIC", raising=False)
assert notify_mod.send("t", "m") is False
assert notify_mod.is_configured() is False
def test_network_failure_never_raises(self, monkeypatch):
"""A down ntfy server must not fail a run that captured data."""
from src import notify as notify_mod
monkeypatch.setenv("NTFY_TOPIC", "unit-test-topic")
monkeypatch.setenv("NTFY_SERVER", "http://127.0.0.1:9")
assert notify_mod.send("t", "m") is False
def test_off_policy_disables(self, monkeypatch):
from src import notify as notify_mod
monkeypatch.setenv("NTFY_TOPIC", "unit-test-topic")
monkeypatch.setenv("NTFY_NOTIFY", "off")
assert notify_mod.is_configured() is False