Files
AIChatExporter/tests/test_cli.py
T

538 lines
20 KiB
Python

"""CLI-level tests using Click's CliRunner — no live API calls required."""
from pathlib import Path
import pytest
from click.testing import CliRunner
from src.cache import Cache
from src.main import _filter_by_project, cli
# ---------------------------------------------------------------------------
# _filter_by_project (T-27)
# ---------------------------------------------------------------------------
class TestFilterByProject:
"""Unit tests for the project filter logic used by export/list/joplin."""
# ChatGPT conversations use the _project_name annotation key
def _chatgpt(self, conv_id, project_name):
return {"id": conv_id, "_project_name": project_name}
# Claude conversations use the project dict key
def _claude(self, conv_id, project_name):
proj = {"name": project_name} if project_name else None
return {"id": conv_id, "project": proj}
def test_none_filter_keeps_no_project_chatgpt(self):
convs = [self._chatgpt("a", None), self._chatgpt("b", "Python Course")]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_none_filter_keeps_no_project_claude(self):
convs = [self._claude("a", None), self._claude("b", "Python Course")]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_name_filter_case_insensitive(self):
convs = [
self._chatgpt("a", "Python Course"),
self._chatgpt("b", "Java Course"),
self._chatgpt("c", None),
]
result = _filter_by_project(convs, "PYTHON")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_name_filter_substring_match(self):
convs = [
self._chatgpt("a", "Python Advanced Course"),
self._chatgpt("b", "Python Basics"),
self._chatgpt("c", "JavaScript"),
]
result = _filter_by_project(convs, "python")
assert len(result) == 2
assert {c["id"] for c in result} == {"a", "b"}
def test_no_matches_returns_empty(self):
convs = [self._chatgpt("a", "Python Course"), self._chatgpt("b", None)]
result = _filter_by_project(convs, "ruby")
assert result == []
def test_none_filter_excludes_all_with_projects(self):
convs = [self._chatgpt("a", "Project A"), self._chatgpt("b", "Project B")]
result = _filter_by_project(convs, "none")
assert result == []
def test_empty_string_project_treated_as_no_project(self):
convs = [{"id": "a", "_project_name": ""}, {"id": "b", "_project_name": "Real"}]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_claude_project_string_matched(self):
# Claude can also have project as a plain string
convs = [{"id": "a", "project": "python-course"}, {"id": "b", "project": None}]
result = _filter_by_project(convs, "python")
assert len(result) == 1
assert result[0]["id"] == "a"
# ---------------------------------------------------------------------------
# export --since validation (T-25)
# ---------------------------------------------------------------------------
class TestExportSinceValidation:
"""Test that --since with an invalid date exits cleanly with an error message."""
def _pre_populated_cache(self, tmp_path) -> Cache:
"""Create a cache that passes the ToS gate and first-run doctor check."""
cache = Cache(tmp_path)
cache.acknowledge_tos()
cache.mark_exported("chatgpt", "dummy-conv", {"updated_at": "2024-01-01T00:00:00Z"})
return cache
def test_invalid_since_date_exits_with_error(self, tmp_path):
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--since", "notadate"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 1
assert "Invalid --since date" in result.output
assert "YYYY-MM-DD" in result.output
def test_valid_since_date_does_not_error(self, tmp_path):
"""A valid date should not produce the invalid-date error (may fail later on API)."""
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--since", "2024-01-01"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
# This test hits the real (failing) auth endpoint with retries;
# don't add politeness pacing on top of the backoff sleeps.
"REQUEST_DELAY": "0",
},
)
assert "Invalid --since date" not in result.output
def test_max_conversations_zero_rejected(self, tmp_path):
"""--max-conversations uses IntRange(min=1); 0 must be rejected by click."""
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--max-conversations", "0"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 2
assert "max-conversations" in result.output
# ---------------------------------------------------------------------------
# LossReport summary
# ---------------------------------------------------------------------------
class TestLossReportSummary:
"""The LossReport's format_summary() pinned format covers zero, top-5, and overflow cases."""
def test_zero_summary_uses_none_sentinel(self):
from src.loss_report import LossReport
report = LossReport()
out = report.format_summary()
assert "[export] Run summary:" in out
assert "conversations: 0" in out
assert "messages rendered: 0" in out
# All three "(none)" sentinels present — never empty parens
# (unknown blocks, extraction failures, collapsed by policy)
assert out.count("(none)") == 3
def test_top_5_breakdown(self):
from src.loss_report import LossReport
report = LossReport()
for raw_type in ("a", "b", "c", "d", "e", "f", "g"):
report.record_unknown(raw_type)
if raw_type == "a":
# Make 'a' the most common
for _ in range(4):
report.record_unknown("a")
out = report.format_summary()
# Top entry shown
assert "a=5" in out
# Overflow line present (7 types, top 5 + 2 more)
assert "+ 2 more types" in out
def test_messages_and_conversations_recorded(self):
from src.loss_report import LossReport
report = LossReport()
report.record_conversation()
report.record_message()
report.record_message()
out = report.format_summary()
assert "conversations: 1" in out
assert "messages rendered: 2" in out
# ---------------------------------------------------------------------------
# prune command
# ---------------------------------------------------------------------------
class TestPrune:
"""prune deletes export files not referenced by the manifest."""
def _setup(self, tmp_path):
"""Cache with one referenced file; one stale file + empty-dir candidate."""
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
export_dir = tmp_path / "exports"
keep = export_dir / "chatgpt" / "proj.2026" / "keep.md"
keep.parent.mkdir(parents=True)
keep.write_text("kept")
stale = export_dir / "chatgpt" / "proj" / "2026" / "old-layout.md"
stale.parent.mkdir(parents=True)
stale.write_text("stale")
cache.mark_exported("chatgpt", "conv-1", {"file_path": str(keep)})
return export_dir, keep, stale
def _invoke(self, tmp_path, *args):
runner = CliRunner(mix_stderr=True)
return runner.invoke(
cli,
["--no-log-file", "prune", *args],
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
def test_dry_run_lists_but_keeps_files(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
result = self._invoke(tmp_path, "--dry-run")
assert result.exit_code == 0
assert "old-layout.md" in result.output
assert "Dry run" in result.output
assert stale.exists() and keep.exists()
def test_yes_deletes_stale_keeps_referenced_sweeps_dirs(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
result = self._invoke(tmp_path, "--yes")
assert result.exit_code == 0
assert not stale.exists()
assert keep.exists()
# Old-layout dirs are now empty and swept
assert not (export_dir / "chatgpt" / "proj").exists()
def test_refuses_with_empty_manifest(self, tmp_path):
"""Footgun guard: after cache --clear, prune must not wipe the archive."""
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
export_dir = tmp_path / "exports"
f = export_dir / "chatgpt" / "a.md"
f.parent.mkdir(parents=True)
f.write_text("data")
result = self._invoke(tmp_path, "--yes")
assert result.exit_code == 1
assert "Refusing to prune" in result.output
assert f.exists()
def test_aborts_without_confirmation(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "prune"],
input="n\n",
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 0
assert "Aborted" in result.output
assert stale.exists()
class TestCanaryCommand:
"""`canary` wiring — no live API calls (no tokens configured)."""
def test_no_tokens_exits_nonzero(self, tmp_path):
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "canary"],
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
# Empty so .env (override=False) can't repopulate real tokens.
"CHATGPT_SESSION_TOKEN": "",
"CLAUDE_SESSION_KEY": "",
},
)
assert result.exit_code == 1
assert "No web-API provider tokens" in result.output
class TestProjectsCommand:
"""`projects` discovers project IDs missing from CHATGPT_PROJECT_IDS."""
def _patch_provider(self, monkeypatch, summaries, details=None, names=None):
import src.providers.chatgpt as chatgpt_mod
names = names or {}
details = details or {}
class FakeProvider:
def __init__(self, **kwargs):
self._project_ids = kwargs.get("project_ids") or []
def fetch_all_conversations(self, since=None):
return summaries
def get_conversation(self, conv_id):
return details.get(conv_id, {})
def _fetch_project_name(self, gizmo_id):
return names.get(gizmo_id, gizmo_id)
monkeypatch.setattr(chatgpt_mod, "ChatGPTProvider", FakeProvider)
return FakeProvider
def _env(self, tmp_path, **extra):
# Clears the ToS gate and the first-run doctor check.
cache = Cache(tmp_path)
cache.acknowledge_tos()
cache.mark_exported("chatgpt", "dummy", {"updated_at": "2024-01-01T00:00:00Z"})
env = {
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
}
env.update(extra)
return env
def test_reports_project_absent_from_config(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-missing"}],
names={"g-p-missing": "Tech Questions"},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert result.exit_code == 0
assert "Tech Questions" in result.output
assert "g-p-missing" in result.output
assert "CHATGPT_PROJECT_IDS=g-p-missing" in result.output
def test_quiet_when_everything_is_configured(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-known"}],
names={"g-p-known": "Known"},
)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "projects"],
env=self._env(tmp_path, CHATGPT_PROJECT_IDS="g-p-known"),
)
assert result.exit_code == 0
assert "already configured" in result.output
def test_custom_gpt_ids_are_ignored(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-notaproject"}],
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert "g-notaproject" not in result.output
def test_deep_reads_conversation_details(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1"}], # listing carries no gizmo_id
details={"c1": {"gizmo_id": "g-p-fromdetail"}},
names={"g-p-fromdetail": "Found Deep"},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects", "--deep"], env=self._env(tmp_path)
)
assert "Found Deep" in result.output
assert "CHATGPT_PROJECT_IDS=g-p-fromdetail" in result.output
def test_without_deep_says_so_rather_than_reporting_nothing(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1"}],
details={"c1": {"gizmo_id": "g-p-fromdetail"}},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert "--deep" in result.output
def test_write_updates_env(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-new"}],
names={"g-p-new": "New Project"},
)
runner = CliRunner(mix_stderr=True)
with runner.isolated_filesystem(temp_dir=tmp_path) as fs:
result = runner.invoke(
cli, ["--no-log-file", "projects", "--write"], env=self._env(tmp_path)
)
assert result.exit_code == 0
env_text = (Path(fs) / ".env").read_text(encoding="utf-8")
assert "CHATGPT_PROJECT_IDS=g-p-new" in env_text
# ---------------------------------------------------------------------------
# sync command + non-interactive ToS gate
# ---------------------------------------------------------------------------
class TestSyncCommand:
"""`sync` chains export → joplin for schedulers, with a real exit code."""
def _cache(self, tmp_path) -> Cache:
cache = Cache(tmp_path)
cache.acknowledge_tos()
# Non-empty last_run so the first-run doctor gate stays out of the way.
cache.mark_exported("codex", "dummy", {"updated_at": "2024-01-01T00:00:00Z"})
return cache
def _env(self, tmp_path) -> dict:
"""A real (minimal) codex session — `export` exits 1 on no providers at
all, so an empty directory would test the wrong failure."""
import json
day = tmp_path / "sessions" / "2026" / "08" / "17"
day.mkdir(parents=True, exist_ok=True)
sid = "01a00e3f-a309-74a3-bf32-06c2cd87faa3"
records = [
{
"timestamp": "2026-08-17T05:44:06.666Z",
"type": "session_meta",
"payload": {"session_id": sid, "timestamp": "2026-08-17T05:44:06.666Z",
"cwd": str(tmp_path / "ws")},
},
{
"timestamp": "2026-08-17T05:44:09.000Z",
"type": "event_msg",
"payload": {"type": "item_completed", "item": {
"type": "UserMessage", "id": "u1",
"content": [{"type": "text", "text": "hello"}]}},
},
]
(day / f"rollout-2026-08-17T01-44-06-{sid}.jsonl").write_text(
"\n".join(json.dumps(r) for r in records), encoding="utf-8"
)
return {
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
"CODEX_DIR": str(tmp_path / "sessions"),
}
def test_skip_joplin_exits_zero(self, tmp_path):
self._cache(tmp_path)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex", "--skip-joplin"],
env=self._env(tmp_path),
)
assert result.exit_code == 0
assert "Skipping Joplin sync" in result.output
assert "Sync complete" in result.output
def test_joplin_optional_survives_unreachable_joplin(self, tmp_path):
"""Joplin being closed must not fail a scheduled run — the export is done."""
self._cache(tmp_path)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex", "--joplin-optional"],
env={**self._env(tmp_path), "JOPLIN_API_URL": "http://127.0.0.1:9"},
)
assert result.exit_code == 0
assert "Joplin sync skipped" in result.output
def test_unreachable_joplin_fails_without_the_flag(self, tmp_path):
self._cache(tmp_path)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex"],
env={**self._env(tmp_path), "JOPLIN_API_URL": "http://127.0.0.1:9"},
)
assert result.exit_code == 1
def test_export_failures_set_nonzero_exit(self, tmp_path, monkeypatch):
"""A scheduler must be able to tell a real run from a silent no-op."""
self._cache(tmp_path)
import src.main as main_mod
real_export = main_mod.export.callback
def fake_export(*args, **kwargs):
import click
ctx = click.get_current_context()
ctx.obj["last_export_summary"] = {
"codex": {"exported": 0, "skipped": 0, "failed": 3}
}
monkeypatch.setattr(main_mod.export, "callback", fake_export)
try:
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "sync", "--provider", "codex", "--skip-joplin"],
env=self._env(tmp_path),
)
finally:
monkeypatch.setattr(main_mod.export, "callback", real_export)
assert result.exit_code == 1
assert "3 conversation(s) failed to export" in result.output
class TestNonInteractiveTosGate:
"""Without a TTY the gate must fail loudly, not exit 0 having done nothing."""
def test_no_tty_exits_one_with_explanation(self, tmp_path, monkeypatch):
Cache(tmp_path) # fresh cache: ToS not acknowledged
monkeypatch.setattr("sys.stdin.isatty", lambda: False)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "doctor"],
env={"CACHE_DIR": str(tmp_path), "EXPORT_DIR": str(tmp_path / "exports")},
)
assert result.exit_code == 1
assert "no terminal to" in result.output