Files
AIChatExporter/tests/test_cli.py
T
JesseMarkowitz d083aea135 feat: projects command — find the project IDs the config is missing
Reading gizmo_id during normalization fixed attribution, but it cannot
answer "which projects am I missing?": normalization only runs on
conversations being exported, and a normal run skips everything already
cached. Discovering the gaps would have meant --force re-exporting the
whole archive.

`ai-chat-exporter projects` does it directly. It lists conversations,
collects the g-p- ids they belong to, resolves display names, and prints a
table marking which are absent from .env plus a paste-ready
CHATGPT_PROJECT_IDS line. --write applies it; --deep falls back to one
detail request per conversation when the listing does not carry gizmo_id
(unverified which shape this account returns, so the command reports which
path it took rather than assuming).

This matters beyond tidiness: attribution is now self-correcting, but the
listing pass still needs the ids. Conversations that live only inside a
project never appear in the default listing, so an unlisted project's chats
are not merely misfiled — they are never fetched.

6 CLI tests: reporting, the already-configured case, the g-p- guard, --deep,
the hint when --deep is needed, and --write. 324 pass.
2026-08-17 13:19:34 -04:00

418 lines
16 KiB
Python

"""CLI-level tests using Click's CliRunner — no live API calls required."""
from pathlib import Path
import pytest
from click.testing import CliRunner
from src.cache import Cache
from src.main import _filter_by_project, cli
# ---------------------------------------------------------------------------
# _filter_by_project (T-27)
# ---------------------------------------------------------------------------
class TestFilterByProject:
"""Unit tests for the project filter logic used by export/list/joplin."""
# ChatGPT conversations use the _project_name annotation key
def _chatgpt(self, conv_id, project_name):
return {"id": conv_id, "_project_name": project_name}
# Claude conversations use the project dict key
def _claude(self, conv_id, project_name):
proj = {"name": project_name} if project_name else None
return {"id": conv_id, "project": proj}
def test_none_filter_keeps_no_project_chatgpt(self):
convs = [self._chatgpt("a", None), self._chatgpt("b", "Python Course")]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_none_filter_keeps_no_project_claude(self):
convs = [self._claude("a", None), self._claude("b", "Python Course")]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_name_filter_case_insensitive(self):
convs = [
self._chatgpt("a", "Python Course"),
self._chatgpt("b", "Java Course"),
self._chatgpt("c", None),
]
result = _filter_by_project(convs, "PYTHON")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_name_filter_substring_match(self):
convs = [
self._chatgpt("a", "Python Advanced Course"),
self._chatgpt("b", "Python Basics"),
self._chatgpt("c", "JavaScript"),
]
result = _filter_by_project(convs, "python")
assert len(result) == 2
assert {c["id"] for c in result} == {"a", "b"}
def test_no_matches_returns_empty(self):
convs = [self._chatgpt("a", "Python Course"), self._chatgpt("b", None)]
result = _filter_by_project(convs, "ruby")
assert result == []
def test_none_filter_excludes_all_with_projects(self):
convs = [self._chatgpt("a", "Project A"), self._chatgpt("b", "Project B")]
result = _filter_by_project(convs, "none")
assert result == []
def test_empty_string_project_treated_as_no_project(self):
convs = [{"id": "a", "_project_name": ""}, {"id": "b", "_project_name": "Real"}]
result = _filter_by_project(convs, "none")
assert len(result) == 1
assert result[0]["id"] == "a"
def test_claude_project_string_matched(self):
# Claude can also have project as a plain string
convs = [{"id": "a", "project": "python-course"}, {"id": "b", "project": None}]
result = _filter_by_project(convs, "python")
assert len(result) == 1
assert result[0]["id"] == "a"
# ---------------------------------------------------------------------------
# export --since validation (T-25)
# ---------------------------------------------------------------------------
class TestExportSinceValidation:
"""Test that --since with an invalid date exits cleanly with an error message."""
def _pre_populated_cache(self, tmp_path) -> Cache:
"""Create a cache that passes the ToS gate and first-run doctor check."""
cache = Cache(tmp_path)
cache.acknowledge_tos()
cache.mark_exported("chatgpt", "dummy-conv", {"updated_at": "2024-01-01T00:00:00Z"})
return cache
def test_invalid_since_date_exits_with_error(self, tmp_path):
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--since", "notadate"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 1
assert "Invalid --since date" in result.output
assert "YYYY-MM-DD" in result.output
def test_valid_since_date_does_not_error(self, tmp_path):
"""A valid date should not produce the invalid-date error (may fail later on API)."""
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--since", "2024-01-01"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
# This test hits the real (failing) auth endpoint with retries;
# don't add politeness pacing on top of the backoff sleeps.
"REQUEST_DELAY": "0",
},
)
assert "Invalid --since date" not in result.output
def test_max_conversations_zero_rejected(self, tmp_path):
"""--max-conversations uses IntRange(min=1); 0 must be rejected by click."""
self._pre_populated_cache(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "export", "--max-conversations", "0"],
env={
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 2
assert "max-conversations" in result.output
# ---------------------------------------------------------------------------
# LossReport summary
# ---------------------------------------------------------------------------
class TestLossReportSummary:
"""The LossReport's format_summary() pinned format covers zero, top-5, and overflow cases."""
def test_zero_summary_uses_none_sentinel(self):
from src.loss_report import LossReport
report = LossReport()
out = report.format_summary()
assert "[export] Run summary:" in out
assert "conversations: 0" in out
assert "messages rendered: 0" in out
# All three "(none)" sentinels present — never empty parens
# (unknown blocks, extraction failures, collapsed by policy)
assert out.count("(none)") == 3
def test_top_5_breakdown(self):
from src.loss_report import LossReport
report = LossReport()
for raw_type in ("a", "b", "c", "d", "e", "f", "g"):
report.record_unknown(raw_type)
if raw_type == "a":
# Make 'a' the most common
for _ in range(4):
report.record_unknown("a")
out = report.format_summary()
# Top entry shown
assert "a=5" in out
# Overflow line present (7 types, top 5 + 2 more)
assert "+ 2 more types" in out
def test_messages_and_conversations_recorded(self):
from src.loss_report import LossReport
report = LossReport()
report.record_conversation()
report.record_message()
report.record_message()
out = report.format_summary()
assert "conversations: 1" in out
assert "messages rendered: 2" in out
# ---------------------------------------------------------------------------
# prune command
# ---------------------------------------------------------------------------
class TestPrune:
"""prune deletes export files not referenced by the manifest."""
def _setup(self, tmp_path):
"""Cache with one referenced file; one stale file + empty-dir candidate."""
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
export_dir = tmp_path / "exports"
keep = export_dir / "chatgpt" / "proj.2026" / "keep.md"
keep.parent.mkdir(parents=True)
keep.write_text("kept")
stale = export_dir / "chatgpt" / "proj" / "2026" / "old-layout.md"
stale.parent.mkdir(parents=True)
stale.write_text("stale")
cache.mark_exported("chatgpt", "conv-1", {"file_path": str(keep)})
return export_dir, keep, stale
def _invoke(self, tmp_path, *args):
runner = CliRunner(mix_stderr=True)
return runner.invoke(
cli,
["--no-log-file", "prune", *args],
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
def test_dry_run_lists_but_keeps_files(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
result = self._invoke(tmp_path, "--dry-run")
assert result.exit_code == 0
assert "old-layout.md" in result.output
assert "Dry run" in result.output
assert stale.exists() and keep.exists()
def test_yes_deletes_stale_keeps_referenced_sweeps_dirs(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
result = self._invoke(tmp_path, "--yes")
assert result.exit_code == 0
assert not stale.exists()
assert keep.exists()
# Old-layout dirs are now empty and swept
assert not (export_dir / "chatgpt" / "proj").exists()
def test_refuses_with_empty_manifest(self, tmp_path):
"""Footgun guard: after cache --clear, prune must not wipe the archive."""
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
export_dir = tmp_path / "exports"
f = export_dir / "chatgpt" / "a.md"
f.parent.mkdir(parents=True)
f.write_text("data")
result = self._invoke(tmp_path, "--yes")
assert result.exit_code == 1
assert "Refusing to prune" in result.output
assert f.exists()
def test_aborts_without_confirmation(self, tmp_path):
export_dir, keep, stale = self._setup(tmp_path)
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "prune"],
input="n\n",
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
},
)
assert result.exit_code == 0
assert "Aborted" in result.output
assert stale.exists()
class TestCanaryCommand:
"""`canary` wiring — no live API calls (no tokens configured)."""
def test_no_tokens_exits_nonzero(self, tmp_path):
cache = Cache(tmp_path / "cache")
cache.acknowledge_tos()
runner = CliRunner(mix_stderr=True)
result = runner.invoke(
cli,
["--no-log-file", "canary"],
env={
"CACHE_DIR": str(tmp_path / "cache"),
"EXPORT_DIR": str(tmp_path / "exports"),
# Empty so .env (override=False) can't repopulate real tokens.
"CHATGPT_SESSION_TOKEN": "",
"CLAUDE_SESSION_KEY": "",
},
)
assert result.exit_code == 1
assert "No web-API provider tokens" in result.output
class TestProjectsCommand:
"""`projects` discovers project IDs missing from CHATGPT_PROJECT_IDS."""
def _patch_provider(self, monkeypatch, summaries, details=None, names=None):
import src.providers.chatgpt as chatgpt_mod
names = names or {}
details = details or {}
class FakeProvider:
def __init__(self, **kwargs):
self._project_ids = kwargs.get("project_ids") or []
def fetch_all_conversations(self, since=None):
return summaries
def get_conversation(self, conv_id):
return details.get(conv_id, {})
def _fetch_project_name(self, gizmo_id):
return names.get(gizmo_id, gizmo_id)
monkeypatch.setattr(chatgpt_mod, "ChatGPTProvider", FakeProvider)
return FakeProvider
def _env(self, tmp_path, **extra):
# Clears the ToS gate and the first-run doctor check.
cache = Cache(tmp_path)
cache.acknowledge_tos()
cache.mark_exported("chatgpt", "dummy", {"updated_at": "2024-01-01T00:00:00Z"})
env = {
"CHATGPT_SESSION_TOKEN": "eyJtesttoken",
"CACHE_DIR": str(tmp_path),
"EXPORT_DIR": str(tmp_path / "exports"),
}
env.update(extra)
return env
def test_reports_project_absent_from_config(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-missing"}],
names={"g-p-missing": "Tech Questions"},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert result.exit_code == 0
assert "Tech Questions" in result.output
assert "g-p-missing" in result.output
assert "CHATGPT_PROJECT_IDS=g-p-missing" in result.output
def test_quiet_when_everything_is_configured(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-known"}],
names={"g-p-known": "Known"},
)
result = CliRunner(mix_stderr=True).invoke(
cli,
["--no-log-file", "projects"],
env=self._env(tmp_path, CHATGPT_PROJECT_IDS="g-p-known"),
)
assert result.exit_code == 0
assert "already configured" in result.output
def test_custom_gpt_ids_are_ignored(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-notaproject"}],
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert "g-notaproject" not in result.output
def test_deep_reads_conversation_details(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1"}], # listing carries no gizmo_id
details={"c1": {"gizmo_id": "g-p-fromdetail"}},
names={"g-p-fromdetail": "Found Deep"},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects", "--deep"], env=self._env(tmp_path)
)
assert "Found Deep" in result.output
assert "CHATGPT_PROJECT_IDS=g-p-fromdetail" in result.output
def test_without_deep_says_so_rather_than_reporting_nothing(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1"}],
details={"c1": {"gizmo_id": "g-p-fromdetail"}},
)
result = CliRunner(mix_stderr=True).invoke(
cli, ["--no-log-file", "projects"], env=self._env(tmp_path)
)
assert "--deep" in result.output
def test_write_updates_env(self, tmp_path, monkeypatch):
self._patch_provider(
monkeypatch,
summaries=[{"id": "c1", "gizmo_id": "g-p-new"}],
names={"g-p-new": "New Project"},
)
runner = CliRunner(mix_stderr=True)
with runner.isolated_filesystem(temp_dir=tmp_path) as fs:
result = runner.invoke(
cli, ["--no-log-file", "projects", "--write"], env=self._env(tmp_path)
)
assert result.exit_code == 0
env_text = (Path(fs) / ".env").read_text(encoding="utf-8")
assert "CHATGPT_PROJECT_IDS=g-p-new" in env_text