From d083aea135d84bb7c7cee13079d7630f066c6aef Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 13:19:34 -0400 Subject: [PATCH] =?UTF-8?q?feat:=20projects=20command=20=E2=80=94=20find?= =?UTF-8?q?=20the=20project=20IDs=20the=20config=20is=20missing?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reading gizmo_id during normalization fixed attribution, but it cannot answer "which projects am I missing?": normalization only runs on conversations being exported, and a normal run skips everything already cached. Discovering the gaps would have meant --force re-exporting the whole archive. `ai-chat-exporter projects` does it directly. It lists conversations, collects the g-p- ids they belong to, resolves display names, and prints a table marking which are absent from .env plus a paste-ready CHATGPT_PROJECT_IDS line. --write applies it; --deep falls back to one detail request per conversation when the listing does not carry gizmo_id (unverified which shape this account returns, so the command reports which path it took rather than assuming). This matters beyond tidiness: attribution is now self-correcting, but the listing pass still needs the ids. Conversations that live only inside a project never appear in the default listing, so an unlisted project's chats are not merely misfiled — they are never fetched. 6 CLI tests: reporting, the already-configured case, the g-p- guard, --deep, the hint when --deep is needed, and --write. 324 pass. --- CHANGELOG.md | 1 + README.md | 22 ++++++++ src/main.py | 137 ++++++++++++++++++++++++++++++++++++++++++++++ tests/test_cli.py | 118 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 278 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index a225047..45d560a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,7 @@ Format follows [Keep a Changelog](https://keepachangelog.com/en/1.0.0/). - **`tests/test_config.py::TestSessionLimiterConfig::test_defaults` depended on the developer's `.env`.** `load_config()` calls `load_dotenv(override=False)`, which re-populated the variable the test had just deleted — so it passed only on a machine with no `.env`. The test now stubs dotenv discovery. ### Added +- **`projects` command — discover the project IDs your config is missing.** `CHATGPT_PROJECT_IDS` is maintained by hand, and a project missing from it is invisible to the listing pass, so its conversations are never fetched. The command reports every project your conversations belong to, marks which are absent from `.env`, and prints a paste-ready line (`--write` applies it). It reads project ids from the conversation listing when they are there and falls back to `--deep`, one detail request per conversation, when they are not. - **Project attribution now reads the conversation's own `gizmo_id`.** Previously the project name came only from `CHATGPT_PROJECT_IDS`, so a conversation in a project you had not listed exported into `no-project/` even though its payload names its project. The detail response carries `gizmo_id`, so it is used as a fallback after the listing annotation and the project map — attribution stays correct without maintaining a list, and moving a chat into a new project no longer silently misfiles it. Only `g-p-` ids count: a custom GPT is not a project and must not become a folder. Each unconfigured project is reported once per run, naming the id to add, because the *listing* pass still needs `CHATGPT_PROJECT_IDS` — conversations that live only inside a project never appear in the default listing. ### Changed diff --git a/README.md b/README.md index 1d2884b..587bb61 100644 --- a/README.md +++ b/README.md @@ -201,6 +201,28 @@ ChatGPT project conversations are stored separately from your main conversation ### Finding your project IDs +The quickest way is to let the exporter find them: + +``` +ai-chat-exporter projects +``` + +It lists every project your conversations actually belong to, marks which are +missing from `.env`, and prints a paste-ready `CHATGPT_PROJECT_IDS=` line +(`--write` updates `.env` for you). If your ChatGPT account does not include +the project on conversation summaries, add `--deep` and it reads each +conversation's detail instead — slower, one request per conversation, but +complete. + +Why it matters: project attribution is resolved from each conversation's own +`gizmo_id`, so exports file correctly whether or not a project is configured. +But the *listing* pass still needs `CHATGPT_PROJECT_IDS` — conversations that +live only inside a project never appear in the default conversation list, so an +unlisted project's chats are never fetched at all. Every export run also names +any unconfigured project it encounters. + +To find them by hand instead: + 1. Open ChatGPT and click a Project in the left sidebar 2. Look at the browser URL — it will look like: `https://chatgpt.com/g/g-p-68c2b2b3037c8191890036fb4ae3ed9f-my-project/project` diff --git a/src/main.py b/src/main.py index e718972..d31604e 100644 --- a/src/main.py +++ b/src/main.py @@ -332,6 +332,143 @@ def doctor(ctx: click.Context) -> None: sys.exit(1) +@cli.command() +@click.option( + "--deep", + is_flag=True, + help=( + "Fetch each conversation's detail when the listing does not name its " + "project. Slow (one request per conversation) but complete." + ), +) +@click.option("--write", is_flag=True, help="Write the discovered IDs to .env.") +@click.pass_context +def projects(ctx: click.Context, deep: bool, write: bool) -> None: + """Discover ChatGPT project IDs, including ones missing from your config. + + CHATGPT_PROJECT_IDS has to be maintained by hand, and a project missing + from it is invisible to the listing pass: conversations that live only + inside that project are never fetched at all. This finds the projects your + account actually uses and prints a paste-ready line. + """ + cfg = _load_config_or_exit(ctx.obj["debug"]) + if not cfg.chatgpt_session_token: + console.print("[red]CHATGPT_SESSION_TOKEN is not set — run 'ai-chat-exporter auth'.[/red]") + sys.exit(1) + + from src.providers.chatgpt import ChatGPTProvider + + try: + prov = ChatGPTProvider( + session_token=cfg.chatgpt_session_token, + session_token_1=cfg.chatgpt_session_token_1, + project_ids=cfg.chatgpt_project_ids, + ) + except ProviderError as e: + _handle_provider_error(e, ctx.obj["debug"]) + sys.exit(1) + + configured = list(cfg.chatgpt_project_ids) + console.print(f"\n[bold cyan][CHATGPT][/bold cyan] {len(configured)} project ID(s) configured") + console.print("Listing conversations…") + + try: + summaries = prov.fetch_all_conversations(since=None) + except ProviderError as e: + _handle_provider_error(e, ctx.obj["debug"]) + sys.exit(1) + + found: dict[str, str] = {} + missing_gizmo: list[dict] = [] + for conv in summaries: + gizmo_id = conv.get("gizmo_id") + if gizmo_id and str(gizmo_id).startswith("g-p-"): + found.setdefault(gizmo_id, "") + elif gizmo_id is None: + missing_gizmo.append(conv) + + if found: + console.print( + f" [dim]The listing names a project on {len(summaries) - len(missing_gizmo)} " + f"conversation(s) — no detail fetches needed.[/dim]" + ) + else: + console.print( + " [yellow]The listing does not carry gizmo_id, so projects can only be " + "read from conversation details.[/yellow]" + ) + + if deep and missing_gizmo: + console.print( + f" Fetching detail for {len(missing_gizmo)} conversation(s) " + f"(~{len(missing_gizmo) * cfg.request_delay / 60:.0f} min at " + f"{cfg.request_delay}s pacing)…" + ) + from rich.progress import ( + BarColumn, Progress, SpinnerColumn, TaskProgressColumn, TextColumn, + ) + + with Progress( + SpinnerColumn(), + TextColumn("[progress.description]{task.description}"), + BarColumn(), + TaskProgressColumn(), + console=console, + ) as progress: + task = progress.add_task("Scanning…", total=len(missing_gizmo)) + for conv in missing_gizmo: + conv_id = conv.get("id") or conv.get("conversation_id") + if conv_id: + try: + raw = prov.get_conversation(conv_id) + gizmo_id = raw.get("gizmo_id") + if gizmo_id and str(gizmo_id).startswith("g-p-"): + found.setdefault(gizmo_id, "") + except ProviderError: + pass + progress.advance(task) + elif missing_gizmo and not found: + console.print( + f" [dim]Re-run with --deep to read {len(missing_gizmo)} conversation " + "detail(s). Nothing to report without it.[/dim]" + ) + + for gizmo_id in list(found) + [p for p in configured if p not in found]: + found[gizmo_id] = prov._fetch_project_name(gizmo_id) + + table = Table(title="ChatGPT Projects", show_header=True) + table.add_column("Project", style="cyan") + table.add_column("ID", style="dim") + table.add_column("In .env") + for gizmo_id, name in sorted(found.items(), key=lambda kv: kv[1].lower()): + in_env = gizmo_id in configured + table.add_row( + name, + gizmo_id, + "[green]yes[/green]" if in_env else "[yellow]NO — add it[/yellow]", + ) + console.print(table) + + new_ids = [g for g in found if g not in configured] + if not new_ids: + console.print("[green]Every project found is already configured.[/green]") + return + + combined = configured + new_ids + line = "CHATGPT_PROJECT_IDS=" + ",".join(combined) + console.print( + f"\n[bold]{len(new_ids)} project(s) missing from your config.[/bold] " + "Conversations that live only inside them are not being exported." + ) + console.print("\n[dim]Paste into .env:[/dim]") + console.print(line) + + if write: + _set_env_key("CHATGPT_PROJECT_IDS", ",".join(combined)) + else: + console.print("\n[dim]Or re-run with --write to update .env directly.[/dim]") + + @cli.command() @click.option( "--provider", diff --git a/tests/test_cli.py b/tests/test_cli.py index 0de89a5..c0639b5 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1,5 +1,7 @@ """CLI-level tests using Click's CliRunner — no live API calls required.""" +from pathlib import Path + import pytest from click.testing import CliRunner @@ -297,3 +299,119 @@ class TestCanaryCommand: ) assert result.exit_code == 1 assert "No web-API provider tokens" in result.output + + +class TestProjectsCommand: + """`projects` discovers project IDs missing from CHATGPT_PROJECT_IDS.""" + + def _patch_provider(self, monkeypatch, summaries, details=None, names=None): + import src.providers.chatgpt as chatgpt_mod + + names = names or {} + details = details or {} + + class FakeProvider: + def __init__(self, **kwargs): + self._project_ids = kwargs.get("project_ids") or [] + + def fetch_all_conversations(self, since=None): + return summaries + + def get_conversation(self, conv_id): + return details.get(conv_id, {}) + + def _fetch_project_name(self, gizmo_id): + return names.get(gizmo_id, gizmo_id) + + monkeypatch.setattr(chatgpt_mod, "ChatGPTProvider", FakeProvider) + return FakeProvider + + def _env(self, tmp_path, **extra): + # Clears the ToS gate and the first-run doctor check. + cache = Cache(tmp_path) + cache.acknowledge_tos() + cache.mark_exported("chatgpt", "dummy", {"updated_at": "2024-01-01T00:00:00Z"}) + env = { + "CHATGPT_SESSION_TOKEN": "eyJtesttoken", + "CACHE_DIR": str(tmp_path), + "EXPORT_DIR": str(tmp_path / "exports"), + } + env.update(extra) + return env + + def test_reports_project_absent_from_config(self, tmp_path, monkeypatch): + self._patch_provider( + monkeypatch, + summaries=[{"id": "c1", "gizmo_id": "g-p-missing"}], + names={"g-p-missing": "Tech Questions"}, + ) + result = CliRunner(mix_stderr=True).invoke( + cli, ["--no-log-file", "projects"], env=self._env(tmp_path) + ) + assert result.exit_code == 0 + assert "Tech Questions" in result.output + assert "g-p-missing" in result.output + assert "CHATGPT_PROJECT_IDS=g-p-missing" in result.output + + def test_quiet_when_everything_is_configured(self, tmp_path, monkeypatch): + self._patch_provider( + monkeypatch, + summaries=[{"id": "c1", "gizmo_id": "g-p-known"}], + names={"g-p-known": "Known"}, + ) + result = CliRunner(mix_stderr=True).invoke( + cli, + ["--no-log-file", "projects"], + env=self._env(tmp_path, CHATGPT_PROJECT_IDS="g-p-known"), + ) + assert result.exit_code == 0 + assert "already configured" in result.output + + def test_custom_gpt_ids_are_ignored(self, tmp_path, monkeypatch): + self._patch_provider( + monkeypatch, + summaries=[{"id": "c1", "gizmo_id": "g-notaproject"}], + ) + result = CliRunner(mix_stderr=True).invoke( + cli, ["--no-log-file", "projects"], env=self._env(tmp_path) + ) + assert "g-notaproject" not in result.output + + def test_deep_reads_conversation_details(self, tmp_path, monkeypatch): + self._patch_provider( + monkeypatch, + summaries=[{"id": "c1"}], # listing carries no gizmo_id + details={"c1": {"gizmo_id": "g-p-fromdetail"}}, + names={"g-p-fromdetail": "Found Deep"}, + ) + result = CliRunner(mix_stderr=True).invoke( + cli, ["--no-log-file", "projects", "--deep"], env=self._env(tmp_path) + ) + assert "Found Deep" in result.output + assert "CHATGPT_PROJECT_IDS=g-p-fromdetail" in result.output + + def test_without_deep_says_so_rather_than_reporting_nothing(self, tmp_path, monkeypatch): + self._patch_provider( + monkeypatch, + summaries=[{"id": "c1"}], + details={"c1": {"gizmo_id": "g-p-fromdetail"}}, + ) + result = CliRunner(mix_stderr=True).invoke( + cli, ["--no-log-file", "projects"], env=self._env(tmp_path) + ) + assert "--deep" in result.output + + def test_write_updates_env(self, tmp_path, monkeypatch): + self._patch_provider( + monkeypatch, + summaries=[{"id": "c1", "gizmo_id": "g-p-new"}], + names={"g-p-new": "New Project"}, + ) + runner = CliRunner(mix_stderr=True) + with runner.isolated_filesystem(temp_dir=tmp_path) as fs: + result = runner.invoke( + cli, ["--no-log-file", "projects", "--write"], env=self._env(tmp_path) + ) + assert result.exit_code == 0 + env_text = (Path(fs) / ".env").read_text(encoding="utf-8") + assert "CHATGPT_PROJECT_IDS=g-p-new" in env_text