From 710889b65fad6d2635274841f11ea4c219ace486 Mon Sep 17 00:00:00 2001 From: JesseMarkowitz Date: Mon, 17 Aug 2026 12:49:04 -0400 Subject: [PATCH] fix: read project attribution from the conversation's gizmo_id MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A conversation in a project absent from CHATGPT_PROJECT_IDS exported into no-project/ even though its own payload names the project. Found while investigating the media 403s: bpi-f3-case-options sits in g-p-6a4edf5160848191927b05da20f49151, which is not among the 13 configured ids, so it filed under no-project.2026. Both existing sources are bounded by what the user configured. The detail response is not: it carries gizmo_id. Use it as a third fallback, after the listing annotation and the project map, and cache the result into the map. Attribution now stays correct with no list to maintain, and moving a chat into a new project stops silently misfiling it. Only g-p- ids are treated as projects — a custom GPT is not a project and must not become a folder. CHATGPT_PROJECT_IDS still matters for the listing pass: conversations that live only inside a project never appear in the default listing, so an unconfigured project's chats can be missed entirely. Each one is now reported once per run with the id to add, which turns "some chats are missing" into a line to paste. 8 tests cover precedence, the g-p- guard, map caching and the once-per-run report. 318 pass. --- CHANGELOG.md | 3 ++ src/providers/chatgpt.py | 60 +++++++++++++++++++++++++++--- tests/test_providers.py | 80 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 138 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ecd1962..a225047 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,6 +11,9 @@ Format follows [Keep a Changelog](https://keepachangelog.com/en/1.0.0/). - **`redact_secrets` missed compound key names.** It matched keys exactly, so `access_token`, `api_key`, and `session-token` passed through un-redacted into debug-logged response bodies; matching now applies per word ("keywords", "monkey", "tokenizer" stay intact). - **`tests/test_config.py::TestSessionLimiterConfig::test_defaults` depended on the developer's `.env`.** `load_config()` calls `load_dotenv(override=False)`, which re-populated the variable the test had just deleted — so it passed only on a machine with no `.env`. The test now stubs dotenv discovery. +### Added +- **Project attribution now reads the conversation's own `gizmo_id`.** Previously the project name came only from `CHATGPT_PROJECT_IDS`, so a conversation in a project you had not listed exported into `no-project/` even though its payload names its project. The detail response carries `gizmo_id`, so it is used as a fallback after the listing annotation and the project map — attribution stays correct without maintaining a list, and moving a chat into a new project no longer silently misfiles it. Only `g-p-` ids count: a custom GPT is not a project and must not become a folder. Each unconfigured project is reported once per run, naming the id to add, because the *listing* pass still needs `CHATGPT_PROJECT_IDS` — conversations that live only inside a project never appear in the default listing. + ### Changed - Media download failures are bucketed as `forbidden` (403 — the file record survives) separately from `download-error`, so the run summary distinguishes it from `expired-or-missing` (404). diff --git a/src/providers/chatgpt.py b/src/providers/chatgpt.py index c7a7752..fbd3ffa 100644 --- a/src/providers/chatgpt.py +++ b/src/providers/chatgpt.py @@ -401,6 +401,34 @@ class ChatGPTProvider(BaseProvider): self._project_name_cache[project_id] = name return name + def _note_unconfigured_project(self, gizmo_id: str, name: str) -> None: + """Report a project discovered from a conversation but absent from config. + + Attribution now works without CHATGPT_PROJECT_IDS, but the listing pass + still uses it: project-only conversations never appear in the default + listing, so an unconfigured project's chats are exported only if they + surface some other way. Naming the gap once per run is the difference + between "some chats are missing" and knowing which line to add. + """ + configured = getattr(self, "_project_ids", None) or [] + if gizmo_id in configured: + return + seen = getattr(self, "_unconfigured_projects", None) + if seen is None: + seen = set() + self._unconfigured_projects = seen + if gizmo_id in seen: + return + seen.add(gizmo_id) + logger.info( + "[chatgpt] Project '%s' (%s) is not in CHATGPT_PROJECT_IDS — " + "attribution resolved from the conversation, but conversations " + "that live only in this project are not being listed. Add it to " + "CHATGPT_PROJECT_IDS to fetch them.", + name, + gizmo_id, + ) + def list_project_conversations( self, project_id: str, cursor: str = "0" ) -> tuple[list[dict], str | None]: @@ -797,15 +825,37 @@ class ChatGPTProvider(BaseProvider): updated_at = _ts_to_iso(raw.get("update_time")) # Prefer _project_name annotation injected from the listing summary - # (propagated by the export loop). Fall back to _project_map lookup. - project = raw.get("_project_name") or ( - self._project_map.get(conv_id) if conv_id else None - ) + # (propagated by the export loop), then the _project_map built from + # CHATGPT_PROJECT_IDS, then the conversation's own gizmo_id. + # + # That last fallback matters: both earlier sources are limited to the + # projects the user configured, so a conversation in an unconfigured + # project still exports (it appears in the default listing) but landed + # under no-project. The detail payload names its own project, so read + # it from there and the attribution stays correct without the user + # having to maintain a list. Observed 2026-08-17: a conversation whose + # gizmo_id was absent from CHATGPT_PROJECT_IDS filed under no-project. + source = "_project_name" + project = raw.get("_project_name") + if not project and conv_id: + project = self._project_map.get(conv_id) + source = "_project_map" + if not project: + gizmo_id = raw.get("gizmo_id") + # gizmo_type distinguishes a project from a custom GPT; only a + # project should become a folder. Absent on older payloads, so + # treat "unknown but has a project-shaped id" as a project. + if gizmo_id and str(gizmo_id).startswith("g-p-"): + project = self._fetch_project_name(gizmo_id) + source = "gizmo_id" + if conv_id: + self._project_map[conv_id] = project + self._note_unconfigured_project(gizmo_id, project) logger.debug( "[chatgpt] normalize_conversation[%s]: project=%r (source=%s)", conv_id[:8] if conv_id else "?", project, - "_project_name" if raw.get("_project_name") else "_project_map", + source, ) mapping: dict = raw.get("mapping", {}) diff --git a/tests/test_providers.py b/tests/test_providers.py index 7a1e6c9..3ba3722 100644 --- a/tests/test_providers.py +++ b/tests/test_providers.py @@ -1299,3 +1299,83 @@ class TestDeletedAssetReporting: ), ) assert _classify_failure(err) == "expired-or-missing" + + +# --------------------------------------------------------------------------- +# Project attribution: a conversation names its own project via gizmo_id, so +# it should not depend on the user having listed that project in config. +# --------------------------------------------------------------------------- + + +class TestProjectAttribution: + def _provider(self, *, project_ids=None, gizmo_name="My Project"): + from src.providers.chatgpt import ChatGPTProvider + + p = ChatGPTProvider.__new__(ChatGPTProvider) + p._project_map = {} + p._project_name_cache = {} + p._project_ids = project_ids or [] + p._hidden_content = "placeholder" + p._fetch_project_name = lambda gid: gizmo_name + return p + + def _raw(self, **extra): + raw = { + "conversation_id": "conv-1", + "title": "T", + "create_time": 1_700_000_000, + "update_time": 1_700_000_000, + "mapping": {}, + } + raw.update(extra) + return raw + + def test_gizmo_id_supplies_project_when_unconfigured(self): + p = self._provider(gizmo_name="Tech Questions") + out = p.normalize_conversation(self._raw(gizmo_id="g-p-abc123")) + assert out["project"] == "Tech Questions" + + def test_explicit_annotation_still_wins(self): + p = self._provider(gizmo_name="From Gizmo") + out = p.normalize_conversation( + self._raw(gizmo_id="g-p-abc123", _project_name="From Listing") + ) + assert out["project"] == "From Listing" + + def test_project_map_beats_gizmo_lookup(self): + p = self._provider(gizmo_name="From Gizmo") + p._project_map["conv-1"] = "From Map" + out = p.normalize_conversation(self._raw(gizmo_id="g-p-abc123")) + assert out["project"] == "From Map" + + def test_custom_gpt_is_not_treated_as_a_project(self): + """Only g-p- ids are projects; a custom GPT must not become a folder.""" + p = self._provider() + out = p.normalize_conversation(self._raw(gizmo_id="g-xyz789")) + assert out["project"] is None + + def test_no_gizmo_id_stays_unprojected(self): + p = self._provider() + assert p.normalize_conversation(self._raw())["project"] is None + + def test_resolved_project_is_cached_into_the_map(self): + p = self._provider(gizmo_name="Cached") + p.normalize_conversation(self._raw(gizmo_id="g-p-abc123")) + assert p._project_map["conv-1"] == "Cached" + + def test_unconfigured_project_is_reported_once(self, caplog): + p = self._provider(project_ids=["g-p-known"], gizmo_name="Surprise") + with caplog.at_level(logging.INFO): + p.normalize_conversation(self._raw(gizmo_id="g-p-surprise")) + p.normalize_conversation( + self._raw(conversation_id="conv-2", gizmo_id="g-p-surprise") + ) + hits = [r for r in caplog.records if "not in CHATGPT_PROJECT_IDS" in r.message] + assert len(hits) == 1 + assert "Surprise" in hits[0].message + + def test_configured_project_is_not_reported(self, caplog): + p = self._provider(project_ids=["g-p-known"], gizmo_name="Known") + with caplog.at_level(logging.INFO): + p.normalize_conversation(self._raw(gizmo_id="g-p-known")) + assert not [r for r in caplog.records if "not in CHATGPT_PROJECT_IDS" in r.message]