fix: read project attribution from the conversation's gizmo_id

A conversation in a project absent from CHATGPT_PROJECT_IDS exported into
no-project/ even though its own payload names the project. Found while
investigating the media 403s: bpi-f3-case-options sits in
g-p-6a4edf5160848191927b05da20f49151, which is not among the 13 configured
ids, so it filed under no-project.2026.

Both existing sources are bounded by what the user configured. The detail
response is not: it carries gizmo_id. Use it as a third fallback, after the
listing annotation and the project map, and cache the result into the map.
Attribution now stays correct with no list to maintain, and moving a chat
into a new project stops silently misfiling it.

Only g-p- ids are treated as projects — a custom GPT is not a project and
must not become a folder.

CHATGPT_PROJECT_IDS still matters for the listing pass: conversations that
live only inside a project never appear in the default listing, so an
unconfigured project's chats can be missed entirely. Each one is now
reported once per run with the id to add, which turns "some chats are
missing" into a line to paste.

8 tests cover precedence, the g-p- guard, map caching and the once-per-run
report. 318 pass.
This commit is contained in:
JesseMarkowitz
2026-08-17 12:49:04 -04:00
parent dfa0645fba
commit 710889b65f
3 changed files with 138 additions and 5 deletions
+3
View File
@@ -11,6 +11,9 @@ Format follows [Keep a Changelog](https://keepachangelog.com/en/1.0.0/).
- **`redact_secrets` missed compound key names.** It matched keys exactly, so `access_token`, `api_key`, and `session-token` passed through un-redacted into debug-logged response bodies; matching now applies per word ("keywords", "monkey", "tokenizer" stay intact). - **`redact_secrets` missed compound key names.** It matched keys exactly, so `access_token`, `api_key`, and `session-token` passed through un-redacted into debug-logged response bodies; matching now applies per word ("keywords", "monkey", "tokenizer" stay intact).
- **`tests/test_config.py::TestSessionLimiterConfig::test_defaults` depended on the developer's `.env`.** `load_config()` calls `load_dotenv(override=False)`, which re-populated the variable the test had just deleted — so it passed only on a machine with no `.env`. The test now stubs dotenv discovery. - **`tests/test_config.py::TestSessionLimiterConfig::test_defaults` depended on the developer's `.env`.** `load_config()` calls `load_dotenv(override=False)`, which re-populated the variable the test had just deleted — so it passed only on a machine with no `.env`. The test now stubs dotenv discovery.
### Added
- **Project attribution now reads the conversation's own `gizmo_id`.** Previously the project name came only from `CHATGPT_PROJECT_IDS`, so a conversation in a project you had not listed exported into `no-project/` even though its payload names its project. The detail response carries `gizmo_id`, so it is used as a fallback after the listing annotation and the project map — attribution stays correct without maintaining a list, and moving a chat into a new project no longer silently misfiles it. Only `g-p-` ids count: a custom GPT is not a project and must not become a folder. Each unconfigured project is reported once per run, naming the id to add, because the *listing* pass still needs `CHATGPT_PROJECT_IDS` — conversations that live only inside a project never appear in the default listing.
### Changed ### Changed
- Media download failures are bucketed as `forbidden` (403 — the file record survives) separately from `download-error`, so the run summary distinguishes it from `expired-or-missing` (404). - Media download failures are bucketed as `forbidden` (403 — the file record survives) separately from `download-error`, so the run summary distinguishes it from `expired-or-missing` (404).
+55 -5
View File
@@ -401,6 +401,34 @@ class ChatGPTProvider(BaseProvider):
self._project_name_cache[project_id] = name self._project_name_cache[project_id] = name
return name return name
def _note_unconfigured_project(self, gizmo_id: str, name: str) -> None:
"""Report a project discovered from a conversation but absent from config.
Attribution now works without CHATGPT_PROJECT_IDS, but the listing pass
still uses it: project-only conversations never appear in the default
listing, so an unconfigured project's chats are exported only if they
surface some other way. Naming the gap once per run is the difference
between "some chats are missing" and knowing which line to add.
"""
configured = getattr(self, "_project_ids", None) or []
if gizmo_id in configured:
return
seen = getattr(self, "_unconfigured_projects", None)
if seen is None:
seen = set()
self._unconfigured_projects = seen
if gizmo_id in seen:
return
seen.add(gizmo_id)
logger.info(
"[chatgpt] Project '%s' (%s) is not in CHATGPT_PROJECT_IDS — "
"attribution resolved from the conversation, but conversations "
"that live only in this project are not being listed. Add it to "
"CHATGPT_PROJECT_IDS to fetch them.",
name,
gizmo_id,
)
def list_project_conversations( def list_project_conversations(
self, project_id: str, cursor: str = "0" self, project_id: str, cursor: str = "0"
) -> tuple[list[dict], str | None]: ) -> tuple[list[dict], str | None]:
@@ -797,15 +825,37 @@ class ChatGPTProvider(BaseProvider):
updated_at = _ts_to_iso(raw.get("update_time")) updated_at = _ts_to_iso(raw.get("update_time"))
# Prefer _project_name annotation injected from the listing summary # Prefer _project_name annotation injected from the listing summary
# (propagated by the export loop). Fall back to _project_map lookup. # (propagated by the export loop), then the _project_map built from
project = raw.get("_project_name") or ( # CHATGPT_PROJECT_IDS, then the conversation's own gizmo_id.
self._project_map.get(conv_id) if conv_id else None #
) # That last fallback matters: both earlier sources are limited to the
# projects the user configured, so a conversation in an unconfigured
# project still exports (it appears in the default listing) but landed
# under no-project. The detail payload names its own project, so read
# it from there and the attribution stays correct without the user
# having to maintain a list. Observed 2026-08-17: a conversation whose
# gizmo_id was absent from CHATGPT_PROJECT_IDS filed under no-project.
source = "_project_name"
project = raw.get("_project_name")
if not project and conv_id:
project = self._project_map.get(conv_id)
source = "_project_map"
if not project:
gizmo_id = raw.get("gizmo_id")
# gizmo_type distinguishes a project from a custom GPT; only a
# project should become a folder. Absent on older payloads, so
# treat "unknown but has a project-shaped id" as a project.
if gizmo_id and str(gizmo_id).startswith("g-p-"):
project = self._fetch_project_name(gizmo_id)
source = "gizmo_id"
if conv_id:
self._project_map[conv_id] = project
self._note_unconfigured_project(gizmo_id, project)
logger.debug( logger.debug(
"[chatgpt] normalize_conversation[%s]: project=%r (source=%s)", "[chatgpt] normalize_conversation[%s]: project=%r (source=%s)",
conv_id[:8] if conv_id else "?", conv_id[:8] if conv_id else "?",
project, project,
"_project_name" if raw.get("_project_name") else "_project_map", source,
) )
mapping: dict = raw.get("mapping", {}) mapping: dict = raw.get("mapping", {})
+80
View File
@@ -1299,3 +1299,83 @@ class TestDeletedAssetReporting:
), ),
) )
assert _classify_failure(err) == "expired-or-missing" assert _classify_failure(err) == "expired-or-missing"
# ---------------------------------------------------------------------------
# Project attribution: a conversation names its own project via gizmo_id, so
# it should not depend on the user having listed that project in config.
# ---------------------------------------------------------------------------
class TestProjectAttribution:
def _provider(self, *, project_ids=None, gizmo_name="My Project"):
from src.providers.chatgpt import ChatGPTProvider
p = ChatGPTProvider.__new__(ChatGPTProvider)
p._project_map = {}
p._project_name_cache = {}
p._project_ids = project_ids or []
p._hidden_content = "placeholder"
p._fetch_project_name = lambda gid: gizmo_name
return p
def _raw(self, **extra):
raw = {
"conversation_id": "conv-1",
"title": "T",
"create_time": 1_700_000_000,
"update_time": 1_700_000_000,
"mapping": {},
}
raw.update(extra)
return raw
def test_gizmo_id_supplies_project_when_unconfigured(self):
p = self._provider(gizmo_name="Tech Questions")
out = p.normalize_conversation(self._raw(gizmo_id="g-p-abc123"))
assert out["project"] == "Tech Questions"
def test_explicit_annotation_still_wins(self):
p = self._provider(gizmo_name="From Gizmo")
out = p.normalize_conversation(
self._raw(gizmo_id="g-p-abc123", _project_name="From Listing")
)
assert out["project"] == "From Listing"
def test_project_map_beats_gizmo_lookup(self):
p = self._provider(gizmo_name="From Gizmo")
p._project_map["conv-1"] = "From Map"
out = p.normalize_conversation(self._raw(gizmo_id="g-p-abc123"))
assert out["project"] == "From Map"
def test_custom_gpt_is_not_treated_as_a_project(self):
"""Only g-p- ids are projects; a custom GPT must not become a folder."""
p = self._provider()
out = p.normalize_conversation(self._raw(gizmo_id="g-xyz789"))
assert out["project"] is None
def test_no_gizmo_id_stays_unprojected(self):
p = self._provider()
assert p.normalize_conversation(self._raw())["project"] is None
def test_resolved_project_is_cached_into_the_map(self):
p = self._provider(gizmo_name="Cached")
p.normalize_conversation(self._raw(gizmo_id="g-p-abc123"))
assert p._project_map["conv-1"] == "Cached"
def test_unconfigured_project_is_reported_once(self, caplog):
p = self._provider(project_ids=["g-p-known"], gizmo_name="Surprise")
with caplog.at_level(logging.INFO):
p.normalize_conversation(self._raw(gizmo_id="g-p-surprise"))
p.normalize_conversation(
self._raw(conversation_id="conv-2", gizmo_id="g-p-surprise")
)
hits = [r for r in caplog.records if "not in CHATGPT_PROJECT_IDS" in r.message]
assert len(hits) == 1
assert "Surprise" in hits[0].message
def test_configured_project_is_not_reported(self, caplog):
p = self._provider(project_ids=["g-p-known"], gizmo_name="Known")
with caplog.at_level(logging.INFO):
p.normalize_conversation(self._raw(gizmo_id="g-p-known"))
assert not [r for r in caplog.records if "not in CHATGPT_PROJECT_IDS" in r.message]