fix: surface the response body on 4xx so 403s are diagnosable

Media downloads logged "HTTP Error 403:" with no reason. That string is
curl_cffi's raise_for_status() format, "HTTP Error {code}: {reason}", and
HTTP/2 carries no reason phrase — so the message said nothing, and
_make_request threw the response body away. The provider's JSON `detail`
is the only explanation available for a refused asset.

- base._make_request: end non-retryable statuses with a ProviderError
  carrying the body's detail/error/message (redacted, truncated to 300
  chars) instead of a bare raise_for_status().
- media: bucket 403 as `forbidden` in the run summary, separately from
  `download-error` — "the asset is gone" and "we were refused" are
  different problems.
- utils.redact_secrets: match secret key names per word. Exact matching
  let access_token, api_key, and session-token through into logged
  bodies; "keywords"/"monkey"/"tokenizer" stay intact.
- tests/test_config.py: test_defaults depended on the absence of a local
  .env — load_config() calls load_dotenv(override=False), which restored
  the variable the test had just deleted. Stub dotenv discovery.

305 tests pass.
This commit is contained in:
JesseMarkowitz
2026-08-17 07:58:02 -04:00
parent 1f5a445ada
commit 395ea19ca8
8 changed files with 193 additions and 8 deletions
+6
View File
@@ -60,7 +60,13 @@ class TestSessionLimiterConfig:
"""MAX_CONVERSATIONS_PER_RUN and REQUEST_DELAY parsing in load_config."""
def _load(self, monkeypatch, tmp_path, **env):
from src import config as config_module
from src.config import load_config
# load_config() calls load_dotenv(override=False), which re-populates
# any variable this test just deleted from the developer's real .env —
# so test_defaults only saw defaults on a machine without one. Stub it:
# these tests are about parsing the environment, not discovering .env.
monkeypatch.setattr(config_module, "load_dotenv", lambda *a, **k: False)
monkeypatch.setenv("EXPORT_DIR", str(tmp_path / "exports"))
monkeypatch.setenv("CACHE_DIR", str(tmp_path / "cache"))
for key in (
+16
View File
@@ -170,6 +170,22 @@ class TestResolveMedia:
# Still renders as a placeholder, not a broken image link
assert render_blocks_to_markdown([block]).startswith("> 🖼️")
def test_forbidden_counted_separately_from_generic_error(self, tmp_path):
"""403 is a distinct bucket: the asset exists, we were refused."""
ref = "sediment://file_denied"
provider = _FakeProvider(
fail_refs={ref: RuntimeError("HTTP 403 — detail: unauthorized")}
)
block = make_image_placeholder(ref=ref, source="model_generated")
report = LossReport()
resolve_media(
_conv_with([block]), provider, tmp_path, "provider/project/year",
"images", report,
)
assert report.media_failed["forbidden"] == 1
assert "download-error" not in report.media_failed
def test_provider_without_download_asset(self, tmp_path):
"""claude-code has no remote assets — resolve_media must no-op."""
class NoDownload:
+66
View File
@@ -1116,3 +1116,69 @@ class TestClaudeDriftCanary:
p = self._provider([{"uuid": "u1", "name": "N", "updated_at": "z"}],
self._detail([]))
assert DRIFT_ERROR in _sev(p.check_drift())
# ---------------------------------------------------------------------------
# 4xx diagnostics: curl_cffi renders raise_for_status() as
# "HTTP Error {code}: {reason}", and HTTP/2 has no reason phrase — so a bare
# 403 logged as "HTTP Error 403:" says nothing. The body carries the cause.
# ---------------------------------------------------------------------------
class TestErrorBodyDiagnostics:
class _Resp:
ok = False
status_code = 403
reason = ""
headers: dict = {}
def __init__(self, payload=None, text=""):
self._payload = payload
self.text = text
def json(self):
if self._payload is None:
raise ValueError("not json")
return self._payload
def _provider(self, response):
from src.providers.chatgpt import ChatGPTProvider
p = ChatGPTProvider.__new__(ChatGPTProvider)
p._request_delay = 0
p._last_request_at = None
p._session = type("S", (), {"request": lambda *a, **k: response})()
return p
def test_detail_field_surfaces_in_error(self):
from src.providers.base import ProviderError
resp = self._Resp(payload={"detail": "File not accessible to this account"})
with pytest.raises(ProviderError) as exc:
self._provider(resp)._make_request("GET", "https://x/files/f1/download")
message = str(exc.value.original)
assert "403" in message
assert "File not accessible to this account" in message
def test_non_json_body_excerpted(self):
from src.providers.base import ProviderError
resp = self._Resp(text="<html>Forbidden</html>")
with pytest.raises(ProviderError) as exc:
self._provider(resp)._make_request("GET", "https://x/files/f1/download")
assert "Forbidden" in str(exc.value.original)
def test_empty_body_says_so_rather_than_nothing(self):
from src.providers.base import ProviderError
resp = self._Resp(text="")
with pytest.raises(ProviderError) as exc:
self._provider(resp)._make_request("GET", "https://x/files/f1/download")
assert "empty response body" in str(exc.value.original)
def test_secrets_in_error_body_are_redacted(self):
from src.providers.base import _describe_error_body
resp = self._Resp(payload={"error": {"message": "no", "access_token": "sk-abc"}})
described = _describe_error_body(resp)
assert "sk-abc" not in described
assert "[REDACTED]" in described
+18
View File
@@ -145,3 +145,21 @@ class TestFormatTokenStatus:
expiry = datetime.now(tz=timezone.utc) + timedelta(days=10, hours=12)
result = format_token_status("tok", expiry)
assert "10 days" in result
class TestRedactCompoundKeys:
"""Exact-match redaction let compound secret names through into logs."""
def test_compound_secret_keys_redacted(self):
result = redact_secrets(
{"access_token": "sk-abc", "api_key": "k1", "session-token": "s1"}
)
assert result == {
"access_token": "[REDACTED]",
"api_key": "[REDACTED]",
"session-token": "[REDACTED]",
}
def test_innocent_keys_containing_a_secret_word_kept(self):
result = redact_secrets({"keywords": ["a"], "monkey": "b", "tokenizer": "c"})
assert result == {"keywords": ["a"], "monkey": "b", "tokenizer": "c"}