fix: surface the response body on 4xx so 403s are diagnosable

Media downloads logged "HTTP Error 403:" with no reason. That string is
curl_cffi's raise_for_status() format, "HTTP Error {code}: {reason}", and
HTTP/2 carries no reason phrase — so the message said nothing, and
_make_request threw the response body away. The provider's JSON `detail`
is the only explanation available for a refused asset.

- base._make_request: end non-retryable statuses with a ProviderError
  carrying the body's detail/error/message (redacted, truncated to 300
  chars) instead of a bare raise_for_status().
- media: bucket 403 as `forbidden` in the run summary, separately from
  `download-error` — "the asset is gone" and "we were refused" are
  different problems.
- utils.redact_secrets: match secret key names per word. Exact matching
  let access_token, api_key, and session-token through into logged
  bodies; "keywords"/"monkey"/"tokenizer" stay intact.
- tests/test_config.py: test_defaults depended on the absence of a local
  .env — load_config() calls load_dotenv(override=False), which restored
  the variable the test had just deleted. Stub dotenv discovery.

305 tests pass.
This commit is contained in:
JesseMarkowitz
2026-08-17 07:58:02 -04:00
parent 1f5a445ada
commit 395ea19ca8
8 changed files with 193 additions and 8 deletions
+16 -4
View File
@@ -131,10 +131,7 @@ def resolve_media(
logger.warning(
"[media] Could not download %s: %s", ref[:60], e.original
)
reason = "expired-or-missing" if "404" in str(e.original) or "not found" in str(
e.original
).lower() else "download-error"
report.record_media_failed(reason)
report.record_media_failed(_classify_failure(e))
continue
ext = _pick_extension(mime, file_name)
@@ -153,6 +150,21 @@ def resolve_media(
return downloaded
def _classify_failure(error: ProviderError) -> str:
"""Bucket a download failure for the run summary.
The buckets separate "the asset is gone" from "the asset is there but we
were refused" — different causes, different fixes, so lumping both into
download-error hides which one you have.
"""
detail = str(error.original).lower()
if "404" in detail or "not found" in detail:
return "expired-or-missing"
if "403" in detail or "forbidden" in detail:
return "forbidden"
return "download-error"
def _safe_asset_name(provider, ref: str) -> str | None:
"""A stable, filesystem-safe name for the asset (the provider file ID)."""
parser = getattr(provider, "parse_asset_file_id", None)
+42 -1
View File
@@ -33,6 +33,9 @@ logger = logging.getLogger(__name__)
# Request timeouts (connect, read) in seconds
REQUEST_TIMEOUT = (10, 30)
# Longest error-body excerpt to carry into a ProviderError message.
_ERROR_BODY_CHARS = 300
# Retry configuration
MAX_RETRIES = 3
BACKOFF_BASE = 2.0
@@ -115,6 +118,32 @@ def resolve_request_delay() -> float:
return DEFAULT_REQUEST_DELAY
return value
def _describe_error_body(response: Any) -> str:
"""Summarise an error response body for a log line.
Providers explain 4xx in the body (ChatGPT uses ``detail``), so a bare
status code is not a diagnosis. Prefers ``detail``/``error``/``message``,
falls back to a truncated raw excerpt, and redacts before it is logged.
"""
try:
body = response.json()
except Exception:
try:
text = (response.text or "").strip()
except Exception:
return "no response body"
if not text:
return "empty response body"
return f"body: {text[:_ERROR_BODY_CHARS]}"
if isinstance(body, dict):
for key in ("detail", "error", "message"):
if key in body:
value = redact_secrets(body[key])
return f"{key}: {str(value)[:_ERROR_BODY_CHARS]}"
return f"body: {str(redact_secrets(body))[:_ERROR_BODY_CHARS]}"
# Realistic Chrome User-Agent
USER_AGENT = (
"Mozilla/5.0 (X11; Linux x86_64) "
@@ -384,7 +413,19 @@ class BaseProvider(ABC):
continue
# ── Other HTTP errors ──────────────────────────────────────
response.raise_for_status()
# Raise with the response body attached. curl_cffi formats
# raise_for_status() as "HTTP Error {code}: {reason}", and
# HTTP/2 carries no reason phrase — so the bare exception
# reads "HTTP Error 403:" and says nothing about the cause.
# The provider's JSON `detail` is the only explanation there is.
if not response.ok:
raise ProviderError(
self.provider_name,
f"{method} {url}",
RuntimeError(
f"HTTP {response.status_code} — {_describe_error_body(response)}"
),
)
# ── Success ────────────────────────────────────────────────
body = response.json()
+19 -3
View File
@@ -76,11 +76,27 @@ def build_export_path(
return base_dir.joinpath(*parts) / filename
def _is_sensitive_key(key: object) -> bool:
"""True if a mapping key names a secret.
Matches the whole key and each of its underscore/dash-separated words, so
compound names carry too: exact-match alone let ``access_token`` and
``api_key`` through into logged response bodies. Word-level matching keeps
innocent keys ("keywords", "monkey") intact.
"""
if not isinstance(key, str):
return False
lowered = key.lower()
if lowered in _SENSITIVE_KEYS:
return True
return any(part in _SENSITIVE_KEYS for part in re.split(r"[^a-z0-9]+", lowered))
def redact_secrets(data: object) -> object:
"""Recursively redact sensitive values from a dict/list for safe logging.
Keys matching _SENSITIVE_KEYS (case-insensitive) have their values
replaced with "[REDACTED]".
Keys naming a secret (see _is_sensitive_key) have their values replaced
with "[REDACTED]".
Args:
data: Any JSON-serializable object.
@@ -90,7 +106,7 @@ def redact_secrets(data: object) -> object:
"""
if isinstance(data, dict):
return {
k: "[REDACTED]" if k.lower() in _SENSITIVE_KEYS else redact_secrets(v)
k: "[REDACTED]" if _is_sensitive_key(k) else redact_secrets(v)
for k, v in data.items()
}
if isinstance(data, list):