"""WCAG contrast for the design tokens, computed rather than eyeballed. python -m tools.contrast_audit M8 recorded contrast and visible focus as "checked by eye" and handed the measurement to M11. There are two halves to doing that properly, and this is the cheap half: the palette itself, checked as pairs, with no browser and no dependency. The other half — the colours that actually reach the screen after inheritance, opacity and layering — is measured on the rendered page by `tools/m11_browser.py`, because a token pair says nothing about what a specific element ended up with. Thresholds are WCAG 2.1 AA: 4.5:1 for body text, 3:1 for large text (>=24px, or >=18.66px bold) and for the non-text parts of a control's boundary. """ from __future__ import annotations import re import sys from pathlib import Path TOKENS = Path(__file__).resolve().parent.parent.parent / "frontend/src/styles/tokens.css" #: Foreground/background pairs the design actually puts together. Written out #: rather than combinatorial, because "every colour against every other" reports #: pairs that never meet on screen. #: #: The `kind` records which success criterion a pair is measured against — #: **text** is WCAG 1.4.3 Contrast (Minimum), **boundary** is WCAG 1.4.11 #: Non-text Contrast — and **both are pass/fail**. #: #: v1.1 WP-E overturned the earlier position here, which was that a boundary #: below 3:1 could be recorded rather than failed because "a control is #: identified by its label, not by its edge". That argument understates what #: 1.4.11 asks: the criterion covers the visual information needed to identify #: a component *and its boundary*, and a reader who cannot see where a text box #: ends cannot see that there is a text box to type into, label or no label. #: The edges were at 1.33:1 and 1.75:1 — the tokens were raised instead. #: #: Borders are measured against **every background they are drawn on**, and the #: floor is the worst of them. Inputs and buttons sit on --bg-input, which is #: lighter than --bg-panel and so the harder case; checking only --bg-panel #: would have let a token pass the audit while the real control failed. #: `--bg-panel-glass` is translucent and cannot be resolved from tokens alone; #: that edge is measured on the rendered page by `tools/m11_browser.py`. PAIRS = [ ("text", "--text", "--bg", 4.5, "body text on the page"), ("text", "--text", "--bg-panel", 4.5, "body text in a panel"), ("text", "--text", "--bg-input", 4.5, "text typed into a field"), ("text", "--text-dim", "--bg", 4.5, "secondary text on the page"), ("text", "--text-dim", "--bg-panel", 4.5, "secondary text in a panel"), ("text", "--text-dim", "--bg-panel", 4.5, "a control's own label"), ("text", "--accent", "--bg", 4.5, "accent text on the page"), ("text", "--accent", "--bg-panel", 4.5, "accent text in a panel"), ("text", "--danger", "--bg-panel", 4.5, "an error message"), ("text", "--warning", "--bg-panel", 4.5, "a caution message"), ("text", "--player", "--bg", 4.5, "the player's own words"), ("boundary", "--border", "--bg-input", 3.0, "a field or button's resting edge"), ("boundary", "--border-bright", "--bg-input", 3.0, "a field or button's hover edge"), ("boundary", "--border", "--bg-panel", 3.0, "a control's resting edge in a panel"), ("boundary", "--border-bright", "--bg-panel", 3.0, "a control's hover edge in a panel"), ("boundary", "--border", "--bg", 3.0, "a divider on the page"), ("boundary", "--border-bright", "--bg", 3.0, "the scrollbar thumb"), ("boundary", "--accent-dim", "--bg-input", 3.0, "a focused field's edge"), ("boundary", "--accent-dim", "--bg-panel", 3.0, "a focused control's edge in a panel"), ("boundary", "--chart-1", "--bg-panel", 3.0, "a chart bar"), ("boundary", "--chart-2", "--bg-panel", 3.0, "a chart bar"), ("boundary", "--chart-3", "--bg-panel", 3.0, "a chart bar"), ] def read_tokens(path: Path) -> dict[str, str]: found = {} for name, value in re.findall(r"(--[\w-]+):\s*(#[0-9a-fA-F]{6})\s*;", path.read_text()): found[name] = value return found def luminance(hex_colour: str) -> float: r, g, b = (int(hex_colour[i:i + 2], 16) / 255 for i in (1, 3, 5)) def channel(value: float) -> float: return value / 12.92 if value <= 0.03928 else ((value + 0.055) / 1.055) ** 2.4 r, g, b = channel(r), channel(g), channel(b) return 0.2126 * r + 0.7152 * g + 0.0722 * b def ratio(a: str, b: str) -> float: la, lb = luminance(a), luminance(b) high, low = max(la, lb), min(la, lb) return (high + 0.05) / (low + 0.05) def main() -> int: tokens = read_tokens(TOKENS) # Named defensively: the tests run this against a temporary tokens file, # which need not sit four directories deep the way the real one does. label = TOKENS.name if len(TOKENS.parents) > 3: label = TOKENS.relative_to(TOKENS.parents[3]) print(f"{label}: {len(tokens)} colour tokens\n") print(f"{'pair':44} {'kind':9} {'ratio':>7} {'floor':>6} verdict") print("-" * 82) text_failures, boundary_failures = 0, 0 for kind, foreground, background, floor, description in PAIRS: if foreground not in tokens or background not in tokens: print(f"{description:44} {kind:9} {'—':>7} {floor:>6.1f} MISSING TOKEN") text_failures += 1 continue measured = ratio(tokens[foreground], tokens[background]) # Rounded to the two decimals printed, so the verdict matches what the # reader is shown: a pair reported as 3.00:1 is not failed for arithmetic # the output does not display. if round(measured, 2) >= floor: verdict = "pass" else: verdict = "FAIL" if kind == "text": text_failures += 1 else: boundary_failures += 1 print(f"{description:44} {kind:9} {measured:>6.2f}:1 {floor:>6.1f} {verdict}") print() if text_failures: print(f"{text_failures} text pair(s) below WCAG AA (1.4.3) — this is a defect") else: print("every text pair clears WCAG AA (1.4.3)") if boundary_failures: print(f"{boundary_failures} boundary pair(s) below 3:1 (WCAG 1.4.11) — " "this is a defect") else: print("every control boundary clears 3:1 (1.4.11)") return 1 if (text_failures or boundary_failures) else 0 if __name__ == "__main__": raise SystemExit(main())