v1.1: harden recovery and control boundaries

WP-D and WP-E complete the planned v1.1 implementation packages.

WP-D — recovery honesty:
- backups verify the completed copy with PRAGMA integrity_check
- corruption missed by quick_check is detected by the full check
- existing good backups remain protected
- oversized exports are still delivered but declare whether this version can
  import them, while the 20 MB import limit remains unchanged
- backup was exercised through the real browser UI on both the normal campaign
  database and a campaign-shaped database over 100 MB

WP-E — control-boundary contrast:
- interactive control boundaries meet the WCAG 1.4.11 3:1 target
- the contrast audit is now a failing gate rather than an advisory
- rendered browser measurements pass for the composer, controls, tabs and nav
- text contrast and focus visibility remain intact
- owner reviewed and approved the before/after screenshots

Reports:
- planning/reports/v1.1/V1.1-WP-D-REPORT.md
- planning/reports/v1.1/V1.1-WP-E-REPORT.md

All planned v1.1 work packages A-E are now complete. Release validation has not
yet begun.
This commit is contained in:
JesseMarkowitz
2026-09-16 05:37:13 -04:00
parent 59b5ebc2d8
commit 87a40326a2
21 changed files with 2401 additions and 60 deletions
+52 -31
View File
@@ -26,16 +26,24 @@ TOKENS = Path(__file__).resolve().parent.parent.parent / "frontend/src/styles/to
#: rather than combinatorial, because "every colour against every other" reports
#: pairs that never meet on screen.
#:
#: The `kind` matters and is not a way of grading on a curve. **text** pairs are
#: WCAG 1.4.3 Contrast (Minimum) and are what §21 of the M11 brief asks about;
#: they are pass/fail. **boundary** pairs are WCAG 1.4.11 Non-text Contrast,
#: which applies to "visual information required to identify user interface
#: components" — and in this design a control is identified by its *label*,
#: which is measured above and passes, not by its edge. So a boundary below 3:1
#: is reported with its number and does not fail the run; what it would take to
#: turn it into a real failure is a control with no visible label, and there is
#: no such control (`tools/m11_browser.py` asserts every visible control has an
#: accessible name, and the story controls are text buttons).
#: The `kind` records which success criterion a pair is measured against —
#: **text** is WCAG 1.4.3 Contrast (Minimum), **boundary** is WCAG 1.4.11
#: Non-text Contrast — and **both are pass/fail**.
#:
#: v1.1 WP-E overturned the earlier position here, which was that a boundary
#: below 3:1 could be recorded rather than failed because "a control is
#: identified by its label, not by its edge". That argument understates what
#: 1.4.11 asks: the criterion covers the visual information needed to identify
#: a component *and its boundary*, and a reader who cannot see where a text box
#: ends cannot see that there is a text box to type into, label or no label.
#: The edges were at 1.33:1 and 1.75:1 — the tokens were raised instead.
#:
#: Borders are measured against **every background they are drawn on**, and the
#: floor is the worst of them. Inputs and buttons sit on --bg-input, which is
#: lighter than --bg-panel and so the harder case; checking only --bg-panel
#: would have let a token pass the audit while the real control failed.
#: `--bg-panel-glass` is translucent and cannot be resolved from tokens alone;
#: that edge is measured on the rendered page by `tools/m11_browser.py`.
PAIRS = [
("text", "--text", "--bg", 4.5, "body text on the page"),
("text", "--text", "--bg-panel", 4.5, "body text in a panel"),
@@ -48,8 +56,14 @@ PAIRS = [
("text", "--danger", "--bg-panel", 4.5, "an error message"),
("text", "--warning", "--bg-panel", 4.5, "a caution message"),
("text", "--player", "--bg", 4.5, "the player's own words"),
("boundary", "--border", "--bg-panel", 3.0, "a control's resting edge"),
("boundary", "--border-bright", "--bg-panel", 3.0, "a control's hover edge"),
("boundary", "--border", "--bg-input", 3.0, "a field or button's resting edge"),
("boundary", "--border-bright", "--bg-input", 3.0, "a field or button's hover edge"),
("boundary", "--border", "--bg-panel", 3.0, "a control's resting edge in a panel"),
("boundary", "--border-bright", "--bg-panel", 3.0, "a control's hover edge in a panel"),
("boundary", "--border", "--bg", 3.0, "a divider on the page"),
("boundary", "--border-bright", "--bg", 3.0, "the scrollbar thumb"),
("boundary", "--accent-dim", "--bg-input", 3.0, "a focused field's edge"),
("boundary", "--accent-dim", "--bg-panel", 3.0, "a focused control's edge in a panel"),
("boundary", "--chart-1", "--bg-panel", 3.0, "a chart bar"),
("boundary", "--chart-2", "--bg-panel", 3.0, "a chart bar"),
("boundary", "--chart-3", "--bg-panel", 3.0, "a chart bar"),
@@ -81,37 +95,44 @@ def ratio(a: str, b: str) -> float:
def main() -> int:
tokens = read_tokens(TOKENS)
print(f"{TOKENS.relative_to(TOKENS.parents[3])}: {len(tokens)} colour tokens\n")
# Named defensively: the tests run this against a temporary tokens file,
# which need not sit four directories deep the way the real one does.
label = TOKENS.name
if len(TOKENS.parents) > 3:
label = TOKENS.relative_to(TOKENS.parents[3])
print(f"{label}: {len(tokens)} colour tokens\n")
print(f"{'pair':44} {'kind':9} {'ratio':>7} {'floor':>6} verdict")
print("-" * 82)
failures, advisories = 0, 0
text_failures, boundary_failures = 0, 0
for kind, foreground, background, floor, description in PAIRS:
if foreground not in tokens or background not in tokens:
print(f"{description:44} {kind:9} {'—':>7} {floor:>6.1f} MISSING TOKEN")
failures += 1
text_failures += 1
continue
measured = ratio(tokens[foreground], tokens[background])
ok = measured >= floor
if not ok:
if kind == "text":
failures += 1
verdict = "FAIL"
else:
advisories += 1
verdict = "below 1.4.11 (label carries it)"
else:
# Rounded to the two decimals printed, so the verdict matches what the
# reader is shown: a pair reported as 3.00:1 is not failed for arithmetic
# the output does not display.
if round(measured, 2) >= floor:
verdict = "pass"
else:
verdict = "FAIL"
if kind == "text":
text_failures += 1
else:
boundary_failures += 1
print(f"{description:44} {kind:9} {measured:>6.2f}:1 {floor:>6.1f} {verdict}")
print()
if failures:
print(f"{failures} text pair(s) below WCAG AA — this is a defect")
if text_failures:
print(f"{text_failures} text pair(s) below WCAG AA (1.4.3) — this is a defect")
else:
print("every text pair clears WCAG AA (1.4.3)")
if advisories:
print(f"{advisories} boundary pair(s) below 3:1 (1.4.11). Recorded rather "
"than failed: every control in this design carries a visible text "
"label, which is measured above and passes.")
return 1 if failures else 0
if boundary_failures:
print(f"{boundary_failures} boundary pair(s) below 3:1 (WCAG 1.4.11) — "
"this is a defect")
else:
print("every control boundary clears 3:1 (1.4.11)")
return 1 if (text_failures or boundary_failures) else 0
if __name__ == "__main__":
+229 -2
View File
@@ -665,6 +665,228 @@ def check_accessibility(browser: Browser, site: Site, adv: int, checks: Checks):
checks.record("A11y", "the story input takes keyboard focus", typed is True)
# -------------------------------------------------------------- WP-E boundaries
#: The controls whose edges WP-E measures, as the stylesheets actually draw them.
#: `side` is the border the rule sets: .topnav paints only a bottom edge.
BOUNDARY_TARGETS = [
("E01", "the story composer", ".input-bar", "Top"),
("E02", "a story control", ".story-controls button:not(:disabled)", "Top"),
("E04", "the open panel tab", ".panel-tabs button.active", "Top"),
]
#: `.topnav` is measured on the library route, because the play route does not
#: render it — the play page has its own `.play-header`, which story.css keeps
#: deliberately opaque. So the translucent case exists on exactly one screen,
#: and it is the case token arithmetic cannot answer: --bg-panel-glass is
#: rgba(...,0.82) over a gradient, so only the rendered page knows what is
#: behind that edge.
TRANSLUCENT_TARGET = ("E03", "the top navigation's edge", ".topnav", "Bottom")
#: Measured in the browser rather than from tokens, because two of these cannot
#: be derived from tokens at all: .topnav sits on --bg-panel-glass, which is
#: translucent, so its effective background is a composite of what is behind it;
#: and a rendered edge can be changed by opacity, a transition mid-flight, or a
#: rule the token file knows nothing about.
#:
#: A boundary is measured against **both** adjacent colours — the control's own
#: fill inside it and the background outside it — and passes on the better of
#: the two. An edge that matches its fill but contrasts with the page is still a
#: visible outline, and vice versa; what 1.4.11 asks is that the component's
#: extent be perceivable, not that every neighbouring surface differ from it.
BOUNDARY_JS = """
const parse = (c) => (c.match(/[\\d.]+/g) || []).map(Number);
const alphaOf = (c) => {
if (!c || c === 'transparent') return 0;
const p = parse(c);
return p.length > 3 ? p[3] : 1;
};
function lum(c) {
const [r, g, b] = parse(c).slice(0, 3).map(v => v / 255)
.map(v => v <= 0.03928 ? v / 12.92 : Math.pow((v + 0.055) / 1.055, 2.4));
return 0.2126 * r + 0.7152 * g + 0.0722 * b;
}
function over(fg, bg) {
const f = parse(fg), b = parse(bg), a = alphaOf(fg);
return 'rgb(' + [0, 1, 2].map(i => Math.round(f[i] * a + b[i] * (1 - a))).join(', ') + ')';
}
function ratio(x, y) {
const a = lum(x), b = lum(y);
return (Math.max(a, b) + 0.05) / (Math.min(a, b) + 0.05);
}
// Everything painted behind `el`, composited bottom-up, so a translucent
// panel reports the colour a reader actually sees rather than its own rgba.
function behind(el) {
const layers = [];
let node = el;
while (node && node !== document.documentElement) {
const c = getComputedStyle(node).backgroundColor;
if (alphaOf(c) > 0) {
layers.push(c);
if (alphaOf(c) >= 1) break;
}
node = node.parentElement;
}
let result = 'rgb(10, 10, 15)';
const root = getComputedStyle(document.documentElement).backgroundColor;
const body = getComputedStyle(document.body).backgroundColor;
if (alphaOf(root) >= 1) result = root;
else if (alphaOf(body) >= 1) result = body;
for (let i = layers.length - 1; i >= 0; i--) {
result = alphaOf(layers[i]) >= 1 ? layers[i] : over(layers[i], result);
}
return result;
}
const [selector, side] = arguments;
const el = document.querySelector(selector);
if (!el) return null;
const s = getComputedStyle(el);
const edge = s['border' + side + 'Color'];
const width = parseFloat(s['border' + side + 'Width']) || 0;
const opacity = parseFloat(s.opacity);
const inside = alphaOf(s.backgroundColor) >= 1
? s.backgroundColor : over(s.backgroundColor, behind(el.parentElement || el));
const outside = behind(el.parentElement || el);
return {
selector, edge, width, opacity, inside, outside,
insideRatio: Math.round(ratio(edge, inside) * 100) / 100,
outsideRatio: Math.round(ratio(edge, outside) * 100) / 100,
outline: s.outlineStyle + ' ' + s.outlineWidth + ' ' + s.outlineColor,
shadow: s.boxShadow,
};
"""
def _settled(browser: Browser, selector: str, side: str, *, timeout: float = 5) -> None:
"""Waits until the edge colour stops moving.
Every one of these controls carries `transition: border-color 0.15s`, so a
measurement taken the instant after a hover or a focus reads a colour part
way between the two states — a value no state actually has. The first run of
this check did exactly that: it reported the hover edge as rgb(114,114,160),
which is neither --border nor --border-bright but a frame between them.
Polled rather than slept, per this harness's rule.
"""
script = ("(() => { const el = document.querySelector(%s);"
" if (!el) return true;"
" const c = getComputedStyle(el)['border%sColor'];"
" const was = window.__wpeEdge; window.__wpeEdge = c;"
" return was === c; })()" % (json.dumps(selector), side))
# Cleared first, so the comparison starts from no previous reading rather
# than from whatever the last measured control left behind.
browser.js("window.__wpeEdge = undefined")
browser.wait_js(script, timeout=timeout)
def _boundary(browser: Browser, selector: str, side: str) -> dict | None:
_settled(browser, selector, side)
return browser.js(BOUNDARY_JS, selector, side)
def _record_boundary(checks: Checks, test: str, what: str, row: dict | None) -> None:
if row is None:
checks.skip(test, what, "the control was not on the page")
return
best = max(row["insideRatio"], row["outsideRatio"])
detail = (f"{best:.2f}:1 (edge {row['edge']} — {row['insideRatio']}:1 against its fill "
f"{row['inside']}, {row['outsideRatio']}:1 against {row['outside']}), "
f"{row['width']}px")
checks.record("WCAG 1.4.11", what, best >= 3.0 and row["width"] > 0, detail)
def check_control_boundaries(browser: Browser, site: Site, adv: int, checks: Checks,
evidence: dict, out: Path) -> None:
"""v1.1 WP-E §21: a control's edge is visible on its own, measured.
M11 measured text contrast on the rendered page and left boundaries to the
token audit, which checked them against one background and reported a
shortfall without failing. These are the rendered edges, at rest, on hover
and while focused, against what is actually behind them.
"""
_open_play(browser, site, adv)
measured: dict[str, dict] = {}
# A panel tab's edge is `transparent` until the panel is open, which makes
# the open tab the one control here whose boundary is the only thing marking
# it — exactly what 1.4.11 is about. So one is opened rather than skipped.
if not _open_panel(browser, "State"):
checks.skip("E04", "the open panel tab — resting edge", "the State panel did not open")
for test, what, selector, side in BOUNDARY_TARGETS:
row = _boundary(browser, selector, side)
_record_boundary(checks, test, f"{what} — resting edge", row)
if row:
measured[f"{what} (rest)"] = row
# Hover, with a real pointer: `:hover` follows the browser's pointer state,
# so a synthetic mouseover would silently re-measure the resting edge.
control = browser.find(".story-controls button:not(:disabled)", required=False)
if control is None:
checks.skip("E02", "a story control — hover edge", "no enabled control on the page")
else:
browser.hover(control)
hovered = browser.js(
"return document.querySelector('.story-controls button:not(:disabled)')"
" .matches(':hover');")
if hovered is not True:
checks.skip("E02", "a story control — hover edge",
"the pointer did not land on the control")
else:
row = _boundary(browser, ".story-controls button:not(:disabled)", "Top")
_record_boundary(checks, "E02", "a story control — hover edge", row)
if row:
measured["a story control (hover)"] = row
browser.unhover()
# Focus: the composer's edge changes colour while it holds focus, and that
# edge is what tells a keyboard reader where they are.
focused = browser.js("""
const box = document.querySelector('.input-main textarea');
if (!box) return false;
box.focus();
return document.activeElement === box;
""")
if focused is not True:
checks.skip("E01", "the story composer — focused edge", "the composer did not take focus")
else:
row = _boundary(browser, ".input-bar", "Top")
_record_boundary(checks, "E01", "the story composer — focused edge", row)
if row:
measured["the story composer (focus)"] = row
# A focus ring that is only a colour change is not enough on its own;
# M11 already asserts a visible focus indicator, and this says the
# focused edge is also measurably distinct from the resting one.
rest = measured.get("the story composer (rest)")
if rest:
checks.record("WCAG 1.4.11", "the focused composer edge differs from its resting edge",
row["edge"] != rest["edge"],
f"rest {rest['edge']} -> focus {row['edge']}")
shot = browser.screenshot(out / "control-boundaries.png")
checks.record("WP-E", "a screenshot of the measured controls was captured",
shot.exists() and shot.stat().st_size > 0, str(shot))
# The translucent edge, on the only screen that has it. Token arithmetic
# cannot reach this one: --bg-panel-glass is rgba over a gradient, so what
# is behind the nav's bottom edge is known only to the rendered page.
test, what, selector, side = TRANSLUCENT_TARGET
browser.go(site.url)
browser.wait_for(".topnav", timeout=30)
row = _boundary(browser, selector, side)
_record_boundary(checks, test, f"{what} — over a translucent panel", row)
if row:
measured[f"{what} (translucent)"] = row
checks.record("WP-E", "the nav's background really is translucent",
row["outside"] != row["inside"],
f"composited to {row['inside']} over {row['outside']}")
nav_shot = browser.screenshot(out / "control-boundaries-nav.png")
checks.record("WP-E", "a screenshot of the navigation edge was captured",
nav_shot.exists() and nav_shot.stat().st_size > 0, str(nav_shot))
evidence["control_boundaries"] = {"measured": measured,
"screenshots": [str(shot), str(nav_shot)]}
# -------------------------------------------------------------- WP-C scenarios
def _take_count_is(count: str) -> str:
@@ -1354,7 +1576,11 @@ def main() -> int:
"C5", "failed generation and recovery"),
("export", lambda: check_export_download(browser, site, adv, checks, evidence, out)),
]
for suite, scenarios in (("M11", m11), ("WP-C", wpc)):
wpe = [
("boundaries", lambda: check_control_boundaries(browser, site, adv, checks,
evidence, out)),
]
for suite, scenarios in (("M11", m11), ("WP-C", wpc), ("WP-E", wpe)):
checks.suite = suite
for name, scenario in scenarios:
if not wanted(name):
@@ -1383,7 +1609,8 @@ def main() -> int:
"kind": ("release regression" if narrated and not only else
"partial (no narrator)" if not narrated else f"development (only {sorted(only)})"),
"checks": checks.rows,
"suites": {"M11": checks.counts("M11"), "WP-C": checks.counts("WP-C")},
"suites": {"M11": checks.counts("M11"), "WP-C": checks.counts("WP-C"),
"WP-E": checks.counts("WP-E")},
"passed": checks.counts()["passed"],
"failed": checks.counts()["failed"],
"skipped": checks.counts()["skipped"],
+40
View File
@@ -25,6 +25,7 @@ has actually finished arriving — never the click that started it.
from __future__ import annotations
import base64
import json
import os
import shutil
@@ -226,6 +227,45 @@ class Browser:
def source(self) -> str:
return self._call("GET", self._s("/source"))["value"]
def screenshot(self, path) -> Path:
"""The viewport as a PNG, written where you ask (v1.1 WP-E).
Evidence for a change a reader judges by looking at it: a contrast ratio
says a boundary is measurable, and a picture says what it looks like.
"""
encoded = self._call("GET", self._s("/screenshot"))["value"]
target = Path(path)
target.parent.mkdir(parents=True, exist_ok=True)
target.write_bytes(base64.b64decode(encoded))
return target
def hover(self, element: str) -> None:
"""A real pointer over an element, so `:hover` actually applies.
Dispatching a mouseover event from JavaScript does not do this: CSS
`:hover` follows the browser's own pointer state, not a synthetic event,
so a measurement taken after `dispatchEvent` reads the resting style and
reports it as the hover style. This moves the pointer (v1.1 WP-E).
"""
self._call("POST", self._s("/execute/sync"), {
"script": "arguments[0].scrollIntoView({block: 'center', inline: 'nearest'})",
"args": [{ELEMENT_KEY: element}]})
self._call("POST", self._s("/actions"), {"actions": [{
"type": "pointer", "id": "mouse", "parameters": {"pointerType": "mouse"},
"actions": [{"type": "pointerMove", "duration": 60,
"origin": {ELEMENT_KEY: element}, "x": 0, "y": 0}]}]})
def unhover(self) -> None:
"""Move the pointer off whatever it was over, and forget the input state."""
self._call("POST", self._s("/actions"), {"actions": [{
"type": "pointer", "id": "mouse", "parameters": {"pointerType": "mouse"},
"actions": [{"type": "pointerMove", "duration": 30,
"origin": "viewport", "x": 0, "y": 0}]}]})
try:
self._call("DELETE", self._s("/actions"))
except WebDriverError:
pass
def js(self, script: str, *args):
return self._call("POST", self._s("/execute/sync"),
{"script": script, "args": list(args)})["value"]
+353
View File
@@ -0,0 +1,353 @@
"""v1.1 WP-D criterion 3: the backup completes **through the UI**.
python -m tools.wpd_backup_ui --out <dir under $HOME> [--case real|large|both] [--show]
Run from `backend/`, with `frontend/dist` already built.
WP-D's first pass drove `POST /api/backups` — the endpoint the *Back up now*
button calls — on a 2.3 MB campaign database and a 117 MB one. That is evidence
about the implementation, and the plan's criterion 3 asks for something else:
that the backup *completes through the UI* on both. A reader does not call an
endpoint. This drives the reader-facing control in a real Firefox, against the
production build served by FastAPI, exactly as WP-C's harness does.
**No narrator and no inference.** A backup needs neither, so nothing here
touches a model host.
What it refuses to call a pass, per the brief:
- the click does nothing (no toast, no file);
- the request fails (an error toast);
- no backup file appears on disk;
- the finished copy does not pass a full `PRAGMA integrity_check`;
- the UI reports an error.
Every wait is on a condition the page or the filesystem can show. Nothing here
sleeps and then asserts.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import shutil
import sqlite3
import sys
import time
from datetime import datetime
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from tools.m11_webdriver import ( # noqa: E402
Browser, Site, geckodriver_version, require_under_home,
)
BACKEND = Path(__file__).resolve().parent.parent
HOME = Path.home()
#: The two databases criterion 3 names. Both are copied before use: the first is
#: the M11 evidence campaign and must not be written to, and the second is the
#: 117 MB application database built for WP-D's timing.
DEFAULT_REAL = HOME / "m11-evidence/m04-final/campaign.db"
DEFAULT_LARGE = HOME / "v11-evidence/wp-d/endpoint/campaign-100mb/campaign.db"
BLOCK = '[data-testid="database-backup"]'
BUTTON = f'{BLOCK} button.primary'
class Checks:
"""Results, with the discipline that an unrun check is not a passing one."""
def __init__(self) -> None:
self.rows: list[dict] = []
def record(self, case: str, name: str, ok: bool, detail: str = "") -> bool:
self.rows.append({"case": case, "check": name,
"result": "PASS" if ok else "FAIL", "detail": detail})
print(f" {'ok ' if ok else 'FAIL'} {case:6} {name}"
+ (f" — {detail}" if detail else ""), flush=True)
return ok
@property
def failed(self) -> list[dict]:
return [r for r in self.rows if r["result"] == "FAIL"]
# ----------------------------------------------------------------- helpers
def integrity_of(path: Path) -> tuple[str, float]:
"""The full check on a finished copy, and what it cost, measured here.
The application runs its own `integrity_check` before keeping the file; this
is an independent second opinion on the artefact the UI produced, and it is
where criterion 3's 'time for integrity_check' comes from.
"""
connection = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
try:
started = time.perf_counter()
rows = connection.execute("PRAGMA integrity_check").fetchall()
elapsed = time.perf_counter() - started
finally:
connection.close()
return ", ".join(str(r[0]) for r in rows), elapsed
def digest(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()[:16]
def backups_in(db_path: Path) -> dict[str, int]:
directory = db_path.parent / "backups"
if not directory.exists():
return {}
return {p.name: p.stat().st_size for p in directory.iterdir() if p.is_file()}
def wait_for_new_backup(db_path: Path, before: set[str], *, timeout: float = 300,
poll: float = 0.2) -> Path | None:
"""The backup file the application wrote, once it is really there.
A file condition, not a sleep: a name that was not there before, not a
`.partial`, and a size that has stopped growing.
"""
directory = db_path.parent / "backups"
deadline = time.monotonic() + timeout
sizes: dict[str, int] = {}
while time.monotonic() < deadline:
if directory.exists():
for path in directory.iterdir():
if not path.is_file() or path.name in before:
continue
if path.name.endswith(".partial"):
continue
size = path.stat().st_size
if size > 0 and sizes.get(path.name) == size:
return path
sizes[path.name] = size
time.sleep(poll)
return None
def toasts(browser: Browser) -> list[dict]:
"""What the page is telling the reader — the message apart from its mark.
A toast renders a decorative mark before the message:
`<span class="toast-mark" aria-hidden="true">❖</span><span>…</span>`. So the
button's `textContent` begins with that character, and the first run of this
tool matched `textContent.startswith('Backup written:')` and reported six
*passing* behaviours as failures — the UI had said exactly what it should,
and the assertion was reading the mark. `message` is the message span alone;
`text` is kept whole for the evidence record.
"""
return browser.js("""
const host = document.querySelector('.toast-host');
if (!host) return [];
return [...host.querySelectorAll('button.toast')].map(b => {
const span = b.querySelector('span:not(.toast-mark)');
return {
error: b.classList.contains('toast-error'),
message: (span ? span.textContent : b.textContent).trim(),
text: b.textContent.trim(),
};
});
""") or []
def open_backup_panel(browser: Browser, site: Site, checks: Checks, case: str) -> bool:
"""Reach the control the way a reader does: the nav link, then the panel."""
browser.go(site.url)
browser.wait_for(".topnav", timeout=60)
link = browser.find('.nav-links a[href="/settings"]', required=False)
if link is None:
return checks.record(case, "the Settings link is in the navigation", False)
browser.click(link)
opened = browser.wait_js(f"!!document.querySelector('{BLOCK}')", timeout=60)
if not checks.record(case, "Settings opens from the navigation link", opened,
browser.url):
return False
summary = browser.find(f"{BLOCK} summary", required=False)
if summary is None:
return checks.record(case, "the backup panel has a disclosure", False)
browser.click(summary)
# The panel is a <details>: it loads what is already on disk when it opens,
# so waiting for the directory line proves the application answered.
shown = browser.wait_js(
f"document.querySelector('{BLOCK}').open === true"
f" && !!document.querySelector('{BUTTON}')", timeout=30)
checks.record(case, "the backup panel opens and shows its control", shown)
listed = browser.wait_js(f"!!document.querySelector('{BLOCK} code')", timeout=30)
checks.record(case, "the panel reports where backups are written", listed)
return shown
def click_back_up_now(browser: Browser, site: Site, db: Path, checks: Checks,
case: str, label: str) -> dict:
"""One press of the button, judged by what the page and the disk then show."""
before = set(backups_in(db))
# Clear anything still on screen, so the toast this press produces is the
# one that is read back rather than a leftover from the previous press.
browser.js("""
for (const b of document.querySelectorAll('.toast-host button.toast')) b.click();
return true;
""")
button = browser.find(BUTTON, required=False)
if button is None:
checks.record(case, f"{label}: the control is on the page", False)
return {}
started = time.perf_counter()
browser.click(button)
# Either outcome ends the wait, so a failure is reported as a failure rather
# than as a timeout.
settled = browser.wait_js(
"(() => { const t = [...document.querySelectorAll('.toast-host button.toast')];"
" return t.length > 0; })()", timeout=300)
elapsed = time.perf_counter() - started
shown = toasts(browser)
errors = [t for t in shown if t["error"]]
written = [t for t in shown if t["message"].startswith("Backup written:")]
checks.record(case, f"{label}: the click produced a visible result", settled,
json.dumps(shown)[:200])
checks.record(case, f"{label}: the UI reports no error",
not errors, json.dumps(errors)[:300])
checks.record(case, f"{label}: the UI reports the backup was written",
bool(written), json.dumps(written)[:200])
idle = browser.wait_js(
"(() => { const b = document.querySelector(%s);"
" return !!b && b.textContent.trim() === 'Back up now'; })()"
% json.dumps(BUTTON), timeout=120)
checks.record(case, f"{label}: the control returns from 'Backing up…'", idle)
produced = wait_for_new_backup(db, before)
checks.record(case, f"{label}: a backup file was physically produced",
produced is not None, str(produced))
if produced is None:
return {"seconds": round(elapsed, 3), "toasts": shown}
named = any(produced.name in t["message"] for t in written)
checks.record(case, f"{label}: the UI names the file that appeared", named,
f"{produced.name} — {written[0]['message'] if written else ''}")
relisted = browser.wait_js(
"[...document.querySelectorAll('%s .backup-list code')]"
".some(c => c.textContent.trim() === %s)" % (BLOCK, json.dumps(produced.name)),
timeout=60)
checks.record(case, f"{label}: the new backup appears in the panel's list", relisted)
verdict, integrity_seconds = integrity_of(produced)
checks.record(case, f"{label}: the finished copy passes full integrity_check",
verdict == "ok", f"{verdict} in {integrity_seconds * 1000:.1f} ms")
return {
"file": str(produced),
"bytes": produced.stat().st_size,
"sha256_16": digest(produced),
"seconds": round(elapsed, 3),
"integrity": verdict,
"integrity_seconds": round(integrity_seconds, 4),
"toasts": shown,
}
def run_case(case: str, source: Path, out: Path, checks: Checks, *, show: bool,
twice: bool) -> dict:
print(f"\n=== {case}: {source} ===", flush=True)
work = out / case
shutil.rmtree(work, ignore_errors=True)
(work / "data").mkdir(parents=True)
db = work / "data" / "campaign.db"
copy_started = time.perf_counter()
shutil.copy2(source, db)
copy_seconds = time.perf_counter() - copy_started
size = db.stat().st_size
print(f" copied {size:,} bytes in {copy_seconds:.2f}s -> {db}", flush=True)
site = Site(BACKEND, db, work / "server.log")
browser = Browser(headless=not show, log=work / "geckodriver.log")
result: dict = {"database": str(source), "bytes": size,
"served_at": site.url, "firefox": browser.version}
try:
if not open_backup_panel(browser, site, checks, case):
return result
result["first"] = click_back_up_now(browser, site, db, checks, case, "backup")
browser.screenshot(work / "back-up-now.png")
if twice and result["first"].get("file"):
kept = Path(result["first"]["file"])
before_bytes, before_digest = kept.stat().st_size, digest(kept)
result["second"] = click_back_up_now(browser, site, db, checks, case,
"second backup")
still_there = kept.exists()
checks.record(case, "the earlier backup still exists", still_there)
if still_there:
checks.record(
case, "and is byte-for-byte what it was",
kept.stat().st_size == before_bytes and digest(kept) == before_digest,
f"{before_bytes:,} bytes, sha256:{before_digest}")
verdict, _ = integrity_of(kept)
checks.record(case, "and still passes integrity_check", verdict == "ok",
verdict)
if result["second"].get("file"):
checks.record(case, "the second backup is a different file",
result["second"]["file"] != result["first"]["file"],
Path(result["second"]["file"]).name)
finally:
browser.quit()
site.stop()
return result
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--out", required=True,
help="evidence directory, which must be under $HOME")
parser.add_argument("--case", default="both", choices=["real", "large", "both"])
parser.add_argument("--show", action="store_true", help="run Firefox visibly")
parser.add_argument("--real-db", default=str(DEFAULT_REAL))
parser.add_argument("--large-db", default=str(DEFAULT_LARGE))
args = parser.parse_args()
out = require_under_home(Path(args.out).expanduser())
out.mkdir(parents=True, exist_ok=True)
if not (BACKEND.parent / "frontend/dist/index.html").exists():
print("frontend/dist is not built", file=sys.stderr)
return 2
checks = Checks()
started = datetime.now()
results: dict[str, dict] = {}
wanted = [("real", Path(args.real_db).expanduser(), True)] if args.case != "large" else []
if args.case != "real":
wanted.append(("large", Path(args.large_db).expanduser(), False))
print(f"WP-D criterion 3 — the backup through the UI. geckodriver "
f"{geckodriver_version()}", flush=True)
for case, source, twice in wanted:
if not source.exists():
checks.record(case, "the database is present", False, str(source))
continue
results[case] = run_case(case, source, out, checks, show=args.show, twice=twice)
report = {
"started": started.isoformat(timespec="seconds"),
"seconds": round((datetime.now() - started).total_seconds()),
"cases": results,
"checks": checks.rows,
"passed": len([r for r in checks.rows if r["result"] == "PASS"]),
"failed": len(checks.failed),
}
(out / "backup-ui-report.json").write_text(json.dumps(report, indent=2))
print(f"\n{report['passed']} passed, {report['failed']} failed "
f"-> {out / 'backup-ui-report.json'}")
return 1 if checks.failed else 0
if __name__ == "__main__":
raise SystemExit(main())