"""Guards on the offline baseline: nothing a story turn needs is fetched. Phase 0B ran the upstream application on a network with no route out and the first turn died in `tiktoken`, which downloads its BPE table the first time anything counts a token — invisible on a machine that had already been online once. The browser separately pulled three font families from Google on every page load. Both are now local, and these tests fail if either comes back. The distinction that matters here is *runtime* assets. Downloading a dependency at install time is fine; downloading one while the user is playing is not. python -m pytest tests/test_offline_assets.py -v Acceptance tests A01, H01 and H11. """ import hashlib import inspect import re import socket from pathlib import Path import pytest from fastapi.testclient import TestClient from app.context import builder, encoding from app.main import app REPO = Path(__file__).resolve().parents[2] FRONTEND = REPO / "frontend" # Any absolute http(s) URL. Matched against the places a browser would actually # be told to fetch from: markup attributes and CSS url()/@import. REMOTE_URL = re.compile(rb"https?://[^\s\"'()>]+") @pytest.fixture def client(): with TestClient(app) as c: yield c # --- the tokenizer ------------------------------------------------------- def test_vendored_table_is_present_and_intact(): """The digest is the whole reason the vendored copy is trustworthy.""" data = encoding.BPE_PATH.read_bytes() assert hashlib.sha256(data).hexdigest() == encoding.BPE_SHA256 def test_pinned_digest_still_matches_tiktokens_own(): """`tiktoken` hardcodes the digest it expects for `cl100k_base`. If an upgrade ever points that name at a different table, the vendored copy is stale and every token count would quietly disagree with upstream's.""" import tiktoken_ext.openai_public source = inspect.getsource(tiktoken_ext.openai_public.cl100k_base) assert encoding.BPE_SHA256 in source assert encoding.SOURCE_URL in source def test_token_counting_makes_no_network_call(monkeypatch): """The real regression. `count_tokens` runs on every turn; on a host with no route out, the upstream version raised ConnectionError instead of narrating.""" encoding.get_encoding.cache_clear() def refuse(*args, **kwargs): raise AssertionError("the tokenizer tried to open a socket") monkeypatch.setattr(socket, "socket", refuse) monkeypatch.setattr(socket, "create_connection", refuse) monkeypatch.setattr(socket, "getaddrinfo", refuse) try: assert builder.count_tokens("The lighthouse keeper's lamp guttered.") > 0 finally: encoding.get_encoding.cache_clear() def test_token_counts_are_unchanged(): """Golden values from the encoding built by upstream's own `tiktoken.get_encoding("cl100k_base")`, checked equal token for token across ASCII and accented text. The context budget is computed from these numbers, so a silently different tokenizer would silently change what fits in a prompt.""" assert builder.count_tokens("") == 0 assert builder.count_tokens("hello world") == 2 assert builder.count_tokens( "The lighthouse keeper's lamp guttered in the salt wind." ) == 13 assert builder.count_tokens("naïve café — résumé") == 8 def test_round_trip_survives_the_awkward_characters(): enc = encoding.get_encoding() for text in ("", "naïve café — résumé 🜂 龍", "tabs\tand\r\nCRLF", "a" * 500): assert enc.decode(enc.encode(text)) == text # --- the browser --------------------------------------------------------- def test_csp_names_no_remote_origin(client): csp = client.get("/api/health").headers["content-security-policy"] assert "fonts.googleapis.com" not in csp assert "fonts.gstatic.com" not in csp assert "//" not in csp, f"CSP still allows a remote origin: {csp}" assert "font-src 'self'" in csp assert "connect-src 'self'" in csp def test_index_html_fetches_nothing_remote(): html = (FRONTEND / "index.html").read_bytes() # The comment explaining where the fonts went is not a fetch, so match # attributes rather than the whole file. for attr in re.findall(rb"(?:href|src)\s*=\s*\"([^\"]*)\"", html): assert not REMOTE_URL.match(attr), attr def test_stylesheets_fetch_nothing_remote(): for css in sorted((FRONTEND / "src").rglob("*.css")): text = css.read_bytes() for statement in re.findall(rb"(?:url|@import)\s*\(?[^;{}]*", text): assert not REMOTE_URL.search(statement), f"{css.name}: {statement}" def test_every_declared_font_file_exists(): """A @font-face pointing at a file that is not in the tree falls back to a system font on the developer's machine and 404s on a user's.""" css = (FRONTEND / "src" / "styles" / "fonts.css").read_text() declared = re.findall(r"url\('/fonts/([^']+)'\)", css) assert declared, "fonts.css declares no faces" for name in declared: assert (FRONTEND / "public" / "fonts" / name).is_file(), name def _without_comments(text: bytes) -> bytes: """Comments name the hosts these files no longer talk to — the note in `index.html` saying where the fonts went, and the `/* from ... */` line recording where each vendored face was downloaded from. Both are documentation. Strip them, then the check below can be blunt.""" text = re.sub(rb"", b"", text, flags=re.S) return re.sub(rb"/\*.*?\*/", b"", text, flags=re.S) @pytest.mark.skipif( not (Path(__file__).resolve().parents[2] / "frontend" / "dist" / "index.html").exists(), reason="SPA not built; run `npm run build` in frontend/ to check the shipped bundle", ) def test_built_spa_fetches_no_fonts_remotely(): """The source is what the tests above read; this is what users are served.""" dist = FRONTEND / "dist" for path in list(dist.rglob("*.html")) + list(dist.rglob("*.css")): text = _without_comments(path.read_bytes()) assert b"fonts.googleapis.com" not in text, path assert b"fonts.gstatic.com" not in text, path for path in dist.rglob("*.css"): for url in re.findall(rb"url\(([^)]*)\)", _without_comments(path.read_bytes())): assert not REMOTE_URL.search(url), f"{path}: {url}"