"""M9: a consistent copy of the whole database, taken while the app is running. This is **not** the campaign bundle, and the two are not alternatives. They are different recovery tools and M9 keeps them apart deliberately: campaign bundle one campaign, logical, portable between installations, importable into a clean data directory on another machine, readable by a human and by a later build database backup every campaign, every setting, physical, this machine, restored by putting the file back The bundle is the primary cross-install recovery path and is what the acceptance tests measure. This exists for the other question: the reader has one database holding everything they have ever played, and wants a copy of it before they upgrade, move a disk, or try something they might regret. ## Why not `cp data.db backup.db` Because a copy taken with the application running is a copy of a moving target. SQLite writes a database in pages, and a plain file copy can read page 5 before a transaction and page 900 after it — the result is a file that opens, reports a schema, and is silently missing or duplicating rows. In WAL mode it is worse: the committed data may be in a `-wal` file the copy never touched. Nothing warns anyone. The corruption is found later, by which time the original may be gone. So this uses SQLite's own **online backup API** (`sqlite3.Connection.backup`), which is the supported mechanism for exactly this: it copies page by page while holding the right locks, restarts if a write moves the source underneath it, and produces a file that is a transactionally consistent snapshot of some committed point. The application keeps running throughout; no session is closed and no turn is blocked. ## What the procedure guarantees 1. The source database is opened **read-only** and is never written to. A backup that could damage what it is backing up would be worse than no backup. 2. The copy is written to a temporary file beside the destination and renamed into place only after it has been verified, so an interrupted or failed run never leaves a half-written file wearing a backup's name. `os.replace` is atomic on the same filesystem, which is why the temporary sits in the destination's own directory rather than in `/tmp`. 3. `PRAGMA quick_check` runs against the finished copy, opened as its own database, before it is renamed. A backup nobody verified is a belief. 4. An existing file is never overwritten. Each run writes a new name stamped with the time, so yesterday's backup survives today's mistake — which is most of what a backup is for. 5. Failure is reported and leaves nothing behind but the log line. ## What it does not do There is no restore endpoint. Restoring a whole database means replacing the file the running application has open, and doing that from inside that application is a way to lose both copies. The procedure is in `DEVELOPMENT.md`: stop the app, move the file into place, start it. Campaign-level recovery — the common case, and the one that crosses machines — is the bundle. No path comes from a caller. The destination directory is derived from the database the application is already using and the filename is generated here, so there is no request that can direct a write anywhere else (H08). """ from __future__ import annotations import logging import os import sqlite3 from dataclasses import dataclass from datetime import datetime from pathlib import Path from .database import DB_PATH log = logging.getLogger(__name__) #: Where backups go: a directory beside the database itself. Beside, rather than #: inside a configurable location, because the one thing this must not do is #: write somewhere a request can name. DIRECTORY_NAME = "backups" #: The stem every backup file carries, so a directory listing sorts by date and #: says what these files are without being opened. PREFIX = "adventure-storyteller" class BackupError(RuntimeError): """A backup did not complete. The source database is untouched.""" @dataclass(frozen=True) class Backup: """One finished, verified backup file.""" path: Path bytes: int pages: int seconds: float integrity: str def as_dict(self) -> dict: return { # The name alone, not the path. The full path is a fact about this # machine's filesystem, and the reader is told the directory once by # the endpoint that lists them. "filename": self.path.name, "bytes": self.bytes, "pages": self.pages, "seconds": round(self.seconds, 3), "integrity": self.integrity, } def directory(db_path: Path | None = None) -> Path: """The backup directory for a database, created if it does not exist.""" root = (db_path or DB_PATH).parent / DIRECTORY_NAME root.mkdir(parents=True, exist_ok=True) return root def create(db_path: Path | None = None, *, now: datetime | None = None) -> Backup: """Takes one verified backup of the live database, and returns it. Raises `BackupError` on any failure, having removed whatever it had written. The source database is opened read-only and is never modified, so a failure here costs the backup and nothing else. """ source_path = db_path or DB_PATH if not source_path.exists(): raise BackupError(f"There is no database at {source_path}.") stamp = (now or datetime.now()).strftime("%Y%m%d-%H%M%S") target = _unused_name(directory(source_path), stamp) # The temporary sits in the destination directory so the rename below is a # rename rather than a copy across filesystems, which would not be atomic. working = target.with_name(target.name + ".partial") started = datetime.now() try: pages = _copy(source_path, working) integrity = _verify(working) except BackupError: _discard(working) raise except Exception as exc: # noqa: BLE001 - reported, never raised raw _discard(working) log.exception("Backup of %s failed", source_path) raise BackupError(f"{type(exc).__name__}: {exc}") from exc size = working.stat().st_size # Only now does the file get the name a reader would trust. os.replace(working, target) return Backup( path=target, bytes=size, pages=pages, seconds=(datetime.now() - started).total_seconds(), integrity=integrity, ) def _copy(source_path: Path, working: Path) -> int: """Runs SQLite's online backup from `source_path` into a new file. The source is opened through a URI with `mode=ro`, so this connection cannot write to it even by accident. The destination is a fresh database that this function creates; `backup()` overwrites whatever is in it, and the caller has guaranteed the name is unused. Returns the number of pages copied, which is the one honest measure of how much was actually written — the file size counts pages the source had already allocated. """ source = sqlite3.connect(f"file:{source_path}?mode=ro", uri=True) try: destination = sqlite3.connect(working) try: copied = 0 def progress(_status, remaining, total): nonlocal copied copied = total - remaining # `pages=-1` copies the whole database in one step while holding the # source's read lock, which is the right trade for a local # single-user database: it is the fastest option, it cannot restart # partway, and the lock it holds does not block readers. source.backup(destination, pages=-1, progress=progress) return copied finally: destination.close() finally: source.close() def _verify(working: Path) -> str: """Runs `PRAGMA quick_check` against the finished copy. Opened as its own connection, so what is checked is the file on disk rather than any page cache the copy left behind. `quick_check` rather than `integrity_check` because it does the structural work — every page reachable, every record readable — without the full index cross-check, which on a large database is minutes rather than moments. A backup nobody verified is a belief; a backup verified slowly enough that nobody takes one is worse. """ connection = sqlite3.connect(f"file:{working}?mode=ro", uri=True) try: rows = connection.execute("PRAGMA quick_check").fetchall() finally: connection.close() result = ", ".join(str(row[0]) for row in rows) if rows else "no result" if result != "ok": raise BackupError( f"The backup was written but did not verify: {result}. It has been " f"discarded; the original database is untouched." ) return result def _unused_name(root: Path, stamp: str) -> Path: """A name in `root` that nothing is using. An existing backup is never overwritten. Two backups taken inside one second are the only way to collide, and the counter settles that rather than one of them silently replacing the other. """ candidate = root / f"{PREFIX}-{stamp}.db" counter = 2 while candidate.exists() or candidate.with_name(candidate.name + ".partial").exists(): candidate = root / f"{PREFIX}-{stamp}-{counter}.db" counter += 1 return candidate def _discard(working: Path) -> None: """Removes a partial file, ignoring a file that is already gone.""" try: working.unlink() except OSError: pass def existing(db_path: Path | None = None) -> list[dict]: """Every backup in the directory, newest first. Names and sizes only. Reading one to report what is inside it would mean opening a database on every page load for a screen that is a list. `taken_at` is read out of the **filename**, which is the stamp `create` wrote when it took the backup, and falls back to the file's modification time only for a name that does not parse. The two usually agree, and where they disagree the name is the one telling the truth: copying a backup to another disk, restoring it from an archive, or touching it all move the mtime, and a list that then reordered itself would report when the file was last handled rather than when the backup was taken. """ root = directory(db_path) rows = [] for path in root.glob(f"{PREFIX}-*.db"): try: stat = path.stat() except OSError: continue rows.append({ "filename": path.name, "bytes": stat.st_size, "taken_at": ( _stamp_in(path.name) or datetime.fromtimestamp(stat.st_mtime) ).isoformat(timespec="seconds"), }) rows.sort(key=lambda row: (row["taken_at"], row["filename"]), reverse=True) return rows def _stamp_in(filename: str) -> datetime | None: """The time in a backup's name, or `None` if it does not carry one.""" rest = filename[len(PREFIX) + 1:].removesuffix(".db") # A collision within one second gets a `-2` suffix, which is not the stamp. stamp = "-".join(rest.split("-")[:2]) try: return datetime.strptime(stamp, "%Y%m%d-%H%M%S") except ValueError: return None