From 20403c5132fe9fdea093df06f52a98d4c799513c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Cau=C3=AA=20Faleiros?= Date: Tue, 29 Sep 2026 18:49:21 -0300 Subject: [PATCH] feat: daily encrypted database backup to a bucket of its own MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A backup service runs pg_dump every day at 03:00 Brasília, checks the archive, encrypts it with age to a public key and uploads it with a token for that bucket only. The server cannot read or delete backups: the private key stays with the owner, the bucket's lifecycle rule expires copies and its lock stops early deletion. Each run is recorded and shown on the Kanban's Integrations tab. tests/backup_test.py backs up, restores into a scratch database and compares the rows in CI. Setup and restore: docs/BACKUP.md. Co-Authored-By: Claude Opus 5.5 --- .gitea/workflows/deploy.yml | 3 +- app/api/operator.py | 13 +- app/schema.sql | 8 + compose.local.yaml | 25 +++ deploy/Dockerfile.api | 3 + deploy/portainer.env.example | 7 + docker-compose.yml | 28 ++++ docs/BACKUP.md | 87 +++++++++++ docs/ROADMAP.md | 6 + infra/Dockerfile | 3 + infra/storage-init.sh | 9 ++ ops/db_backup.py | 289 +++++++++++++++++++++++++++++++++++ tests/backup_test.py | 79 ++++++++++ web/kanban.js | 7 + 14 files changed, 565 insertions(+), 2 deletions(-) create mode 100644 docs/BACKUP.md create mode 100644 ops/db_backup.py create mode 100644 tests/backup_test.py diff --git a/.gitea/workflows/deploy.yml b/.gitea/workflows/deploy.yml index 80d80a8..892b157 100644 --- a/.gitea/workflows/deploy.yml +++ b/.gitea/workflows/deploy.yml @@ -93,6 +93,7 @@ jobs: $COMPOSE exec -T api python -m tests.retention_test $COMPOSE exec -T api python -m tests.runtime_security_test $COMPOSE exec -T api python -m tests.tiny_oauth_test + $COMPOSE exec -T backup python -m tests.backup_test # Run Chrome on the Compose network. It must resolve the same storage:9000 # hostname used in presigned URLs, and absence of Chrome must fail CI. @@ -106,7 +107,7 @@ jobs: if: failure() run: | $COMPOSE ps || true - $COMPOSE logs --tail 200 api worker site kanban || true + $COMPOSE logs --tail 200 api worker backup site kanban || true - name: Tear down if: always() diff --git a/app/api/operator.py b/app/api/operator.py index 2c90e1a..d12341d 100644 --- a/app/api/operator.py +++ b/app/api/operator.py @@ -67,6 +67,16 @@ def freight_status(): 'production_days': freight.production_days} +def backup_status(c): + """The latest database backup and the latest run, which may have failed.""" + ok = c.execute('''SELECT finished_at,bytes FROM dtf_local.backups WHERE status='ok' + ORDER BY finished_at DESC LIMIT 1''').fetchone() + last = c.execute('SELECT finished_at,status,detail FROM dtf_local.backups ORDER BY finished_at DESC LIMIT 1').fetchone() + return {'last_ok': ok['finished_at'].isoformat() if ok else None, 'bytes': ok['bytes'] if ok else None, + 'failed': last['detail'] if last and last['status'] == 'failed' else None, + 'failed_at': last['finished_at'].isoformat() if last and last['status'] == 'failed' else None} + + @router.get('/api/operator/board') def board(user=Depends(operator)): with db.connect() as c: @@ -97,9 +107,10 @@ def board(user=Depends(operator)): # Pagamentos tab pages through them until someone records a resolution. issues_total = c.execute('''SELECT count(*) AS n FROM dtf_local.payment_events WHERE (outcome LIKE 'refused%' OR outcome LIKE 'attention%') AND resolved_at IS NULL''').fetchone()['n'] + backup = backup_status(c) return {'states': STATES, 'transitions': TRANSITIONS, 'back': BACK, 'orders': orders, 'payment_issues_total': issues_total, 'tiny': tiny_status(), 'operator': user, - 'providers': {'payment': payment.name, 'freight': freight_status()}, 'environment': ENVIRONMENT, + 'providers': {'payment': payment.name, 'freight': freight_status(), 'backup': backup}, 'environment': ENVIRONMENT, 'finished_shown': len(finished), 'finished_total': finished_total, 'quotes': [quote_view(c, q) for q in pending + approved], 'pending_total': pending_total, 'approved_total': approved_total} diff --git a/app/schema.sql b/app/schema.sql index 5ff9bbc..6c1f47f 100644 --- a/app/schema.sql +++ b/app/schema.sql @@ -134,6 +134,14 @@ CREATE TABLE IF NOT EXISTS dtf_local.print_files ( UNIQUE(order_id,item_index) ); +-- Each run of the off-server database backup (ops/db_backup.py), which the +-- Kanban shows so a backup that stopped working does not go unnoticed. +CREATE TABLE IF NOT EXISTS dtf_local.backups ( + id uuid PRIMARY KEY, started_at timestamptz NOT NULL, finished_at timestamptz NOT NULL, + status text NOT NULL CHECK(status IN ('ok','failed')), object_key text, bytes bigint, detail text +); +CREATE INDEX IF NOT EXISTS backups_finished ON dtf_local.backups(finished_at DESC); + CREATE INDEX IF NOT EXISTS uploads_owner ON dtf_local.uploads(owner); -- Indexes follow the queries the application actually issues. Only these; every diff --git a/compose.local.yaml b/compose.local.yaml index 13cb2f4..5f02e10 100644 --- a/compose.local.yaml +++ b/compose.local.yaml @@ -148,6 +148,9 @@ services: S3_APP_USER: ${S3_APP_USER:-dtf_app} S3_APP_PASSWORD: ${S3_APP_PASSWORD:-local-app-storage-only} S3_BUCKET: ${S3_BUCKET:-dtf-local-artwork} + S3_BACKUP_BUCKET: dtf-local-backups + S3_BACKUP_USER: dtf_backup + S3_BACKUP_PASSWORD: local-backup-storage-only networks: [local] depends_on: storage: {condition: service_healthy} @@ -197,6 +200,28 @@ services: timeout: 3s retries: 12 + # The daily database backup, against the local backup bucket. Idle until + # BACKUP_AGE_RECIPIENT is set; tests/backup_test.py runs it with a throwaway key. + backup: + build: + context: . + dockerfile: infra/Dockerfile + command: python -m ops.db_backup serve + environment: + DATABASE_ADMIN_URL: postgresql://${POSTGRES_USER:-dtf_local}:${POSTGRES_PASSWORD:-local-database-only}@db:5432/${POSTGRES_DB:-dtf_local} + BACKUP_S3_ENDPOINT: http://storage:9000 + BACKUP_BUCKET: dtf-local-backups + BACKUP_ACCESS_KEY_ID: dtf_backup + BACKUP_SECRET_ACCESS_KEY: local-backup-storage-only + BACKUP_REGION: us-east-1 + BACKUP_AGE_RECIPIENT: ${BACKUP_AGE_RECIPIENT:-} + networks: [local] + read_only: true + tmpfs: [/tmp] + depends_on: + db-init: {condition: service_completed_successfully} + storage-init: {condition: service_completed_successfully} + site: build: context: . diff --git a/deploy/Dockerfile.api b/deploy/Dockerfile.api index 93a1f7c..074dc7c 100644 --- a/deploy/Dockerfile.api +++ b/deploy/Dockerfile.api @@ -13,8 +13,11 @@ LABEL org.opencontainers.image.title="DTF Portal/API" \ # The base is pinned, so its OS packages are frozen at the digest's build date. # Upgrade them here or the image ships known-fixed Debian vulnerabilities, which # is what the production image was doing while the local one already did this. +# pg_dump and age are for the database backup (ops/db_backup.py). Debian 13 +# ships PostgreSQL 17, the server's major version, which pg_dump must match. RUN apt-get update \ && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends postgresql-client-17 age \ && rm -rf /var/lib/apt/lists/* WORKDIR /app diff --git a/deploy/portainer.env.example b/deploy/portainer.env.example index 3694ecb..bcb2dde 100644 --- a/deploy/portainer.env.example +++ b/deploy/portainer.env.example @@ -51,6 +51,13 @@ MAX_PENDING_UPLOADS=10 MAX_UPLOAD_BYTES=5368709120 UPLOAD_PART_BYTES=8388608 SCAN_MAX_BYTES=134217728 +# Daily encrypted database backup to its own bucket (docs/BACKUP.md). +BACKUP_S3_ENDPOINT=https://.r2.cloudflarestorage.com +BACKUP_BUCKET=dtf-backups +BACKUP_ACCESS_KEY_ID=TBD +BACKUP_SECRET_ACCESS_KEY=TBD +BACKUP_AGE_RECIPIENT=age1... +BACKUP_HOUR=3 # Names of external Portainer/Docker Swarm secrets, never their values. DATABASE_URL_SECRET=TBD diff --git a/docker-compose.yml b/docker-compose.yml index 6471eba..fa775e3 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -183,6 +183,34 @@ services: replicas: 1 restart_policy: {condition: on-failure, delay: 5s} + # Daily encrypted copy of the database to a bucket of its own, off this + # server (ops/db_backup.py). Idle until the BACKUP_* settings are set. The + # bucket's lifecycle rule removes old copies; this credential never deletes. + backup: + image: ${API_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-api}:${IMAGE_TAG:-latest} + command: python -m ops.db_backup serve + environment: + DATABASE_ADMIN_HOST: db + DATABASE_ADMIN_NAME: dtf + DATABASE_ADMIN_USER: dtf_admin + DATABASE_ADMIN_PASSWORD: ${POSTGRES_PASSWORD:?set POSTGRES_PASSWORD} + # The R2 account endpoint (the same as R2_ENDPOINT) and a bucket and API + # token for backups only, never the artwork bucket's. + BACKUP_S3_ENDPOINT: ${BACKUP_S3_ENDPOINT:-} + BACKUP_BUCKET: ${BACKUP_BUCKET:-} + BACKUP_ACCESS_KEY_ID: ${BACKUP_ACCESS_KEY_ID:-} + BACKUP_SECRET_ACCESS_KEY: ${BACKUP_SECRET_ACCESS_KEY:-} + # The public key (age1...). Its private key stays off this server. + BACKUP_AGE_RECIPIENT: ${BACKUP_AGE_RECIPIENT:-} + # Hour of day in Brasília. + BACKUP_HOUR: ${BACKUP_HOUR:-3} + networks: [backend, egress] + read_only: true + tmpfs: [/tmp] + deploy: + replicas: 1 + restart_policy: {condition: on-failure, delay: 30s} + site: image: ${WEB_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-web}:${IMAGE_TAG:-latest} environment: diff --git a/docs/BACKUP.md b/docs/BACKUP.md new file mode 100644 index 0000000..e78605a --- /dev/null +++ b/docs/BACKUP.md @@ -0,0 +1,87 @@ +# Database backup + +The `backup` service copies the database off the server once a day +(`ops/db_backup.py`). It runs `pg_dump`, checks the archive, encrypts it with +[age](https://age-encryption.org) and uploads it to an R2 bucket used for +nothing else. Each run is recorded, and the Kanban shows the latest one under +Integrações → "Backup do banco" ("Em dia", "Verificar" or "Não configurado"). + +The artwork is not backed up. Originals and print files are temporary (7 to 30 +days) and already stored on R2. The database holds what cannot be recreated: +orders, customers, quotes, payments, the Tiny connection and the operators. + +## Why the backup cannot be read or deleted from the server + +- **The server only encrypts.** It holds the public key (`age1…`). The private + key, which is the only way to open a backup, stays with the owner. +- **Its own bucket and token.** The backup token reaches the backup bucket + only, and the artwork token cannot reach it. +- **Nothing is deleted by the job.** The bucket's lifecycle rule removes old + copies. The bucket lock keeps anyone, including someone holding the token, + from deleting or overwriting a copy before its time. + +## Setup (once) + +1. **Key pair.** Generate it on your own computer, not on the server: + + ```bash + docker run --rm gitea.blyzer.com.br/blyzer/dtf-api:latest age-keygen + ``` + + Or, with age installed (`sudo pacman -S age`, `apt install age`), just + run `age-keygen`. + + Save the whole output (the `AGE-SECRET-KEY-1…` line is the private key) in + the password manager. Only the `public key: age1…` value goes to Portainer. + Without the private key, no backup can ever be restored. +2. **Cloudflare R2 → Create bucket.** Use a name such as `dtf-backups`. +3. **Settings on that bucket:** + - Object lifecycle rules: delete objects after 35 days. + - Bucket lock rules: retain for 30 days, prefix `db/`. +4. **R2 → Manage API tokens → Create API token.** + - Permission: Object Read & Write. + - Scope: the `dtf-backups` bucket only. +5. **Portainer.** Add these to the stack's environment, then redeploy: + + | Variable | Value | + |---|---| + | `BACKUP_S3_ENDPOINT` | same as `R2_ENDPOINT` | + | `BACKUP_BUCKET` | `dtf-backups` | + | `BACKUP_ACCESS_KEY_ID` | the new token's Access Key ID | + | `BACKUP_SECRET_ACCESS_KEY` | the new token's Secret Access Key | + | `BACKUP_AGE_RECIPIENT` | the public key, `age1…` | + | `BACKUP_HOUR` | optional, default `3` (03:00 Brasília) | + +The first backup runs as soon as the service starts. After that it runs daily, +and a failed run is retried an hour later. + +## Restore test + +Do this after setup, and then once a month. Open the console of the `backup` +container in Portainer (Containers → `…_backup…` → Console) and run: + +```bash +python -m ops.db_backup list +python -m ops.db_backup verify --identity - +``` + +`--identity -` asks for the private key to be pasted. It is kept only in the +container's memory while the command runs. `verify` restores the latest backup +into a scratch database, prints its row counts and drops it. The live database +is not touched. + +## Restoring for real + +From the `backup` container's console: + +```bash +python -m ops.db_backup restore db/2026/10/dtf-20261001T060000Z.dump.age --into dtf_restored --identity - +``` + +This restores into a new database (or an empty one), never over the running +one. Then do one of these: + +- **Check the restored data**, then point the stack at it. +- **On a new server**, deploy only the `db` service first. Restore into `dtf` + while it is still empty, then deploy the rest. `db-init` then applies the + schema and grants on top. diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 451a50d..48a17a7 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -864,6 +864,12 @@ print-file evidence still need correction before this item can close. - `[ ]` 5.13 — Define production recovery: scheduled encrypted offsite database and object backups, a consistent snapshot boundary, Swarm data placement and a restore rehearsal that opens every required live order file. + **2026-09-29:** database part built. The `backup` service dumps daily, + encrypts with age (the server holds only the public key) and uploads to a + bucket of its own, whose lifecycle rule expires copies and whose lock keeps + them from being deleted early. The Kanban shows the latest run, and + `tests/backup_test.py` restores a copy in CI. Setup and restore: + docs/BACKUP.md. Artwork is not copied: it is temporary and already on R2. - `[ ]` 5.14 — Promote and verify one immutable release. **2026-09-24:** green pushes to `main` now publish images and Portainer's pull-and-redeploy is the release gate; the source preflight is advisory unless enforced by variable, diff --git a/infra/Dockerfile b/infra/Dockerfile index 4bd181d..7c10073 100644 --- a/infra/Dockerfile +++ b/infra/Dockerfile @@ -1,8 +1,11 @@ # Same pinned base as deploy/Dockerfile.api, so the integration suite exercises # the image that ships rather than a different one. FROM python:3.12-slim@sha256:2f17fc044b579bab302c2e8054d3a686e2cb9a83de48e70534b94cd8ebbe06a9 +# pg_dump and age are for the database backup (ops/db_backup.py). Debian 13 +# ships PostgreSQL 17, the server's major version, which pg_dump must match. RUN apt-get update \ && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends postgresql-client-17 age \ && rm -rf /var/lib/apt/lists/* WORKDIR /app COPY infra/requirements.txt infra/requirements.lock /app/infra/ diff --git a/infra/storage-init.sh b/infra/storage-init.sh index 9757c1b..6c82c4c 100644 --- a/infra/storage-init.sh +++ b/infra/storage-init.sh @@ -21,4 +21,13 @@ printf '%s' "$policy" > /tmp/policy.json mc admin policy create local dtf-artwork /tmp/policy.json >/dev/null mc admin policy attach local dtf-artwork --user "$S3_APP_USER" >/dev/null mc ilm import "local/$S3_BUCKET" < /lifecycle.json +# The backup bucket and its own account, which can write and read backups but +# not delete them: the same separation as the backup token on R2. +case "$S3_BACKUP_BUCKET" in *[!a-z0-9.-]*|'') echo 'Invalid local backup bucket name' >&2; exit 1;; esac +mc mb --ignore-existing "local/$S3_BACKUP_BUCKET" >/dev/null +mc anonymous set none "local/$S3_BACKUP_BUCKET" >/dev/null +mc admin user add local "$S3_BACKUP_USER" "$S3_BACKUP_PASSWORD" >/dev/null +printf '%s' '{"Version":"2012-10-17","Statement":[{"Effect":"Allow","Action":["s3:ListBucket","s3:GetBucketLocation"],"Resource":["arn:aws:s3:::'"$S3_BACKUP_BUCKET"'"]},{"Effect":"Allow","Action":["s3:GetObject","s3:PutObject","s3:AbortMultipartUpload","s3:ListMultipartUploadParts"],"Resource":["arn:aws:s3:::'"$S3_BACKUP_BUCKET"'/*"]}]}' > /tmp/backup-policy.json +mc admin policy create local dtf-backup /tmp/backup-policy.json >/dev/null +mc admin policy attach local dtf-backup --user "$S3_BACKUP_USER" >/dev/null echo 'Local storage runtime account provisioned.' diff --git a/ops/db_backup.py b/ops/db_backup.py new file mode 100644 index 0000000..0e5a51f --- /dev/null +++ b/ops/db_backup.py @@ -0,0 +1,289 @@ +"""Daily off-server backup of the database to its own bucket. + + python -m ops.db_backup serve # the stack's backup service + python -m ops.db_backup once # one backup now + python -m ops.db_backup list # what the bucket holds + python -m ops.db_backup verify --identity F # restore the latest into a scratch database, then drop it + python -m ops.db_backup restore KEY --identity F --into NAME + +Each backup is `pg_dump --format=custom`, checked with `pg_restore --list`, +encrypted with age to BACKUP_AGE_RECIPIENT (a public key) and uploaded to +BACKUP_BUCKET with credentials of its own. The server holds only the public +key: it can write backups but not read them. The private key (the identity) +stays with the owner, off the server, and is needed only to restore. + +Old backups are removed by the bucket's lifecycle rule, not by this job, so +its credential never needs to delete; a bucket lock keeps a stolen one from +deleting either. Every run is recorded in dtf_local.backups, which the Kanban +shows on its Integrations tab. +""" +import argparse +import hashlib +import os +import subprocess +import sys +import tempfile +import time +from datetime import datetime, timedelta, timezone +from pathlib import Path +from uuid import uuid4 + +import boto3 +import psycopg +from botocore.config import Config +from psycopg import sql +from psycopg.conninfo import conninfo_to_dict + +from app.bootstrap import admin_connect +from app.core.secrets import load as load_secret_files + +BRASILIA = timezone(timedelta(hours=-3)) +PREFIX = 'db/' +# After a failure the next attempt comes this much later, not a day later. +RETRY = timedelta(hours=1) +# A backup older than this means the daily run has stopped working. +STALE = timedelta(hours=26) +TABLES = ('orders', 'quotes', 'uploads', 'accounts', 'movements', 'payment_intents') + + +def configured(): + return all(os.environ.get(k) for k in + ('BACKUP_S3_ENDPOINT', 'BACKUP_BUCKET', 'BACKUP_ACCESS_KEY_ID', + 'BACKUP_SECRET_ACCESS_KEY', 'BACKUP_AGE_RECIPIENT')) + + +def bucket(): + client = boto3.client('s3', endpoint_url=os.environ['BACKUP_S3_ENDPOINT'], + aws_access_key_id=os.environ['BACKUP_ACCESS_KEY_ID'], + aws_secret_access_key=os.environ['BACKUP_SECRET_ACCESS_KEY'], + region_name=os.environ.get('BACKUP_REGION', 'auto'), + config=Config(signature_version='s3v4', s3={'addressing_style': 'path'}, + retries={'max_attempts': 5, 'mode': 'standard'})) + return client, os.environ['BACKUP_BUCKET'] + + +def admin_params(database=None): + """The administrator's connection, optionally to another database.""" + if os.environ.get('DATABASE_ADMIN_HOST'): + params = {'host': os.environ['DATABASE_ADMIN_HOST'], 'user': os.environ['DATABASE_ADMIN_USER'], + 'password': os.environ['DATABASE_ADMIN_PASSWORD'], 'dbname': os.environ['DATABASE_ADMIN_NAME']} + else: + params = conninfo_to_dict(os.environ['DATABASE_ADMIN_URL']) + if database: + params['dbname'] = database + return params + + +def libpq_env(database=None): + """pg_dump and pg_restore read the credentials from the environment, so the + password never appears in a process list.""" + names = {'host': 'PGHOST', 'port': 'PGPORT', 'user': 'PGUSER', 'password': 'PGPASSWORD', 'dbname': 'PGDATABASE'} + return {**os.environ, **{names[k]: str(v) for k, v in admin_params(database).items() if k in names}} + + +def run(command, **kwargs): + result = subprocess.run(command, capture_output=True, **kwargs) + if result.returncode: + detail = result.stderr.decode(errors='replace').strip().splitlines() + raise RuntimeError(f'{command[0]} failed: ' + (detail[-1] if detail else f'exit {result.returncode}')) + return result + + +def sha256(path): + digest = hashlib.sha256() + with open(path, 'rb') as stream: + for block in iter(lambda: stream.read(1 << 20), b''): + digest.update(block) + return digest.hexdigest() + + +def record(started, status, key=None, size=None, detail=None): + with admin_connect() as c: + c.execute('''INSERT INTO dtf_local.backups(id,started_at,finished_at,status,object_key,bytes,detail) + VALUES(%s,%s,now(),%s,%s,%s,%s)''', + (uuid4(), started, status, key, size, detail)) + + +def run_once(): + """Dump, check, encrypt and upload one backup. Returns its object key.""" + started = datetime.now(timezone.utc) + key = PREFIX + started.strftime('%Y/%m/dtf-%Y%m%dT%H%M%SZ') + '.dump.age' + try: + client, name = bucket() + with tempfile.TemporaryDirectory(dir=os.environ.get('BACKUP_TMP')) as work: + dump, sealed = Path(work, 'dtf.dump'), Path(work, 'dtf.dump.age') + run(['pg_dump', '--format=custom', '--compress=6', '--file', str(dump)], env=libpq_env()) + # A truncated or damaged archive fails here, before it is kept. + run(['pg_restore', '--list', str(dump)]) + run(['age', '--encrypt', '--recipient', os.environ['BACKUP_AGE_RECIPIENT'], + '--output', str(sealed), str(dump)]) + size = sealed.stat().st_size + client.upload_file(str(sealed), name, key, ExtraArgs={'Metadata': { + 'sha256': sha256(sealed), 'dump-sha256': sha256(dump), 'format': 'pg-custom+age'}}) + record(started, 'ok', key, size) + print(f'Backup {key} uploaded ({size} bytes).', flush=True) + return key + except Exception as error: + # The message names a command or a provider error, never a credential. + detail = str(error)[:500] + try: + record(started, 'failed', key, None, detail) + finally: + print(f'Backup failed: {detail}', file=sys.stderr, flush=True) + raise + + +def last_ok(): + with admin_connect() as c: + row = c.execute("SELECT max(finished_at) FROM dtf_local.backups WHERE status='ok'").fetchone() + return row[0] + + +def next_run(now, hour): + """The next `hour` o'clock in Brasília after `now`.""" + local = now.astimezone(BRASILIA) + at = local.replace(hour=hour, minute=0, second=0, microsecond=0) + if at <= local: + at += timedelta(days=1) + return at.astimezone(timezone.utc) + + +def serve(): + if not configured(): + print('Backup not configured: set BACKUP_S3_ENDPOINT, BACKUP_BUCKET, BACKUP_ACCESS_KEY_ID, ' + 'BACKUP_SECRET_ACCESS_KEY and BACKUP_AGE_RECIPIENT. Idle.', flush=True) + while True: + time.sleep(3600) + hour = int(os.environ.get('BACKUP_HOUR', '3')) + # Wait for the schema (db-init) before the first query. + for _ in range(60): + try: + previous = last_ok() + break + except psycopg.Error: + time.sleep(10) + else: + previous = last_ok() + now = datetime.now(timezone.utc) + # A fresh deployment, or one that missed a day, backs up straight away. + due = now if previous is None or now - previous > STALE else next_run(now, hour) + while True: + time.sleep(max(0, (due - datetime.now(timezone.utc)).total_seconds())) + try: + run_once() + due = next_run(datetime.now(timezone.utc), hour) + except Exception: + due = datetime.now(timezone.utc) + RETRY + + +def listing(): + client, name = bucket() + keys = [] + for page in client.get_paginator('list_objects_v2').paginate(Bucket=name, Prefix=PREFIX): + keys += [(o['Key'], o['Size'], o['LastModified']) for o in page.get('Contents', [])] + return sorted(keys) + + +def counts(database): + with psycopg.connect(**admin_params(database)) as c: + return {t: c.execute(sql.SQL('SELECT count(*) FROM dtf_local.{}').format(sql.Identifier(t))).fetchone()[0] + for t in TABLES} + + +def restore(key, identity, into): + """Restore one backup into a database that is new or holds no DTF data.""" + client, name = bucket() + with admin_connect() as c: + c.autocommit = True + if into == c.execute('SELECT current_database()').fetchone()[0]: + raise SystemExit('Refusing to restore over the database the stack is using.') + exists = c.execute('SELECT 1 FROM pg_database WHERE datname=%s', (into,)).fetchone() + if not exists: + c.execute(sql.SQL('CREATE DATABASE {}').format(sql.Identifier(into))) + if exists: + with psycopg.connect(**admin_params(into)) as c: + if c.execute("SELECT 1 FROM pg_namespace WHERE nspname='dtf_local'").fetchone(): + raise SystemExit(f'{into} already holds DTF data; choose an empty or new database.') + with tempfile.TemporaryDirectory(dir=os.environ.get('BACKUP_TMP')) as work: + sealed, dump = Path(work, 'dtf.dump.age'), Path(work, 'dtf.dump') + client.download_file(name, key, str(sealed)) + expected = client.head_object(Bucket=name, Key=key)['Metadata'].get('sha256') + if expected and sha256(sealed) != expected: + raise SystemExit('The downloaded backup does not match its checksum.') + run(['age', '--decrypt', '--identity', identity, '--output', str(dump), str(sealed)]) + run(['pg_restore', '--exit-on-error', '--no-owner', '--no-privileges', '--dbname', into, str(dump)], + env=libpq_env(into)) + return counts(into) + + +def drop(database): + with admin_connect() as c: + c.autocommit = True + c.execute(sql.SQL('DROP DATABASE IF EXISTS {} WITH (FORCE)').format(sql.Identifier(database))) + + +def verify(identity, key=None): + """The restore test: the latest backup into a scratch database, counted, dropped.""" + key = key or listing()[-1][0] + scratch = 'dtf_verify_' + uuid4().hex[:12] + try: + restored = restore(key, identity, scratch) + finally: + drop(scratch) + print(f'PASS: {key} decrypted and restored into a scratch database (now dropped). Rows: ' + + ', '.join(f'{t} {n}' for t, n in restored.items())) + return restored + + +def identity_file(value): + """The identity as a path, or '-' to paste it: nothing is written to disk + except a private temporary file removed when the command ends.""" + if value != '-': + return value, None + print('Paste the private key (AGE-SECRET-KEY-...) and press Enter:', file=sys.stderr) + line = sys.stdin.readline().strip() + handle = tempfile.NamedTemporaryFile('w', prefix='identity-', dir=os.environ.get('BACKUP_TMP'), delete=False) + os.chmod(handle.name, 0o600) + handle.write(line + '\n') + handle.close() + return handle.name, handle.name + + +def main(argv=None): + load_secret_files() + parser = argparse.ArgumentParser(prog='python -m ops.db_backup') + sub = parser.add_subparsers(dest='command', required=True) + sub.add_parser('serve') + sub.add_parser('once') + sub.add_parser('list') + for name in ('verify', 'restore'): + p = sub.add_parser(name) + if name == 'restore': + p.add_argument('key') + p.add_argument('--into', required=True, help='a new or empty database') + else: + p.add_argument('key', nargs='?', help='default: the latest backup') + p.add_argument('--identity', required=True, help="the private key file, or '-' to paste it") + args = parser.parse_args(argv) + if args.command == 'serve': + serve() + elif args.command == 'once': + run_once() + elif args.command == 'list': + for key, size, modified in listing(): + print(f'{modified:%Y-%m-%d %H:%M} UTC {size:>12} {key}') + else: + path, temporary = identity_file(args.identity) + try: + if args.command == 'verify': + verify(path, args.key) + else: + rows = restore(args.key, path, args.into) + print(f'Restored {args.key} into {args.into}. Rows: ' + ', '.join(f'{t} {n}' for t, n in rows.items())) + finally: + if temporary: + os.unlink(temporary) + + +if __name__ == '__main__': + main() diff --git a/tests/backup_test.py b/tests/backup_test.py new file mode 100644 index 0000000..ca38884 --- /dev/null +++ b/tests/backup_test.py @@ -0,0 +1,79 @@ +"""The database backup against the local stack: dump, encrypt, upload, restore. + + docker compose -f compose.local.yaml exec -T backup python -m tests.backup_test + +Uses a throwaway age key made here, so nothing secret is kept in the repository. +""" +import os +import subprocess +import tempfile +from datetime import datetime, timezone +from pathlib import Path + +from botocore.exceptions import ClientError + +from app.bootstrap import admin_connect +from ops import db_backup + + +def check(condition, message): + if not condition: + raise AssertionError(message) + print('PASS:', message) + + +def main(): + with tempfile.TemporaryDirectory() as work: + identity = Path(work, 'identity.txt') + subprocess.run(['age-keygen', '-o', str(identity)], check=True, capture_output=True) + public = next(line.split(': ', 1)[1] for line in identity.read_text().splitlines() + if line.startswith('# public key: ')) + os.environ['BACKUP_AGE_RECIPIENT'] = public + check(db_backup.configured(), 'the local backup service is configured once a recipient is set') + + live = db_backup.counts(None) + key = db_backup.run_once() + with admin_connect() as c: + row = c.execute('SELECT status,bytes FROM dtf_local.backups WHERE object_key=%s', (key,)).fetchone() + check(row is not None and row[0] == 'ok' and row[1] > 0, 'the run is recorded for the Kanban') + + client, bucket = db_backup.bucket() + head = client.get_object(Bucket=bucket, Key=key, Range='bytes=0-20')['Body'].read() + check(head.startswith(b'age-encryption.org/v1'), 'what leaves the server is encrypted') + check(key in [k for k, _, _ in db_backup.listing()], 'the backup is listed in its bucket') + try: + client.delete_object(Bucket=bucket, Key=key) + deleted = True + except ClientError: + deleted = False + check(not deleted and key in [k for k, _, _ in db_backup.listing()], + 'the backup credential cannot delete backups') + + restored = db_backup.verify(str(identity), key) + check(restored == live, f'the restore has the same rows as the live database ({restored})') + + live_name = db_backup.admin_params()['dbname'] + try: + db_backup.restore(key, str(identity), live_name) + refused = False + except SystemExit: + refused = True + check(refused, 'a restore over the live database is refused') + + other = Path(work, 'other.txt') + subprocess.run(['age-keygen', '-o', str(other)], check=True, capture_output=True) + try: + db_backup.verify(str(other), key) + opened = True + except RuntimeError: + opened = False + check(not opened, 'another key cannot open the backup') + + brt = lambda h, m=0: datetime(2026, 9, 29, h + 3, m, tzinfo=timezone.utc) + check(db_backup.next_run(brt(2), 3) == brt(3), 'before 03:00 in Brasília the run is that day') + check(db_backup.next_run(brt(3), 3) == datetime(2026, 9, 30, 6, tzinfo=timezone.utc), + 'at or after 03:00 the run is the next day') + + +if __name__ == '__main__': + main() diff --git a/web/kanban.js b/web/kanban.js index 1a41b80..ae9ab12 100644 --- a/web/kanban.js +++ b/web/kanban.js @@ -706,6 +706,13 @@ function renderIntegrations(){ ' + '+kg(fr.per_metre_kg)+' por metro · prazo da Jadlog + '+fr.production_days+' dia(s) de produção.') : integration('Frete','Não configurado','off','Somente retirada.')); cards.push(integration('WhatsApp','Não configurado','off','Mensagens registradas, sem envio.')); + const bk=board.providers?.backup||{}; + const ontem=bk.last_ok&&Date.now()-new Date(bk.last_ok)<26*3600e3; + const tam=bk.bytes?' · '+(bk.bytes<1048576?Math.max(1,Math.round(bk.bytes/1024))+' KB':(bk.bytes/1048576).toFixed(1).replace('.',',')+' MB'):''; + cards.push(integration('Backup do banco',!bk.last_ok&&!bk.failed?'Não configurado':ontem&&!bk.failed?'Em dia':'Verificar', + !bk.last_ok&&!bk.failed?'off':ontem&&!bk.failed?'ok':'warn', + (bk.last_ok?'Último backup em '+when(bk.last_ok)+tam+'.':'Nenhum backup feito ainda.')+ + (bk.failed?' A última tentativa falhou ('+when(bk.failed_at)+'): '+bk.failed:'')+' Cópia diária, criptografada, fora do servidor.')); $('integrations').replaceChildren(...cards); eventFilters(); }