diff --git a/.gitea/workflows/deploy.yml b/.gitea/workflows/deploy.yml index 80d80a8..892b157 100644 --- a/.gitea/workflows/deploy.yml +++ b/.gitea/workflows/deploy.yml @@ -93,6 +93,7 @@ jobs: $COMPOSE exec -T api python -m tests.retention_test $COMPOSE exec -T api python -m tests.runtime_security_test $COMPOSE exec -T api python -m tests.tiny_oauth_test + $COMPOSE exec -T backup python -m tests.backup_test # Run Chrome on the Compose network. It must resolve the same storage:9000 # hostname used in presigned URLs, and absence of Chrome must fail CI. @@ -106,7 +107,7 @@ jobs: if: failure() run: | $COMPOSE ps || true - $COMPOSE logs --tail 200 api worker site kanban || true + $COMPOSE logs --tail 200 api worker backup site kanban || true - name: Tear down if: always() diff --git a/app/api/operator.py b/app/api/operator.py index 2c90e1a..d12341d 100644 --- a/app/api/operator.py +++ b/app/api/operator.py @@ -67,6 +67,16 @@ def freight_status(): 'production_days': freight.production_days} +def backup_status(c): + """The latest database backup and the latest run, which may have failed.""" + ok = c.execute('''SELECT finished_at,bytes FROM dtf_local.backups WHERE status='ok' + ORDER BY finished_at DESC LIMIT 1''').fetchone() + last = c.execute('SELECT finished_at,status,detail FROM dtf_local.backups ORDER BY finished_at DESC LIMIT 1').fetchone() + return {'last_ok': ok['finished_at'].isoformat() if ok else None, 'bytes': ok['bytes'] if ok else None, + 'failed': last['detail'] if last and last['status'] == 'failed' else None, + 'failed_at': last['finished_at'].isoformat() if last and last['status'] == 'failed' else None} + + @router.get('/api/operator/board') def board(user=Depends(operator)): with db.connect() as c: @@ -97,9 +107,10 @@ def board(user=Depends(operator)): # Pagamentos tab pages through them until someone records a resolution. issues_total = c.execute('''SELECT count(*) AS n FROM dtf_local.payment_events WHERE (outcome LIKE 'refused%' OR outcome LIKE 'attention%') AND resolved_at IS NULL''').fetchone()['n'] + backup = backup_status(c) return {'states': STATES, 'transitions': TRANSITIONS, 'back': BACK, 'orders': orders, 'payment_issues_total': issues_total, 'tiny': tiny_status(), 'operator': user, - 'providers': {'payment': payment.name, 'freight': freight_status()}, 'environment': ENVIRONMENT, + 'providers': {'payment': payment.name, 'freight': freight_status(), 'backup': backup}, 'environment': ENVIRONMENT, 'finished_shown': len(finished), 'finished_total': finished_total, 'quotes': [quote_view(c, q) for q in pending + approved], 'pending_total': pending_total, 'approved_total': approved_total} diff --git a/app/schema.sql b/app/schema.sql index 5ff9bbc..6c1f47f 100644 --- a/app/schema.sql +++ b/app/schema.sql @@ -134,6 +134,14 @@ CREATE TABLE IF NOT EXISTS dtf_local.print_files ( UNIQUE(order_id,item_index) ); +-- Each run of the off-server database backup (ops/db_backup.py), which the +-- Kanban shows so a backup that stopped working does not go unnoticed. +CREATE TABLE IF NOT EXISTS dtf_local.backups ( + id uuid PRIMARY KEY, started_at timestamptz NOT NULL, finished_at timestamptz NOT NULL, + status text NOT NULL CHECK(status IN ('ok','failed')), object_key text, bytes bigint, detail text +); +CREATE INDEX IF NOT EXISTS backups_finished ON dtf_local.backups(finished_at DESC); + CREATE INDEX IF NOT EXISTS uploads_owner ON dtf_local.uploads(owner); -- Indexes follow the queries the application actually issues. Only these; every diff --git a/compose.local.yaml b/compose.local.yaml index 13cb2f4..5f02e10 100644 --- a/compose.local.yaml +++ b/compose.local.yaml @@ -148,6 +148,9 @@ services: S3_APP_USER: ${S3_APP_USER:-dtf_app} S3_APP_PASSWORD: ${S3_APP_PASSWORD:-local-app-storage-only} S3_BUCKET: ${S3_BUCKET:-dtf-local-artwork} + S3_BACKUP_BUCKET: dtf-local-backups + S3_BACKUP_USER: dtf_backup + S3_BACKUP_PASSWORD: local-backup-storage-only networks: [local] depends_on: storage: {condition: service_healthy} @@ -197,6 +200,28 @@ services: timeout: 3s retries: 12 + # The daily database backup, against the local backup bucket. Idle until + # BACKUP_AGE_RECIPIENT is set; tests/backup_test.py runs it with a throwaway key. + backup: + build: + context: . + dockerfile: infra/Dockerfile + command: python -m ops.db_backup serve + environment: + DATABASE_ADMIN_URL: postgresql://${POSTGRES_USER:-dtf_local}:${POSTGRES_PASSWORD:-local-database-only}@db:5432/${POSTGRES_DB:-dtf_local} + BACKUP_S3_ENDPOINT: http://storage:9000 + BACKUP_BUCKET: dtf-local-backups + BACKUP_ACCESS_KEY_ID: dtf_backup + BACKUP_SECRET_ACCESS_KEY: local-backup-storage-only + BACKUP_REGION: us-east-1 + BACKUP_AGE_RECIPIENT: ${BACKUP_AGE_RECIPIENT:-} + networks: [local] + read_only: true + tmpfs: [/tmp] + depends_on: + db-init: {condition: service_completed_successfully} + storage-init: {condition: service_completed_successfully} + site: build: context: . diff --git a/deploy/Dockerfile.api b/deploy/Dockerfile.api index 93a1f7c..074dc7c 100644 --- a/deploy/Dockerfile.api +++ b/deploy/Dockerfile.api @@ -13,8 +13,11 @@ LABEL org.opencontainers.image.title="DTF Portal/API" \ # The base is pinned, so its OS packages are frozen at the digest's build date. # Upgrade them here or the image ships known-fixed Debian vulnerabilities, which # is what the production image was doing while the local one already did this. +# pg_dump and age are for the database backup (ops/db_backup.py). Debian 13 +# ships PostgreSQL 17, the server's major version, which pg_dump must match. RUN apt-get update \ && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends postgresql-client-17 age \ && rm -rf /var/lib/apt/lists/* WORKDIR /app diff --git a/deploy/portainer.env.example b/deploy/portainer.env.example index 3694ecb..bcb2dde 100644 --- a/deploy/portainer.env.example +++ b/deploy/portainer.env.example @@ -51,6 +51,13 @@ MAX_PENDING_UPLOADS=10 MAX_UPLOAD_BYTES=5368709120 UPLOAD_PART_BYTES=8388608 SCAN_MAX_BYTES=134217728 +# Daily encrypted database backup to its own bucket (docs/BACKUP.md). +BACKUP_S3_ENDPOINT=https://.r2.cloudflarestorage.com +BACKUP_BUCKET=dtf-backups +BACKUP_ACCESS_KEY_ID=TBD +BACKUP_SECRET_ACCESS_KEY=TBD +BACKUP_AGE_RECIPIENT=age1... +BACKUP_HOUR=3 # Names of external Portainer/Docker Swarm secrets, never their values. DATABASE_URL_SECRET=TBD diff --git a/docker-compose.yml b/docker-compose.yml index 6471eba..fa775e3 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -183,6 +183,34 @@ services: replicas: 1 restart_policy: {condition: on-failure, delay: 5s} + # Daily encrypted copy of the database to a bucket of its own, off this + # server (ops/db_backup.py). Idle until the BACKUP_* settings are set. The + # bucket's lifecycle rule removes old copies; this credential never deletes. + backup: + image: ${API_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-api}:${IMAGE_TAG:-latest} + command: python -m ops.db_backup serve + environment: + DATABASE_ADMIN_HOST: db + DATABASE_ADMIN_NAME: dtf + DATABASE_ADMIN_USER: dtf_admin + DATABASE_ADMIN_PASSWORD: ${POSTGRES_PASSWORD:?set POSTGRES_PASSWORD} + # The R2 account endpoint (the same as R2_ENDPOINT) and a bucket and API + # token for backups only, never the artwork bucket's. + BACKUP_S3_ENDPOINT: ${BACKUP_S3_ENDPOINT:-} + BACKUP_BUCKET: ${BACKUP_BUCKET:-} + BACKUP_ACCESS_KEY_ID: ${BACKUP_ACCESS_KEY_ID:-} + BACKUP_SECRET_ACCESS_KEY: ${BACKUP_SECRET_ACCESS_KEY:-} + # The public key (age1...). Its private key stays off this server. + BACKUP_AGE_RECIPIENT: ${BACKUP_AGE_RECIPIENT:-} + # Hour of day in Brasília. + BACKUP_HOUR: ${BACKUP_HOUR:-3} + networks: [backend, egress] + read_only: true + tmpfs: [/tmp] + deploy: + replicas: 1 + restart_policy: {condition: on-failure, delay: 30s} + site: image: ${WEB_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-web}:${IMAGE_TAG:-latest} environment: diff --git a/docs/BACKUP.md b/docs/BACKUP.md new file mode 100644 index 0000000..e78605a --- /dev/null +++ b/docs/BACKUP.md @@ -0,0 +1,87 @@ +# Database backup + +The `backup` service copies the database off the server once a day +(`ops/db_backup.py`). It runs `pg_dump`, checks the archive, encrypts it with +[age](https://age-encryption.org) and uploads it to an R2 bucket used for +nothing else. Each run is recorded, and the Kanban shows the latest one under +Integrações → "Backup do banco" ("Em dia", "Verificar" or "Não configurado"). + +The artwork is not backed up. Originals and print files are temporary (7 to 30 +days) and already stored on R2. The database holds what cannot be recreated: +orders, customers, quotes, payments, the Tiny connection and the operators. + +## Why the backup cannot be read or deleted from the server + +- **The server only encrypts.** It holds the public key (`age1…`). The private + key, which is the only way to open a backup, stays with the owner. +- **Its own bucket and token.** The backup token reaches the backup bucket + only, and the artwork token cannot reach it. +- **Nothing is deleted by the job.** The bucket's lifecycle rule removes old + copies. The bucket lock keeps anyone, including someone holding the token, + from deleting or overwriting a copy before its time. + +## Setup (once) + +1. **Key pair.** Generate it on your own computer, not on the server: + + ```bash + docker run --rm gitea.blyzer.com.br/blyzer/dtf-api:latest age-keygen + ``` + + Or, with age installed (`sudo pacman -S age`, `apt install age`), just + run `age-keygen`. + + Save the whole output (the `AGE-SECRET-KEY-1…` line is the private key) in + the password manager. Only the `public key: age1…` value goes to Portainer. + Without the private key, no backup can ever be restored. +2. **Cloudflare R2 → Create bucket.** Use a name such as `dtf-backups`. +3. **Settings on that bucket:** + - Object lifecycle rules: delete objects after 35 days. + - Bucket lock rules: retain for 30 days, prefix `db/`. +4. **R2 → Manage API tokens → Create API token.** + - Permission: Object Read & Write. + - Scope: the `dtf-backups` bucket only. +5. **Portainer.** Add these to the stack's environment, then redeploy: + + | Variable | Value | + |---|---| + | `BACKUP_S3_ENDPOINT` | same as `R2_ENDPOINT` | + | `BACKUP_BUCKET` | `dtf-backups` | + | `BACKUP_ACCESS_KEY_ID` | the new token's Access Key ID | + | `BACKUP_SECRET_ACCESS_KEY` | the new token's Secret Access Key | + | `BACKUP_AGE_RECIPIENT` | the public key, `age1…` | + | `BACKUP_HOUR` | optional, default `3` (03:00 Brasília) | + +The first backup runs as soon as the service starts. After that it runs daily, +and a failed run is retried an hour later. + +## Restore test + +Do this after setup, and then once a month. Open the console of the `backup` +container in Portainer (Containers → `…_backup…` → Console) and run: + +```bash +python -m ops.db_backup list +python -m ops.db_backup verify --identity - +``` + +`--identity -` asks for the private key to be pasted. It is kept only in the +container's memory while the command runs. `verify` restores the latest backup +into a scratch database, prints its row counts and drops it. The live database +is not touched. + +## Restoring for real + +From the `backup` container's console: + +```bash +python -m ops.db_backup restore db/2026/10/dtf-20261001T060000Z.dump.age --into dtf_restored --identity - +``` + +This restores into a new database (or an empty one), never over the running +one. Then do one of these: + +- **Check the restored data**, then point the stack at it. +- **On a new server**, deploy only the `db` service first. Restore into `dtf` + while it is still empty, then deploy the rest. `db-init` then applies the + schema and grants on top. diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 451a50d..48a17a7 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -864,6 +864,12 @@ print-file evidence still need correction before this item can close. - `[ ]` 5.13 — Define production recovery: scheduled encrypted offsite database and object backups, a consistent snapshot boundary, Swarm data placement and a restore rehearsal that opens every required live order file. + **2026-09-29:** database part built. The `backup` service dumps daily, + encrypts with age (the server holds only the public key) and uploads to a + bucket of its own, whose lifecycle rule expires copies and whose lock keeps + them from being deleted early. The Kanban shows the latest run, and + `tests/backup_test.py` restores a copy in CI. Setup and restore: + docs/BACKUP.md. Artwork is not copied: it is temporary and already on R2. - `[ ]` 5.14 — Promote and verify one immutable release. **2026-09-24:** green pushes to `main` now publish images and Portainer's pull-and-redeploy is the release gate; the source preflight is advisory unless enforced by variable, diff --git a/infra/Dockerfile b/infra/Dockerfile index 4bd181d..7c10073 100644 --- a/infra/Dockerfile +++ b/infra/Dockerfile @@ -1,8 +1,11 @@ # Same pinned base as deploy/Dockerfile.api, so the integration suite exercises # the image that ships rather than a different one. FROM python:3.12-slim@sha256:2f17fc044b579bab302c2e8054d3a686e2cb9a83de48e70534b94cd8ebbe06a9 +# pg_dump and age are for the database backup (ops/db_backup.py). Debian 13 +# ships PostgreSQL 17, the server's major version, which pg_dump must match. RUN apt-get update \ && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends postgresql-client-17 age \ && rm -rf /var/lib/apt/lists/* WORKDIR /app COPY infra/requirements.txt infra/requirements.lock /app/infra/ diff --git a/infra/storage-init.sh b/infra/storage-init.sh index 9757c1b..6c82c4c 100644 --- a/infra/storage-init.sh +++ b/infra/storage-init.sh @@ -21,4 +21,13 @@ printf '%s' "$policy" > /tmp/policy.json mc admin policy create local dtf-artwork /tmp/policy.json >/dev/null mc admin policy attach local dtf-artwork --user "$S3_APP_USER" >/dev/null mc ilm import "local/$S3_BUCKET" < /lifecycle.json +# The backup bucket and its own account, which can write and read backups but +# not delete them: the same separation as the backup token on R2. +case "$S3_BACKUP_BUCKET" in *[!a-z0-9.-]*|'') echo 'Invalid local backup bucket name' >&2; exit 1;; esac +mc mb --ignore-existing "local/$S3_BACKUP_BUCKET" >/dev/null +mc anonymous set none "local/$S3_BACKUP_BUCKET" >/dev/null +mc admin user add local "$S3_BACKUP_USER" "$S3_BACKUP_PASSWORD" >/dev/null +printf '%s' '{"Version":"2012-10-17","Statement":[{"Effect":"Allow","Action":["s3:ListBucket","s3:GetBucketLocation"],"Resource":["arn:aws:s3:::'"$S3_BACKUP_BUCKET"'"]},{"Effect":"Allow","Action":["s3:GetObject","s3:PutObject","s3:AbortMultipartUpload","s3:ListMultipartUploadParts"],"Resource":["arn:aws:s3:::'"$S3_BACKUP_BUCKET"'/*"]}]}' > /tmp/backup-policy.json +mc admin policy create local dtf-backup /tmp/backup-policy.json >/dev/null +mc admin policy attach local dtf-backup --user "$S3_BACKUP_USER" >/dev/null echo 'Local storage runtime account provisioned.' diff --git a/ops/db_backup.py b/ops/db_backup.py new file mode 100644 index 0000000..0e5a51f --- /dev/null +++ b/ops/db_backup.py @@ -0,0 +1,289 @@ +"""Daily off-server backup of the database to its own bucket. + + python -m ops.db_backup serve # the stack's backup service + python -m ops.db_backup once # one backup now + python -m ops.db_backup list # what the bucket holds + python -m ops.db_backup verify --identity F # restore the latest into a scratch database, then drop it + python -m ops.db_backup restore KEY --identity F --into NAME + +Each backup is `pg_dump --format=custom`, checked with `pg_restore --list`, +encrypted with age to BACKUP_AGE_RECIPIENT (a public key) and uploaded to +BACKUP_BUCKET with credentials of its own. The server holds only the public +key: it can write backups but not read them. The private key (the identity) +stays with the owner, off the server, and is needed only to restore. + +Old backups are removed by the bucket's lifecycle rule, not by this job, so +its credential never needs to delete; a bucket lock keeps a stolen one from +deleting either. Every run is recorded in dtf_local.backups, which the Kanban +shows on its Integrations tab. +""" +import argparse +import hashlib +import os +import subprocess +import sys +import tempfile +import time +from datetime import datetime, timedelta, timezone +from pathlib import Path +from uuid import uuid4 + +import boto3 +import psycopg +from botocore.config import Config +from psycopg import sql +from psycopg.conninfo import conninfo_to_dict + +from app.bootstrap import admin_connect +from app.core.secrets import load as load_secret_files + +BRASILIA = timezone(timedelta(hours=-3)) +PREFIX = 'db/' +# After a failure the next attempt comes this much later, not a day later. +RETRY = timedelta(hours=1) +# A backup older than this means the daily run has stopped working. +STALE = timedelta(hours=26) +TABLES = ('orders', 'quotes', 'uploads', 'accounts', 'movements', 'payment_intents') + + +def configured(): + return all(os.environ.get(k) for k in + ('BACKUP_S3_ENDPOINT', 'BACKUP_BUCKET', 'BACKUP_ACCESS_KEY_ID', + 'BACKUP_SECRET_ACCESS_KEY', 'BACKUP_AGE_RECIPIENT')) + + +def bucket(): + client = boto3.client('s3', endpoint_url=os.environ['BACKUP_S3_ENDPOINT'], + aws_access_key_id=os.environ['BACKUP_ACCESS_KEY_ID'], + aws_secret_access_key=os.environ['BACKUP_SECRET_ACCESS_KEY'], + region_name=os.environ.get('BACKUP_REGION', 'auto'), + config=Config(signature_version='s3v4', s3={'addressing_style': 'path'}, + retries={'max_attempts': 5, 'mode': 'standard'})) + return client, os.environ['BACKUP_BUCKET'] + + +def admin_params(database=None): + """The administrator's connection, optionally to another database.""" + if os.environ.get('DATABASE_ADMIN_HOST'): + params = {'host': os.environ['DATABASE_ADMIN_HOST'], 'user': os.environ['DATABASE_ADMIN_USER'], + 'password': os.environ['DATABASE_ADMIN_PASSWORD'], 'dbname': os.environ['DATABASE_ADMIN_NAME']} + else: + params = conninfo_to_dict(os.environ['DATABASE_ADMIN_URL']) + if database: + params['dbname'] = database + return params + + +def libpq_env(database=None): + """pg_dump and pg_restore read the credentials from the environment, so the + password never appears in a process list.""" + names = {'host': 'PGHOST', 'port': 'PGPORT', 'user': 'PGUSER', 'password': 'PGPASSWORD', 'dbname': 'PGDATABASE'} + return {**os.environ, **{names[k]: str(v) for k, v in admin_params(database).items() if k in names}} + + +def run(command, **kwargs): + result = subprocess.run(command, capture_output=True, **kwargs) + if result.returncode: + detail = result.stderr.decode(errors='replace').strip().splitlines() + raise RuntimeError(f'{command[0]} failed: ' + (detail[-1] if detail else f'exit {result.returncode}')) + return result + + +def sha256(path): + digest = hashlib.sha256() + with open(path, 'rb') as stream: + for block in iter(lambda: stream.read(1 << 20), b''): + digest.update(block) + return digest.hexdigest() + + +def record(started, status, key=None, size=None, detail=None): + with admin_connect() as c: + c.execute('''INSERT INTO dtf_local.backups(id,started_at,finished_at,status,object_key,bytes,detail) + VALUES(%s,%s,now(),%s,%s,%s,%s)''', + (uuid4(), started, status, key, size, detail)) + + +def run_once(): + """Dump, check, encrypt and upload one backup. Returns its object key.""" + started = datetime.now(timezone.utc) + key = PREFIX + started.strftime('%Y/%m/dtf-%Y%m%dT%H%M%SZ') + '.dump.age' + try: + client, name = bucket() + with tempfile.TemporaryDirectory(dir=os.environ.get('BACKUP_TMP')) as work: + dump, sealed = Path(work, 'dtf.dump'), Path(work, 'dtf.dump.age') + run(['pg_dump', '--format=custom', '--compress=6', '--file', str(dump)], env=libpq_env()) + # A truncated or damaged archive fails here, before it is kept. + run(['pg_restore', '--list', str(dump)]) + run(['age', '--encrypt', '--recipient', os.environ['BACKUP_AGE_RECIPIENT'], + '--output', str(sealed), str(dump)]) + size = sealed.stat().st_size + client.upload_file(str(sealed), name, key, ExtraArgs={'Metadata': { + 'sha256': sha256(sealed), 'dump-sha256': sha256(dump), 'format': 'pg-custom+age'}}) + record(started, 'ok', key, size) + print(f'Backup {key} uploaded ({size} bytes).', flush=True) + return key + except Exception as error: + # The message names a command or a provider error, never a credential. + detail = str(error)[:500] + try: + record(started, 'failed', key, None, detail) + finally: + print(f'Backup failed: {detail}', file=sys.stderr, flush=True) + raise + + +def last_ok(): + with admin_connect() as c: + row = c.execute("SELECT max(finished_at) FROM dtf_local.backups WHERE status='ok'").fetchone() + return row[0] + + +def next_run(now, hour): + """The next `hour` o'clock in Brasília after `now`.""" + local = now.astimezone(BRASILIA) + at = local.replace(hour=hour, minute=0, second=0, microsecond=0) + if at <= local: + at += timedelta(days=1) + return at.astimezone(timezone.utc) + + +def serve(): + if not configured(): + print('Backup not configured: set BACKUP_S3_ENDPOINT, BACKUP_BUCKET, BACKUP_ACCESS_KEY_ID, ' + 'BACKUP_SECRET_ACCESS_KEY and BACKUP_AGE_RECIPIENT. Idle.', flush=True) + while True: + time.sleep(3600) + hour = int(os.environ.get('BACKUP_HOUR', '3')) + # Wait for the schema (db-init) before the first query. + for _ in range(60): + try: + previous = last_ok() + break + except psycopg.Error: + time.sleep(10) + else: + previous = last_ok() + now = datetime.now(timezone.utc) + # A fresh deployment, or one that missed a day, backs up straight away. + due = now if previous is None or now - previous > STALE else next_run(now, hour) + while True: + time.sleep(max(0, (due - datetime.now(timezone.utc)).total_seconds())) + try: + run_once() + due = next_run(datetime.now(timezone.utc), hour) + except Exception: + due = datetime.now(timezone.utc) + RETRY + + +def listing(): + client, name = bucket() + keys = [] + for page in client.get_paginator('list_objects_v2').paginate(Bucket=name, Prefix=PREFIX): + keys += [(o['Key'], o['Size'], o['LastModified']) for o in page.get('Contents', [])] + return sorted(keys) + + +def counts(database): + with psycopg.connect(**admin_params(database)) as c: + return {t: c.execute(sql.SQL('SELECT count(*) FROM dtf_local.{}').format(sql.Identifier(t))).fetchone()[0] + for t in TABLES} + + +def restore(key, identity, into): + """Restore one backup into a database that is new or holds no DTF data.""" + client, name = bucket() + with admin_connect() as c: + c.autocommit = True + if into == c.execute('SELECT current_database()').fetchone()[0]: + raise SystemExit('Refusing to restore over the database the stack is using.') + exists = c.execute('SELECT 1 FROM pg_database WHERE datname=%s', (into,)).fetchone() + if not exists: + c.execute(sql.SQL('CREATE DATABASE {}').format(sql.Identifier(into))) + if exists: + with psycopg.connect(**admin_params(into)) as c: + if c.execute("SELECT 1 FROM pg_namespace WHERE nspname='dtf_local'").fetchone(): + raise SystemExit(f'{into} already holds DTF data; choose an empty or new database.') + with tempfile.TemporaryDirectory(dir=os.environ.get('BACKUP_TMP')) as work: + sealed, dump = Path(work, 'dtf.dump.age'), Path(work, 'dtf.dump') + client.download_file(name, key, str(sealed)) + expected = client.head_object(Bucket=name, Key=key)['Metadata'].get('sha256') + if expected and sha256(sealed) != expected: + raise SystemExit('The downloaded backup does not match its checksum.') + run(['age', '--decrypt', '--identity', identity, '--output', str(dump), str(sealed)]) + run(['pg_restore', '--exit-on-error', '--no-owner', '--no-privileges', '--dbname', into, str(dump)], + env=libpq_env(into)) + return counts(into) + + +def drop(database): + with admin_connect() as c: + c.autocommit = True + c.execute(sql.SQL('DROP DATABASE IF EXISTS {} WITH (FORCE)').format(sql.Identifier(database))) + + +def verify(identity, key=None): + """The restore test: the latest backup into a scratch database, counted, dropped.""" + key = key or listing()[-1][0] + scratch = 'dtf_verify_' + uuid4().hex[:12] + try: + restored = restore(key, identity, scratch) + finally: + drop(scratch) + print(f'PASS: {key} decrypted and restored into a scratch database (now dropped). Rows: ' + + ', '.join(f'{t} {n}' for t, n in restored.items())) + return restored + + +def identity_file(value): + """The identity as a path, or '-' to paste it: nothing is written to disk + except a private temporary file removed when the command ends.""" + if value != '-': + return value, None + print('Paste the private key (AGE-SECRET-KEY-...) and press Enter:', file=sys.stderr) + line = sys.stdin.readline().strip() + handle = tempfile.NamedTemporaryFile('w', prefix='identity-', dir=os.environ.get('BACKUP_TMP'), delete=False) + os.chmod(handle.name, 0o600) + handle.write(line + '\n') + handle.close() + return handle.name, handle.name + + +def main(argv=None): + load_secret_files() + parser = argparse.ArgumentParser(prog='python -m ops.db_backup') + sub = parser.add_subparsers(dest='command', required=True) + sub.add_parser('serve') + sub.add_parser('once') + sub.add_parser('list') + for name in ('verify', 'restore'): + p = sub.add_parser(name) + if name == 'restore': + p.add_argument('key') + p.add_argument('--into', required=True, help='a new or empty database') + else: + p.add_argument('key', nargs='?', help='default: the latest backup') + p.add_argument('--identity', required=True, help="the private key file, or '-' to paste it") + args = parser.parse_args(argv) + if args.command == 'serve': + serve() + elif args.command == 'once': + run_once() + elif args.command == 'list': + for key, size, modified in listing(): + print(f'{modified:%Y-%m-%d %H:%M} UTC {size:>12} {key}') + else: + path, temporary = identity_file(args.identity) + try: + if args.command == 'verify': + verify(path, args.key) + else: + rows = restore(args.key, path, args.into) + print(f'Restored {args.key} into {args.into}. Rows: ' + ', '.join(f'{t} {n}' for t, n in rows.items())) + finally: + if temporary: + os.unlink(temporary) + + +if __name__ == '__main__': + main() diff --git a/tests/backup_test.py b/tests/backup_test.py new file mode 100644 index 0000000..ca38884 --- /dev/null +++ b/tests/backup_test.py @@ -0,0 +1,79 @@ +"""The database backup against the local stack: dump, encrypt, upload, restore. + + docker compose -f compose.local.yaml exec -T backup python -m tests.backup_test + +Uses a throwaway age key made here, so nothing secret is kept in the repository. +""" +import os +import subprocess +import tempfile +from datetime import datetime, timezone +from pathlib import Path + +from botocore.exceptions import ClientError + +from app.bootstrap import admin_connect +from ops import db_backup + + +def check(condition, message): + if not condition: + raise AssertionError(message) + print('PASS:', message) + + +def main(): + with tempfile.TemporaryDirectory() as work: + identity = Path(work, 'identity.txt') + subprocess.run(['age-keygen', '-o', str(identity)], check=True, capture_output=True) + public = next(line.split(': ', 1)[1] for line in identity.read_text().splitlines() + if line.startswith('# public key: ')) + os.environ['BACKUP_AGE_RECIPIENT'] = public + check(db_backup.configured(), 'the local backup service is configured once a recipient is set') + + live = db_backup.counts(None) + key = db_backup.run_once() + with admin_connect() as c: + row = c.execute('SELECT status,bytes FROM dtf_local.backups WHERE object_key=%s', (key,)).fetchone() + check(row is not None and row[0] == 'ok' and row[1] > 0, 'the run is recorded for the Kanban') + + client, bucket = db_backup.bucket() + head = client.get_object(Bucket=bucket, Key=key, Range='bytes=0-20')['Body'].read() + check(head.startswith(b'age-encryption.org/v1'), 'what leaves the server is encrypted') + check(key in [k for k, _, _ in db_backup.listing()], 'the backup is listed in its bucket') + try: + client.delete_object(Bucket=bucket, Key=key) + deleted = True + except ClientError: + deleted = False + check(not deleted and key in [k for k, _, _ in db_backup.listing()], + 'the backup credential cannot delete backups') + + restored = db_backup.verify(str(identity), key) + check(restored == live, f'the restore has the same rows as the live database ({restored})') + + live_name = db_backup.admin_params()['dbname'] + try: + db_backup.restore(key, str(identity), live_name) + refused = False + except SystemExit: + refused = True + check(refused, 'a restore over the live database is refused') + + other = Path(work, 'other.txt') + subprocess.run(['age-keygen', '-o', str(other)], check=True, capture_output=True) + try: + db_backup.verify(str(other), key) + opened = True + except RuntimeError: + opened = False + check(not opened, 'another key cannot open the backup') + + brt = lambda h, m=0: datetime(2026, 9, 29, h + 3, m, tzinfo=timezone.utc) + check(db_backup.next_run(brt(2), 3) == brt(3), 'before 03:00 in Brasília the run is that day') + check(db_backup.next_run(brt(3), 3) == datetime(2026, 9, 30, 6, tzinfo=timezone.utc), + 'at or after 03:00 the run is the next day') + + +if __name__ == '__main__': + main() diff --git a/web/kanban.js b/web/kanban.js index 1a41b80..ae9ab12 100644 --- a/web/kanban.js +++ b/web/kanban.js @@ -706,6 +706,13 @@ function renderIntegrations(){ ' + '+kg(fr.per_metre_kg)+' por metro · prazo da Jadlog + '+fr.production_days+' dia(s) de produção.') : integration('Frete','Não configurado','off','Somente retirada.')); cards.push(integration('WhatsApp','Não configurado','off','Mensagens registradas, sem envio.')); + const bk=board.providers?.backup||{}; + const ontem=bk.last_ok&&Date.now()-new Date(bk.last_ok)<26*3600e3; + const tam=bk.bytes?' · '+(bk.bytes<1048576?Math.max(1,Math.round(bk.bytes/1024))+' KB':(bk.bytes/1048576).toFixed(1).replace('.',',')+' MB'):''; + cards.push(integration('Backup do banco',!bk.last_ok&&!bk.failed?'Não configurado':ontem&&!bk.failed?'Em dia':'Verificar', + !bk.last_ok&&!bk.failed?'off':ontem&&!bk.failed?'ok':'warn', + (bk.last_ok?'Último backup em '+when(bk.last_ok)+tam+'.':'Nenhum backup feito ainda.')+ + (bk.failed?' A última tentativa falhou ('+when(bk.failed_at)+'): '+bk.failed:'')+' Cópia diária, criptografada, fora do servidor.')); $('integrations').replaceChildren(...cards); eventFilters(); }