feat: daily encrypted database backup to a bucket of its own
All checks were successful
Build and deploy / Validate source (push) Successful in 8s
Build and deploy / Integration suite on a real stack (push) Successful in 3m38s
Build and deploy / Secret scan and release gate (push) Successful in 7s
Build and deploy / Publish images (push) Successful in 1m2s

A backup service runs pg_dump every day at 03:00 Brasília, checks the archive,
encrypts it with age to a public key and uploads it with a token for that
bucket only. The server cannot read or delete backups: the private key stays
with the owner, the bucket's lifecycle rule expires copies and its lock stops
early deletion. Each run is recorded and shown on the Kanban's Integrations
tab. tests/backup_test.py backs up, restores into a scratch database and
compares the rows in CI. Setup and restore: docs/BACKUP.md.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Cauê Faleiros
2026-09-29 18:49:21 -03:00
parent 1b57313496
commit 20403c5132
14 changed files with 565 additions and 2 deletions

View File

@@ -93,6 +93,7 @@ jobs:
$COMPOSE exec -T api python -m tests.retention_test $COMPOSE exec -T api python -m tests.retention_test
$COMPOSE exec -T api python -m tests.runtime_security_test $COMPOSE exec -T api python -m tests.runtime_security_test
$COMPOSE exec -T api python -m tests.tiny_oauth_test $COMPOSE exec -T api python -m tests.tiny_oauth_test
$COMPOSE exec -T backup python -m tests.backup_test
# Run Chrome on the Compose network. It must resolve the same storage:9000 # Run Chrome on the Compose network. It must resolve the same storage:9000
# hostname used in presigned URLs, and absence of Chrome must fail CI. # hostname used in presigned URLs, and absence of Chrome must fail CI.
@@ -106,7 +107,7 @@ jobs:
if: failure() if: failure()
run: | run: |
$COMPOSE ps || true $COMPOSE ps || true
$COMPOSE logs --tail 200 api worker site kanban || true $COMPOSE logs --tail 200 api worker backup site kanban || true
- name: Tear down - name: Tear down
if: always() if: always()

View File

@@ -67,6 +67,16 @@ def freight_status():
'production_days': freight.production_days} 'production_days': freight.production_days}
def backup_status(c):
"""The latest database backup and the latest run, which may have failed."""
ok = c.execute('''SELECT finished_at,bytes FROM dtf_local.backups WHERE status='ok'
ORDER BY finished_at DESC LIMIT 1''').fetchone()
last = c.execute('SELECT finished_at,status,detail FROM dtf_local.backups ORDER BY finished_at DESC LIMIT 1').fetchone()
return {'last_ok': ok['finished_at'].isoformat() if ok else None, 'bytes': ok['bytes'] if ok else None,
'failed': last['detail'] if last and last['status'] == 'failed' else None,
'failed_at': last['finished_at'].isoformat() if last and last['status'] == 'failed' else None}
@router.get('/api/operator/board') @router.get('/api/operator/board')
def board(user=Depends(operator)): def board(user=Depends(operator)):
with db.connect() as c: with db.connect() as c:
@@ -97,9 +107,10 @@ def board(user=Depends(operator)):
# Pagamentos tab pages through them until someone records a resolution. # Pagamentos tab pages through them until someone records a resolution.
issues_total = c.execute('''SELECT count(*) AS n FROM dtf_local.payment_events issues_total = c.execute('''SELECT count(*) AS n FROM dtf_local.payment_events
WHERE (outcome LIKE 'refused%' OR outcome LIKE 'attention%') AND resolved_at IS NULL''').fetchone()['n'] WHERE (outcome LIKE 'refused%' OR outcome LIKE 'attention%') AND resolved_at IS NULL''').fetchone()['n']
backup = backup_status(c)
return {'states': STATES, 'transitions': TRANSITIONS, 'back': BACK, return {'states': STATES, 'transitions': TRANSITIONS, 'back': BACK,
'orders': orders, 'payment_issues_total': issues_total, 'tiny': tiny_status(), 'operator': user, 'orders': orders, 'payment_issues_total': issues_total, 'tiny': tiny_status(), 'operator': user,
'providers': {'payment': payment.name, 'freight': freight_status()}, 'environment': ENVIRONMENT, 'providers': {'payment': payment.name, 'freight': freight_status(), 'backup': backup}, 'environment': ENVIRONMENT,
'finished_shown': len(finished), 'finished_total': finished_total, 'finished_shown': len(finished), 'finished_total': finished_total,
'quotes': [quote_view(c, q) for q in pending + approved], 'quotes': [quote_view(c, q) for q in pending + approved],
'pending_total': pending_total, 'approved_total': approved_total} 'pending_total': pending_total, 'approved_total': approved_total}

View File

@@ -134,6 +134,14 @@ CREATE TABLE IF NOT EXISTS dtf_local.print_files (
UNIQUE(order_id,item_index) UNIQUE(order_id,item_index)
); );
-- Each run of the off-server database backup (ops/db_backup.py), which the
-- Kanban shows so a backup that stopped working does not go unnoticed.
CREATE TABLE IF NOT EXISTS dtf_local.backups (
id uuid PRIMARY KEY, started_at timestamptz NOT NULL, finished_at timestamptz NOT NULL,
status text NOT NULL CHECK(status IN ('ok','failed')), object_key text, bytes bigint, detail text
);
CREATE INDEX IF NOT EXISTS backups_finished ON dtf_local.backups(finished_at DESC);
CREATE INDEX IF NOT EXISTS uploads_owner ON dtf_local.uploads(owner); CREATE INDEX IF NOT EXISTS uploads_owner ON dtf_local.uploads(owner);
-- Indexes follow the queries the application actually issues. Only these; every -- Indexes follow the queries the application actually issues. Only these; every

View File

@@ -148,6 +148,9 @@ services:
S3_APP_USER: ${S3_APP_USER:-dtf_app} S3_APP_USER: ${S3_APP_USER:-dtf_app}
S3_APP_PASSWORD: ${S3_APP_PASSWORD:-local-app-storage-only} S3_APP_PASSWORD: ${S3_APP_PASSWORD:-local-app-storage-only}
S3_BUCKET: ${S3_BUCKET:-dtf-local-artwork} S3_BUCKET: ${S3_BUCKET:-dtf-local-artwork}
S3_BACKUP_BUCKET: dtf-local-backups
S3_BACKUP_USER: dtf_backup
S3_BACKUP_PASSWORD: local-backup-storage-only
networks: [local] networks: [local]
depends_on: depends_on:
storage: {condition: service_healthy} storage: {condition: service_healthy}
@@ -197,6 +200,28 @@ services:
timeout: 3s timeout: 3s
retries: 12 retries: 12
# The daily database backup, against the local backup bucket. Idle until
# BACKUP_AGE_RECIPIENT is set; tests/backup_test.py runs it with a throwaway key.
backup:
build:
context: .
dockerfile: infra/Dockerfile
command: python -m ops.db_backup serve
environment:
DATABASE_ADMIN_URL: postgresql://${POSTGRES_USER:-dtf_local}:${POSTGRES_PASSWORD:-local-database-only}@db:5432/${POSTGRES_DB:-dtf_local}
BACKUP_S3_ENDPOINT: http://storage:9000
BACKUP_BUCKET: dtf-local-backups
BACKUP_ACCESS_KEY_ID: dtf_backup
BACKUP_SECRET_ACCESS_KEY: local-backup-storage-only
BACKUP_REGION: us-east-1
BACKUP_AGE_RECIPIENT: ${BACKUP_AGE_RECIPIENT:-}
networks: [local]
read_only: true
tmpfs: [/tmp]
depends_on:
db-init: {condition: service_completed_successfully}
storage-init: {condition: service_completed_successfully}
site: site:
build: build:
context: . context: .

View File

@@ -13,8 +13,11 @@ LABEL org.opencontainers.image.title="DTF Portal/API" \
# The base is pinned, so its OS packages are frozen at the digest's build date. # The base is pinned, so its OS packages are frozen at the digest's build date.
# Upgrade them here or the image ships known-fixed Debian vulnerabilities, which # Upgrade them here or the image ships known-fixed Debian vulnerabilities, which
# is what the production image was doing while the local one already did this. # is what the production image was doing while the local one already did this.
# pg_dump and age are for the database backup (ops/db_backup.py). Debian 13
# ships PostgreSQL 17, the server's major version, which pg_dump must match.
RUN apt-get update \ RUN apt-get update \
&& apt-get upgrade -y \ && apt-get upgrade -y \
&& apt-get install -y --no-install-recommends postgresql-client-17 age \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
WORKDIR /app WORKDIR /app

View File

@@ -51,6 +51,13 @@ MAX_PENDING_UPLOADS=10
MAX_UPLOAD_BYTES=5368709120 MAX_UPLOAD_BYTES=5368709120
UPLOAD_PART_BYTES=8388608 UPLOAD_PART_BYTES=8388608
SCAN_MAX_BYTES=134217728 SCAN_MAX_BYTES=134217728
# Daily encrypted database backup to its own bucket (docs/BACKUP.md).
BACKUP_S3_ENDPOINT=https://<account>.r2.cloudflarestorage.com
BACKUP_BUCKET=dtf-backups
BACKUP_ACCESS_KEY_ID=TBD
BACKUP_SECRET_ACCESS_KEY=TBD
BACKUP_AGE_RECIPIENT=age1...
BACKUP_HOUR=3
# Names of external Portainer/Docker Swarm secrets, never their values. # Names of external Portainer/Docker Swarm secrets, never their values.
DATABASE_URL_SECRET=TBD DATABASE_URL_SECRET=TBD

View File

@@ -183,6 +183,34 @@ services:
replicas: 1 replicas: 1
restart_policy: {condition: on-failure, delay: 5s} restart_policy: {condition: on-failure, delay: 5s}
# Daily encrypted copy of the database to a bucket of its own, off this
# server (ops/db_backup.py). Idle until the BACKUP_* settings are set. The
# bucket's lifecycle rule removes old copies; this credential never deletes.
backup:
image: ${API_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-api}:${IMAGE_TAG:-latest}
command: python -m ops.db_backup serve
environment:
DATABASE_ADMIN_HOST: db
DATABASE_ADMIN_NAME: dtf
DATABASE_ADMIN_USER: dtf_admin
DATABASE_ADMIN_PASSWORD: ${POSTGRES_PASSWORD:?set POSTGRES_PASSWORD}
# The R2 account endpoint (the same as R2_ENDPOINT) and a bucket and API
# token for backups only, never the artwork bucket's.
BACKUP_S3_ENDPOINT: ${BACKUP_S3_ENDPOINT:-}
BACKUP_BUCKET: ${BACKUP_BUCKET:-}
BACKUP_ACCESS_KEY_ID: ${BACKUP_ACCESS_KEY_ID:-}
BACKUP_SECRET_ACCESS_KEY: ${BACKUP_SECRET_ACCESS_KEY:-}
# The public key (age1...). Its private key stays off this server.
BACKUP_AGE_RECIPIENT: ${BACKUP_AGE_RECIPIENT:-}
# Hour of day in Brasília.
BACKUP_HOUR: ${BACKUP_HOUR:-3}
networks: [backend, egress]
read_only: true
tmpfs: [/tmp]
deploy:
replicas: 1
restart_policy: {condition: on-failure, delay: 30s}
site: site:
image: ${WEB_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-web}:${IMAGE_TAG:-latest} image: ${WEB_IMAGE:-gitea.blyzer.com.br/blyzer/dtf-web}:${IMAGE_TAG:-latest}
environment: environment:

87
docs/BACKUP.md Normal file
View File

@@ -0,0 +1,87 @@
# Database backup
The `backup` service copies the database off the server once a day
(`ops/db_backup.py`). It runs `pg_dump`, checks the archive, encrypts it with
[age](https://age-encryption.org) and uploads it to an R2 bucket used for
nothing else. Each run is recorded, and the Kanban shows the latest one under
Integrações → "Backup do banco" ("Em dia", "Verificar" or "Não configurado").
The artwork is not backed up. Originals and print files are temporary (7 to 30
days) and already stored on R2. The database holds what cannot be recreated:
orders, customers, quotes, payments, the Tiny connection and the operators.
## Why the backup cannot be read or deleted from the server
- **The server only encrypts.** It holds the public key (`age1…`). The private
key, which is the only way to open a backup, stays with the owner.
- **Its own bucket and token.** The backup token reaches the backup bucket
only, and the artwork token cannot reach it.
- **Nothing is deleted by the job.** The bucket's lifecycle rule removes old
copies. The bucket lock keeps anyone, including someone holding the token,
from deleting or overwriting a copy before its time.
## Setup (once)
1. **Key pair.** Generate it on your own computer, not on the server:
```bash
docker run --rm gitea.blyzer.com.br/blyzer/dtf-api:latest age-keygen
```
Or, with age installed (`sudo pacman -S age`, `apt install age`), just
run `age-keygen`.
Save the whole output (the `AGE-SECRET-KEY-1…` line is the private key) in
the password manager. Only the `public key: age1…` value goes to Portainer.
Without the private key, no backup can ever be restored.
2. **Cloudflare R2 → Create bucket.** Use a name such as `dtf-backups`.
3. **Settings on that bucket:**
- Object lifecycle rules: delete objects after 35 days.
- Bucket lock rules: retain for 30 days, prefix `db/`.
4. **R2 → Manage API tokens → Create API token.**
- Permission: Object Read & Write.
- Scope: the `dtf-backups` bucket only.
5. **Portainer.** Add these to the stack's environment, then redeploy:
| Variable | Value |
|---|---|
| `BACKUP_S3_ENDPOINT` | same as `R2_ENDPOINT` |
| `BACKUP_BUCKET` | `dtf-backups` |
| `BACKUP_ACCESS_KEY_ID` | the new token's Access Key ID |
| `BACKUP_SECRET_ACCESS_KEY` | the new token's Secret Access Key |
| `BACKUP_AGE_RECIPIENT` | the public key, `age1…` |
| `BACKUP_HOUR` | optional, default `3` (03:00 Brasília) |
The first backup runs as soon as the service starts. After that it runs daily,
and a failed run is retried an hour later.
## Restore test
Do this after setup, and then once a month. Open the console of the `backup`
container in Portainer (Containers → `…_backup…` → Console) and run:
```bash
python -m ops.db_backup list
python -m ops.db_backup verify --identity -
```
`--identity -` asks for the private key to be pasted. It is kept only in the
container's memory while the command runs. `verify` restores the latest backup
into a scratch database, prints its row counts and drops it. The live database
is not touched.
## Restoring for real
From the `backup` container's console:
```bash
python -m ops.db_backup restore db/2026/10/dtf-20261001T060000Z.dump.age --into dtf_restored --identity -
```
This restores into a new database (or an empty one), never over the running
one. Then do one of these:
- **Check the restored data**, then point the stack at it.
- **On a new server**, deploy only the `db` service first. Restore into `dtf`
while it is still empty, then deploy the rest. `db-init` then applies the
schema and grants on top.

View File

@@ -864,6 +864,12 @@ print-file evidence still need correction before this item can close.
- `[ ]` 5.13 — Define production recovery: scheduled encrypted offsite database - `[ ]` 5.13 — Define production recovery: scheduled encrypted offsite database
and object backups, a consistent snapshot boundary, Swarm data placement and and object backups, a consistent snapshot boundary, Swarm data placement and
a restore rehearsal that opens every required live order file. a restore rehearsal that opens every required live order file.
**2026-09-29:** database part built. The `backup` service dumps daily,
encrypts with age (the server holds only the public key) and uploads to a
bucket of its own, whose lifecycle rule expires copies and whose lock keeps
them from being deleted early. The Kanban shows the latest run, and
`tests/backup_test.py` restores a copy in CI. Setup and restore:
docs/BACKUP.md. Artwork is not copied: it is temporary and already on R2.
- `[ ]` 5.14 — Promote and verify one immutable release. **2026-09-24:** green - `[ ]` 5.14 — Promote and verify one immutable release. **2026-09-24:** green
pushes to `main` now publish images and Portainer's pull-and-redeploy is the pushes to `main` now publish images and Portainer's pull-and-redeploy is the
release gate; the source preflight is advisory unless enforced by variable, release gate; the source preflight is advisory unless enforced by variable,

View File

@@ -1,8 +1,11 @@
# Same pinned base as deploy/Dockerfile.api, so the integration suite exercises # Same pinned base as deploy/Dockerfile.api, so the integration suite exercises
# the image that ships rather than a different one. # the image that ships rather than a different one.
FROM python:3.12-slim@sha256:2f17fc044b579bab302c2e8054d3a686e2cb9a83de48e70534b94cd8ebbe06a9 FROM python:3.12-slim@sha256:2f17fc044b579bab302c2e8054d3a686e2cb9a83de48e70534b94cd8ebbe06a9
# pg_dump and age are for the database backup (ops/db_backup.py). Debian 13
# ships PostgreSQL 17, the server's major version, which pg_dump must match.
RUN apt-get update \ RUN apt-get update \
&& apt-get upgrade -y \ && apt-get upgrade -y \
&& apt-get install -y --no-install-recommends postgresql-client-17 age \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
WORKDIR /app WORKDIR /app
COPY infra/requirements.txt infra/requirements.lock /app/infra/ COPY infra/requirements.txt infra/requirements.lock /app/infra/

View File

@@ -21,4 +21,13 @@ printf '%s' "$policy" > /tmp/policy.json
mc admin policy create local dtf-artwork /tmp/policy.json >/dev/null mc admin policy create local dtf-artwork /tmp/policy.json >/dev/null
mc admin policy attach local dtf-artwork --user "$S3_APP_USER" >/dev/null mc admin policy attach local dtf-artwork --user "$S3_APP_USER" >/dev/null
mc ilm import "local/$S3_BUCKET" < /lifecycle.json mc ilm import "local/$S3_BUCKET" < /lifecycle.json
# The backup bucket and its own account, which can write and read backups but
# not delete them: the same separation as the backup token on R2.
case "$S3_BACKUP_BUCKET" in *[!a-z0-9.-]*|'') echo 'Invalid local backup bucket name' >&2; exit 1;; esac
mc mb --ignore-existing "local/$S3_BACKUP_BUCKET" >/dev/null
mc anonymous set none "local/$S3_BACKUP_BUCKET" >/dev/null
mc admin user add local "$S3_BACKUP_USER" "$S3_BACKUP_PASSWORD" >/dev/null
printf '%s' '{"Version":"2012-10-17","Statement":[{"Effect":"Allow","Action":["s3:ListBucket","s3:GetBucketLocation"],"Resource":["arn:aws:s3:::'"$S3_BACKUP_BUCKET"'"]},{"Effect":"Allow","Action":["s3:GetObject","s3:PutObject","s3:AbortMultipartUpload","s3:ListMultipartUploadParts"],"Resource":["arn:aws:s3:::'"$S3_BACKUP_BUCKET"'/*"]}]}' > /tmp/backup-policy.json
mc admin policy create local dtf-backup /tmp/backup-policy.json >/dev/null
mc admin policy attach local dtf-backup --user "$S3_BACKUP_USER" >/dev/null
echo 'Local storage runtime account provisioned.' echo 'Local storage runtime account provisioned.'

289
ops/db_backup.py Normal file
View File

@@ -0,0 +1,289 @@
"""Daily off-server backup of the database to its own bucket.
python -m ops.db_backup serve # the stack's backup service
python -m ops.db_backup once # one backup now
python -m ops.db_backup list # what the bucket holds
python -m ops.db_backup verify --identity F # restore the latest into a scratch database, then drop it
python -m ops.db_backup restore KEY --identity F --into NAME
Each backup is `pg_dump --format=custom`, checked with `pg_restore --list`,
encrypted with age to BACKUP_AGE_RECIPIENT (a public key) and uploaded to
BACKUP_BUCKET with credentials of its own. The server holds only the public
key: it can write backups but not read them. The private key (the identity)
stays with the owner, off the server, and is needed only to restore.
Old backups are removed by the bucket's lifecycle rule, not by this job, so
its credential never needs to delete; a bucket lock keeps a stolen one from
deleting either. Every run is recorded in dtf_local.backups, which the Kanban
shows on its Integrations tab.
"""
import argparse
import hashlib
import os
import subprocess
import sys
import tempfile
import time
from datetime import datetime, timedelta, timezone
from pathlib import Path
from uuid import uuid4
import boto3
import psycopg
from botocore.config import Config
from psycopg import sql
from psycopg.conninfo import conninfo_to_dict
from app.bootstrap import admin_connect
from app.core.secrets import load as load_secret_files
BRASILIA = timezone(timedelta(hours=-3))
PREFIX = 'db/'
# After a failure the next attempt comes this much later, not a day later.
RETRY = timedelta(hours=1)
# A backup older than this means the daily run has stopped working.
STALE = timedelta(hours=26)
TABLES = ('orders', 'quotes', 'uploads', 'accounts', 'movements', 'payment_intents')
def configured():
return all(os.environ.get(k) for k in
('BACKUP_S3_ENDPOINT', 'BACKUP_BUCKET', 'BACKUP_ACCESS_KEY_ID',
'BACKUP_SECRET_ACCESS_KEY', 'BACKUP_AGE_RECIPIENT'))
def bucket():
client = boto3.client('s3', endpoint_url=os.environ['BACKUP_S3_ENDPOINT'],
aws_access_key_id=os.environ['BACKUP_ACCESS_KEY_ID'],
aws_secret_access_key=os.environ['BACKUP_SECRET_ACCESS_KEY'],
region_name=os.environ.get('BACKUP_REGION', 'auto'),
config=Config(signature_version='s3v4', s3={'addressing_style': 'path'},
retries={'max_attempts': 5, 'mode': 'standard'}))
return client, os.environ['BACKUP_BUCKET']
def admin_params(database=None):
"""The administrator's connection, optionally to another database."""
if os.environ.get('DATABASE_ADMIN_HOST'):
params = {'host': os.environ['DATABASE_ADMIN_HOST'], 'user': os.environ['DATABASE_ADMIN_USER'],
'password': os.environ['DATABASE_ADMIN_PASSWORD'], 'dbname': os.environ['DATABASE_ADMIN_NAME']}
else:
params = conninfo_to_dict(os.environ['DATABASE_ADMIN_URL'])
if database:
params['dbname'] = database
return params
def libpq_env(database=None):
"""pg_dump and pg_restore read the credentials from the environment, so the
password never appears in a process list."""
names = {'host': 'PGHOST', 'port': 'PGPORT', 'user': 'PGUSER', 'password': 'PGPASSWORD', 'dbname': 'PGDATABASE'}
return {**os.environ, **{names[k]: str(v) for k, v in admin_params(database).items() if k in names}}
def run(command, **kwargs):
result = subprocess.run(command, capture_output=True, **kwargs)
if result.returncode:
detail = result.stderr.decode(errors='replace').strip().splitlines()
raise RuntimeError(f'{command[0]} failed: ' + (detail[-1] if detail else f'exit {result.returncode}'))
return result
def sha256(path):
digest = hashlib.sha256()
with open(path, 'rb') as stream:
for block in iter(lambda: stream.read(1 << 20), b''):
digest.update(block)
return digest.hexdigest()
def record(started, status, key=None, size=None, detail=None):
with admin_connect() as c:
c.execute('''INSERT INTO dtf_local.backups(id,started_at,finished_at,status,object_key,bytes,detail)
VALUES(%s,%s,now(),%s,%s,%s,%s)''',
(uuid4(), started, status, key, size, detail))
def run_once():
"""Dump, check, encrypt and upload one backup. Returns its object key."""
started = datetime.now(timezone.utc)
key = PREFIX + started.strftime('%Y/%m/dtf-%Y%m%dT%H%M%SZ') + '.dump.age'
try:
client, name = bucket()
with tempfile.TemporaryDirectory(dir=os.environ.get('BACKUP_TMP')) as work:
dump, sealed = Path(work, 'dtf.dump'), Path(work, 'dtf.dump.age')
run(['pg_dump', '--format=custom', '--compress=6', '--file', str(dump)], env=libpq_env())
# A truncated or damaged archive fails here, before it is kept.
run(['pg_restore', '--list', str(dump)])
run(['age', '--encrypt', '--recipient', os.environ['BACKUP_AGE_RECIPIENT'],
'--output', str(sealed), str(dump)])
size = sealed.stat().st_size
client.upload_file(str(sealed), name, key, ExtraArgs={'Metadata': {
'sha256': sha256(sealed), 'dump-sha256': sha256(dump), 'format': 'pg-custom+age'}})
record(started, 'ok', key, size)
print(f'Backup {key} uploaded ({size} bytes).', flush=True)
return key
except Exception as error:
# The message names a command or a provider error, never a credential.
detail = str(error)[:500]
try:
record(started, 'failed', key, None, detail)
finally:
print(f'Backup failed: {detail}', file=sys.stderr, flush=True)
raise
def last_ok():
with admin_connect() as c:
row = c.execute("SELECT max(finished_at) FROM dtf_local.backups WHERE status='ok'").fetchone()
return row[0]
def next_run(now, hour):
"""The next `hour` o'clock in Brasília after `now`."""
local = now.astimezone(BRASILIA)
at = local.replace(hour=hour, minute=0, second=0, microsecond=0)
if at <= local:
at += timedelta(days=1)
return at.astimezone(timezone.utc)
def serve():
if not configured():
print('Backup not configured: set BACKUP_S3_ENDPOINT, BACKUP_BUCKET, BACKUP_ACCESS_KEY_ID, '
'BACKUP_SECRET_ACCESS_KEY and BACKUP_AGE_RECIPIENT. Idle.', flush=True)
while True:
time.sleep(3600)
hour = int(os.environ.get('BACKUP_HOUR', '3'))
# Wait for the schema (db-init) before the first query.
for _ in range(60):
try:
previous = last_ok()
break
except psycopg.Error:
time.sleep(10)
else:
previous = last_ok()
now = datetime.now(timezone.utc)
# A fresh deployment, or one that missed a day, backs up straight away.
due = now if previous is None or now - previous > STALE else next_run(now, hour)
while True:
time.sleep(max(0, (due - datetime.now(timezone.utc)).total_seconds()))
try:
run_once()
due = next_run(datetime.now(timezone.utc), hour)
except Exception:
due = datetime.now(timezone.utc) + RETRY
def listing():
client, name = bucket()
keys = []
for page in client.get_paginator('list_objects_v2').paginate(Bucket=name, Prefix=PREFIX):
keys += [(o['Key'], o['Size'], o['LastModified']) for o in page.get('Contents', [])]
return sorted(keys)
def counts(database):
with psycopg.connect(**admin_params(database)) as c:
return {t: c.execute(sql.SQL('SELECT count(*) FROM dtf_local.{}').format(sql.Identifier(t))).fetchone()[0]
for t in TABLES}
def restore(key, identity, into):
"""Restore one backup into a database that is new or holds no DTF data."""
client, name = bucket()
with admin_connect() as c:
c.autocommit = True
if into == c.execute('SELECT current_database()').fetchone()[0]:
raise SystemExit('Refusing to restore over the database the stack is using.')
exists = c.execute('SELECT 1 FROM pg_database WHERE datname=%s', (into,)).fetchone()
if not exists:
c.execute(sql.SQL('CREATE DATABASE {}').format(sql.Identifier(into)))
if exists:
with psycopg.connect(**admin_params(into)) as c:
if c.execute("SELECT 1 FROM pg_namespace WHERE nspname='dtf_local'").fetchone():
raise SystemExit(f'{into} already holds DTF data; choose an empty or new database.')
with tempfile.TemporaryDirectory(dir=os.environ.get('BACKUP_TMP')) as work:
sealed, dump = Path(work, 'dtf.dump.age'), Path(work, 'dtf.dump')
client.download_file(name, key, str(sealed))
expected = client.head_object(Bucket=name, Key=key)['Metadata'].get('sha256')
if expected and sha256(sealed) != expected:
raise SystemExit('The downloaded backup does not match its checksum.')
run(['age', '--decrypt', '--identity', identity, '--output', str(dump), str(sealed)])
run(['pg_restore', '--exit-on-error', '--no-owner', '--no-privileges', '--dbname', into, str(dump)],
env=libpq_env(into))
return counts(into)
def drop(database):
with admin_connect() as c:
c.autocommit = True
c.execute(sql.SQL('DROP DATABASE IF EXISTS {} WITH (FORCE)').format(sql.Identifier(database)))
def verify(identity, key=None):
"""The restore test: the latest backup into a scratch database, counted, dropped."""
key = key or listing()[-1][0]
scratch = 'dtf_verify_' + uuid4().hex[:12]
try:
restored = restore(key, identity, scratch)
finally:
drop(scratch)
print(f'PASS: {key} decrypted and restored into a scratch database (now dropped). Rows: ' +
', '.join(f'{t} {n}' for t, n in restored.items()))
return restored
def identity_file(value):
"""The identity as a path, or '-' to paste it: nothing is written to disk
except a private temporary file removed when the command ends."""
if value != '-':
return value, None
print('Paste the private key (AGE-SECRET-KEY-...) and press Enter:', file=sys.stderr)
line = sys.stdin.readline().strip()
handle = tempfile.NamedTemporaryFile('w', prefix='identity-', dir=os.environ.get('BACKUP_TMP'), delete=False)
os.chmod(handle.name, 0o600)
handle.write(line + '\n')
handle.close()
return handle.name, handle.name
def main(argv=None):
load_secret_files()
parser = argparse.ArgumentParser(prog='python -m ops.db_backup')
sub = parser.add_subparsers(dest='command', required=True)
sub.add_parser('serve')
sub.add_parser('once')
sub.add_parser('list')
for name in ('verify', 'restore'):
p = sub.add_parser(name)
if name == 'restore':
p.add_argument('key')
p.add_argument('--into', required=True, help='a new or empty database')
else:
p.add_argument('key', nargs='?', help='default: the latest backup')
p.add_argument('--identity', required=True, help="the private key file, or '-' to paste it")
args = parser.parse_args(argv)
if args.command == 'serve':
serve()
elif args.command == 'once':
run_once()
elif args.command == 'list':
for key, size, modified in listing():
print(f'{modified:%Y-%m-%d %H:%M} UTC {size:>12} {key}')
else:
path, temporary = identity_file(args.identity)
try:
if args.command == 'verify':
verify(path, args.key)
else:
rows = restore(args.key, path, args.into)
print(f'Restored {args.key} into {args.into}. Rows: ' + ', '.join(f'{t} {n}' for t, n in rows.items()))
finally:
if temporary:
os.unlink(temporary)
if __name__ == '__main__':
main()

79
tests/backup_test.py Normal file
View File

@@ -0,0 +1,79 @@
"""The database backup against the local stack: dump, encrypt, upload, restore.
docker compose -f compose.local.yaml exec -T backup python -m tests.backup_test
Uses a throwaway age key made here, so nothing secret is kept in the repository.
"""
import os
import subprocess
import tempfile
from datetime import datetime, timezone
from pathlib import Path
from botocore.exceptions import ClientError
from app.bootstrap import admin_connect
from ops import db_backup
def check(condition, message):
if not condition:
raise AssertionError(message)
print('PASS:', message)
def main():
with tempfile.TemporaryDirectory() as work:
identity = Path(work, 'identity.txt')
subprocess.run(['age-keygen', '-o', str(identity)], check=True, capture_output=True)
public = next(line.split(': ', 1)[1] for line in identity.read_text().splitlines()
if line.startswith('# public key: '))
os.environ['BACKUP_AGE_RECIPIENT'] = public
check(db_backup.configured(), 'the local backup service is configured once a recipient is set')
live = db_backup.counts(None)
key = db_backup.run_once()
with admin_connect() as c:
row = c.execute('SELECT status,bytes FROM dtf_local.backups WHERE object_key=%s', (key,)).fetchone()
check(row is not None and row[0] == 'ok' and row[1] > 0, 'the run is recorded for the Kanban')
client, bucket = db_backup.bucket()
head = client.get_object(Bucket=bucket, Key=key, Range='bytes=0-20')['Body'].read()
check(head.startswith(b'age-encryption.org/v1'), 'what leaves the server is encrypted')
check(key in [k for k, _, _ in db_backup.listing()], 'the backup is listed in its bucket')
try:
client.delete_object(Bucket=bucket, Key=key)
deleted = True
except ClientError:
deleted = False
check(not deleted and key in [k for k, _, _ in db_backup.listing()],
'the backup credential cannot delete backups')
restored = db_backup.verify(str(identity), key)
check(restored == live, f'the restore has the same rows as the live database ({restored})')
live_name = db_backup.admin_params()['dbname']
try:
db_backup.restore(key, str(identity), live_name)
refused = False
except SystemExit:
refused = True
check(refused, 'a restore over the live database is refused')
other = Path(work, 'other.txt')
subprocess.run(['age-keygen', '-o', str(other)], check=True, capture_output=True)
try:
db_backup.verify(str(other), key)
opened = True
except RuntimeError:
opened = False
check(not opened, 'another key cannot open the backup')
brt = lambda h, m=0: datetime(2026, 9, 29, h + 3, m, tzinfo=timezone.utc)
check(db_backup.next_run(brt(2), 3) == brt(3), 'before 03:00 in Brasília the run is that day')
check(db_backup.next_run(brt(3), 3) == datetime(2026, 9, 30, 6, tzinfo=timezone.utc),
'at or after 03:00 the run is the next day')
if __name__ == '__main__':
main()

View File

@@ -706,6 +706,13 @@ function renderIntegrations(){
' + '+kg(fr.per_metre_kg)+' por metro · prazo da Jadlog + '+fr.production_days+' dia(s) de produção.') ' + '+kg(fr.per_metre_kg)+' por metro · prazo da Jadlog + '+fr.production_days+' dia(s) de produção.')
: integration('Frete','Não configurado','off','Somente retirada.')); : integration('Frete','Não configurado','off','Somente retirada.'));
cards.push(integration('WhatsApp','Não configurado','off','Mensagens registradas, sem envio.')); cards.push(integration('WhatsApp','Não configurado','off','Mensagens registradas, sem envio.'));
const bk=board.providers?.backup||{};
const ontem=bk.last_ok&&Date.now()-new Date(bk.last_ok)<26*3600e3;
const tam=bk.bytes?' · '+(bk.bytes<1048576?Math.max(1,Math.round(bk.bytes/1024))+' KB':(bk.bytes/1048576).toFixed(1).replace('.',',')+' MB'):'';
cards.push(integration('Backup do banco',!bk.last_ok&&!bk.failed?'Não configurado':ontem&&!bk.failed?'Em dia':'Verificar',
!bk.last_ok&&!bk.failed?'off':ontem&&!bk.failed?'ok':'warn',
(bk.last_ok?'Último backup em '+when(bk.last_ok)+tam+'.':'Nenhum backup feito ainda.')+
(bk.failed?' A última tentativa falhou ('+when(bk.failed_at)+'): '+bk.failed:'')+' Cópia diária, criptografada, fora do servidor.'));
$('integrations').replaceChildren(...cards); $('integrations').replaceChildren(...cards);
eventFilters(); eventFilters();
} }