feat: accept sheets of up to 5 GB end to end
Some checks failed
Build and deploy / Validate source (push) Successful in 12s
Build and deploy / Integration suite on a real stack (push) Failing after 2m36s
Build and deploy / Secret scan and release gate (push) Successful in 7s
Build and deploy / Publish images (push) Has been skipped

Sheets of several GB are the normal order. The upload limit is now 5 GB.
ClamAV scans files up to 2 GB; a larger file is released only when its
first bytes match the format its name claims, and a disguised file is
refused. The Site grades a sheet over 150 MB from the pixel size in its
PNG, JPEG or WebP header without decoding it, and reads large PDFs in
ranges. The worker never opens a source over 300 MB: a finished sheet
placed whole becomes its own print file, which the Kanban offers to approve
as the final, and anything else goes to hand preparation. Files start
uploading as they enter the cart, with progress in the summary, and each
part renews the reservation so slow uploads do not expire. Quotas grow to
50 GB per customer and 500 GB in total; the Swarm config for ClamAV is
renamed because a deployed config cannot change in place.

Verified locally with a 386 MB and a 1.8 GB PNG (scanned, paid, original
as print file), a 2.3 GB PNG (format check) and a disguised 2.3 GB file
(refused).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Cauê Faleiros
2026-09-29 13:18:18 -03:00
parent 4437232d27
commit 5f2be7ea20
21 changed files with 329 additions and 54 deletions

View File

@@ -57,8 +57,11 @@ def upload_status(uid: UUID, session_id=Depends(owner)):
def part_url(uid: UUID, part: int, session_id=Depends(owner)):
with db.connect() as c:
row = upload_row(c, uid, session_id)
if row['complete'] or not 1 <= part <= math.ceil(row['size'] / PART_BYTES):
raise HTTPException(409, 'Invalid part or completed upload')
if row['complete'] or not 1 <= part <= math.ceil(row['size'] / PART_BYTES):
raise HTTPException(409, 'Invalid part or completed upload')
# The reservation lease is an hour; a multi-GB upload on a slow line
# takes longer, so each part it asks for keeps it alive.
c.execute("UPDATE dtf_local.uploads SET expires_at=GREATEST(expires_at,now()+interval '1 hour') WHERE id=%s", (uid,))
size = min(PART_BYTES, row['size']-(part-1)*PART_BYTES)
return {'url': storage.part_url(row['object_key'], row['multipart_id'], part, size)}

View File

@@ -9,7 +9,6 @@ from uuid import UUID, uuid4
from fastapi import HTTPException
from .printjobs import generated_identity
from .runtime import upload_row
from .scanning import require_clean
@@ -30,7 +29,8 @@ def generated_owner(c, order, ref, kind):
if c.execute("SELECT 1 FROM dtf_local.order_files WHERE order_id=%s AND kind='correction' LIMIT 1",
(order['id'],)).fetchone():
raise HTTPException(409, 'A customer correction replaced the artwork this file was generated from')
return generated_identity(order['id'])
# A large sheet's print file is its original, owned by the customer.
return c.execute('SELECT owner FROM dtf_local.uploads WHERE id=%s', (ref.upload_id,)).fetchone()['owner']
def submit_files(c, order, body, identity, kind, actor):
if order['version'] != body.version:
@@ -71,5 +71,8 @@ def submit_files(c, order, body, identity, kind, actor):
c.execute('UPDATE dtf_local.orders SET version=version+1,updated_at=now() WHERE id=%s', (order['id'],))
if kind == 'final':
# Artwork approval, not commercial quote approval, starts original cleanup.
c.execute("UPDATE dtf_local.uploads SET expires_at=LEAST(expires_at,now()+interval '7 days') WHERE id=ANY(%s)", (original_ids,))
# An original approved as its own print file is kept as the final.
finals = [ref.upload_id for ref in body.files]
c.execute("UPDATE dtf_local.uploads SET expires_at=LEAST(expires_at,now()+interval '7 days') WHERE id=ANY(%s) AND NOT id=ANY(%s)",
(original_ids, finals))
return {'ok': True, 'version': order['version']+1, 'expires_at': expiry}

View File

@@ -1,13 +1,13 @@
"""Limits shared by upload admission and the malware scanner."""
import os
CLAMAV_STREAM_MAX_BYTES = 128 * 1024 * 1024 # infra/clamd.conf
CLAMAV_STREAM_MAX_BYTES = 2000 * 1024 * 1024 # infra/clamd.conf StreamMaxLength
def scan_limit_bytes():
return min(CLAMAV_STREAM_MAX_BYTES, int(os.environ.get('SCAN_MAX_BYTES', '134217728')))
"""The largest file ClamAV scans; above it the format check releases it."""
return min(CLAMAV_STREAM_MAX_BYTES, int(os.environ.get('SCAN_MAX_BYTES', str(CLAMAV_STREAM_MAX_BYTES))))
def upload_limit_bytes():
transport = int(os.environ.get('MAX_UPLOAD_BYTES', '5368709120'))
return min(transport, scan_limit_bytes())
return int(os.environ.get('MAX_UPLOAD_BYTES', '5368709120'))

View File

@@ -9,6 +9,12 @@ The result is an ordinary upload row owned by an identity derived from the
order, already marked clean: its only inputs are artwork that passed the
malware scan, and the bytes are written here. The operator still decides
whether it becomes the final file; generation never approves anything.
Sheets of several GB are the normal order, and decoding one would take more
memory than the worker has. A source above LARGE_SOURCE_BYTES is never
opened: a finished sheet placed whole on the film is already its own print
file, so the original becomes the print file; any other layout is prepared
by hand from the original.
"""
import logging
import os
@@ -25,6 +31,9 @@ from .printfile import Unsupported, render
CLAIM_TIMEOUT = timedelta(minutes=15)
MAX_ATTEMPTS = 3
LARGE_SOURCE_BYTES = int(os.environ.get('PRINT_DECODE_MAX_BYTES', str(300 * 1024 * 1024)))
# Formats the operator can import as they are.
PRINTABLE_ORIGINAL = ('.png', '.jpg', '.jpeg', '.tif', '.tiff', '.pdf')
def generated_identity(order_id):
@@ -60,13 +69,26 @@ def render_one(storage):
c.execute('''UPDATE dtf_local.print_files SET status='rendering', claimed_at=now(),
attempts=attempts+1 WHERE id=%s''', (job['id'],))
item = job['snapshot']['items'][job['item_index']]
uploads = c.execute('''SELECT id,name,object_key,scan_state,purged_at,expires_at,
uploads = c.execute('''SELECT id,name,size,object_key,scan_state,purged_at,expires_at,
(expires_at<=now()) AS expired FROM dtf_local.uploads WHERE id=ANY(%s)''',
([UUID(u) for u in item['uploads']],)).fetchall()
# Every generated file shares its order's artwork retention deadline.
expiry = c.execute('SELECT min(created_at)+interval \'30 days\' AS e FROM dtf_local.uploads WHERE id=ANY(%s)',
([UUID(u) for u in item['uploads']],)).fetchone()['e']
by_id = {str(row['id']): row for row in uploads}
if any(row['size'] > LARGE_SOURCE_BYTES for row in uploads):
original = whole_sheet(item, by_id)
if original:
with connect() as c:
c.execute('''UPDATE dtf_local.print_files SET status='ready', upload_id=%s, detail=%s,
finished_at=now(), claimed_at=NULL WHERE id=%s''',
(original['id'], Jsonb({'source': 'original', 'name': original['name']}), job['id']))
audit('print_file_original', order=str(job['order_id']), item=job['item_index'])
else:
finish(job, 'manual', {'reason': f'arquivo acima de {LARGE_SOURCE_BYTES // 1048576} MB: '
'monte a folha a partir do original'})
audit('print_file_manual', order=str(job['order_id']), item=job['item_index'])
return True
try:
result = produce(storage, job, item, by_id)
except Unsupported as reason:
@@ -96,6 +118,25 @@ def render_one(storage):
return True
def whole_sheet(item, uploads):
"""The original, when the item is one finished sheet placed whole, once,
unrotated and unmirrored, across the film: then it is the print file."""
spec = item.get('production') or {}
sources, placements = spec.get('sources') or [], spec.get('placements') or []
if len(item['uploads']) != 1 or len(sources) != 1 or len(placements) != 1:
return None
source, place = sources[0], placements[0]
row = uploads.get(item['uploads'][0])
if (not row or row['scan_state'] != 'clean' or row['purged_at'] or row['expired']
or source.get('kind') != 'sheet' or int(source.get('copies', 1)) != 1
or not row['name'].lower().endswith(PRINTABLE_ORIGINAL)):
return None
if (float(place['x_cm']) != 0 or float(place['y_cm']) != 0 or int(place['rotation_degrees']) != 0
or place['mirrored'] or abs(float(source['width_cm']) - float(spec['film_width_cm'])) > 0.5):
return None
return row
def produce(storage, job, item, uploads):
"""Fetch the item's artwork and render it. Returns (pdf path, name, size, evidence)."""
for upload_id in item['uploads']:

View File

@@ -1,4 +1,13 @@
"""Local ClamAV boundary. Unknown/error/over-limit results NEVER release artwork."""
"""Releasing artwork: ClamAV up to its size limit, a format check above it.
Unknown or error results NEVER release artwork. ClamAV scans files up to
scan_limit_bytes() (2 GB). Sheets of several GB are the normal order and
ClamAV cannot take them, so a larger file is released only if its first bytes
are those of the format its name claims (a PNG that really is a PNG, not a
program renamed .png). That is the check the client chose for large files; it
does not look for malware inside a valid file.
"""
import re
import socket
import struct
import time
@@ -30,10 +39,9 @@ class ClamAV:
return self.command(b'VERSION').decode('utf-8','replace')
def scan(self, stream, size):
if size > scan_limit_bytes():
return 'rejected', 'File exceeds the malware scan limit'
with socket.create_connection(('scanner',3310),timeout=10) as sock:
sock.settimeout(150)
# A 2 GB file takes minutes to stream and scan.
sock.settimeout(900)
sock.sendall(b'zINSTREAM\0')
sent=0
for chunk in stream.iter_chunks(chunk_size=65536):
@@ -52,6 +60,37 @@ class ClamAV:
if result.endswith(b' FOUND'):return 'rejected','Malware or unsafe scan condition detected'
return 'error','Scanner could not verify this file'
# What each accepted extension must start with. AI files are PDF or PostScript;
# CDR and WebP are RIFF containers with their own form type.
TIFF = (b'II*\x00', b'MM\x00*', b'II+\x00', b'MM\x00+')
SIGNATURES = {
'png': (b'\x89PNG\r\n\x1a\n',),
'jpg': (b'\xff\xd8\xff',), 'jpeg': (b'\xff\xd8\xff',),
'tif': TIFF, 'tiff': TIFF,
'pdf': (b'%PDF-',), 'ai': (b'%PDF-', b'%!PS'),
'psd': (b'8BPS',), 'psb': (b'8BPS',),
}
RIFF_FORMS = {'cdr': re.compile(rb'^RIFF....CDR', re.S), 'webp': re.compile(rb'^RIFF....WEBP', re.S)}
HEAD_BYTES = 64
def format_matches(name, head):
"""Whether a file's first bytes are those of the format its name claims."""
ext = name.rsplit('.', 1)[-1].lower() if '.' in name else ''
if ext in RIFF_FORMS:
return bool(RIFF_FORMS[ext].match(head))
return any(head.startswith(sig) for sig in SIGNATURES.get(ext, ()))
def check_large(storage, row):
"""Release decision for a file above the antivirus limit."""
head = storage.client.get_object(Bucket=storage.bucket, Key=row['object_key'],
Range=f'bytes=0-{HEAD_BYTES - 1}')['Body'].read()
if format_matches(row['name'], head):
return 'clean', 'Acima do limite do antivírus; formato do arquivo conferido'
return 'rejected', 'O conteúdo do arquivo não corresponde ao formato do nome'
def scan_one(storage, scanner=None):
scanner=scanner or ClamAV()
with connect() as c:
@@ -60,9 +99,12 @@ def scan_one(storage, scanner=None):
ORDER BY created_at FOR UPDATE SKIP LOCKED LIMIT 1''').fetchone()
if not row:return False
try:
stream=storage.client.get_object(Bucket=storage.bucket,Key=row['object_key'])['Body']
try:state,reason=scanner.scan(stream,row['size'])
finally:stream.close()
if row['size'] > scan_limit_bytes():
state,reason=check_large(storage,row)
else:
stream=storage.client.get_object(Bucket=storage.bucket,Key=row['object_key'])['Body']
try:state,reason=scanner.scan(stream,row['size'])
finally:stream.close()
except Exception:
state,reason='error','Malware scanner unavailable; file remains blocked'
c.execute("""UPDATE dtf_local.uploads SET scan_state=%s,scan_reason=%s,scanned_at=now(),