feat: accept sheets of up to 5 GB end to end
Some checks failed
Build and deploy / Validate source (push) Successful in 12s
Build and deploy / Integration suite on a real stack (push) Failing after 2m36s
Build and deploy / Secret scan and release gate (push) Successful in 7s
Build and deploy / Publish images (push) Has been skipped
Some checks failed
Build and deploy / Validate source (push) Successful in 12s
Build and deploy / Integration suite on a real stack (push) Failing after 2m36s
Build and deploy / Secret scan and release gate (push) Successful in 7s
Build and deploy / Publish images (push) Has been skipped
Sheets of several GB are the normal order. The upload limit is now 5 GB. ClamAV scans files up to 2 GB; a larger file is released only when its first bytes match the format its name claims, and a disguised file is refused. The Site grades a sheet over 150 MB from the pixel size in its PNG, JPEG or WebP header without decoding it, and reads large PDFs in ranges. The worker never opens a source over 300 MB: a finished sheet placed whole becomes its own print file, which the Kanban offers to approve as the final, and anything else goes to hand preparation. Files start uploading as they enter the cart, with progress in the summary, and each part renews the reservation so slow uploads do not expire. Quotas grow to 50 GB per customer and 500 GB in total; the Swarm config for ClamAV is renamed because a deployed config cannot change in place. Verified locally with a 386 MB and a 1.8 GB PNG (scanned, paid, original as print file), a 2.3 GB PNG (format check) and a disguised 2.3 GB file (refused). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -57,8 +57,11 @@ def upload_status(uid: UUID, session_id=Depends(owner)):
|
||||
def part_url(uid: UUID, part: int, session_id=Depends(owner)):
|
||||
with db.connect() as c:
|
||||
row = upload_row(c, uid, session_id)
|
||||
if row['complete'] or not 1 <= part <= math.ceil(row['size'] / PART_BYTES):
|
||||
raise HTTPException(409, 'Invalid part or completed upload')
|
||||
if row['complete'] or not 1 <= part <= math.ceil(row['size'] / PART_BYTES):
|
||||
raise HTTPException(409, 'Invalid part or completed upload')
|
||||
# The reservation lease is an hour; a multi-GB upload on a slow line
|
||||
# takes longer, so each part it asks for keeps it alive.
|
||||
c.execute("UPDATE dtf_local.uploads SET expires_at=GREATEST(expires_at,now()+interval '1 hour') WHERE id=%s", (uid,))
|
||||
size = min(PART_BYTES, row['size']-(part-1)*PART_BYTES)
|
||||
return {'url': storage.part_url(row['object_key'], row['multipart_id'], part, size)}
|
||||
|
||||
|
||||
@@ -9,7 +9,6 @@ from uuid import UUID, uuid4
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
from .printjobs import generated_identity
|
||||
from .runtime import upload_row
|
||||
from .scanning import require_clean
|
||||
|
||||
@@ -30,7 +29,8 @@ def generated_owner(c, order, ref, kind):
|
||||
if c.execute("SELECT 1 FROM dtf_local.order_files WHERE order_id=%s AND kind='correction' LIMIT 1",
|
||||
(order['id'],)).fetchone():
|
||||
raise HTTPException(409, 'A customer correction replaced the artwork this file was generated from')
|
||||
return generated_identity(order['id'])
|
||||
# A large sheet's print file is its original, owned by the customer.
|
||||
return c.execute('SELECT owner FROM dtf_local.uploads WHERE id=%s', (ref.upload_id,)).fetchone()['owner']
|
||||
|
||||
def submit_files(c, order, body, identity, kind, actor):
|
||||
if order['version'] != body.version:
|
||||
@@ -71,5 +71,8 @@ def submit_files(c, order, body, identity, kind, actor):
|
||||
c.execute('UPDATE dtf_local.orders SET version=version+1,updated_at=now() WHERE id=%s', (order['id'],))
|
||||
if kind == 'final':
|
||||
# Artwork approval, not commercial quote approval, starts original cleanup.
|
||||
c.execute("UPDATE dtf_local.uploads SET expires_at=LEAST(expires_at,now()+interval '7 days') WHERE id=ANY(%s)", (original_ids,))
|
||||
# An original approved as its own print file is kept as the final.
|
||||
finals = [ref.upload_id for ref in body.files]
|
||||
c.execute("UPDATE dtf_local.uploads SET expires_at=LEAST(expires_at,now()+interval '7 days') WHERE id=ANY(%s) AND NOT id=ANY(%s)",
|
||||
(original_ids, finals))
|
||||
return {'ok': True, 'version': order['version']+1, 'expires_at': expiry}
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
"""Limits shared by upload admission and the malware scanner."""
|
||||
import os
|
||||
|
||||
CLAMAV_STREAM_MAX_BYTES = 128 * 1024 * 1024 # infra/clamd.conf
|
||||
CLAMAV_STREAM_MAX_BYTES = 2000 * 1024 * 1024 # infra/clamd.conf StreamMaxLength
|
||||
|
||||
|
||||
def scan_limit_bytes():
|
||||
return min(CLAMAV_STREAM_MAX_BYTES, int(os.environ.get('SCAN_MAX_BYTES', '134217728')))
|
||||
"""The largest file ClamAV scans; above it the format check releases it."""
|
||||
return min(CLAMAV_STREAM_MAX_BYTES, int(os.environ.get('SCAN_MAX_BYTES', str(CLAMAV_STREAM_MAX_BYTES))))
|
||||
|
||||
|
||||
def upload_limit_bytes():
|
||||
transport = int(os.environ.get('MAX_UPLOAD_BYTES', '5368709120'))
|
||||
return min(transport, scan_limit_bytes())
|
||||
return int(os.environ.get('MAX_UPLOAD_BYTES', '5368709120'))
|
||||
|
||||
@@ -9,6 +9,12 @@ The result is an ordinary upload row owned by an identity derived from the
|
||||
order, already marked clean: its only inputs are artwork that passed the
|
||||
malware scan, and the bytes are written here. The operator still decides
|
||||
whether it becomes the final file; generation never approves anything.
|
||||
|
||||
Sheets of several GB are the normal order, and decoding one would take more
|
||||
memory than the worker has. A source above LARGE_SOURCE_BYTES is never
|
||||
opened: a finished sheet placed whole on the film is already its own print
|
||||
file, so the original becomes the print file; any other layout is prepared
|
||||
by hand from the original.
|
||||
"""
|
||||
import logging
|
||||
import os
|
||||
@@ -25,6 +31,9 @@ from .printfile import Unsupported, render
|
||||
|
||||
CLAIM_TIMEOUT = timedelta(minutes=15)
|
||||
MAX_ATTEMPTS = 3
|
||||
LARGE_SOURCE_BYTES = int(os.environ.get('PRINT_DECODE_MAX_BYTES', str(300 * 1024 * 1024)))
|
||||
# Formats the operator can import as they are.
|
||||
PRINTABLE_ORIGINAL = ('.png', '.jpg', '.jpeg', '.tif', '.tiff', '.pdf')
|
||||
|
||||
|
||||
def generated_identity(order_id):
|
||||
@@ -60,13 +69,26 @@ def render_one(storage):
|
||||
c.execute('''UPDATE dtf_local.print_files SET status='rendering', claimed_at=now(),
|
||||
attempts=attempts+1 WHERE id=%s''', (job['id'],))
|
||||
item = job['snapshot']['items'][job['item_index']]
|
||||
uploads = c.execute('''SELECT id,name,object_key,scan_state,purged_at,expires_at,
|
||||
uploads = c.execute('''SELECT id,name,size,object_key,scan_state,purged_at,expires_at,
|
||||
(expires_at<=now()) AS expired FROM dtf_local.uploads WHERE id=ANY(%s)''',
|
||||
([UUID(u) for u in item['uploads']],)).fetchall()
|
||||
# Every generated file shares its order's artwork retention deadline.
|
||||
expiry = c.execute('SELECT min(created_at)+interval \'30 days\' AS e FROM dtf_local.uploads WHERE id=ANY(%s)',
|
||||
([UUID(u) for u in item['uploads']],)).fetchone()['e']
|
||||
by_id = {str(row['id']): row for row in uploads}
|
||||
if any(row['size'] > LARGE_SOURCE_BYTES for row in uploads):
|
||||
original = whole_sheet(item, by_id)
|
||||
if original:
|
||||
with connect() as c:
|
||||
c.execute('''UPDATE dtf_local.print_files SET status='ready', upload_id=%s, detail=%s,
|
||||
finished_at=now(), claimed_at=NULL WHERE id=%s''',
|
||||
(original['id'], Jsonb({'source': 'original', 'name': original['name']}), job['id']))
|
||||
audit('print_file_original', order=str(job['order_id']), item=job['item_index'])
|
||||
else:
|
||||
finish(job, 'manual', {'reason': f'arquivo acima de {LARGE_SOURCE_BYTES // 1048576} MB: '
|
||||
'monte a folha a partir do original'})
|
||||
audit('print_file_manual', order=str(job['order_id']), item=job['item_index'])
|
||||
return True
|
||||
try:
|
||||
result = produce(storage, job, item, by_id)
|
||||
except Unsupported as reason:
|
||||
@@ -96,6 +118,25 @@ def render_one(storage):
|
||||
return True
|
||||
|
||||
|
||||
def whole_sheet(item, uploads):
|
||||
"""The original, when the item is one finished sheet placed whole, once,
|
||||
unrotated and unmirrored, across the film: then it is the print file."""
|
||||
spec = item.get('production') or {}
|
||||
sources, placements = spec.get('sources') or [], spec.get('placements') or []
|
||||
if len(item['uploads']) != 1 or len(sources) != 1 or len(placements) != 1:
|
||||
return None
|
||||
source, place = sources[0], placements[0]
|
||||
row = uploads.get(item['uploads'][0])
|
||||
if (not row or row['scan_state'] != 'clean' or row['purged_at'] or row['expired']
|
||||
or source.get('kind') != 'sheet' or int(source.get('copies', 1)) != 1
|
||||
or not row['name'].lower().endswith(PRINTABLE_ORIGINAL)):
|
||||
return None
|
||||
if (float(place['x_cm']) != 0 or float(place['y_cm']) != 0 or int(place['rotation_degrees']) != 0
|
||||
or place['mirrored'] or abs(float(source['width_cm']) - float(spec['film_width_cm'])) > 0.5):
|
||||
return None
|
||||
return row
|
||||
|
||||
|
||||
def produce(storage, job, item, uploads):
|
||||
"""Fetch the item's artwork and render it. Returns (pdf path, name, size, evidence)."""
|
||||
for upload_id in item['uploads']:
|
||||
|
||||
@@ -1,4 +1,13 @@
|
||||
"""Local ClamAV boundary. Unknown/error/over-limit results NEVER release artwork."""
|
||||
"""Releasing artwork: ClamAV up to its size limit, a format check above it.
|
||||
|
||||
Unknown or error results NEVER release artwork. ClamAV scans files up to
|
||||
scan_limit_bytes() (2 GB). Sheets of several GB are the normal order and
|
||||
ClamAV cannot take them, so a larger file is released only if its first bytes
|
||||
are those of the format its name claims (a PNG that really is a PNG, not a
|
||||
program renamed .png). That is the check the client chose for large files; it
|
||||
does not look for malware inside a valid file.
|
||||
"""
|
||||
import re
|
||||
import socket
|
||||
import struct
|
||||
import time
|
||||
@@ -30,10 +39,9 @@ class ClamAV:
|
||||
return self.command(b'VERSION').decode('utf-8','replace')
|
||||
|
||||
def scan(self, stream, size):
|
||||
if size > scan_limit_bytes():
|
||||
return 'rejected', 'File exceeds the malware scan limit'
|
||||
with socket.create_connection(('scanner',3310),timeout=10) as sock:
|
||||
sock.settimeout(150)
|
||||
# A 2 GB file takes minutes to stream and scan.
|
||||
sock.settimeout(900)
|
||||
sock.sendall(b'zINSTREAM\0')
|
||||
sent=0
|
||||
for chunk in stream.iter_chunks(chunk_size=65536):
|
||||
@@ -52,6 +60,37 @@ class ClamAV:
|
||||
if result.endswith(b' FOUND'):return 'rejected','Malware or unsafe scan condition detected'
|
||||
return 'error','Scanner could not verify this file'
|
||||
|
||||
# What each accepted extension must start with. AI files are PDF or PostScript;
|
||||
# CDR and WebP are RIFF containers with their own form type.
|
||||
TIFF = (b'II*\x00', b'MM\x00*', b'II+\x00', b'MM\x00+')
|
||||
SIGNATURES = {
|
||||
'png': (b'\x89PNG\r\n\x1a\n',),
|
||||
'jpg': (b'\xff\xd8\xff',), 'jpeg': (b'\xff\xd8\xff',),
|
||||
'tif': TIFF, 'tiff': TIFF,
|
||||
'pdf': (b'%PDF-',), 'ai': (b'%PDF-', b'%!PS'),
|
||||
'psd': (b'8BPS',), 'psb': (b'8BPS',),
|
||||
}
|
||||
RIFF_FORMS = {'cdr': re.compile(rb'^RIFF....CDR', re.S), 'webp': re.compile(rb'^RIFF....WEBP', re.S)}
|
||||
HEAD_BYTES = 64
|
||||
|
||||
|
||||
def format_matches(name, head):
|
||||
"""Whether a file's first bytes are those of the format its name claims."""
|
||||
ext = name.rsplit('.', 1)[-1].lower() if '.' in name else ''
|
||||
if ext in RIFF_FORMS:
|
||||
return bool(RIFF_FORMS[ext].match(head))
|
||||
return any(head.startswith(sig) for sig in SIGNATURES.get(ext, ()))
|
||||
|
||||
|
||||
def check_large(storage, row):
|
||||
"""Release decision for a file above the antivirus limit."""
|
||||
head = storage.client.get_object(Bucket=storage.bucket, Key=row['object_key'],
|
||||
Range=f'bytes=0-{HEAD_BYTES - 1}')['Body'].read()
|
||||
if format_matches(row['name'], head):
|
||||
return 'clean', 'Acima do limite do antivírus; formato do arquivo conferido'
|
||||
return 'rejected', 'O conteúdo do arquivo não corresponde ao formato do nome'
|
||||
|
||||
|
||||
def scan_one(storage, scanner=None):
|
||||
scanner=scanner or ClamAV()
|
||||
with connect() as c:
|
||||
@@ -60,9 +99,12 @@ def scan_one(storage, scanner=None):
|
||||
ORDER BY created_at FOR UPDATE SKIP LOCKED LIMIT 1''').fetchone()
|
||||
if not row:return False
|
||||
try:
|
||||
stream=storage.client.get_object(Bucket=storage.bucket,Key=row['object_key'])['Body']
|
||||
try:state,reason=scanner.scan(stream,row['size'])
|
||||
finally:stream.close()
|
||||
if row['size'] > scan_limit_bytes():
|
||||
state,reason=check_large(storage,row)
|
||||
else:
|
||||
stream=storage.client.get_object(Bucket=storage.bucket,Key=row['object_key'])['Body']
|
||||
try:state,reason=scanner.scan(stream,row['size'])
|
||||
finally:stream.close()
|
||||
except Exception:
|
||||
state,reason='error','Malware scanner unavailable; file remains blocked'
|
||||
c.execute("""UPDATE dtf_local.uploads SET scan_state=%s,scan_reason=%s,scanned_at=now(),
|
||||
|
||||
Reference in New Issue
Block a user