From 5f2be7ea208a478a2ed971eaa53d865fa502e486 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Cau=C3=AA=20Faleiros?= Date: Tue, 29 Sep 2026 13:18:18 -0300 Subject: [PATCH] feat: accept sheets of up to 5 GB end to end Sheets of several GB are the normal order. The upload limit is now 5 GB. ClamAV scans files up to 2 GB; a larger file is released only when its first bytes match the format its name claims, and a disguised file is refused. The Site grades a sheet over 150 MB from the pixel size in its PNG, JPEG or WebP header without decoding it, and reads large PDFs in ranges. The worker never opens a source over 300 MB: a finished sheet placed whole becomes its own print file, which the Kanban offers to approve as the final, and anything else goes to hand preparation. Files start uploading as they enter the cart, with progress in the summary, and each part renews the reservation so slow uploads do not expire. Quotas grow to 50 GB per customer and 500 GB in total; the Swarm config for ClamAV is renamed because a deployed config cannot change in place. Verified locally with a 386 MB and a 1.8 GB PNG (scanned, paid, original as print file), a 2.3 GB PNG (format check) and a disguised 2.3 GB file (refused). Co-Authored-By: Claude Opus 5.5 --- .gitea/workflows/deploy.yml | 2 +- app/api/uploads.py | 7 +++-- app/artwork.py | 9 ++++-- app/core/limits.py | 8 ++--- app/printjobs.py | 43 +++++++++++++++++++++++++- app/scanning.py | 56 +++++++++++++++++++++++++++++----- compose.local.yaml | 2 +- docker-compose.yml | 15 ++++++--- docs/ROADMAP.md | 19 +++++++++++- infra/clamd.conf | 10 +++--- tests/runtime_security_test.py | 10 ++++-- tests/smoke_test.py | 3 +- tests/test_large_files.py | 42 +++++++++++++++++++++++++ web/checkout.js | 51 ++++++++++++++++++++++++++++--- web/index.html | 3 +- web/kanban.js | 10 +++--- web/site-pdf.js | 53 ++++++++++++++++++++++++++++++-- web/site-quality.js | 9 +++--- web/site-upload.js | 18 ++++++++--- web/site-v2.css | 2 ++ web/upload.js | 11 +++++-- 21 files changed, 329 insertions(+), 54 deletions(-) create mode 100644 tests/test_large_files.py diff --git a/.gitea/workflows/deploy.yml b/.gitea/workflows/deploy.yml index 043ec34..80d80a8 100644 --- a/.gitea/workflows/deploy.yml +++ b/.gitea/workflows/deploy.yml @@ -86,7 +86,7 @@ jobs: # the generator's geometry. The provider suites use a fake transport: they # prove the documented contract, not the integration. - name: Print-file geometry and provider adapters - run: $COMPOSE exec -T api python -m unittest tests.test_printfile tests.test_mercadopago tests.test_tiny tests.test_jadlog tests.test_quote_review -v + run: $COMPOSE exec -T api python -m unittest tests.test_printfile tests.test_mercadopago tests.test_tiny tests.test_jadlog tests.test_quote_review tests.test_large_files -v - name: Runtime and retention regressions run: | diff --git a/app/api/uploads.py b/app/api/uploads.py index cf2ef26..232d3df 100644 --- a/app/api/uploads.py +++ b/app/api/uploads.py @@ -57,8 +57,11 @@ def upload_status(uid: UUID, session_id=Depends(owner)): def part_url(uid: UUID, part: int, session_id=Depends(owner)): with db.connect() as c: row = upload_row(c, uid, session_id) - if row['complete'] or not 1 <= part <= math.ceil(row['size'] / PART_BYTES): - raise HTTPException(409, 'Invalid part or completed upload') + if row['complete'] or not 1 <= part <= math.ceil(row['size'] / PART_BYTES): + raise HTTPException(409, 'Invalid part or completed upload') + # The reservation lease is an hour; a multi-GB upload on a slow line + # takes longer, so each part it asks for keeps it alive. + c.execute("UPDATE dtf_local.uploads SET expires_at=GREATEST(expires_at,now()+interval '1 hour') WHERE id=%s", (uid,)) size = min(PART_BYTES, row['size']-(part-1)*PART_BYTES) return {'url': storage.part_url(row['object_key'], row['multipart_id'], part, size)} diff --git a/app/artwork.py b/app/artwork.py index f7a0ebd..bf2c3ac 100644 --- a/app/artwork.py +++ b/app/artwork.py @@ -9,7 +9,6 @@ from uuid import UUID, uuid4 from fastapi import HTTPException -from .printjobs import generated_identity from .runtime import upload_row from .scanning import require_clean @@ -30,7 +29,8 @@ def generated_owner(c, order, ref, kind): if c.execute("SELECT 1 FROM dtf_local.order_files WHERE order_id=%s AND kind='correction' LIMIT 1", (order['id'],)).fetchone(): raise HTTPException(409, 'A customer correction replaced the artwork this file was generated from') - return generated_identity(order['id']) + # A large sheet's print file is its original, owned by the customer. + return c.execute('SELECT owner FROM dtf_local.uploads WHERE id=%s', (ref.upload_id,)).fetchone()['owner'] def submit_files(c, order, body, identity, kind, actor): if order['version'] != body.version: @@ -71,5 +71,8 @@ def submit_files(c, order, body, identity, kind, actor): c.execute('UPDATE dtf_local.orders SET version=version+1,updated_at=now() WHERE id=%s', (order['id'],)) if kind == 'final': # Artwork approval, not commercial quote approval, starts original cleanup. - c.execute("UPDATE dtf_local.uploads SET expires_at=LEAST(expires_at,now()+interval '7 days') WHERE id=ANY(%s)", (original_ids,)) + # An original approved as its own print file is kept as the final. + finals = [ref.upload_id for ref in body.files] + c.execute("UPDATE dtf_local.uploads SET expires_at=LEAST(expires_at,now()+interval '7 days') WHERE id=ANY(%s) AND NOT id=ANY(%s)", + (original_ids, finals)) return {'ok': True, 'version': order['version']+1, 'expires_at': expiry} diff --git a/app/core/limits.py b/app/core/limits.py index d1a1063..e154c86 100644 --- a/app/core/limits.py +++ b/app/core/limits.py @@ -1,13 +1,13 @@ """Limits shared by upload admission and the malware scanner.""" import os -CLAMAV_STREAM_MAX_BYTES = 128 * 1024 * 1024 # infra/clamd.conf +CLAMAV_STREAM_MAX_BYTES = 2000 * 1024 * 1024 # infra/clamd.conf StreamMaxLength def scan_limit_bytes(): - return min(CLAMAV_STREAM_MAX_BYTES, int(os.environ.get('SCAN_MAX_BYTES', '134217728'))) + """The largest file ClamAV scans; above it the format check releases it.""" + return min(CLAMAV_STREAM_MAX_BYTES, int(os.environ.get('SCAN_MAX_BYTES', str(CLAMAV_STREAM_MAX_BYTES)))) def upload_limit_bytes(): - transport = int(os.environ.get('MAX_UPLOAD_BYTES', '5368709120')) - return min(transport, scan_limit_bytes()) + return int(os.environ.get('MAX_UPLOAD_BYTES', '5368709120')) diff --git a/app/printjobs.py b/app/printjobs.py index 03667d8..09e1fb3 100644 --- a/app/printjobs.py +++ b/app/printjobs.py @@ -9,6 +9,12 @@ The result is an ordinary upload row owned by an identity derived from the order, already marked clean: its only inputs are artwork that passed the malware scan, and the bytes are written here. The operator still decides whether it becomes the final file; generation never approves anything. + +Sheets of several GB are the normal order, and decoding one would take more +memory than the worker has. A source above LARGE_SOURCE_BYTES is never +opened: a finished sheet placed whole on the film is already its own print +file, so the original becomes the print file; any other layout is prepared +by hand from the original. """ import logging import os @@ -25,6 +31,9 @@ from .printfile import Unsupported, render CLAIM_TIMEOUT = timedelta(minutes=15) MAX_ATTEMPTS = 3 +LARGE_SOURCE_BYTES = int(os.environ.get('PRINT_DECODE_MAX_BYTES', str(300 * 1024 * 1024))) +# Formats the operator can import as they are. +PRINTABLE_ORIGINAL = ('.png', '.jpg', '.jpeg', '.tif', '.tiff', '.pdf') def generated_identity(order_id): @@ -60,13 +69,26 @@ def render_one(storage): c.execute('''UPDATE dtf_local.print_files SET status='rendering', claimed_at=now(), attempts=attempts+1 WHERE id=%s''', (job['id'],)) item = job['snapshot']['items'][job['item_index']] - uploads = c.execute('''SELECT id,name,object_key,scan_state,purged_at,expires_at, + uploads = c.execute('''SELECT id,name,size,object_key,scan_state,purged_at,expires_at, (expires_at<=now()) AS expired FROM dtf_local.uploads WHERE id=ANY(%s)''', ([UUID(u) for u in item['uploads']],)).fetchall() # Every generated file shares its order's artwork retention deadline. expiry = c.execute('SELECT min(created_at)+interval \'30 days\' AS e FROM dtf_local.uploads WHERE id=ANY(%s)', ([UUID(u) for u in item['uploads']],)).fetchone()['e'] by_id = {str(row['id']): row for row in uploads} + if any(row['size'] > LARGE_SOURCE_BYTES for row in uploads): + original = whole_sheet(item, by_id) + if original: + with connect() as c: + c.execute('''UPDATE dtf_local.print_files SET status='ready', upload_id=%s, detail=%s, + finished_at=now(), claimed_at=NULL WHERE id=%s''', + (original['id'], Jsonb({'source': 'original', 'name': original['name']}), job['id'])) + audit('print_file_original', order=str(job['order_id']), item=job['item_index']) + else: + finish(job, 'manual', {'reason': f'arquivo acima de {LARGE_SOURCE_BYTES // 1048576} MB: ' + 'monte a folha a partir do original'}) + audit('print_file_manual', order=str(job['order_id']), item=job['item_index']) + return True try: result = produce(storage, job, item, by_id) except Unsupported as reason: @@ -96,6 +118,25 @@ def render_one(storage): return True +def whole_sheet(item, uploads): + """The original, when the item is one finished sheet placed whole, once, + unrotated and unmirrored, across the film: then it is the print file.""" + spec = item.get('production') or {} + sources, placements = spec.get('sources') or [], spec.get('placements') or [] + if len(item['uploads']) != 1 or len(sources) != 1 or len(placements) != 1: + return None + source, place = sources[0], placements[0] + row = uploads.get(item['uploads'][0]) + if (not row or row['scan_state'] != 'clean' or row['purged_at'] or row['expired'] + or source.get('kind') != 'sheet' or int(source.get('copies', 1)) != 1 + or not row['name'].lower().endswith(PRINTABLE_ORIGINAL)): + return None + if (float(place['x_cm']) != 0 or float(place['y_cm']) != 0 or int(place['rotation_degrees']) != 0 + or place['mirrored'] or abs(float(source['width_cm']) - float(spec['film_width_cm'])) > 0.5): + return None + return row + + def produce(storage, job, item, uploads): """Fetch the item's artwork and render it. Returns (pdf path, name, size, evidence).""" for upload_id in item['uploads']: diff --git a/app/scanning.py b/app/scanning.py index 30dfe5c..207c21d 100644 --- a/app/scanning.py +++ b/app/scanning.py @@ -1,4 +1,13 @@ -"""Local ClamAV boundary. Unknown/error/over-limit results NEVER release artwork.""" +"""Releasing artwork: ClamAV up to its size limit, a format check above it. + +Unknown or error results NEVER release artwork. ClamAV scans files up to +scan_limit_bytes() (2 GB). Sheets of several GB are the normal order and +ClamAV cannot take them, so a larger file is released only if its first bytes +are those of the format its name claims (a PNG that really is a PNG, not a +program renamed .png). That is the check the client chose for large files; it +does not look for malware inside a valid file. +""" +import re import socket import struct import time @@ -30,10 +39,9 @@ class ClamAV: return self.command(b'VERSION').decode('utf-8','replace') def scan(self, stream, size): - if size > scan_limit_bytes(): - return 'rejected', 'File exceeds the malware scan limit' with socket.create_connection(('scanner',3310),timeout=10) as sock: - sock.settimeout(150) + # A 2 GB file takes minutes to stream and scan. + sock.settimeout(900) sock.sendall(b'zINSTREAM\0') sent=0 for chunk in stream.iter_chunks(chunk_size=65536): @@ -52,6 +60,37 @@ class ClamAV: if result.endswith(b' FOUND'):return 'rejected','Malware or unsafe scan condition detected' return 'error','Scanner could not verify this file' +# What each accepted extension must start with. AI files are PDF or PostScript; +# CDR and WebP are RIFF containers with their own form type. +TIFF = (b'II*\x00', b'MM\x00*', b'II+\x00', b'MM\x00+') +SIGNATURES = { + 'png': (b'\x89PNG\r\n\x1a\n',), + 'jpg': (b'\xff\xd8\xff',), 'jpeg': (b'\xff\xd8\xff',), + 'tif': TIFF, 'tiff': TIFF, + 'pdf': (b'%PDF-',), 'ai': (b'%PDF-', b'%!PS'), + 'psd': (b'8BPS',), 'psb': (b'8BPS',), +} +RIFF_FORMS = {'cdr': re.compile(rb'^RIFF....CDR', re.S), 'webp': re.compile(rb'^RIFF....WEBP', re.S)} +HEAD_BYTES = 64 + + +def format_matches(name, head): + """Whether a file's first bytes are those of the format its name claims.""" + ext = name.rsplit('.', 1)[-1].lower() if '.' in name else '' + if ext in RIFF_FORMS: + return bool(RIFF_FORMS[ext].match(head)) + return any(head.startswith(sig) for sig in SIGNATURES.get(ext, ())) + + +def check_large(storage, row): + """Release decision for a file above the antivirus limit.""" + head = storage.client.get_object(Bucket=storage.bucket, Key=row['object_key'], + Range=f'bytes=0-{HEAD_BYTES - 1}')['Body'].read() + if format_matches(row['name'], head): + return 'clean', 'Acima do limite do antivírus; formato do arquivo conferido' + return 'rejected', 'O conteúdo do arquivo não corresponde ao formato do nome' + + def scan_one(storage, scanner=None): scanner=scanner or ClamAV() with connect() as c: @@ -60,9 +99,12 @@ def scan_one(storage, scanner=None): ORDER BY created_at FOR UPDATE SKIP LOCKED LIMIT 1''').fetchone() if not row:return False try: - stream=storage.client.get_object(Bucket=storage.bucket,Key=row['object_key'])['Body'] - try:state,reason=scanner.scan(stream,row['size']) - finally:stream.close() + if row['size'] > scan_limit_bytes(): + state,reason=check_large(storage,row) + else: + stream=storage.client.get_object(Bucket=storage.bucket,Key=row['object_key'])['Body'] + try:state,reason=scanner.scan(stream,row['size']) + finally:stream.close() except Exception: state,reason='error','Malware scanner unavailable; file remains blocked' c.execute("""UPDATE dtf_local.uploads SET scan_state=%s,scan_reason=%s,scanned_at=now(), diff --git a/compose.local.yaml b/compose.local.yaml index f8f4fd0..13cb2f4 100644 --- a/compose.local.yaml +++ b/compose.local.yaml @@ -64,7 +64,7 @@ x-app: &app STORAGE_QUOTA_BYTES: ${STORAGE_QUOTA_BYTES:-53687091200} OWNER_UPLOAD_QUOTA_BYTES: ${OWNER_UPLOAD_QUOTA_BYTES:-10737418240} MAX_PENDING_UPLOADS: ${MAX_PENDING_UPLOADS:-10} - SCAN_MAX_BYTES: ${SCAN_MAX_BYTES:-134217728} + SCAN_MAX_BYTES: ${SCAN_MAX_BYTES:-2097152000} networks: [local] init: true security_opt: [no-new-privileges:true] diff --git a/docker-compose.yml b/docker-compose.yml index 404c9e2..6471eba 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -82,10 +82,13 @@ x-app-environment: &app-environment COOKIE_SECURE: "true" MAX_UPLOAD_BYTES: "5368709120" UPLOAD_PART_BYTES: "8388608" - STORAGE_QUOTA_BYTES: "53687091200" - OWNER_UPLOAD_QUOTA_BYTES: "10737418240" + # Sheets of several GB are the normal order, so room for many of them. + STORAGE_QUOTA_BYTES: "${STORAGE_QUOTA_BYTES:-536870912000}" + OWNER_UPLOAD_QUOTA_BYTES: "${OWNER_UPLOAD_QUOTA_BYTES:-53687091200}" MAX_PENDING_UPLOADS: "10" - SCAN_MAX_BYTES: "134217728" + # ClamAV scans up to this; larger files (up to MAX_UPLOAD_BYTES) are released + # after a file-format check instead (app/scanning.py). + SCAN_MAX_BYTES: "2097152000" services: db: @@ -131,7 +134,7 @@ services: image: clamav/clamav@sha256:9cb27d7660bdf66e9878c832cb433dd8aa152cfbe16f3c2c0084c80b04ae22b4 entrypoint: [clamd, --foreground=true, --config-file=/etc/clamav/clamd.conf] configs: - - source: clamd_config + - source: clamd_config_2gb target: /etc/clamav/clamd.conf mode: 0444 networks: [backend] @@ -234,7 +237,9 @@ services: restart_policy: {condition: on-failure, delay: 5s} configs: - clamd_config: + # Renamed whenever infra/clamd.conf changes: Swarm cannot update a deployed + # config in place, and a redeploy with new content under the old name fails. + clamd_config_2gb: file: ./infra/clamd.conf volumes: diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 417ec0b..451a50d 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -627,7 +627,7 @@ the site already did. generator, with the browser as preview only. This is the single largest gap between what was promised in the meeting and what exists. -### `[?]` 3.3 — The 5 GB problem is unsolved `(F19)` +### `[~]` 3.3 — The 5 GB problem is unsolved `(F19)` Transport accepts 5 GiB; `SCAN_MAX_BYTES` / ClamAV `StreamMaxLength` release only ≤ 128 MiB. As of 2026-09-23, customer selection and API reservation reject files @@ -638,6 +638,23 @@ This is exactly the risk Jorge raised in the meeting. **Decide:** raise the scan ceiling with a resource/timeout design, or define an explicit large-file path (staged scan, sampled scan, operator override with audit). +**Built 2026-09-29 (sheets of several GB are the normal order, not the +exception):** files up to 5 GB. ClamAV scans up to 2 GB (`StreamMaxLength +2000M`); above that a file is released only when its first bytes match the +format its name claims (option A, the user's choice). The Site grades a sheet +over 150 MB from the pixel size in the PNG/JPEG/WebP header without decoding +it, and measures large PDFs through ranged reads; the worker never opens a +source over 300 MB: a single finished sheet placed whole becomes its own print +file, anything else goes to hand preparation. Uploads start as items enter the +cart and the lease renews with each part. Verified on the local stack: 386 MB +(ClamAV 78 s), 1.8 GB (ClamAV 6 min 18 s, scanner under 430 MB of memory, the +original as print file), 2.3 GB PNG released by the format check and a +disguised 2.3 GB file refused, and the header grade of a 200 MB file in 0.5 s. +**Open:** large PDFs get no automatic grade (the DPI of images inside is not +read without rendering); the scanner takes one file at a time, so several +multi-GB uploads queue; pieces, residue and background of a large sheet are +not checked automatically. + ### `[ ]` 3.4 — Upload throughput `(F20)` 8 MiB parts, strictly sequential in `local/static/upload.js:21`, one presign diff --git a/infra/clamd.conf b/infra/clamd.conf index 55c0d00..6b34c33 100644 --- a/infra/clamd.conf +++ b/infra/clamd.conf @@ -5,10 +5,12 @@ TCPSocket 3310 TCPAddr 0.0.0.0 MaxThreads 2 MaxQueue 8 -StreamMaxLength 128M -MaxFileSize 128M -MaxScanSize 256M -MaxScanTime 120000 +StreamMaxLength 2000M +MaxFileSize 2000M +MaxScanSize 4000M +MaxScanTime 900000 +ReadTimeout 900 +CommandReadTimeout 900 AlertExceedsMax yes AlertEncrypted yes ScanPDF yes diff --git a/tests/runtime_security_test.py b/tests/runtime_security_test.py index 21bb929..553e387 100644 --- a/tests/runtime_security_test.py +++ b/tests/runtime_security_test.py @@ -5,7 +5,7 @@ from botocore.exceptions import ClientError from app.core.db import connect from app.adapters import LocalS3Storage from app.core.auth import client_ip, password_hash, password_matches -from app.scanning import ClamAV, require_clean +from app.scanning import ClamAV, format_matches, require_clean from fastapi import HTTPException class FakeRequest: @@ -78,7 +78,13 @@ def run(): except HTTPException as error:assert error.status_code==409 require_clean({'complete':True,'scan_state':'clean'}) assert ClamAV().ping() and ClamAV().version().startswith('ClamAV ') - assert ClamAV().scan(None,134217729)[0]=='rejected' + # Above the antivirus limit only a file whose bytes match its name is released. + for name,head,ok in (('folha.png',b'\x89PNG\r\n\x1a\n\x00',True),('folha.png',b'MZ\x90\x00',False), + ('folha.jpg',b'\xff\xd8\xff\xe0',True),('folha.pdf',b'%PDF-1.7',True), + ('folha.pdf',b'#!/bin/sh',False),('folha.tif',b'II*\x00',True),('folha.psd',b'8BPS',True), + ('folha.ai',b'%!PS-Adobe',True),('folha.cdr',b'RIFF\x10\x00\x00\x00CDRv',True), + ('folha.cdr',b'RIFF\x10\x00\x00\x00WEBP',False),('folha.exe',b'MZ',False)): + assert format_matches(name,head)==ok,(name,head) with patch('app.scanning.socket.create_connection',side_effect=OSError('offline')): try:ClamAV().scan(None,1);raise AssertionError('Offline scanner returned success') except OSError:pass diff --git a/tests/smoke_test.py b/tests/smoke_test.py index 2374328..84ec872 100644 --- a/tests/smoke_test.py +++ b/tests/smoke_test.py @@ -104,7 +104,8 @@ def run(): client=Client();other=Client() config=client.call('/session');other.call('/session') assert client.call('/health')['integrations']=='fake' - assert 0 < config['max_upload_bytes'] <= 128 * 1024 * 1024 + # Sheets of several GB are the normal order: 5 GB per file. + assert config['max_upload_bytes'] == 5 * 1024 ** 3 client.call('/uploads',{'name':'too-large.cdr', 'size':config['max_upload_bytes']+1},expected=413) cancelled=client.call('/uploads',{'name':'CANCELLED-PART.cdr','size':3})['id'] diff --git a/tests/test_large_files.py b/tests/test_large_files.py new file mode 100644 index 0000000..58f463a --- /dev/null +++ b/tests/test_large_files.py @@ -0,0 +1,42 @@ +"""Large sheets: when the original is its own print file, and the format check.""" +import unittest + +from app.printjobs import whole_sheet +from app.scanning import format_matches + +ROW = {'id': 'u1', 'name': 'folha.png', 'size': 3 * 1024 ** 3, 'scan_state': 'clean', + 'purged_at': None, 'expired': False} + + +def sheet(**changes): + source = {'kind': 'sheet', 'width_cm': 57, 'length_cm': 500, 'copies': 1} + place = {'x_cm': 0, 'y_cm': 0, 'rotation_degrees': 0, 'mirrored': False} + for key, value in changes.items(): + (source if key in source else place)[key] = value + return {'uploads': ['u1'], 'production': {'film_width_cm': 57, 'sources': [source], 'placements': [place]}} + + +class WholeSheetTest(unittest.TestCase): + def test_a_finished_sheet_placed_whole_is_its_own_print_file(self): + self.assertIs(whole_sheet(sheet(), {'u1': ROW}), ROW) + + def test_anything_else_is_prepared_by_hand(self): + for changes in ({'copies': 2}, {'kind': 'artwork'}, {'rotation_degrees': 90}, {'mirrored': True}, + {'x_cm': 1}, {'width_cm': 50}): + with self.subTest(changes=changes): + self.assertIsNone(whole_sheet(sheet(**changes), {'u1': ROW})) + self.assertIsNone(whole_sheet(sheet(), {'u1': {**ROW, 'name': 'folha.cdr'}})) + self.assertIsNone(whole_sheet(sheet(), {'u1': {**ROW, 'scan_state': 'pending'}})) + self.assertIsNone(whole_sheet(sheet(), {'u1': {**ROW, 'expired': True}})) + + +class FormatTest(unittest.TestCase): + def test_bytes_must_match_the_name(self): + self.assertTrue(format_matches('A.PNG', b'\x89PNG\r\n\x1a\n')) + self.assertTrue(format_matches('a.tiff', b'MM\x00*')) + self.assertFalse(format_matches('a.png', b'%PDF-1.4')) + self.assertFalse(format_matches('semextensao', b'\x89PNG\r\n\x1a\n')) + + +if __name__ == '__main__': + unittest.main() diff --git a/web/checkout.js b/web/checkout.js index b4d7804..abcdbd6 100644 --- a/web/checkout.js +++ b/web/checkout.js @@ -45,8 +45,7 @@ ready.then(session=>{ window.dtfUploadMaxBytes=session.max_upload_bytes; const limit=document.getElementById('zLimite'); - if(limit)limit.textContent='Até '+(session.max_upload_bytes/1048576).toFixed(0)+ - ' MB por arquivo enquanto a verificação de segurança para arquivos grandes é preparada.'; + if(limit)limit.textContent='Até '+(session.max_upload_bytes/1073741824).toFixed(0)+' GB por arquivo.'; }).catch(()=>{}); window.dtfSessionReady=ready; window.dtfApi=api; @@ -93,9 +92,52 @@ message('Nenhum pedido aguardando pagamento.'); button('Ir para o carrinho', () => vaiPara(CARRINHO)); } + // Files start uploading as soon as they are in the cart, so a sheet of + // several GB is on its way while the customer fills in the order. The + // checkout waits for whatever is still going. + const envios=new Map(); + const chaveArquivo=f=>[f.name,f.size,f.lastModified].join('|'); + function enviar(file) { + const k=chaveArquivo(file); + let e=envios.get(k); + if (!e) { + e={file, sent:0, done:false, failed:null}; + e.promise=(async()=>{ + const session=await ready; + return window.dtfUpload(file,{api,scope:session.cart_scope,progress:()=>{}, + onBytes:n=>{e.sent=n;pintaEnvio();}}); + })(); + e.promise.then(()=>{e.done=true;pintaEnvio();}, + error=>{e.failed=error;envios.delete(k);pintaEnvio();}); + envios.set(k,e); + } + return e.promise; + } + const arquivosDoCarrinho=()=>[...pedido,...(busy&&itemAtual?[itemAtual]:[])].flatMap(it=>it.localFiles||[]); + const gb=n=>(n/1073741824).toLocaleString('pt-BR',{maximumFractionDigits:1})+' GB'; + const mb=n=>n>=1073741824 ? gb(n) : Math.round(n/1048576)+' MB'; + function textoEnvio() { + const files=arquivosDoCarrinho(); if(!files.length) return ''; + let total=0, sent=0, pendentes=0; + for (const f of files) { + const e=envios.get(chaveArquivo(f)); + total+=f.size; sent+=e ? Math.min(e.sent,f.size) : 0; + if (!e || !e.done) pendentes++; + } + if (!pendentes) return 'Arquivos enviados.'; + if (sent>=total) return 'Arquivos enviados · verificando a segurança…'; + return 'Enviando seus arquivos: '+Math.floor(sent/total*100)+'% ('+mb(sent)+' de '+mb(total)+')'; + } + function pintaEnvio() { + const el=document.getElementById('envioArq'); + const texto=textoEnvio(); + if (el) el.textContent=texto; + if (busy && texto) message(texto); + } + // Only what is in the cart: the item on the product page may still change. + window.addEventListener('dtf-cart-changed',()=>{ arquivosDoCarrinho().forEach(f=>{ enviar(f).catch(()=>{}); }); pintaEnvio(); }); async function upload(file) { - const session=await ready; - return window.dtfUpload(file,{api,progress:message,scope:session.cart_scope}); + return enviar(file); } // The cart's package: billed metres and value. The charged freight is // quoted again by the server from the approved items. @@ -148,6 +190,7 @@ await ready; if((await api('/session')).cart_scope !== (await ready).cart_scope) throw new Error('Sua conta ou sessão mudou. Recarregue a página antes de enviar o carrinho.'); const items=[]; + pintaEnvio(); for (const item of cart) { if (!item.localFiles?.length) throw new Error('Selecione novamente os arquivos deste item.'); const uploads=[]; diff --git a/web/index.html b/web/index.html index 1d20403..a4f9c6d 100644 --- a/web/index.html +++ b/web/index.html @@ -969,7 +969,7 @@ footer a:hover{color:var(--laranja2)}
↑
Arraste aqui - Até 128 MB por arquivo enquanto a verificação de segurança para arquivos grandes é preparada. + Até 5 GB por arquivo. ou escolher no computador
@@ -1203,6 +1203,7 @@ footer a:hover{color:var(--laranja2)} +

diff --git a/web/kanban.js b/web/kanban.js index ea2af95..a5f4448 100644 --- a/web/kanban.js +++ b/web/kanban.js @@ -324,14 +324,16 @@ function itemCard(order,item,index){ const file=node('div',undefined,'file'); if(row){ const tone=row.status==='ready'?'ok':row.status==='manual'||row.status==='failed'?'warn':'muted'; - const label=node('span',PRINT_TEXT[row.status]+(row.status==='ready'&&row.name?' · '+row.name:'')+(row.status==='manual'&&row.detail.reason?': '+row.detail.reason:''),tone); + const original=row.status==='ready'&&row.detail?.source==='original'; + const label=node('span',(original?'Arquivo grande: o original é o arquivo de impressão':PRINT_TEXT[row.status])+ + (row.status==='ready'&&row.name?' · '+row.name:'')+(row.status==='manual'&&row.detail.reason?': '+row.detail.reason:''),tone); file.append(tone==='ok'?icon(ICON_OK):tone==='warn'?icon(ICON_WARN):'',label); - if(row.status==='ready'&&row.detail){body.append(file,node('div',[row.detail.film_width_cm+' × '+row.detail.height_cm+' cm', + if(row.status==='ready'&&row.detail&&!original){body.append(file,node('div',[row.detail.film_width_cm+' × '+row.detail.height_cm+' cm', row.detail.min_dpi?'menor resolução '+row.detail.min_dpi+' DPI':'',row.detail.vector_sources?'PDF vetorial':''].filter(Boolean).join(' · '),'line'));} else body.append(file); }else body.append(node('div','Arquivo de impressão ainda não gerado','line')); const buttons=node('div',undefined,'buttons'); - if(row?.status==='ready')buttons.append(button('Baixar PDF',download(row.upload_id))); + if(row?.status==='ready'&&row.detail?.source!=='original')buttons.append(button('Baixar PDF',download(row.upload_id))); item.uploads.forEach((uid,i)=>buttons.append(button(item.uploads.length>1?'Original '+(i+1):'Baixar original',download(uid),'btn ghost'))); if(spec?.placements)buttons.append(button('Manifesto',()=>manifest(spec),'btn ghost')); if(['rec','tra'].includes(order.state)&&(!row||row.status==='manual'||row.status==='failed')) @@ -360,7 +362,7 @@ function finalsSection(order,revisions,section){ if(generated){ const use=node('input');use.type='checkbox';use.checked=true;use.dataset.useGenerated=index;input.required=false;input.hidden=true; use.onchange=()=>{input.hidden=use.checked;input.required=!use.checked;}; - const choice=node('label',undefined,'check');choice.append(use,'Usar o PDF gerado ('+generated.name+')'); + const choice=node('label',undefined,'check');choice.append(use,(generated.detail?.source==='original'?'Usar o original como arquivo final (':'Usar o PDF gerado (')+generated.name+')'); slot.append(choice);input.generated=()=>use.checked?generated.upload_id:null; } slot.append(input);form.append(slot);return input; diff --git a/web/site-pdf.js b/web/site-pdf.js index ca9b8cb..97d3576 100644 --- a/web/site-pdf.js +++ b/web/site-pdf.js @@ -139,9 +139,55 @@ async function rasterizarPdf(file, larguraCm, alturaCm){ return result||{erro:'não deu para conferir',semWorker:temWorker===false}; } +// Sheets of several GB are the normal order. The browser cannot decode an image +// that size (it would freeze or crash the tab), and it does not need to: the +// grade comes from the width in pixels, which PNG, JPEG and WebP store in their +// first bytes. Above GRANDE_BYTES only those bytes are read. +const GRANDE_BYTES=150*1048576; +async function dimensoesImagem(file){ + const b=new Uint8Array(await file.slice(0,Math.min(file.size,4*1048576)).arrayBuffer()); + const u16=(i,le)=>le? b[i]|(b[i+1]<<8) : (b[i]<<8)|b[i+1]; + const u32=(i)=>((b[i]<<24)>>>0)+(b[i+1]<<16)+(b[i+2]<<8)+b[i+3]; + if(b[0]===0x89 && b[1]===0x50 && b[2]===0x4E && b[3]===0x47) // PNG · IHDR + return {w:u32(16), h:u32(20)}; + if(b[0]===0xFF && b[1]===0xD8){ // JPEG · SOFn + let i=2; + while(i+9=0xC0 && m<=0xCF && ![0xC4,0xC8,0xCC].includes(m)) return {w:u16(i+7), h:u16(i+5)}; + i+=2+len; + } + return null; + } + const tag=String.fromCharCode(...b.slice(8,16)); + if(String.fromCharCode(...b.slice(0,4))==='RIFF' && tag.startsWith('WEBP')){ // WebP + if(tag==='WEBPVP8X') return {w:1+(b[24]|(b[25]<<8)|(b[26]<<16)), h:1+(b[27]|(b[28]<<8)|(b[29]<<16))}; + if(tag==='WEBPVP8L'){ const n=b[21]|(b[22]<<8)|(b[23]<<16)|(b[24]<<24); return {w:(n&0x3FFF)+1, h:((n>>14)&0x3FFF)+1}; } + if(tag==='WEBPVP8 ') return {w:u16(26,true)&0x3FFF, h:u16(28,true)&0x3FFF}; + } + return null; +} +// Large PDFs are read in ranges, never loaded whole into memory. +async function fontePdf(lib, file){ + if(file.size<=GRANDE_BYTES) return {data:await file.arrayBuffer()}; + const inicio=new Uint8Array(await file.slice(0,262144).arrayBuffer()); + const t=new lib.PDFDataRangeTransport(file.size, inicio); + t.requestDataRange=(de,ate)=>{ file.slice(de,ate).arrayBuffer().then(buf=>t.onDataRange(de,new Uint8Array(buf))); }; + return {range:t, rangeChunkSize:262144, disableAutoFetch:true, disableStream:true}; +} function medirFolha(file){ return new Promise(res=>{ const nome=(file.name||'').toLowerCase(); + if(RENDERIZA.test(nome) && file.size>GRANDE_BYTES){ + dimensoesImagem(file).then(d=>{ + if(!d || !(d.w>0) || !(d.h>0)) return res(null); + const L=larguraFilme(); + res({larg:L, alt:+(d.h/d.w*L).toFixed(1), fonte:'dimensões do arquivo', + dpiFolha:Math.round(d.w/(L/2.54)), px:d.w, grande:true}); + }).catch(()=>res(null)); + return; + } if(RENDERIZA.test(nome)){ carregarImagem(file).then(img=>{ if(!img || !img.width) return res(null); @@ -160,7 +206,7 @@ function medirFolha(file){ let doc=null; carregarPdfJs().then(async lib=>{ if(!lib) throw new Error('Não foi possível carregar o leitor de PDF.'); - doc=await lib.getDocument({data:await file.arrayBuffer(), + doc=await lib.getDocument({...await fontePdf(lib,file), disableFontFace:true, isEvalSupported:false, useSystemFonts:false, verbosity:0}).promise; if(doc.numPages!==1) @@ -207,7 +253,10 @@ function pintaFolha(){ let medida=''; if(lido){ const a=x.an; - const conferido = a + const conferido = a && a.grande + ? '
Arquivo grande · '+a.dpi+' DPI'+ + 'nota pela resolução · peças, resíduos e fundo conferidos pela equipe
' + : a ? '
Conferido de verdade'+ ''+a.artes+' peça'+(a.artes===1?'':'s')+' · '+a.aproveitamento+ '% da folha virou arte · '+ diff --git a/web/site-quality.js b/web/site-quality.js index f7fcaac..6bb31bb 100644 --- a/web/site-quality.js +++ b/web/site-quality.js @@ -45,15 +45,16 @@ function avaliar(){ ? 'média por área das imagens dentro do PDF · pior em '+ Math.min(...ans.map(a=>a.res? a.res.pior : a.dpi))+' DPI' : 'resolução do arquivo · arte ampliada antes de exportar não aparece aqui'], - ['ok', soma('artes')+' peças na folha', + ans.some(a=>a.grande) ? ['ok','Arquivo grande','peças, resíduos e fundo conferidos pela equipe'] : null, + ans.some(a=>a.grande) ? null : ['ok', soma('artes')+' peças na folha', Math.round(soma('aproveitamento')/ans.length)+'% da folha virou arte'], - soma('residuos') ? ['er','Resíduo de recorte', + ans.some(a=>a.grande) ? null : soma('residuos') ? ['er','Resíduo de recorte', soma('residuos')+' ponto'+(soma('residuos')>1?'s':'')+' abaixo de 2 mm · a impressora imprime'] : ['ok','Sem resíduo de recorte','nada solto na folha'], - soma('encostadas') ? ['fix', soma('encostadas')+' peças a menos de 5 mm', + ans.some(a=>a.grande) ? null : soma('encostadas') ? ['fix', soma('encostadas')+' peças a menos de 5 mm', 'se forem artes diferentes, separamos com 5 mm sem custo'] : ['ok','Espaço entre peças','nenhuma abaixo de 6 mm'], - ans.some(a=>a.fundoChapado) ? ['er','Fundo chapado','a folha está sem transparência'] + ans.some(a=>a.grande) ? null : ans.some(a=>a.fundoChapado) ? ['er','Fundo chapado','a folha está sem transparência'] : ['ok','Fundo transparente','canal alfa conferido'], null ].filter(Boolean); diff --git a/web/site-upload.js b/web/site-upload.js index 5fb4631..b783afd 100644 --- a/web/site-upload.js +++ b/web/site-upload.js @@ -37,14 +37,13 @@ function sel(fs){ const fora = fs.filter(f=>!regra.test(f.name)); fs = fs.filter(f=>regra.test(f.name)); recusa(fora); - const max=window.dtfUploadMaxBytes||128*1048576; + const max=window.dtfUploadMaxBytes||5*1073741824; const grandes=fs.filter(f=>f.size>max); if(grandes.length){ const el=$('recusa'); el.style.display='block'; - el.innerHTML='Arquivo acima do limite de '+(max/1048576).toFixed(0)+ - ' MB. A verificação de segurança ainda não consegue liberar arquivos maiores. '+ - 'Divida ou compacte a arte antes de enviar.'; + el.innerHTML='Arquivo acima do limite de '+(max/1073741824).toFixed(0)+ + ' GB. Divida a folha em partes menores antes de enviar.'; fs=fs.filter(f=>f.size<=max); } if(!fs.length) return; @@ -75,12 +74,21 @@ function sel(fs){ } if(md && md.dpiFolha!=null && md.dpiFolha{ if(img){ x.previewSrc=img.src; try{ x.an=analisarFolha(img, md.larg); if(x.an) x.an.fonteDpi='arquivo'; } catch(e){} } x.pct=null; pintaFolha(); }); + }else if(md && /\.pdf$/i.test(x.f.name) && x.f.size>GRANDE_BYTES){ + x.semAnalise='PDF grande: a resolução das imagens de dentro é conferida pela equipe'; + x.pct=null; pintaFolha(); }else if(md && /\.pdf$/i.test(x.f.name)){ x.pct=70; pintaFolha(); rasterizarPdf(x.f, md.larg, md.alt).then(r=>{ diff --git a/web/site-v2.css b/web/site-v2.css index 9d15982..f91eb01 100644 --- a/web/site-v2.css +++ b/web/site-v2.css @@ -228,6 +228,8 @@ section+section{border-top:0} /* The payment pages: paying on the left, the order summary on the right */ html:is([data-rota="pagamento"],[data-rota="pix"]) .checkout{max-width:1040px;margin-top:8px} html:is([data-rota="pagamento"],[data-rota="pix"]) .pagGrid{display:grid;grid-template-columns:minmax(0,1fr) 340px;gap:32px;align-items:start} +.envioArq{font-size:13px;color:var(--texto2);margin-top:12px;line-height:1.5} +.envioArq:empty{display:none} .dicaEnd{font-size:13px;color:var(--texto2);margin-top:12px} .dicaEnd:empty{display:none} .pagCab{display:flex;align-items:baseline;justify-content:space-between;gap:16px;margin-bottom:20px} diff --git a/web/upload.js b/web/upload.js index a3db838..fd434b8 100644 --- a/web/upload.js +++ b/web/upload.js @@ -1,5 +1,5 @@ /* Shared direct multipart transport for customer originals/corrections and operator finals. */ -window.dtfUpload = async (file, {api, prefix='/uploads', startPath=prefix, progress=()=>{}, scope='guest', resume=true}) => { +window.dtfUpload = async (file, {api, prefix='/uploads', startPath=prefix, progress=()=>{}, onBytes=()=>{}, scope='guest', resume=true}) => { const samples=new Blob([file.slice(0,65536),file.slice(Math.max(0,file.size-65536))]); const hash=Array.from(new Uint8Array(await crypto.subtle.digest('SHA-256',await samples.arrayBuffer())),b=>b.toString(16).padStart(2,'0')).join(''); const key='dtf-upload:'+JSON.stringify([scope,file.name,file.size,file.lastModified,hash]); @@ -7,7 +7,11 @@ window.dtfUpload = async (file, {api, prefix='/uploads', startPath=prefix, progr if(id){try{state=await api(prefix+'/'+id);}catch(error){if(![404,410].includes(error.status))throw error;id=null;}} if(!id){state=await api(startPath,{name:file.name,size:file.size});id=state.id;if(resume)localStorage.setItem(key,id);} async function waitForScan(){ - for(let attempt=0;attempt<150;attempt++){ + onBytes(file.size); + // The antivirus streams the whole file: about a second per 2 MB, and at + // least two and a half minutes. + const attempts=Math.max(150,Math.ceil(file.size/2e6)); + for(let attempt=0;attempt