diff --git a/.gitea/workflows/deploy.yml b/.gitea/workflows/deploy.yml index 892b157..038a928 100644 --- a/.gitea/workflows/deploy.yml +++ b/.gitea/workflows/deploy.yml @@ -86,7 +86,7 @@ jobs: # the generator's geometry. The provider suites use a fake transport: they # prove the documented contract, not the integration. - name: Print-file geometry and provider adapters - run: $COMPOSE exec -T api python -m unittest tests.test_printfile tests.test_mercadopago tests.test_tiny tests.test_jadlog tests.test_quote_review tests.test_large_files -v + run: $COMPOSE exec -T api python -m unittest tests.test_printfile tests.test_mercadopago tests.test_tiny tests.test_jadlog tests.test_quote_review tests.test_large_files tests.test_grade_check -v - name: Runtime and retention regressions run: | diff --git a/app/api/quotes.py b/app/api/quotes.py index 79c8c4d..dbfb6fe 100644 --- a/app/api/quotes.py +++ b/app/api/quotes.py @@ -13,7 +13,8 @@ from ..core.auth import owner from ..core.models import QuoteRequest from ..core.pricing import price from .. import quote_review -from ..runtime import freight, quote_view, require_delivery_available, upload_row +from ..grade_check import Files, mismatch +from ..runtime import freight, quote_view, require_delivery_available, storage, upload_row from ..scanning import require_clean router = APIRouter() @@ -36,15 +37,21 @@ def create_quote(body: QuoteRequest, session_id=Depends(owner)): except ValueError as exc: raise HTTPException(422, str(exc)) with db.connect() as c: + rows = {} for item in body.items: for uid in item.uploads: row = upload_row(c, uid, session_id) if not row['complete']: raise HTTPException(409, 'Complete every upload before requesting a quote') require_clean(row) + rows[str(uid)] = row + # The grade sets the price and came from the browser: before a cart is + # approved at checkout, the server works it out from the files. + note = mismatch(Files(storage), draft['items'], rows) if reason is None else None + reason = reason or note uid = uuid4() - c.execute('INSERT INTO dtf_local.quotes(id,owner,request_key,request_hash,draft) VALUES(%s,%s,%s,%s,%s) ON CONFLICT(owner,request_key) DO NOTHING', - (uid, session_id, body.request_key, digest, Jsonb(draft))) + c.execute('INSERT INTO dtf_local.quotes(id,owner,request_key,request_hash,draft,review_note) VALUES(%s,%s,%s,%s,%s,%s) ON CONFLICT(owner,request_key) DO NOTHING', + (uid, session_id, body.request_key, digest, Jsonb(draft), note)) row = c.execute('SELECT * FROM dtf_local.quotes WHERE owner=%s AND request_key=%s FOR UPDATE', (session_id, body.request_key)).fetchone() if row['request_hash'] != digest: raise HTTPException(409, 'Request key already used for a different cart') diff --git a/app/grade_check.py b/app/grade_check.py new file mode 100644 index 0000000..b7a63e3 --- /dev/null +++ b/app/grade_check.py @@ -0,0 +1,189 @@ +"""The grade (the resolution discount) recomputed from the uploaded files. + +The Site grades the art in the customer's browser and sends the grade with the +cart, and the price depends on it. A customer who edits the page could claim a +better grade, so before a cart is approved at checkout the server reads the +files it already has and works the grade out again, by the Site's own rules: + +- a finished sheet: its pixel width over the sheet width is its DPI, and the + item's grade comes from the worst sheet (web/site-quality.js); +- artworks: each one's DPI along the width it is printed at, rotation + included; the item's grade is the average of the artworks' grades; +- a PDF: the area-weighted DPI of the images drawn on its page, 300 when it + is only vectors (web/site-pdf.js, dpiDasImagens and resumoDpi). + +Images are measured from their first bytes, never decoded. A PDF is opened +only up to PDF_MAX_BYTES, the size the Site itself analyses. A grade that +cannot be recomputed is None; the caller decides what that means. +""" +import math +import os +import struct +import tempfile + +DPI_IDEAL = 300 +# The Site analyses PDFs up to 150 MB (GRANDE_BYTES); above that it does not +# grade them, so neither does this. +PDF_MAX_BYTES = 150 * 1048576 +HEAD_BYTES = 1048576 +# Rounding in the browser and here can differ by a point on the same file. +TOLERANCE = 2 + + +def js_round(value): + """Math.round, which rounds halves up; Python's round() does not.""" + return math.floor(value + 0.5) + + +def grade_of(dpi): + return max(6, min(100, js_round(dpi / DPI_IDEAL * 100))) + + +def image_size(head): + """(width, height) in pixels from a PNG, JPEG or WebP's first bytes.""" + if head[:8] == b'\x89PNG\r\n\x1a\n' and head[12:16] == b'IHDR': + return struct.unpack('>II', head[16:24]) + if head[:2] == b'\xff\xd8': + i = 2 + while i + 9 < len(head): + if head[i] != 0xFF: + i += 1 + continue + marker = head[i + 1] + if marker in (0xD8, 0x01) or 0xD0 <= marker <= 0xD7 or marker == 0xFF: + i += 1 if marker == 0xFF else 2 + continue + length = struct.unpack('>H', head[i + 2:i + 4])[0] + if 0xC0 <= marker <= 0xCF and marker not in (0xC4, 0xC8, 0xCC): + h, w = struct.unpack('>HH', head[i + 5:i + 9]) + return w, h + i += 2 + length + return None + if head[:4] == b'RIFF' and head[8:12] == b'WEBP': + chunk = head[12:16] + if chunk == b'VP8X': + return (int.from_bytes(head[24:27], 'little') + 1, int.from_bytes(head[27:30], 'little') + 1) + if chunk == b'VP8L' and head[20] == 0x2F: + bits = int.from_bytes(head[21:25], 'little') + return (bits & 0x3FFF) + 1, ((bits >> 14) & 0x3FFF) + 1 + if chunk == b'VP8 ' and head[23:26] == b'\x9d\x01\x2a': + w, h = struct.unpack('= 1: + found.append((int(xobject.get('/Width', 0)) / (width_pt * unit / 72), + abs(ctm[0] * ctm[3] - ctm[1] * ctm[2]))) + elif subtype == '/Form' and depth < 8: + matrix = [float(v) for v in xobject.get('/Matrix', [1, 0, 0, 1, 0, 0])] + walk(xobject.get('/Resources', resources), + pikepdf.parse_content_stream(xobject), _multiply(ctm, matrix), depth + 1) + + walk(page.obj.get('/Resources'), pikepdf.parse_content_stream(page), [1, 0, 0, 1, 0, 0], 0) + except Exception: + return None + found = [(dpi, area) for dpi, area in found if dpi > 0] + if not found: + return DPI_IDEAL + total = sum(area for _, area in found) or 1 + return js_round(sum(dpi * area for dpi, area in found) / total) + + +class Files: + """The uploaded files, read from storage as little as needed.""" + + def __init__(self, storage): + self.storage = storage + + def head(self, row): + return self.storage.client.get_object(Bucket=self.storage.bucket, Key=row['object_key'], + Range=f'bytes=0-{HEAD_BYTES - 1}')['Body'].read() + + def pdf_dpi(self, row): + if row['size'] > PDF_MAX_BYTES: + return None + with tempfile.NamedTemporaryFile(suffix='.pdf') as handle: + self.storage.client.download_fileobj(self.storage.bucket, row['object_key'], handle) + handle.flush() + return pdf_dpi(handle.name) + + +def source_dpi(files, row, source): + """One source's DPI across the width it is printed at, or None.""" + name = row['name'].lower() + width_in = float(source['width_cm']) / 2.54 + if name.endswith('.pdf'): + return files.pdf_dpi(row) if source['kind'] == 'sheet' else None + if not name.endswith(('.png', '.jpg', '.jpeg', '.webp')): + return None + size = image_size(files.head(row)) + if not size or not all(size): + return None + across = size[1] if source['rotation_degrees'] in (90, 270) else size[0] + return across / width_in + + +def item_grade(files, item, rows): + """The grade the Site gives this item, recomputed; None if it cannot be.""" + sources = item['production']['sources'] + dpis = [] + for source in sources: + row = rows.get(str(source['upload_id'])) + dpi = source_dpi(files, row, source) if row else None + if dpi is None: + return None + dpis.append(dpi) + if all(source['kind'] == 'sheet' for source in sources): + # A sheet's DPI is shown and compared as a whole number. + return grade_of(min(js_round(d) for d in dpis)) + return js_round(sum(grade_of(d) for d in dpis) / len(dpis)) + + +def mismatch(files, items, rows): + """Why a cart's grades cannot be approved at checkout, or None.""" + for item in items: + if item['grade'] == 0: + continue # full price: nothing to verify + verified = item_grade(files, item, rows) + if verified is None: + return 'Nota não conferida no servidor' + if item['grade'] > verified + TOLERANCE: + return f"Nota {item['grade']} maior que a do arquivo ({verified})" + return None diff --git a/app/quote_review.py b/app/quote_review.py index 0d445a2..95f5156 100644 --- a/app/quote_review.py +++ b/app/quote_review.py @@ -6,11 +6,11 @@ moment it was created. The Site is a shop: a customer who can price the order should be able to pay for it straight away, at any hour, so only orders a person must look at before charging wait for the Kanban (review_reason). -Known limit: the grade (and so the discount) and the layout are still worked -out in the customer's browser (roadmap 3.2, 3.9). The API checks the layout's -geometry and that an unanalysed item carries no discount, but a customer who -edits the page can claim a better grade. Operators see every order's grade -and artwork at Arte recebida. +The grade (and so the discount) is worked out in the customer's browser and +checked against the files before an automatic approval (app/grade_check.py); +a grade the files do not support waits for review. The layout is still the +browser's (roadmap 3.9): the API checks its geometry. Operators see every +order's grade and artwork at Arte recebida. """ import os from decimal import Decimal diff --git a/app/runtime.py b/app/runtime.py index 3049fa7..eac94fc 100644 --- a/app/runtime.py +++ b/app/runtime.py @@ -85,7 +85,7 @@ def quote_view(c, row): 'draft': row['draft'], 'approved': row['approved'], 'status': 'paid' if order else 'expired' if expired else 'approved' if row['approved'] else 'pending_review', 'auto_approved': row.get('reviewed_by') == 'auto', - 'review_reason': None if row['approved'] else review_reason(row['draft']), + 'review_reason': None if row['approved'] else (row.get('review_note') or review_reason(row['draft'])), 'payment': attempt, 'order': order} diff --git a/app/schema.sql b/app/schema.sql index 6c1f47f..b6ea0e1 100644 --- a/app/schema.sql +++ b/app/schema.sql @@ -142,6 +142,10 @@ CREATE TABLE IF NOT EXISTS dtf_local.backups ( ); CREATE INDEX IF NOT EXISTS backups_finished ON dtf_local.backups(finished_at DESC); +-- Why a quote waits for review, when the reason came from its files (the grade +-- the server recomputed) and cannot be worked out again from the draft alone. +ALTER TABLE dtf_local.quotes ADD COLUMN IF NOT EXISTS review_note text; + CREATE INDEX IF NOT EXISTS uploads_owner ON dtf_local.uploads(owner); -- Indexes follow the queries the application actually issues. Only these; every diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 48a17a7..ff00694 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -627,6 +627,13 @@ the site already did. generator, with the browser as preview only. This is the single largest gap between what was promised in the meeting and what exists. +**2026-09-30:** the grade (the resolution discount) is no longer taken on trust. +Before an automatic approval the API recomputes it from the uploaded files by the +Site's rules (`app/grade_check.py`): image headers for PNG/JPG/WebP, the placed +images of a PDF up to 150 MB. A claim more than 2 points above the file's grade, +or a discount on a file the server cannot grade, waits for review with the reason +on the Kanban. The metres (the packing) are still the browser's. + ### `[~]` 3.3 — The 5 GB problem is unsolved `(F19)` Transport accepts 5 GiB; `SCAN_MAX_BYTES` / ClamAV `StreamMaxLength` release only diff --git a/tests/smoke_test.py b/tests/smoke_test.py index 84ec872..2f16042 100644 --- a/tests/smoke_test.py +++ b/tests/smoke_test.py @@ -175,7 +175,11 @@ def run(): assert not client.call('/quotes/'+qid)['auto_approved'] # A cart the Site priced is approved at once and can be paid straight away, # at the server's own prices; no operator can then change it. - small={**draft,'request_key':str(uuid4()),'items':[item_spec('avulsa','2.75',90,uid)]} + # The server works the grade out from the file: this PNG is 6059 px across + # 57 cm, 270 DPI, which is grade 90. + png=b'\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR'+(6059).to_bytes(4,'big')+(2000).to_bytes(4,'big')+b'\x08\x06\x00\x00\x00'+b'\x00'*4 + graded=upload_bytes(client,png,name='arte-270dpi.png') + small={**draft,'request_key':str(uuid4()),'items':[item_spec('avulsa','2.75',90,graded)]} auto=client.call('/quotes',small) assert auto['status']=='approved' seen=client.call('/quotes/'+auto['id']) @@ -188,6 +192,12 @@ def run(): held=client.call('/quotes',{**small,'request_key':str(uuid4()),'items':[claimed]}) assert held['status']=='pending_review' assert client.call('/quotes/'+held['id'])['review_reason']=='Nota informada sem análise da arte' + # A better grade than the file supports, or one the server cannot check, waits too. + inflated=client.call('/quotes',{**small,'request_key':str(uuid4()),'items':[item_spec('avulsa','2.75',100,graded)]}) + assert inflated['status']=='pending_review' + assert client.call('/quotes/'+inflated['id'])['review_reason']=='Nota 100 maior que a do arquivo (90)' + unchecked=client.call('/quotes',{**small,'request_key':str(uuid4()),'items':[item_spec('avulsa','2.75',90,uid)]}) + assert client.call('/quotes/'+unchecked['id'])['review_reason']=='Nota não conferida no servidor' print('PASS: priced carts are approved at checkout; large or inconsistent ones wait for review') client.call('/orders/dev-paid',{'quote_id':qid,'total_cents':1},expected=422) other.call('/orders/dev-paid',{'quote_id':qid},expected=404) diff --git a/tests/test_grade_check.py b/tests/test_grade_check.py new file mode 100644 index 0000000..2a61a20 --- /dev/null +++ b/tests/test_grade_check.py @@ -0,0 +1,86 @@ +"""The server's grade against real image and PDF files. + +Runs where Pillow and pikepdf are installed (the API image). +""" +import io +import tempfile +import unittest + +import pikepdf +from PIL import Image + +from app import grade_check + + +class FakeFiles: + def __init__(self, blobs): + self.blobs = blobs + + def head(self, row): + return self.blobs[row['name']][:grade_check.HEAD_BYTES] + + def pdf_dpi(self, row): + with tempfile.NamedTemporaryFile(suffix='.pdf') as handle: + handle.write(self.blobs[row['name']]) + handle.flush() + return grade_check.pdf_dpi(handle.name) + + +def encoded(fmt, size, **options): + out = io.BytesIO() + Image.new('RGBA' if fmt in ('PNG', 'WEBP') else 'RGB', size, (200, 30, 30)).save(out, fmt, **options) + return out.getvalue() + + +def item(kind, grade, sources): + return {'grade': grade, 'production': {'sources': [ + {'upload_id': name, 'kind': kind, 'width_cm': width, 'rotation_degrees': rotation} + for name, width, rotation in sources]}} + + +class GradeCheckTests(unittest.TestCase): + def test_image_sizes_from_the_first_bytes(self): + for fmt, options in (('PNG', {}), ('JPEG', {'quality': 80}), ('JPEG', {'progressive': True}), + ('WEBP', {'lossless': True}), ('WEBP', {'quality': 80})): + self.assertEqual(grade_check.image_size(encoded(fmt, (1234, 567), **options)), (1234, 567), (fmt, options)) + # A JPEG with a large metadata block before the frame header. + exif = Image.Exif() + exif[0x010E] = 'x' * 60000 + self.assertEqual(grade_check.image_size(encoded('JPEG', (800, 600), exif=exif)), (800, 600)) + self.assertIsNone(grade_check.image_size(b'not an image at all')) + + def test_pdf_dpi_is_the_images_area_weighted(self): + out = io.BytesIO() + Image.new('RGB', (1500, 750), 'white').save(out, 'PDF', resolution=150) + with tempfile.NamedTemporaryFile(suffix='.pdf') as handle: + handle.write(out.getvalue()); handle.flush() + self.assertEqual(grade_check.pdf_dpi(handle.name), 150) + vector = pikepdf.new() + vector.add_blank_page(page_size=(1615, 850)) + with tempfile.NamedTemporaryFile(suffix='.pdf') as handle: + vector.save(handle.name) + self.assertEqual(grade_check.pdf_dpi(handle.name), 300) + + def test_grades_follow_the_site(self): + # 57 cm is 22.44 in: 6059 px is 270 DPI, grade 90; 3366 px is 150 DPI, grade 50. + files = FakeFiles({'a.png': encoded('PNG', (6059, 100)), 'b.jpg': encoded('JPEG', (3366, 100)), + 'tall.png': encoded('PNG', (100, 2362)), 'x.cdr': b'CDR'}) + rows = {n: {'name': n, 'size': 1} for n in files.blobs} + # A sheet takes the worst sheet's grade. + self.assertEqual(grade_check.item_grade(files, item('sheet', 0, [('a.png', 57, 0)]), rows), 90) + self.assertEqual(grade_check.item_grade(files, item('sheet', 0, [('a.png', 57, 0), ('b.jpg', 57, 0)]), rows), 50) + # Artworks average; a rotated one is measured along its height. + self.assertEqual(grade_check.item_grade(files, item('artwork', 0, [('a.png', 57, 0), ('b.jpg', 57, 0)]), rows), 70) + self.assertEqual(grade_check.item_grade(files, item('artwork', 0, [('tall.png', 20, 90)]), rows), 100) + self.assertIsNone(grade_check.item_grade(files, item('sheet', 0, [('x.cdr', 57, 0)]), rows)) + # What the cart claims is checked, with a point or two of rounding. + claim = lambda g, src: grade_check.mismatch(files, [item('sheet', g, src)], rows) + self.assertIsNone(claim(90, [('a.png', 57, 0)])) + self.assertIsNone(claim(92, [('a.png', 57, 0)])) + self.assertEqual(claim(100, [('a.png', 57, 0)]), 'Nota 100 maior que a do arquivo (90)') + self.assertEqual(claim(40, [('x.cdr', 57, 0)]), 'Nota não conferida no servidor') + self.assertIsNone(claim(0, [('x.cdr', 57, 0)])) + + +if __name__ == '__main__': + unittest.main()