"""The print file: the reviewed layout, reproduced exactly, at the film's size. The customer is quoted on a layout the Site computes: each copy of each artwork at a width, rotation, mirror and position on the film. Production spec v2 keeps that layout through the approved order. This module turns it into one PDF per order item, with a page exactly as wide as the film and as long as the layout, so the operator imports what the customer approved instead of rebuilding it. Each source image is embedded once, at its original resolution, and every copy is a placement of it. Nothing is resampled: a 300 DPI artwork is still 300 DPI in the file, a JPEG keeps its original bytes, and transparency survives as a soft mask. The output is therefore about the size of the artwork, not of a 57 cm x 20 m raster, and it never needs that raster in memory. Raster sources are JPEG, PNG, WebP and TIFF. A single-page PDF is placed as a vector form, never rasterised. Anything else (PSD, AI, CDR, multi-page or protected PDFs), or a file whose proportions do not match the size it was quoted at, is refused with a reason, and the operator prepares that item by hand exactly as before. """ import math import os import tempfile import zlib from decimal import Decimal from PIL import Image, ImageOps PT_PER_CM = 72 / 2.54 # Acrobat's page limit. Longer layouts scale user space with /UserUnit (PDF 1.6) # rather than cutting the film into pages a RIP might print with gaps. MAX_PAGE_PT = 14400 # How far a file's proportions may drift from the quoted size before the item # is refused: rounding in the Site keeps real files well inside this. ASPECT_TOLERANCE = 0.01 # The quote may round up (10 cm steps, 1 m minimum) but never down. HEIGHT_TOLERANCE_CM = Decimal('0.05') SUPPORTED_FORMATS = {'JPEG', 'PNG', 'WEBP', 'TIFF'} STRIP_ROWS = 256 # Decoding holds the whole image in memory (about 4 bytes a pixel). 57 cm x 3 m # at 300 DPI is 239 Mpx; above this the item goes to the operator instead. MAX_DECODED_PIXELS = int(os.environ.get('PRINT_MAX_PIXELS', '250000000')) # Pillow's own bomb guard would refuse a genuine long sheet before we can # decide; the explicit limit above is the one that applies. Image.MAX_IMAGE_PIXELS = None class Unsupported(Exception): """This item cannot be generated automatically; the reason is for the operator.""" class Name(str): pass class Ref(int): pass def serialize(value): if isinstance(value, Ref): return b'%d 0 R' % value if isinstance(value, Name): return b'/' + value.encode('ascii') if isinstance(value, bool): return b'true' if value else b'false' if isinstance(value, int): return b'%d' % value if isinstance(value, (float, Decimal)): text = f'{float(value):.4f}'.rstrip('0').rstrip('.') return (text if text not in ('', '-0') else '0').encode('ascii') if isinstance(value, dict): return b'<<' + b''.join(b'/' + k.encode('ascii') + b' ' + serialize(v) for k, v in value.items()) + b'>>' if isinstance(value, (list, tuple)): return b'[' + b' '.join(serialize(v) for v in value) + b']' if isinstance(value, str): escaped = value.replace('\\', '\\\\').replace('(', '\\(').replace(')', '\\)') return b'(' + escaped.encode('latin-1', 'replace') + b')' raise TypeError(f'Cannot serialize {type(value).__name__}') class PdfWriter: """A sequential PDF writer: objects go straight to the file, streams included. Stream lengths are indirect objects written after the data, so an image is compressed strip by strip into the output without being held in memory. """ def __init__(self, fp): self.fp = fp self.offsets = {} self.next_id = 1 self.write(b'%PDF-1.6\n%\xe2\xe3\xcf\xd3\n') def write(self, data): self.fp.write(data) def tell(self): return self.fp.tell() def alloc(self): ref = Ref(self.next_id) self.next_id += 1 return ref def obj(self, value, ref=None): ref = ref or self.alloc() self.offsets[ref] = self.tell() self.write(b'%d 0 obj\n' % ref + serialize(value) + b'\nendobj\n') return ref def stream(self, dictionary, chunks, ref=None): """Write a stream from an iterable of already-encoded byte chunks.""" ref = ref or self.alloc() length = self.alloc() self.offsets[ref] = self.tell() self.write(b'%d 0 obj\n' % ref + serialize({**dictionary, 'Length': length}) + b'\nstream\n') start = self.tell() for chunk in chunks: self.write(chunk) size = self.tell() - start self.write(b'\nendstream\nendobj\n') self.obj(size, length) return ref def finish(self, root, info): xref = self.tell() count = self.next_id self.write(b'xref\n0 %d\n0000000000 65535 f \n' % count) for ref in range(1, count): self.write(b'%010d 00000 n \n' % self.offsets[ref]) self.write(b'trailer\n' + serialize({'Size': count, 'Root': root, 'Info': info}) + b'\nstartxref\n%d\n%%%%EOF\n' % xref) def deflate(pieces): compressor = zlib.compressobj(6) for piece in pieces: out = compressor.compress(piece) if out: yield out yield compressor.flush() class SourceImage: """One customer file, opened for reading pixels and measured as the browser sees it.""" def __init__(self, path, name): self.path = path self.name = name try: self.image = Image.open(path) self.format = self.image.format except Image.DecompressionBombError as exc: raise Unsupported(f'"{name}" is too large to generate automatically') from exc except Exception as exc: raise Unsupported(f'"{name}" is not an image this generator can read') from exc if self.format not in SUPPORTED_FORMATS: raise Unsupported(f'"{name}" is {self.format or "an unknown format"}; ' 'only JPEG, PNG, WebP and TIFF are generated automatically') if getattr(self.image, 'n_frames', 1) > 1 and self.format != 'TIFF': raise Unsupported(f'"{name}" is animated or has several frames') # Browsers draw a photo upright according to its EXIF orientation, and # the Site measured it that way, so the print must too. try: self.orientation = self.image.getexif().get(0x0112, 1) except Exception: self.orientation = 1 width, height = self.image.size self.size = (height, width) if self.orientation in (5, 6, 7, 8) else (width, height) def passthrough(self): """Whether the original JPEG bytes can go into the PDF unchanged.""" return (self.format == 'JPEG' and self.orientation == 1 and self.image.mode in ('L', 'RGB', 'CMYK')) def embed(self, pdf): """Write this image (and its alpha) as XObjects; returns the image reference.""" if self.passthrough(): return self._embed_jpeg(pdf) width, height = self.image.size if width * height > MAX_DECODED_PIXELS: raise Unsupported(f'"{self.name}" has {width} x {height} px, more than the ' 'generator decodes; prepare this item by hand') return self._embed_pixels(pdf) def _colorspace(self, pdf, mode): components = {'L': 1, 'RGB': 3, 'CMYK': 4}[mode] device = Name({'L': 'DeviceGray', 'RGB': 'DeviceRGB', 'CMYK': 'DeviceCMYK'}[mode]) profile = self.image.info.get('icc_profile') if not profile: return device icc = pdf.stream({'N': components, 'Alternate': device, 'Filter': Name('FlateDecode')}, deflate([profile])) return [Name('ICCBased'), icc] def _embed_jpeg(self, pdf): image = self.image extra = {} if image.mode == 'CMYK' and 'adobe' in image.info: # Adobe writes CMYK JPEGs inverted; PDF readers expect the Decode flip. extra['Decode'] = [1, 0, 1, 0, 1, 0, 1, 0] def chunks(): with open(self.path, 'rb') as source: while block := source.read(1 << 20): yield block return pdf.stream({'Type': Name('XObject'), 'Subtype': Name('Image'), 'Width': image.width, 'Height': image.height, 'ColorSpace': self._colorspace(pdf, image.mode), 'BitsPerComponent': 8, 'Filter': Name('DCTDecode'), **extra}, chunks()) def _upright(self): image = self.image image.load() if self.orientation != 1: image = ImageOps.exif_transpose(image) return image def _embed_pixels(self, pdf): image = self._upright() mode = image.mode has_alpha = mode in ('RGBA', 'LA', 'PA', 'RGBa', 'La') or ( mode == 'P' and 'transparency' in image.info) or ( mode in ('L', 'RGB') and 'transparency' in image.info) if mode in ('I;16', 'I;16B', 'I;16L', 'I'): image = image.convert('I').point(lambda value: value * (1 / 257)).convert('L') mode = 'L' if mode == 'CMYK': color_mode = 'CMYK' elif mode in ('1', 'L', 'LA', 'La'): color_mode = 'L' elif mode in ('P', 'PA', 'RGB', 'RGBA', 'RGBa'): color_mode = 'RGB' else: raise Unsupported(f'"{self.name}" uses the {mode} colour mode, which is not generated automatically') if has_alpha: image = image.convert('RGBA' if color_mode == 'RGB' else 'LA') width, height = image.size def strips(convert): for top in range(0, height, STRIP_ROWS): yield convert(image.crop((0, top, width, min(height, top + STRIP_ROWS)))).tobytes() smask = None if has_alpha: smask = pdf.stream({'Type': Name('XObject'), 'Subtype': Name('Image'), 'Width': width, 'Height': height, 'ColorSpace': Name('DeviceGray'), 'BitsPerComponent': 8, 'Filter': Name('FlateDecode')}, deflate(strips(lambda strip: strip.getchannel('A')))) colorspace = self._colorspace(pdf, color_mode) dictionary = {'Type': Name('XObject'), 'Subtype': Name('Image'), 'Width': width, 'Height': height, 'ColorSpace': colorspace, 'BitsPerComponent': 8, 'Filter': Name('FlateDecode')} if smask: dictionary['SMask'] = smask return pdf.stream(dictionary, deflate(strips(lambda strip: strip.convert(color_mode)))) def close(self): self.image.close() def placement_matrix(placement, page_height_pt): """Map the image's unit square onto its box on the film. Matches the Site's canvas: the artwork is mirrored first, then turned clockwise about the centre of its box, and the turned image fills the box. """ x = float(placement['x_cm']) * PT_PER_CM top = page_height_pt - float(placement['y_cm']) * PT_PER_CM w = float(placement['width_cm']) * PT_PER_CM h = float(placement['length_cm']) * PT_PER_CM rotation = placement['rotation_degrees'] % 360 mirrored = placement['mirrored'] def to_page(u, v): # Unit square (v up) -> image frame (s right, t down). s, t = u, 1 - v if mirrored: s = 1 - s s, t = {0: (s, t), 90: (1 - t, s), 180: (1 - s, 1 - t), 270: (t, 1 - s)}[rotation] return x + s * w, top - t * h ox, oy = to_page(0, 0) ax, ay = to_page(1, 0) cx, cy = to_page(0, 1) return [ax - ox, ay - oy, cx - ox, cy - oy, ox, oy] class PdfPage: """One page of a customer PDF, placed as a vector form: nothing is rasterised. The box and orientation are the ones the Site measured with pdf.js: the CropBox (which pikepdf uses for the form's BBox), turned by the page's /Rotate, inherited or not. """ def __init__(self, path, name): import pikepdf self.name = name try: self.pdf = pikepdf.open(path) except pikepdf.PasswordError as exc: raise Unsupported(f'"{name}" is password-protected') from exc except Exception as exc: raise Unsupported(f'"{name}" is not a PDF this generator can read') from exc try: pages = len(self.pdf.pages) if pages != 1: raise Unsupported(f'"{name}" has {pages} pages; only single-page PDFs are generated automatically') self.page = self.pdf.pages[0] self.rotation = int(self.page.rotation) % 360 if self.rotation not in (0, 90, 180, 270): raise Unsupported(f'"{name}" has an unsupported page rotation') box = [float(v) for v in self.page.cropbox] except Unsupported: self.pdf.close() raise except Exception as exc: self.pdf.close() raise Unsupported(f'"{name}" has a page this generator cannot read') from exc self.x0, self.y0 = min(box[0], box[2]), min(box[1], box[3]) self.w, self.h = abs(box[2] - box[0]), abs(box[3] - box[1]) if self.w <= 0 or self.h <= 0: self.pdf.close() raise Unsupported(f'"{name}" has an empty page') # As displayed, which is what the proportions are checked against. self.size = (self.h, self.w) if self.rotation in (90, 270) else (self.w, self.h) def normalise(self): """Matrix from the page's box, as displayed, onto the unit square.""" a, d = 1 / self.w, 1 / self.h scale = [a, 0, 0, d, -self.x0 * a, -self.y0 * d] turn = {0: [1, 0, 0, 1, 0, 0], 90: [0, -1, 1, 0, 0, 1], 180: [-1, 0, 0, -1, 1, 1], 270: [0, 1, -1, 0, 1, 0]}[self.rotation] return multiply(scale, turn) def form(self, target): """The page as a form XObject copied into the target document.""" form = self.page.as_form_xobject(handle_transformations=False) group = self.page.obj.get('/Group') if group is not None and '/Group' not in form: form.Group = group return target.copy_foreign(form) def close(self): self.pdf.close() def multiply(first, then): """PDF matrices: a point transformed by `first`, then by `then`.""" a1, b1, c1, d1, e1, f1 = first a2, b2, c2, d2, e2, f2 = then return [a1 * a2 + b1 * c2, a1 * b2 + b1 * d2, c1 * a2 + d1 * c2, c1 * b2 + d1 * d2, e1 * a2 + f1 * c2 + e2, e1 * b2 + f1 * d2 + f2] def open_source(path, name): with open(path, 'rb') as handle: head = handle.read(1024) if b'%PDF-' in head: return PdfPage(path, name) return SourceImage(path, name) def check_layout(item, sizes, vector=()): """Refuse a layout the file cannot reproduce faithfully. Returns the lowest DPI of the raster sources (vector pages have none).""" production = item['production'] height = Decimal(str(production['height_cm'])) billed = Decimal(str(item['billed_metres'])) * 100 if height > billed + HEIGHT_TOLERANCE_CM: raise Unsupported(f'layout is {height} cm long but only {billed} cm were billed') lowest = None for placement in production['placements']: width_px, height_px = sizes[placement['source_index']] if placement['rotation_degrees'] % 180 == 90: width_px, height_px = height_px, width_px width_cm = float(placement['width_cm']) length_cm = float(placement['length_cm']) drift = abs((width_px / height_px) / (width_cm / length_cm) - 1) if drift > ASPECT_TOLERANCE: source = production['sources'][placement['source_index']] raise Unsupported(f'file {placement["source_index"] + 1} has proportions that do not match ' f'the quoted {source["width_cm"]} x {source["length_cm"]} cm') if placement['source_index'] in vector: continue dpi = width_px / (width_cm / 2.54) lowest = dpi if lowest is None else min(lowest, dpi) return lowest def render(item, files, out, title): """Write the PDF for one approved order item. `files` maps each source index to (local path, original name). Returns the evidence the Kanban shows: page size, what was billed, and the lowest DPI. Raster sources are written by the streaming writer above. PDF sources are then added as vector forms by a second pass through pikepdf, so a sheet exported as PDF keeps its vectors, fonts and transparency. """ production = item['production'] if production.get('version') != 2: raise Unsupported('item uses an obsolete production layout') sources = [] try: for index in range(len(production['sources'])): path, name = files[index] sources.append(open_source(path, name)) vector = {index for index, source in enumerate(sources) if isinstance(source, PdfPage)} lowest_dpi = check_layout(item, [source.size for source in sources], vector) width_pt = float(production['film_width_cm']) * PT_PER_CM height_pt = float(production['height_cm']) * PT_PER_CM unit = max(1, math.ceil(max(width_pt, height_pt) / MAX_PAGE_PT)) first = tempfile.TemporaryFile() if vector else out try: write_rasters(production, sources, first, title, width_pt, height_pt, unit) if vector: first.seek(0) add_vector_pages(production, sources, first, out, height_pt) finally: if vector: first.close() finally: for source in sources: source.close() return {'film_width_cm': str(production['film_width_cm']), 'height_cm': str(production['height_cm']), 'billed_metres': str(item['billed_metres']), 'placements': len(production['placements']), 'sources': len(sources), 'vector_sources': len(vector), 'min_dpi': round(lowest_dpi) if lowest_dpi else None, 'user_unit': unit} def write_rasters(production, sources, out, title, width_pt, height_pt, unit): pdf = PdfWriter(out) images = {index: source.embed(pdf) for index, source in enumerate(sources) if isinstance(source, SourceImage)} # The user-space scale is not wrapped in q/Q, so it also applies to the # vector placements appended by the second pass. commands = [b'%s 0 0 %s 0 0 cm\n' % (serialize(1 / unit), serialize(1 / unit))] if unit > 1 else [] for placement in production['placements']: if placement['source_index'] not in images: continue matrix = placement_matrix(placement, height_pt) commands.append(b'q ' + b' '.join(serialize(v) for v in matrix) + b' cm /Im%d Do Q\n' % placement['source_index']) content = pdf.stream({'Filter': Name('FlateDecode')}, deflate(commands)) pages = pdf.alloc() page_box = [0, 0, width_pt / unit, height_pt / unit] page = {'Type': Name('Page'), 'Parent': pages, 'MediaBox': page_box, 'TrimBox': page_box, 'Resources': {'XObject': {f'Im{index}': ref for index, ref in images.items()}}, 'Contents': content} if unit > 1: page['UserUnit'] = unit page_ref = pdf.obj(page) pdf.obj({'Type': Name('Pages'), 'Kids': [page_ref], 'Count': 1}, pages) root = pdf.obj({'Type': Name('Catalog'), 'Pages': pages}) info = pdf.obj({'Title': title, 'Producer': 'DTF System print-file generator'}) pdf.finish(root, info) def add_vector_pages(production, sources, first, out, height_pt): import pikepdf with pikepdf.open(first) as document: page = document.pages[0] xobjects = page.obj.Resources.XObject for index, source in enumerate(sources): if isinstance(source, PdfPage): xobjects[f'/Pdf{index}'] = source.form(document) commands = [] for placement in production['placements']: source = sources[placement['source_index']] if not isinstance(source, PdfPage): continue matrix = multiply(source.normalise(), placement_matrix(placement, height_pt)) commands.append(b'q ' + b' '.join(serialize(v) for v in matrix) + b' cm /Pdf%d Do Q\n' % placement['source_index']) page.contents_add(pikepdf.Stream(document, b''.join(commands)), prepend=False) document.save(out, min_version='1.6')