"""The print file: the reviewed layout, reproduced exactly, at the film's size. The customer is quoted on a layout the Site computes: each copy of each artwork at a width, rotation, mirror and position on the film. Production spec v2 keeps that layout through the approved order. This module turns it into one PDF per order item, with a page exactly as wide as the film and as long as the layout, so the operator imports what the customer approved instead of rebuilding it. Each source image is embedded once, at its original resolution, and every copy is a placement of it. Nothing is resampled: a 300 DPI artwork is still 300 DPI in the file, a JPEG keeps its original bytes, and transparency survives as a soft mask. The output is therefore about the size of the artwork, not of a 57 cm x 20 m raster, and it never needs that raster in memory. Only formats whose pixels can be read here are generated: JPEG, PNG, WebP and TIFF. Anything else (PDF, PSD, AI, CDR), or a file whose proportions do not match the size it was quoted at, is refused with a reason, and the operator prepares that item by hand exactly as before. """ import math import os import zlib from decimal import Decimal from PIL import Image, ImageOps PT_PER_CM = 72 / 2.54 # Acrobat's page limit. Longer layouts scale user space with /UserUnit (PDF 1.6) # rather than cutting the film into pages a RIP might print with gaps. MAX_PAGE_PT = 14400 # How far a file's proportions may drift from the quoted size before the item # is refused: rounding in the Site keeps real files well inside this. ASPECT_TOLERANCE = 0.01 # The quote may round up (10 cm steps, 1 m minimum) but never down. HEIGHT_TOLERANCE_CM = Decimal('0.05') SUPPORTED_FORMATS = {'JPEG', 'PNG', 'WEBP', 'TIFF'} STRIP_ROWS = 256 # Decoding holds the whole image in memory (about 4 bytes a pixel). 57 cm x 3 m # at 300 DPI is 239 Mpx; above this the item goes to the operator instead. MAX_DECODED_PIXELS = int(os.environ.get('PRINT_MAX_PIXELS', '250000000')) # Pillow's own bomb guard would refuse a genuine long sheet before we can # decide; the explicit limit above is the one that applies. Image.MAX_IMAGE_PIXELS = None class Unsupported(Exception): """This item cannot be generated automatically; the reason is for the operator.""" class Name(str): pass class Ref(int): pass def serialize(value): if isinstance(value, Ref): return b'%d 0 R' % value if isinstance(value, Name): return b'/' + value.encode('ascii') if isinstance(value, bool): return b'true' if value else b'false' if isinstance(value, int): return b'%d' % value if isinstance(value, (float, Decimal)): text = f'{float(value):.4f}'.rstrip('0').rstrip('.') return (text if text not in ('', '-0') else '0').encode('ascii') if isinstance(value, dict): return b'<<' + b''.join(b'/' + k.encode('ascii') + b' ' + serialize(v) for k, v in value.items()) + b'>>' if isinstance(value, (list, tuple)): return b'[' + b' '.join(serialize(v) for v in value) + b']' if isinstance(value, str): escaped = value.replace('\\', '\\\\').replace('(', '\\(').replace(')', '\\)') return b'(' + escaped.encode('latin-1', 'replace') + b')' raise TypeError(f'Cannot serialize {type(value).__name__}') class PdfWriter: """A sequential PDF writer: objects go straight to the file, streams included. Stream lengths are indirect objects written after the data, so an image is compressed strip by strip into the output without being held in memory. """ def __init__(self, fp): self.fp = fp self.offsets = {} self.next_id = 1 self.write(b'%PDF-1.6\n%\xe2\xe3\xcf\xd3\n') def write(self, data): self.fp.write(data) def tell(self): return self.fp.tell() def alloc(self): ref = Ref(self.next_id) self.next_id += 1 return ref def obj(self, value, ref=None): ref = ref or self.alloc() self.offsets[ref] = self.tell() self.write(b'%d 0 obj\n' % ref + serialize(value) + b'\nendobj\n') return ref def stream(self, dictionary, chunks, ref=None): """Write a stream from an iterable of already-encoded byte chunks.""" ref = ref or self.alloc() length = self.alloc() self.offsets[ref] = self.tell() self.write(b'%d 0 obj\n' % ref + serialize({**dictionary, 'Length': length}) + b'\nstream\n') start = self.tell() for chunk in chunks: self.write(chunk) size = self.tell() - start self.write(b'\nendstream\nendobj\n') self.obj(size, length) return ref def finish(self, root, info): xref = self.tell() count = self.next_id self.write(b'xref\n0 %d\n0000000000 65535 f \n' % count) for ref in range(1, count): self.write(b'%010d 00000 n \n' % self.offsets[ref]) self.write(b'trailer\n' + serialize({'Size': count, 'Root': root, 'Info': info}) + b'\nstartxref\n%d\n%%%%EOF\n' % xref) def deflate(pieces): compressor = zlib.compressobj(6) for piece in pieces: out = compressor.compress(piece) if out: yield out yield compressor.flush() class SourceImage: """One customer file, opened for reading pixels and measured as the browser sees it.""" def __init__(self, path, name): self.path = path self.name = name try: self.image = Image.open(path) self.format = self.image.format except Image.DecompressionBombError as exc: raise Unsupported(f'"{name}" is too large to generate automatically') from exc except Exception as exc: raise Unsupported(f'"{name}" is not an image this generator can read') from exc if self.format not in SUPPORTED_FORMATS: raise Unsupported(f'"{name}" is {self.format or "an unknown format"}; ' 'only JPEG, PNG, WebP and TIFF are generated automatically') if getattr(self.image, 'n_frames', 1) > 1 and self.format != 'TIFF': raise Unsupported(f'"{name}" is animated or has several frames') # Browsers draw a photo upright according to its EXIF orientation, and # the Site measured it that way, so the print must too. try: self.orientation = self.image.getexif().get(0x0112, 1) except Exception: self.orientation = 1 width, height = self.image.size self.size = (height, width) if self.orientation in (5, 6, 7, 8) else (width, height) def passthrough(self): """Whether the original JPEG bytes can go into the PDF unchanged.""" return (self.format == 'JPEG' and self.orientation == 1 and self.image.mode in ('L', 'RGB', 'CMYK')) def embed(self, pdf): """Write this image (and its alpha) as XObjects; returns the image reference.""" if self.passthrough(): return self._embed_jpeg(pdf) width, height = self.image.size if width * height > MAX_DECODED_PIXELS: raise Unsupported(f'"{self.name}" has {width} x {height} px, more than the ' 'generator decodes; prepare this item by hand') return self._embed_pixels(pdf) def _colorspace(self, pdf, mode): components = {'L': 1, 'RGB': 3, 'CMYK': 4}[mode] device = Name({'L': 'DeviceGray', 'RGB': 'DeviceRGB', 'CMYK': 'DeviceCMYK'}[mode]) profile = self.image.info.get('icc_profile') if not profile: return device icc = pdf.stream({'N': components, 'Alternate': device, 'Filter': Name('FlateDecode')}, deflate([profile])) return [Name('ICCBased'), icc] def _embed_jpeg(self, pdf): image = self.image extra = {} if image.mode == 'CMYK' and 'adobe' in image.info: # Adobe writes CMYK JPEGs inverted; PDF readers expect the Decode flip. extra['Decode'] = [1, 0, 1, 0, 1, 0, 1, 0] def chunks(): with open(self.path, 'rb') as source: while block := source.read(1 << 20): yield block return pdf.stream({'Type': Name('XObject'), 'Subtype': Name('Image'), 'Width': image.width, 'Height': image.height, 'ColorSpace': self._colorspace(pdf, image.mode), 'BitsPerComponent': 8, 'Filter': Name('DCTDecode'), **extra}, chunks()) def _upright(self): image = self.image image.load() if self.orientation != 1: image = ImageOps.exif_transpose(image) return image def _embed_pixels(self, pdf): image = self._upright() mode = image.mode has_alpha = mode in ('RGBA', 'LA', 'PA', 'RGBa', 'La') or ( mode == 'P' and 'transparency' in image.info) or ( mode in ('L', 'RGB') and 'transparency' in image.info) if mode in ('I;16', 'I;16B', 'I;16L', 'I'): image = image.convert('I').point(lambda value: value * (1 / 257)).convert('L') mode = 'L' if mode == 'CMYK': color_mode = 'CMYK' elif mode in ('1', 'L', 'LA', 'La'): color_mode = 'L' elif mode in ('P', 'PA', 'RGB', 'RGBA', 'RGBa'): color_mode = 'RGB' else: raise Unsupported(f'"{self.name}" uses the {mode} colour mode, which is not generated automatically') if has_alpha: image = image.convert('RGBA' if color_mode == 'RGB' else 'LA') width, height = image.size def strips(convert): for top in range(0, height, STRIP_ROWS): yield convert(image.crop((0, top, width, min(height, top + STRIP_ROWS)))).tobytes() smask = None if has_alpha: smask = pdf.stream({'Type': Name('XObject'), 'Subtype': Name('Image'), 'Width': width, 'Height': height, 'ColorSpace': Name('DeviceGray'), 'BitsPerComponent': 8, 'Filter': Name('FlateDecode')}, deflate(strips(lambda strip: strip.getchannel('A')))) colorspace = self._colorspace(pdf, color_mode) dictionary = {'Type': Name('XObject'), 'Subtype': Name('Image'), 'Width': width, 'Height': height, 'ColorSpace': colorspace, 'BitsPerComponent': 8, 'Filter': Name('FlateDecode')} if smask: dictionary['SMask'] = smask return pdf.stream(dictionary, deflate(strips(lambda strip: strip.convert(color_mode)))) def close(self): self.image.close() def placement_matrix(placement, page_height_pt): """Map the image's unit square onto its box on the film. Matches the Site's canvas: the artwork is mirrored first, then turned clockwise about the centre of its box, and the turned image fills the box. """ x = float(placement['x_cm']) * PT_PER_CM top = page_height_pt - float(placement['y_cm']) * PT_PER_CM w = float(placement['width_cm']) * PT_PER_CM h = float(placement['length_cm']) * PT_PER_CM rotation = placement['rotation_degrees'] % 360 mirrored = placement['mirrored'] def to_page(u, v): # Unit square (v up) -> image frame (s right, t down). s, t = u, 1 - v if mirrored: s = 1 - s s, t = {0: (s, t), 90: (1 - t, s), 180: (1 - s, 1 - t), 270: (t, 1 - s)}[rotation] return x + s * w, top - t * h ox, oy = to_page(0, 0) ax, ay = to_page(1, 0) cx, cy = to_page(0, 1) return [ax - ox, ay - oy, cx - ox, cy - oy, ox, oy] def check_layout(item, sizes): """Refuse a layout the file cannot reproduce faithfully. Returns the lowest DPI.""" production = item['production'] height = Decimal(str(production['height_cm'])) billed = Decimal(str(item['billed_metres'])) * 100 if height > billed + HEIGHT_TOLERANCE_CM: raise Unsupported(f'layout is {height} cm long but only {billed} cm were billed') lowest = None for placement in production['placements']: width_px, height_px = sizes[placement['source_index']] if placement['rotation_degrees'] % 180 == 90: width_px, height_px = height_px, width_px width_cm = float(placement['width_cm']) length_cm = float(placement['length_cm']) drift = abs((width_px / height_px) / (width_cm / length_cm) - 1) if drift > ASPECT_TOLERANCE: source = production['sources'][placement['source_index']] raise Unsupported(f'file {placement["source_index"] + 1} has proportions that do not match ' f'the quoted {source["width_cm"]} x {source["length_cm"]} cm') dpi = width_px / (width_cm / 2.54) lowest = dpi if lowest is None else min(lowest, dpi) return lowest def render(item, files, out, title): """Write the PDF for one approved order item. `files` maps each source index to (local path, original name). Returns the evidence the Kanban shows: page size, what was billed, and the lowest DPI. """ production = item['production'] if production.get('version') != 2: raise Unsupported('item uses an obsolete production layout') sources = [] try: for index in range(len(production['sources'])): path, name = files[index] sources.append(SourceImage(path, name)) lowest_dpi = check_layout(item, [source.size for source in sources]) width_pt = float(production['film_width_cm']) * PT_PER_CM height_pt = float(production['height_cm']) * PT_PER_CM unit = max(1, math.ceil(max(width_pt, height_pt) / MAX_PAGE_PT)) pdf = PdfWriter(out) images = [source.embed(pdf) for source in sources] commands = [b'%s 0 0 %s 0 0 cm\n' % (serialize(1 / unit), serialize(1 / unit))] if unit > 1 else [] for placement in production['placements']: matrix = placement_matrix(placement, height_pt) commands.append(b'q ' + b' '.join(serialize(v) for v in matrix) + b' cm /Im%d Do Q\n' % placement['source_index']) content = pdf.stream({'Filter': Name('FlateDecode')}, deflate(commands)) pages = pdf.alloc() page_box = [0, 0, width_pt / unit, height_pt / unit] page = {'Type': Name('Page'), 'Parent': pages, 'MediaBox': page_box, 'TrimBox': page_box, 'Resources': {'XObject': {f'Im{index}': ref for index, ref in enumerate(images)}}, 'Contents': content} if unit > 1: page['UserUnit'] = unit page_ref = pdf.obj(page) pdf.obj({'Type': Name('Pages'), 'Kids': [page_ref], 'Count': 1}, pages) root = pdf.obj({'Type': Name('Catalog'), 'Pages': pages}) info = pdf.obj({'Title': title, 'Producer': 'DTF System print-file generator'}) pdf.finish(root, info) finally: for source in sources: source.close() return {'film_width_cm': str(production['film_width_cm']), 'height_cm': str(production['height_cm']), 'billed_metres': str(item['billed_metres']), 'placements': len(production['placements']), 'sources': len(sources), 'min_dpi': round(lowest_dpi) if lowest_dpi else None, 'user_unit': unit}