| 1 | #!/usr/bin/env python3 |
| 2 | """Compare PDF baselines using identical opaque image anchors, independently of text layout.""" |
| 3 | import argparse |
| 4 | import base64 |
| 5 | from collections import Counter, defaultdict |
| 6 | import hashlib |
| 7 | import io |
| 8 | import json |
| 9 | import math |
| 10 | from pathlib import Path |
| 11 | import statistics |
| 12 | import xml.etree.ElementTree as ET |
| 13 | |
| 14 | import pdfplumber |
| 15 | from PIL import Image |
| 16 | |
| 17 | from compare_page_pdf import compare |
| 18 | from native_fixture import NS |
| 19 | |
| 20 | |
| 21 | def pdf_rgb(image): |
| 22 | stream = image['stream'] |
| 23 | if any(key in stream.attrs for key in ['Mask', 'SMask', 'Decode']) or image.get('imagemask'): |
| 24 | return None |
| 25 | space = image['colorspace'] |
| 26 | names = [getattr(value, 'name', None) for value in space] |
| 27 | size = tuple(image['srcsize']) |
| 28 | if names == ['DeviceRGB'] and image['bits'] == 8: |
| 29 | return Image.frombytes('RGB', size, stream.get_data()).tobytes() |
| 30 | if names[:2] == ['Indexed', 'DeviceRGB'] and len(space) == 4 and image['bits'] in (4, 8) and isinstance(space[3], bytes): |
| 31 | pixels = Image.frombytes('P', size, stream.get_data(), 'raw', 'P;4' if image['bits'] == 4 else 'P') |
| 32 | if len(space[3]) != (space[2] + 1) * 3 or pixels.getextrema()[1] > space[2]: |
| 33 | return None |
| 34 | pixels.putpalette(space[3]) |
| 35 | return pixels.convert('RGB').tobytes() |
| 36 | return None |
| 37 | |
| 38 | |
| 39 | def register_images(xml, pages): |
| 40 | sources = [] |
| 41 | for node in xml.findall(f'{{{NS}}}Image'): |
| 42 | data = node.findtext(f'{{{NS}}}Data') |
| 43 | if not data: |
| 44 | continue |
| 45 | with Image.open(io.BytesIO(base64.b64decode(''.join(data.split()), validate=True))) as image: |
| 46 | if image.convert('RGBA').getchannel('A').getextrema() != (255, 255): |
| 47 | continue |
| 48 | key = (image.size, hashlib.sha256(image.convert('RGB').tobytes()).hexdigest()) |
| 49 | position = node.find(f'{{{NS}}}Position') |
| 50 | size = node.find(f'{{{NS}}}Size') |
| 51 | rect = [float(position.get('x')), float(position.get('y')), float(size.get('width')), float(size.get('height'))] |
| 52 | if not all(map(math.isfinite, rect)) or min(rect[2:]) <= 0: |
| 53 | raise ValueError('Invalid source image geometry') |
| 54 | sources.append((key, node.get('objectID'), rect)) |
| 55 | counts = Counter(key for key, _, _ in sources) |
| 56 | candidates = defaultdict(list) |
| 57 | for index, page in enumerate(pages): |
| 58 | for image in page.images: |
| 59 | pixels = pdf_rgb(image) |
| 60 | if pixels is not None: |
| 61 | candidates[(tuple(image['srcsize']), hashlib.sha256(pixels).hexdigest())].append((index, image)) |
| 62 | registrations = defaultdict(list) |
| 63 | for key, identity, rect in sources: |
| 64 | matches = candidates[key] |
| 65 | if counts[key] != 1 or len(matches) != 1: |
| 66 | continue |
| 67 | index, image = matches[0] |
| 68 | pdf_rect = [image['x0'], image['top'], image['x1'], image['bottom']] |
| 69 | if not all(map(math.isfinite, pdf_rect)) or pdf_rect[2] <= pdf_rect[0] or pdf_rect[3] <= pdf_rect[1]: |
| 70 | raise ValueError('Invalid PDF image geometry') |
| 71 | registrations[index].append({ |
| 72 | 'source_id': identity, 'pixel_sha256': key[1], 'source_xywh': rect, 'pdf_rect': pdf_rect, |
| 73 | 'translation': [pdf_rect[0] - rect[0], pdf_rect[1] - rect[1]], |
| 74 | 'extent_scale': [(pdf_rect[2] - pdf_rect[0]) / rect[2], (pdf_rect[3] - pdf_rect[1]) / rect[3]], |
| 75 | }) |
| 76 | return dict(registrations) |
| 77 | |
| 78 | |
| 79 | def baseline_report(probe, xml, pages): |
| 80 | anchors = register_images(xml, pages) |
| 81 | wraps = compare(probe, pages) |
| 82 | outlines = {o['id']: o for o in probe['objects'] if o['kind'] == 'outline' and not o['is_title']} |
| 83 | rows = [] |
| 84 | for paragraph in wraps['paragraphs']: |
| 85 | if paragraph['status'] != 'matched': |
| 86 | rows.append({**paragraph, 'lines': []}) |
| 87 | continue |
| 88 | outline = outlines[paragraph['outline_id']] |
| 89 | for instance in paragraph['native_instances']: |
| 90 | page = instance['pdf_page'] - 1 |
| 91 | registration = anchors.get(page, []) |
| 92 | row = {'outline_id': paragraph['outline_id'], 'paragraph_id': paragraph['paragraph_id'], 'pdf_page': page + 1, 'lines': []} |
| 93 | rows.append(row) |
| 94 | if len({a['source_xywh'][1] for a in registration}) < 2: |
| 95 | row['status'] = 'insufficient_image_anchors' |
| 96 | continue |
| 97 | row['status'] = 'measured_translation_residuals' |
| 98 | offsets = [a['translation'][1] for a in registration] |
| 99 | for native, local in zip(instance['baseline_ranges_from_pdf_top'], paragraph['canvas_baselines_in_outline'], strict=True): |
| 100 | document = outline['layout']['y'] + local |
| 101 | row['lines'].append({ |
| 102 | 'canvas_document_baseline': document, 'native_pdf_baseline_range': native, |
| 103 | 'residual_using_median_anchor_translation': [v - document - statistics.median(offsets) for v in native], |
| 104 | 'residual_range_over_anchor_translations': [native[0] - document - max(offsets), native[1] - document - min(offsets)], |
| 105 | }) |
| 106 | return {'anchors_by_pdf_page': {str(page + 1): rows for page, rows in anchors.items()}, |
| 107 | 'wrap_counts': wraps['counts'], 'paragraphs': rows, |
| 108 | 'limits': 'Translation candidates come only from identical opaque images. Their spread and extent scales remain visible; no text-fitted registration or baseline acceptance tolerance is applied. PDF coordinates do not establish screen glyph baselines.'} |
| 109 | |
| 110 | |
| 111 | if __name__ == '__main__': |
| 112 | parser = argparse.ArgumentParser(description=__doc__) |
| 113 | parser.add_argument('probe', type=Path) |
| 114 | parser.add_argument('native_xml', type=Path) |
| 115 | parser.add_argument('native_pdf', type=Path) |
| 116 | args = parser.parse_args() |
| 117 | with pdfplumber.open(args.native_pdf) as pdf: |
| 118 | print(json.dumps(baseline_report(json.loads(args.probe.read_text()), ET.parse(args.native_xml).getroot(), pdf.pages), indent=2)) |