| 1 | #!/usr/bin/env python3 |
| 2 | """Compare text probe output with native XML geometry and recoverable PDF line ranges.""" |
| 3 | import argparse |
| 4 | import json |
| 5 | from pathlib import Path |
| 6 | import xml.etree.ElementTree as ET |
| 7 | |
| 8 | import pdfplumber |
| 9 | from native_fixture import build |
| 10 | |
| 11 | NS = {'one': 'http://schemas.microsoft.com/office/onenote/2010/onenote'} |
| 12 | |
| 13 | |
| 14 | def pdf_lines(chars): |
| 15 | lines = [] |
| 16 | for char in sorted(chars, key=lambda c: c['matrix'][5], reverse=True): |
| 17 | overlaps = [line for line in lines if max(line[0], char['y0']) < min(line[1], char['y1'])] |
| 18 | if len(overlaps) > 1: |
| 19 | return None |
| 20 | if overlaps: |
| 21 | line = overlaps[0] |
| 22 | line[0] = max(line[0], char['y0']) |
| 23 | line[1] = min(line[1], char['y1']) |
| 24 | line[2].append(char) |
| 25 | else: |
| 26 | lines.append([char['y0'], char['y1'], [char]]) |
| 27 | return [sorted(line[2], key=lambda c: c['x0']) for line in lines] |
| 28 | |
| 29 | |
| 30 | def pdf_line_ends(page, source): |
| 31 | if not source or not source.isascii(): |
| 32 | return None |
| 33 | groups = pdf_lines(page.chars) |
| 34 | if groups is None: |
| 35 | return None |
| 36 | lines = [''.join(c['text'] for c in chars) for chars in groups] |
| 37 | compact_source = ''.join(source.split()) |
| 38 | matches = [] |
| 39 | for start in range(len(lines)): |
| 40 | for end in range(start + 1, len(lines) + 1): |
| 41 | if ''.join(''.join(lines[start:end]).split()) == compact_source: |
| 42 | matches.append(lines[start:end]) |
| 43 | if len(matches) != 1: |
| 44 | return None |
| 45 | offset = 0 |
| 46 | ends = [] |
| 47 | for line in matches[0]: |
| 48 | for character in ''.join(line.split()): |
| 49 | while offset < len(source) and source[offset].isspace(): |
| 50 | offset += 1 |
| 51 | if offset == len(source) or source[offset] != character: |
| 52 | return None |
| 53 | offset += 1 |
| 54 | while offset < len(source) and source[offset].isspace(): |
| 55 | offset += 1 |
| 56 | ends.append(offset) |
| 57 | return ends if offset == len(source) else None |
| 58 | |
| 59 | |
| 60 | def compare(cases, parley, core_text, native): |
| 61 | if json.loads((native.parent / 'notebook' / 'fixture.json').read_text(encoding='utf-8-sig')) != build(cases): |
| 62 | raise ValueError('Native capture was generated from different input cases') |
| 63 | parley = {case['id']: case for case in parley['cases']} |
| 64 | core_text = {case['id']: case for case in core_text['cases']} |
| 65 | pages = {} |
| 66 | for path in native.glob('page-*.xml'): |
| 67 | page = ET.parse(path).getroot() |
| 68 | name = page.get('name') |
| 69 | if name in pages: |
| 70 | raise ValueError(f'Duplicate native page: {name}') |
| 71 | pages[name] = (path, page) |
| 72 | ids = {case['id'] for case in cases} |
| 73 | if len(ids) != len(cases) or ids != parley.keys() or ids != core_text.keys() or ids != pages.keys(): |
| 74 | raise ValueError('Input and captured case identities must match exactly') |
| 75 | results = [] |
| 76 | for case in cases: |
| 77 | name = case['id'] |
| 78 | p, c = parley[name], core_text[name] |
| 79 | for result in (p, c): |
| 80 | if result['width'] != case['width']: |
| 81 | raise ValueError(f'{name}: probe used a different width') |
| 82 | actual = [{k: v for k, v in run.items() if k != 'font'} for run in result['requested_runs']] |
| 83 | expected = [{k: v for k, v in run.items() if k != 'font'} for run in case['runs']] |
| 84 | if actual != expected: |
| 85 | raise ValueError(f'{name}: probe used different text or styles') |
| 86 | path, page = pages[name] |
| 87 | sizes = page.findall('one:Outline/one:Size', NS) |
| 88 | source = ''.join(run['text'] for run in case['runs']) |
| 89 | native_ends = None |
| 90 | with pdfplumber.open(path.with_suffix('.pdf')) as pdf: |
| 91 | if len(pdf.pages) == 1: |
| 92 | native_ends = pdf_line_ends(pdf.pages[0], source) |
| 93 | dimensions = [{k: float(size.attrib[k]) for k in ('width', 'height')} for size in sizes] |
| 94 | width_changed = len(dimensions) != 1 or abs(dimensions[0]['width'] - case['width']) > 0.001 |
| 95 | windows_height = 0.0 |
| 96 | for line in p['lines']: |
| 97 | runs = line['runs'] |
| 98 | if not runs or any(r[k] is None for r in runs for k in ('win_ascent', 'win_descent', 'units_per_em')): |
| 99 | windows_height = None |
| 100 | break |
| 101 | windows_height += max(r['win_ascent'] * r['size'] / r['units_per_em'] for r in runs) |
| 102 | windows_height += max(r['win_descent'] * r['size'] / r['units_per_em'] for r in runs) |
| 103 | ends = {engine: [line['end_utf16'] for line in result['lines']] |
| 104 | for engine, result in [('parley', p), ('core_text', c)]} |
| 105 | results.append({ |
| 106 | 'id': name, |
| 107 | 'requested_font_override': {engine: [r['font'] for r in result['requested_runs']] |
| 108 | for engine, result in [('parley', p), ('core_text', c)]} |
| 109 | if any(result['requested_runs'] != case['runs'] for result in (p, c)) else None, |
| 110 | 'requested_width': case['width'], |
| 111 | 'native_outlines': dimensions, |
| 112 | 'native_changed_width_or_removed_outline': width_changed, |
| 113 | 'native_pdf_end_utf16': native_ends, |
| 114 | 'end_utf16': ends, |
| 115 | 'parley_default_height': p['height'], |
| 116 | 'canvas_height': p.get('windows_height'), |
| 117 | 'canvas_height_residual': (p['windows_height'] - dimensions[0]['height']) |
| 118 | if p.get('windows_height') is not None and not width_changed else None, |
| 119 | 'windows_metric_height_hypothesis': windows_height, |
| 120 | 'height_residual_hypothesis': (windows_height - dimensions[0]['height']) |
| 121 | if windows_height is not None and not width_changed else None, |
| 122 | 'parley_resolved_faces': sorted({r['face'] for line in p['lines'] for r in line['runs']}), |
| 123 | 'core_text_requested_faces_resolved': c['requested_faces_resolved'], |
| 124 | 'same_width_pdf_break_agreement': {engine: values == native_ends for engine, values in ends.items()} |
| 125 | if native_ends is not None and not width_changed else None, |
| 126 | }) |
| 127 | return {'cases': results, |
| 128 | 'limits': ['PDF ranges are supplementary evidence, recovered only for unambiguous ASCII text.', |
| 129 | 'Metric hypotheses use the Mac-resolved fonts and do not establish matching native font identity.', |
| 130 | 'Width changes and removed empty outlines require separate native cases.', |
| 131 | 'XML height alone does not establish first-baseline placement.']} |
| 132 | |
| 133 | |
| 134 | if __name__ == '__main__': |
| 135 | parser = argparse.ArgumentParser(description=__doc__) |
| 136 | parser.add_argument('cases', type=Path) |
| 137 | parser.add_argument('parley', type=Path) |
| 138 | parser.add_argument('core_text', type=Path) |
| 139 | parser.add_argument('native_read', type=Path) |
| 140 | args = parser.parse_args() |
| 141 | result = compare(json.loads(args.cases.read_text()), json.loads(args.parley.read_text()), |
| 142 | json.loads(args.core_text.read_text()), args.native_read) |
| 143 | print(json.dumps(result, indent=2)) |