| 1 | #!/usr/bin/env python3 |
| 2 | """Recover page-paragraph wrap evidence from the matching native PDF export.""" |
| 3 | import argparse |
| 4 | from collections import Counter |
| 5 | import json |
| 6 | from pathlib import Path |
| 7 | from types import SimpleNamespace |
| 8 | import xml.etree.ElementTree as ET |
| 9 | |
| 10 | import pdfplumber |
| 11 | from compare import pdf_line_ends, pdf_lines |
| 12 | from compare_page import compare as compare_xml |
| 13 | |
| 14 | |
| 15 | def occurrences(text, needle): |
| 16 | start = 0 |
| 17 | while needle and (start := text.find(needle, start)) != -1: |
| 18 | yield start |
| 19 | start += 1 |
| 20 | |
| 21 | |
| 22 | def compare(probe, pages, include_titles=False): |
| 23 | outlines = [o for o in probe['objects'] if o['kind'] == 'outline' and (include_titles or not o['is_title'])] |
| 24 | source_paragraphs = [''.join(p['visible_text'].split()) for o in outlines for p in o['paragraphs']] |
| 25 | streams = [] |
| 26 | for page in pages: |
| 27 | chars = [dict(char, text=scalar) for char in page.chars |
| 28 | for scalar in char['text'] if not scalar.isspace()] |
| 29 | streams.append((''.join(c['text'] for c in chars), chars, page.height)) |
| 30 | results = [] |
| 31 | for outline in outlines: |
| 32 | compact = [''.join(p['visible_text'].split()) for p in outline['paragraphs']] |
| 33 | whole = ''.join(compact) |
| 34 | outline_matches = [(page, start) for page, (text, _, _) in enumerate(streams) |
| 35 | for start in occurrences(text, whole)] |
| 36 | offset = 0 |
| 37 | for index, (paragraph, text) in enumerate(zip(outline['paragraphs'], compact)): |
| 38 | row = {'outline_id': outline['id'], 'paragraph_id': paragraph['id'], 'paragraph_index': index, |
| 39 | 'is_title': outline['is_title'], |
| 40 | 'canvas_end_utf16': [line['end_utf16'] for line in paragraph['lines']], |
| 41 | 'native_end_utf16': None, 'breaks_match': None} |
| 42 | results.append(row) |
| 43 | if not text: |
| 44 | row['status'] = 'no_visible_glyphs' |
| 45 | continue |
| 46 | if outline_matches: |
| 47 | matches = [(page, start + offset) for page, start in outline_matches] |
| 48 | row['match_context'] = 'complete_outline' |
| 49 | else: |
| 50 | matches = [(page, start) for page, (stream, _, _) in enumerate(streams) |
| 51 | for start in occurrences(stream, text)] |
| 52 | row['match_context'] = 'unique_paragraph' |
| 53 | if sum(len(list(occurrences(source, text))) for source in source_paragraphs) != 1: |
| 54 | matches = [] |
| 55 | row['match_context'] = 'ambiguous_source_context' |
| 56 | offset += len(text) |
| 57 | if not matches: |
| 58 | row['status'] = 'unresolved_match' |
| 59 | continue |
| 60 | first_byte = len(paragraph['visible_text'][:next(i for i, c in enumerate(paragraph['visible_text']) |
| 61 | if not c.isspace())].encode('utf-8')) |
| 62 | span = next(s for s in paragraph['spans'] if first_byte < s['end_utf8']) |
| 63 | size = span['format'].get('font_size') |
| 64 | if size is None: |
| 65 | size = 11.0 |
| 66 | instances = [] |
| 67 | row['native_instances'] = instances |
| 68 | for page, start in matches: |
| 69 | _, chars, height = streams[page] |
| 70 | matched = chars[start:start + len(text)] |
| 71 | ends = pdf_line_ends(SimpleNamespace(chars=matched), paragraph['visible_text']) |
| 72 | instance = {'pdf_page': page + 1, 'end_utf16': ends} |
| 73 | instances.append(instance) |
| 74 | groups = pdf_lines(matched) |
| 75 | if groups is None: |
| 76 | continue |
| 77 | baselines = [(min(c['matrix'][5] for c in group), max(c['matrix'][5] for c in group)) |
| 78 | for group in groups] |
| 79 | instance['baseline_ranges_from_pdf_top'] = [[height - high, height - low] for low, high in baselines] |
| 80 | instance['line_advance_ranges_pdf'] = [[a[0] - b[1], a[1] - b[0]] |
| 81 | for a, b in zip(baselines, baselines[1:])] |
| 82 | # Exported font sizes expose print scaling without fitting against canvas geometry. |
| 83 | instance['first_glyph_nominal_font_scale'] = matched[0]['size'] / size |
| 84 | ends = instances[0]['end_utf16'] |
| 85 | if ends is None or any(instance['end_utf16'] != ends for instance in instances) or any( |
| 86 | not line['text'].strip() for line in paragraph['lines']): |
| 87 | row['status'] = 'unrecoverable_line_ranges' |
| 88 | continue |
| 89 | row['native_end_utf16'] = ends |
| 90 | row['breaks_match'] = ends == row['canvas_end_utf16'] |
| 91 | row['status'] = 'matched' if row['breaks_match'] else 'different_wraps' |
| 92 | row['canvas_baselines_in_outline'] = [paragraph['origin'][1] + line['baseline'] |
| 93 | for line in paragraph['lines']] |
| 94 | row['canvas_line_advances'] = [b['baseline'] - a['baseline'] |
| 95 | for a, b in zip(paragraph['lines'], paragraph['lines'][1:])] |
| 96 | return {'title': probe['title'], 'counts': dict(Counter(row['status'] for row in results)), |
| 97 | 'paragraphs': results, |
| 98 | 'limits': ['Recovered wrap offsets require unambiguous ASCII paragraph text.', |
| 99 | 'Complete-outline context disambiguates repeated paragraph text when available.', |
| 100 | 'Repeated PDF occurrences must agree on every recovered break; all instances remain in the report.', |
| 101 | 'PDF baseline coordinates include print scaling and pagination; they are not screen-baseline acceptance.', |
| 102 | 'Empty paragraphs have no PDF glyph position; their geometry requires other evidence.']} |
| 103 | |
| 104 | |
| 105 | if __name__ == '__main__': |
| 106 | parser = argparse.ArgumentParser(description=__doc__) |
| 107 | parser.add_argument('probe', type=Path) |
| 108 | parser.add_argument('native_xml', type=Path) |
| 109 | parser.add_argument('--allow-native-empty-nbsp', action='store_true') |
| 110 | parser.add_argument('--include-titles', action='store_true') |
| 111 | args = parser.parse_args() |
| 112 | probe = json.loads(args.probe.read_text()) |
| 113 | compare_xml(probe, ET.parse(args.native_xml).getroot(), args.allow_native_empty_nbsp) |
| 114 | with pdfplumber.open(args.native_xml.with_suffix('.pdf')) as pdf: |
| 115 | result = compare(probe, pdf.pages, args.include_titles) |
| 116 | print(json.dumps(result, indent=2)) |