1#!/usr/bin/env python3
2"""Recover page-paragraph wrap evidence from the matching native PDF export."""
3import argparse
4from collections import Counter
5import json
6from pathlib import Path
7from types import SimpleNamespace
8import xml.etree.ElementTree as ET
9
10import pdfplumber
11from compare import pdf_line_ends, pdf_lines
12from compare_page import compare as compare_xml
13
14
15def occurrences(text, needle):
16 start = 0
17 while needle and (start := text.find(needle, start)) != -1:
18 yield start
19 start += 1
20
21
22def compare(probe, pages, include_titles=False):
23 outlines = [o for o in probe['objects'] if o['kind'] == 'outline' and (include_titles or not o['is_title'])]
24 source_paragraphs = [''.join(p['visible_text'].split()) for o in outlines for p in o['paragraphs']]
25 streams = []
26 for page in pages:
27 chars = [dict(char, text=scalar) for char in page.chars
28 for scalar in char['text'] if not scalar.isspace()]
29 streams.append((''.join(c['text'] for c in chars), chars, page.height))
30 results = []
31 for outline in outlines:
32 compact = [''.join(p['visible_text'].split()) for p in outline['paragraphs']]
33 whole = ''.join(compact)
34 outline_matches = [(page, start) for page, (text, _, _) in enumerate(streams)
35 for start in occurrences(text, whole)]
36 offset = 0
37 for index, (paragraph, text) in enumerate(zip(outline['paragraphs'], compact)):
38 row = {'outline_id': outline['id'], 'paragraph_id': paragraph['id'], 'paragraph_index': index,
39 'is_title': outline['is_title'],
40 'canvas_end_utf16': [line['end_utf16'] for line in paragraph['lines']],
41 'native_end_utf16': None, 'breaks_match': None}
42 results.append(row)
43 if not text:
44 row['status'] = 'no_visible_glyphs'
45 continue
46 if outline_matches:
47 matches = [(page, start + offset) for page, start in outline_matches]
48 row['match_context'] = 'complete_outline'
49 else:
50 matches = [(page, start) for page, (stream, _, _) in enumerate(streams)
51 for start in occurrences(stream, text)]
52 row['match_context'] = 'unique_paragraph'
53 if sum(len(list(occurrences(source, text))) for source in source_paragraphs) != 1:
54 matches = []
55 row['match_context'] = 'ambiguous_source_context'
56 offset += len(text)
57 if not matches:
58 row['status'] = 'unresolved_match'
59 continue
60 first_byte = len(paragraph['visible_text'][:next(i for i, c in enumerate(paragraph['visible_text'])
61 if not c.isspace())].encode('utf-8'))
62 span = next(s for s in paragraph['spans'] if first_byte < s['end_utf8'])
63 size = span['format'].get('font_size')
64 if size is None:
65 size = 11.0
66 instances = []
67 row['native_instances'] = instances
68 for page, start in matches:
69 _, chars, height = streams[page]
70 matched = chars[start:start + len(text)]
71 ends = pdf_line_ends(SimpleNamespace(chars=matched), paragraph['visible_text'])
72 instance = {'pdf_page': page + 1, 'end_utf16': ends}
73 instances.append(instance)
74 groups = pdf_lines(matched)
75 if groups is None:
76 continue
77 baselines = [(min(c['matrix'][5] for c in group), max(c['matrix'][5] for c in group))
78 for group in groups]
79 instance['baseline_ranges_from_pdf_top'] = [[height - high, height - low] for low, high in baselines]
80 instance['line_advance_ranges_pdf'] = [[a[0] - b[1], a[1] - b[0]]
81 for a, b in zip(baselines, baselines[1:])]
82 # Exported font sizes expose print scaling without fitting against canvas geometry.
83 instance['first_glyph_nominal_font_scale'] = matched[0]['size'] / size
84 ends = instances[0]['end_utf16']
85 if ends is None or any(instance['end_utf16'] != ends for instance in instances) or any(
86 not line['text'].strip() for line in paragraph['lines']):
87 row['status'] = 'unrecoverable_line_ranges'
88 continue
89 row['native_end_utf16'] = ends
90 row['breaks_match'] = ends == row['canvas_end_utf16']
91 row['status'] = 'matched' if row['breaks_match'] else 'different_wraps'
92 row['canvas_baselines_in_outline'] = [paragraph['origin'][1] + line['baseline']
93 for line in paragraph['lines']]
94 row['canvas_line_advances'] = [b['baseline'] - a['baseline']
95 for a, b in zip(paragraph['lines'], paragraph['lines'][1:])]
96 return {'title': probe['title'], 'counts': dict(Counter(row['status'] for row in results)),
97 'paragraphs': results,
98 'limits': ['Recovered wrap offsets require unambiguous ASCII paragraph text.',
99 'Complete-outline context disambiguates repeated paragraph text when available.',
100 'Repeated PDF occurrences must agree on every recovered break; all instances remain in the report.',
101 'PDF baseline coordinates include print scaling and pagination; they are not screen-baseline acceptance.',
102 'Empty paragraphs have no PDF glyph position; their geometry requires other evidence.']}
103
104
105if __name__ == '__main__':
106 parser = argparse.ArgumentParser(description=__doc__)
107 parser.add_argument('probe', type=Path)
108 parser.add_argument('native_xml', type=Path)
109 parser.add_argument('--allow-native-empty-nbsp', action='store_true')
110 parser.add_argument('--include-titles', action='store_true')
111 args = parser.parse_args()
112 probe = json.loads(args.probe.read_text())
113 compare_xml(probe, ET.parse(args.native_xml).getroot(), args.allow_native_empty_nbsp)
114 with pdfplumber.open(args.native_xml.with_suffix('.pdf')) as pdf:
115 result = compare(probe, pdf.pages, args.include_titles)
116 print(json.dumps(result, indent=2))