1#!/usr/bin/env python3
2"""Compare owned-page probe geometry with native XML after exact text matching."""
3import argparse
4from html.parser import HTMLParser
5import json
6from pathlib import Path
7import xml.etree.ElementTree as ET
8
9NS = {'one': 'http://schemas.microsoft.com/office/onenote/2010/onenote'}
10
11
12class Text(HTMLParser):
13 def __init__(self, markup):
14 super().__init__(convert_charrefs=True)
15 self.parts = []
16 self.feed(markup)
17 self.close()
18
19 def handle_data(self, data):
20 self.parts.append(data)
21
22 def handle_starttag(self, tag, attrs):
23 if tag == 'br':
24 self.parts.append('\n')
25
26
27def compare(probe, native, allow_native_empty_nbsp=False):
28 if probe['title'] != native.get('name'):
29 raise ValueError('Source and native page titles differ')
30 outlines = native.findall('one:Outline', NS)
31 available = []
32 for outline in outlines:
33 paragraphs = outline.findall('.//one:OE', NS)
34 text = tuple(''.join(''.join(Text(t.text or '').parts)
35 for t in p.findall('one:T', NS)) for p in paragraphs)
36 if any(p.find('one:Image', NS) is not None or p.find('one:Table', NS) is not None
37 for p in paragraphs):
38 raise ValueError('Native outline includes non-text paragraph content')
39 available.append((text, outline))
40 results = []
41 for outline in probe['objects']:
42 if outline['kind'] != 'outline' or outline['is_title']:
43 continue
44 paragraphs = outline['paragraphs']
45 if outline['unsupported'] or any(p['unsupported'] for p in paragraphs):
46 raise ValueError('Source outline includes unsupported content')
47 text = tuple(p['visible_text'] for p in paragraphs)
48 match_text = tuple('' if t == '\u00a0' else t for t in text) if allow_native_empty_nbsp else text
49 matches = [(index, native) for index, (candidate, native) in enumerate(available)
50 if candidate == match_text]
51 if len(matches) != 1:
52 raise ValueError(f'{outline["id"]}: expected one native outline text match; got {len(matches)}')
53 index, matched = matches[0]
54 del available[index]
55 position = matched.find('one:Position', NS).attrib
56 size = matched.find('one:Size', NS).attrib
57 layout = outline['layout']
58 height = outline['size'][1] if outline.get('size') else sum(p['height'] for p in paragraphs)
59 results.append({
60 'source_id': outline['id'], 'native_id': matched.get('objectID'),
61 'paragraphs': len(paragraphs),
62 'text_matches_exactly': text == match_text,
63 'native_empty_nbsp_paragraphs': [i for i, (a, b) in enumerate(zip(text, match_text)) if a != b],
64 'source_width': layout['max_width'], 'native_width': float(size['width']),
65 'layout_height': height, 'native_height': float(size['height']),
66 'height_residual': height - float(size['height']),
67 'inferred_native_margin_origin': [float(position[axis]) - layout[axis] + margin
68 for axis, margin in zip(('x', 'y'), probe['margin_origin'])],
69 'nested_paragraphs': sum(p['parent'] is not None for p in paragraphs),
70 'list_paragraphs': sum(bool(p['lists']) for p in paragraphs),
71 })
72 if available:
73 raise ValueError('Native capture contains unmatched outlines')
74 return {'title': probe['title'], 'outlines': results,
75 'limits': ['Text matching identifies outlines; this comparison does not verify run formatting.',
76 probe['measurement'],
77 'Total height does not establish wrap offsets or first-baseline placement.',
78 'Inferred margin origins are observations, not a general page-origin rule.']}
79
80
81if __name__ == '__main__':
82 parser = argparse.ArgumentParser(description=__doc__)
83 parser.add_argument('probe', type=Path)
84 parser.add_argument('native_xml', type=Path)
85 parser.add_argument('--allow-native-empty-nbsp', action='store_true',
86 help='Record native empty paragraphs replacing a sole source NBSP')
87 args = parser.parse_args()
88 print(json.dumps(compare(json.loads(args.probe.read_text()), ET.parse(args.native_xml).getroot(),
89 args.allow_native_empty_nbsp), indent=2))