1#!/usr/bin/env python3
2"""Compare document order and text with independently captured OneNote XML."""
3import argparse
4import base64
5import hashlib
6from io import BytesIO
7from tempfile import TemporaryDirectory
8from PIL import Image
9from datetime import datetime, timezone
10import json
11import math
12import re
13from pathlib import Path, PureWindowsPath
14from zoneinfo import ZoneInfo
15import subprocess
16import xml.etree.ElementTree as ET
17import uuid
18import unicodedata
19from native_xml import ns, pages, texts, project_text
20from document_model import EXPORTER, ordered_pages, version_pages, walk
21from native_format import native_characters, compare_formats
22
23ROOT = Path(__file__).resolve().parent.parent
24
25
26def verify_pdf_black(path, paragraphs):
27 import pdfplumber
28 characters = []
29 with pdfplumber.open(path) as pdf:
30 for page in pdf.pages:
31 black = [r for r in page.rects if r.get('fill') and r.get('non_stroking_color') in (0, (0, 0, 0))]
32 for char in page.chars:
33 if char['text'].isspace():
34 continue
35 x, y = (char['x0'] + char['x1']) / 2, (char['top'] + char['bottom']) / 2
36 covered = any(r['x0'] - .2 <= x <= r['x1'] + .2 and r['top'] - .2 <= y <= r['bottom'] + .2 for r in black)
37 characters.extend((c, covered, (page.page_number, char, black)) for c in char['text'])
38 text = ''.join(c for c, _, _ in characters)
39 checked = 0
40 for runs in paragraphs:
41 source = [(c, run['format']['highlight'] == 0) for run in runs if not run['format']['hidden']
42 for c in run['text'] if not c.isspace()]
43 if not any(black for _, black in source):
44 continue
45 value = ''.join(c for c, _ in source)
46 omitted = []
47 if value not in text:
48 # OneNote can map rendered combining clusters to spaces in PDF text.
49 projected = []
50 index = 0
51 while index < len(source):
52 end = index + 1
53 while end < len(source) and unicodedata.combining(source[end][0]):
54 end += 1
55 if end > index + 1:
56 assert not any(black for _, black in source[index:end]), 'Black-highlighted unmapped text requires visual verification'
57 omitted.append(len(projected))
58 else:
59 projected.append(source[index])
60 index = end
61 source = projected
62 value = ''.join(c for c, _ in source)
63 start = text.find(value)
64 assert start >= 0 and text.find(value, start + 1) < 0, 'Native PDF paragraph text is missing or ambiguous'
65 assert [(c, covered) for c, covered, _ in characters[start:start + len(source)]] == source, 'Native PDF black rectangles disagree with the stored character positions'
66 for index in omitted:
67 assert 0 < index < len(source), 'Unmapped text has no surrounding PDF characters'
68 left_page, left, black = characters[start + index - 1][2]
69 right_page, right, _ = characters[start + index][2]
70 assert left_page == right_page and abs(left['top'] - right['top']) < .2 and left['x1'] < right['x0'], 'Unmapped text has no unambiguous inline PDF position'
71 assert not any(r['x0'] < right['x0'] - .2 and r['x1'] > left['x1'] + .2
72 and r['top'] < min(left['bottom'], right['bottom']) - .2
73 and r['bottom'] > max(left['top'], right['top']) + .2 for r in black), 'Native PDF black rectangles cover unhighlighted unmapped text'
74 checked += 1
75 assert checked, 'No stored black-highlight paragraph was checked'
76 return checked
77
78
79def visible_text(node, space):
80 kind = node['kind']
81 data = kind['text'].encode('utf-16-le')
82 # Hidden runs and equation runs (exported as MathML) have no visible native text.
83 text = ''.join(data[run['start'] * 2:run['end'] * 2].decode('utf-16-le') for run in kind['runs']
84 if not run['format'] or not (space['nodes'][run['format']]['format']['hidden']
85 or space['nodes'][run['format']]['format']['math']))
86 # A trailing CR ends the stored text; one inside it, between equations, is a line break.
87 if kind['text'].endswith('\r'):
88 text = text.removesuffix('\r')
89 return "" if text == "\u00a0" else project_text(text)
90
91
92ISF_X = bytes.fromhex('8f6a8a59c052a04b93afaf357411a561')
93ISF_Y = bytes.fromhex('759f3fb5e0049844a7eec30dbb5a9011')
94
95
96def multi_byte(data):
97 """ISF multi-byte signed integers: a count, then 7-bit little-endian varints with the sign in bit 0."""
98 values = []
99 index = 0
100 while index < len(data):
101 value = shift = 0
102 while True:
103 byte = data[index]
104 index += 1
105 value |= (byte & 0x7f) << shift
106 shift += 7
107 if not byte & 0x80:
108 break
109 values.append(-(value >> 1) if value & 1 else value >> 1)
110 count, values = values[0], values[1:]
111 assert count == len(values), 'Ink packet count differs'
112 return values
113
114
115def ink_extent(space, container):
116 """Every stroke point of an ink container in points, as [left, top, width, height]."""
117 points = []
118 scale = [container['kind']['scale_x'] or 1.0, container['kind']['scale_y'] or 1.0]
119 for stroke_id in space['nodes'][container['kind']['data']]['kind']['strokes']:
120 stroke = space['nodes'][stroke_id]['kind']
121 style = space['nodes'][stroke['style']]['kind']
122 dimensions = bytes(style['dimensions'])
123 guids = [dimensions[i:i + 16] for i in range(0, len(dimensions), 32)]
124 values = multi_byte(bytes(stroke['path']))
125 per_dimension = len(values) // len(guids)
126 axes = []
127 for guid, factor in ((ISF_X, scale[0]), (ISF_Y, scale[1])):
128 start = guids.index(guid) * per_dimension
129 position, coordinates = 0, []
130 for delta in values[start:start + per_dimension]:
131 position += delta
132 coordinates.append(position * factor * 72 / 2540)
133 axes.append(coordinates)
134 points.extend(zip(*axes))
135 for child in container['content']:
136 nested = space['nodes'][child]
137 if nested['kind']['type'] == 'Ink':
138 x, y, w, h = ink_extent(space, nested)
139 points.extend([(x, y), (x + w, y + h)])
140 xs, ys = [p[0] for p in points], [p[1] for p in points]
141 return [min(xs), min(ys), max(xs) - min(xs), max(ys) - min(ys)]
142
143
144def compare_objects(space, roots, page, native_roots, assets, native_payloads, autofit):
145 types = {'T': 'RichText', 'Image': 'Image', 'InsertedFile': 'Attachment', 'MediaFile': 'Attachment', 'Table': 'Table'}
146 actual = [n for root in roots for _, n in walk(space, root)
147 if n['kind']['type'] in types.values() and not n['kind'].get('boilerplate')]
148 wanted = [n for root in native_roots for n in root.iter() if n.tag.rsplit('}', 1)[-1] in types]
149 assert [n['kind']['type'] for n in actual] == [types[n.tag.rsplit('}', 1)[-1]] for n in wanted], 'Content object order differs'
150 parents = {child: parent for parent in page.iter() for child in parent}
151 definitions = {n.attrib['index']: n for n in page.findall('one:TagDef', ns)}
152 count = 0
153 for node, native in zip(actual, wanted, strict=True):
154 kind = node['kind']
155 if kind['type'] in ('Image', 'Attachment'):
156 container = space['nodes'][kind['container']]['kind']
157 data = assets[json.dumps(container['reference'], sort_keys=True)]
158 if kind['type'] == 'Image':
159 expected = base64.b64decode(native.find('one:Data', ns).text)
160 if kind['printout'] and data.startswith(b'PK\x03\x04'):
161 # A printout page stores its XPS package; the read shows OneNote's rendering
162 # of it, kept as the picture's WebPictureContainer14.
163 web = next(field['value']['Objects'][0] for group in node['extra']
164 for field in group if field['id'] == 0x200034c8)
165 data = assets[json.dumps(space['nodes'][web]['kind']['reference'], sort_keys=True)]
166 if data != expected:
167 actual_image = Image.open(BytesIO(data)).convert('RGBA')
168 expected_image = Image.open(BytesIO(expected)).convert('RGBA')
169 assert actual_image.size == expected_image.size and actual_image.tobytes() == expected_image.tobytes(), 'Image pixels differ'
170 assert bool(kind['background']) == (native.get('backgroundImage') == 'true'), 'Image background state differs'
171 assert bool(kind['printout']) == (native.get('isPrintOut') == 'true'), 'Image printout state differs'
172 assert kind['alt'] == native.get('alt'), 'Image alternative text differs'
173 assert kind['link'] == native.get('hyperlink'), 'Image hyperlink differs'
174 size = native.find('one:Size', ns)
175 if size is None:
176 assert native.get('format') == 'png' and Image.open(BytesIO(expected)).size == (1, 1), 'Native image size is unavailable'
177 assert node['layout']['max_width'] is None and node['layout']['max_height'] is None, 'Native image omitted an explicit size'
178 for axis in ('width', 'height'):
179 native_size = .75 if size is None else float(size.get(axis))
180 stored_size = node['layout']['max_' + axis]
181 if stored_size is None:
182 stored_size = kind['picture_' + axis]
183 if stored_size is not None:
184 # Native printout XML includes the 0.72-point border on each side.
185 frame_size = stored_size + (1.44 if kind['printout'] else 0)
186 assert abs(frame_size - native_size) < .002, ('Image display dimension differs', axis, frame_size, native_size)
187 else:
188 owner = native if parents[native] is page else parents[native]
189 expected = [p for p in native_payloads if p['page'] == page.get('ID') and p['object'] == owner.get('objectID')]
190 assert len(expected) == 1, 'Native attachment association is ambiguous'
191 assert hashlib.sha256(data).hexdigest() == expected[0]['sha256'], 'Associated attachment bytes differ'
192 assert kind['filename'] == expected[0]['name'], 'Attachment filename differs'
193 media = native.find('one:MediaReference', ns)
194 assert (kind['recording_id'] is None) == (media is None), 'Recording identity presence differs'
195 if media is not None:
196 assert uuid.UUID(bytes_le=bytes(kind['recording_id'])) == uuid.UUID(media.get('mediaID')), 'Recording identity differs'
197 elif kind['type'] == 'Table':
198 rows = native.findall('one:Row', ns)
199 columns = native.findall('one:Columns/one:Column', ns)
200 assert len(rows) == kind['rows'] and len(columns) == kind['columns'], 'Table dimensions differ'
201 assert (kind['locked'] or [False] * len(columns)) == [c.get('isLocked') == 'true' for c in columns], 'Table column lock state differs'
202 for column, width in zip(columns, kind['widths'], strict=True):
203 measured = float(column.get('width'))
204 assert math.isfinite(measured) and measured >= 0, 'Invalid native table column width'
205 if abs(measured - width) >= .002:
206 if column.get('isLocked') == 'true':
207 raise AssertionError('Locked table column width differs')
208 autofit.append({'page': page.get('ID'), 'table': parents[native].get('objectID'),
209 'column': column.get('index'), 'stored': width, 'native': measured})
210 assert len(node['children']) == len(rows), 'Table row count differs'
211 for oid, row in zip(node['children'], rows, strict=True):
212 assert len(space['nodes'][oid]['children']) == len(row.findall('one:Cell', ns)), 'Table cell count differs'
213 tags = native.findall('one:Tag', ns) + parents[native].findall('one:Tag', ns)
214 tasks = native.findall('one:OutlookTask', ns) + parents[native].findall('one:OutlookTask', ns)
215 assert len(node['tags']) == len(tags) + len(tasks), 'Associated note tag count differs'
216 by_type = {int(definitions[t.attrib['index']].attrib['type']): t for t in tags}
217 assert len(by_type) == len(tags), 'Native note tag types are repeated'
218 for tag in node['tags']:
219 if tag['status'] & 4:
220 assert len(tasks) == 1 and tag['definition'] is None, 'Task association differs'
221 task = tasks[0]
222 assert uuid.UUID(bytes_le=bytes(tag['task_id'])) == uuid.UUID(task.get('guidTask')), 'Task identity differs'
223 for mask, name in [(1, 'completed'), (2, 'disabled')]:
224 assert bool(tag['status'] & mask) == (task.get(name) == 'true'), 'Task state differs'
225 for field, name in [('created', 'creationDate'), ('completed', 'completionDate'), ('start', 'startDate'), ('due', 'dueDate')]:
226 if tag[field]:
227 actual = datetime.fromtimestamp(tag[field] + 315532800, timezone.utc)
228 assert actual == datetime.fromisoformat(task.get(name).replace('Z', '+00:00')), 'Task date differs'
229 else:
230 assert task.get(name) is None, 'Unset task date differs'
231 count += 1
232 continue
233 definition = space['nodes'][tag['definition']]['kind']
234 expected = by_type[definition['action_type']]
235 native_definition = definitions[expected.attrib['index']]
236 assert definition['label'] == native_definition.attrib['name'], 'Tag label differs'
237 assert definition['shape'] == int(native_definition.attrib['symbol']), 'Tag shape differs'
238 assert definition['action_type'] == int(native_definition.attrib['type']), 'Tag type differs'
239 assert bool(tag['status'] & 1) == (expected.attrib['completed'] == 'true'), 'Tag completion differs'
240 assert bool(tag['status'] & 2) == (expected.attrib['disabled'] == 'true'), 'Tag disabled state differs'
241 for field, key in [('created', 'creationDate'), ('completed', 'completionDate')]:
242 if field == 'completed' and tag[field] == 0:
243 assert key not in expected.attrib, 'Unset tag completion date was exported'
244 continue
245 if tag[field] is not None:
246 timestamp = datetime.fromtimestamp(tag[field] + 315532800, timezone.utc)
247 assert timestamp == datetime.fromisoformat(expected.attrib[key].replace('Z', '+00:00')), 'Tag date differs'
248 for field, key in [('color', 'fontColor'), ('highlight', 'highlightColor')]:
249 value = definition[field]
250 if value is not None:
251 color = native_definition.attrib[key].lower()
252 assert (value == 0xff000000 and color in ('automatic', 'none')) or color == '#' + ''.join(f'{(value >> (8 * i)) & 255:02x}' for i in range(3)), 'Tag color differs'
253 count += 1
254 indexes = native.findall('one:MediaIndex', ns) + parents[native].findall('one:MediaIndex', ns)
255 assert [uuid.UUID(bytes_le=bytes(value)) for value in node['media_ids']] == [uuid.UUID(value.find('one:MediaReference', ns).get('mediaID')) for value in indexes], 'Recording annotation association differs'
256 for index in indexes:
257 assert node['media_time_ms'] == int(index.get('timeIndex')), 'Recording annotation time differs'
258 return count
259
260
261def exported_links(runs):
262 """Characters and links as OneNote 2010 exports them: an address without a scheme opens
263 over http, and URL text it finds in a page's text is a link, a removed link's included
264 (`corpus/link-edit/native-typed`)."""
265 def href(link):
266 # A file address naming a drive gains the third slash of its URL.
267 if link is not None and re.match(r'file://[A-Za-z]:', link):
268 return 'file:///' + link[len('file://'):]
269 if link is None or ':' in link.split('/')[0] or link.startswith('\\\\'):
270 return link
271 return 'http://' + link
272 out = [(char, href(link)) for char, link in runs]
273 start = 0
274 while start < len(out):
275 end = start
276 while end < len(out) and not out[end][0].isspace():
277 end += 1
278 word = ''.join(char for char, _ in out[start:end])
279 # The scheme stays with the text it links; trailing punctuation does not.
280 for scheme in ('http://', 'https://', 'ftp://', 'mailto:', 'news:'):
281 at = word.lower().find(scheme)
282 if at >= 0 and all(link is None for _, link in out[start:end]):
283 url = word[at:].rstrip('.,;:!?\'")]}')
284 for index in range(start + at, start + at + len(url)):
285 out[index] = (out[index][0], url)
286 break
287 start = end + 1
288 return out
289
290
291def compare(notebook, native, versions=None, password_file=None):
292 notebook = notebook.resolve(strict=True)
293 native = native.resolve(strict=True)
294 sections = sorted(notebook.rglob('*.one'))
295 assert sections, 'No notebook sections were supplied for comparison'
296 captures = {page.attrib['ID']: (page, path.with_suffix('.pdf'))
297 for path, page in zip(sorted(native.glob('page-*.xml')), pages(native), strict=True)}
298 native_payloads = json.loads((native / 'payloads.json').read_text(encoding='utf-8-sig')) if (native / 'payloads.json').exists() else []
299 hierarchy = ET.parse(native / 'hierarchy.xml').getroot()
300 compared = formatting = tags = 0
301 discrepancies = []
302 pdf_checks = []
303 geometry = []
304 autofit = []
305 for path in sections:
306 relative = path.relative_to(notebook)
307 if versions is not None and relative.as_posix() != versions['section']:
308 continue
309 with TemporaryDirectory() as temporary:
310 exported = Path(temporary) / 'document'
311 command = [EXPORTER, path, exported]
312 if password_file is not None:
313 command.extend(['--password-file', password_file])
314 subprocess.run(command, check=True)
315 document = json.loads((exported / 'document.json').read_text())
316 resolved_text = json.loads((exported / 'text.json').read_text())
317 assets = {json.dumps(a['reference'], sort_keys=True): (exported / a['path']).read_bytes()
318 for a in json.loads((exported / 'assets.json').read_text())}
319 candidates = [n for n in hierarchy.iter('{' + ns['one'] + '}Section')
320 if PureWindowsPath(n.attrib['path']).parts[-len(relative.parts):] == relative.parts]
321 assert len(candidates) == 1, (relative, 'native section identity')
322 expected = [captures[n.attrib['ID']][0] for n in candidates[0].findall('one:Page', ns)]
323 actual = list(ordered_pages(document))
324 if versions is not None:
325 assert hashlib.sha256(path.read_bytes()).hexdigest() == versions['source_sha256'], 'Historical source changed'
326 section_pages = {n.get('ID') for n in candidates[0].findall('one:Page', ns)}
327 historical = []
328 expected = []
329 associations = []
330 for row in versions['pages']:
331 assert row['native_id'] in section_pages, 'Copied version belongs to a different section'
332 sid, _, revision, _ = actual[row['source_ordinal']]
333 choices = list(version_pages(document, sid, revision))
334 if row.get('selection') == 'sole-version':
335 assert len(choices) == 1, 'The native sole-version selection has multiple stored candidates'
336 matches = [(context, rid, page, oid) for context, rid, page, oid, _ in choices]
337 else:
338 matches = [(context, rid, page, oid) for context, rid, page, oid, proxy in choices
339 if proxy['kind']['modified_filetime'] is not None
340 and datetime.fromtimestamp(proxy['kind']['modified_filetime'] / 10_000_000 - 11644473600, ZoneInfo(versions['timezone'])).date().isoformat() == row['displayed_date']]
341 assert len(matches) == 1, 'Native version association is ambiguous'
342 context, rid, page, oid = matches[0]
343 historical.append((sid, rid, page, oid))
344 expected.append(captures[row['native_id']][0])
345 associations.append({**row, 'space': sid, 'context': context, 'revision': rid, 'object': oid})
346 available = {(sid, context, oid) for sid, _, revision, _ in actual
347 for context, _, _, oid, _ in version_pages(document, sid, revision)}
348 assert {(row['space'], row['context'], row['object']) for row in associations} == available, 'Native captures do not cover every historical page'
349 assert len(historical) == len(available), 'Historical page is captured more than once'
350 actual = historical
351 assert len(actual) == len(expected), (relative, 'page count', len(actual), len(expected))
352 for ordinal, ((sid, rid, space, oid), page) in enumerate(zip(actual, expected, strict=True)):
353 title_nodes = [n for _, n in walk(space, oid) if n['kind']['type'] == 'RichText'
354 and any(field['id'] == 0x88001cb4 for field in n['extra'][0])]
355 title = title_nodes[0]['kind']['text'].lstrip().split('\r')[0] if len(title_nodes) == 1 else ''
356 if title:
357 assert space['nodes'][space['roots']['2']]['kind']['title'] == title, 'Cached page title differs from visible title text'
358 cached = space['nodes'][space['roots']['2']]['kind']['title'] or ''
359 assert page.get('name') == (cached.replace('\t', ' ') or 'Untitled page'), 'Cached page title differs from native navigation'
360 native_z = {int(n.find('one:Position', ns).attrib['z']) for n in page if n.find('one:Position', ns) is not None}
361 source_page = space['nodes'][oid]
362 assert bool(source_page['kind']['rtl']) == (page.find('one:PageSettings', ns).get('RTL') == 'true'), 'Page direction differs'
363 if source_page['kind']['rtl']:
364 # Storage orders columns visually left to right; native XML follows page direction.
365 for table in page.findall('.//one:Table', ns):
366 columns = table.find('one:Columns', ns)
367 columns[:] = reversed(columns[:])
368 for row in table.findall('one:Row', ns):
369 row[:] = reversed(row[:])
370 assert sum(space['nodes'][child]['kind']['type'] == 'Ink' for child in source_page['children']) == len(page.findall('one:InkDrawing', ns)), 'Opaque ink object count differs'
371 for child in page:
372 position = child.find('one:Position', ns)
373 if position is None:
374 continue
375 z = int(position.attrib['z'])
376 source = space['nodes'][source_page['children'][z]]
377 if source['kind']['type'] == 'Ink':
378 if child.tag == '{%s}InkDrawing' % ns['one']:
379 # A drawing lies in the page's frame as outlines do: its strokes plus the
380 # offset a move gives it, reported from the canonical margin origin.
381 extent = ink_extent(space, source)
382 for index, axis in enumerate(('x', 'y')):
383 origin = source_page['kind']['margin_origin_' + axis]
384 canonical_origin = (-36.0 if source_page['kind']['rtl'] else 36.0) if axis == 'x' else 14.4
385 extent[index] += (source['layout'][axis] or 0.0) + (canonical_origin - origin if origin is not None else 0)
386 size = child.find('one:Size', ns)
387 # Native sizes are one HIMETRIC unit larger than the point extent.
388 reported = [float(position.attrib['x']), float(position.attrib['y']), float(size.get('width')) - 72 / 2540, float(size.get('height')) - 72 / 2540]
389 for index, key in enumerate(('x', 'y', 'width', 'height')):
390 if abs(extent[index] - reported[index]) > 0.01:
391 geometry.append({'file': str(relative), 'page': ordinal, 'z': z, 'axis': key,
392 'stored': extent[index], 'native': reported[index]})
393 continue
394 for axis in ('x', 'y'):
395 if source['layout'][axis] is not None:
396 stored_position, native_position = source['layout'][axis], float(position.attrib[axis])
397 origin = source_page['kind']['margin_origin_' + axis]
398 canonical_origin = (-36.0 if source_page['kind']['rtl'] else 36.0) if axis == 'x' else 14.4
399 projected = stored_position + (canonical_origin - origin if origin is not None else 0)
400 if axis == 'x' and source_page['kind']['rtl'] and source['kind']['type'] == 'Outline':
401 assert not any(p['id'] == 0x14001c84 for group in source['extra'] for p in group), 'Explicit RTL outline alignment requires a native control'
402 # An outline laid out wider than its stored width keeps its right edge.
403 width = source['layout']['max_width']
404 projected -= max(0.0, float(child.find('one:Size', ns).get('width')) - width) if width is not None else 0.0
405 if abs(projected - native_position) > 0.002:
406 geometry.append({'file': str(relative), 'page': ordinal, 'z': z, 'axis': axis,
407 'stored': stored_position, 'origin': origin, 'projected': projected, 'native': native_position})
408 roots = list(source_page['structure'])
409 for z, child in enumerate(source_page['children']):
410 subtree = list(walk(space, child))
411 if z not in native_z:
412 assert all(n['kind']['type'] in ('Outline', 'Paragraph', 'RichText') for _, n in subtree)
413 # Native XML omits otherwise empty outlines containing only ASCII spaces.
414 assert not any(n['kind'].get('text', '').strip(' ') or n['kind'].get('lists') or n['tags'] for _, n in subtree)
415 continue
416 roots.append(child)
417 text_nodes = [n for root in roots for _, n in walk(space, root)
418 if n['kind']['type'] == 'RichText' and not n['kind']['boilerplate']]
419 observed = [visible_text(n, space) for n in text_nodes]
420 native_children = list(page)
421 content = [n for n in native_children if n.find('one:Position', ns) is not None]
422 content.sort(key=lambda n: int(n.find('one:Position', ns).attrib['z']))
423 wanted = [text for n in native_children if n.tag == '{' + ns['one'] + '}Title' for text in texts(n)]
424 wanted.extend(text for n in content for text in texts(n))
425 if observed != wanted:
426 destination = native.parent / 'text-order-diff.json'
427 destination.write_text(json.dumps({'file': str(relative), 'page': ordinal, 'space': sid,
428 'actual': observed, 'native': wanted}, indent=2))
429 raise AssertionError(f'{relative}: page {ordinal}: ordered text differs; inspect {destination}')
430 native_roots = [n for n in native_children if n.tag == '{' + ns['one'] + '}Title'] + content
431 try:
432 tags += compare_objects(space, roots, page, native_roots, assets, native_payloads, autofit)
433 native_runs = native_characters(page, native_roots)
434 text_ids = [oid for root in roots for oid, n in walk(space, root)
435 if n['kind']['type'] == 'RichText' and not n['kind']['boilerplate']]
436 for text_id, expected_runs in zip(text_ids, native_runs, strict=True):
437 # Native XML exports equation runs as MathML, not text (tools/test_math_edit.py compares that).
438 observed_runs = [(char, run['link']) for run in resolved_text[sid][rid][text_id]
439 if not run['format']['hidden'] and not run['format']['math'] for char in run['text']]
440 stored = ''.join(run['text'] for run in resolved_text[sid][rid][text_id])
441 if observed_runs and observed_runs[-1][0] == '\r' and stored.endswith('\r'):
442 observed_runs.pop()
443 observed_runs = [(projected, link) for char, link in observed_runs for projected in project_text(char)]
444 expected_links = [(char, style.get('link')) for char, style in expected_runs]
445 if observed_runs == [('\u00a0', None)] and not expected_links:
446 observed_runs = []
447 observed_runs = exported_links(observed_runs)
448 assert observed_runs == expected_links, 'Resolved text or associated hyperlink differs'
449 count, differences = compare_formats(space, text_nodes, native_runs)
450 formatting += count
451 pdf = captures[page.attrib['ID']][1]
452 if differences and pdf.exists() and all(d['field'] == 'highlight' and d['stored'] == '#000000' and d['native'] == 'automatic' for d in differences):
453 paragraphs = verify_pdf_black(pdf, [resolved_text[sid][rid][i] for i in text_ids])
454 pdf_checks.append({'file': str(relative), 'page': ordinal, 'paragraphs': paragraphs,
455 'pdf': pdf.name, 'pdf_sha256': hashlib.sha256(pdf.read_bytes()).hexdigest(),
456 'source_sha256': hashlib.sha256(path.read_bytes()).hexdigest(),
457 'xml_omissions': differences})
458 differences = []
459 discrepancies.extend({"file": str(relative), "page": ordinal, **d} for d in differences)
460 except AssertionError as error:
461 raise AssertionError((str(relative), ordinal, str(error))) from error
462 compared += 1
463 if versions is not None:
464 assert compared == len(versions['pages']) > 0, 'No historical source section was compared'
465 (native.parent / 'geometry-differences.json').write_text(json.dumps(geometry, indent=2))
466 (native.parent / 'table-autofit.json').write_text(json.dumps(autofit, indent=2))
467 (native.parent / 'pdf-format-checks.json').write_text(json.dumps(pdf_checks, indent=2))
468 destination = native.parent / 'format-differences.json'
469 destination.write_text(json.dumps(discrepancies, indent=2))
470 if geometry:
471 raise AssertionError(f'{len(geometry)} coordinate differences; inspect {native.parent / "geometry-differences.json"}')
472 if discrepancies:
473 raise AssertionError(f'{len(discrepancies)} formatting differences across {compared} pages; inspect {destination}')
474 if versions is not None:
475 (native.parent / 'version-associations.json').write_text(json.dumps(associations, indent=2))
476 print(f'Passed: {compared} pages with native-normalized paragraph text/order; {tags} associated tags; {formatting} explicit character-format comparisons')
477 if pdf_checks:
478 print(f'Native PDF rectangles independently verified black-highlight character positions in {sum(p["paragraphs"] for p in pdf_checks)} paragraphs where XML omitted formatting.')
479
480
481if __name__ == '__main__':
482 parser = argparse.ArgumentParser(description=__doc__)
483 parser.add_argument('notebook', type=Path)
484 parser.add_argument('native', type=Path)
485 parser.add_argument('--versions', type=Path, help='Native history UI date and copied-page associations.')
486 parser.add_argument('--password-file', type=Path, help='Exact UTF-8 password bytes; requires the protected exporter feature.')
487 args = parser.parse_args()
488 compare(args.notebook.resolve(), args.native.resolve(), json.loads(args.versions.read_text()) if args.versions else None, args.password_file)