| 1 | #!/usr/bin/env python3 |
| 2 | """Compare document order and text with independently captured OneNote XML.""" |
| 3 | import argparse |
| 4 | import base64 |
| 5 | import hashlib |
| 6 | from io import BytesIO |
| 7 | from tempfile import TemporaryDirectory |
| 8 | from PIL import Image |
| 9 | from datetime import datetime, timezone |
| 10 | import json |
| 11 | import math |
| 12 | import re |
| 13 | from pathlib import Path, PureWindowsPath |
| 14 | from zoneinfo import ZoneInfo |
| 15 | import subprocess |
| 16 | import xml.etree.ElementTree as ET |
| 17 | import uuid |
| 18 | import unicodedata |
| 19 | from native_xml import ns, pages, texts, project_text |
| 20 | from document_model import EXPORTER, ordered_pages, version_pages, walk |
| 21 | from native_format import native_characters, compare_formats |
| 22 | |
| 23 | ROOT = Path(__file__).resolve().parent.parent |
| 24 | |
| 25 | |
| 26 | def verify_pdf_black(path, paragraphs): |
| 27 | import pdfplumber |
| 28 | characters = [] |
| 29 | with pdfplumber.open(path) as pdf: |
| 30 | for page in pdf.pages: |
| 31 | black = [r for r in page.rects if r.get('fill') and r.get('non_stroking_color') in (0, (0, 0, 0))] |
| 32 | for char in page.chars: |
| 33 | if char['text'].isspace(): |
| 34 | continue |
| 35 | x, y = (char['x0'] + char['x1']) / 2, (char['top'] + char['bottom']) / 2 |
| 36 | covered = any(r['x0'] - .2 <= x <= r['x1'] + .2 and r['top'] - .2 <= y <= r['bottom'] + .2 for r in black) |
| 37 | characters.extend((c, covered, (page.page_number, char, black)) for c in char['text']) |
| 38 | text = ''.join(c for c, _, _ in characters) |
| 39 | checked = 0 |
| 40 | for runs in paragraphs: |
| 41 | source = [(c, run['format']['highlight'] == 0) for run in runs if not run['format']['hidden'] |
| 42 | for c in run['text'] if not c.isspace()] |
| 43 | if not any(black for _, black in source): |
| 44 | continue |
| 45 | value = ''.join(c for c, _ in source) |
| 46 | omitted = [] |
| 47 | if value not in text: |
| 48 | # OneNote can map rendered combining clusters to spaces in PDF text. |
| 49 | projected = [] |
| 50 | index = 0 |
| 51 | while index < len(source): |
| 52 | end = index + 1 |
| 53 | while end < len(source) and unicodedata.combining(source[end][0]): |
| 54 | end += 1 |
| 55 | if end > index + 1: |
| 56 | assert not any(black for _, black in source[index:end]), 'Black-highlighted unmapped text requires visual verification' |
| 57 | omitted.append(len(projected)) |
| 58 | else: |
| 59 | projected.append(source[index]) |
| 60 | index = end |
| 61 | source = projected |
| 62 | value = ''.join(c for c, _ in source) |
| 63 | start = text.find(value) |
| 64 | assert start >= 0 and text.find(value, start + 1) < 0, 'Native PDF paragraph text is missing or ambiguous' |
| 65 | assert [(c, covered) for c, covered, _ in characters[start:start + len(source)]] == source, 'Native PDF black rectangles disagree with the stored character positions' |
| 66 | for index in omitted: |
| 67 | assert 0 < index < len(source), 'Unmapped text has no surrounding PDF characters' |
| 68 | left_page, left, black = characters[start + index - 1][2] |
| 69 | right_page, right, _ = characters[start + index][2] |
| 70 | assert left_page == right_page and abs(left['top'] - right['top']) < .2 and left['x1'] < right['x0'], 'Unmapped text has no unambiguous inline PDF position' |
| 71 | assert not any(r['x0'] < right['x0'] - .2 and r['x1'] > left['x1'] + .2 |
| 72 | and r['top'] < min(left['bottom'], right['bottom']) - .2 |
| 73 | and r['bottom'] > max(left['top'], right['top']) + .2 for r in black), 'Native PDF black rectangles cover unhighlighted unmapped text' |
| 74 | checked += 1 |
| 75 | assert checked, 'No stored black-highlight paragraph was checked' |
| 76 | return checked |
| 77 | |
| 78 | |
| 79 | def visible_text(node, space): |
| 80 | kind = node['kind'] |
| 81 | data = kind['text'].encode('utf-16-le') |
| 82 | # Hidden runs and equation runs (exported as MathML) have no visible native text. |
| 83 | text = ''.join(data[run['start'] * 2:run['end'] * 2].decode('utf-16-le') for run in kind['runs'] |
| 84 | if not run['format'] or not (space['nodes'][run['format']]['format']['hidden'] |
| 85 | or space['nodes'][run['format']]['format']['math'])) |
| 86 | # A trailing CR ends the stored text; one inside it, between equations, is a line break. |
| 87 | if kind['text'].endswith('\r'): |
| 88 | text = text.removesuffix('\r') |
| 89 | return "" if text == "\u00a0" else project_text(text) |
| 90 | |
| 91 | |
| 92 | ISF_X = bytes.fromhex('8f6a8a59c052a04b93afaf357411a561') |
| 93 | ISF_Y = bytes.fromhex('759f3fb5e0049844a7eec30dbb5a9011') |
| 94 | |
| 95 | |
| 96 | def multi_byte(data): |
| 97 | """ISF multi-byte signed integers: a count, then 7-bit little-endian varints with the sign in bit 0.""" |
| 98 | values = [] |
| 99 | index = 0 |
| 100 | while index < len(data): |
| 101 | value = shift = 0 |
| 102 | while True: |
| 103 | byte = data[index] |
| 104 | index += 1 |
| 105 | value |= (byte & 0x7f) << shift |
| 106 | shift += 7 |
| 107 | if not byte & 0x80: |
| 108 | break |
| 109 | values.append(-(value >> 1) if value & 1 else value >> 1) |
| 110 | count, values = values[0], values[1:] |
| 111 | assert count == len(values), 'Ink packet count differs' |
| 112 | return values |
| 113 | |
| 114 | |
| 115 | def ink_extent(space, container): |
| 116 | """Every stroke point of an ink container in points, as [left, top, width, height].""" |
| 117 | points = [] |
| 118 | scale = [container['kind']['scale_x'] or 1.0, container['kind']['scale_y'] or 1.0] |
| 119 | for stroke_id in space['nodes'][container['kind']['data']]['kind']['strokes']: |
| 120 | stroke = space['nodes'][stroke_id]['kind'] |
| 121 | style = space['nodes'][stroke['style']]['kind'] |
| 122 | dimensions = bytes(style['dimensions']) |
| 123 | guids = [dimensions[i:i + 16] for i in range(0, len(dimensions), 32)] |
| 124 | values = multi_byte(bytes(stroke['path'])) |
| 125 | per_dimension = len(values) // len(guids) |
| 126 | axes = [] |
| 127 | for guid, factor in ((ISF_X, scale[0]), (ISF_Y, scale[1])): |
| 128 | start = guids.index(guid) * per_dimension |
| 129 | position, coordinates = 0, [] |
| 130 | for delta in values[start:start + per_dimension]: |
| 131 | position += delta |
| 132 | coordinates.append(position * factor * 72 / 2540) |
| 133 | axes.append(coordinates) |
| 134 | points.extend(zip(*axes)) |
| 135 | for child in container['content']: |
| 136 | nested = space['nodes'][child] |
| 137 | if nested['kind']['type'] == 'Ink': |
| 138 | x, y, w, h = ink_extent(space, nested) |
| 139 | points.extend([(x, y), (x + w, y + h)]) |
| 140 | xs, ys = [p[0] for p in points], [p[1] for p in points] |
| 141 | return [min(xs), min(ys), max(xs) - min(xs), max(ys) - min(ys)] |
| 142 | |
| 143 | |
| 144 | def compare_objects(space, roots, page, native_roots, assets, native_payloads, autofit): |
| 145 | types = {'T': 'RichText', 'Image': 'Image', 'InsertedFile': 'Attachment', 'MediaFile': 'Attachment', 'Table': 'Table'} |
| 146 | actual = [n for root in roots for _, n in walk(space, root) |
| 147 | if n['kind']['type'] in types.values() and not n['kind'].get('boilerplate')] |
| 148 | wanted = [n for root in native_roots for n in root.iter() if n.tag.rsplit('}', 1)[-1] in types] |
| 149 | assert [n['kind']['type'] for n in actual] == [types[n.tag.rsplit('}', 1)[-1]] for n in wanted], 'Content object order differs' |
| 150 | parents = {child: parent for parent in page.iter() for child in parent} |
| 151 | definitions = {n.attrib['index']: n for n in page.findall('one:TagDef', ns)} |
| 152 | count = 0 |
| 153 | for node, native in zip(actual, wanted, strict=True): |
| 154 | kind = node['kind'] |
| 155 | if kind['type'] in ('Image', 'Attachment'): |
| 156 | container = space['nodes'][kind['container']]['kind'] |
| 157 | data = assets[json.dumps(container['reference'], sort_keys=True)] |
| 158 | if kind['type'] == 'Image': |
| 159 | expected = base64.b64decode(native.find('one:Data', ns).text) |
| 160 | if kind['printout'] and data.startswith(b'PK\x03\x04'): |
| 161 | # A printout page stores its XPS package; the read shows OneNote's rendering |
| 162 | # of it, kept as the picture's WebPictureContainer14. |
| 163 | web = next(field['value']['Objects'][0] for group in node['extra'] |
| 164 | for field in group if field['id'] == 0x200034c8) |
| 165 | data = assets[json.dumps(space['nodes'][web]['kind']['reference'], sort_keys=True)] |
| 166 | if data != expected: |
| 167 | actual_image = Image.open(BytesIO(data)).convert('RGBA') |
| 168 | expected_image = Image.open(BytesIO(expected)).convert('RGBA') |
| 169 | assert actual_image.size == expected_image.size and actual_image.tobytes() == expected_image.tobytes(), 'Image pixels differ' |
| 170 | assert bool(kind['background']) == (native.get('backgroundImage') == 'true'), 'Image background state differs' |
| 171 | assert bool(kind['printout']) == (native.get('isPrintOut') == 'true'), 'Image printout state differs' |
| 172 | assert kind['alt'] == native.get('alt'), 'Image alternative text differs' |
| 173 | assert kind['link'] == native.get('hyperlink'), 'Image hyperlink differs' |
| 174 | size = native.find('one:Size', ns) |
| 175 | if size is None: |
| 176 | assert native.get('format') == 'png' and Image.open(BytesIO(expected)).size == (1, 1), 'Native image size is unavailable' |
| 177 | assert node['layout']['max_width'] is None and node['layout']['max_height'] is None, 'Native image omitted an explicit size' |
| 178 | for axis in ('width', 'height'): |
| 179 | native_size = .75 if size is None else float(size.get(axis)) |
| 180 | stored_size = node['layout']['max_' + axis] |
| 181 | if stored_size is None: |
| 182 | stored_size = kind['picture_' + axis] |
| 183 | if stored_size is not None: |
| 184 | # Native printout XML includes the 0.72-point border on each side. |
| 185 | frame_size = stored_size + (1.44 if kind['printout'] else 0) |
| 186 | assert abs(frame_size - native_size) < .002, ('Image display dimension differs', axis, frame_size, native_size) |
| 187 | else: |
| 188 | owner = native if parents[native] is page else parents[native] |
| 189 | expected = [p for p in native_payloads if p['page'] == page.get('ID') and p['object'] == owner.get('objectID')] |
| 190 | assert len(expected) == 1, 'Native attachment association is ambiguous' |
| 191 | assert hashlib.sha256(data).hexdigest() == expected[0]['sha256'], 'Associated attachment bytes differ' |
| 192 | assert kind['filename'] == expected[0]['name'], 'Attachment filename differs' |
| 193 | media = native.find('one:MediaReference', ns) |
| 194 | assert (kind['recording_id'] is None) == (media is None), 'Recording identity presence differs' |
| 195 | if media is not None: |
| 196 | assert uuid.UUID(bytes_le=bytes(kind['recording_id'])) == uuid.UUID(media.get('mediaID')), 'Recording identity differs' |
| 197 | elif kind['type'] == 'Table': |
| 198 | rows = native.findall('one:Row', ns) |
| 199 | columns = native.findall('one:Columns/one:Column', ns) |
| 200 | assert len(rows) == kind['rows'] and len(columns) == kind['columns'], 'Table dimensions differ' |
| 201 | assert (kind['locked'] or [False] * len(columns)) == [c.get('isLocked') == 'true' for c in columns], 'Table column lock state differs' |
| 202 | for column, width in zip(columns, kind['widths'], strict=True): |
| 203 | measured = float(column.get('width')) |
| 204 | assert math.isfinite(measured) and measured >= 0, 'Invalid native table column width' |
| 205 | if abs(measured - width) >= .002: |
| 206 | if column.get('isLocked') == 'true': |
| 207 | raise AssertionError('Locked table column width differs') |
| 208 | autofit.append({'page': page.get('ID'), 'table': parents[native].get('objectID'), |
| 209 | 'column': column.get('index'), 'stored': width, 'native': measured}) |
| 210 | assert len(node['children']) == len(rows), 'Table row count differs' |
| 211 | for oid, row in zip(node['children'], rows, strict=True): |
| 212 | assert len(space['nodes'][oid]['children']) == len(row.findall('one:Cell', ns)), 'Table cell count differs' |
| 213 | tags = native.findall('one:Tag', ns) + parents[native].findall('one:Tag', ns) |
| 214 | tasks = native.findall('one:OutlookTask', ns) + parents[native].findall('one:OutlookTask', ns) |
| 215 | assert len(node['tags']) == len(tags) + len(tasks), 'Associated note tag count differs' |
| 216 | by_type = {int(definitions[t.attrib['index']].attrib['type']): t for t in tags} |
| 217 | assert len(by_type) == len(tags), 'Native note tag types are repeated' |
| 218 | for tag in node['tags']: |
| 219 | if tag['status'] & 4: |
| 220 | assert len(tasks) == 1 and tag['definition'] is None, 'Task association differs' |
| 221 | task = tasks[0] |
| 222 | assert uuid.UUID(bytes_le=bytes(tag['task_id'])) == uuid.UUID(task.get('guidTask')), 'Task identity differs' |
| 223 | for mask, name in [(1, 'completed'), (2, 'disabled')]: |
| 224 | assert bool(tag['status'] & mask) == (task.get(name) == 'true'), 'Task state differs' |
| 225 | for field, name in [('created', 'creationDate'), ('completed', 'completionDate'), ('start', 'startDate'), ('due', 'dueDate')]: |
| 226 | if tag[field]: |
| 227 | actual = datetime.fromtimestamp(tag[field] + 315532800, timezone.utc) |
| 228 | assert actual == datetime.fromisoformat(task.get(name).replace('Z', '+00:00')), 'Task date differs' |
| 229 | else: |
| 230 | assert task.get(name) is None, 'Unset task date differs' |
| 231 | count += 1 |
| 232 | continue |
| 233 | definition = space['nodes'][tag['definition']]['kind'] |
| 234 | expected = by_type[definition['action_type']] |
| 235 | native_definition = definitions[expected.attrib['index']] |
| 236 | assert definition['label'] == native_definition.attrib['name'], 'Tag label differs' |
| 237 | assert definition['shape'] == int(native_definition.attrib['symbol']), 'Tag shape differs' |
| 238 | assert definition['action_type'] == int(native_definition.attrib['type']), 'Tag type differs' |
| 239 | assert bool(tag['status'] & 1) == (expected.attrib['completed'] == 'true'), 'Tag completion differs' |
| 240 | assert bool(tag['status'] & 2) == (expected.attrib['disabled'] == 'true'), 'Tag disabled state differs' |
| 241 | for field, key in [('created', 'creationDate'), ('completed', 'completionDate')]: |
| 242 | if field == 'completed' and tag[field] == 0: |
| 243 | assert key not in expected.attrib, 'Unset tag completion date was exported' |
| 244 | continue |
| 245 | if tag[field] is not None: |
| 246 | timestamp = datetime.fromtimestamp(tag[field] + 315532800, timezone.utc) |
| 247 | assert timestamp == datetime.fromisoformat(expected.attrib[key].replace('Z', '+00:00')), 'Tag date differs' |
| 248 | for field, key in [('color', 'fontColor'), ('highlight', 'highlightColor')]: |
| 249 | value = definition[field] |
| 250 | if value is not None: |
| 251 | color = native_definition.attrib[key].lower() |
| 252 | assert (value == 0xff000000 and color in ('automatic', 'none')) or color == '#' + ''.join(f'{(value >> (8 * i)) & 255:02x}' for i in range(3)), 'Tag color differs' |
| 253 | count += 1 |
| 254 | indexes = native.findall('one:MediaIndex', ns) + parents[native].findall('one:MediaIndex', ns) |
| 255 | assert [uuid.UUID(bytes_le=bytes(value)) for value in node['media_ids']] == [uuid.UUID(value.find('one:MediaReference', ns).get('mediaID')) for value in indexes], 'Recording annotation association differs' |
| 256 | for index in indexes: |
| 257 | assert node['media_time_ms'] == int(index.get('timeIndex')), 'Recording annotation time differs' |
| 258 | return count |
| 259 | |
| 260 | |
| 261 | def exported_links(runs): |
| 262 | """Characters and links as OneNote 2010 exports them: an address without a scheme opens |
| 263 | over http, and URL text it finds in a page's text is a link, a removed link's included |
| 264 | (`corpus/link-edit/native-typed`).""" |
| 265 | def href(link): |
| 266 | # A file address naming a drive gains the third slash of its URL. |
| 267 | if link is not None and re.match(r'file://[A-Za-z]:', link): |
| 268 | return 'file:///' + link[len('file://'):] |
| 269 | if link is None or ':' in link.split('/')[0] or link.startswith('\\\\'): |
| 270 | return link |
| 271 | return 'http://' + link |
| 272 | out = [(char, href(link)) for char, link in runs] |
| 273 | start = 0 |
| 274 | while start < len(out): |
| 275 | end = start |
| 276 | while end < len(out) and not out[end][0].isspace(): |
| 277 | end += 1 |
| 278 | word = ''.join(char for char, _ in out[start:end]) |
| 279 | # The scheme stays with the text it links; trailing punctuation does not. |
| 280 | for scheme in ('http://', 'https://', 'ftp://', 'mailto:', 'news:'): |
| 281 | at = word.lower().find(scheme) |
| 282 | if at >= 0 and all(link is None for _, link in out[start:end]): |
| 283 | url = word[at:].rstrip('.,;:!?\'")]}') |
| 284 | for index in range(start + at, start + at + len(url)): |
| 285 | out[index] = (out[index][0], url) |
| 286 | break |
| 287 | start = end + 1 |
| 288 | return out |
| 289 | |
| 290 | |
| 291 | def compare(notebook, native, versions=None, password_file=None): |
| 292 | notebook = notebook.resolve(strict=True) |
| 293 | native = native.resolve(strict=True) |
| 294 | sections = sorted(notebook.rglob('*.one')) |
| 295 | assert sections, 'No notebook sections were supplied for comparison' |
| 296 | captures = {page.attrib['ID']: (page, path.with_suffix('.pdf')) |
| 297 | for path, page in zip(sorted(native.glob('page-*.xml')), pages(native), strict=True)} |
| 298 | native_payloads = json.loads((native / 'payloads.json').read_text(encoding='utf-8-sig')) if (native / 'payloads.json').exists() else [] |
| 299 | hierarchy = ET.parse(native / 'hierarchy.xml').getroot() |
| 300 | compared = formatting = tags = 0 |
| 301 | discrepancies = [] |
| 302 | pdf_checks = [] |
| 303 | geometry = [] |
| 304 | autofit = [] |
| 305 | for path in sections: |
| 306 | relative = path.relative_to(notebook) |
| 307 | if versions is not None and relative.as_posix() != versions['section']: |
| 308 | continue |
| 309 | with TemporaryDirectory() as temporary: |
| 310 | exported = Path(temporary) / 'document' |
| 311 | command = [EXPORTER, path, exported] |
| 312 | if password_file is not None: |
| 313 | command.extend(['--password-file', password_file]) |
| 314 | subprocess.run(command, check=True) |
| 315 | document = json.loads((exported / 'document.json').read_text()) |
| 316 | resolved_text = json.loads((exported / 'text.json').read_text()) |
| 317 | assets = {json.dumps(a['reference'], sort_keys=True): (exported / a['path']).read_bytes() |
| 318 | for a in json.loads((exported / 'assets.json').read_text())} |
| 319 | candidates = [n for n in hierarchy.iter('{' + ns['one'] + '}Section') |
| 320 | if PureWindowsPath(n.attrib['path']).parts[-len(relative.parts):] == relative.parts] |
| 321 | assert len(candidates) == 1, (relative, 'native section identity') |
| 322 | expected = [captures[n.attrib['ID']][0] for n in candidates[0].findall('one:Page', ns)] |
| 323 | actual = list(ordered_pages(document)) |
| 324 | if versions is not None: |
| 325 | assert hashlib.sha256(path.read_bytes()).hexdigest() == versions['source_sha256'], 'Historical source changed' |
| 326 | section_pages = {n.get('ID') for n in candidates[0].findall('one:Page', ns)} |
| 327 | historical = [] |
| 328 | expected = [] |
| 329 | associations = [] |
| 330 | for row in versions['pages']: |
| 331 | assert row['native_id'] in section_pages, 'Copied version belongs to a different section' |
| 332 | sid, _, revision, _ = actual[row['source_ordinal']] |
| 333 | choices = list(version_pages(document, sid, revision)) |
| 334 | if row.get('selection') == 'sole-version': |
| 335 | assert len(choices) == 1, 'The native sole-version selection has multiple stored candidates' |
| 336 | matches = [(context, rid, page, oid) for context, rid, page, oid, _ in choices] |
| 337 | else: |
| 338 | matches = [(context, rid, page, oid) for context, rid, page, oid, proxy in choices |
| 339 | if proxy['kind']['modified_filetime'] is not None |
| 340 | and datetime.fromtimestamp(proxy['kind']['modified_filetime'] / 10_000_000 - 11644473600, ZoneInfo(versions['timezone'])).date().isoformat() == row['displayed_date']] |
| 341 | assert len(matches) == 1, 'Native version association is ambiguous' |
| 342 | context, rid, page, oid = matches[0] |
| 343 | historical.append((sid, rid, page, oid)) |
| 344 | expected.append(captures[row['native_id']][0]) |
| 345 | associations.append({**row, 'space': sid, 'context': context, 'revision': rid, 'object': oid}) |
| 346 | available = {(sid, context, oid) for sid, _, revision, _ in actual |
| 347 | for context, _, _, oid, _ in version_pages(document, sid, revision)} |
| 348 | assert {(row['space'], row['context'], row['object']) for row in associations} == available, 'Native captures do not cover every historical page' |
| 349 | assert len(historical) == len(available), 'Historical page is captured more than once' |
| 350 | actual = historical |
| 351 | assert len(actual) == len(expected), (relative, 'page count', len(actual), len(expected)) |
| 352 | for ordinal, ((sid, rid, space, oid), page) in enumerate(zip(actual, expected, strict=True)): |
| 353 | title_nodes = [n for _, n in walk(space, oid) if n['kind']['type'] == 'RichText' |
| 354 | and any(field['id'] == 0x88001cb4 for field in n['extra'][0])] |
| 355 | title = title_nodes[0]['kind']['text'].lstrip().split('\r')[0] if len(title_nodes) == 1 else '' |
| 356 | if title: |
| 357 | assert space['nodes'][space['roots']['2']]['kind']['title'] == title, 'Cached page title differs from visible title text' |
| 358 | cached = space['nodes'][space['roots']['2']]['kind']['title'] or '' |
| 359 | assert page.get('name') == (cached.replace('\t', ' ') or 'Untitled page'), 'Cached page title differs from native navigation' |
| 360 | native_z = {int(n.find('one:Position', ns).attrib['z']) for n in page if n.find('one:Position', ns) is not None} |
| 361 | source_page = space['nodes'][oid] |
| 362 | assert bool(source_page['kind']['rtl']) == (page.find('one:PageSettings', ns).get('RTL') == 'true'), 'Page direction differs' |
| 363 | if source_page['kind']['rtl']: |
| 364 | # Storage orders columns visually left to right; native XML follows page direction. |
| 365 | for table in page.findall('.//one:Table', ns): |
| 366 | columns = table.find('one:Columns', ns) |
| 367 | columns[:] = reversed(columns[:]) |
| 368 | for row in table.findall('one:Row', ns): |
| 369 | row[:] = reversed(row[:]) |
| 370 | assert sum(space['nodes'][child]['kind']['type'] == 'Ink' for child in source_page['children']) == len(page.findall('one:InkDrawing', ns)), 'Opaque ink object count differs' |
| 371 | for child in page: |
| 372 | position = child.find('one:Position', ns) |
| 373 | if position is None: |
| 374 | continue |
| 375 | z = int(position.attrib['z']) |
| 376 | source = space['nodes'][source_page['children'][z]] |
| 377 | if source['kind']['type'] == 'Ink': |
| 378 | if child.tag == '{%s}InkDrawing' % ns['one']: |
| 379 | # A drawing lies in the page's frame as outlines do: its strokes plus the |
| 380 | # offset a move gives it, reported from the canonical margin origin. |
| 381 | extent = ink_extent(space, source) |
| 382 | for index, axis in enumerate(('x', 'y')): |
| 383 | origin = source_page['kind']['margin_origin_' + axis] |
| 384 | canonical_origin = (-36.0 if source_page['kind']['rtl'] else 36.0) if axis == 'x' else 14.4 |
| 385 | extent[index] += (source['layout'][axis] or 0.0) + (canonical_origin - origin if origin is not None else 0) |
| 386 | size = child.find('one:Size', ns) |
| 387 | # Native sizes are one HIMETRIC unit larger than the point extent. |
| 388 | reported = [float(position.attrib['x']), float(position.attrib['y']), float(size.get('width')) - 72 / 2540, float(size.get('height')) - 72 / 2540] |
| 389 | for index, key in enumerate(('x', 'y', 'width', 'height')): |
| 390 | if abs(extent[index] - reported[index]) > 0.01: |
| 391 | geometry.append({'file': str(relative), 'page': ordinal, 'z': z, 'axis': key, |
| 392 | 'stored': extent[index], 'native': reported[index]}) |
| 393 | continue |
| 394 | for axis in ('x', 'y'): |
| 395 | if source['layout'][axis] is not None: |
| 396 | stored_position, native_position = source['layout'][axis], float(position.attrib[axis]) |
| 397 | origin = source_page['kind']['margin_origin_' + axis] |
| 398 | canonical_origin = (-36.0 if source_page['kind']['rtl'] else 36.0) if axis == 'x' else 14.4 |
| 399 | projected = stored_position + (canonical_origin - origin if origin is not None else 0) |
| 400 | if axis == 'x' and source_page['kind']['rtl'] and source['kind']['type'] == 'Outline': |
| 401 | assert not any(p['id'] == 0x14001c84 for group in source['extra'] for p in group), 'Explicit RTL outline alignment requires a native control' |
| 402 | # An outline laid out wider than its stored width keeps its right edge. |
| 403 | width = source['layout']['max_width'] |
| 404 | projected -= max(0.0, float(child.find('one:Size', ns).get('width')) - width) if width is not None else 0.0 |
| 405 | if abs(projected - native_position) > 0.002: |
| 406 | geometry.append({'file': str(relative), 'page': ordinal, 'z': z, 'axis': axis, |
| 407 | 'stored': stored_position, 'origin': origin, 'projected': projected, 'native': native_position}) |
| 408 | roots = list(source_page['structure']) |
| 409 | for z, child in enumerate(source_page['children']): |
| 410 | subtree = list(walk(space, child)) |
| 411 | if z not in native_z: |
| 412 | assert all(n['kind']['type'] in ('Outline', 'Paragraph', 'RichText') for _, n in subtree) |
| 413 | # Native XML omits otherwise empty outlines containing only ASCII spaces. |
| 414 | assert not any(n['kind'].get('text', '').strip(' ') or n['kind'].get('lists') or n['tags'] for _, n in subtree) |
| 415 | continue |
| 416 | roots.append(child) |
| 417 | text_nodes = [n for root in roots for _, n in walk(space, root) |
| 418 | if n['kind']['type'] == 'RichText' and not n['kind']['boilerplate']] |
| 419 | observed = [visible_text(n, space) for n in text_nodes] |
| 420 | native_children = list(page) |
| 421 | content = [n for n in native_children if n.find('one:Position', ns) is not None] |
| 422 | content.sort(key=lambda n: int(n.find('one:Position', ns).attrib['z'])) |
| 423 | wanted = [text for n in native_children if n.tag == '{' + ns['one'] + '}Title' for text in texts(n)] |
| 424 | wanted.extend(text for n in content for text in texts(n)) |
| 425 | if observed != wanted: |
| 426 | destination = native.parent / 'text-order-diff.json' |
| 427 | destination.write_text(json.dumps({'file': str(relative), 'page': ordinal, 'space': sid, |
| 428 | 'actual': observed, 'native': wanted}, indent=2)) |
| 429 | raise AssertionError(f'{relative}: page {ordinal}: ordered text differs; inspect {destination}') |
| 430 | native_roots = [n for n in native_children if n.tag == '{' + ns['one'] + '}Title'] + content |
| 431 | try: |
| 432 | tags += compare_objects(space, roots, page, native_roots, assets, native_payloads, autofit) |
| 433 | native_runs = native_characters(page, native_roots) |
| 434 | text_ids = [oid for root in roots for oid, n in walk(space, root) |
| 435 | if n['kind']['type'] == 'RichText' and not n['kind']['boilerplate']] |
| 436 | for text_id, expected_runs in zip(text_ids, native_runs, strict=True): |
| 437 | # Native XML exports equation runs as MathML, not text (tools/test_math_edit.py compares that). |
| 438 | observed_runs = [(char, run['link']) for run in resolved_text[sid][rid][text_id] |
| 439 | if not run['format']['hidden'] and not run['format']['math'] for char in run['text']] |
| 440 | stored = ''.join(run['text'] for run in resolved_text[sid][rid][text_id]) |
| 441 | if observed_runs and observed_runs[-1][0] == '\r' and stored.endswith('\r'): |
| 442 | observed_runs.pop() |
| 443 | observed_runs = [(projected, link) for char, link in observed_runs for projected in project_text(char)] |
| 444 | expected_links = [(char, style.get('link')) for char, style in expected_runs] |
| 445 | if observed_runs == [('\u00a0', None)] and not expected_links: |
| 446 | observed_runs = [] |
| 447 | observed_runs = exported_links(observed_runs) |
| 448 | assert observed_runs == expected_links, 'Resolved text or associated hyperlink differs' |
| 449 | count, differences = compare_formats(space, text_nodes, native_runs) |
| 450 | formatting += count |
| 451 | pdf = captures[page.attrib['ID']][1] |
| 452 | if differences and pdf.exists() and all(d['field'] == 'highlight' and d['stored'] == '#000000' and d['native'] == 'automatic' for d in differences): |
| 453 | paragraphs = verify_pdf_black(pdf, [resolved_text[sid][rid][i] for i in text_ids]) |
| 454 | pdf_checks.append({'file': str(relative), 'page': ordinal, 'paragraphs': paragraphs, |
| 455 | 'pdf': pdf.name, 'pdf_sha256': hashlib.sha256(pdf.read_bytes()).hexdigest(), |
| 456 | 'source_sha256': hashlib.sha256(path.read_bytes()).hexdigest(), |
| 457 | 'xml_omissions': differences}) |
| 458 | differences = [] |
| 459 | discrepancies.extend({"file": str(relative), "page": ordinal, **d} for d in differences) |
| 460 | except AssertionError as error: |
| 461 | raise AssertionError((str(relative), ordinal, str(error))) from error |
| 462 | compared += 1 |
| 463 | if versions is not None: |
| 464 | assert compared == len(versions['pages']) > 0, 'No historical source section was compared' |
| 465 | (native.parent / 'geometry-differences.json').write_text(json.dumps(geometry, indent=2)) |
| 466 | (native.parent / 'table-autofit.json').write_text(json.dumps(autofit, indent=2)) |
| 467 | (native.parent / 'pdf-format-checks.json').write_text(json.dumps(pdf_checks, indent=2)) |
| 468 | destination = native.parent / 'format-differences.json' |
| 469 | destination.write_text(json.dumps(discrepancies, indent=2)) |
| 470 | if geometry: |
| 471 | raise AssertionError(f'{len(geometry)} coordinate differences; inspect {native.parent / "geometry-differences.json"}') |
| 472 | if discrepancies: |
| 473 | raise AssertionError(f'{len(discrepancies)} formatting differences across {compared} pages; inspect {destination}') |
| 474 | if versions is not None: |
| 475 | (native.parent / 'version-associations.json').write_text(json.dumps(associations, indent=2)) |
| 476 | print(f'Passed: {compared} pages with native-normalized paragraph text/order; {tags} associated tags; {formatting} explicit character-format comparisons') |
| 477 | if pdf_checks: |
| 478 | print(f'Native PDF rectangles independently verified black-highlight character positions in {sum(p["paragraphs"] for p in pdf_checks)} paragraphs where XML omitted formatting.') |
| 479 | |
| 480 | |
| 481 | if __name__ == '__main__': |
| 482 | parser = argparse.ArgumentParser(description=__doc__) |
| 483 | parser.add_argument('notebook', type=Path) |
| 484 | parser.add_argument('native', type=Path) |
| 485 | parser.add_argument('--versions', type=Path, help='Native history UI date and copied-page associations.') |
| 486 | parser.add_argument('--password-file', type=Path, help='Exact UTF-8 password bytes; requires the protected exporter feature.') |
| 487 | args = parser.parse_args() |
| 488 | compare(args.notebook.resolve(), args.native.resolve(), json.loads(args.versions.read_text()) if args.versions else None, args.password_file) |