| 1 | #!/usr/bin/env python3 |
| 2 | """Compare current Rust page content against independent native XML; requires Pillow.""" |
| 3 | import base64 |
| 4 | from collections import Counter |
| 5 | import io |
| 6 | from pathlib import Path |
| 7 | import subprocess |
| 8 | from PIL import Image |
| 9 | from native_xml import Text, ns, pages, texts |
| 10 | |
| 11 | root = Path(__file__).resolve().parent.parent |
| 12 | private = root / 'corpus/private' |
| 13 | paths = sorted((private / 'original').rglob('*.one')) |
| 14 | output = subprocess.check_output( |
| 15 | ['cargo', 'run', '-p', 'onestore', '--quiet', '--example', 'inventory', '--', *map(str, paths)], |
| 16 | cwd=root, text=True, |
| 17 | ) |
| 18 | records = [line.split('\t') for line in output.splitlines()] |
| 19 | regular = {row[1] for row in records if row[0] == 'page' and row[2] == '20030'} |
| 20 | assert len(regular) == 26 |
| 21 | assert Counter(row[2] for row in records if row[0] == 'page') == {'20030': 26, '2003e': 1} |
| 22 | content = {osid: Counter() for osid in regular} |
| 23 | payloads = {osid: [] for osid in regular} |
| 24 | links = [] |
| 25 | for kind, osid, value in records: |
| 26 | if osid not in regular: |
| 27 | continue |
| 28 | if kind == 'file': |
| 29 | payloads[osid].append(bytes.fromhex(value)) |
| 30 | elif kind in ('ascii', 'unicode'): |
| 31 | data = bytes.fromhex(value) |
| 32 | text = data.decode('utf-16-le').removesuffix('\0') if kind == 'unicode' else data.decode('latin1') |
| 33 | if text.startswith('\ufddfHYPERLINK "'): |
| 34 | target, text = text[len('\ufddfHYPERLINK "'):].rsplit('"', 1) |
| 35 | links.append(target) |
| 36 | if text.strip(): |
| 37 | content[osid][text] += 1 |
| 38 | |
| 39 | native = pages(private / 'exact-native/read') |
| 40 | native_content = [Counter(text for text in texts(page) if text.strip()) for page in native] |
| 41 | assert Counter(tuple(sorted(c.items())) for c in content.values()) == Counter(tuple(sorted(c.items())) for c in native_content) |
| 42 | native_links = [target for page in native for node in page.findall('.//one:T', ns) for target in Text(node.text or '').links] |
| 43 | assert len(links) == 3 and all(target in native_links for target in links) |
| 44 | |
| 45 | exact = converted = 0 |
| 46 | for page, expected in zip(native, native_content, strict=True): |
| 47 | candidates = [data for osid, text in content.items() if text == expected for data in payloads[osid]] |
| 48 | for node in page.findall('.//one:Image/one:Data', ns): |
| 49 | data = base64.b64decode(node.text) |
| 50 | if data in candidates: |
| 51 | exact += 1 |
| 52 | continue |
| 53 | image = Image.open(io.BytesIO(data)).convert('RGBA') |
| 54 | matched = False |
| 55 | for candidate in candidates: |
| 56 | if not candidate.startswith((b'GIF87a', b'GIF89a')): |
| 57 | continue |
| 58 | gif = Image.open(io.BytesIO(candidate)).convert('RGBA') |
| 59 | if gif.size == image.size and gif.tobytes() == image.tobytes(): |
| 60 | matched = True |
| 61 | break |
| 62 | assert matched, 'Native image differs from its stored payload' |
| 63 | converted += 1 |
| 64 | |
| 65 | assert exact == 18 and converted == 3 |
| 66 | assert sum(sum(c.values()) for c in content.values()) == 581 |
| 67 | print('Passed: 26 pages, 581 nonempty text objects by page, 3 hyperlink targets, 18 exact images, 3 native GIF conversions') |