1#!/usr/bin/env python3
2"""Compare current Rust page content against independent native XML; requires Pillow."""
3import base64
4from collections import Counter
5import io
6from pathlib import Path
7import subprocess
8from PIL import Image
9from native_xml import Text, ns, pages, texts
10
11root = Path(__file__).resolve().parent.parent
12private = root / 'corpus/private'
13paths = sorted((private / 'original').rglob('*.one'))
14output = subprocess.check_output(
15 ['cargo', 'run', '-p', 'onestore', '--quiet', '--example', 'inventory', '--', *map(str, paths)],
16 cwd=root, text=True,
17)
18records = [line.split('\t') for line in output.splitlines()]
19regular = {row[1] for row in records if row[0] == 'page' and row[2] == '20030'}
20assert len(regular) == 26
21assert Counter(row[2] for row in records if row[0] == 'page') == {'20030': 26, '2003e': 1}
22content = {osid: Counter() for osid in regular}
23payloads = {osid: [] for osid in regular}
24links = []
25for kind, osid, value in records:
26 if osid not in regular:
27 continue
28 if kind == 'file':
29 payloads[osid].append(bytes.fromhex(value))
30 elif kind in ('ascii', 'unicode'):
31 data = bytes.fromhex(value)
32 text = data.decode('utf-16-le').removesuffix('\0') if kind == 'unicode' else data.decode('latin1')
33 if text.startswith('\ufddfHYPERLINK "'):
34 target, text = text[len('\ufddfHYPERLINK "'):].rsplit('"', 1)
35 links.append(target)
36 if text.strip():
37 content[osid][text] += 1
38
39native = pages(private / 'exact-native/read')
40native_content = [Counter(text for text in texts(page) if text.strip()) for page in native]
41assert Counter(tuple(sorted(c.items())) for c in content.values()) == Counter(tuple(sorted(c.items())) for c in native_content)
42native_links = [target for page in native for node in page.findall('.//one:T', ns) for target in Text(node.text or '').links]
43assert len(links) == 3 and all(target in native_links for target in links)
44
45exact = converted = 0
46for page, expected in zip(native, native_content, strict=True):
47 candidates = [data for osid, text in content.items() if text == expected for data in payloads[osid]]
48 for node in page.findall('.//one:Image/one:Data', ns):
49 data = base64.b64decode(node.text)
50 if data in candidates:
51 exact += 1
52 continue
53 image = Image.open(io.BytesIO(data)).convert('RGBA')
54 matched = False
55 for candidate in candidates:
56 if not candidate.startswith((b'GIF87a', b'GIF89a')):
57 continue
58 gif = Image.open(io.BytesIO(candidate)).convert('RGBA')
59 if gif.size == image.size and gif.tobytes() == image.tobytes():
60 matched = True
61 break
62 assert matched, 'Native image differs from its stored payload'
63 converted += 1
64
65assert exact == 18 and converted == 3
66assert sum(sum(c.values()) for c in content.values()) == 581
67print('Passed: 26 pages, 581 nonempty text objects by page, 3 hyperlink targets, 18 exact images, 3 native GIF conversions')