1#!/usr/bin/env python3
2"""Export a copied notebook to an offline, read-only document report."""
3import argparse
4from collections import Counter
5from datetime import datetime, timedelta, timezone
6import hashlib
7from html import escape as esc
8import json
9import os
10from io import BytesIO
11from pathlib import Path, PureWindowsPath
12from PIL import Image
13import xml.etree.ElementTree as ET
14from native_xml import ns
15import subprocess
16from urllib.parse import urlsplit
17from zoneinfo import ZoneInfo
18
19from document_model import BRIDGE, DEFAULT_CONTEXT, EXPORTER, ordered_pages, version_pages, view, walk
20
21ROOT = Path(__file__).resolve().parent.parent
22
23
24def hyperlink(fragment, target):
25 if urlsplit(target).scheme.lower() in ('http', 'https', 'mailto', 'onenote'):
26 return '<a href="' + esc(target) + '" rel="noreferrer">' + fragment + '</a>'
27 return fragment + ' <span class="meta">(' + esc(target) + ')</span>'
28
29
30def color(value):
31 if value is None or value >> 24:
32 return None
33 return '#' + ''.join(f'{value >> shift & 255:02x}' for shift in (0, 8, 16))
34
35
36def css(fmt):
37 rules = []
38 for field, prop, yes, no in [('bold', 'font-weight', 'bold', 'normal'),
39 ('italic', 'font-style', 'italic', 'normal')]:
40 if fmt.get(field) is not None:
41 rules.append(f'{prop}:{yes if fmt[field] else no}')
42 decorations = [v for k, v in [('underline', 'underline'), ('strike', 'line-through')] if fmt.get(k)]
43 if decorations:
44 rules.append('text-decoration:' + ' '.join(decorations))
45 for field, prop in [('font', 'font-family'), ('font_size', 'font-size'),
46 ('color', 'color'), ('highlight', 'background-color')]:
47 value = fmt.get(field)
48 if value is not None:
49 if field == 'font':
50 fallback = 'monospace' if value in ('Consolas', 'Courier New', 'Lucida Console') else 'serif' if value in ('Times New Roman', 'Cambria', 'Georgia') else 'sans-serif'
51 value = json.dumps(value) + ',' + fallback
52 else:
53 value = f'{value:g}pt' if field == 'font_size' else color(value)
54 if value:
55 rules.append(f'{prop}:{value}')
56 for field, prop in [('space_before', 'margin-top'), ('space_after', 'margin-bottom')]:
57 if fmt.get(field) is not None:
58 rules.append(f'{prop}:{fmt[field]:g}pt')
59 if fmt.get('alignment') is not None:
60 rules.append('text-align:' + {0: 'left', 1: 'center', 2: 'right'}.get(fmt['alignment'], 'start'))
61 if fmt.get('line_spacing'):
62 rules.append(f'line-height:max(1.5em,{fmt["line_spacing"]:g}pt)')
63 if fmt.get('rtl'):
64 rules.append('direction:rtl')
65 return ';'.join(rules)
66
67
68def html_page(title, nav, body, editable=False):
69 policy = "default-src 'none'; img-src 'self' data:; style-src 'self' 'unsafe-inline'; media-src 'self'; base-uri 'none'"
70 if editable:
71 policy += "; script-src 'self'; connect-src 'self'"
72 return '''<!doctype html><html lang="en"><meta charset="utf-8">
73<meta name="viewport" content="width=device-width,initial-scale=1">
74<meta http-equiv="Content-Security-Policy" content="''' + policy + '''">
75<title>''' + esc(title) + '''</title><style>
76*{box-sizing:border-box}body{margin:0;color:#000;background:#fff;font:15px/1.5 system-ui,sans-serif}a{color:#175bb2}nav{position:fixed;inset:0 auto 0 0;width:240px;overflow:auto;background:#f5f6f7;padding:20px}nav a{display:block;padding:3px 0;overflow-wrap:anywhere}nav h2{font-size:14px;margin:20px 0 4px}main{margin-left:240px;padding:30px 40px;max-width:1250px}h1{font-size:28px;line-height:1.2}h2{font-size:18px}p{margin:6px 0}.outline{margin:24px 0;border-top:1px solid #d9dde2;padding-top:12px}.location,.meta{font:12px/1.5 system-ui;color:#666;margin:6px 0}.paragraph{min-height:1.3em;position:relative;overflow-wrap:anywhere}.nested{margin-left:24px}.text{white-space:pre-wrap}.tag{display:inline-block;font:12px system-ui;padding:2px 5px;border:1px solid #aaa;border-radius:3px;margin-right:5px}.list-marker{display:inline-block;min-width:22px;margin-left:-22px}.listed{margin-left:22px}img.content{max-width:100%;height:auto;vertical-align:top}figure{margin:12px 0}figcaption{font-size:12px;color:#666}table{border-collapse:collapse;margin:8px 0;max-width:100%}td{border:1px solid #bbb;padding:5px 8px;vertical-align:top;min-width:30px}table.no-borders td{border-color:transparent}.opaque{border:1px dashed #9aa2ad;padding:12px;margin:12px 0;background:#f7f8fa}details{margin:14px 0}summary{cursor:pointer;font:13px system-ui}pre{white-space:pre-wrap;overflow-wrap:anywhere;font:11px/1.4 ui-monospace,monospace;max-height:560px;overflow:auto}.page-link{padding-left:12px}code{font-size:12px}.references a{margin-right:14px}@media(max-width:750px){nav{position:static;width:auto;max-height:220px}main{margin:0;padding:20px}}@media print{nav{display:none}main{margin:0}details{display:none}}
77</style>''' + ('<link rel="stylesheet" href="/editor.css"><script src="/editor.js" defer></script>' if editable else '') + '<nav>' + nav + '</nav><main>' + body + '</main></html>'
78
79
80class Page:
81 def __init__(self, section, sid, rid):
82 self.section = section
83 self.space = section['document']['spaces'][sid]['revisions'][rid]
84 self.text = section['text'][sid][rid]
85 self.counts = Counter()
86 self.numbering = {}
87 self.has_paragraph = False
88 self.active = set()
89
90 def asset(self, container):
91 if container is None:
92 return None
93 node = self.space['nodes'][container]
94 if node['kind']['type'] != 'File':
95 raise ValueError('The asset container is not a file object.')
96 return self.section['assets'].get(json.dumps(node['kind']['reference'], sort_keys=True))
97
98 def render(self, oid, depth=0, level=0):
99 if oid in self.active or depth > 256:
100 raise ValueError('The content graph is cyclic or exceeds the report depth limit.')
101 self.active.add(oid)
102 node = self.space['nodes'][oid]
103 kind = node['kind']; typ = kind['type']
104 self.counts[typ] += 1
105 children = lambda refs: ''.join(self.render(x, depth + 1, level) for x in refs)
106 tags = ''
107 tag_format = {}
108 for tag in node['tags']:
109 definition = self.space['nodes'][tag['definition']]['kind'] if tag['definition'] else {}
110 label = definition.get('label') or 'Task'
111 if not tag['status'] & 2:
112 for field in ('color', 'highlight'):
113 if color(definition.get(field)):
114 tag_format.setdefault(field, definition[field])
115 if tag['due']:
116 label += ' · Due ' + datetime.fromtimestamp(tag['due'] + 315532800, timezone.utc).strftime('%Y-%m-%d')
117 if tag['status'] & 2:
118 label += ' · Disabled'
119 tags += '<span class="tag" title="' + esc(json.dumps(tag)) + '">' + (('☑ ' if tag['status'] & 1 else '☐ ') if definition.get('shape') == 3 or tag['status'] & 4 else '') + esc(label) + '</span>'
120 if typ == 'RichText':
121 body = ''
122 structured_math = any(c in kind['text'] for c in '\ufdd0\ufdee\ufdef')
123 equation = False
124 for run_index, run in enumerate(self.text[oid]):
125 if run['format']['hidden']:
126 continue
127 if structured_math and run['format'].get('math'):
128 if not equation:
129 body += '<span class="meta">[Equation · see native reference]</span>'
130 equation = True
131 continue
132 equation = False
133 style = {**run['format'], **tag_format}
134 if (style['superscript'] or style['subscript']) and style['font_size'] is not None:
135 style['font_size'] *= 2 / 3
136 fragment = '<span data-text-object="' + esc(oid) + '" data-run="' + str(run_index) + '" style="' + esc(css(style)) + '">' + esc(run['text']) + '</span>'
137 if run['format']['superscript']:
138 fragment = '<sup>' + fragment + '</sup>'
139 if run['format']['subscript']:
140 fragment = '<sub>' + fragment + '</sub>'
141 if run['link']:
142 fragment = hyperlink(fragment, run['link'])
143 body += fragment
144 body = tags + '<span class="text">' + body + '</span>'
145 elif typ == 'Paragraph':
146 body = children(node['content'])
147 marker = ''
148 for list_id in kind['lists']:
149 item = self.space['nodes'][list_id]['kind']
150 fmt = item['format'] or ''
151 if '\ufffd' in fmt:
152 position = fmt.index('\ufffd'); style = ord(fmt[position + 1])
153 key = (fmt, level)
154 value = item['restart'] if item['restart'] is not None else self.numbering.get(key, 0) + 1
155 self.numbering[key] = value
156 number = str(value)
157 if style in (1, 2):
158 number = ''; rest = value
159 for quantity, letters in [(1000, 'M'), (900, 'CM'), (500, 'D'), (400, 'CD'), (100, 'C'), (90, 'XC'), (50, 'L'), (40, 'XL'), (10, 'X'), (9, 'IX'), (5, 'V'), (4, 'IV'), (1, 'I')]:
160 times, rest = divmod(rest, quantity)
161 number += letters * times
162 if style == 2: number = number.lower()
163 elif style in (3, 4):
164 number = ''; rest = value
165 while rest:
166 rest, digit = divmod(rest - 1, 26)
167 number = chr(65 + digit) + number
168 if style == 4: number = number.lower()
169 elif style != 0:
170 raise ValueError(f'Number format {style} requires interpretation before report generation.')
171 marker += fmt[:position] + number + fmt[position + 2:]
172 else:
173 marker += {1: '•', 2: '◦', 3: '●', 4: '○', 5: '◉', 6: '◎', 7: '▪', 8: '▫', 9: '■', 10: '□', 11: '▸', 12: '▶', 13: '◇', 14: '♢', 15: '◆', 16: '❖', 17: '★', 18: '☆', 19: '☀', 20: '>', 21: '→', 22: '⇒', 23: '⇨', 24: '*', 25: '-', 26: '–', 27: '—'}.get(item['bullet'], fmt)
174 if marker:
175 body = '<span class="list-marker">' + esc(marker) + '</span>' + body
176 fmt = node['format']
177 if kind['paragraph_style']:
178 parent = self.space['nodes'][kind['paragraph_style']]['format']
179 fmt = {k: v if v is not None else parent[k] for k, v in fmt.items()}
180 for child in node['content']:
181 if child in self.text and self.text[child]:
182 resolved = self.text[child][0]['format']
183 fmt = {**fmt, **{k: resolved[k] for k in ('alignment', 'rtl', 'space_before', 'space_after', 'line_spacing') if resolved[k] is not None}}
184 break
185 if not self.has_paragraph:
186 fmt = {**fmt, 'space_before': None}
187 self.has_paragraph = True
188 body = '<div class="paragraph' + (' listed' if marker else '') + '" style="' + esc(css(fmt)) + '">' + tags + body + '</div>'
189 if node['children']:
190 child_level = node['child_level'] or 1
191 descendants = '<div class="nested" style="margin-left:' + str(child_level * 24) + 'px">' + ''.join(self.render(x, depth + 1, level + child_level) for x in node['children']) + '</div>'
192 state = kind.get('collapse_state')
193 if state == 1:
194 descendants = '<details><summary>Collapsed paragraphs</summary>' + descendants + '</details>'
195 elif state not in (None, 0):
196 descendants = '<div class="opaque">Uninterpreted collapse state: ' + str(state) + '</div>' + descendants
197 body += descendants
198 elif typ in ('Page', 'Title', 'Outline'):
199 if typ == 'Outline':
200 previous = self.numbering, self.has_paragraph
201 self.numbering, self.has_paragraph = {}, False
202 body = children(node['structure'] + node['content'] + node['children'])
203 if typ == 'Outline':
204 self.numbering, self.has_paragraph = previous
205 if typ == 'Outline' and node['layout']['x'] is not None:
206 values = ', '.join(f'{k} {v:g} pt' for k, v in node['layout'].items() if v is not None)
207 body = '<section class="outline"><div class="location">' + esc(values) + '</div>' + body + '</section>'
208 elif typ == 'Title':
209 body = '<header>' + body + '</header>'
210 elif typ == 'OutlineGroup':
211 child_level = node['child_level'] or 1
212 body = '<div class="nested" style="margin-left:' + str(child_level * 24) + 'px">' + ''.join(self.render(x, depth + 1, level + child_level) for x in node['children']) + '</div>'
213 elif typ == 'Table':
214 cols = ''.join(f'<col style="width:{width:g}pt">' if kind['locked'] and kind['locked'][i] else '<col>' for i, width in enumerate(kind['widths']))
215 body = '<table class="' + ('no-borders' if kind['borders'] is False else '') + '"><colgroup>' + cols + '</colgroup>' + children(node['children']) + '</table>'
216 elif typ == 'Row':
217 body = '<tr>' + children(node['children']) + '</tr>'
218 elif typ == 'Cell':
219 shade = color(kind['shading'])
220 body = '<td' + (' style="background:' + shade + '"' if shade else '') + '>' + children(node['children']) + '</td>'
221 elif typ == 'Image':
222 asset = self.asset(kind['container'])
223 if asset:
224 width = node['layout']['max_width'] if node['layout']['max_width'] is not None else kind['picture_width']
225 height = node['layout']['max_height'] if node['layout']['max_height'] is not None else kind['picture_height']
226 style = f'width:{width:g}pt' if width is not None else ''
227 if width and height:
228 style += f';aspect-ratio:{width:g}/{height:g}'
229 preview = self.section['previews'].get(asset, asset)
230 if preview.endswith('.bin'):
231 content = '<span class="opaque">Image format is uninterpreted; the original payload is retained in the asset references.</span>'
232 else:
233 content = '<img class="content" src="' + esc(preview) + '" style="' + style + '" alt="' + esc(kind['alt'] or '') + '">'
234 body = '<figure>' + tags + (hyperlink(content, kind['link']) if kind['link'] else content)
235 labels = [kind['filename']] if kind['filename'] else []
236 if kind['background']: labels.append('Background image')
237 if kind['printout']: labels.append('Printout image')
238 if labels:
239 body += '<figcaption>' + esc(' · '.join(labels)) + '</figcaption>'
240 body += '</figure>'
241 if kind['background'] and not kind['printout']:
242 body = '<details><summary>Background image' + (' · ' + esc(kind['filename']) if kind['filename'] else '') + '</summary>' + body + '</details>'
243 else:
244 body = '<div class="opaque">Image payload is external; its source reference is retained.</div>'
245 elif typ == 'Attachment':
246 asset = self.asset(kind['container'])
247 label = kind['filename'] or 'Attached file'
248 body = tags + ('<a download="' + esc(label) + '" href="' + esc(asset) + '">' + esc(label) + '</a>' if asset else esc(label) + ' · External payload')
249 elif typ == 'Ink':
250 body = '<div class="opaque">Ink drawing · Stroke data is retained in the document structure.</div>'
251 else:
252 body = '<div class="opaque">' + ('Encrypted content' if typ == 'Encrypted' else f'Uninterpreted content · JCID {node["jcid"]:#x}') + '<br><code>' + esc(oid) + '</code>' + children(node['structure'] + node['content'] + node['children']) + '</div>'
253 for recording in node['media_ids']:
254 matches = [n['kind'] for n in self.space['nodes'].values()
255 if n['kind']['type'] == 'Attachment' and n['kind']['recording_id'] == recording]
256 label = 'Recording reference'
257 asset = None
258 if len(matches) == 1:
259 label = matches[0]['filename'] or label
260 asset = self.asset(matches[0]['container'])
261 if node['media_time_ms'] is not None:
262 seconds = node['media_time_ms'] / 1000
263 label += f' · {seconds:g} s'
264 body += '<div class="meta">' + ('<a href="' + esc(asset) + '" download>' + esc(label) + '</a>' if asset else esc(label)) + '</div>'
265 self.active.remove(oid)
266 return '<div data-object="' + esc(oid) + '">' + body + '</div>' if typ not in ('RichText', 'Row', 'Cell') else body
267
268
269def generate(source, destination, native=None, versions=(), zone=timezone.utc, editable=False, previous=None):
270 source = source.resolve(strict=True)
271 if destination.resolve().is_relative_to(source):
272 raise ValueError('Choose an export directory outside the source notebook.')
273 destination.mkdir(parents=True, exist_ok=False)
274 (destination / 'model').mkdir(); (destination / 'assets').mkdir()
275 cached = {row['path']: (row['sha256'], previous / 'model' / str(index))
276 for index, row in enumerate(json.loads((previous / 'source.json').read_text()))} if previous else {}
277 result = json.loads(subprocess.check_output([BRIDGE, 'catalog', source], timeout=120))
278 if not result['ok']: raise ValueError(result['error'])
279 catalog = result['catalog']
280 (destination / 'catalog.json').write_text(json.dumps(catalog, indent=2, ensure_ascii=False))
281 files = []; catalog_sections = {}
282 def collect(folder):
283 if folder['toc'] is not None: files.append(Path(folder['path']) / folder['toc']['filename'])
284 for section in folder['sections']:
285 relative = Path(section['path'])
286 files.append(relative); catalog_sections[relative] = section
287 for group in folder['groups']: collect(group)
288 collect(catalog)
289 sections = []; unavailable = []; manifest = []
290 for index, relative in enumerate(sorted(files)):
291 path = source / relative
292 before = path.read_bytes()
293 manifest.append({'path': relative.as_posix(), 'sha256': hashlib.sha256(before).hexdigest(), 'bytes': len(before)})
294 state = catalog_sections.get(relative, {}).get('state', {})
295 if isinstance(state, dict) and 'Unreadable' in state:
296 unavailable.append((relative, state['Unreadable']))
297 continue
298 exported = destination / 'model' / str(index)
299 reusable = cached.get(relative.as_posix())
300 reused = reusable is not None and reusable[0] == manifest[-1]['sha256']
301 if reused and any('External' in asset['reference'] for asset in json.loads((reusable[1] / 'assets.json').read_text())):
302 reused = False
303 if reused:
304 exported.mkdir()
305 for name in ('document.json', 'text.json', 'assets.json'):
306 os.link(reusable[1] / name, exported / name)
307 else:
308 subprocess.run([EXPORTER, path, exported], check=True)
309 document = json.loads((exported / 'document.json').read_text())
310 if path.read_bytes() != before:
311 raise ValueError('A source file changed during export.')
312 assets = {}
313 previews = {}
314 image_references = {
315 json.dumps(revision['nodes'][node['kind']['container']]['kind']['reference'], sort_keys=True)
316 for space in document['spaces'].values() for revision in space['revisions'].values()
317 for node in revision['nodes'].values()
318 if node['kind']['type'] == 'Image' and node['kind']['container'] is not None
319 }
320 rows = json.loads((exported / 'assets.json').read_text())
321 for asset in rows:
322 if asset['path'] is None: continue
323 reference = json.dumps(asset['reference'], sort_keys=True)
324 if reused:
325 name = Path(asset['path']).name
326 assets[reference] = 'assets/' + name
327 if not (destination / 'assets' / name).exists():
328 os.link(previous / 'assets' / name, destination / 'assets' / name)
329 preview = name + '.png'
330 if name.endswith('.tiff') and reference in image_references:
331 if not (destination / 'assets' / preview).exists():
332 os.link(previous / 'assets' / preview, destination / 'assets' / preview)
333 previews['assets/' + name] = 'assets/' + preview
334 continue
335 original = exported / asset['path']; data = original.read_bytes()
336 extension = '.png' if data.startswith(b'\x89PNG') else '.jpg' if data.startswith(b'\xff\xd8') else '.gif' if data.startswith(b'GIF8') else '.bmp' if data.startswith(b'BM') else '.tiff' if data.startswith((b'II*\0', b'MM\0*')) else '.bin'
337 name = hashlib.sha256(data).hexdigest() + extension
338 target = destination / 'assets' / name
339 if target.exists():
340 original.unlink()
341 else:
342 original.rename(target)
343 asset['path'] = '../../assets/' + name
344 assets[reference] = 'assets/' + name
345 if extension == '.tiff' and reference in image_references:
346 preview = name + '.png'
347 with Image.open(BytesIO(data)) as image:
348 if image.n_frames != 1:
349 raise ValueError('Multipage TIFF requires frame interpretation before report generation.')
350 image.convert('RGBA').save(destination / 'assets' / preview)
351 previews['assets/' + name] = 'assets/' + preview
352 if not reused:
353 (exported / 'assets').rmdir()
354 (exported / 'assets.json').write_text(json.dumps(rows, indent=2))
355 if path.suffix.lower() == '.one':
356 sections.append({'path': relative, 'export': exported.relative_to(destination), 'document': document,
357 'text': json.loads((exported / 'text.json').read_text()), 'assets': assets, 'previews': previews})
358 order = {path: index for index, path in enumerate(catalog_sections)}
359 sections.sort(key=lambda section: order[section['path']])
360 pages = []; locked_sections = []; histories = {}; nav = '<a href="index.html">Notebook review</a>'
361 for section in sections:
362 state = catalog_sections[section['path']]['state']
363 name = state.get('Readable', {}).get('name') if isinstance(state, dict) else None
364 if name is None: name = section['path'].stem
365 nav += '<h2>' + esc(str(section['path'].parent) + ' / ' + name) + '</h2>'
366 _, root_space = view(section['document'], section['document']['root'])
367 if root_space['nodes'][root_space['roots']['1']]['kind']['type'] == 'Encrypted':
368 locked_sections.append(str(section['path']))
369 nav += '<p>Locked section · Page count unavailable</p><a href="' + section['export'].as_posix() + '/document.json">Encrypted structure</a>'
370 continue
371 ordinary = list(ordered_pages(section['document']))
372 known = {(sid, rid, oid) for sid, rid, _, oid in ordinary}
373 additional = [(sid, rid, revision, oid) for sid in section['document']['spaces']
374 for rid, revision in [view(section['document'], sid)]
375 for oid, node in revision['nodes'].items() if node['kind']['type'] == 'Page' and (sid, rid, oid) not in known]
376 historical = []
377 for sid, _, current, parent_oid in ordinary + additional:
378 for context, rid, revision, oid, proxy in version_pages(section['document'], sid, current):
379 historical.append((sid, context, rid, revision, oid))
380 modified = proxy['kind']['modified_filetime']
381 histories[(section['path'], sid, context, oid)] = {
382 'modified': (datetime(1601, 1, 1, tzinfo=timezone.utc) + timedelta(microseconds=modified // 10)).astimezone(zone).isoformat() if modified is not None else None,
383 'source': (section['path'], sid, DEFAULT_CONTEXT, parent_oid),
384 }
385 if not ordinary and not additional:
386 nav += '<p>Empty section</p>'
387 current_pages = [(sid, DEFAULT_CONTEXT, rid, revision, oid) for sid, rid, revision, oid in ordinary + additional]
388 for ordinal, (sid, context, rid, space, oid) in enumerate(current_pages + historical):
389 category = 'Historical version' if ordinal >= len(ordinary) + len(additional) else 'Additional stored page' if ordinal >= len(ordinary) else 'Recycle bin' if 'OneNote_RecycleBin' in section['path'].parts else 'Page'
390 metadata = space['nodes'].get(space['roots'].get('2'), {}).get('kind', {})
391 if sid == root_space['nodes'][root_space['roots']['1']]['kind'].get('default_template'):
392 if metadata.get('type') != 'TemplateMetadata':
393 raise ValueError('The default page template has unrecognized metadata.')
394 category = 'Default page template'
395 titles = [n for root in space['nodes'][oid]['structure'] for _, n in walk(space, root) if n['kind']['type'] == 'RichText' and not n['kind']['boilerplate']]
396 title = ' '.join(n['kind']['text'] for n in titles) or space['nodes'][oid]['kind']['alternate_title'] or metadata.get('title') or metadata.get('name') or 'Untitled page'
397 filename = f'page-{len(pages):03}.html'
398 if metadata.get('type') == 'ConflictMetadata':
399 category = 'Conflict page'
400 level = metadata.get('level') or 1
401 label = category + ' · ' + title if category in ('Additional stored page', 'Default page template', 'Conflict page', 'Historical version') else title
402 version = histories.get((section['path'], sid, context, oid))
403 if version and version['modified']:
404 label += ' · ' + version['modified'][:10]
405 nav += '<a class="page-link" style="margin-left:' + str((level - 1) * 12) + 'px" href="' + filename + '">' + esc(label) + '</a>'
406 pages.append((section, ordinal, sid, context, rid, oid, title, filename, category))
407 references = {}
408 if native is not None:
409 native = native.resolve(strict=True)
410 captured = native.parent / 'notebook'
411 if captured.exists():
412 hashes = {p.relative_to(captured).as_posix(): hashlib.sha256(p.read_bytes()).hexdigest() for p in captured.rglob('*') if p.is_file()}
413 else:
414 hashes = {item['path']: item['sha256'] for item in json.loads((native.parent / 'source.json').read_text())}
415 if any(hashes.get(item['path']) != item['sha256'] for item in manifest):
416 raise ValueError('The native capture does not describe these source files.')
417 captures = {ET.parse(p).getroot().get('ID'): p.stem for p in native.glob('page-*.xml')}
418 hierarchy = ET.parse(native / 'hierarchy.xml').getroot()
419 for section in sections:
420 relative = section['path']
421 matches = [node for node in hierarchy.findall('.//one:Section', ns)
422 if PureWindowsPath(node.attrib['path']).parts[-len(relative.parts):] == relative.parts]
423 if len(matches) != 1:
424 raise ValueError('The native section association is ambiguous.')
425 native_pages = matches[0].findall('one:Page', ns)
426 for (sid, _, _, oid), page in zip(ordered_pages(section['document']), native_pages, strict=True):
427 references[(relative, sid, DEFAULT_CONTEXT, oid)] = Path('reference') / captures[page.get('ID')]
428 (destination / 'reference').symlink_to(native, target_is_directory=True)
429 for index, capture in enumerate(versions):
430 capture = capture.resolve(strict=True)
431 config = json.loads((capture / 'version-ui.json').read_text())
432 matching = [section for section in sections if section['path'].name == Path(config['section']).name
433 and hashlib.sha256((source / section['path']).read_bytes()).hexdigest() == config['source_sha256']]
434 if len(matching) != 1:
435 raise ValueError('The historical capture source association is ambiguous.')
436 section, = matching
437 captures = {ET.parse(p).getroot().get('ID'): p.stem for p in (capture / 'read').glob('page-*.xml')}
438 directory = Path(f'reference-history-{index}')
439 (destination / directory).symlink_to(capture / 'read', target_is_directory=True)
440 for row in json.loads((capture / 'version-associations.json').read_text()):
441 key = (section['path'], row['space'], row['context'], row['object'])
442 rid, revision = view(section['document'], row['space'], row['context'])
443 if key not in histories or rid != row['revision'] or revision['nodes'][row['object']]['kind']['type'] != 'Page':
444 raise ValueError('The historical capture does not describe this page revision.')
445 if key in references:
446 raise ValueError('A page has multiple native reference captures.')
447 references[key] = directory / captures[row['native_id']]
448 accounting = []
449 source_pages = {(section['path'], conflict): filename
450 for section, _, sid, _, rid, _, _, filename, category in pages if category != 'Historical version'
451 for revision in [section['document']['spaces'][sid]['revisions'][rid]]
452 for conflict in revision['nodes'][revision['roots']['1']]['spaces']}
453 page_files = {(section['path'], sid, context, oid): filename for section, _, sid, context, _, oid, _, filename, _ in pages}
454 for section, ordinal, sid, context, rid, oid, title, filename, category in pages:
455 page = Page(section, sid, rid)
456 body = '<div class="meta">' + esc(str(section['path'])) + ' · ' + category + ' ' + str(ordinal + 1) + '</div>'
457 if not page.space['nodes'][oid]['structure']:
458 body += '<h1>' + esc(title) + '</h1>'
459 metadata = page.space['nodes'].get(page.space['roots'].get('2'), {}).get('kind', {})
460 if metadata.get('type') == 'ConflictMetadata' and metadata['author']:
461 body += '<p class="meta">Conflicting author: ' + esc(metadata['author']) + '</p>'
462 source_page = source_pages.get((section['path'], sid))
463 version = histories.get((section['path'], sid, context, oid))
464 if version:
465 source_page = page_files[version['source']]
466 if version['modified']:
467 body += '<p class="meta">Version modified: ' + esc(version['modified']) + '</p>'
468 if source_page:
469 body += '<p><a href="' + source_page + '">Source page</a></p>'
470 reference = references.get((section['path'], sid, context, oid))
471 if reference:
472 body += '<p class="references">'
473 for suffix, label in [('.pdf', 'Native PDF'), ('.png', 'Native screenshot'), ('.xml', 'Native XML')]:
474 if (destination / reference.with_suffix(suffix)).exists():
475 body += '<a href="' + reference.with_suffix(suffix).as_posix() + '">' + label + '</a>'
476 body += '</p>'
477 body += page.render(oid)
478 body += '<details><summary>Document structure and source identities</summary><pre>' + esc(json.dumps(page.space, indent=2, ensure_ascii=False)) + '</pre></details>'
479 body += '<p class="references"><a href="' + section['export'].as_posix() + '/document.json">Document JSON</a><a href="' + section['export'].as_posix() + '/assets.json">Asset references</a></p>'
480 (destination / filename).write_text(html_page(title, nav, body, editable and category == 'Page'))
481 accounting.append({'section': str(section['path']), 'ordinal': ordinal, 'space': sid, 'revision': rid, 'object': oid, 'title': title, 'report': filename, 'category': category, 'context': context, 'version_modified': version['modified'] if version else None, 'source_report': source_page, 'native_reference': reference.as_posix() if reference else None, 'rendered': dict(page.counts)})
482 (destination / 'source.json').write_text(json.dumps(manifest, indent=2))
483 (destination / 'pages.json').write_text(json.dumps(accounting, indent=2, ensure_ascii=False))
484 intro = '<h1>Notebook review</h1><p>' + str(len(sections) + len(unavailable)) + ' sections · ' + str(len(pages)) + ' stored pages</p><p>Readable content follows the stored object order. Outline positions are shown in points. The document structure retains properties and identities that the readable view does not interpret.</p><p><a href="source.json">Source hashes</a> · <a href="pages.json">Page inventory</a> · <a href="catalog.json">Notebook catalog</a></p>'
485 if locked_sections:
486 intro += '<p>Page counts are unavailable for locked sections: ' + esc(', '.join(locked_sections)) + '.</p>'
487 for path, error in unavailable:
488 intro += '<p>Cannot read ' + esc(str(path)) + ': ' + esc(error['message']) + ' (offset ' + str(error['offset']) + ').</p>'
489 (destination / 'index.html').write_text(html_page('Notebook review', nav, intro, editable))
490
491
492if __name__ == '__main__':
493 parser = argparse.ArgumentParser(description=__doc__)
494 parser.add_argument('source', type=Path)
495 parser.add_argument('destination', type=Path)
496 parser.add_argument('--native', type=Path, help='Associate a matching native capture directory.')
497 parser.add_argument('--versions', type=Path, action='append', default=[], help='Associate a verified historical capture; repeat for each section.')
498 parser.add_argument('--timezone', type=ZoneInfo, default=timezone.utc, help='Display historical timestamps in this time zone; default UTC.')
499 args = parser.parse_args()
500 generate(args.source, args.destination, args.native, args.versions, args.timezone)