| 1 | #!/usr/bin/env python3 |
| 2 | """Export a copied notebook to an offline, read-only document report.""" |
| 3 | import argparse |
| 4 | from collections import Counter |
| 5 | from datetime import datetime, timedelta, timezone |
| 6 | import hashlib |
| 7 | from html import escape as esc |
| 8 | import json |
| 9 | import os |
| 10 | from io import BytesIO |
| 11 | from pathlib import Path, PureWindowsPath |
| 12 | from PIL import Image |
| 13 | import xml.etree.ElementTree as ET |
| 14 | from native_xml import ns |
| 15 | import subprocess |
| 16 | from urllib.parse import urlsplit |
| 17 | from zoneinfo import ZoneInfo |
| 18 | |
| 19 | from document_model import BRIDGE, DEFAULT_CONTEXT, EXPORTER, ordered_pages, version_pages, view, walk |
| 20 | |
| 21 | ROOT = Path(__file__).resolve().parent.parent |
| 22 | |
| 23 | |
| 24 | def hyperlink(fragment, target): |
| 25 | if urlsplit(target).scheme.lower() in ('http', 'https', 'mailto', 'onenote'): |
| 26 | return '<a href="' + esc(target) + '" rel="noreferrer">' + fragment + '</a>' |
| 27 | return fragment + ' <span class="meta">(' + esc(target) + ')</span>' |
| 28 | |
| 29 | |
| 30 | def color(value): |
| 31 | if value is None or value >> 24: |
| 32 | return None |
| 33 | return '#' + ''.join(f'{value >> shift & 255:02x}' for shift in (0, 8, 16)) |
| 34 | |
| 35 | |
| 36 | def css(fmt): |
| 37 | rules = [] |
| 38 | for field, prop, yes, no in [('bold', 'font-weight', 'bold', 'normal'), |
| 39 | ('italic', 'font-style', 'italic', 'normal')]: |
| 40 | if fmt.get(field) is not None: |
| 41 | rules.append(f'{prop}:{yes if fmt[field] else no}') |
| 42 | decorations = [v for k, v in [('underline', 'underline'), ('strike', 'line-through')] if fmt.get(k)] |
| 43 | if decorations: |
| 44 | rules.append('text-decoration:' + ' '.join(decorations)) |
| 45 | for field, prop in [('font', 'font-family'), ('font_size', 'font-size'), |
| 46 | ('color', 'color'), ('highlight', 'background-color')]: |
| 47 | value = fmt.get(field) |
| 48 | if value is not None: |
| 49 | if field == 'font': |
| 50 | fallback = 'monospace' if value in ('Consolas', 'Courier New', 'Lucida Console') else 'serif' if value in ('Times New Roman', 'Cambria', 'Georgia') else 'sans-serif' |
| 51 | value = json.dumps(value) + ',' + fallback |
| 52 | else: |
| 53 | value = f'{value:g}pt' if field == 'font_size' else color(value) |
| 54 | if value: |
| 55 | rules.append(f'{prop}:{value}') |
| 56 | for field, prop in [('space_before', 'margin-top'), ('space_after', 'margin-bottom')]: |
| 57 | if fmt.get(field) is not None: |
| 58 | rules.append(f'{prop}:{fmt[field]:g}pt') |
| 59 | if fmt.get('alignment') is not None: |
| 60 | rules.append('text-align:' + {0: 'left', 1: 'center', 2: 'right'}.get(fmt['alignment'], 'start')) |
| 61 | if fmt.get('line_spacing'): |
| 62 | rules.append(f'line-height:max(1.5em,{fmt["line_spacing"]:g}pt)') |
| 63 | if fmt.get('rtl'): |
| 64 | rules.append('direction:rtl') |
| 65 | return ';'.join(rules) |
| 66 | |
| 67 | |
| 68 | def html_page(title, nav, body, editable=False): |
| 69 | policy = "default-src 'none'; img-src 'self' data:; style-src 'self' 'unsafe-inline'; media-src 'self'; base-uri 'none'" |
| 70 | if editable: |
| 71 | policy += "; script-src 'self'; connect-src 'self'" |
| 72 | return '''<!doctype html><html lang="en"><meta charset="utf-8"> |
| 73 | <meta name="viewport" content="width=device-width,initial-scale=1"> |
| 74 | <meta http-equiv="Content-Security-Policy" content="''' + policy + '''"> |
| 75 | <title>''' + esc(title) + '''</title><style> |
| 76 | *{box-sizing:border-box}body{margin:0;color:#000;background:#fff;font:15px/1.5 system-ui,sans-serif}a{color:#175bb2}nav{position:fixed;inset:0 auto 0 0;width:240px;overflow:auto;background:#f5f6f7;padding:20px}nav a{display:block;padding:3px 0;overflow-wrap:anywhere}nav h2{font-size:14px;margin:20px 0 4px}main{margin-left:240px;padding:30px 40px;max-width:1250px}h1{font-size:28px;line-height:1.2}h2{font-size:18px}p{margin:6px 0}.outline{margin:24px 0;border-top:1px solid #d9dde2;padding-top:12px}.location,.meta{font:12px/1.5 system-ui;color:#666;margin:6px 0}.paragraph{min-height:1.3em;position:relative;overflow-wrap:anywhere}.nested{margin-left:24px}.text{white-space:pre-wrap}.tag{display:inline-block;font:12px system-ui;padding:2px 5px;border:1px solid #aaa;border-radius:3px;margin-right:5px}.list-marker{display:inline-block;min-width:22px;margin-left:-22px}.listed{margin-left:22px}img.content{max-width:100%;height:auto;vertical-align:top}figure{margin:12px 0}figcaption{font-size:12px;color:#666}table{border-collapse:collapse;margin:8px 0;max-width:100%}td{border:1px solid #bbb;padding:5px 8px;vertical-align:top;min-width:30px}table.no-borders td{border-color:transparent}.opaque{border:1px dashed #9aa2ad;padding:12px;margin:12px 0;background:#f7f8fa}details{margin:14px 0}summary{cursor:pointer;font:13px system-ui}pre{white-space:pre-wrap;overflow-wrap:anywhere;font:11px/1.4 ui-monospace,monospace;max-height:560px;overflow:auto}.page-link{padding-left:12px}code{font-size:12px}.references a{margin-right:14px}@media(max-width:750px){nav{position:static;width:auto;max-height:220px}main{margin:0;padding:20px}}@media print{nav{display:none}main{margin:0}details{display:none}} |
| 77 | </style>''' + ('<link rel="stylesheet" href="/editor.css"><script src="/editor.js" defer></script>' if editable else '') + '<nav>' + nav + '</nav><main>' + body + '</main></html>' |
| 78 | |
| 79 | |
| 80 | class Page: |
| 81 | def __init__(self, section, sid, rid): |
| 82 | self.section = section |
| 83 | self.space = section['document']['spaces'][sid]['revisions'][rid] |
| 84 | self.text = section['text'][sid][rid] |
| 85 | self.counts = Counter() |
| 86 | self.numbering = {} |
| 87 | self.has_paragraph = False |
| 88 | self.active = set() |
| 89 | |
| 90 | def asset(self, container): |
| 91 | if container is None: |
| 92 | return None |
| 93 | node = self.space['nodes'][container] |
| 94 | if node['kind']['type'] != 'File': |
| 95 | raise ValueError('The asset container is not a file object.') |
| 96 | return self.section['assets'].get(json.dumps(node['kind']['reference'], sort_keys=True)) |
| 97 | |
| 98 | def render(self, oid, depth=0, level=0): |
| 99 | if oid in self.active or depth > 256: |
| 100 | raise ValueError('The content graph is cyclic or exceeds the report depth limit.') |
| 101 | self.active.add(oid) |
| 102 | node = self.space['nodes'][oid] |
| 103 | kind = node['kind']; typ = kind['type'] |
| 104 | self.counts[typ] += 1 |
| 105 | children = lambda refs: ''.join(self.render(x, depth + 1, level) for x in refs) |
| 106 | tags = '' |
| 107 | tag_format = {} |
| 108 | for tag in node['tags']: |
| 109 | definition = self.space['nodes'][tag['definition']]['kind'] if tag['definition'] else {} |
| 110 | label = definition.get('label') or 'Task' |
| 111 | if not tag['status'] & 2: |
| 112 | for field in ('color', 'highlight'): |
| 113 | if color(definition.get(field)): |
| 114 | tag_format.setdefault(field, definition[field]) |
| 115 | if tag['due']: |
| 116 | label += ' · Due ' + datetime.fromtimestamp(tag['due'] + 315532800, timezone.utc).strftime('%Y-%m-%d') |
| 117 | if tag['status'] & 2: |
| 118 | label += ' · Disabled' |
| 119 | tags += '<span class="tag" title="' + esc(json.dumps(tag)) + '">' + (('☑ ' if tag['status'] & 1 else '☐ ') if definition.get('shape') == 3 or tag['status'] & 4 else '') + esc(label) + '</span>' |
| 120 | if typ == 'RichText': |
| 121 | body = '' |
| 122 | structured_math = any(c in kind['text'] for c in '\ufdd0\ufdee\ufdef') |
| 123 | equation = False |
| 124 | for run_index, run in enumerate(self.text[oid]): |
| 125 | if run['format']['hidden']: |
| 126 | continue |
| 127 | if structured_math and run['format'].get('math'): |
| 128 | if not equation: |
| 129 | body += '<span class="meta">[Equation · see native reference]</span>' |
| 130 | equation = True |
| 131 | continue |
| 132 | equation = False |
| 133 | style = {**run['format'], **tag_format} |
| 134 | if (style['superscript'] or style['subscript']) and style['font_size'] is not None: |
| 135 | style['font_size'] *= 2 / 3 |
| 136 | fragment = '<span data-text-object="' + esc(oid) + '" data-run="' + str(run_index) + '" style="' + esc(css(style)) + '">' + esc(run['text']) + '</span>' |
| 137 | if run['format']['superscript']: |
| 138 | fragment = '<sup>' + fragment + '</sup>' |
| 139 | if run['format']['subscript']: |
| 140 | fragment = '<sub>' + fragment + '</sub>' |
| 141 | if run['link']: |
| 142 | fragment = hyperlink(fragment, run['link']) |
| 143 | body += fragment |
| 144 | body = tags + '<span class="text">' + body + '</span>' |
| 145 | elif typ == 'Paragraph': |
| 146 | body = children(node['content']) |
| 147 | marker = '' |
| 148 | for list_id in kind['lists']: |
| 149 | item = self.space['nodes'][list_id]['kind'] |
| 150 | fmt = item['format'] or '' |
| 151 | if '\ufffd' in fmt: |
| 152 | position = fmt.index('\ufffd'); style = ord(fmt[position + 1]) |
| 153 | key = (fmt, level) |
| 154 | value = item['restart'] if item['restart'] is not None else self.numbering.get(key, 0) + 1 |
| 155 | self.numbering[key] = value |
| 156 | number = str(value) |
| 157 | if style in (1, 2): |
| 158 | number = ''; rest = value |
| 159 | for quantity, letters in [(1000, 'M'), (900, 'CM'), (500, 'D'), (400, 'CD'), (100, 'C'), (90, 'XC'), (50, 'L'), (40, 'XL'), (10, 'X'), (9, 'IX'), (5, 'V'), (4, 'IV'), (1, 'I')]: |
| 160 | times, rest = divmod(rest, quantity) |
| 161 | number += letters * times |
| 162 | if style == 2: number = number.lower() |
| 163 | elif style in (3, 4): |
| 164 | number = ''; rest = value |
| 165 | while rest: |
| 166 | rest, digit = divmod(rest - 1, 26) |
| 167 | number = chr(65 + digit) + number |
| 168 | if style == 4: number = number.lower() |
| 169 | elif style != 0: |
| 170 | raise ValueError(f'Number format {style} requires interpretation before report generation.') |
| 171 | marker += fmt[:position] + number + fmt[position + 2:] |
| 172 | else: |
| 173 | marker += {1: '•', 2: '◦', 3: '●', 4: '○', 5: '◉', 6: '◎', 7: '▪', 8: '▫', 9: '■', 10: '□', 11: '▸', 12: '▶', 13: '◇', 14: '♢', 15: '◆', 16: '❖', 17: '★', 18: '☆', 19: '☀', 20: '>', 21: '→', 22: '⇒', 23: '⇨', 24: '*', 25: '-', 26: '–', 27: '—'}.get(item['bullet'], fmt) |
| 174 | if marker: |
| 175 | body = '<span class="list-marker">' + esc(marker) + '</span>' + body |
| 176 | fmt = node['format'] |
| 177 | if kind['paragraph_style']: |
| 178 | parent = self.space['nodes'][kind['paragraph_style']]['format'] |
| 179 | fmt = {k: v if v is not None else parent[k] for k, v in fmt.items()} |
| 180 | for child in node['content']: |
| 181 | if child in self.text and self.text[child]: |
| 182 | resolved = self.text[child][0]['format'] |
| 183 | fmt = {**fmt, **{k: resolved[k] for k in ('alignment', 'rtl', 'space_before', 'space_after', 'line_spacing') if resolved[k] is not None}} |
| 184 | break |
| 185 | if not self.has_paragraph: |
| 186 | fmt = {**fmt, 'space_before': None} |
| 187 | self.has_paragraph = True |
| 188 | body = '<div class="paragraph' + (' listed' if marker else '') + '" style="' + esc(css(fmt)) + '">' + tags + body + '</div>' |
| 189 | if node['children']: |
| 190 | child_level = node['child_level'] or 1 |
| 191 | descendants = '<div class="nested" style="margin-left:' + str(child_level * 24) + 'px">' + ''.join(self.render(x, depth + 1, level + child_level) for x in node['children']) + '</div>' |
| 192 | state = kind.get('collapse_state') |
| 193 | if state == 1: |
| 194 | descendants = '<details><summary>Collapsed paragraphs</summary>' + descendants + '</details>' |
| 195 | elif state not in (None, 0): |
| 196 | descendants = '<div class="opaque">Uninterpreted collapse state: ' + str(state) + '</div>' + descendants |
| 197 | body += descendants |
| 198 | elif typ in ('Page', 'Title', 'Outline'): |
| 199 | if typ == 'Outline': |
| 200 | previous = self.numbering, self.has_paragraph |
| 201 | self.numbering, self.has_paragraph = {}, False |
| 202 | body = children(node['structure'] + node['content'] + node['children']) |
| 203 | if typ == 'Outline': |
| 204 | self.numbering, self.has_paragraph = previous |
| 205 | if typ == 'Outline' and node['layout']['x'] is not None: |
| 206 | values = ', '.join(f'{k} {v:g} pt' for k, v in node['layout'].items() if v is not None) |
| 207 | body = '<section class="outline"><div class="location">' + esc(values) + '</div>' + body + '</section>' |
| 208 | elif typ == 'Title': |
| 209 | body = '<header>' + body + '</header>' |
| 210 | elif typ == 'OutlineGroup': |
| 211 | child_level = node['child_level'] or 1 |
| 212 | body = '<div class="nested" style="margin-left:' + str(child_level * 24) + 'px">' + ''.join(self.render(x, depth + 1, level + child_level) for x in node['children']) + '</div>' |
| 213 | elif typ == 'Table': |
| 214 | cols = ''.join(f'<col style="width:{width:g}pt">' if kind['locked'] and kind['locked'][i] else '<col>' for i, width in enumerate(kind['widths'])) |
| 215 | body = '<table class="' + ('no-borders' if kind['borders'] is False else '') + '"><colgroup>' + cols + '</colgroup>' + children(node['children']) + '</table>' |
| 216 | elif typ == 'Row': |
| 217 | body = '<tr>' + children(node['children']) + '</tr>' |
| 218 | elif typ == 'Cell': |
| 219 | shade = color(kind['shading']) |
| 220 | body = '<td' + (' style="background:' + shade + '"' if shade else '') + '>' + children(node['children']) + '</td>' |
| 221 | elif typ == 'Image': |
| 222 | asset = self.asset(kind['container']) |
| 223 | if asset: |
| 224 | width = node['layout']['max_width'] if node['layout']['max_width'] is not None else kind['picture_width'] |
| 225 | height = node['layout']['max_height'] if node['layout']['max_height'] is not None else kind['picture_height'] |
| 226 | style = f'width:{width:g}pt' if width is not None else '' |
| 227 | if width and height: |
| 228 | style += f';aspect-ratio:{width:g}/{height:g}' |
| 229 | preview = self.section['previews'].get(asset, asset) |
| 230 | if preview.endswith('.bin'): |
| 231 | content = '<span class="opaque">Image format is uninterpreted; the original payload is retained in the asset references.</span>' |
| 232 | else: |
| 233 | content = '<img class="content" src="' + esc(preview) + '" style="' + style + '" alt="' + esc(kind['alt'] or '') + '">' |
| 234 | body = '<figure>' + tags + (hyperlink(content, kind['link']) if kind['link'] else content) |
| 235 | labels = [kind['filename']] if kind['filename'] else [] |
| 236 | if kind['background']: labels.append('Background image') |
| 237 | if kind['printout']: labels.append('Printout image') |
| 238 | if labels: |
| 239 | body += '<figcaption>' + esc(' · '.join(labels)) + '</figcaption>' |
| 240 | body += '</figure>' |
| 241 | if kind['background'] and not kind['printout']: |
| 242 | body = '<details><summary>Background image' + (' · ' + esc(kind['filename']) if kind['filename'] else '') + '</summary>' + body + '</details>' |
| 243 | else: |
| 244 | body = '<div class="opaque">Image payload is external; its source reference is retained.</div>' |
| 245 | elif typ == 'Attachment': |
| 246 | asset = self.asset(kind['container']) |
| 247 | label = kind['filename'] or 'Attached file' |
| 248 | body = tags + ('<a download="' + esc(label) + '" href="' + esc(asset) + '">' + esc(label) + '</a>' if asset else esc(label) + ' · External payload') |
| 249 | elif typ == 'Ink': |
| 250 | body = '<div class="opaque">Ink drawing · Stroke data is retained in the document structure.</div>' |
| 251 | else: |
| 252 | body = '<div class="opaque">' + ('Encrypted content' if typ == 'Encrypted' else f'Uninterpreted content · JCID {node["jcid"]:#x}') + '<br><code>' + esc(oid) + '</code>' + children(node['structure'] + node['content'] + node['children']) + '</div>' |
| 253 | for recording in node['media_ids']: |
| 254 | matches = [n['kind'] for n in self.space['nodes'].values() |
| 255 | if n['kind']['type'] == 'Attachment' and n['kind']['recording_id'] == recording] |
| 256 | label = 'Recording reference' |
| 257 | asset = None |
| 258 | if len(matches) == 1: |
| 259 | label = matches[0]['filename'] or label |
| 260 | asset = self.asset(matches[0]['container']) |
| 261 | if node['media_time_ms'] is not None: |
| 262 | seconds = node['media_time_ms'] / 1000 |
| 263 | label += f' · {seconds:g} s' |
| 264 | body += '<div class="meta">' + ('<a href="' + esc(asset) + '" download>' + esc(label) + '</a>' if asset else esc(label)) + '</div>' |
| 265 | self.active.remove(oid) |
| 266 | return '<div data-object="' + esc(oid) + '">' + body + '</div>' if typ not in ('RichText', 'Row', 'Cell') else body |
| 267 | |
| 268 | |
| 269 | def generate(source, destination, native=None, versions=(), zone=timezone.utc, editable=False, previous=None): |
| 270 | source = source.resolve(strict=True) |
| 271 | if destination.resolve().is_relative_to(source): |
| 272 | raise ValueError('Choose an export directory outside the source notebook.') |
| 273 | destination.mkdir(parents=True, exist_ok=False) |
| 274 | (destination / 'model').mkdir(); (destination / 'assets').mkdir() |
| 275 | cached = {row['path']: (row['sha256'], previous / 'model' / str(index)) |
| 276 | for index, row in enumerate(json.loads((previous / 'source.json').read_text()))} if previous else {} |
| 277 | result = json.loads(subprocess.check_output([BRIDGE, 'catalog', source], timeout=120)) |
| 278 | if not result['ok']: raise ValueError(result['error']) |
| 279 | catalog = result['catalog'] |
| 280 | (destination / 'catalog.json').write_text(json.dumps(catalog, indent=2, ensure_ascii=False)) |
| 281 | files = []; catalog_sections = {} |
| 282 | def collect(folder): |
| 283 | if folder['toc'] is not None: files.append(Path(folder['path']) / folder['toc']['filename']) |
| 284 | for section in folder['sections']: |
| 285 | relative = Path(section['path']) |
| 286 | files.append(relative); catalog_sections[relative] = section |
| 287 | for group in folder['groups']: collect(group) |
| 288 | collect(catalog) |
| 289 | sections = []; unavailable = []; manifest = [] |
| 290 | for index, relative in enumerate(sorted(files)): |
| 291 | path = source / relative |
| 292 | before = path.read_bytes() |
| 293 | manifest.append({'path': relative.as_posix(), 'sha256': hashlib.sha256(before).hexdigest(), 'bytes': len(before)}) |
| 294 | state = catalog_sections.get(relative, {}).get('state', {}) |
| 295 | if isinstance(state, dict) and 'Unreadable' in state: |
| 296 | unavailable.append((relative, state['Unreadable'])) |
| 297 | continue |
| 298 | exported = destination / 'model' / str(index) |
| 299 | reusable = cached.get(relative.as_posix()) |
| 300 | reused = reusable is not None and reusable[0] == manifest[-1]['sha256'] |
| 301 | if reused and any('External' in asset['reference'] for asset in json.loads((reusable[1] / 'assets.json').read_text())): |
| 302 | reused = False |
| 303 | if reused: |
| 304 | exported.mkdir() |
| 305 | for name in ('document.json', 'text.json', 'assets.json'): |
| 306 | os.link(reusable[1] / name, exported / name) |
| 307 | else: |
| 308 | subprocess.run([EXPORTER, path, exported], check=True) |
| 309 | document = json.loads((exported / 'document.json').read_text()) |
| 310 | if path.read_bytes() != before: |
| 311 | raise ValueError('A source file changed during export.') |
| 312 | assets = {} |
| 313 | previews = {} |
| 314 | image_references = { |
| 315 | json.dumps(revision['nodes'][node['kind']['container']]['kind']['reference'], sort_keys=True) |
| 316 | for space in document['spaces'].values() for revision in space['revisions'].values() |
| 317 | for node in revision['nodes'].values() |
| 318 | if node['kind']['type'] == 'Image' and node['kind']['container'] is not None |
| 319 | } |
| 320 | rows = json.loads((exported / 'assets.json').read_text()) |
| 321 | for asset in rows: |
| 322 | if asset['path'] is None: continue |
| 323 | reference = json.dumps(asset['reference'], sort_keys=True) |
| 324 | if reused: |
| 325 | name = Path(asset['path']).name |
| 326 | assets[reference] = 'assets/' + name |
| 327 | if not (destination / 'assets' / name).exists(): |
| 328 | os.link(previous / 'assets' / name, destination / 'assets' / name) |
| 329 | preview = name + '.png' |
| 330 | if name.endswith('.tiff') and reference in image_references: |
| 331 | if not (destination / 'assets' / preview).exists(): |
| 332 | os.link(previous / 'assets' / preview, destination / 'assets' / preview) |
| 333 | previews['assets/' + name] = 'assets/' + preview |
| 334 | continue |
| 335 | original = exported / asset['path']; data = original.read_bytes() |
| 336 | extension = '.png' if data.startswith(b'\x89PNG') else '.jpg' if data.startswith(b'\xff\xd8') else '.gif' if data.startswith(b'GIF8') else '.bmp' if data.startswith(b'BM') else '.tiff' if data.startswith((b'II*\0', b'MM\0*')) else '.bin' |
| 337 | name = hashlib.sha256(data).hexdigest() + extension |
| 338 | target = destination / 'assets' / name |
| 339 | if target.exists(): |
| 340 | original.unlink() |
| 341 | else: |
| 342 | original.rename(target) |
| 343 | asset['path'] = '../../assets/' + name |
| 344 | assets[reference] = 'assets/' + name |
| 345 | if extension == '.tiff' and reference in image_references: |
| 346 | preview = name + '.png' |
| 347 | with Image.open(BytesIO(data)) as image: |
| 348 | if image.n_frames != 1: |
| 349 | raise ValueError('Multipage TIFF requires frame interpretation before report generation.') |
| 350 | image.convert('RGBA').save(destination / 'assets' / preview) |
| 351 | previews['assets/' + name] = 'assets/' + preview |
| 352 | if not reused: |
| 353 | (exported / 'assets').rmdir() |
| 354 | (exported / 'assets.json').write_text(json.dumps(rows, indent=2)) |
| 355 | if path.suffix.lower() == '.one': |
| 356 | sections.append({'path': relative, 'export': exported.relative_to(destination), 'document': document, |
| 357 | 'text': json.loads((exported / 'text.json').read_text()), 'assets': assets, 'previews': previews}) |
| 358 | order = {path: index for index, path in enumerate(catalog_sections)} |
| 359 | sections.sort(key=lambda section: order[section['path']]) |
| 360 | pages = []; locked_sections = []; histories = {}; nav = '<a href="index.html">Notebook review</a>' |
| 361 | for section in sections: |
| 362 | state = catalog_sections[section['path']]['state'] |
| 363 | name = state.get('Readable', {}).get('name') if isinstance(state, dict) else None |
| 364 | if name is None: name = section['path'].stem |
| 365 | nav += '<h2>' + esc(str(section['path'].parent) + ' / ' + name) + '</h2>' |
| 366 | _, root_space = view(section['document'], section['document']['root']) |
| 367 | if root_space['nodes'][root_space['roots']['1']]['kind']['type'] == 'Encrypted': |
| 368 | locked_sections.append(str(section['path'])) |
| 369 | nav += '<p>Locked section · Page count unavailable</p><a href="' + section['export'].as_posix() + '/document.json">Encrypted structure</a>' |
| 370 | continue |
| 371 | ordinary = list(ordered_pages(section['document'])) |
| 372 | known = {(sid, rid, oid) for sid, rid, _, oid in ordinary} |
| 373 | additional = [(sid, rid, revision, oid) for sid in section['document']['spaces'] |
| 374 | for rid, revision in [view(section['document'], sid)] |
| 375 | for oid, node in revision['nodes'].items() if node['kind']['type'] == 'Page' and (sid, rid, oid) not in known] |
| 376 | historical = [] |
| 377 | for sid, _, current, parent_oid in ordinary + additional: |
| 378 | for context, rid, revision, oid, proxy in version_pages(section['document'], sid, current): |
| 379 | historical.append((sid, context, rid, revision, oid)) |
| 380 | modified = proxy['kind']['modified_filetime'] |
| 381 | histories[(section['path'], sid, context, oid)] = { |
| 382 | 'modified': (datetime(1601, 1, 1, tzinfo=timezone.utc) + timedelta(microseconds=modified // 10)).astimezone(zone).isoformat() if modified is not None else None, |
| 383 | 'source': (section['path'], sid, DEFAULT_CONTEXT, parent_oid), |
| 384 | } |
| 385 | if not ordinary and not additional: |
| 386 | nav += '<p>Empty section</p>' |
| 387 | current_pages = [(sid, DEFAULT_CONTEXT, rid, revision, oid) for sid, rid, revision, oid in ordinary + additional] |
| 388 | for ordinal, (sid, context, rid, space, oid) in enumerate(current_pages + historical): |
| 389 | category = 'Historical version' if ordinal >= len(ordinary) + len(additional) else 'Additional stored page' if ordinal >= len(ordinary) else 'Recycle bin' if 'OneNote_RecycleBin' in section['path'].parts else 'Page' |
| 390 | metadata = space['nodes'].get(space['roots'].get('2'), {}).get('kind', {}) |
| 391 | if sid == root_space['nodes'][root_space['roots']['1']]['kind'].get('default_template'): |
| 392 | if metadata.get('type') != 'TemplateMetadata': |
| 393 | raise ValueError('The default page template has unrecognized metadata.') |
| 394 | category = 'Default page template' |
| 395 | titles = [n for root in space['nodes'][oid]['structure'] for _, n in walk(space, root) if n['kind']['type'] == 'RichText' and not n['kind']['boilerplate']] |
| 396 | title = ' '.join(n['kind']['text'] for n in titles) or space['nodes'][oid]['kind']['alternate_title'] or metadata.get('title') or metadata.get('name') or 'Untitled page' |
| 397 | filename = f'page-{len(pages):03}.html' |
| 398 | if metadata.get('type') == 'ConflictMetadata': |
| 399 | category = 'Conflict page' |
| 400 | level = metadata.get('level') or 1 |
| 401 | label = category + ' · ' + title if category in ('Additional stored page', 'Default page template', 'Conflict page', 'Historical version') else title |
| 402 | version = histories.get((section['path'], sid, context, oid)) |
| 403 | if version and version['modified']: |
| 404 | label += ' · ' + version['modified'][:10] |
| 405 | nav += '<a class="page-link" style="margin-left:' + str((level - 1) * 12) + 'px" href="' + filename + '">' + esc(label) + '</a>' |
| 406 | pages.append((section, ordinal, sid, context, rid, oid, title, filename, category)) |
| 407 | references = {} |
| 408 | if native is not None: |
| 409 | native = native.resolve(strict=True) |
| 410 | captured = native.parent / 'notebook' |
| 411 | if captured.exists(): |
| 412 | hashes = {p.relative_to(captured).as_posix(): hashlib.sha256(p.read_bytes()).hexdigest() for p in captured.rglob('*') if p.is_file()} |
| 413 | else: |
| 414 | hashes = {item['path']: item['sha256'] for item in json.loads((native.parent / 'source.json').read_text())} |
| 415 | if any(hashes.get(item['path']) != item['sha256'] for item in manifest): |
| 416 | raise ValueError('The native capture does not describe these source files.') |
| 417 | captures = {ET.parse(p).getroot().get('ID'): p.stem for p in native.glob('page-*.xml')} |
| 418 | hierarchy = ET.parse(native / 'hierarchy.xml').getroot() |
| 419 | for section in sections: |
| 420 | relative = section['path'] |
| 421 | matches = [node for node in hierarchy.findall('.//one:Section', ns) |
| 422 | if PureWindowsPath(node.attrib['path']).parts[-len(relative.parts):] == relative.parts] |
| 423 | if len(matches) != 1: |
| 424 | raise ValueError('The native section association is ambiguous.') |
| 425 | native_pages = matches[0].findall('one:Page', ns) |
| 426 | for (sid, _, _, oid), page in zip(ordered_pages(section['document']), native_pages, strict=True): |
| 427 | references[(relative, sid, DEFAULT_CONTEXT, oid)] = Path('reference') / captures[page.get('ID')] |
| 428 | (destination / 'reference').symlink_to(native, target_is_directory=True) |
| 429 | for index, capture in enumerate(versions): |
| 430 | capture = capture.resolve(strict=True) |
| 431 | config = json.loads((capture / 'version-ui.json').read_text()) |
| 432 | matching = [section for section in sections if section['path'].name == Path(config['section']).name |
| 433 | and hashlib.sha256((source / section['path']).read_bytes()).hexdigest() == config['source_sha256']] |
| 434 | if len(matching) != 1: |
| 435 | raise ValueError('The historical capture source association is ambiguous.') |
| 436 | section, = matching |
| 437 | captures = {ET.parse(p).getroot().get('ID'): p.stem for p in (capture / 'read').glob('page-*.xml')} |
| 438 | directory = Path(f'reference-history-{index}') |
| 439 | (destination / directory).symlink_to(capture / 'read', target_is_directory=True) |
| 440 | for row in json.loads((capture / 'version-associations.json').read_text()): |
| 441 | key = (section['path'], row['space'], row['context'], row['object']) |
| 442 | rid, revision = view(section['document'], row['space'], row['context']) |
| 443 | if key not in histories or rid != row['revision'] or revision['nodes'][row['object']]['kind']['type'] != 'Page': |
| 444 | raise ValueError('The historical capture does not describe this page revision.') |
| 445 | if key in references: |
| 446 | raise ValueError('A page has multiple native reference captures.') |
| 447 | references[key] = directory / captures[row['native_id']] |
| 448 | accounting = [] |
| 449 | source_pages = {(section['path'], conflict): filename |
| 450 | for section, _, sid, _, rid, _, _, filename, category in pages if category != 'Historical version' |
| 451 | for revision in [section['document']['spaces'][sid]['revisions'][rid]] |
| 452 | for conflict in revision['nodes'][revision['roots']['1']]['spaces']} |
| 453 | page_files = {(section['path'], sid, context, oid): filename for section, _, sid, context, _, oid, _, filename, _ in pages} |
| 454 | for section, ordinal, sid, context, rid, oid, title, filename, category in pages: |
| 455 | page = Page(section, sid, rid) |
| 456 | body = '<div class="meta">' + esc(str(section['path'])) + ' · ' + category + ' ' + str(ordinal + 1) + '</div>' |
| 457 | if not page.space['nodes'][oid]['structure']: |
| 458 | body += '<h1>' + esc(title) + '</h1>' |
| 459 | metadata = page.space['nodes'].get(page.space['roots'].get('2'), {}).get('kind', {}) |
| 460 | if metadata.get('type') == 'ConflictMetadata' and metadata['author']: |
| 461 | body += '<p class="meta">Conflicting author: ' + esc(metadata['author']) + '</p>' |
| 462 | source_page = source_pages.get((section['path'], sid)) |
| 463 | version = histories.get((section['path'], sid, context, oid)) |
| 464 | if version: |
| 465 | source_page = page_files[version['source']] |
| 466 | if version['modified']: |
| 467 | body += '<p class="meta">Version modified: ' + esc(version['modified']) + '</p>' |
| 468 | if source_page: |
| 469 | body += '<p><a href="' + source_page + '">Source page</a></p>' |
| 470 | reference = references.get((section['path'], sid, context, oid)) |
| 471 | if reference: |
| 472 | body += '<p class="references">' |
| 473 | for suffix, label in [('.pdf', 'Native PDF'), ('.png', 'Native screenshot'), ('.xml', 'Native XML')]: |
| 474 | if (destination / reference.with_suffix(suffix)).exists(): |
| 475 | body += '<a href="' + reference.with_suffix(suffix).as_posix() + '">' + label + '</a>' |
| 476 | body += '</p>' |
| 477 | body += page.render(oid) |
| 478 | body += '<details><summary>Document structure and source identities</summary><pre>' + esc(json.dumps(page.space, indent=2, ensure_ascii=False)) + '</pre></details>' |
| 479 | body += '<p class="references"><a href="' + section['export'].as_posix() + '/document.json">Document JSON</a><a href="' + section['export'].as_posix() + '/assets.json">Asset references</a></p>' |
| 480 | (destination / filename).write_text(html_page(title, nav, body, editable and category == 'Page')) |
| 481 | accounting.append({'section': str(section['path']), 'ordinal': ordinal, 'space': sid, 'revision': rid, 'object': oid, 'title': title, 'report': filename, 'category': category, 'context': context, 'version_modified': version['modified'] if version else None, 'source_report': source_page, 'native_reference': reference.as_posix() if reference else None, 'rendered': dict(page.counts)}) |
| 482 | (destination / 'source.json').write_text(json.dumps(manifest, indent=2)) |
| 483 | (destination / 'pages.json').write_text(json.dumps(accounting, indent=2, ensure_ascii=False)) |
| 484 | intro = '<h1>Notebook review</h1><p>' + str(len(sections) + len(unavailable)) + ' sections · ' + str(len(pages)) + ' stored pages</p><p>Readable content follows the stored object order. Outline positions are shown in points. The document structure retains properties and identities that the readable view does not interpret.</p><p><a href="source.json">Source hashes</a> · <a href="pages.json">Page inventory</a> · <a href="catalog.json">Notebook catalog</a></p>' |
| 485 | if locked_sections: |
| 486 | intro += '<p>Page counts are unavailable for locked sections: ' + esc(', '.join(locked_sections)) + '.</p>' |
| 487 | for path, error in unavailable: |
| 488 | intro += '<p>Cannot read ' + esc(str(path)) + ': ' + esc(error['message']) + ' (offset ' + str(error['offset']) + ').</p>' |
| 489 | (destination / 'index.html').write_text(html_page('Notebook review', nav, intro, editable)) |
| 490 | |
| 491 | |
| 492 | if __name__ == '__main__': |
| 493 | parser = argparse.ArgumentParser(description=__doc__) |
| 494 | parser.add_argument('source', type=Path) |
| 495 | parser.add_argument('destination', type=Path) |
| 496 | parser.add_argument('--native', type=Path, help='Associate a matching native capture directory.') |
| 497 | parser.add_argument('--versions', type=Path, action='append', default=[], help='Associate a verified historical capture; repeat for each section.') |
| 498 | parser.add_argument('--timezone', type=ZoneInfo, default=timezone.utc, help='Display historical timestamps in this time zone; default UTC.') |
| 499 | args = parser.parse_args() |
| 500 | generate(args.source, args.destination, args.native, args.versions, args.timezone) |