| 1 | """Account for offline document intents, receipts, reader states and native formatting.""" |
| 2 | import uuid |
| 3 | |
| 4 | from document_model import ordered_pages, walk |
| 5 | |
| 6 | |
| 7 | def identity(insertion, extension): |
| 8 | return '{' + str(uuid.UUID(bytes_le=bytes(insertion['guid']))).upper() + '},' + str(extension) |
| 9 | |
| 10 | |
| 11 | def characters(observed): |
| 12 | actual = [(char, run['bold'], run['size'], run['color']) |
| 13 | for run in observed['runs'] for char in run['text']] |
| 14 | assert ''.join(row[0] for row in actual) == observed['text'], 'Document runs omit or duplicate text' |
| 15 | return actual |
| 16 | |
| 17 | |
| 18 | def operation_kinds(events): |
| 19 | kinds = tuple(events[0].get('document_kinds', ('insert', 'format'))) |
| 20 | assert kinds in (('insert', 'format'), ('insert', 'format', 'text'), |
| 21 | ('insert', 'format', 'text', 'split', 'right_text', 'join'), |
| 22 | ('insert', 'format', 'text', 'split', 'right_text', 'nest', 'unnest', 'join', 'tail_split', 'delete')), 'Unknown document workload' |
| 23 | return kinds |
| 24 | |
| 25 | |
| 26 | def document_history(logs, operations): |
| 27 | documents = {} |
| 28 | for actor, events in logs.items(): |
| 29 | if not actor.startswith('w'): continue |
| 30 | assert events[0].get('document_operations') is True, 'Writer omitted document operations' |
| 31 | kinds = operation_kinds(events) |
| 32 | edits = [row for row in events if row['event'] == 'local_document_commit'] |
| 33 | assert [(row['operation'], row['kind']) for row in edits] == [ |
| 34 | (i, kind) for i in range(operations) for kind in kinds], 'Missing or duplicate document intent' |
| 35 | ids = [row['id'] for row in edits] |
| 36 | assert ids == sorted(set(ids)), 'Document intent IDs are duplicated or unordered' |
| 37 | assert not set(ids) & {row['id'] for row in events if row['event'] == 'local_commit'}, 'Text and document intents share an ID' |
| 38 | receipts = [row for row in events if row['event'] == 'document_receipt'] |
| 39 | reopened = [row for row in events if row['event'] == 'reopened_document_receipt'] |
| 40 | assert [row['id'] for row in receipts] == ids, 'Document receipt inventory differs' |
| 41 | assert [(r['id'], r['revision']) for r in reopened] == [(r['id'], r['revision']) for r in receipts], 'Document receipts changed across reopen' |
| 42 | linked = {} |
| 43 | for intent, receipt in zip(edits, receipts, strict=True): |
| 44 | changed_targets = {intent['object']} |
| 45 | if intent['kind'] in ('split', 'tail_split'): changed_targets.add(identity(intent['split'], 2)) |
| 46 | if intent['kind'] == 'join': changed_targets = set(intent['joined']) |
| 47 | if intent['kind'] in ('nest', 'unnest'): changed_targets = set() |
| 48 | attempts = [row for row in events if row['event'] == 'remote_attempt' and row['revision'] == receipt['revision'] |
| 49 | and set(row.get('document_changes') or {}) == changed_targets] |
| 50 | if not attempts and intent['kind'] == 'format': |
| 51 | attempts = [row for row in events if row['event'] == 'remote_attempt' and row['state'] == 'Unknown' |
| 52 | and set(row.get('document_changes') or {}) == changed_targets] |
| 53 | assert len(attempts) == 1, 'Document receipt lacks one publication attempt' |
| 54 | attempt, = attempts |
| 55 | assert attempt['state'] in ('Committed', 'Unknown'), 'Receipt identifies an unpublished document operation' |
| 56 | assert (intent['object'] in attempt['documents']) == (intent['kind'] != 'delete'), 'Document publication retained or lost its target' |
| 57 | assert attempt['document_changes'] == {target: attempt['documents'].get(target) for target in changed_targets}, 'Document publication changed another target' |
| 58 | successful = [row for row in events if row['event'] == 'remote_attempt' |
| 59 | and row['state'] in ('Committed', 'Unknown') |
| 60 | and row.get('document_changes') == attempt['document_changes'] |
| 61 | and row.get('document_graph_changes') == attempt.get('document_graph_changes')] |
| 62 | assert successful == [attempt], 'Document intent was published or attempted uncertainly more than once' |
| 63 | assert intent['started_us'] <= attempt['started_us'] <= attempt['finished_us'] <= receipt['at_us'] and intent['started_us'] <= intent['finished_us'] <= receipt['at_us'], 'Document acknowledgement order is invalid' |
| 64 | if attempt['state'] == 'Unknown': |
| 65 | confirmed, observed = False, None |
| 66 | for row in events: |
| 67 | if row['event'] == 'read': observed = row |
| 68 | if (row['event'] != 'remote_confirm' or row['state'] != 'Committed' |
| 69 | or receipt['revision'] not in row.get('revisions', {}).get(intent['space'], []) |
| 70 | or not attempt['finished_us'] <= row['started_us'] <= row['finished_us'] <= receipt['at_us']): |
| 71 | continue |
| 72 | if receipt['revision'] != attempt['revision']: |
| 73 | assert intent['kind'] == 'format' and attempt['revision'] not in row['revisions'][intent['space']], 'Replacement receipt did not retire the original attempt' |
| 74 | assert row.get('current_revisions', {}).get(intent['space']) == receipt['revision'], 'Effect receipt does not identify the confirmed current revision' |
| 75 | assert observed and observed['finished_us'] <= row['started_us'] and observed['text'] == row['text'] |
| 76 | assert observed.get('documents', {}).get(intent['object']) == attempt['document_changes'][intent['object']], 'Effect confirmation differs from the uncertain formatting intent' |
| 77 | confirmed = True |
| 78 | assert confirmed, 'Uncertain document publication lacks confirmation' |
| 79 | else: |
| 80 | assert receipt['revision'] == attempt['revision'] |
| 81 | linked[intent['id']] = {**attempt, 'acknowledged_us': receipt['at_us'], 'receipt_revision': receipt['revision']} |
| 82 | assert {(row['revision'], row['started_us']) for row in events |
| 83 | if row['event'] == 'remote_attempt' and row['state'] in ('Committed', 'Unknown') |
| 84 | and (row.get('document_changes') or row.get('document_graph_changes'))} == { |
| 85 | (attempt['revision'], attempt['started_us']) for attempt in linked.values() |
| 86 | }, 'A document publication lacks its recorded intent and receipt' |
| 87 | for at in range(0, len(edits), len(kinds)): |
| 88 | inserted, formatted, *replaced = edits[at:at + len(kinds)] |
| 89 | number = inserted['operation'] |
| 90 | text = f'Document {actor}:{number} 🦀' |
| 91 | insertion = inserted['insertion'] |
| 92 | target = identity(insertion, 2) |
| 93 | assert target == inserted['object'] == formatted['object'], 'Dependent formatting addresses another object' |
| 94 | assert inserted['space'] == formatted['space'] and inserted['text'] == formatted['text'] == insertion['text'] == text |
| 95 | assert insertion['author'] == 'Offline document writer' |
| 96 | if number % 2 == 0: |
| 97 | assert insertion['placement'] == {'Outline': {'x': 144 + int(actor[1:]) * 240, 'y': 144 + number * 72}}, 'Outline placement differs from intent' |
| 98 | else: |
| 99 | assert insertion['placement'] == {'Paragraph': {'before': None}} |
| 100 | assert insertion['parent'] == identity(edits[(number-1)*len(kinds)]['insertion'], 1), 'Paragraph lost its outline parent' |
| 101 | assert formatted['range'] == [1, len(text.encode('utf-16-le')) // 2 - 2] |
| 102 | assert formatted['attributes'] == [{'Bold': True}, {'FontSize': 18 + number % 9}, {'Color': [18, 52, 86]}] |
| 103 | old = [(char, False, 11, 0xff000000) for char in text] |
| 104 | if insertion.get('formats'): |
| 105 | span, = insertion['formats'] |
| 106 | assert span['range'] == {'start': 0, 'end': len(text.encode('utf-16-le')) // 2}, 'Initial formatting range differs from workload' |
| 107 | assert span['attributes'] == [{'FontSize': 13.5}, {'Color': [68, 85, 102]}], 'Initial formatting attributes differ from workload' |
| 108 | old = [(char, False, 13.5, 0x665544) for char in text] |
| 109 | new = [(char, True, 18 + number % 9, 0x563412) if 0 < i < len(text)-1 else old[i] for i, char in enumerate(text)] |
| 110 | assert target not in documents, 'Two insertion intents share an object identity' |
| 111 | created, changed = linked[inserted['id']], linked[formatted['id']] |
| 112 | assert created['finished_us'] <= changed['started_us'], 'Formatting preceded its insertion' |
| 113 | assert characters(created['documents'][target]) == old, 'Insertion publication differs from its local intent' |
| 114 | assert characters(changed['documents'][target]) == new, 'Formatting publication differs from its local intent' |
| 115 | states = {'insert': {'characters': old, 'attempt': created}, |
| 116 | 'format': {'characters': new, 'attempt': changed}} |
| 117 | if replaced: |
| 118 | replacement, *boundaries = replaced |
| 119 | end = len(text.encode('utf-16-le')) // 2 |
| 120 | assert replacement['object'] == target and replacement['space'] == inserted['space'], 'Text edit addresses another object' |
| 121 | assert replacement['text'] == text and replacement['range'] == [end-3, end], 'Cross-run text range differs from workload' |
| 122 | assert replacement['replacement'] == ' e\u0301🐈', 'Cross-run replacement differs from workload' |
| 123 | final = new[:-2] + [(char, *new[-2][1:]) for char in replacement['replacement']] |
| 124 | attempt = linked[replacement['id']] |
| 125 | assert changed['finished_us'] <= attempt['started_us'], 'Text replacement preceded its formatting' |
| 126 | assert characters(attempt['documents'][target]) == final, 'Text publication differs from its local intent' |
| 127 | states['text'] = {'characters': final, 'attempt': attempt} |
| 128 | paragraph = identity(insertion, 3 if 'Outline' in insertion['placement'] else 1) |
| 129 | for state in states.values(): |
| 130 | state['parts'] = [(paragraph, target, 0, len(state['characters']))] |
| 131 | if replaced and boundaries: |
| 132 | split, right_edit, *following = boundaries |
| 133 | join = next(row for row in following if row['kind'] == 'join') |
| 134 | assert events[0].get('document_graph') is True, 'Boundary workload requires structural observations' |
| 135 | assert all(row.get('document') == target and row['space'] == inserted['space'] |
| 136 | for row in edits[at:at + len(kinds)]), 'Boundary operation lost its owning document' |
| 137 | intent = split['split'] |
| 138 | right, right_paragraph = identity(intent, 2), identity(intent, 1) |
| 139 | assert split['object'] == intent['text'] == target and intent['author'] == insertion['author'], 'Split addresses another text or author' |
| 140 | assert intent['offset'] == end-2 and split['range'] == [end-2, end-2], 'Split boundary differs from workload' |
| 141 | boundary = len(text)-1 |
| 142 | assert right_edit['object'] == right and right_edit['range'] == [0, 2] and right_edit['replacement'] == 'B🦋', 'Dependent right edit differs from workload' |
| 143 | assert join['object'] == target and join['joined'] == [target, right], 'Join does not retain its original targets' |
| 144 | updated = final[:boundary] + [(char, *final[boundary][1:]) for char in 'B🦋'] + final[boundary+2:] |
| 145 | split_parts = [(paragraph, target, 0, boundary), (right_paragraph, right, boundary, len(final))] |
| 146 | edited_parts = [(paragraph, target, 0, boundary), (right_paragraph, right, boundary, len(updated))] |
| 147 | stages = [(split, final, split_parts, {}), (right_edit, updated, edited_parts, {})] |
| 148 | if len(following) > 1: |
| 149 | nest, unnest, _, tail_split, deleted = following |
| 150 | outline = identity(insertion, 1) if 'Outline' in insertion['placement'] else insertion['parent'] |
| 151 | for row, parent in [(nest, paragraph), (unnest, outline)]: |
| 152 | tree = row['tree'] |
| 153 | assert row['object'] == target and tree['object'] == right_paragraph and tree['author'] == insertion['author'], 'Tree move addresses another subtree or author' |
| 154 | assert tree['placement'] == {'Move': {'parent': parent, 'before': None}}, 'Tree destination differs from workload' |
| 155 | stages += [(nest, updated, edited_parts, {right_paragraph: paragraph}), |
| 156 | (unnest, updated, edited_parts, {})] |
| 157 | stages.append((join, updated, [(paragraph, target, 0, len(updated))], {})) |
| 158 | if len(following) > 1: |
| 159 | tail = tail_split['split'] |
| 160 | tail_text, tail_paragraph = identity(tail, 2), identity(tail, 1) |
| 161 | assert tail_split['object'] == tail['text'] == target and tail['author'] == insertion['author'], 'Tail split addresses another text or author' |
| 162 | assert tail['offset'] == end+1 and tail_split['range'] == [end+1, end+1], 'Tail split boundary differs from workload' |
| 163 | tree = deleted['tree'] |
| 164 | assert deleted['object'] == tail_text and tree['object'] == tail_paragraph and tree['author'] == insertion['author'], 'Deletion addresses another subtree or author' |
| 165 | assert tree['placement'] == 'Delete', 'Deletion became a move' |
| 166 | stages += [(tail_split, updated, [(paragraph, target, 0, len(updated)-1), (tail_paragraph, tail_text, len(updated)-1, len(updated))], {}), |
| 167 | (deleted, updated[:-1], [(paragraph, target, 0, len(updated)-1)], {})] |
| 168 | known_texts = {oid for _, _, pieces, _ in stages for _, oid, _, _ in pieces} |
| 169 | for row, value, parts, parents in stages: |
| 170 | attempt = linked[row['id']] |
| 171 | assert list(states.values())[-1]['attempt']['finished_us'] <= attempt['started_us'], 'Boundary publication preceded its dependency' |
| 172 | assert set(attempt['documents']) & known_texts == {oid for _, oid, _, _ in parts}, 'Boundary publication omitted or resurrected a text object' |
| 173 | for _, oid, start, stop in parts: |
| 174 | assert characters(attempt['documents'][oid]) == value[start:stop], 'Boundary publication differs from its local intent' |
| 175 | states[row['kind']] = {'characters': value, 'parts': parts, 'parents': parents, 'attempt': attempt} |
| 176 | allocated = {oid for state in states.values() for paragraph, text, _, _ in state['parts'] for oid in (paragraph, text)} |
| 177 | existing = {oid for document in documents.values() for state in document['states'].values() |
| 178 | for paragraph, text, _, _ in state['parts'] for oid in (paragraph, text)} |
| 179 | assert not allocated & existing, 'Two document operations share allocated identities' |
| 180 | documents[target] = {'insertion': insertion, 'space': inserted['space'], 'states': states} |
| 181 | assert documents, 'No document operations were recorded' |
| 182 | structural = any(events[0].get('document_graph') for events in logs.values()) |
| 183 | allowed_texts = {oid for document in documents.values() for state in document['states'].values() |
| 184 | for _, oid, _, _ in state['parts']} |
| 185 | for actor, events in logs.items(): |
| 186 | if structural: |
| 187 | assert events[0].get('document_graph') is True, 'Client omitted structural observations' |
| 188 | before, previous = None, {} |
| 189 | reads = [row for row in events if row['event'] in ('read', 'document_read') and row.get('documents') is not None] |
| 190 | assert reads, f'{actor} did not observe document snapshots' |
| 191 | assert any(row['documents'] for row in reads), f'{actor} never observed a created document object' |
| 192 | for row in events: |
| 193 | if row['event'] not in ('read', 'document_read', 'remote_attempt'): continue |
| 194 | is_read = row['event'] != 'remote_attempt' |
| 195 | observed = row.get('documents') |
| 196 | assert isinstance(observed, dict), 'Snapshot omitted document text' |
| 197 | assert set(previous) <= set(observed) <= allowed_texts, 'Reader lost an object or observed an unrecorded insertion' |
| 198 | graph = row.get('document_graph') |
| 199 | if structural: assert isinstance(graph, dict), 'Snapshot omitted document graph' |
| 200 | expected_graph = {} |
| 201 | for target, document in documents.items(): |
| 202 | states = list(document['states'].values()) |
| 203 | known_texts = {oid for state in states for _, oid, _, _ in state['parts']} |
| 204 | known_paragraphs = {oid for state in states for oid, _, _, _ in state['parts']} |
| 205 | insertion = document['insertion'] |
| 206 | outline = identity(insertion, 1) if 'Outline' in insertion['placement'] else insertion['parent'] |
| 207 | if is_read and row['started_us'] > states[0]['attempt']['acknowledged_us']: |
| 208 | assert target in observed, 'Reader missed an acknowledged insertion' |
| 209 | if target not in observed: |
| 210 | assert not known_texts & observed.keys(), 'Snapshot omitted the original paragraph text' |
| 211 | continue |
| 212 | actual = {oid: characters(observed[oid]) for oid in known_texts & observed.keys()} |
| 213 | matches = [i for i, state in enumerate(states) |
| 214 | if actual == {oid: state['characters'][start:stop] for _, oid, start, stop in state['parts']} |
| 215 | and (not structural or (known_paragraphs & graph.keys() == {oid for oid, _, _, _ in state['parts']} |
| 216 | and all(graph[p]['parent'] == state.get('parents', {}).get(p, outline) |
| 217 | for p, _, _, _ in state['parts'])))] |
| 218 | assert matches, 'Reader observed partial or invented document content' |
| 219 | if is_read: |
| 220 | matches = [i for i in matches if i >= previous.get(target, 0)] |
| 221 | assert matches, 'Reader reverted document content' |
| 222 | matches = [i for i in matches if row['finished_us'] >= states[i]['attempt']['started_us']] |
| 223 | assert matches, 'Reader observed future document content' |
| 224 | acknowledged = max((i for i, state in enumerate(states) if row['started_us'] > state['attempt']['acknowledged_us']), default=0) |
| 225 | matches = [i for i in matches if i >= acknowledged] |
| 226 | assert matches, 'Reader missed acknowledged document content' |
| 227 | previous[target] = min(matches) |
| 228 | if structural: |
| 229 | if 'Outline' in insertion['placement']: |
| 230 | expected_graph[outline] = {'parent': insertion['parent'], 'children': [], 'content': [], |
| 231 | 'child_level': 1, 'position': insertion['placement']['Outline']} |
| 232 | assert outline in expected_graph, 'Snapshot omitted the inserted outline' |
| 233 | for paragraph, oid, _, _ in states[min(matches)]['parts']: |
| 234 | parent = states[min(matches)].get('parents', {}).get(paragraph, outline) |
| 235 | expected_graph[parent]['children'].append(paragraph) |
| 236 | expected_graph[paragraph] = {'parent': parent, 'children': [], 'content': [oid], |
| 237 | 'child_level': 1, 'position': None} |
| 238 | if structural: |
| 239 | assert graph == expected_graph, 'Snapshot contains partial, reordered or invented paragraph structure' |
| 240 | if not is_read: |
| 241 | assert before is not None, 'Publication omitted its observed source' |
| 242 | for image, delta in [('documents', 'document_changes'), ('document_graph', 'document_graph_changes')]: |
| 243 | expected_delta = {key: row[image].get(key) for key in before[image].keys() | row[image].keys() |
| 244 | if before[image].get(key) != row[image].get(key)} |
| 245 | assert row.get(delta) == expected_delta, 'Publication diff omitted or invented a changed object' |
| 246 | if is_read: before = row |
| 247 | return documents |
| 248 | |
| 249 | |
| 250 | def verify_model(model, documents): |
| 251 | children = {} |
| 252 | for document in documents.values(): |
| 253 | insertion = document['insertion'] |
| 254 | if 'Outline' in insertion['placement']: |
| 255 | children[identity(insertion, 1)] = [identity(insertion, 3)] |
| 256 | for document in documents.values(): |
| 257 | insertion = document['insertion'] |
| 258 | if 'Paragraph' in insertion['placement']: |
| 259 | children[insertion['parent']].append(identity(insertion, 1)) |
| 260 | retired = {oid for document in documents.values() for state in document['states'].values() |
| 261 | for paragraph, text, _, _ in state['parts'] for oid in (paragraph, text)} - { |
| 262 | oid for document in documents.values() for paragraph, text, _, _ in list(document['states'].values())[-1]['parts'] |
| 263 | for oid in (paragraph, text)} |
| 264 | found = set() |
| 265 | for sid, _, revision, page in ordered_pages(model): |
| 266 | nodes = revision['nodes'] |
| 267 | for target, node in walk(revision, page): |
| 268 | assert target not in retired, 'Final model resurrected a retired paragraph or text' |
| 269 | if target not in documents: continue |
| 270 | assert target not in found, 'Inserted text is reachable twice' |
| 271 | found.add(target) |
| 272 | expected = documents[target] |
| 273 | insertion = expected['insertion'] |
| 274 | final = list(expected['states'].values())[-1]['characters'] |
| 275 | assert sid == expected['space'] and node['kind']['text'] == ''.join(char for char, *_ in final) |
| 276 | object_id = identity(insertion, 1) |
| 277 | assert object_id in nodes[insertion['parent']]['children'], 'Insertion lost its parent' |
| 278 | paragraph = identity(insertion, 3) if 'Outline' in insertion['placement'] else object_id |
| 279 | assert not nodes[paragraph]['children'], 'Final paragraph gained an unrecorded child' |
| 280 | assert nodes[paragraph]['content'] == [target], 'Inserted paragraph content changed' |
| 281 | if 'Outline' in insertion['placement']: |
| 282 | position = insertion['placement']['Outline'] |
| 283 | assert all(nodes[object_id]['layout'][key] == position[key] for key in ('x', 'y')), 'Outline coordinates changed' |
| 284 | assert nodes[object_id]['children'] == children[object_id], 'Inserted paragraph order changed' |
| 285 | assert found == set(documents), 'Final model omitted an inserted object' |
| 286 | |
| 287 | |
| 288 | def verify_native(paragraphs, expected): |
| 289 | from PIL import ImageColor |
| 290 | by_text = {''.join(char for char, _ in paragraph): paragraph for paragraph in paragraphs} |
| 291 | checks = 0 |
| 292 | for final in expected: |
| 293 | actual = by_text[''.join(char for char, *_ in final)] |
| 294 | for (char, style), (wanted, bold, size, color) in zip(actual, final, strict=True): |
| 295 | assert char == wanted and bool(style.get('bold')) == bold, 'Native text or bold differs from intent' |
| 296 | assert style.get('font_size', 11) == size, 'Native font size differs from intent' |
| 297 | native_color = style.get('color', 'automatic') |
| 298 | if color == 0xff000000: |
| 299 | assert native_color in ('automatic', None), 'Native automatic color changed' |
| 300 | else: |
| 301 | assert ImageColor.getrgb(native_color) == (color & 255, (color >> 8) & 255, (color >> 16) & 255), 'Native color differs from intent' |
| 302 | checks += 3 |
| 303 | return checks |