| 1 | #!/usr/bin/env python3 |
| 2 | """Generate and compare independently specified native edit histories.""" |
| 3 | import argparse |
| 4 | import copy |
| 5 | import json |
| 6 | from pathlib import Path |
| 7 | import random |
| 8 | import shutil |
| 9 | import signal |
| 10 | import subprocess |
| 11 | import tempfile |
| 12 | import xml.etree.ElementTree as ET |
| 13 | |
| 14 | from document_model import EXPORTER, ordered_pages |
| 15 | from native_xml import ns, Text, project_text |
| 16 | |
| 17 | ROOT = Path(__file__).resolve().parent.parent |
| 18 | |
| 19 | |
| 20 | def reduce_history(history): |
| 21 | state = copy.deepcopy(history['initial']) |
| 22 | for op in history['operations']: |
| 23 | kind = op['kind'] |
| 24 | if kind == 'insert': state['paragraphs'].insert(op['index'], copy.deepcopy(op['paragraph'])) |
| 25 | elif kind == 'delete': state['paragraphs'].pop(op['index']) |
| 26 | elif kind == 'move': state['paragraphs'].insert(op['to'], state['paragraphs'].pop(op['index'])) |
| 27 | elif kind == 'position': state['x'], state['y'] = op['x'], op['y'] |
| 28 | elif kind == 'cell': state['table'][op['row']][op['column']] = op['text'] |
| 29 | else: state['paragraphs'][op['index']][kind] = op['value'] |
| 30 | return state |
| 31 | |
| 32 | |
| 33 | def generate(destination, count, start=0): |
| 34 | destination.mkdir(parents=True, exist_ok=False) |
| 35 | notebook = destination / 'notebook' |
| 36 | shutil.copytree(ROOT / 'corpus/writer/create-notebook-01/notebook', notebook) |
| 37 | histories = [] |
| 38 | samples = ['Repeated words', 'café 東京 مرحبا', 'A😀e\u0301Z', 'literal <b> & "quotes"', 'tabs\there', ' ]]> end', 'simple text'] |
| 39 | def paragraph(identifier, text): |
| 40 | return {'id': identifier, 'text': text, 'bold': False, 'italic': False, 'underline': False, |
| 41 | 'strike': False, 'color': '#000000', 'highlight': '#ffff00', 'font_size': 11, |
| 42 | 'link': '', 'tag': False} |
| 43 | kinds = ['text', 'bold', 'italic', 'underline', 'strike', 'color', 'highlight', 'font_size', 'link', 'tag', 'insert', 'delete', 'move', 'position', 'cell'] |
| 44 | for seed in range(start, start + count): |
| 45 | rng = random.Random(seed) |
| 46 | history = {'seed': seed, 'title': f'History {seed:06}', 'initial': { |
| 47 | 'paragraphs': [paragraph(f'p{i}', f'Initial {i}') for i in range(4)], |
| 48 | 'table': [[f'Cell {row},{column}' for column in range(2)] for row in range(2)], 'x': 72, 'y': 108}, 'operations': []} |
| 49 | for step in range(20): |
| 50 | state = reduce_history(history) |
| 51 | kind = kinds[step % len(kinds)] if seed < 3 else rng.choice(kinds) |
| 52 | index = rng.randrange(len(state['paragraphs'])) |
| 53 | op = {'kind': kind, 'index': index} |
| 54 | if kind == 'insert': op['paragraph'] = paragraph(f'new{step}', rng.choice(samples)) |
| 55 | elif kind == 'delete' and len(state['paragraphs']) == 1: op = {'kind': 'text', 'index': 0, 'value': rng.choice(samples)} |
| 56 | elif kind == 'move': op['to'] = rng.randrange(len(state['paragraphs'])) |
| 57 | elif kind == 'position': op.update(x=rng.randrange(1, 7) * 36, y=rng.randrange(2, 10) * 36) |
| 58 | elif kind == 'cell': op.update(row=rng.randrange(2), column=rng.randrange(2), text=rng.choice(samples)) |
| 59 | elif kind == 'text': op['value'] = rng.choice(samples) |
| 60 | elif kind in ('bold', 'italic', 'underline', 'strike', 'tag'): op['value'] = not state['paragraphs'][index][kind] |
| 61 | elif kind in ('color', 'highlight'): op['value'] = rng.choice(['#000000', '#ff0000', '#00ffff', '#ffff00', '#00ff00']) |
| 62 | elif kind == 'font_size': op['value'] = rng.choice([9, 11, 14, 18]) |
| 63 | elif kind == 'link': op['value'] = '' if state['paragraphs'][index]['link'] else f'https://example.invalid/{seed}/{step}?a=1&b=2' |
| 64 | history['operations'].append(op) |
| 65 | histories.append(history) |
| 66 | (notebook / 'histories.json').write_text(json.dumps(histories, ensure_ascii=False, indent=2)) |
| 67 | |
| 68 | |
| 69 | def compare(capture, seed=None): |
| 70 | histories = json.loads((capture / 'notebook/histories.json').read_text()) |
| 71 | if seed is not None: |
| 72 | histories = [h for h in histories if h['seed'] == seed] |
| 73 | if len(histories) != 1: |
| 74 | raise ValueError('The capture must contain exactly one history with this seed.') |
| 75 | native = {ET.parse(p).getroot().get('name'): ET.parse(p).getroot() for p in (capture / 'read').glob('page-*.xml')} |
| 76 | with tempfile.TemporaryDirectory() as temporary: |
| 77 | export = Path(temporary) / 'model' |
| 78 | subprocess.run([EXPORTER, capture / 'notebook/synthetic.one', export], check=True) |
| 79 | document = json.loads((export / 'document.json').read_text()) |
| 80 | resolved = json.loads((export / 'text.json').read_text()) |
| 81 | actual = {} |
| 82 | for sid, rid, space, oid in ordered_pages(document): |
| 83 | title = ''.join(n['kind']['text'] for n in space['nodes'].values() if n['kind']['type'] == 'RichText' and n['kind']['text'].startswith('History ')) |
| 84 | if title: actual[title] = (sid, rid, space, oid) |
| 85 | failures = [] |
| 86 | for history in histories: |
| 87 | try: |
| 88 | expected = reduce_history(history) |
| 89 | title = history['title']; page = native[title] |
| 90 | sid, rid, space, oid = actual[title] |
| 91 | outlines = [space['nodes'][i] for i in space['nodes'][oid]['children'] if space['nodes'][i]['kind']['type'] == 'Outline'] |
| 92 | assert len(outlines) == 1, (title, 'outline count') |
| 93 | native_outline = page.find('one:Outline', ns) |
| 94 | native_oes = native_outline.findall('one:OEChildren/one:OE', ns) |
| 95 | native_text = [n.find('one:T', ns) for n in native_oes if n.find('one:T', ns) is not None] |
| 96 | paragraphs = [space['nodes'][i] for i in outlines[0]['children']] |
| 97 | content = [p['content'][0] for p in paragraphs] |
| 98 | text_ids = [i for i in content if space['nodes'][i]['kind']['type'] == 'RichText'] |
| 99 | assert len(text_ids) == len(native_text) == len(expected['paragraphs']), (title, 'paragraph count') |
| 100 | for wanted, oid, native_t in zip(expected['paragraphs'], text_ids, native_text, strict=True): |
| 101 | runs = [r for r in resolved[sid][rid][oid] if not r['format']['hidden']] |
| 102 | assert ''.join(r['text'] for r in runs) == wanted['text'], (title, wanted['id'], 'Rust text') |
| 103 | assert ''.join(Text(native_t.text or '').parts) == project_text(wanted['text']), (title, wanted['id'], 'native text') |
| 104 | leading = len(wanted['text']) - len(wanted['text'].lstrip(' \t')) |
| 105 | actual_links = [r['link'] or '' for r in runs for _ in r['text']] |
| 106 | expected_links = ['' if index < leading else wanted['link'] for index in range(len(wanted['text']))] |
| 107 | assert actual_links == expected_links, (title, wanted['id'], 'hyperlink', actual_links, expected_links) |
| 108 | for r in runs: |
| 109 | if not r['text']: |
| 110 | continue |
| 111 | for flag in ('bold', 'italic', 'underline', 'strike'): |
| 112 | assert bool(r['format'][flag]) == (False if flag == 'underline' and wanted['link'] else wanted[flag]), (title, wanted['id'], flag) |
| 113 | assert r['format']['font_size'] == wanted['font_size'], (title, wanted['id'], 'font size') |
| 114 | for field in ('color', 'highlight'): |
| 115 | value = '#dbdb00' if field == 'color' and wanted[field] == '#ffff00' else wanted[field] |
| 116 | rgb = bytes.fromhex(value[1:]) |
| 117 | assert r['format'][field] == (0xff000000 if field == 'color' and wanted['link'] else int.from_bytes(rgb, 'little')), (title, wanted['id'], field, r['format'][field], wanted[field]) |
| 118 | tags = space['nodes'][oid]['tags'] |
| 119 | assert bool(tags) == wanted['tag'], (title, wanted['id'], 'tag') |
| 120 | table = [space['nodes'][i] for i in content if space['nodes'][i]['kind']['type'] == 'Table'] |
| 121 | assert len(table) == 1, (title, 'table count') |
| 122 | cells = [[ ''.join(r['text'] for paragraph in space['nodes'][cell]['children'] |
| 123 | for text in space['nodes'][paragraph]['content'] for r in resolved[sid][rid][text] if not r['format']['hidden']) |
| 124 | for cell in space['nodes'][row]['children']] for row in table[0]['children']] |
| 125 | assert cells == expected['table'], (title, 'table contents') |
| 126 | for axis in ('x', 'y'): |
| 127 | assert abs(outlines[0]['layout'][axis] - expected[axis]) < .002, (title, axis) |
| 128 | except AssertionError as failure: |
| 129 | failures.append(failure.args[0]) |
| 130 | (capture / 'history-differences.json').write_text(json.dumps(failures, ensure_ascii=False, indent=2)) |
| 131 | if failures: |
| 132 | raise AssertionError(failures[0]) |
| 133 | print(f'Passed {len(histories)} independently specified native histories, {sum(len(h["operations"]) for h in histories)} operations.') |
| 134 | |
| 135 | |
| 136 | def shrink(capture, destination, seed): |
| 137 | from native_runner import capture as native_capture |
| 138 | histories = json.loads((capture / 'notebook/histories.json').read_text()) |
| 139 | history, = [h for h in histories if h['seed'] == seed] |
| 140 | try: |
| 141 | compare(capture, seed) |
| 142 | except AssertionError as failure: |
| 143 | signature = failure.args[0][:3] |
| 144 | else: |
| 145 | raise ValueError('The selected history passes; there is no discrepancy to shrink.') |
| 146 | destination.mkdir(parents=True, exist_ok=False) |
| 147 | (destination / 'source.json').write_text(json.dumps({'capture': str(capture.resolve()), 'seed': seed, |
| 148 | 'signature': signature}, indent=2)) |
| 149 | attempt = 0 |
| 150 | |
| 151 | def reproduces(candidate): |
| 152 | nonlocal attempt |
| 153 | try: |
| 154 | reduce_history(candidate) |
| 155 | except IndexError: |
| 156 | return False |
| 157 | attempt += 1 |
| 158 | root = destination / f'attempt-{attempt:03}' |
| 159 | root.mkdir() |
| 160 | shutil.copytree(ROOT / 'corpus/writer/create-notebook-01/notebook', root / 'input') |
| 161 | (root / 'input/histories.json').write_text(json.dumps([candidate], ensure_ascii=False, indent=2)) |
| 162 | native_capture(root / 'input', root / 'capture', 2, ROOT / 'tools/native/history.ps1') |
| 163 | observed = None |
| 164 | try: |
| 165 | compare(root / 'capture', seed) |
| 166 | except AssertionError as failure: |
| 167 | observed = failure.args[0] |
| 168 | matches = observed is not None and observed[:3] == signature |
| 169 | (root / 'comparison.json').write_text(json.dumps({'discrepancy': observed, 'same_failure': matches}, indent=2)) |
| 170 | print(f'Shrink attempt {attempt}: {len(candidate["operations"])} operations; same discrepancy: {matches}', flush=True) |
| 171 | return matches |
| 172 | |
| 173 | if not reproduces(history): |
| 174 | raise ValueError('The original discrepancy did not reproduce in a fresh clone.') |
| 175 | granularity = 2 |
| 176 | while history['operations']: |
| 177 | operations = history['operations'] |
| 178 | width = max(1, (len(operations) + granularity - 1) // granularity) |
| 179 | for start in range(0, len(operations), width): |
| 180 | candidate = {**history, 'operations': operations[:start] + operations[start + width:]} |
| 181 | if reproduces(candidate): |
| 182 | history = candidate |
| 183 | granularity = max(2, granularity - 1) |
| 184 | break |
| 185 | else: |
| 186 | if width == 1: |
| 187 | break |
| 188 | granularity = min(len(operations), granularity * 2) |
| 189 | (destination / 'minimal.json').write_text(json.dumps(history, ensure_ascii=False, indent=2)) |
| 190 | for index, operation in enumerate(history['operations']): |
| 191 | choices = [] |
| 192 | if operation['kind'] == 'text': choices = [{'value': ''}, {'value': 'A'}] |
| 193 | elif operation['kind'] == 'cell': choices = [{'text': ''}, {'text': 'A'}] |
| 194 | elif operation['kind'] == 'position': choices = [{'x': 0, 'y': 0}] |
| 195 | elif operation['kind'] == 'font_size': choices = [{'value': 11}] |
| 196 | for replacement in choices: |
| 197 | candidate = copy.deepcopy(history) |
| 198 | candidate['operations'][index].update(replacement) |
| 199 | if candidate != history and reproduces(candidate): |
| 200 | history = candidate |
| 201 | break |
| 202 | (destination / 'minimal.json').write_text(json.dumps(history, ensure_ascii=False, indent=2)) |
| 203 | |
| 204 | |
| 205 | if __name__ == '__main__': |
| 206 | parser = argparse.ArgumentParser(description=__doc__) |
| 207 | sub = parser.add_subparsers(dest='command', required=True) |
| 208 | generation = sub.add_parser('generate'); generation.add_argument('destination', type=Path) |
| 209 | generation.add_argument('--count', type=int, default=100); generation.add_argument('--start', type=int, default=0) |
| 210 | comparison = sub.add_parser('compare'); comparison.add_argument('capture', type=Path) |
| 211 | comparison.add_argument('--seed', type=int) |
| 212 | shrinking = sub.add_parser('shrink'); shrinking.add_argument('capture', type=Path) |
| 213 | shrinking.add_argument('destination', type=Path); shrinking.add_argument('--seed', type=int, required=True) |
| 214 | args = parser.parse_args() |
| 215 | if args.command == 'generate': generate(args.destination, args.count, args.start) |
| 216 | elif args.command == 'compare': compare(args.capture, args.seed) |
| 217 | else: |
| 218 | def interrupted(_signal, _frame): |
| 219 | raise KeyboardInterrupt |
| 220 | signal.signal(signal.SIGTERM, interrupted) |
| 221 | shrink(args.capture, args.destination, args.seed) |