1import unittest
2from types import SimpleNamespace
3
4from compare import pdf_line_ends
5from range_stress import check
6
7
8def page(lines):
9 return SimpleNamespace(chars=[
10 {'text': character, 'x0': x, 'y0': baseline - 2, 'y1': baseline + 8, 'matrix': (1, 0, 0, 1, x, baseline)}
11 for baseline, text in lines for x, character in enumerate(text)
12 ])
13
14
15class PdfRanges(unittest.TestCase):
16 def test_preserves_source_spaces_at_wraps(self):
17 native = page([(700, 'Title'), (600, 'one two'), (580, 'three'), (30, 'Footer')])
18 self.assertEqual(pdf_line_ends(native, ' one two three '), [13, 20])
19
20 def test_style_baseline_offsets_stay_on_the_same_visual_line(self):
21 native = page([(600, 'a'), (599.83, 'b'), (580, 'c')])
22 native.chars[1]['x0'] = 1
23 self.assertEqual(pdf_line_ends(native, 'ab c'), [3, 4])
24
25 def test_long_word_has_no_invented_whitespace(self):
26 self.assertEqual(pdf_line_ends(page([(600, 'extra'), (580, 'ordinary')]), 'extraordinary'), [5, 13])
27
28 def test_missing_pdf_characters_are_not_accepted(self):
29 self.assertIsNone(pdf_line_ends(page([(600, 'one tree')]), 'one three'))
30
31 def test_ambiguous_match_is_not_selected_arbitrarily(self):
32 self.assertIsNone(pdf_line_ends(page([(600, 'same'), (580, 'same')]), 'same'))
33
34 def test_non_ascii_requires_other_evidence(self):
35 self.assertIsNone(pdf_line_ends(page([(600, 'café')]), 'café'))
36
37
38class SourceCoverage(unittest.TestCase):
39 def test_utf16_offsets_preserve_supplementary_characters(self):
40 source = [{'id': 'emoji', 'runs': [{'text': 'a🌳b'}]}]
41 output = {'cases': [{'id': 'emoji', 'lines': [
42 {'start_utf16': 0, 'end_utf16': 3, 'text': 'a🌳'},
43 {'start_utf16': 3, 'end_utf16': 4, 'text': 'b'},
44 ]}]}
45 self.assertEqual(check(source, output), 2)
46 output['cases'][0]['lines'][0]['end_utf16'] = 2
47 with self.assertRaises(UnicodeDecodeError):
48 check(source, output)
49
50 def test_dropped_leading_space_is_detected(self):
51 source = [{'id': 'space', 'runs': [{'text': ' a'}]}]
52 output = {'cases': [{'id': 'space', 'lines': [
53 {'start_utf16': 1, 'end_utf16': 2, 'text': 'a'},
54 ]}]}
55 with self.assertRaisesRegex(ValueError, 'noncontiguous'):
56 check(source, output)
57
58 def test_empty_paragraph_covers_an_empty_source(self):
59 source = [{'id': 'empty', 'runs': [{'text': ''}]}]
60 output = {'cases': [{'id': 'empty', 'lines': [
61 {'start_utf16': 0, 'end_utf16': 0, 'text': ''},
62 ]}]}
63 self.assertEqual(check(source, output), 1)
64
65
66if __name__ == '__main__':
67 unittest.main()