| 1 | import unittest |
| 2 | from types import SimpleNamespace |
| 3 | |
| 4 | from compare import pdf_lines |
| 5 | from compare_page_pdf import compare |
| 6 | |
| 7 | |
| 8 | def probe(texts, ends=None): |
| 9 | return {'title': 'Test', 'objects': [{'kind': 'outline', 'is_title': False, 'id': 'outline', |
| 10 | 'paragraphs': [{'id': str(i), 'visible_text': text, 'origin': [0, i * 12], |
| 11 | 'spans': [{'end_utf8': len(text.encode()), 'format': {'font_size': 10}}], |
| 12 | 'lines': [{'text': text, 'end_utf16': len(text) if ends is None else ends[i], |
| 13 | 'baseline': 8}]} |
| 14 | for i, text in enumerate(texts)]}]} |
| 15 | |
| 16 | |
| 17 | def page(lines): |
| 18 | return SimpleNamespace(height=800, chars=[ |
| 19 | {'text': ch, 'x0': x * 5, 'y0': y - 2, 'y1': y + 8, 'size': 10, |
| 20 | 'matrix': (1, 0, 0, 1, x * 5, y)} |
| 21 | for y, text in lines for x, ch in enumerate(text)]) |
| 22 | |
| 23 | |
| 24 | class PagePdf(unittest.TestCase): |
| 25 | def test_title_opt_in_requires_its_own_native_glyphs(self): |
| 26 | source = probe(['body']) |
| 27 | title = probe(['title'])['objects'][0] |
| 28 | title.update(id='title', is_title=True) |
| 29 | source['objects'].append(title) |
| 30 | native = page([(700, 'title'), (680, 'body')]) |
| 31 | self.assertEqual(compare(source, [native])['counts'], {'matched': 1}) |
| 32 | result = compare(source, [native], include_titles=True) |
| 33 | self.assertEqual(result['counts'], {'matched': 2}) |
| 34 | self.assertTrue(result['paragraphs'][1]['is_title']) |
| 35 | missing = compare(source, [page([(680, 'body')])], include_titles=True) |
| 36 | self.assertEqual(missing['paragraphs'][1]['status'], 'unresolved_match') |
| 37 | |
| 38 | def test_outline_context_resolves_repeated_paragraphs(self): |
| 39 | result = compare(probe(['same', 'same']), [page([(700, 'same'), (680, 'same')])]) |
| 40 | self.assertEqual(result['counts'], {'matched': 2}) |
| 41 | self.assertEqual([p['native_instances'][0]['baseline_ranges_from_pdf_top'] |
| 42 | for p in result['paragraphs']], [[[100, 100]], [[120, 120]]]) |
| 43 | |
| 44 | def test_different_wraps_are_reported(self): |
| 45 | result = compare(probe(['one two']), [page([(700, 'one'), (680, 'two')])]) |
| 46 | row = result['paragraphs'][0] |
| 47 | self.assertEqual(row['status'], 'different_wraps') |
| 48 | self.assertEqual(row['native_end_utf16'], [4, 7]) |
| 49 | self.assertFalse(row['breaks_match']) |
| 50 | |
| 51 | def test_repeated_exports_must_agree(self): |
| 52 | source = probe(['one two']) |
| 53 | one_line = page([(700, 'one two')]) |
| 54 | matching = compare(source, [one_line, one_line])['paragraphs'][0] |
| 55 | self.assertTrue(matching['breaks_match']) |
| 56 | self.assertEqual(len(matching['native_instances']), 2) |
| 57 | different = compare(source, [one_line, page([(700, 'one'), (680, 'two')])])['paragraphs'][0] |
| 58 | self.assertEqual(different['status'], 'unrecoverable_line_ranges') |
| 59 | self.assertIsNone(different['breaks_match']) |
| 60 | |
| 61 | def test_incomplete_outline_does_not_use_ambiguous_source_substrings(self): |
| 62 | result = compare(probe(['one', 'one other']), [page([(700, 'one extra one other')])]) |
| 63 | self.assertEqual(result['paragraphs'][0]['status'], 'unresolved_match') |
| 64 | self.assertEqual(result['paragraphs'][0]['match_context'], 'ambiguous_source_context') |
| 65 | |
| 66 | def test_empty_text_is_not_a_geometry_pass(self): |
| 67 | result = compare(probe(['', ' ', '\u00a0']), [page([])]) |
| 68 | self.assertEqual(result['counts'], {'no_visible_glyphs': 3}) |
| 69 | self.assertTrue(all(p['breaks_match'] is None for p in result['paragraphs'])) |
| 70 | |
| 71 | def test_overlapping_line_candidates_are_not_chosen_arbitrarily(self): |
| 72 | native = page([(700, 'a'), (680, 'b'), (660, 'c')]) |
| 73 | native.chars[-1]['y1'] = 720 |
| 74 | self.assertIsNone(pdf_lines(native.chars)) |
| 75 | |
| 76 | |
| 77 | if __name__ == '__main__': |
| 78 | unittest.main() |