1from importlib.util import find_spec
2from pathlib import Path
3import runpy
4import unittest
5from types import SimpleNamespace
6from unittest.mock import patch
7
8verify = runpy.run_path(str(Path(__file__).with_name('verify-document.py')))['verify_pdf_black']
9
10requires_pdfplumber = unittest.skipUnless(
11 find_spec('pdfplumber'), 'pdfplumber is not installed; tools/TESTING.md sets up the Python lane')
12
13
14@requires_pdfplumber
15class PdfFormatOracleTest(unittest.TestCase):
16 def test_unmapped_text_cannot_hide_missing_text_or_incorrect_highlighting(self):
17 chars = [{'text': c, 'x0': i * 2, 'x1': i * 2 + 1, 'top': 0, 'bottom': 1}
18 for i, c in enumerate('BeforeSilk after tail') if not c.isspace()]
19 rect = {'fill': True, 'non_stroking_color': 0, 'x0': 12, 'x1': 19, 'top': 0, 'bottom': 1}
20 page = SimpleNamespace(page_number=1, chars=chars, rects=[rect])
21 runs = [{'text': text, 'format': {'hidden': False, 'highlight': highlight}}
22 for text, highlight in [('Before', None), ('Silk', 0), (' after e\u0301 tail', None)]]
23 with patch('pdfplumber.open') as opened:
24 opened.return_value.__enter__.return_value.pages = [page]
25 self.assertEqual(verify('unused', [runs]), 1)
26 page.rects.append({**rect, 'x0': 32, 'x1': 35})
27 with self.assertRaisesRegex(AssertionError, 'unhighlighted unmapped text'):
28 verify('unused', [runs])
29 page.rects.pop()
30 runs[2]['text'] = ' after ordinary tail'
31 with self.assertRaisesRegex(AssertionError, 'ambiguous'):
32 verify('unused', [runs])
33 runs[2].update(text=' after e\u0301 tail', format={'hidden': False, 'highlight': 0})
34 with self.assertRaisesRegex(AssertionError, 'visual verification'):
35 verify('unused', [runs])
36 runs[2]['format']['highlight'] = None
37 page.chars += chars
38 with self.assertRaisesRegex(AssertionError, 'ambiguous'):
39 verify('unused', [runs])
40 page.chars = [{**chars[0], 'text': 'a', 'x0': i * 2, 'x1': i * 2 + 1} for i in range(3)]
41 page.rects = [{**rect, 'x0': 0, 'x1': 5}]
42 with self.assertRaisesRegex(AssertionError, 'ambiguous'):
43 verify('unused', [[{'text': 'aa', 'format': {'hidden': False, 'highlight': 0}}]])
44
45 def test_native_pdf_locates_partial_black_by_paragraph_and_character(self):
46 pdf = Path(__file__).resolve().parent.parent / 'corpus/m6/native-structure-01/read/page-001.pdf'
47 runs = [{'text': text, 'format': {'hidden': False, 'highlight': highlight}}
48 for text, highlight in [('Before ', None), ('middle', 0), (' after', None)]]
49 self.assertEqual(verify(pdf, [runs]), 1)
50 runs[0]['text'] = 'Before m'
51 runs[1]['text'] = 'iddle '
52 runs[2]['text'] = 'after'
53 with self.assertRaisesRegex(AssertionError, 'character positions'):
54 verify(pdf, [runs])
55
56
57if __name__ == '__main__':
58 unittest.main()