| 1 | from importlib.util import find_spec |
| 2 | from pathlib import Path |
| 3 | import runpy |
| 4 | import unittest |
| 5 | from types import SimpleNamespace |
| 6 | from unittest.mock import patch |
| 7 | |
| 8 | verify = runpy.run_path(str(Path(__file__).with_name('verify-document.py')))['verify_pdf_black'] |
| 9 | |
| 10 | requires_pdfplumber = unittest.skipUnless( |
| 11 | find_spec('pdfplumber'), 'pdfplumber is not installed; tools/TESTING.md sets up the Python lane') |
| 12 | |
| 13 | |
| 14 | @requires_pdfplumber |
| 15 | class PdfFormatOracleTest(unittest.TestCase): |
| 16 | def test_unmapped_text_cannot_hide_missing_text_or_incorrect_highlighting(self): |
| 17 | chars = [{'text': c, 'x0': i * 2, 'x1': i * 2 + 1, 'top': 0, 'bottom': 1} |
| 18 | for i, c in enumerate('BeforeSilk after tail') if not c.isspace()] |
| 19 | rect = {'fill': True, 'non_stroking_color': 0, 'x0': 12, 'x1': 19, 'top': 0, 'bottom': 1} |
| 20 | page = SimpleNamespace(page_number=1, chars=chars, rects=[rect]) |
| 21 | runs = [{'text': text, 'format': {'hidden': False, 'highlight': highlight}} |
| 22 | for text, highlight in [('Before', None), ('Silk', 0), (' after e\u0301 tail', None)]] |
| 23 | with patch('pdfplumber.open') as opened: |
| 24 | opened.return_value.__enter__.return_value.pages = [page] |
| 25 | self.assertEqual(verify('unused', [runs]), 1) |
| 26 | page.rects.append({**rect, 'x0': 32, 'x1': 35}) |
| 27 | with self.assertRaisesRegex(AssertionError, 'unhighlighted unmapped text'): |
| 28 | verify('unused', [runs]) |
| 29 | page.rects.pop() |
| 30 | runs[2]['text'] = ' after ordinary tail' |
| 31 | with self.assertRaisesRegex(AssertionError, 'ambiguous'): |
| 32 | verify('unused', [runs]) |
| 33 | runs[2].update(text=' after e\u0301 tail', format={'hidden': False, 'highlight': 0}) |
| 34 | with self.assertRaisesRegex(AssertionError, 'visual verification'): |
| 35 | verify('unused', [runs]) |
| 36 | runs[2]['format']['highlight'] = None |
| 37 | page.chars += chars |
| 38 | with self.assertRaisesRegex(AssertionError, 'ambiguous'): |
| 39 | verify('unused', [runs]) |
| 40 | page.chars = [{**chars[0], 'text': 'a', 'x0': i * 2, 'x1': i * 2 + 1} for i in range(3)] |
| 41 | page.rects = [{**rect, 'x0': 0, 'x1': 5}] |
| 42 | with self.assertRaisesRegex(AssertionError, 'ambiguous'): |
| 43 | verify('unused', [[{'text': 'aa', 'format': {'hidden': False, 'highlight': 0}}]]) |
| 44 | |
| 45 | def test_native_pdf_locates_partial_black_by_paragraph_and_character(self): |
| 46 | pdf = Path(__file__).resolve().parent.parent / 'corpus/m6/native-structure-01/read/page-001.pdf' |
| 47 | runs = [{'text': text, 'format': {'hidden': False, 'highlight': highlight}} |
| 48 | for text, highlight in [('Before ', None), ('middle', 0), (' after', None)]] |
| 49 | self.assertEqual(verify(pdf, [runs]), 1) |
| 50 | runs[0]['text'] = 'Before m' |
| 51 | runs[1]['text'] = 'iddle ' |
| 52 | runs[2]['text'] = 'after' |
| 53 | with self.assertRaisesRegex(AssertionError, 'character positions'): |
| 54 | verify(pdf, [runs]) |
| 55 | |
| 56 | |
| 57 | if __name__ == '__main__': |
| 58 | unittest.main() |