Files
pdfplumber/tests/test_dedupe_chars.py
T
2023-02-07 17:06:59 -05:00

75 lines
3.0 KiB
Python

#!/usr/bin/env python
import logging
import os
import unittest
import pdfplumber
logging.disable(logging.ERROR)
HERE = os.path.abspath(os.path.dirname(__file__))
class Test(unittest.TestCase):
@classmethod
def setup_class(self):
path = os.path.join(HERE, "pdfs/issue-71-duplicate-chars.pdf")
self.pdf = pdfplumber.open(path)
@classmethod
def teardown_class(self):
self.pdf.close()
def test_extract_table(self):
page = self.pdf.pages[0]
table_without_drop_duplicates = page.extract_table()
table_with_drop_duplicates = page.dedupe_chars().extract_table()
last_line_without_drop = table_without_drop_duplicates[1][1].split("\n")[-1]
last_line_with_drop = table_with_drop_duplicates[1][1].split("\n")[-1]
assert last_line_without_drop == "微微软软 培培训训课课程程:: 名名模模意意义义一一些些有有意意义义一一些些"
assert last_line_with_drop == "微软 培训课程: 名模意义一些有意义一些"
def test_extract_words(self):
page = self.pdf.pages[0]
x0 = 440.143
x1_without_drop = 534.992
x1_with_drop = 534.719
top_windows = 791.849
top_linux = 794.357
bottom = 802.961
last_words_without_drop = page.extract_words()[-1]
last_words_with_drop = page.dedupe_chars().extract_words()[-1]
assert round(last_words_without_drop["x0"], 3) == x0
assert round(last_words_without_drop["x1"], 3) == x1_without_drop
assert round(last_words_without_drop["top"], 3) in (top_windows, top_linux)
assert round(last_words_without_drop["bottom"], 3) == bottom
assert last_words_without_drop["upright"] == 1
assert last_words_without_drop["text"] == "名名模模意意义义一一些些有有意意义义一一些些"
assert round(last_words_with_drop["x0"], 3) == x0
assert round(last_words_with_drop["x1"], 3) == x1_with_drop
assert round(last_words_with_drop["top"], 3) in (top_windows, top_linux)
assert round(last_words_with_drop["bottom"], 3) == bottom
assert last_words_with_drop["upright"] == 1
assert last_words_with_drop["text"] == "名模意义一些有意义一些"
def test_extract_text(self):
page = self.pdf.pages[0]
last_line_without_drop = page.extract_text().split("\n")[-1]
last_line_with_drop = page.dedupe_chars().extract_text().split("\n")[-1]
assert last_line_without_drop == "微微软软 培培训训课课程程:: 名名模模意意义义一一些些有有意意义义一一些些"
assert last_line_with_drop == "微软 培训课程: 名模意义一些有意义一些"
def test_extract_text2(self):
path = os.path.join(HERE, "pdfs/issue-71-duplicate-chars-2.pdf")
pdf = pdfplumber.open(path)
page = pdf.pages[0]
assert (
page.dedupe_chars().extract_text(y_tolerance=6).splitlines()[4]
== "UE 8. Circulation - Métabolismes"
)