mirror of
https://github.com/jsvine/pdfplumber.git
synced 2026-08-29 16:40:24 +08:00
87b947f8f6
Per discussion at https://github.com/jsvine/pdfplumber/discussions/346 and input from @ramcdona, this commit changes pdfplumber's behavior regarding floating point numbers. Specifically, it removes all conversion of floats to Decimal objects. This brings several advantages: - Increased precision (where applicable) - Decreased code complexity - Increased performance (~10% speedup on test suite) - Increased fidelity to `pdfminer.six` output These seem to outweigh the disadvantages: - Some tests break (but have been easily fixed) due to increased precision and/or floating point arithmetic artifacts - Some users' scripts may also break, if they depend on strict equality testing, though these *should* also be easily fixable Because some form of automatic rounding may still be desirable for the pdfplumber CLI utility, the conversion methods (.to_csv, .to_json) have been adjusted to accept a `precision` argument.
97 lines
2.7 KiB
Python
97 lines
2.7 KiB
Python
#!/usr/bin/env python
|
|
import unittest
|
|
import pdfplumber
|
|
from subprocess import Popen, PIPE
|
|
from io import StringIO
|
|
import json
|
|
import sys
|
|
import os
|
|
|
|
import logging
|
|
|
|
logging.disable(logging.ERROR)
|
|
|
|
HERE = os.path.abspath(os.path.dirname(__file__))
|
|
|
|
|
|
def run(cmd):
|
|
return Popen(cmd, stdout=PIPE).communicate()[0]
|
|
|
|
|
|
class Test(unittest.TestCase):
|
|
@classmethod
|
|
def setup_class(self):
|
|
self.path = os.path.join(HERE, "pdfs/pdffill-demo.pdf")
|
|
self.pdf = pdfplumber.open(self.path, pages=[1, 2, 5])
|
|
|
|
@classmethod
|
|
def teardown_class(self):
|
|
self.pdf.close()
|
|
|
|
def test_json(self):
|
|
c = json.loads(self.pdf.to_json())
|
|
assert (
|
|
c["pages"][0]["rects"][0]["bottom"] == self.pdf.pages[0].rects[0]["bottom"]
|
|
)
|
|
|
|
def test_json_all_types(self):
|
|
c = json.loads(self.pdf.to_json(types=None))
|
|
found_types = c["pages"][0].keys()
|
|
assert "curves" in found_types
|
|
assert "chars" in found_types
|
|
assert "lines" in found_types
|
|
assert "rects" in found_types
|
|
assert "images" in found_types
|
|
|
|
def test_single_pages(self):
|
|
c = json.loads(self.pdf.pages[0].to_json())
|
|
assert c["rects"][0]["bottom"] == self.pdf.pages[0].rects[0]["bottom"]
|
|
|
|
def test_additional_attr_types(self):
|
|
path = os.path.join(HERE, "pdfs/issue-67-example.pdf")
|
|
with pdfplumber.open(path, pages=[1]) as pdf:
|
|
c = json.loads(pdf.to_json())
|
|
assert len(c["pages"][0]["images"])
|
|
|
|
def test_csv(self):
|
|
c = self.pdf.to_csv(precision=3)
|
|
assert c.split("\r\n")[9] == (
|
|
"char,1,45.83,58.826,656.82,674.82,117.18,117.18,135.18,12.996,"
|
|
'18.0,12.996,,,,,,TimesNewRomanPSMT,,,,"(0, 0, 0)",,,18.0,,,,,Y,,1,'
|
|
)
|
|
|
|
io = StringIO()
|
|
self.pdf.to_csv(io, precision=3)
|
|
io.seek(0)
|
|
c_from_io = io.read()
|
|
assert c == c_from_io
|
|
|
|
def test_csv_all_types(self):
|
|
c = self.pdf.to_csv(types=None)
|
|
assert c.split("\r\n")[1].split(",")[0] == "line"
|
|
|
|
def test_cli(self):
|
|
res = run(
|
|
[
|
|
sys.executable,
|
|
"-m",
|
|
"pdfplumber.cli",
|
|
self.path,
|
|
"--format",
|
|
"json",
|
|
"--pages",
|
|
"1-2",
|
|
"5",
|
|
"--indent",
|
|
"2",
|
|
]
|
|
)
|
|
|
|
c = json.loads(res)
|
|
assert c["pages"][0]["page_number"] == 1
|
|
assert c["pages"][1]["page_number"] == 2
|
|
assert c["pages"][2]["page_number"] == 5
|
|
assert c["pages"][0]["rects"][0]["bottom"] == float(
|
|
self.pdf.pages[0].rects[0]["bottom"]
|
|
)
|