2020-08-01 12:27:07 -04:00
|
|
|
|
#!/usr/bin/env python
|
2021-12-09 22:27:51 -05:00
|
|
|
|
import logging
|
|
|
|
|
|
import os
|
2022-05-13 16:04:03 -04:00
|
|
|
|
import re
|
2020-08-01 12:27:07 -04:00
|
|
|
|
import unittest
|
2021-12-09 22:27:51 -05:00
|
|
|
|
from itertools import groupby
|
|
|
|
|
|
from operator import itemgetter
|
|
|
|
|
|
|
2022-05-06 09:43:50 -04:00
|
|
|
|
import pandas as pd
|
2020-08-01 12:27:07 -04:00
|
|
|
|
import pytest
|
|
|
|
|
|
from pdfminer.pdfparser import PDFObjRef
|
|
|
|
|
|
from pdfminer.psparser import PSLiteral
|
|
|
|
|
|
|
2021-12-09 22:27:51 -05:00
|
|
|
|
import pdfplumber
|
|
|
|
|
|
from pdfplumber import utils
|
2020-12-16 22:19:17 -05:00
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
logging.disable(logging.ERROR)
|
|
|
|
|
|
|
|
|
|
|
|
HERE = os.path.abspath(os.path.dirname(__file__))
|
|
|
|
|
|
|
|
|
|
|
|
|
2020-12-16 22:19:17 -05:00
|
|
|
|
class Test(unittest.TestCase):
|
2020-08-01 12:27:07 -04:00
|
|
|
|
@classmethod
|
|
|
|
|
|
def setup_class(self):
|
2022-05-13 16:04:03 -04:00
|
|
|
|
self.pdf = pdfplumber.open(os.path.join(HERE, "pdfs/pdffill-demo.pdf"))
|
|
|
|
|
|
self.pdf_scotus = pdfplumber.open(
|
|
|
|
|
|
os.path.join(HERE, "pdfs/scotus-transcript-p1.pdf")
|
|
|
|
|
|
)
|
2020-08-01 12:27:07 -04:00
|
|
|
|
|
|
|
|
|
|
@classmethod
|
|
|
|
|
|
def teardown_class(self):
|
|
|
|
|
|
self.pdf.close()
|
|
|
|
|
|
|
|
|
|
|
|
def test_cluster_list(self):
|
|
|
|
|
|
a = [1, 2, 3, 4]
|
|
|
|
|
|
assert utils.cluster_list(a) == [[x] for x in a]
|
|
|
|
|
|
assert utils.cluster_list(a, tolerance=1) == [a]
|
|
|
|
|
|
|
|
|
|
|
|
a = [1, 2, 5, 6]
|
|
|
|
|
|
assert utils.cluster_list(a, tolerance=1) == [[1, 2], [5, 6]]
|
|
|
|
|
|
|
|
|
|
|
|
def test_cluster_objects(self):
|
|
|
|
|
|
a = ["a", "ab", "abc", "b"]
|
|
|
|
|
|
assert utils.cluster_objects(a, len, 0) == [["a", "b"], ["ab"], ["abc"]]
|
|
|
|
|
|
|
2022-10-01 09:35:42 -04:00
|
|
|
|
b = [{"x": 1, 7: "a"}, {"x": 1, 7: "b"}, {"x": 2, 7: "b"}, {"x": 2, 7: "b"}]
|
|
|
|
|
|
assert utils.cluster_objects(b, "x", 0) == [[b[0], b[1]], [b[2], b[3]]]
|
|
|
|
|
|
assert utils.cluster_objects(b, 7, 0) == [[b[0]], [b[1], b[2], b[3]]]
|
|
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
def test_resolve(self):
|
|
|
|
|
|
annot = self.pdf.annots[0]
|
|
|
|
|
|
annot_ad0 = utils.resolve(annot["data"]["A"]["D"][0])
|
|
|
|
|
|
assert annot_ad0["MediaBox"] == [0, 0, 612, 792]
|
2020-08-13 08:24:41 -04:00
|
|
|
|
assert utils.resolve(1) == 1
|
2020-08-01 12:27:07 -04:00
|
|
|
|
|
|
|
|
|
|
def test_resolve_all(self):
|
|
|
|
|
|
info = self.pdf.doc.xrefs[0].trailer["Info"]
|
|
|
|
|
|
assert type(info) == PDFObjRef
|
2020-12-16 22:19:17 -05:00
|
|
|
|
a = [{"info": info}]
|
2020-08-01 12:27:07 -04:00
|
|
|
|
a_res = utils.resolve_all(a)
|
|
|
|
|
|
assert a_res[0]["info"]["Producer"] == self.pdf.doc.info[0]["Producer"]
|
|
|
|
|
|
|
|
|
|
|
|
def test_decode_psl_list(self):
|
2020-12-16 22:19:17 -05:00
|
|
|
|
a = [PSLiteral("test"), "test_2"]
|
2020-08-01 12:27:07 -04:00
|
|
|
|
assert utils.decode_psl_list(a) == ["test", "test_2"]
|
|
|
|
|
|
|
|
|
|
|
|
def test_extract_words(self):
|
|
|
|
|
|
path = os.path.join(HERE, "pdfs/issue-192-example.pdf")
|
|
|
|
|
|
with pdfplumber.open(path) as pdf:
|
|
|
|
|
|
p = pdf.pages[0]
|
|
|
|
|
|
words = p.extract_words(vertical_ttb=False)
|
2020-12-16 22:19:17 -05:00
|
|
|
|
words_attr = p.extract_words(vertical_ttb=False, extra_attrs=["size"])
|
2020-08-29 16:00:41 -04:00
|
|
|
|
words_w_spaces = p.extract_words(vertical_ttb=False, keep_blank_chars=True)
|
2020-08-01 12:27:07 -04:00
|
|
|
|
words_rtl = p.extract_words(horizontal_ltr=False)
|
|
|
|
|
|
|
|
|
|
|
|
assert words[0]["text"] == "Agaaaaa:"
|
2020-08-30 22:30:14 -04:00
|
|
|
|
assert words[0]["direction"] == 1
|
2020-08-29 16:00:41 -04:00
|
|
|
|
|
|
|
|
|
|
assert "size" not in words[0]
|
2021-02-07 16:33:54 -05:00
|
|
|
|
assert round(words_attr[0]["size"], 2) == 9.96
|
2020-08-29 16:00:41 -04:00
|
|
|
|
|
|
|
|
|
|
assert words_w_spaces[0]["text"] == "Agaaaaa: AAAA"
|
|
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
vertical = [w for w in words if w["upright"] == 0]
|
|
|
|
|
|
assert vertical[0]["text"] == "Aaaaaabag8"
|
2020-08-30 22:30:14 -04:00
|
|
|
|
assert vertical[0]["direction"] == -1
|
2020-08-29 16:00:41 -04:00
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
assert words_rtl[1]["text"] == "baaabaaA/AAA"
|
2020-08-30 22:30:14 -04:00
|
|
|
|
assert words_rtl[1]["direction"] == -1
|
2020-08-01 12:27:07 -04:00
|
|
|
|
|
2022-07-16 06:58:10 -07:00
|
|
|
|
def test_extract_words_punctuation(self):
|
|
|
|
|
|
path = os.path.join(HERE, "pdfs/test-punkt.pdf")
|
|
|
|
|
|
with pdfplumber.open(path) as pdf:
|
|
|
|
|
|
|
|
|
|
|
|
wordsA = pdf.pages[0].extract_words(split_at_punctuation=True)
|
|
|
|
|
|
wordsB = pdf.pages[0].extract_words(split_at_punctuation=False)
|
|
|
|
|
|
wordsC = pdf.pages[0].extract_words(
|
|
|
|
|
|
split_at_punctuation=r"!\"&'()*+,.:;<=>?@[]^`{|}~"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert wordsA[0]["text"] == "https"
|
|
|
|
|
|
assert (
|
|
|
|
|
|
wordsB[0]["text"]
|
|
|
|
|
|
== "https://dell-research-harvard.github.io/HJDataset/"
|
|
|
|
|
|
)
|
|
|
|
|
|
assert wordsC[2]["text"] == "//dell-research-harvard"
|
|
|
|
|
|
|
|
|
|
|
|
wordsA = pdf.pages[1].extract_words(split_at_punctuation=True)
|
|
|
|
|
|
wordsB = pdf.pages[1].extract_words(split_at_punctuation=False)
|
|
|
|
|
|
wordsC = pdf.pages[1].extract_words(
|
|
|
|
|
|
split_at_punctuation=r"!\"&'()*+,.:;<=>?@[]^`{|}~"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert len(wordsA) == 4
|
|
|
|
|
|
assert len(wordsB) == 2
|
|
|
|
|
|
assert len(wordsC) == 2
|
|
|
|
|
|
|
|
|
|
|
|
wordsA = pdf.pages[2].extract_words(split_at_punctuation=True)
|
|
|
|
|
|
wordsB = pdf.pages[2].extract_words(split_at_punctuation=False)
|
|
|
|
|
|
wordsC = pdf.pages[2].extract_words(
|
|
|
|
|
|
split_at_punctuation=r"!\"&'()*+,.:;<=>?@[]^`{|}~"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert wordsA[1]["text"] == "["
|
|
|
|
|
|
assert wordsB[1]["text"] == "[2,"
|
|
|
|
|
|
assert wordsC[1]["text"] == "["
|
|
|
|
|
|
|
|
|
|
|
|
wordsA = pdf.pages[3].extract_words(split_at_punctuation=True)
|
|
|
|
|
|
wordsB = pdf.pages[3].extract_words(split_at_punctuation=False)
|
|
|
|
|
|
wordsC = pdf.pages[3].extract_words(
|
|
|
|
|
|
split_at_punctuation=r"!\"&'()*+,.:;<=>?@[]^`{|}~"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert wordsA[2]["text"] == "al"
|
|
|
|
|
|
assert wordsB[2]["text"] == "al."
|
|
|
|
|
|
assert wordsC[2]["text"] == "al"
|
|
|
|
|
|
|
2022-07-17 12:51:07 -04:00
|
|
|
|
def test_extract_text_punctuation(self):
|
|
|
|
|
|
path = os.path.join(HERE, "pdfs/test-punkt.pdf")
|
|
|
|
|
|
with pdfplumber.open(path) as pdf:
|
|
|
|
|
|
text = pdf.pages[0].extract_text(
|
|
|
|
|
|
layout=True,
|
|
|
|
|
|
split_at_punctuation=True,
|
|
|
|
|
|
)
|
|
|
|
|
|
assert "https " in text
|
|
|
|
|
|
|
2020-08-30 18:16:50 -04:00
|
|
|
|
def test_text_flow(self):
|
|
|
|
|
|
path = os.path.join(HERE, "pdfs/federal-register-2020-17221.pdf")
|
|
|
|
|
|
|
|
|
|
|
|
def words_to_text(words):
|
2020-12-16 22:19:17 -05:00
|
|
|
|
grouped = groupby(words, key=itemgetter("top"))
|
|
|
|
|
|
lines = [" ".join(word["text"] for word in grp) for top, grp in grouped]
|
2020-08-30 18:16:50 -04:00
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
|
|
|
|
|
|
with pdfplumber.open(path) as pdf:
|
|
|
|
|
|
p0 = pdf.pages[0]
|
2020-12-16 22:19:17 -05:00
|
|
|
|
using_flow = p0.extract_words(use_text_flow=True)
|
2020-08-30 18:16:50 -04:00
|
|
|
|
not_using_flow = p0.extract_words()
|
|
|
|
|
|
|
|
|
|
|
|
target_text = (
|
|
|
|
|
|
"The FAA proposes to\n"
|
|
|
|
|
|
"supersede Airworthiness Directive (AD)\n"
|
|
|
|
|
|
"2018–23–51, which applies to all The\n"
|
|
|
|
|
|
"Boeing Company Model 737–8 and 737–\n"
|
|
|
|
|
|
"9 (737 MAX) airplanes. Since AD 2018–\n"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert target_text in words_to_text(using_flow)
|
|
|
|
|
|
assert target_text not in words_to_text(not_using_flow)
|
|
|
|
|
|
|
2023-07-01 18:41:58 -04:00
|
|
|
|
def test_text_flow_overlapping(self):
|
|
|
|
|
|
path = os.path.join(HERE, "pdfs/issue-912.pdf")
|
|
|
|
|
|
|
|
|
|
|
|
with pdfplumber.open(path) as pdf:
|
|
|
|
|
|
p0 = pdf.pages[0]
|
|
|
|
|
|
using_flow = p0.extract_text(use_text_flow=True, layout=True, x_tolerance=1)
|
|
|
|
|
|
not_using_flow = p0.extract_text(layout=True, x_tolerance=1)
|
|
|
|
|
|
|
|
|
|
|
|
assert re.search("2015 RICE PAYMENT 26406576 0 1207631 Cr", using_flow)
|
|
|
|
|
|
assert re.search("124644,06155766", using_flow) is None
|
|
|
|
|
|
|
|
|
|
|
|
assert re.search("124644,06155766", not_using_flow)
|
|
|
|
|
|
assert (
|
|
|
|
|
|
re.search("2015 RICE PAYMENT 26406576 0 1207631 Cr", not_using_flow) is None
|
|
|
|
|
|
)
|
|
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
def test_extract_text(self):
|
|
|
|
|
|
text = self.pdf.pages[0].extract_text()
|
2020-12-16 22:29:51 -05:00
|
|
|
|
goal_lines = [
|
|
|
|
|
|
"First Page Previous Page Next Page Last Page",
|
|
|
|
|
|
"Print",
|
|
|
|
|
|
"PDFill: PDF Drawing",
|
|
|
|
|
|
"You can open a PDF or create a blank PDF by PDFill.",
|
|
|
|
|
|
"Online Help",
|
|
|
|
|
|
"Here are the PDF drawings created by PDFill",
|
|
|
|
|
|
"Please save into a new PDF to see the effect!",
|
|
|
|
|
|
"Goto Page 2: Line Tool",
|
|
|
|
|
|
"Goto Page 3: Arrow Tool",
|
|
|
|
|
|
"Goto Page 4: Tool for Rectangle, Square and Rounded Corner",
|
|
|
|
|
|
"Goto Page 5: Tool for Circle, Ellipse, Arc, Pie",
|
|
|
|
|
|
"Goto Page 6: Tool for Basic Shapes",
|
|
|
|
|
|
"Goto Page 7: Tool for Curves",
|
|
|
|
|
|
"Here are the tools to change line width, style, arrow style and colors",
|
|
|
|
|
|
]
|
|
|
|
|
|
goal = "\n".join(goal_lines)
|
2020-08-01 12:27:07 -04:00
|
|
|
|
|
|
|
|
|
|
assert text == goal
|
2023-02-07 17:06:59 -05:00
|
|
|
|
|
|
|
|
|
|
text_simple = self.pdf.pages[0].extract_text_simple()
|
|
|
|
|
|
assert text_simple == goal
|
|
|
|
|
|
|
2021-10-22 18:51:32 -04:00
|
|
|
|
assert self.pdf.pages[0].crop((0, 0, 1, 1)).extract_text() == ""
|
|
|
|
|
|
|
2023-02-07 17:06:59 -05:00
|
|
|
|
def test_extract_text_blank(self):
|
|
|
|
|
|
assert utils.extract_text([]) == ""
|
|
|
|
|
|
|
2021-10-22 18:51:32 -04:00
|
|
|
|
def test_extract_text_layout(self):
|
2023-02-13 17:38:40 -05:00
|
|
|
|
target = (
|
|
|
|
|
|
open(os.path.join(HERE, "comparisons/scotus-transcript-p1.txt"))
|
|
|
|
|
|
.read()
|
|
|
|
|
|
.strip("\n")
|
|
|
|
|
|
)
|
2022-05-13 16:04:03 -04:00
|
|
|
|
page = self.pdf_scotus.pages[0]
|
|
|
|
|
|
text = page.extract_text(layout=True)
|
2023-02-13 17:38:40 -05:00
|
|
|
|
utils_text = utils.extract_text(
|
|
|
|
|
|
page.chars, layout=True, layout_width=page.width, layout_height=page.height
|
|
|
|
|
|
)
|
2022-05-13 16:04:03 -04:00
|
|
|
|
assert text == utils_text
|
2021-10-22 18:51:32 -04:00
|
|
|
|
assert text == target
|
|
|
|
|
|
|
|
|
|
|
|
def test_extract_text_layout_cropped(self):
|
2023-02-13 17:38:40 -05:00
|
|
|
|
target = (
|
|
|
|
|
|
open(os.path.join(HERE, "comparisons/scotus-transcript-p1-cropped.txt"))
|
|
|
|
|
|
.read()
|
|
|
|
|
|
.strip("\n")
|
|
|
|
|
|
)
|
2022-05-13 16:04:03 -04:00
|
|
|
|
p = self.pdf_scotus.pages[0]
|
2021-10-22 18:51:32 -04:00
|
|
|
|
cropped = p.crop((90, 70, p.width, 300))
|
|
|
|
|
|
text = cropped.extract_text(layout=True)
|
|
|
|
|
|
assert text == target
|
2020-08-01 12:27:07 -04:00
|
|
|
|
|
2023-02-13 17:38:40 -05:00
|
|
|
|
def test_extract_text_layout_widths(self):
|
|
|
|
|
|
p = self.pdf_scotus.pages[0]
|
|
|
|
|
|
text = p.extract_text(layout=True, layout_width_chars=75)
|
|
|
|
|
|
assert all(len(line) == 75 for line in text.splitlines())
|
|
|
|
|
|
with pytest.raises(ValueError):
|
|
|
|
|
|
p.extract_text(layout=True, layout_width=300, layout_width_chars=50)
|
|
|
|
|
|
with pytest.raises(ValueError):
|
|
|
|
|
|
p.extract_text(layout=True, layout_height=300, layout_height_chars=50)
|
|
|
|
|
|
|
2022-05-27 14:41:57 -04:00
|
|
|
|
def test_extract_text_nochars(self):
|
|
|
|
|
|
charless = self.pdf.pages[0].filter(lambda df: df["object_type"] != "char")
|
|
|
|
|
|
assert charless.extract_text() == ""
|
|
|
|
|
|
assert charless.extract_text(layout=True) == ""
|
|
|
|
|
|
|
2022-05-13 16:04:03 -04:00
|
|
|
|
def test_search_regex_compiled(self):
|
|
|
|
|
|
page = self.pdf_scotus.pages[0]
|
|
|
|
|
|
pat = re.compile(r"supreme\s+(\w+)", re.I)
|
|
|
|
|
|
results = page.search(pat)
|
|
|
|
|
|
assert results[0]["text"] == "SUPREME COURT"
|
|
|
|
|
|
assert results[0]["groups"] == ("COURT",)
|
|
|
|
|
|
assert results[1]["text"] == "Supreme Court"
|
|
|
|
|
|
assert results[1]["groups"] == ("Court",)
|
|
|
|
|
|
|
|
|
|
|
|
with pytest.raises(ValueError):
|
|
|
|
|
|
page.search(re.compile(r"x"), regex=False)
|
|
|
|
|
|
|
|
|
|
|
|
with pytest.raises(ValueError):
|
|
|
|
|
|
page.search(re.compile(r"x"), case=False)
|
|
|
|
|
|
|
|
|
|
|
|
def test_search_regex_uncompiled(self):
|
|
|
|
|
|
page = self.pdf_scotus.pages[0]
|
|
|
|
|
|
pat = r"supreme\s+(\w+)"
|
|
|
|
|
|
results = page.search(pat, case=False)
|
|
|
|
|
|
assert results[0]["text"] == "SUPREME COURT"
|
|
|
|
|
|
assert results[0]["groups"] == ("COURT",)
|
|
|
|
|
|
assert results[1]["text"] == "Supreme Court"
|
|
|
|
|
|
assert results[1]["groups"] == ("Court",)
|
|
|
|
|
|
|
|
|
|
|
|
def test_search_string(self):
|
|
|
|
|
|
page = self.pdf_scotus.pages[0]
|
|
|
|
|
|
results = page.search("SUPREME COURT", regex=False)
|
|
|
|
|
|
assert results[0]["text"] == "SUPREME COURT"
|
|
|
|
|
|
assert results[0]["groups"] == tuple()
|
|
|
|
|
|
|
|
|
|
|
|
results = page.search("supreme court", regex=False)
|
|
|
|
|
|
assert len(results) == 0
|
|
|
|
|
|
|
|
|
|
|
|
results = page.search("supreme court", regex=False, case=False)
|
|
|
|
|
|
assert len(results) == 2
|
|
|
|
|
|
|
|
|
|
|
|
results = page.search("supreme court", regex=True, case=False)
|
|
|
|
|
|
assert len(results) == 2
|
|
|
|
|
|
|
|
|
|
|
|
results = page.search(r"supreme\s+(\w+)", regex=False)
|
|
|
|
|
|
assert len(results) == 0
|
|
|
|
|
|
|
2023-04-12 14:10:10 -04:00
|
|
|
|
results = page.search(r"10 Tuesday", layout=False)
|
|
|
|
|
|
assert len(results) == 1
|
|
|
|
|
|
|
|
|
|
|
|
results = page.search(r"10 Tuesday", layout=True)
|
|
|
|
|
|
assert len(results) == 0
|
|
|
|
|
|
|
|
|
|
|
|
def test_extract_text_lines(self):
|
|
|
|
|
|
page = self.pdf_scotus.pages[0]
|
|
|
|
|
|
results = page.extract_text_lines()
|
|
|
|
|
|
assert len(results) == 28
|
|
|
|
|
|
assert "chars" in results[0]
|
|
|
|
|
|
assert results[0]["text"] == "Official - Subject to Final Review"
|
|
|
|
|
|
|
|
|
|
|
|
alt = page.extract_text_lines(layout=True, strip=False, return_chars=False)
|
|
|
|
|
|
assert "chars" not in alt[0]
|
|
|
|
|
|
assert (
|
|
|
|
|
|
alt[0]["text"]
|
|
|
|
|
|
== " Official - Subject to Final Review " # noqa: E501
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert results[10]["text"] == "10 Tuesday, January 13, 2009"
|
|
|
|
|
|
assert (
|
|
|
|
|
|
alt[10]["text"]
|
|
|
|
|
|
== " 10 Tuesday, January 13, 2009 " # noqa: E501
|
|
|
|
|
|
)
|
|
|
|
|
|
assert (
|
|
|
|
|
|
page.extract_text_lines(layout=True)[10]["text"]
|
|
|
|
|
|
== "10 Tuesday, January 13, 2009"
|
|
|
|
|
|
) # noqa: E501
|
|
|
|
|
|
|
2023-04-07 16:02:47 -04:00
|
|
|
|
def test_handle_empty_and_whitespace_search_results(self):
|
|
|
|
|
|
# via https://github.com/jsvine/pdfplumber/discussions/853
|
|
|
|
|
|
# The searches below should not raise errors but instead
|
|
|
|
|
|
# should return empty result-sets.
|
|
|
|
|
|
page = self.pdf_scotus.pages[0]
|
|
|
|
|
|
for regex in [True, False]:
|
|
|
|
|
|
results = page.search("\n", regex=regex)
|
|
|
|
|
|
assert len(results) == 0
|
|
|
|
|
|
|
|
|
|
|
|
assert len(page.search("(sdfsd)?")) == 0
|
|
|
|
|
|
assert len(page.search("")) == 0
|
|
|
|
|
|
|
2020-08-14 21:10:17 -04:00
|
|
|
|
def test_intersects_bbox(self):
|
|
|
|
|
|
objs = [
|
|
|
|
|
|
# Is same as bbox
|
2020-12-16 22:19:17 -05:00
|
|
|
|
{
|
2020-08-14 21:10:17 -04:00
|
|
|
|
"x0": 0,
|
|
|
|
|
|
"top": 0,
|
|
|
|
|
|
"x1": 20,
|
|
|
|
|
|
"bottom": 20,
|
|
|
|
|
|
},
|
|
|
|
|
|
# Inside bbox
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 10,
|
|
|
|
|
|
"top": 10,
|
|
|
|
|
|
"x1": 15,
|
|
|
|
|
|
"bottom": 15,
|
|
|
|
|
|
},
|
|
|
|
|
|
# Overlaps bbox
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 10,
|
|
|
|
|
|
"top": 10,
|
|
|
|
|
|
"x1": 30,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
},
|
|
|
|
|
|
# Touching on one side
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 20,
|
|
|
|
|
|
"top": 0,
|
|
|
|
|
|
"x1": 40,
|
|
|
|
|
|
"bottom": 20,
|
|
|
|
|
|
},
|
|
|
|
|
|
# Touching on one corner
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 20,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"x1": 40,
|
|
|
|
|
|
"bottom": 40,
|
|
|
|
|
|
},
|
|
|
|
|
|
# Fully outside
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 21,
|
|
|
|
|
|
"top": 21,
|
|
|
|
|
|
"x1": 40,
|
|
|
|
|
|
"bottom": 40,
|
|
|
|
|
|
},
|
|
|
|
|
|
]
|
|
|
|
|
|
bbox = utils.obj_to_bbox(objs[0])
|
|
|
|
|
|
|
|
|
|
|
|
assert utils.intersects_bbox(objs, bbox) == objs[:4]
|
|
|
|
|
|
|
2023-04-08 09:34:52 -04:00
|
|
|
|
def test_merge_bboxes(self):
|
|
|
|
|
|
bboxes = [
|
|
|
|
|
|
(0, 10, 20, 20),
|
|
|
|
|
|
(10, 5, 10, 30),
|
|
|
|
|
|
]
|
|
|
|
|
|
merged = utils.merge_bboxes(bboxes)
|
|
|
|
|
|
assert merged == (0, 5, 20, 30)
|
|
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
def test_resize_object(self):
|
|
|
|
|
|
obj = {
|
|
|
|
|
|
"x0": 5,
|
|
|
|
|
|
"x1": 10,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
"width": 5,
|
|
|
|
|
|
"height": 10,
|
|
|
|
|
|
"doctop": 120,
|
|
|
|
|
|
"y0": 40,
|
|
|
|
|
|
"y1": 50,
|
|
|
|
|
|
}
|
|
|
|
|
|
assert utils.resize_object(obj, "x0", 0) == {
|
|
|
|
|
|
"x0": 0,
|
|
|
|
|
|
"x1": 10,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"doctop": 120,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
"width": 10,
|
|
|
|
|
|
"height": 10,
|
|
|
|
|
|
"y0": 40,
|
|
|
|
|
|
"y1": 50,
|
|
|
|
|
|
}
|
|
|
|
|
|
assert utils.resize_object(obj, "x1", 50) == {
|
|
|
|
|
|
"x0": 5,
|
|
|
|
|
|
"x1": 50,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"doctop": 120,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
"width": 45,
|
|
|
|
|
|
"height": 10,
|
|
|
|
|
|
"y0": 40,
|
|
|
|
|
|
"y1": 50,
|
|
|
|
|
|
}
|
|
|
|
|
|
assert utils.resize_object(obj, "top", 0) == {
|
|
|
|
|
|
"x0": 5,
|
|
|
|
|
|
"x1": 10,
|
|
|
|
|
|
"top": 0,
|
|
|
|
|
|
"doctop": 100,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
"height": 30,
|
|
|
|
|
|
"width": 5,
|
|
|
|
|
|
"y0": 40,
|
|
|
|
|
|
"y1": 70,
|
|
|
|
|
|
}
|
|
|
|
|
|
assert utils.resize_object(obj, "bottom", 40) == {
|
|
|
|
|
|
"x0": 5,
|
|
|
|
|
|
"x1": 10,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"doctop": 120,
|
|
|
|
|
|
"bottom": 40,
|
|
|
|
|
|
"height": 20,
|
|
|
|
|
|
"width": 5,
|
|
|
|
|
|
"y0": 30,
|
|
|
|
|
|
"y1": 50,
|
|
|
|
|
|
}
|
|
|
|
|
|
|
2020-08-05 22:52:36 -04:00
|
|
|
|
def test_move_object(self):
|
|
|
|
|
|
a = {
|
|
|
|
|
|
"x0": 5,
|
|
|
|
|
|
"x1": 10,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
"width": 5,
|
|
|
|
|
|
"height": 10,
|
|
|
|
|
|
"doctop": 120,
|
|
|
|
|
|
"y0": 40,
|
|
|
|
|
|
"y1": 50,
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
b = dict(a)
|
|
|
|
|
|
b["x0"] = 15
|
|
|
|
|
|
b["x1"] = 20
|
|
|
|
|
|
|
|
|
|
|
|
a_new = utils.move_object(a, "h", 10)
|
|
|
|
|
|
assert a_new == b
|
|
|
|
|
|
|
|
|
|
|
|
def test_snap_objects(self):
|
|
|
|
|
|
a = {
|
|
|
|
|
|
"x0": 5,
|
|
|
|
|
|
"x1": 10,
|
|
|
|
|
|
"top": 20,
|
|
|
|
|
|
"bottom": 30,
|
|
|
|
|
|
"width": 5,
|
|
|
|
|
|
"height": 10,
|
|
|
|
|
|
"doctop": 120,
|
|
|
|
|
|
"y0": 40,
|
|
|
|
|
|
"y1": 50,
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
b = dict(a)
|
|
|
|
|
|
b["x0"] = 6
|
|
|
|
|
|
b["x1"] = 11
|
|
|
|
|
|
|
|
|
|
|
|
c = dict(a)
|
|
|
|
|
|
c["x0"] = 7
|
|
|
|
|
|
c["x1"] = 12
|
|
|
|
|
|
|
2020-12-16 22:19:17 -05:00
|
|
|
|
a_new, b_new, c_new = utils.snap_objects([a, b, c], "x0", 1)
|
2020-08-05 22:52:36 -04:00
|
|
|
|
assert a_new == b_new == c_new
|
|
|
|
|
|
|
2020-08-01 12:27:07 -04:00
|
|
|
|
def test_filter_edges(self):
|
|
|
|
|
|
with pytest.raises(ValueError):
|
|
|
|
|
|
utils.filter_edges([], "x")
|
2022-05-06 09:43:50 -04:00
|
|
|
|
|
|
|
|
|
|
def test_to_list(self):
|
|
|
|
|
|
objs = [
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 0,
|
|
|
|
|
|
"top": 0,
|
|
|
|
|
|
"x1": 20,
|
|
|
|
|
|
"bottom": 20,
|
|
|
|
|
|
},
|
|
|
|
|
|
{
|
|
|
|
|
|
"x0": 10,
|
|
|
|
|
|
"top": 10,
|
|
|
|
|
|
"x1": 15,
|
|
|
|
|
|
"bottom": 15,
|
|
|
|
|
|
},
|
|
|
|
|
|
]
|
|
|
|
|
|
assert utils.to_list(objs) == objs
|
|
|
|
|
|
assert utils.to_list(tuple(objs)) == objs
|
|
|
|
|
|
assert utils.to_list((o for o in objs)) == objs
|
|
|
|
|
|
assert utils.to_list(pd.DataFrame(objs)) == objs
|