mirror of
https://github.com/jsvine/pdfplumber.git
synced 2026-08-29 08:34:23 +08:00
Initial commit
This commit is contained in:
+60
@@ -0,0 +1,60 @@
|
|||||||
|
notebooks/
|
||||||
|
.DS_Store
|
||||||
|
# Byte-compiled / optimized / DLL files
|
||||||
|
__pycache__/
|
||||||
|
*.py[cod]
|
||||||
|
*$py.class
|
||||||
|
|
||||||
|
# C extensions
|
||||||
|
*.so
|
||||||
|
|
||||||
|
# Distribution / packaging
|
||||||
|
.Python
|
||||||
|
env/
|
||||||
|
build/
|
||||||
|
develop-eggs/
|
||||||
|
dist/
|
||||||
|
downloads/
|
||||||
|
eggs/
|
||||||
|
.eggs/
|
||||||
|
lib/
|
||||||
|
lib64/
|
||||||
|
parts/
|
||||||
|
sdist/
|
||||||
|
var/
|
||||||
|
*.egg-info/
|
||||||
|
.installed.cfg
|
||||||
|
*.egg
|
||||||
|
|
||||||
|
# PyInstaller
|
||||||
|
# Usually these files are written by a python script from a template
|
||||||
|
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||||
|
*.manifest
|
||||||
|
*.spec
|
||||||
|
|
||||||
|
# Installer logs
|
||||||
|
pip-log.txt
|
||||||
|
pip-delete-this-directory.txt
|
||||||
|
|
||||||
|
# Unit test / coverage reports
|
||||||
|
htmlcov/
|
||||||
|
.tox/
|
||||||
|
.coverage
|
||||||
|
.coverage.*
|
||||||
|
.cache
|
||||||
|
nosetests.xml
|
||||||
|
coverage.xml
|
||||||
|
*,cover
|
||||||
|
|
||||||
|
# Translations
|
||||||
|
*.mo
|
||||||
|
*.pot
|
||||||
|
|
||||||
|
# Django stuff:
|
||||||
|
*.log
|
||||||
|
|
||||||
|
# Sphinx documentation
|
||||||
|
docs/_build/
|
||||||
|
|
||||||
|
# PyBuilder
|
||||||
|
target/
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
The MIT License (MIT)
|
||||||
|
|
||||||
|
Copyright (c) 2015, Jeremy Singer-Vine
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
of this software and associated documentation files (the "Software"), to deal
|
||||||
|
in the Software without restriction, including without limitation the rights
|
||||||
|
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
copies of the Software, and to permit persons to whom the Software is
|
||||||
|
furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be included in all
|
||||||
|
copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
__WARNING: This software is in its very early days, might not work well, and might change dramatically in future versions.__
|
||||||
|
|
||||||
|
# PDFPlumber
|
||||||
|
|
||||||
|
Plumb a PDF for detailed information about each char, rectangle, line, et cetera.
|
||||||
|
|
||||||
|
Built on [`pdfminer`](https://github.com/euske/pdfminer)/[`pdfminer.six`](https://github.com/goulu/pdfminer).
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
```sh
|
||||||
|
pip install git+https://github.com/goulu/pdfminer#egg=pdfminer.six
|
||||||
|
pip install pdfplumber
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
```python
|
||||||
|
import pdfplumber
|
||||||
|
|
||||||
|
pdf = pdfplumber.from_path("path/to/file.pdf")
|
||||||
|
|
||||||
|
# OR
|
||||||
|
|
||||||
|
with open("path/to/file.pdf") as f:
|
||||||
|
pdf = pdfplumber.load(f)
|
||||||
|
|
||||||
|
print(pdf.chars)
|
||||||
|
print(pdf.rects)
|
||||||
|
print(pdf.lines)
|
||||||
|
```
|
||||||
|
|
||||||
|
### Pandas Integration
|
||||||
|
|
||||||
|
By default, `pdf.chars`, etc., will be a plain Python dictionary. But if you `pandas=True` to `pdfplumber.load`/`.from_file`, you'll receive those properties as [Pandas dataframes](http://pandas.pydata.org/pandas-docs/stable/dsintro.html#dataframe).
|
||||||
|
|
||||||
|
## Python Support
|
||||||
|
|
||||||
|
Support for Python 3 is rough around the edges and largely dependent on the progress of [`pdfminer.six`](https://github.com/goulu/pdfminer).
|
||||||
|
|
||||||
|
## Feedback
|
||||||
|
|
||||||
|
Issues and pull requests welcome.
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
from pdfplumber.pdf import PDF
|
||||||
|
import pdfminer
|
||||||
|
import pdfminer.pdftypes
|
||||||
|
pdfminer.pdftypes.STRICT = False
|
||||||
|
|
||||||
|
def load(file_or_buffer, **kwargs):
|
||||||
|
return PDF(file_or_buffer, **kwargs)
|
||||||
|
|
||||||
|
def from_path(path, **kwargs):
|
||||||
|
with open(path, "rb") as f:
|
||||||
|
return PDF(f, **kwargs)
|
||||||
|
|
||||||
|
def set_debug(debug=0):
|
||||||
|
pdfminer.debug = debug
|
||||||
|
|
||||||
|
set_debug(0)
|
||||||
@@ -0,0 +1,125 @@
|
|||||||
|
from six import string_types
|
||||||
|
from six.moves import cStringIO
|
||||||
|
|
||||||
|
try:
|
||||||
|
import pandas
|
||||||
|
except: pass
|
||||||
|
|
||||||
|
from pdfminer.pdfparser import PDFParser
|
||||||
|
from pdfminer.pdfdocument import PDFDocument
|
||||||
|
from pdfminer.pdfpage import PDFPage
|
||||||
|
from pdfminer.pdfinterp import PDFResourceManager, PDFPageInterpreter
|
||||||
|
from pdfminer.layout import LAParams, LTChar, LTImage, LTPage
|
||||||
|
from pdfminer.converter import PDFPageAggregator
|
||||||
|
|
||||||
|
class PDF(object):
|
||||||
|
def __init__(self, file_or_buffer, pandas=False, laparams={}):
|
||||||
|
self.pandas = pandas
|
||||||
|
self.laparams = LAParams(**laparams)
|
||||||
|
|
||||||
|
rsrcmgr = PDFResourceManager()
|
||||||
|
self.doc = PDFDocument(PDFParser(file_or_buffer))
|
||||||
|
self.device = PDFPageAggregator(rsrcmgr, laparams=self.laparams)
|
||||||
|
self.interpreter = PDFPageInterpreter(rsrcmgr, self.device)
|
||||||
|
|
||||||
|
self.pages = []
|
||||||
|
for page in PDFPage.create_pages(self.doc):
|
||||||
|
self.interpreter.process_page(page)
|
||||||
|
layout = self.device.get_result()
|
||||||
|
self.pages.append(layout)
|
||||||
|
|
||||||
|
self.objects = self.parse()
|
||||||
|
|
||||||
|
def parse(self):
|
||||||
|
try:
|
||||||
|
# pdfminer < 20131022
|
||||||
|
_pages = self.doc.get_pages()
|
||||||
|
except AttributeError:
|
||||||
|
# pdfminer >= 20131022
|
||||||
|
_pages = PDFPage.create_pages(self.doc)
|
||||||
|
|
||||||
|
objects = {}
|
||||||
|
|
||||||
|
def process_object(obj, page):
|
||||||
|
_round = lambda x: round(x, 3) if type(x) == float else x
|
||||||
|
|
||||||
|
attr = dict((k, _round(v)) for k, v in obj.__dict__.items()
|
||||||
|
if isinstance(v, (float, int, string_types))
|
||||||
|
and k[0] != "_")
|
||||||
|
|
||||||
|
kind = obj.__class__.__name__
|
||||||
|
attr["kind"] = kind
|
||||||
|
attr["pageid"] = page.pageid
|
||||||
|
|
||||||
|
if hasattr(obj, "get_text"):
|
||||||
|
attr["text"] = obj.get_text()
|
||||||
|
|
||||||
|
if attr.get("y0") != None:
|
||||||
|
page_index = self.pages.index(page)
|
||||||
|
prev_h = sum(p.height for p in self.pages[:page_index])
|
||||||
|
attr["top"] = _round(page.height - attr["y1"])
|
||||||
|
attr["doctop"] = _round(prev_h + attr["top"])
|
||||||
|
|
||||||
|
if objects.get(kind) == None:
|
||||||
|
objects[kind] = []
|
||||||
|
objects[kind].append(attr)
|
||||||
|
|
||||||
|
if hasattr(obj, "_objs"):
|
||||||
|
for child in obj._objs:
|
||||||
|
process_object(child, page)
|
||||||
|
|
||||||
|
def process_page(page):
|
||||||
|
for child in page._objs:
|
||||||
|
process_object(child, page)
|
||||||
|
|
||||||
|
for page in self.pages:
|
||||||
|
process_page(page)
|
||||||
|
|
||||||
|
return objects
|
||||||
|
|
||||||
|
def emit(self, objs):
|
||||||
|
return pandas.DataFrame(objs) if self.pandas else objs
|
||||||
|
|
||||||
|
@property
|
||||||
|
def rects(self):
|
||||||
|
x = self.objects.get("LTRect", [])
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def lines(self):
|
||||||
|
x = self.objects.get("LTLine", [])
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def images(self):
|
||||||
|
x = self.objects.get("LTImage", [])
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def figures(self):
|
||||||
|
x = self.objects.get("LTFigure", [])
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def chars(self):
|
||||||
|
x = self.objects.get("LTChar", [])
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def annos(self):
|
||||||
|
x = self.objects.get("LTAnno", [])
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def text_lines(self):
|
||||||
|
h = self.objects.get("LTTextLineHorizontal", [])
|
||||||
|
v = self.objects.get("LTTextLineVertical", [])
|
||||||
|
x = h + v
|
||||||
|
return self.emit(x)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def text_boxes(self):
|
||||||
|
h = self.objects.get("LTTextBoxHorizontal", [])
|
||||||
|
v = self.objects.get("LTTextBoxVertical", [])
|
||||||
|
x = h + v
|
||||||
|
return self.emit(x)
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
import sys, os
|
||||||
|
from setuptools import setup, find_packages
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
base_reqs = [
|
||||||
|
"chardet",
|
||||||
|
"pdfminer.six"
|
||||||
|
]
|
||||||
|
|
||||||
|
setup(
|
||||||
|
name="pdfplumber",
|
||||||
|
version="0.0.0",
|
||||||
|
packages=find_packages(exclude=["test",]),
|
||||||
|
tests_require=[ "nose", "pandas" ] + base_reqs,
|
||||||
|
install_requires=base_reqs,
|
||||||
|
)
|
||||||
Reference in New Issue
Block a user