unpdf-markdown 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unpdf_markdown-0.13.0/PKG-INFO +207 -0
- unpdf_markdown-0.13.0/README.md +182 -0
- unpdf_markdown-0.13.0/pyproject.toml +40 -0
- unpdf_markdown-0.13.0/setup.cfg +4 -0
- unpdf_markdown-0.13.0/src/unpdf/__init__.py +41 -0
- unpdf_markdown-0.13.0/src/unpdf/_native.py +201 -0
- unpdf_markdown-0.13.0/src/unpdf/lib/linux-musl-x64/libunpdf.so +0 -0
- unpdf_markdown-0.13.0/src/unpdf/lib/linux-x64/libunpdf.so +0 -0
- unpdf_markdown-0.13.0/src/unpdf/lib/osx-arm64/libunpdf.dylib +0 -0
- unpdf_markdown-0.13.0/src/unpdf/lib/osx-x64/libunpdf.dylib +0 -0
- unpdf_markdown-0.13.0/src/unpdf/lib/win-x64/unpdf.dll +0 -0
- unpdf_markdown-0.13.0/src/unpdf/unpdf.py +405 -0
- unpdf_markdown-0.13.0/src/unpdf_markdown.egg-info/PKG-INFO +207 -0
- unpdf_markdown-0.13.0/src/unpdf_markdown.egg-info/SOURCES.txt +15 -0
- unpdf_markdown-0.13.0/src/unpdf_markdown.egg-info/dependency_links.txt +1 -0
- unpdf_markdown-0.13.0/src/unpdf_markdown.egg-info/top_level.txt +1 -0
- unpdf_markdown-0.13.0/tests/test_unpdf.py +390 -0
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: unpdf-markdown
|
|
3
|
+
Version: 0.13.0
|
|
4
|
+
Summary: Python bindings for unpdf - High-performance PDF content extraction
|
|
5
|
+
Author: iyulab
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/iyulab/unpdf
|
|
8
|
+
Project-URL: Documentation, https://github.com/iyulab/unpdf
|
|
9
|
+
Project-URL: Repository, https://github.com/iyulab/unpdf
|
|
10
|
+
Project-URL: Issues, https://github.com/iyulab/unpdf/issues
|
|
11
|
+
Keywords: pdf,markdown,text-extraction,document,parser
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Text Processing
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# unpdf
|
|
27
|
+
|
|
28
|
+
Python bindings for [unpdf](https://github.com/iyulab/unpdf) - High-performance PDF content extraction to Markdown, text, and JSON.
|
|
29
|
+
|
|
30
|
+
## Installation
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install unpdf-markdown
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
The distribution is `unpdf-markdown`; the import name is `unpdf`.
|
|
37
|
+
|
|
38
|
+
## Quick Start
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
import unpdf
|
|
42
|
+
|
|
43
|
+
# Convert PDF to Markdown
|
|
44
|
+
markdown = unpdf.to_markdown("document.pdf")
|
|
45
|
+
print(markdown)
|
|
46
|
+
|
|
47
|
+
# Convert PDF to plain text
|
|
48
|
+
text = unpdf.to_text("document.pdf")
|
|
49
|
+
print(text)
|
|
50
|
+
|
|
51
|
+
# Convert PDF to JSON
|
|
52
|
+
json_data = unpdf.to_json("document.pdf", pretty=True)
|
|
53
|
+
print(json_data)
|
|
54
|
+
|
|
55
|
+
# Get document information
|
|
56
|
+
info = unpdf.get_info("document.pdf")
|
|
57
|
+
print(info)
|
|
58
|
+
|
|
59
|
+
# Get page count
|
|
60
|
+
pages = unpdf.get_page_count("document.pdf")
|
|
61
|
+
print(f"Total pages: {pages}")
|
|
62
|
+
|
|
63
|
+
# Check if file is a valid PDF
|
|
64
|
+
is_valid = unpdf.is_pdf("document.pdf")
|
|
65
|
+
print(f"Is valid PDF: {is_valid}")
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## Input: path or bytes
|
|
69
|
+
|
|
70
|
+
Every function takes its PDF as `PdfSource` — a path (`str`, or any `os.PathLike`
|
|
71
|
+
such as `pathlib.Path`) or the PDF's own bytes. Types are unambiguous: `str` is
|
|
72
|
+
always a path, `bytes` is always content, and bytes go through the native
|
|
73
|
+
in-memory parser rather than a temporary file.
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
from pathlib import Path
|
|
77
|
+
import unpdf
|
|
78
|
+
|
|
79
|
+
unpdf.to_markdown("document.pdf")
|
|
80
|
+
unpdf.to_markdown(Path("document.pdf"))
|
|
81
|
+
unpdf.to_markdown(pdf_bytes)
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## API Reference
|
|
85
|
+
|
|
86
|
+
### `to_markdown(source: PdfSource, flags: int = 0) -> str`
|
|
87
|
+
Convert a PDF file to Markdown format. `flags` is a bitwise OR of:
|
|
88
|
+
|
|
89
|
+
| Constant | Effect |
|
|
90
|
+
|---|---|
|
|
91
|
+
| `UNPDF_FLAG_FRONTMATTER` | Emit YAML frontmatter with document metadata |
|
|
92
|
+
| `UNPDF_FLAG_ESCAPE_SPECIAL` | Escape Markdown special characters |
|
|
93
|
+
| `UNPDF_FLAG_PAGE_MARKERS` | Mark each page boundary with `<!-- page N -->` |
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
import unpdf
|
|
97
|
+
|
|
98
|
+
markdown = unpdf.to_markdown(
|
|
99
|
+
"document.pdf",
|
|
100
|
+
unpdf.UNPDF_FLAG_FRONTMATTER | unpdf.UNPDF_FLAG_PAGE_MARKERS,
|
|
101
|
+
)
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
### `to_text(source: PdfSource) -> str`
|
|
105
|
+
Convert a PDF file to plain text.
|
|
106
|
+
|
|
107
|
+
### `to_json(source: PdfSource, pretty: bool = False) -> str`
|
|
108
|
+
Convert a PDF file to JSON format.
|
|
109
|
+
|
|
110
|
+
### `get_info(source: PdfSource) -> dict`
|
|
111
|
+
Get document metadata. Keys: `section_count` (the page count), `resource_count`,
|
|
112
|
+
plus `title` / `author` only when the document sets them.
|
|
113
|
+
|
|
114
|
+
### `get_page_count(source: PdfSource) -> int`
|
|
115
|
+
Get the number of pages in a PDF file.
|
|
116
|
+
|
|
117
|
+
### `is_pdf(source: PdfSource) -> bool`
|
|
118
|
+
Check if a file is a valid PDF.
|
|
119
|
+
|
|
120
|
+
### `version() -> str`
|
|
121
|
+
Get the version of the native library.
|
|
122
|
+
|
|
123
|
+
### `get_extraction_quality(source: PdfSource) -> dict`
|
|
124
|
+
Document-level extraction diagnostics: `char_count`, `word_count`,
|
|
125
|
+
`replacement_char_count`, `encrypted`, `is_scan_pdf`, `suppressed_ocr_pages`,
|
|
126
|
+
`suppressed_text_runs`,
|
|
127
|
+
`pages_incomplete`, `declared_page_count`, `unresolved_page_nodes`,
|
|
128
|
+
`skipped_object_count`. See "Incomplete extraction" below.
|
|
129
|
+
|
|
130
|
+
### `get_page_stats(source: PdfSource, page_number: int) -> dict`
|
|
131
|
+
Per-page content-stream operator counts (1-indexed): `page`, `text_op_count`,
|
|
132
|
+
`image_op_count`, `ocr_text_suppressed`.
|
|
133
|
+
|
|
134
|
+
## Detecting scanned (image-only) PDFs
|
|
135
|
+
|
|
136
|
+
Empty output can mean a scanned document, a genuinely blank page, or a parse
|
|
137
|
+
failure. The introspection surface tells them apart:
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
import unpdf
|
|
141
|
+
|
|
142
|
+
if unpdf.get_extraction_quality("scan.pdf")["is_scan_pdf"]:
|
|
143
|
+
print("Scanned document - OCR required")
|
|
144
|
+
|
|
145
|
+
stats = unpdf.get_page_stats("scan.pdf", 1)
|
|
146
|
+
if stats["text_op_count"] == 0 and stats["image_op_count"] > 0:
|
|
147
|
+
print("Page 1 is image-only (scanned)")
|
|
148
|
+
elif stats["text_op_count"] == 0:
|
|
149
|
+
print("Page 1 is genuinely blank")
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Note: a *searchable* scan (page image plus an invisible OCR text layer) reports
|
|
153
|
+
`text_op_count > 0` — combine the check with `ocr_text_suppressed`, which flags
|
|
154
|
+
pages whose unreadable OCR layer was dropped.
|
|
155
|
+
|
|
156
|
+
## Incomplete extraction
|
|
157
|
+
|
|
158
|
+
A damaged PDF does not always fail. When the cross-reference table survives but the
|
|
159
|
+
objects it points at do not, the parser returns the pages it could read — a success
|
|
160
|
+
over an incomplete page set. Check before indexing or archiving, because a page that
|
|
161
|
+
silently never arrived is indistinguishable from a page that never existed:
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
quality = unpdf.get_extraction_quality("document.pdf")
|
|
165
|
+
if quality["pages_incomplete"]:
|
|
166
|
+
print(f"incomplete - document declares {quality['declared_page_count']} page(s)")
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
| Field | Meaning |
|
|
170
|
+
|-------|---------|
|
|
171
|
+
| `pages_incomplete` | Pages are known to be missing. The one field to branch on. |
|
|
172
|
+
| `declared_page_count` | Page count the document declares, or `None` if unreadable. |
|
|
173
|
+
| `unresolved_page_nodes` | Unreadable page-tree *nodes* — non-zero means incomplete, **not** a page count. |
|
|
174
|
+
| `skipped_object_count` | Objects that could not be loaded. Most cost no page. |
|
|
175
|
+
|
|
176
|
+
Also note that `get_info()["resource_count"]` counts the extracted-resource
|
|
177
|
+
inventory, which this binding's parse path leaves empty (resource extraction is off
|
|
178
|
+
by default to bound peak memory) — it is not a count of images on the page. Use
|
|
179
|
+
`get_page_stats` for scan detection.
|
|
180
|
+
|
|
181
|
+
## Handling failures
|
|
182
|
+
|
|
183
|
+
`UnpdfError` carries a `kind` so you can branch on the reason a call failed instead
|
|
184
|
+
of matching on message text:
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
from unpdf import to_text, ErrorKind, UnpdfError
|
|
188
|
+
|
|
189
|
+
try:
|
|
190
|
+
text = to_text("document.pdf")
|
|
191
|
+
except UnpdfError as e:
|
|
192
|
+
if e.kind == ErrorKind.ENCRYPTED:
|
|
193
|
+
print("Password required")
|
|
194
|
+
elif e.kind in (ErrorKind.CORRUPTED, ErrorKind.PDF_PARSE):
|
|
195
|
+
print("The file is damaged")
|
|
196
|
+
else:
|
|
197
|
+
print(f"Extraction failed ({e.kind.name}): {e}")
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
`UnpdfError` subclasses `RuntimeError`, so existing `except RuntimeError` handlers
|
|
201
|
+
keep working. `ErrorKind` values are part of the native ABI: new reasons take new
|
|
202
|
+
numbers and existing ones are never renumbered, so treat an unrecognised value as a
|
|
203
|
+
generic failure.
|
|
204
|
+
|
|
205
|
+
## License
|
|
206
|
+
|
|
207
|
+
MIT License
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
# unpdf
|
|
2
|
+
|
|
3
|
+
Python bindings for [unpdf](https://github.com/iyulab/unpdf) - High-performance PDF content extraction to Markdown, text, and JSON.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install unpdf-markdown
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
The distribution is `unpdf-markdown`; the import name is `unpdf`.
|
|
12
|
+
|
|
13
|
+
## Quick Start
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import unpdf
|
|
17
|
+
|
|
18
|
+
# Convert PDF to Markdown
|
|
19
|
+
markdown = unpdf.to_markdown("document.pdf")
|
|
20
|
+
print(markdown)
|
|
21
|
+
|
|
22
|
+
# Convert PDF to plain text
|
|
23
|
+
text = unpdf.to_text("document.pdf")
|
|
24
|
+
print(text)
|
|
25
|
+
|
|
26
|
+
# Convert PDF to JSON
|
|
27
|
+
json_data = unpdf.to_json("document.pdf", pretty=True)
|
|
28
|
+
print(json_data)
|
|
29
|
+
|
|
30
|
+
# Get document information
|
|
31
|
+
info = unpdf.get_info("document.pdf")
|
|
32
|
+
print(info)
|
|
33
|
+
|
|
34
|
+
# Get page count
|
|
35
|
+
pages = unpdf.get_page_count("document.pdf")
|
|
36
|
+
print(f"Total pages: {pages}")
|
|
37
|
+
|
|
38
|
+
# Check if file is a valid PDF
|
|
39
|
+
is_valid = unpdf.is_pdf("document.pdf")
|
|
40
|
+
print(f"Is valid PDF: {is_valid}")
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Input: path or bytes
|
|
44
|
+
|
|
45
|
+
Every function takes its PDF as `PdfSource` — a path (`str`, or any `os.PathLike`
|
|
46
|
+
such as `pathlib.Path`) or the PDF's own bytes. Types are unambiguous: `str` is
|
|
47
|
+
always a path, `bytes` is always content, and bytes go through the native
|
|
48
|
+
in-memory parser rather than a temporary file.
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from pathlib import Path
|
|
52
|
+
import unpdf
|
|
53
|
+
|
|
54
|
+
unpdf.to_markdown("document.pdf")
|
|
55
|
+
unpdf.to_markdown(Path("document.pdf"))
|
|
56
|
+
unpdf.to_markdown(pdf_bytes)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## API Reference
|
|
60
|
+
|
|
61
|
+
### `to_markdown(source: PdfSource, flags: int = 0) -> str`
|
|
62
|
+
Convert a PDF file to Markdown format. `flags` is a bitwise OR of:
|
|
63
|
+
|
|
64
|
+
| Constant | Effect |
|
|
65
|
+
|---|---|
|
|
66
|
+
| `UNPDF_FLAG_FRONTMATTER` | Emit YAML frontmatter with document metadata |
|
|
67
|
+
| `UNPDF_FLAG_ESCAPE_SPECIAL` | Escape Markdown special characters |
|
|
68
|
+
| `UNPDF_FLAG_PAGE_MARKERS` | Mark each page boundary with `<!-- page N -->` |
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
import unpdf
|
|
72
|
+
|
|
73
|
+
markdown = unpdf.to_markdown(
|
|
74
|
+
"document.pdf",
|
|
75
|
+
unpdf.UNPDF_FLAG_FRONTMATTER | unpdf.UNPDF_FLAG_PAGE_MARKERS,
|
|
76
|
+
)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
### `to_text(source: PdfSource) -> str`
|
|
80
|
+
Convert a PDF file to plain text.
|
|
81
|
+
|
|
82
|
+
### `to_json(source: PdfSource, pretty: bool = False) -> str`
|
|
83
|
+
Convert a PDF file to JSON format.
|
|
84
|
+
|
|
85
|
+
### `get_info(source: PdfSource) -> dict`
|
|
86
|
+
Get document metadata. Keys: `section_count` (the page count), `resource_count`,
|
|
87
|
+
plus `title` / `author` only when the document sets them.
|
|
88
|
+
|
|
89
|
+
### `get_page_count(source: PdfSource) -> int`
|
|
90
|
+
Get the number of pages in a PDF file.
|
|
91
|
+
|
|
92
|
+
### `is_pdf(source: PdfSource) -> bool`
|
|
93
|
+
Check if a file is a valid PDF.
|
|
94
|
+
|
|
95
|
+
### `version() -> str`
|
|
96
|
+
Get the version of the native library.
|
|
97
|
+
|
|
98
|
+
### `get_extraction_quality(source: PdfSource) -> dict`
|
|
99
|
+
Document-level extraction diagnostics: `char_count`, `word_count`,
|
|
100
|
+
`replacement_char_count`, `encrypted`, `is_scan_pdf`, `suppressed_ocr_pages`,
|
|
101
|
+
`suppressed_text_runs`,
|
|
102
|
+
`pages_incomplete`, `declared_page_count`, `unresolved_page_nodes`,
|
|
103
|
+
`skipped_object_count`. See "Incomplete extraction" below.
|
|
104
|
+
|
|
105
|
+
### `get_page_stats(source: PdfSource, page_number: int) -> dict`
|
|
106
|
+
Per-page content-stream operator counts (1-indexed): `page`, `text_op_count`,
|
|
107
|
+
`image_op_count`, `ocr_text_suppressed`.
|
|
108
|
+
|
|
109
|
+
## Detecting scanned (image-only) PDFs
|
|
110
|
+
|
|
111
|
+
Empty output can mean a scanned document, a genuinely blank page, or a parse
|
|
112
|
+
failure. The introspection surface tells them apart:
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
import unpdf
|
|
116
|
+
|
|
117
|
+
if unpdf.get_extraction_quality("scan.pdf")["is_scan_pdf"]:
|
|
118
|
+
print("Scanned document - OCR required")
|
|
119
|
+
|
|
120
|
+
stats = unpdf.get_page_stats("scan.pdf", 1)
|
|
121
|
+
if stats["text_op_count"] == 0 and stats["image_op_count"] > 0:
|
|
122
|
+
print("Page 1 is image-only (scanned)")
|
|
123
|
+
elif stats["text_op_count"] == 0:
|
|
124
|
+
print("Page 1 is genuinely blank")
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Note: a *searchable* scan (page image plus an invisible OCR text layer) reports
|
|
128
|
+
`text_op_count > 0` — combine the check with `ocr_text_suppressed`, which flags
|
|
129
|
+
pages whose unreadable OCR layer was dropped.
|
|
130
|
+
|
|
131
|
+
## Incomplete extraction
|
|
132
|
+
|
|
133
|
+
A damaged PDF does not always fail. When the cross-reference table survives but the
|
|
134
|
+
objects it points at do not, the parser returns the pages it could read — a success
|
|
135
|
+
over an incomplete page set. Check before indexing or archiving, because a page that
|
|
136
|
+
silently never arrived is indistinguishable from a page that never existed:
|
|
137
|
+
|
|
138
|
+
```python
|
|
139
|
+
quality = unpdf.get_extraction_quality("document.pdf")
|
|
140
|
+
if quality["pages_incomplete"]:
|
|
141
|
+
print(f"incomplete - document declares {quality['declared_page_count']} page(s)")
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
| Field | Meaning |
|
|
145
|
+
|-------|---------|
|
|
146
|
+
| `pages_incomplete` | Pages are known to be missing. The one field to branch on. |
|
|
147
|
+
| `declared_page_count` | Page count the document declares, or `None` if unreadable. |
|
|
148
|
+
| `unresolved_page_nodes` | Unreadable page-tree *nodes* — non-zero means incomplete, **not** a page count. |
|
|
149
|
+
| `skipped_object_count` | Objects that could not be loaded. Most cost no page. |
|
|
150
|
+
|
|
151
|
+
Also note that `get_info()["resource_count"]` counts the extracted-resource
|
|
152
|
+
inventory, which this binding's parse path leaves empty (resource extraction is off
|
|
153
|
+
by default to bound peak memory) — it is not a count of images on the page. Use
|
|
154
|
+
`get_page_stats` for scan detection.
|
|
155
|
+
|
|
156
|
+
## Handling failures
|
|
157
|
+
|
|
158
|
+
`UnpdfError` carries a `kind` so you can branch on the reason a call failed instead
|
|
159
|
+
of matching on message text:
|
|
160
|
+
|
|
161
|
+
```python
|
|
162
|
+
from unpdf import to_text, ErrorKind, UnpdfError
|
|
163
|
+
|
|
164
|
+
try:
|
|
165
|
+
text = to_text("document.pdf")
|
|
166
|
+
except UnpdfError as e:
|
|
167
|
+
if e.kind == ErrorKind.ENCRYPTED:
|
|
168
|
+
print("Password required")
|
|
169
|
+
elif e.kind in (ErrorKind.CORRUPTED, ErrorKind.PDF_PARSE):
|
|
170
|
+
print("The file is damaged")
|
|
171
|
+
else:
|
|
172
|
+
print(f"Extraction failed ({e.kind.name}): {e}")
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
`UnpdfError` subclasses `RuntimeError`, so existing `except RuntimeError` handlers
|
|
176
|
+
keep working. `ErrorKind` values are part of the native ABI: new reasons take new
|
|
177
|
+
numbers and existing ones are never renumbered, so treat an unrecognised value as a
|
|
178
|
+
generic failure.
|
|
179
|
+
|
|
180
|
+
## License
|
|
181
|
+
|
|
182
|
+
MIT License
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "unpdf-markdown"
|
|
7
|
+
version = "0.13.0"
|
|
8
|
+
description = "Python bindings for unpdf - High-performance PDF content extraction"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = {text = "MIT"}
|
|
11
|
+
authors = [
|
|
12
|
+
{name = "iyulab"}
|
|
13
|
+
]
|
|
14
|
+
requires-python = ">=3.9"
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.9",
|
|
22
|
+
"Programming Language :: Python :: 3.10",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Topic :: Text Processing",
|
|
26
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
27
|
+
]
|
|
28
|
+
keywords = ["pdf", "markdown", "text-extraction", "document", "parser"]
|
|
29
|
+
|
|
30
|
+
[project.urls]
|
|
31
|
+
Homepage = "https://github.com/iyulab/unpdf"
|
|
32
|
+
Documentation = "https://github.com/iyulab/unpdf"
|
|
33
|
+
Repository = "https://github.com/iyulab/unpdf"
|
|
34
|
+
Issues = "https://github.com/iyulab/unpdf/issues"
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.packages.find]
|
|
37
|
+
where = ["src"]
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.package-data]
|
|
40
|
+
unpdf = ["lib/**/*"]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""
|
|
2
|
+
unpdf - Python bindings for unpdf PDF extraction library.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .unpdf import (
|
|
6
|
+
UNPDF_FLAG_ESCAPE_SPECIAL,
|
|
7
|
+
UNPDF_FLAG_FRONTMATTER,
|
|
8
|
+
UNPDF_FLAG_PAGE_MARKERS,
|
|
9
|
+
UNPDF_FLAG_REFINE,
|
|
10
|
+
ErrorKind,
|
|
11
|
+
PdfSource,
|
|
12
|
+
UnpdfError,
|
|
13
|
+
to_markdown,
|
|
14
|
+
to_text,
|
|
15
|
+
to_json,
|
|
16
|
+
get_info,
|
|
17
|
+
get_extraction_quality,
|
|
18
|
+
get_page_stats,
|
|
19
|
+
get_page_count,
|
|
20
|
+
is_pdf,
|
|
21
|
+
version,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"UNPDF_FLAG_ESCAPE_SPECIAL",
|
|
26
|
+
"UNPDF_FLAG_FRONTMATTER",
|
|
27
|
+
"UNPDF_FLAG_PAGE_MARKERS",
|
|
28
|
+
"UNPDF_FLAG_REFINE",
|
|
29
|
+
"ErrorKind",
|
|
30
|
+
"PdfSource",
|
|
31
|
+
"UnpdfError",
|
|
32
|
+
"to_markdown",
|
|
33
|
+
"to_text",
|
|
34
|
+
"to_json",
|
|
35
|
+
"get_info",
|
|
36
|
+
"get_extraction_quality",
|
|
37
|
+
"get_page_stats",
|
|
38
|
+
"get_page_count",
|
|
39
|
+
"is_pdf",
|
|
40
|
+
"version",
|
|
41
|
+
]
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Native library loading for unpdf."""
|
|
2
|
+
|
|
3
|
+
import ctypes
|
|
4
|
+
import platform
|
|
5
|
+
import os
|
|
6
|
+
import subprocess
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
# Library filename by platform
|
|
10
|
+
_LIB_NAMES = {
|
|
11
|
+
"Windows": "unpdf.dll",
|
|
12
|
+
"Linux": "libunpdf.so",
|
|
13
|
+
"Darwin": "libunpdf.dylib",
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _is_musl() -> bool:
|
|
18
|
+
"""Detect if the current Linux system uses musl libc."""
|
|
19
|
+
# Check if /etc/os-release indicates Alpine
|
|
20
|
+
try:
|
|
21
|
+
osrelease = Path("/etc/os-release").read_text()
|
|
22
|
+
if "alpine" in osrelease.lower():
|
|
23
|
+
return True
|
|
24
|
+
except OSError:
|
|
25
|
+
pass
|
|
26
|
+
# Check ldd version output (musl ldd identifies itself)
|
|
27
|
+
try:
|
|
28
|
+
result = subprocess.run(
|
|
29
|
+
["ldd", "--version"], capture_output=True, text=True, timeout=5
|
|
30
|
+
)
|
|
31
|
+
output = result.stdout + result.stderr
|
|
32
|
+
if "musl" in output.lower():
|
|
33
|
+
return True
|
|
34
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
35
|
+
pass
|
|
36
|
+
return False
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _get_linux_runtime_id(machine: str) -> str:
|
|
40
|
+
"""Get the runtime ID for Linux, detecting musl vs glibc."""
|
|
41
|
+
if machine == "x86_64":
|
|
42
|
+
return "linux-musl-x64" if _is_musl() else "linux-x64"
|
|
43
|
+
raise OSError(f"Unsupported Linux architecture: {machine}")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# Runtime identifier
|
|
47
|
+
_RUNTIME_IDS = {
|
|
48
|
+
("Windows", "AMD64"): "win-x64",
|
|
49
|
+
("Windows", "x86_64"): "win-x64",
|
|
50
|
+
("Darwin", "x86_64"): "osx-x64",
|
|
51
|
+
("Darwin", "arm64"): "osx-arm64",
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _get_lib_path() -> Path:
|
|
56
|
+
"""Get the path to the native library."""
|
|
57
|
+
system = platform.system()
|
|
58
|
+
machine = platform.machine()
|
|
59
|
+
|
|
60
|
+
lib_name = _LIB_NAMES.get(system)
|
|
61
|
+
if not lib_name:
|
|
62
|
+
raise OSError(f"Unsupported platform: {system}")
|
|
63
|
+
|
|
64
|
+
# Check UNPDF_LIB_PATH environment variable first
|
|
65
|
+
env_path = os.environ.get("UNPDF_LIB_PATH")
|
|
66
|
+
if env_path:
|
|
67
|
+
p = Path(env_path)
|
|
68
|
+
if p.exists():
|
|
69
|
+
return p
|
|
70
|
+
|
|
71
|
+
if system == "Linux":
|
|
72
|
+
runtime_id = _get_linux_runtime_id(machine)
|
|
73
|
+
else:
|
|
74
|
+
runtime_id = _RUNTIME_IDS.get((system, machine))
|
|
75
|
+
if not runtime_id:
|
|
76
|
+
raise OSError(f"Unsupported architecture: {system}/{machine}")
|
|
77
|
+
|
|
78
|
+
# Look for the library in the package
|
|
79
|
+
package_dir = Path(__file__).parent
|
|
80
|
+
lib_path = package_dir / "lib" / runtime_id / lib_name
|
|
81
|
+
|
|
82
|
+
if lib_path.exists():
|
|
83
|
+
return lib_path
|
|
84
|
+
|
|
85
|
+
# Fallback: look in package root
|
|
86
|
+
lib_path = package_dir / "lib" / lib_name
|
|
87
|
+
if lib_path.exists():
|
|
88
|
+
return lib_path
|
|
89
|
+
|
|
90
|
+
# Fallback: look in current directory
|
|
91
|
+
lib_path = Path(lib_name)
|
|
92
|
+
if lib_path.exists():
|
|
93
|
+
return lib_path
|
|
94
|
+
|
|
95
|
+
# Fallback: system library path
|
|
96
|
+
return Path(lib_name)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _load_library() -> ctypes.CDLL:
|
|
100
|
+
"""Load the native unpdf library."""
|
|
101
|
+
lib_path = _get_lib_path()
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
if platform.system() == "Windows":
|
|
105
|
+
# On Windows, use LoadLibraryEx with LOAD_WITH_ALTERED_SEARCH_PATH
|
|
106
|
+
return ctypes.CDLL(str(lib_path), winmode=0)
|
|
107
|
+
else:
|
|
108
|
+
return ctypes.CDLL(str(lib_path))
|
|
109
|
+
except OSError as e:
|
|
110
|
+
raise OSError(
|
|
111
|
+
f"Failed to load unpdf native library from {lib_path}: {e}\n"
|
|
112
|
+
f"Make sure the library is installed for your platform ({platform.system()}/{platform.machine()})."
|
|
113
|
+
) from e
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# Load the library eagerly
|
|
117
|
+
_lib = _load_library()
|
|
118
|
+
|
|
119
|
+
# Define function signatures
|
|
120
|
+
_lib.unpdf_version.argtypes = []
|
|
121
|
+
_lib.unpdf_version.restype = ctypes.c_char_p
|
|
122
|
+
|
|
123
|
+
_lib.unpdf_last_error.argtypes = []
|
|
124
|
+
_lib.unpdf_last_error.restype = ctypes.c_char_p
|
|
125
|
+
|
|
126
|
+
_lib.unpdf_last_error_kind.argtypes = []
|
|
127
|
+
_lib.unpdf_last_error_kind.restype = ctypes.c_int
|
|
128
|
+
|
|
129
|
+
_lib.unpdf_parse_file.argtypes = [ctypes.c_char_p]
|
|
130
|
+
_lib.unpdf_parse_file.restype = ctypes.c_void_p
|
|
131
|
+
|
|
132
|
+
_lib.unpdf_parse_bytes.argtypes = [ctypes.POINTER(ctypes.c_uint8), ctypes.c_size_t]
|
|
133
|
+
_lib.unpdf_parse_bytes.restype = ctypes.c_void_p
|
|
134
|
+
|
|
135
|
+
_lib.unpdf_free_document.argtypes = [ctypes.c_void_p]
|
|
136
|
+
_lib.unpdf_free_document.restype = None
|
|
137
|
+
|
|
138
|
+
_lib.unpdf_to_markdown.argtypes = [ctypes.c_void_p, ctypes.c_int]
|
|
139
|
+
_lib.unpdf_to_markdown.restype = ctypes.c_char_p
|
|
140
|
+
|
|
141
|
+
_lib.unpdf_to_text.argtypes = [ctypes.c_void_p]
|
|
142
|
+
_lib.unpdf_to_text.restype = ctypes.c_char_p
|
|
143
|
+
|
|
144
|
+
_lib.unpdf_to_json.argtypes = [ctypes.c_void_p, ctypes.c_int]
|
|
145
|
+
_lib.unpdf_to_json.restype = ctypes.c_char_p
|
|
146
|
+
|
|
147
|
+
_lib.unpdf_plain_text.argtypes = [ctypes.c_void_p]
|
|
148
|
+
_lib.unpdf_plain_text.restype = ctypes.c_char_p
|
|
149
|
+
|
|
150
|
+
_lib.unpdf_section_count.argtypes = [ctypes.c_void_p]
|
|
151
|
+
_lib.unpdf_section_count.restype = ctypes.c_int
|
|
152
|
+
|
|
153
|
+
_lib.unpdf_resource_count.argtypes = [ctypes.c_void_p]
|
|
154
|
+
_lib.unpdf_resource_count.restype = ctypes.c_int
|
|
155
|
+
|
|
156
|
+
_lib.unpdf_get_title.argtypes = [ctypes.c_void_p]
|
|
157
|
+
_lib.unpdf_get_title.restype = ctypes.c_char_p
|
|
158
|
+
|
|
159
|
+
_lib.unpdf_get_author.argtypes = [ctypes.c_void_p]
|
|
160
|
+
_lib.unpdf_get_author.restype = ctypes.c_char_p
|
|
161
|
+
|
|
162
|
+
_lib.unpdf_free_string.argtypes = [ctypes.c_char_p]
|
|
163
|
+
_lib.unpdf_free_string.restype = None
|
|
164
|
+
|
|
165
|
+
_lib.unpdf_get_extraction_quality.argtypes = [ctypes.c_void_p]
|
|
166
|
+
_lib.unpdf_get_extraction_quality.restype = ctypes.c_char_p
|
|
167
|
+
|
|
168
|
+
_lib.unpdf_page_stats.argtypes = [ctypes.c_void_p, ctypes.c_int]
|
|
169
|
+
_lib.unpdf_page_stats.restype = ctypes.c_char_p
|
|
170
|
+
|
|
171
|
+
_lib.unpdf_get_resource_ids.argtypes = [ctypes.c_void_p]
|
|
172
|
+
_lib.unpdf_get_resource_ids.restype = ctypes.c_char_p
|
|
173
|
+
|
|
174
|
+
_lib.unpdf_get_resource_info.argtypes = [ctypes.c_void_p, ctypes.c_char_p]
|
|
175
|
+
_lib.unpdf_get_resource_info.restype = ctypes.c_char_p
|
|
176
|
+
|
|
177
|
+
_lib.unpdf_get_resource_data.argtypes = [
|
|
178
|
+
ctypes.c_void_p,
|
|
179
|
+
ctypes.c_char_p,
|
|
180
|
+
ctypes.POINTER(ctypes.c_size_t),
|
|
181
|
+
]
|
|
182
|
+
_lib.unpdf_get_resource_data.restype = ctypes.POINTER(ctypes.c_uint8)
|
|
183
|
+
|
|
184
|
+
_lib.unpdf_free_bytes.argtypes = [ctypes.POINTER(ctypes.c_uint8), ctypes.c_size_t]
|
|
185
|
+
_lib.unpdf_free_bytes.restype = None
|
|
186
|
+
|
|
187
|
+
# Export constants
|
|
188
|
+
UNPDF_FLAG_FRONTMATTER = 1
|
|
189
|
+
UNPDF_FLAG_ESCAPE_SPECIAL = 2
|
|
190
|
+
# Bit 4 is retired: it named a paragraph-spacing option that never reached the
|
|
191
|
+
# renderer. Retired bits are not reused.
|
|
192
|
+
UNPDF_FLAG_PAGE_MARKERS = 8
|
|
193
|
+
UNPDF_FLAG_REFINE = 16
|
|
194
|
+
|
|
195
|
+
UNPDF_JSON_PRETTY = 0
|
|
196
|
+
UNPDF_JSON_COMPACT = 1
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def get_library():
|
|
200
|
+
"""Get the loaded native library."""
|
|
201
|
+
return _lib
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|