edgeparse 0.2.0__cp312-cp312-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgeparse/__init__.py +107 -0
- edgeparse/_edgeparse.cp312-win_amd64.pyd +0 -0
- edgeparse/_types.py +18 -0
- edgeparse/cli.py +50 -0
- edgeparse/py.typed +0 -0
- edgeparse-0.2.0.dist-info/METADATA +72 -0
- edgeparse-0.2.0.dist-info/RECORD +10 -0
- edgeparse-0.2.0.dist-info/WHEEL +4 -0
- edgeparse-0.2.0.dist-info/entry_points.txt +2 -0
- edgeparse-0.2.0.dist-info/sboms/edgeparse-python.cyclonedx.json +6460 -0
edgeparse/__init__.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""edgeparse — High-performance PDF extraction (Rust engine)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import List, Optional, Tuple, Union
|
|
7
|
+
|
|
8
|
+
from edgeparse._edgeparse import (
|
|
9
|
+
convert as _convert,
|
|
10
|
+
convert_file as _convert_file,
|
|
11
|
+
version as _version,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
PathLike = Union[str, Path]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def convert(
|
|
18
|
+
input_path: PathLike,
|
|
19
|
+
*,
|
|
20
|
+
format: str = "markdown",
|
|
21
|
+
pages: Optional[str] = None,
|
|
22
|
+
password: Optional[str] = None,
|
|
23
|
+
reading_order: str = "xycut",
|
|
24
|
+
table_method: str = "default",
|
|
25
|
+
image_output: str = "off",
|
|
26
|
+
) -> str:
|
|
27
|
+
"""Convert a PDF file and return the extracted content as a string.
|
|
28
|
+
|
|
29
|
+
Parameters
|
|
30
|
+
----------
|
|
31
|
+
input_path:
|
|
32
|
+
Path to the PDF file.
|
|
33
|
+
format:
|
|
34
|
+
Output format. Valid values: ``markdown``, ``json``, ``html``, ``text``.
|
|
35
|
+
Default: ``markdown``.
|
|
36
|
+
pages:
|
|
37
|
+
Optional page range string, e.g. ``"1,3,5-7"``.
|
|
38
|
+
password:
|
|
39
|
+
Optional password for encrypted PDFs.
|
|
40
|
+
reading_order:
|
|
41
|
+
Reading order algorithm. ``"xycut"`` (default) or ``"off"``.
|
|
42
|
+
table_method:
|
|
43
|
+
Table detection method. ``"default"`` or ``"cluster"``.
|
|
44
|
+
image_output:
|
|
45
|
+
Image output mode. ``"off"`` (default), ``"embedded"``, or ``"external"``.
|
|
46
|
+
|
|
47
|
+
Returns
|
|
48
|
+
-------
|
|
49
|
+
str
|
|
50
|
+
The extracted content in the requested format.
|
|
51
|
+
"""
|
|
52
|
+
return _convert(
|
|
53
|
+
str(input_path),
|
|
54
|
+
format=format,
|
|
55
|
+
pages=pages,
|
|
56
|
+
password=password,
|
|
57
|
+
reading_order=reading_order,
|
|
58
|
+
table_method=table_method,
|
|
59
|
+
image_output=image_output,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def convert_file(
|
|
64
|
+
input_path: PathLike,
|
|
65
|
+
output_dir: PathLike = "output",
|
|
66
|
+
*,
|
|
67
|
+
format: str = "markdown",
|
|
68
|
+
pages: Optional[str] = None,
|
|
69
|
+
password: Optional[str] = None,
|
|
70
|
+
) -> str:
|
|
71
|
+
"""Convert a PDF file and write the output to a file.
|
|
72
|
+
|
|
73
|
+
Parameters
|
|
74
|
+
----------
|
|
75
|
+
input_path:
|
|
76
|
+
Path to the PDF file.
|
|
77
|
+
output_dir:
|
|
78
|
+
Directory to write output files. Created if it does not exist.
|
|
79
|
+
format:
|
|
80
|
+
Output format. Valid values: ``markdown``, ``json``, ``html``, ``text``.
|
|
81
|
+
Default: ``markdown``.
|
|
82
|
+
pages:
|
|
83
|
+
Optional page range string, e.g. ``"1,3,5-7"``.
|
|
84
|
+
password:
|
|
85
|
+
Optional password for encrypted PDFs.
|
|
86
|
+
|
|
87
|
+
Returns
|
|
88
|
+
-------
|
|
89
|
+
str
|
|
90
|
+
Path to the created output file.
|
|
91
|
+
"""
|
|
92
|
+
return _convert_file(
|
|
93
|
+
str(input_path),
|
|
94
|
+
str(output_dir),
|
|
95
|
+
format=format,
|
|
96
|
+
pages=pages,
|
|
97
|
+
password=password,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def version() -> str:
|
|
102
|
+
"""Return the edgeparse version string."""
|
|
103
|
+
return _version()
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
__all__ = ["convert", "convert_file", "version"]
|
|
107
|
+
__version__ = _version()
|
|
Binary file
|
edgeparse/_types.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Type definitions for edgeparse."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
# Valid output format strings
|
|
9
|
+
FORMATS = ("markdown", "json", "html", "text")
|
|
10
|
+
|
|
11
|
+
# Valid reading order algorithms
|
|
12
|
+
READING_ORDERS = ("xycut", "off")
|
|
13
|
+
|
|
14
|
+
# Valid table detection methods
|
|
15
|
+
TABLE_METHODS = ("default", "cluster")
|
|
16
|
+
|
|
17
|
+
# Valid image output modes
|
|
18
|
+
IMAGE_OUTPUTS = ("off", "embedded", "external")
|
edgeparse/cli.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""CLI entry point — mirrors the Rust CLI but invoked via ``edgeparse`` console script."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
from . import convert_file
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main() -> None:
|
|
12
|
+
parser = argparse.ArgumentParser(
|
|
13
|
+
prog="edgeparse",
|
|
14
|
+
description="Convert PDF files to Markdown, JSON, HTML, or plain text.",
|
|
15
|
+
)
|
|
16
|
+
parser.add_argument("input", nargs="+", help="PDF file(s) to convert")
|
|
17
|
+
parser.add_argument(
|
|
18
|
+
"-o", "--output-dir", default="output", help="Output directory (default: output)"
|
|
19
|
+
)
|
|
20
|
+
parser.add_argument(
|
|
21
|
+
"-f",
|
|
22
|
+
"--format",
|
|
23
|
+
default="markdown",
|
|
24
|
+
help="Output format: markdown, json, html, text (default: markdown)",
|
|
25
|
+
)
|
|
26
|
+
parser.add_argument("--pages", help="Page range, e.g. 1,3,5-7")
|
|
27
|
+
parser.add_argument("-p", "--password", help="Password for encrypted PDFs")
|
|
28
|
+
args = parser.parse_args()
|
|
29
|
+
|
|
30
|
+
has_errors = False
|
|
31
|
+
for input_path in args.input:
|
|
32
|
+
try:
|
|
33
|
+
out = convert_file(
|
|
34
|
+
input_path,
|
|
35
|
+
output_dir=args.output_dir,
|
|
36
|
+
format=args.format,
|
|
37
|
+
pages=args.pages,
|
|
38
|
+
password=args.password,
|
|
39
|
+
)
|
|
40
|
+
print(out)
|
|
41
|
+
except Exception as e:
|
|
42
|
+
print(f"Error processing {input_path}: {e}", file=sys.stderr)
|
|
43
|
+
has_errors = True
|
|
44
|
+
|
|
45
|
+
if has_errors:
|
|
46
|
+
sys.exit(1)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
if __name__ == "__main__":
|
|
50
|
+
main()
|
edgeparse/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: edgeparse
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Classifier: Development Status :: 4 - Beta
|
|
5
|
+
Classifier: Intended Audience :: Developers
|
|
6
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
13
|
+
Classifier: Programming Language :: Rust
|
|
14
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
15
|
+
Classifier: Topic :: Text Processing
|
|
16
|
+
Summary: High-performance PDF-to-structured-data extraction — Rust engine, Python interface
|
|
17
|
+
Keywords: pdf,extraction,markdown,rag,llm
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
20
|
+
Project-URL: Documentation, https://github.com/raphaelmansuy/edgeparse/tree/main/docs
|
|
21
|
+
Project-URL: Homepage, https://github.com/raphaelmansuy/edgeparse
|
|
22
|
+
Project-URL: Repository, https://github.com/raphaelmansuy/edgeparse
|
|
23
|
+
|
|
24
|
+
# edgeparse
|
|
25
|
+
|
|
26
|
+
High-performance PDF-to-structured-data extraction for Python — powered by a Rust engine via PyO3.
|
|
27
|
+
|
|
28
|
+
## Install
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install edgeparse
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Pre-built wheels are available for **macOS**, **Linux** (x86_64, arm64), and **Windows** (x64).
|
|
35
|
+
No system dependencies or compilation required.
|
|
36
|
+
|
|
37
|
+
## Quick start
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
import edgeparse
|
|
41
|
+
|
|
42
|
+
# Convert a PDF to Markdown
|
|
43
|
+
result = edgeparse.convert("document.pdf")
|
|
44
|
+
print(result.markdown)
|
|
45
|
+
|
|
46
|
+
# Convert with options
|
|
47
|
+
result = edgeparse.convert(
|
|
48
|
+
"document.pdf",
|
|
49
|
+
format="markdown", # "markdown" | "json" | "html"
|
|
50
|
+
extract_images=False,
|
|
51
|
+
page_range=None, # None = all pages, or [0, 5] for pages 1–6
|
|
52
|
+
)
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## CLI
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
edgeparse document.pdf # → Markdown on stdout
|
|
59
|
+
edgeparse document.pdf --format json # → JSON
|
|
60
|
+
edgeparse /path/to/dir/ --output-dir out/ # batch convert
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Performance
|
|
64
|
+
|
|
65
|
+
`edgeparse` consistently leads open benchmarks for PDF-to-Markdown extraction quality across 200-document test suites.
|
|
66
|
+
|
|
67
|
+
## Links
|
|
68
|
+
|
|
69
|
+
- [GitHub](https://github.com/raphaelmansuy/edgeparse)
|
|
70
|
+
- [crates.io (Rust)](https://crates.io/crates/edgeparse-core)
|
|
71
|
+
- [npm (edgeparse)](https://www.npmjs.com/package/edgeparse)
|
|
72
|
+
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
edgeparse/__init__.py,sha256=NlSYQTGNeIpGNMINXHWkCsynhbUNEPvZMyerLKz66vc,2766
|
|
2
|
+
edgeparse/_edgeparse.cp312-win_amd64.pyd,sha256=JAjx8mygTlIZ-lqL7t0R8lge7NLcfh1_KgJkjNiR1RA,4347392
|
|
3
|
+
edgeparse/_types.py,sha256=aYnqZW0mYfep8vIItF2cqD2_zNaGveQeXmBM8mFHyB4,416
|
|
4
|
+
edgeparse/cli.py,sha256=LrE7xftA0_4KbkXmMcAz0BxX9JztzGAp6D6yMP3SIsk,1491
|
|
5
|
+
edgeparse/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
edgeparse-0.2.0.dist-info/METADATA,sha256=30OXpQcKMnCTw4QsfMQzQa-yQJVVqrH5qsA3X6rMOI4,2344
|
|
7
|
+
edgeparse-0.2.0.dist-info/WHEEL,sha256=IerCNAQpy9eepfUjoenIR-EbFm_XEgG9BeNfw_zQQO0,97
|
|
8
|
+
edgeparse-0.2.0.dist-info/entry_points.txt,sha256=XyI2HhMiCCI5ArQFB0nbp0Faewr1QKsDO7suWnXuU8A,47
|
|
9
|
+
edgeparse-0.2.0.dist-info/sboms/edgeparse-python.cyclonedx.json,sha256=suJIStkgBxK1hwkH1DKvffn2QEmakIh-gChhe63nK5A,203706
|
|
10
|
+
edgeparse-0.2.0.dist-info/RECORD,,
|