undoc 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- undoc-0.9.0/PKG-INFO +198 -0
- undoc-0.9.0/README.md +166 -0
- undoc-0.9.0/pyproject.toml +57 -0
- undoc-0.9.0/setup.cfg +4 -0
- undoc-0.9.0/src/undoc/__init__.py +10 -0
- undoc-0.9.0/src/undoc/_native.py +195 -0
- undoc-0.9.0/src/undoc/lib/linux-musl-x64/libundoc.so +0 -0
- undoc-0.9.0/src/undoc/lib/linux-x64/libundoc.so +0 -0
- undoc-0.9.0/src/undoc/lib/osx-arm64/libundoc.dylib +0 -0
- undoc-0.9.0/src/undoc/lib/osx-x64/libundoc.dylib +0 -0
- undoc-0.9.0/src/undoc/lib/win-x64/undoc.dll +0 -0
- undoc-0.9.0/src/undoc/undoc.py +373 -0
- undoc-0.9.0/src/undoc.egg-info/PKG-INFO +198 -0
- undoc-0.9.0/src/undoc.egg-info/SOURCES.txt +16 -0
- undoc-0.9.0/src/undoc.egg-info/dependency_links.txt +1 -0
- undoc-0.9.0/src/undoc.egg-info/requires.txt +6 -0
- undoc-0.9.0/src/undoc.egg-info/top_level.txt +1 -0
- undoc-0.9.0/tests/test_undoc.py +538 -0
undoc-0.9.0/PKG-INFO
ADDED
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: undoc
|
|
3
|
+
Version: 0.9.0
|
|
4
|
+
Summary: High-performance Microsoft Office document extraction to Markdown
|
|
5
|
+
Author-email: iyulab <tech@iyulab.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/iyulab/undoc
|
|
8
|
+
Project-URL: Documentation, https://github.com/iyulab/undoc#readme
|
|
9
|
+
Project-URL: Repository, https://github.com/iyulab/undoc
|
|
10
|
+
Project-URL: Issues, https://github.com/iyulab/undoc/issues
|
|
11
|
+
Keywords: office,docx,xlsx,pptx,markdown,extraction
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
21
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
22
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
23
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
24
|
+
Classifier: Topic :: Text Processing :: Markup
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
29
|
+
Requires-Dist: pytest-cov>=4.0; extra == "dev"
|
|
30
|
+
Requires-Dist: black>=23.0; extra == "dev"
|
|
31
|
+
Requires-Dist: mypy>=1.0; extra == "dev"
|
|
32
|
+
|
|
33
|
+
# undoc
|
|
34
|
+
|
|
35
|
+
High-performance Microsoft Office document extraction to Markdown.
|
|
36
|
+
|
|
37
|
+
## Installation
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
pip install undoc
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
### Basic Usage
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
from undoc import parse_file
|
|
49
|
+
|
|
50
|
+
# Parse a document
|
|
51
|
+
doc = parse_file("document.docx")
|
|
52
|
+
|
|
53
|
+
# Convert to Markdown
|
|
54
|
+
markdown = doc.to_markdown()
|
|
55
|
+
print(markdown)
|
|
56
|
+
|
|
57
|
+
# Convert to plain text
|
|
58
|
+
text = doc.to_text()
|
|
59
|
+
|
|
60
|
+
# Convert to JSON
|
|
61
|
+
json_data = doc.to_json()
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### With Context Manager
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from undoc import parse_file
|
|
68
|
+
|
|
69
|
+
with parse_file("document.xlsx") as doc:
|
|
70
|
+
print(doc.to_markdown(frontmatter=True))
|
|
71
|
+
print(f"Sections: {doc.section_count}")
|
|
72
|
+
print(f"Resources: {doc.resource_count}")
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
### Parse from Bytes
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from undoc import parse_bytes
|
|
79
|
+
|
|
80
|
+
with open("document.pptx", "rb") as f:
|
|
81
|
+
data = f.read()
|
|
82
|
+
|
|
83
|
+
doc = parse_bytes(data)
|
|
84
|
+
markdown = doc.to_markdown()
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### Extract Resources (Images)
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from undoc import parse_file
|
|
91
|
+
|
|
92
|
+
doc = parse_file("document.docx")
|
|
93
|
+
|
|
94
|
+
# Get all resource IDs
|
|
95
|
+
resource_ids = doc.get_resource_ids()
|
|
96
|
+
|
|
97
|
+
for rid in resource_ids:
|
|
98
|
+
# Get resource metadata
|
|
99
|
+
info = doc.get_resource_info(rid)
|
|
100
|
+
print(f"Resource: {info['filename']} ({info['mime_type']})")
|
|
101
|
+
|
|
102
|
+
# Get resource binary data
|
|
103
|
+
data = doc.get_resource_data(rid)
|
|
104
|
+
|
|
105
|
+
# Save to file
|
|
106
|
+
with open(info['filename'], 'wb') as f:
|
|
107
|
+
f.write(data)
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
### Document Metadata
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
from undoc import parse_file
|
|
114
|
+
|
|
115
|
+
doc = parse_file("document.docx")
|
|
116
|
+
|
|
117
|
+
print(f"Title: {doc.title}")
|
|
118
|
+
print(f"Author: {doc.author}")
|
|
119
|
+
print(f"Sections: {doc.section_count}")
|
|
120
|
+
print(f"Resources: {doc.resource_count}")
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
### Handling Failures
|
|
124
|
+
|
|
125
|
+
`UndocError.kind` says *why* a call failed, so you can react to the reason instead of
|
|
126
|
+
matching on message text:
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
from undoc import ErrorKind, UndocError, parse_file
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
doc = parse_file(path)
|
|
133
|
+
print(doc.to_markdown())
|
|
134
|
+
except UndocError as err:
|
|
135
|
+
if err.kind is ErrorKind.ZIP_ARCHIVE:
|
|
136
|
+
print("The file is damaged.")
|
|
137
|
+
elif err.kind in (ErrorKind.UNKNOWN_FORMAT, ErrorKind.UNSUPPORTED_FORMAT):
|
|
138
|
+
print("Not a supported Office document.")
|
|
139
|
+
elif err.kind is ErrorKind.ENCRYPTED:
|
|
140
|
+
print("The document is encrypted.")
|
|
141
|
+
else:
|
|
142
|
+
# Also the right branch for a reason this build has no name for.
|
|
143
|
+
print(f"Extraction failed ({err.kind}): {err}")
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
The numbers behind `ErrorKind` are a stable ABI contract: a new reason takes the next free
|
|
147
|
+
number and existing ones are never renumbered. Always keep a final `else` — an
|
|
148
|
+
unrecognised value arrives as a plain `int` rather than an `ErrorKind`, so that a newer
|
|
149
|
+
native library stays usable. `kind` is `ErrorKind.OTHER` for failures raised by the wrapper
|
|
150
|
+
itself, and never `ErrorKind.NONE` (which means success).
|
|
151
|
+
|
|
152
|
+
## Supported Formats
|
|
153
|
+
|
|
154
|
+
- **DOCX** - Microsoft Word documents
|
|
155
|
+
- **XLSX** - Microsoft Excel spreadsheets
|
|
156
|
+
- **PPTX** - Microsoft PowerPoint presentations
|
|
157
|
+
|
|
158
|
+
## Features
|
|
159
|
+
|
|
160
|
+
- **RAG-Ready Output**: Structured Markdown optimized for RAG/LLM applications
|
|
161
|
+
- **High Performance**: Native Rust implementation via FFI
|
|
162
|
+
- **Asset Extraction**: Images and embedded resources
|
|
163
|
+
- **Metadata Preservation**: Document properties, styles, formatting
|
|
164
|
+
- **Cross-Platform**: Windows, Linux, macOS (Intel & ARM)
|
|
165
|
+
|
|
166
|
+
## API Reference
|
|
167
|
+
|
|
168
|
+
### Functions
|
|
169
|
+
|
|
170
|
+
- `parse_file(path)` - Parse document from file path
|
|
171
|
+
- `parse_bytes(data)` - Parse document from bytes
|
|
172
|
+
- `version()` - Get library version
|
|
173
|
+
|
|
174
|
+
### Undoc Class
|
|
175
|
+
|
|
176
|
+
#### Conversion Methods
|
|
177
|
+
|
|
178
|
+
- `to_markdown(frontmatter=False, escape_special=False, paragraph_spacing=False)` - Convert to Markdown
|
|
179
|
+
- `to_text()` - Convert to plain text
|
|
180
|
+
- `to_json(compact=False)` - Convert to JSON
|
|
181
|
+
- `plain_text()` - Get plain text (fast extraction)
|
|
182
|
+
|
|
183
|
+
#### Properties
|
|
184
|
+
|
|
185
|
+
- `title` - Document title
|
|
186
|
+
- `author` - Document author
|
|
187
|
+
- `section_count` - Number of sections
|
|
188
|
+
- `resource_count` - Number of resources
|
|
189
|
+
|
|
190
|
+
#### Resource Methods
|
|
191
|
+
|
|
192
|
+
- `get_resource_ids()` - List of resource IDs
|
|
193
|
+
- `get_resource_info(id)` - Resource metadata
|
|
194
|
+
- `get_resource_data(id)` - Resource binary data
|
|
195
|
+
|
|
196
|
+
## License
|
|
197
|
+
|
|
198
|
+
MIT License - see [LICENSE](../../LICENSE) for details.
|
undoc-0.9.0/README.md
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# undoc
|
|
2
|
+
|
|
3
|
+
High-performance Microsoft Office document extraction to Markdown.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install undoc
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Usage
|
|
12
|
+
|
|
13
|
+
### Basic Usage
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from undoc import parse_file
|
|
17
|
+
|
|
18
|
+
# Parse a document
|
|
19
|
+
doc = parse_file("document.docx")
|
|
20
|
+
|
|
21
|
+
# Convert to Markdown
|
|
22
|
+
markdown = doc.to_markdown()
|
|
23
|
+
print(markdown)
|
|
24
|
+
|
|
25
|
+
# Convert to plain text
|
|
26
|
+
text = doc.to_text()
|
|
27
|
+
|
|
28
|
+
# Convert to JSON
|
|
29
|
+
json_data = doc.to_json()
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
### With Context Manager
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
from undoc import parse_file
|
|
36
|
+
|
|
37
|
+
with parse_file("document.xlsx") as doc:
|
|
38
|
+
print(doc.to_markdown(frontmatter=True))
|
|
39
|
+
print(f"Sections: {doc.section_count}")
|
|
40
|
+
print(f"Resources: {doc.resource_count}")
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
### Parse from Bytes
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from undoc import parse_bytes
|
|
47
|
+
|
|
48
|
+
with open("document.pptx", "rb") as f:
|
|
49
|
+
data = f.read()
|
|
50
|
+
|
|
51
|
+
doc = parse_bytes(data)
|
|
52
|
+
markdown = doc.to_markdown()
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
### Extract Resources (Images)
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
from undoc import parse_file
|
|
59
|
+
|
|
60
|
+
doc = parse_file("document.docx")
|
|
61
|
+
|
|
62
|
+
# Get all resource IDs
|
|
63
|
+
resource_ids = doc.get_resource_ids()
|
|
64
|
+
|
|
65
|
+
for rid in resource_ids:
|
|
66
|
+
# Get resource metadata
|
|
67
|
+
info = doc.get_resource_info(rid)
|
|
68
|
+
print(f"Resource: {info['filename']} ({info['mime_type']})")
|
|
69
|
+
|
|
70
|
+
# Get resource binary data
|
|
71
|
+
data = doc.get_resource_data(rid)
|
|
72
|
+
|
|
73
|
+
# Save to file
|
|
74
|
+
with open(info['filename'], 'wb') as f:
|
|
75
|
+
f.write(data)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### Document Metadata
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from undoc import parse_file
|
|
82
|
+
|
|
83
|
+
doc = parse_file("document.docx")
|
|
84
|
+
|
|
85
|
+
print(f"Title: {doc.title}")
|
|
86
|
+
print(f"Author: {doc.author}")
|
|
87
|
+
print(f"Sections: {doc.section_count}")
|
|
88
|
+
print(f"Resources: {doc.resource_count}")
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
### Handling Failures
|
|
92
|
+
|
|
93
|
+
`UndocError.kind` says *why* a call failed, so you can react to the reason instead of
|
|
94
|
+
matching on message text:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from undoc import ErrorKind, UndocError, parse_file
|
|
98
|
+
|
|
99
|
+
try:
|
|
100
|
+
doc = parse_file(path)
|
|
101
|
+
print(doc.to_markdown())
|
|
102
|
+
except UndocError as err:
|
|
103
|
+
if err.kind is ErrorKind.ZIP_ARCHIVE:
|
|
104
|
+
print("The file is damaged.")
|
|
105
|
+
elif err.kind in (ErrorKind.UNKNOWN_FORMAT, ErrorKind.UNSUPPORTED_FORMAT):
|
|
106
|
+
print("Not a supported Office document.")
|
|
107
|
+
elif err.kind is ErrorKind.ENCRYPTED:
|
|
108
|
+
print("The document is encrypted.")
|
|
109
|
+
else:
|
|
110
|
+
# Also the right branch for a reason this build has no name for.
|
|
111
|
+
print(f"Extraction failed ({err.kind}): {err}")
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
The numbers behind `ErrorKind` are a stable ABI contract: a new reason takes the next free
|
|
115
|
+
number and existing ones are never renumbered. Always keep a final `else` — an
|
|
116
|
+
unrecognised value arrives as a plain `int` rather than an `ErrorKind`, so that a newer
|
|
117
|
+
native library stays usable. `kind` is `ErrorKind.OTHER` for failures raised by the wrapper
|
|
118
|
+
itself, and never `ErrorKind.NONE` (which means success).
|
|
119
|
+
|
|
120
|
+
## Supported Formats
|
|
121
|
+
|
|
122
|
+
- **DOCX** - Microsoft Word documents
|
|
123
|
+
- **XLSX** - Microsoft Excel spreadsheets
|
|
124
|
+
- **PPTX** - Microsoft PowerPoint presentations
|
|
125
|
+
|
|
126
|
+
## Features
|
|
127
|
+
|
|
128
|
+
- **RAG-Ready Output**: Structured Markdown optimized for RAG/LLM applications
|
|
129
|
+
- **High Performance**: Native Rust implementation via FFI
|
|
130
|
+
- **Asset Extraction**: Images and embedded resources
|
|
131
|
+
- **Metadata Preservation**: Document properties, styles, formatting
|
|
132
|
+
- **Cross-Platform**: Windows, Linux, macOS (Intel & ARM)
|
|
133
|
+
|
|
134
|
+
## API Reference
|
|
135
|
+
|
|
136
|
+
### Functions
|
|
137
|
+
|
|
138
|
+
- `parse_file(path)` - Parse document from file path
|
|
139
|
+
- `parse_bytes(data)` - Parse document from bytes
|
|
140
|
+
- `version()` - Get library version
|
|
141
|
+
|
|
142
|
+
### Undoc Class
|
|
143
|
+
|
|
144
|
+
#### Conversion Methods
|
|
145
|
+
|
|
146
|
+
- `to_markdown(frontmatter=False, escape_special=False, paragraph_spacing=False)` - Convert to Markdown
|
|
147
|
+
- `to_text()` - Convert to plain text
|
|
148
|
+
- `to_json(compact=False)` - Convert to JSON
|
|
149
|
+
- `plain_text()` - Get plain text (fast extraction)
|
|
150
|
+
|
|
151
|
+
#### Properties
|
|
152
|
+
|
|
153
|
+
- `title` - Document title
|
|
154
|
+
- `author` - Document author
|
|
155
|
+
- `section_count` - Number of sections
|
|
156
|
+
- `resource_count` - Number of resources
|
|
157
|
+
|
|
158
|
+
#### Resource Methods
|
|
159
|
+
|
|
160
|
+
- `get_resource_ids()` - List of resource IDs
|
|
161
|
+
- `get_resource_info(id)` - Resource metadata
|
|
162
|
+
- `get_resource_data(id)` - Resource binary data
|
|
163
|
+
|
|
164
|
+
## License
|
|
165
|
+
|
|
166
|
+
MIT License - see [LICENSE](../../LICENSE) for details.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "undoc"
|
|
7
|
+
version = "0.9.0"
|
|
8
|
+
description = "High-performance Microsoft Office document extraction to Markdown"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
authors = [{ name = "iyulab", email = "tech@iyulab.com" }]
|
|
12
|
+
keywords = ["office", "docx", "xlsx", "pptx", "markdown", "extraction"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.9",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Operating System :: Microsoft :: Windows",
|
|
23
|
+
"Operating System :: POSIX :: Linux",
|
|
24
|
+
"Operating System :: MacOS :: MacOS X",
|
|
25
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
26
|
+
"Topic :: Text Processing :: Markup",
|
|
27
|
+
]
|
|
28
|
+
requires-python = ">=3.9"
|
|
29
|
+
dependencies = []
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Homepage = "https://github.com/iyulab/undoc"
|
|
33
|
+
Documentation = "https://github.com/iyulab/undoc#readme"
|
|
34
|
+
Repository = "https://github.com/iyulab/undoc"
|
|
35
|
+
Issues = "https://github.com/iyulab/undoc/issues"
|
|
36
|
+
|
|
37
|
+
[project.optional-dependencies]
|
|
38
|
+
dev = ["pytest>=7.0", "pytest-cov>=4.0", "black>=23.0", "mypy>=1.0"]
|
|
39
|
+
|
|
40
|
+
[tool.setuptools.packages.find]
|
|
41
|
+
where = ["src"]
|
|
42
|
+
|
|
43
|
+
[tool.setuptools.package-data]
|
|
44
|
+
undoc = ["lib/**/*"]
|
|
45
|
+
|
|
46
|
+
[tool.black]
|
|
47
|
+
line-length = 88
|
|
48
|
+
target-version = ["py39", "py310", "py311", "py312"]
|
|
49
|
+
|
|
50
|
+
[tool.mypy]
|
|
51
|
+
python_version = "3.9"
|
|
52
|
+
warn_return_any = true
|
|
53
|
+
warn_unused_configs = true
|
|
54
|
+
|
|
55
|
+
[tool.pytest.ini_options]
|
|
56
|
+
testpaths = ["tests"]
|
|
57
|
+
python_files = ["test_*.py"]
|
undoc-0.9.0/setup.cfg
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""undoc - High-performance Microsoft Office document extraction.
|
|
2
|
+
|
|
3
|
+
This package provides Python bindings for the undoc library, which extracts
|
|
4
|
+
DOCX, XLSX, and PPTX documents into structured Markdown with assets.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .undoc import ErrorKind, Undoc, UndocError, parse_file, parse_bytes, version
|
|
8
|
+
|
|
9
|
+
__all__ = ["ErrorKind", "Undoc", "UndocError", "parse_file", "parse_bytes", "version"]
|
|
10
|
+
__version__ = "0.9.0"
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""Native library loading for undoc."""
|
|
2
|
+
|
|
3
|
+
import ctypes
|
|
4
|
+
import platform
|
|
5
|
+
import os
|
|
6
|
+
import subprocess
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
# Library filename by platform
|
|
10
|
+
_LIB_NAMES = {
|
|
11
|
+
"Windows": "undoc.dll",
|
|
12
|
+
"Linux": "libundoc.so",
|
|
13
|
+
"Darwin": "libundoc.dylib",
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _is_musl() -> bool:
|
|
18
|
+
"""Detect if the current Linux system uses musl libc."""
|
|
19
|
+
# Check if /etc/os-release indicates Alpine
|
|
20
|
+
try:
|
|
21
|
+
osrelease = Path("/etc/os-release").read_text()
|
|
22
|
+
if "alpine" in osrelease.lower():
|
|
23
|
+
return True
|
|
24
|
+
except OSError:
|
|
25
|
+
pass
|
|
26
|
+
# Check ldd version output (musl ldd identifies itself)
|
|
27
|
+
try:
|
|
28
|
+
result = subprocess.run(
|
|
29
|
+
["ldd", "--version"], capture_output=True, text=True, timeout=5
|
|
30
|
+
)
|
|
31
|
+
output = result.stdout + result.stderr
|
|
32
|
+
if "musl" in output.lower():
|
|
33
|
+
return True
|
|
34
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
35
|
+
pass
|
|
36
|
+
return False
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _get_linux_runtime_id(machine: str) -> str:
|
|
40
|
+
"""Get the runtime ID for Linux, detecting musl vs glibc."""
|
|
41
|
+
if machine == "x86_64":
|
|
42
|
+
return "linux-musl-x64" if _is_musl() else "linux-x64"
|
|
43
|
+
raise OSError(f"Unsupported Linux architecture: {machine}")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# Runtime identifier
|
|
47
|
+
_RUNTIME_IDS = {
|
|
48
|
+
("Windows", "AMD64"): "win-x64",
|
|
49
|
+
("Windows", "x86_64"): "win-x64",
|
|
50
|
+
("Darwin", "x86_64"): "osx-x64",
|
|
51
|
+
("Darwin", "arm64"): "osx-arm64",
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _get_lib_path() -> Path:
|
|
56
|
+
"""Get the path to the native library."""
|
|
57
|
+
system = platform.system()
|
|
58
|
+
machine = platform.machine()
|
|
59
|
+
|
|
60
|
+
lib_name = _LIB_NAMES.get(system)
|
|
61
|
+
if not lib_name:
|
|
62
|
+
raise OSError(f"Unsupported platform: {system}")
|
|
63
|
+
|
|
64
|
+
# Check UNDOC_LIB_PATH environment variable first
|
|
65
|
+
env_path = os.environ.get("UNDOC_LIB_PATH")
|
|
66
|
+
if env_path:
|
|
67
|
+
p = Path(env_path)
|
|
68
|
+
if p.exists():
|
|
69
|
+
return p
|
|
70
|
+
|
|
71
|
+
if system == "Linux":
|
|
72
|
+
runtime_id = _get_linux_runtime_id(machine)
|
|
73
|
+
else:
|
|
74
|
+
runtime_id = _RUNTIME_IDS.get((system, machine))
|
|
75
|
+
if not runtime_id:
|
|
76
|
+
raise OSError(f"Unsupported architecture: {system}/{machine}")
|
|
77
|
+
|
|
78
|
+
# Look for the library in the package
|
|
79
|
+
package_dir = Path(__file__).parent
|
|
80
|
+
lib_path = package_dir / "lib" / runtime_id / lib_name
|
|
81
|
+
|
|
82
|
+
if lib_path.exists():
|
|
83
|
+
return lib_path
|
|
84
|
+
|
|
85
|
+
# Fallback: look in package root
|
|
86
|
+
lib_path = package_dir / "lib" / lib_name
|
|
87
|
+
if lib_path.exists():
|
|
88
|
+
return lib_path
|
|
89
|
+
|
|
90
|
+
# Fallback: look in current directory
|
|
91
|
+
lib_path = Path(lib_name)
|
|
92
|
+
if lib_path.exists():
|
|
93
|
+
return lib_path
|
|
94
|
+
|
|
95
|
+
# Fallback: system library path
|
|
96
|
+
return Path(lib_name)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _load_library() -> ctypes.CDLL:
|
|
100
|
+
"""Load the native undoc library."""
|
|
101
|
+
lib_path = _get_lib_path()
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
if platform.system() == "Windows":
|
|
105
|
+
# On Windows, use LoadLibraryEx with LOAD_WITH_ALTERED_SEARCH_PATH
|
|
106
|
+
return ctypes.CDLL(str(lib_path), winmode=0)
|
|
107
|
+
else:
|
|
108
|
+
return ctypes.CDLL(str(lib_path))
|
|
109
|
+
except OSError as e:
|
|
110
|
+
raise OSError(
|
|
111
|
+
f"Failed to load undoc native library from {lib_path}: {e}\n"
|
|
112
|
+
f"Make sure the library is installed for your platform ({platform.system()}/{platform.machine()})."
|
|
113
|
+
) from e
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# Load the library
|
|
117
|
+
_lib = _load_library()
|
|
118
|
+
|
|
119
|
+
# Define function signatures
|
|
120
|
+
_lib.undoc_version.argtypes = []
|
|
121
|
+
_lib.undoc_version.restype = ctypes.c_void_p
|
|
122
|
+
|
|
123
|
+
_lib.undoc_last_error.argtypes = []
|
|
124
|
+
_lib.undoc_last_error.restype = ctypes.c_void_p
|
|
125
|
+
|
|
126
|
+
# Returns a raw int, not an enum: a newer native library may report a number this
|
|
127
|
+
# build does not know, and it has to survive the trip rather than fail to convert.
|
|
128
|
+
_lib.undoc_last_error_kind.argtypes = []
|
|
129
|
+
_lib.undoc_last_error_kind.restype = ctypes.c_int
|
|
130
|
+
|
|
131
|
+
_lib.undoc_parse_file.argtypes = [ctypes.c_char_p]
|
|
132
|
+
_lib.undoc_parse_file.restype = ctypes.c_void_p
|
|
133
|
+
|
|
134
|
+
_lib.undoc_parse_bytes.argtypes = [ctypes.POINTER(ctypes.c_uint8), ctypes.c_size_t]
|
|
135
|
+
_lib.undoc_parse_bytes.restype = ctypes.c_void_p
|
|
136
|
+
|
|
137
|
+
_lib.undoc_free_document.argtypes = [ctypes.c_void_p]
|
|
138
|
+
_lib.undoc_free_document.restype = None
|
|
139
|
+
|
|
140
|
+
_lib.undoc_to_markdown.argtypes = [ctypes.c_void_p, ctypes.c_uint]
|
|
141
|
+
_lib.undoc_to_markdown.restype = ctypes.c_void_p
|
|
142
|
+
|
|
143
|
+
_lib.undoc_to_text.argtypes = [ctypes.c_void_p]
|
|
144
|
+
_lib.undoc_to_text.restype = ctypes.c_void_p
|
|
145
|
+
|
|
146
|
+
_lib.undoc_to_json.argtypes = [ctypes.c_void_p, ctypes.c_int]
|
|
147
|
+
_lib.undoc_to_json.restype = ctypes.c_void_p
|
|
148
|
+
|
|
149
|
+
_lib.undoc_plain_text.argtypes = [ctypes.c_void_p]
|
|
150
|
+
_lib.undoc_plain_text.restype = ctypes.c_void_p
|
|
151
|
+
|
|
152
|
+
_lib.undoc_section_count.argtypes = [ctypes.c_void_p]
|
|
153
|
+
_lib.undoc_section_count.restype = ctypes.c_int
|
|
154
|
+
|
|
155
|
+
_lib.undoc_resource_count.argtypes = [ctypes.c_void_p]
|
|
156
|
+
_lib.undoc_resource_count.restype = ctypes.c_int
|
|
157
|
+
|
|
158
|
+
_lib.undoc_get_title.argtypes = [ctypes.c_void_p]
|
|
159
|
+
_lib.undoc_get_title.restype = ctypes.c_void_p
|
|
160
|
+
|
|
161
|
+
_lib.undoc_get_author.argtypes = [ctypes.c_void_p]
|
|
162
|
+
_lib.undoc_get_author.restype = ctypes.c_void_p
|
|
163
|
+
|
|
164
|
+
_lib.undoc_free_string.argtypes = [ctypes.c_void_p]
|
|
165
|
+
_lib.undoc_free_string.restype = None
|
|
166
|
+
|
|
167
|
+
_lib.undoc_get_resource_ids.argtypes = [ctypes.c_void_p]
|
|
168
|
+
_lib.undoc_get_resource_ids.restype = ctypes.c_void_p
|
|
169
|
+
|
|
170
|
+
_lib.undoc_get_resource_info.argtypes = [ctypes.c_void_p, ctypes.c_char_p]
|
|
171
|
+
_lib.undoc_get_resource_info.restype = ctypes.c_void_p
|
|
172
|
+
|
|
173
|
+
_lib.undoc_get_resource_data.argtypes = [
|
|
174
|
+
ctypes.c_void_p,
|
|
175
|
+
ctypes.c_char_p,
|
|
176
|
+
ctypes.POINTER(ctypes.c_size_t),
|
|
177
|
+
]
|
|
178
|
+
_lib.undoc_get_resource_data.restype = ctypes.POINTER(ctypes.c_uint8)
|
|
179
|
+
|
|
180
|
+
_lib.undoc_free_bytes.argtypes = [ctypes.POINTER(ctypes.c_uint8), ctypes.c_size_t]
|
|
181
|
+
_lib.undoc_free_bytes.restype = None
|
|
182
|
+
|
|
183
|
+
# Export constants
|
|
184
|
+
UNDOC_FLAG_FRONTMATTER = 1
|
|
185
|
+
UNDOC_FLAG_ESCAPE_SPECIAL = 2
|
|
186
|
+
UNDOC_FLAG_PARAGRAPH_SPACING = 4
|
|
187
|
+
UNDOC_FLAG_REFINE = 8
|
|
188
|
+
|
|
189
|
+
UNDOC_JSON_PRETTY = 0
|
|
190
|
+
UNDOC_JSON_COMPACT = 1
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def get_library():
|
|
194
|
+
"""Get the loaded native library."""
|
|
195
|
+
return _lib
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|