pdfcraft-dev 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pdfcraft_dev-1.4.0/.gitignore +7 -0
- pdfcraft_dev-1.4.0/LICENSE +21 -0
- pdfcraft_dev-1.4.0/PKG-INFO +241 -0
- pdfcraft_dev-1.4.0/PUBLISHING.md +56 -0
- pdfcraft_dev-1.4.0/README.md +198 -0
- pdfcraft_dev-1.4.0/pyproject.toml +77 -0
- pdfcraft_dev-1.4.0/scripts/publish.sh +307 -0
- pdfcraft_dev-1.4.0/src/pdfcraft/__init__.py +33 -0
- pdfcraft_dev-1.4.0/src/pdfcraft/_client.py +266 -0
- pdfcraft_dev-1.4.0/src/pdfcraft/_contract.py +72 -0
- pdfcraft_dev-1.4.0/src/pdfcraft/_errors.py +66 -0
- pdfcraft_dev-1.4.0/src/pdfcraft/_version.py +6 -0
- pdfcraft_dev-1.4.0/src/pdfcraft/py.typed +0 -0
- pdfcraft_dev-1.4.0/tests/test_client.py +229 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Gaurav Singh
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pdfcraft-dev
|
|
3
|
+
Version: 1.4.0
|
|
4
|
+
Summary: HTML to PDF, and PDF to structured JSON, in one call. The official PDFCraft SDK.
|
|
5
|
+
Project-URL: Homepage, https://pdfcraft.dev
|
|
6
|
+
Project-URL: Documentation, https://pdfcraft.dev/sdk/python/
|
|
7
|
+
Project-URL: Source, https://github.com/igaurav-dev/pdfcraft-python
|
|
8
|
+
Project-URL: Issues, https://github.com/igaurav-dev/pdfcraft-python/issues
|
|
9
|
+
Author: PDFCraft
|
|
10
|
+
License: MIT License
|
|
11
|
+
|
|
12
|
+
Copyright (c) 2026 Gaurav Singh
|
|
13
|
+
|
|
14
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
15
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
16
|
+
in the Software without restriction, including without limitation the rights
|
|
17
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
18
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
19
|
+
furnished to do so, subject to the following conditions:
|
|
20
|
+
|
|
21
|
+
The above copyright notice and this permission notice shall be included in all
|
|
22
|
+
copies or substantial portions of the Software.
|
|
23
|
+
|
|
24
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
25
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
26
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
27
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
28
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
29
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
30
|
+
SOFTWARE.
|
|
31
|
+
License-File: LICENSE
|
|
32
|
+
Keywords: chromium,extract-tables-from-pdf,headless-chrome,html-to-pdf,html-to-pdf-api,invoice,pdf,pdf-accessibility,pdf-api,pdf-extraction,pdf-generation,pdf-table-extraction,pdf-to-json,url-to-pdf
|
|
33
|
+
Classifier: Development Status :: 4 - Beta
|
|
34
|
+
Classifier: Intended Audience :: Developers
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Programming Language :: Python :: 3
|
|
37
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
38
|
+
Classifier: Topic :: Printing
|
|
39
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
40
|
+
Classifier: Typing :: Typed
|
|
41
|
+
Requires-Python: >=3.9
|
|
42
|
+
Description-Content-Type: text/markdown
|
|
43
|
+
|
|
44
|
+
# PDFCraft for Python
|
|
45
|
+
|
|
46
|
+
HTML to PDF, PDF to structured JSON, and accessibility triage for a whole document estate.
|
|
47
|
+
The official Python client for [PDFCraft](https://pdfcraft.dev).
|
|
48
|
+
|
|
49
|
+
**Zero dependencies.** The whole client is `urllib.request` plus a retry loop, so it installs
|
|
50
|
+
into a Lambda or a slim container without dragging a transitive tree behind it.
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
pip install pdfcraft-dev
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Installs as `pdfcraft-dev`, imports as `pdfcraft` — the same split as
|
|
57
|
+
`python-dateutil`/`dateutil`. The plain name was taken on PyPI by an unrelated
|
|
58
|
+
project, and the npm package is `@pdfcraft-dev/pdf`, so the two registries at
|
|
59
|
+
least agree with each other.
|
|
60
|
+
|
|
61
|
+
## Render
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from pdfcraft import PDFCraft
|
|
65
|
+
|
|
66
|
+
client = PDFCraft("sk_live_...")
|
|
67
|
+
|
|
68
|
+
pdf = client.render(html="<h1>Invoice 1042</h1>")
|
|
69
|
+
open("invoice.pdf", "wb").write(pdf)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
A URL instead of HTML, with page options:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
pdf = client.render(
|
|
76
|
+
url="https://example.com/report",
|
|
77
|
+
options={
|
|
78
|
+
"format": "A4",
|
|
79
|
+
"margin": {"top": "20mm", "bottom": "20mm"},
|
|
80
|
+
"printBackground": True,
|
|
81
|
+
"waitFor": {"selector": "#chart-ready", "networkIdle": True},
|
|
82
|
+
},
|
|
83
|
+
filename="report.pdf",
|
|
84
|
+
)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Need a link rather than bytes — for an email, or a file too big to hold in memory:
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
result = client.render_to_url(html=invoice_html)
|
|
91
|
+
print(result["url"], result["expires_at"], result["pages"])
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Extract
|
|
95
|
+
|
|
96
|
+
A PDF in, its tables and labelled fields out, each with a bounding box:
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
data = client.extract_pdf(open("statement.pdf", "rb").read())
|
|
100
|
+
|
|
101
|
+
for table in data["tables"]:
|
|
102
|
+
print(table["header"], len(table["rows"]), table["confidence"])
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Ask for specific fields by name and type:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
data = client.extract(
|
|
109
|
+
file=base64_pdf,
|
|
110
|
+
schema={"invoice_total": "currency", "due_date": "date", "po_number": "string"},
|
|
111
|
+
)
|
|
112
|
+
print(data["fields"]["invoice_total"]["value"])
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
There is **no OCR**. A scanned document has no text layer, comes back as `extraction_failed`,
|
|
116
|
+
and is not billed. Nothing is guessed by a model, so the same document always produces the
|
|
117
|
+
same answer — which is the point if you are reconciling numbers.
|
|
118
|
+
|
|
119
|
+
Extraction is billed per page read, so `options={"pages": "1-3"}` narrows the bill as well as
|
|
120
|
+
the work.
|
|
121
|
+
|
|
122
|
+
## Async
|
|
123
|
+
|
|
124
|
+
For documents slow enough that you would rather not hold the connection:
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
job = client.render_async(url="https://example.com/huge", webhookUrl="https://you/hook")
|
|
128
|
+
status = client.get_render(job["id"])
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
`get_extraction` polls extractions. An extraction id is not a render id — each endpoint 404s
|
|
132
|
+
on the other's ids, deliberately.
|
|
133
|
+
|
|
134
|
+
## Accessibility
|
|
135
|
+
|
|
136
|
+
Point it at a domain and it finds every PDF, checks each against PDF/UA and WCAG 2.1 AA, and
|
|
137
|
+
returns a report ranked by severity weighted by reach, with a remediation cost range.
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
import time
|
|
141
|
+
|
|
142
|
+
scan = client.scan(domain="example.gov", max_documents=500)
|
|
143
|
+
while scan["status"] not in ("succeeded", "failed"):
|
|
144
|
+
time.sleep(10)
|
|
145
|
+
scan = client.get_scan(scan["id"])
|
|
146
|
+
|
|
147
|
+
for doc in scan["documents"][:10]: # already ranked — this is the fix list
|
|
148
|
+
print(doc["severity"], doc["score"], doc["url"])
|
|
149
|
+
print(f" ${doc['cost_low_usd']:.0f}-${doc['cost_high_usd']:.0f} to remediate")
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Exactly one source: `domain=`, `sitemap=` or `urls=`. Passing none or two raises
|
|
153
|
+
`PDFCraftError` before anything is sent, because a round trip to be told you contradicted
|
|
154
|
+
yourself is a round trip wasted.
|
|
155
|
+
|
|
156
|
+
A scan runs for minutes — one request per second per host is a rule we do not break — so it
|
|
157
|
+
returns an id immediately and you poll. `max_documents` is **clamped to your plan rather than
|
|
158
|
+
refused**; `discovered` minus `checked` is what was found and not looked at, which is also the
|
|
159
|
+
upgrade prompt.
|
|
160
|
+
|
|
161
|
+
`report_url` on a finished scan is a share token. Anyone holding it can read the full HTML
|
|
162
|
+
report, and `GET /r/<token>/pdf` renders the same report to PDF through the render API. Treat
|
|
163
|
+
it as a credential, not an identifier.
|
|
164
|
+
|
|
165
|
+
Each finding carries `severity` (`blocker`, `major`, `minor`), the `wcag` criteria it breaks, a
|
|
166
|
+
`message` written for whoever approves the budget, and `technical_detail` for whoever does the
|
|
167
|
+
work. `occurrences` is volume, not severity — one check failing 1,535 times is one thing wrong,
|
|
168
|
+
fixed once, so never rank on it.
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from pdfcraft import A11Y_SEVERITIES, FINDING_LAYERS, SCAN_STATUSES
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
Those are generated from the same contract the API validates against, so comparing against them
|
|
175
|
+
beats comparing against a string you typed.
|
|
176
|
+
|
|
177
|
+
Accessibility is a **separate subscription** from rendering. An account can hold either, both or
|
|
178
|
+
neither, and the free tier is a real scan of 25 documents with full findings.
|
|
179
|
+
|
|
180
|
+
## Errors
|
|
181
|
+
|
|
182
|
+
Every failure raises `PDFCraftError`. Branch on `.code`, which is stable; the message is
|
|
183
|
+
written for a human and may be reworded.
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from pdfcraft import PDFCraft, PDFCraftError
|
|
187
|
+
|
|
188
|
+
try:
|
|
189
|
+
pdf = client.render(html=page)
|
|
190
|
+
except PDFCraftError as error:
|
|
191
|
+
if error.code == "quota_exceeded":
|
|
192
|
+
... # out of renders this period
|
|
193
|
+
elif error.code == "render_failed":
|
|
194
|
+
... # their HTML broke — billable, and worth logging
|
|
195
|
+
elif error.retryable:
|
|
196
|
+
... # already retried; this is after the last attempt
|
|
197
|
+
else:
|
|
198
|
+
raise
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
`error.docs_url` points at the page explaining that specific code.
|
|
202
|
+
|
|
203
|
+
### What is and is not billed
|
|
204
|
+
|
|
205
|
+
A render is billable if Chromium actually ran. Successes and `render_failed` count;
|
|
206
|
+
`render_timeout`, `internal_error` and every 4xx that never reached the browser do not.
|
|
207
|
+
|
|
208
|
+
## Retries
|
|
209
|
+
|
|
210
|
+
Retries happen automatically on `429` and `5xx` and on network failures — never on a `4xx`
|
|
211
|
+
other than `429`, because those fail identically however often you ask. `Retry-After` is
|
|
212
|
+
honoured when the API sends it, otherwise the backoff is exponential with jitter.
|
|
213
|
+
|
|
214
|
+
```python
|
|
215
|
+
client = PDFCraft("sk_live_...", max_retries=0) # off
|
|
216
|
+
client = PDFCraft("sk_live_...", timeout=200.0) # seconds
|
|
217
|
+
client = PDFCraft("sk_live_...", base_url="https://gateway.internal")
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
## Typing
|
|
221
|
+
|
|
222
|
+
Ships `py.typed`. Responses are plain dicts rather than dataclasses: the API's response shape
|
|
223
|
+
grows, and a dict that gains a key is a non-event where a frozen dataclass is a crash. The
|
|
224
|
+
option and error enumerations you might want to validate against are exported:
|
|
225
|
+
|
|
226
|
+
```python
|
|
227
|
+
from pdfcraft import ERROR_CODES, PAGE_FORMATS
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
Those are generated from the same definitions the API validates requests against, so they
|
|
231
|
+
cannot drift from the server.
|
|
232
|
+
|
|
233
|
+
## Links
|
|
234
|
+
|
|
235
|
+
- Docs — https://pdfcraft.dev/docs/
|
|
236
|
+
- Every error code, cause and fix — https://pdfcraft.dev/errors/
|
|
237
|
+
- Extraction guides on 27 real document shapes — https://pdfcraft.dev/guides/
|
|
238
|
+
- TypeScript SDK — `@pdfcraft-dev/pdf`
|
|
239
|
+
- Go SDK — `github.com/igaurav-dev/pdfcraft-go`
|
|
240
|
+
|
|
241
|
+
MIT licensed.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# Publishing `pdfcraft` to PyPI
|
|
2
|
+
|
|
3
|
+
The name **`pdfcraft` is unclaimed** (checked with `pip index versions pdfcraft`, which
|
|
4
|
+
returns nothing). Note that an unrelated project called **`pdf-craft`** exists at 2.2.0 — a
|
|
5
|
+
different name under PyPI's normalisation rules, but close enough that the README and the
|
|
6
|
+
project description should make clear which one this is.
|
|
7
|
+
|
|
8
|
+
## Once, to set up
|
|
9
|
+
|
|
10
|
+
1. Create the PyPI project by uploading the first release (below). There is no way to reserve
|
|
11
|
+
a name without publishing.
|
|
12
|
+
2. Prefer a **Trusted Publisher** (PyPI → the project → Publishing) over a long-lived API
|
|
13
|
+
token, so no credential sits on disk. A token works too; scope it to this project only.
|
|
14
|
+
|
|
15
|
+
## Every release
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
# 1. The contract, first. A release whose generated constants are stale ships
|
|
19
|
+
# something that disagrees with the server.
|
|
20
|
+
node ../pdfcraft/packages/contract/scripts/sync-polyglot.mjs .
|
|
21
|
+
git diff --stat src/pdfcraft/_contract.py # empty is fine; non-empty needs a version bump
|
|
22
|
+
|
|
23
|
+
# 2. Bump the version in ONE place.
|
|
24
|
+
$EDITOR src/pdfcraft/_version.py
|
|
25
|
+
|
|
26
|
+
# 3. Test.
|
|
27
|
+
PYTHONPATH=src python3 -m pytest tests/ -q
|
|
28
|
+
|
|
29
|
+
# 4. Build both artefacts.
|
|
30
|
+
python3 -m pip install --upgrade build twine
|
|
31
|
+
python3 -m build # -> dist/*.whl and dist/*.tar.gz
|
|
32
|
+
|
|
33
|
+
# 5. Check the rendered README before it is public; PyPI will not let you
|
|
34
|
+
# re-upload the same version to fix a broken one.
|
|
35
|
+
python3 -m twine check dist/*
|
|
36
|
+
|
|
37
|
+
# 6. Upload.
|
|
38
|
+
python3 -m twine upload dist/*
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## The rule that has no undo
|
|
42
|
+
|
|
43
|
+
**A version number on PyPI can never be reused**, even after deleting the release. A broken
|
|
44
|
+
1.3.0 means shipping 1.3.1, and 1.3.0 stays visible forever. `twine check` and a real
|
|
45
|
+
`pip install` from `dist/` are the two minutes that prevent it:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
python3 -m venv /tmp/verify && /tmp/verify/bin/pip install dist/*.whl
|
|
49
|
+
/tmp/verify/bin/python -c "import pdfcraft; print(pdfcraft.__version__)"
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Keep in step
|
|
53
|
+
|
|
54
|
+
The three SDKs share a version line so a reader can tell at a glance whether their Python
|
|
55
|
+
client knows about the same contract as their TypeScript one. When the contract changes, sync
|
|
56
|
+
and release all three, or none.
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
# PDFCraft for Python
|
|
2
|
+
|
|
3
|
+
HTML to PDF, PDF to structured JSON, and accessibility triage for a whole document estate.
|
|
4
|
+
The official Python client for [PDFCraft](https://pdfcraft.dev).
|
|
5
|
+
|
|
6
|
+
**Zero dependencies.** The whole client is `urllib.request` plus a retry loop, so it installs
|
|
7
|
+
into a Lambda or a slim container without dragging a transitive tree behind it.
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install pdfcraft-dev
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Installs as `pdfcraft-dev`, imports as `pdfcraft` — the same split as
|
|
14
|
+
`python-dateutil`/`dateutil`. The plain name was taken on PyPI by an unrelated
|
|
15
|
+
project, and the npm package is `@pdfcraft-dev/pdf`, so the two registries at
|
|
16
|
+
least agree with each other.
|
|
17
|
+
|
|
18
|
+
## Render
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
from pdfcraft import PDFCraft
|
|
22
|
+
|
|
23
|
+
client = PDFCraft("sk_live_...")
|
|
24
|
+
|
|
25
|
+
pdf = client.render(html="<h1>Invoice 1042</h1>")
|
|
26
|
+
open("invoice.pdf", "wb").write(pdf)
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
A URL instead of HTML, with page options:
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
pdf = client.render(
|
|
33
|
+
url="https://example.com/report",
|
|
34
|
+
options={
|
|
35
|
+
"format": "A4",
|
|
36
|
+
"margin": {"top": "20mm", "bottom": "20mm"},
|
|
37
|
+
"printBackground": True,
|
|
38
|
+
"waitFor": {"selector": "#chart-ready", "networkIdle": True},
|
|
39
|
+
},
|
|
40
|
+
filename="report.pdf",
|
|
41
|
+
)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Need a link rather than bytes — for an email, or a file too big to hold in memory:
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
result = client.render_to_url(html=invoice_html)
|
|
48
|
+
print(result["url"], result["expires_at"], result["pages"])
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Extract
|
|
52
|
+
|
|
53
|
+
A PDF in, its tables and labelled fields out, each with a bounding box:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
data = client.extract_pdf(open("statement.pdf", "rb").read())
|
|
57
|
+
|
|
58
|
+
for table in data["tables"]:
|
|
59
|
+
print(table["header"], len(table["rows"]), table["confidence"])
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Ask for specific fields by name and type:
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
data = client.extract(
|
|
66
|
+
file=base64_pdf,
|
|
67
|
+
schema={"invoice_total": "currency", "due_date": "date", "po_number": "string"},
|
|
68
|
+
)
|
|
69
|
+
print(data["fields"]["invoice_total"]["value"])
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
There is **no OCR**. A scanned document has no text layer, comes back as `extraction_failed`,
|
|
73
|
+
and is not billed. Nothing is guessed by a model, so the same document always produces the
|
|
74
|
+
same answer — which is the point if you are reconciling numbers.
|
|
75
|
+
|
|
76
|
+
Extraction is billed per page read, so `options={"pages": "1-3"}` narrows the bill as well as
|
|
77
|
+
the work.
|
|
78
|
+
|
|
79
|
+
## Async
|
|
80
|
+
|
|
81
|
+
For documents slow enough that you would rather not hold the connection:
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
job = client.render_async(url="https://example.com/huge", webhookUrl="https://you/hook")
|
|
85
|
+
status = client.get_render(job["id"])
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`get_extraction` polls extractions. An extraction id is not a render id — each endpoint 404s
|
|
89
|
+
on the other's ids, deliberately.
|
|
90
|
+
|
|
91
|
+
## Accessibility
|
|
92
|
+
|
|
93
|
+
Point it at a domain and it finds every PDF, checks each against PDF/UA and WCAG 2.1 AA, and
|
|
94
|
+
returns a report ranked by severity weighted by reach, with a remediation cost range.
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
import time
|
|
98
|
+
|
|
99
|
+
scan = client.scan(domain="example.gov", max_documents=500)
|
|
100
|
+
while scan["status"] not in ("succeeded", "failed"):
|
|
101
|
+
time.sleep(10)
|
|
102
|
+
scan = client.get_scan(scan["id"])
|
|
103
|
+
|
|
104
|
+
for doc in scan["documents"][:10]: # already ranked — this is the fix list
|
|
105
|
+
print(doc["severity"], doc["score"], doc["url"])
|
|
106
|
+
print(f" ${doc['cost_low_usd']:.0f}-${doc['cost_high_usd']:.0f} to remediate")
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Exactly one source: `domain=`, `sitemap=` or `urls=`. Passing none or two raises
|
|
110
|
+
`PDFCraftError` before anything is sent, because a round trip to be told you contradicted
|
|
111
|
+
yourself is a round trip wasted.
|
|
112
|
+
|
|
113
|
+
A scan runs for minutes — one request per second per host is a rule we do not break — so it
|
|
114
|
+
returns an id immediately and you poll. `max_documents` is **clamped to your plan rather than
|
|
115
|
+
refused**; `discovered` minus `checked` is what was found and not looked at, which is also the
|
|
116
|
+
upgrade prompt.
|
|
117
|
+
|
|
118
|
+
`report_url` on a finished scan is a share token. Anyone holding it can read the full HTML
|
|
119
|
+
report, and `GET /r/<token>/pdf` renders the same report to PDF through the render API. Treat
|
|
120
|
+
it as a credential, not an identifier.
|
|
121
|
+
|
|
122
|
+
Each finding carries `severity` (`blocker`, `major`, `minor`), the `wcag` criteria it breaks, a
|
|
123
|
+
`message` written for whoever approves the budget, and `technical_detail` for whoever does the
|
|
124
|
+
work. `occurrences` is volume, not severity — one check failing 1,535 times is one thing wrong,
|
|
125
|
+
fixed once, so never rank on it.
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from pdfcraft import A11Y_SEVERITIES, FINDING_LAYERS, SCAN_STATUSES
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Those are generated from the same contract the API validates against, so comparing against them
|
|
132
|
+
beats comparing against a string you typed.
|
|
133
|
+
|
|
134
|
+
Accessibility is a **separate subscription** from rendering. An account can hold either, both or
|
|
135
|
+
neither, and the free tier is a real scan of 25 documents with full findings.
|
|
136
|
+
|
|
137
|
+
## Errors
|
|
138
|
+
|
|
139
|
+
Every failure raises `PDFCraftError`. Branch on `.code`, which is stable; the message is
|
|
140
|
+
written for a human and may be reworded.
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
from pdfcraft import PDFCraft, PDFCraftError
|
|
144
|
+
|
|
145
|
+
try:
|
|
146
|
+
pdf = client.render(html=page)
|
|
147
|
+
except PDFCraftError as error:
|
|
148
|
+
if error.code == "quota_exceeded":
|
|
149
|
+
... # out of renders this period
|
|
150
|
+
elif error.code == "render_failed":
|
|
151
|
+
... # their HTML broke — billable, and worth logging
|
|
152
|
+
elif error.retryable:
|
|
153
|
+
... # already retried; this is after the last attempt
|
|
154
|
+
else:
|
|
155
|
+
raise
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
`error.docs_url` points at the page explaining that specific code.
|
|
159
|
+
|
|
160
|
+
### What is and is not billed
|
|
161
|
+
|
|
162
|
+
A render is billable if Chromium actually ran. Successes and `render_failed` count;
|
|
163
|
+
`render_timeout`, `internal_error` and every 4xx that never reached the browser do not.
|
|
164
|
+
|
|
165
|
+
## Retries
|
|
166
|
+
|
|
167
|
+
Retries happen automatically on `429` and `5xx` and on network failures — never on a `4xx`
|
|
168
|
+
other than `429`, because those fail identically however often you ask. `Retry-After` is
|
|
169
|
+
honoured when the API sends it, otherwise the backoff is exponential with jitter.
|
|
170
|
+
|
|
171
|
+
```python
|
|
172
|
+
client = PDFCraft("sk_live_...", max_retries=0) # off
|
|
173
|
+
client = PDFCraft("sk_live_...", timeout=200.0) # seconds
|
|
174
|
+
client = PDFCraft("sk_live_...", base_url="https://gateway.internal")
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
## Typing
|
|
178
|
+
|
|
179
|
+
Ships `py.typed`. Responses are plain dicts rather than dataclasses: the API's response shape
|
|
180
|
+
grows, and a dict that gains a key is a non-event where a frozen dataclass is a crash. The
|
|
181
|
+
option and error enumerations you might want to validate against are exported:
|
|
182
|
+
|
|
183
|
+
```python
|
|
184
|
+
from pdfcraft import ERROR_CODES, PAGE_FORMATS
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
Those are generated from the same definitions the API validates requests against, so they
|
|
188
|
+
cannot drift from the server.
|
|
189
|
+
|
|
190
|
+
## Links
|
|
191
|
+
|
|
192
|
+
- Docs — https://pdfcraft.dev/docs/
|
|
193
|
+
- Every error code, cause and fix — https://pdfcraft.dev/errors/
|
|
194
|
+
- Extraction guides on 27 real document shapes — https://pdfcraft.dev/guides/
|
|
195
|
+
- TypeScript SDK — `@pdfcraft-dev/pdf`
|
|
196
|
+
- Go SDK — `github.com/igaurav-dev/pdfcraft-go`
|
|
197
|
+
|
|
198
|
+
MIT licensed.
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
# Upper bound, and it is load-bearing rather than cautious.
|
|
3
|
+
#
|
|
4
|
+
# hatchling 1.28+ emits `Metadata-Version: 2.5`. PyPI's upload endpoint
|
|
5
|
+
# validates that field against a list it knows and currently stops at 2.4, so a
|
|
6
|
+
# 2.5 wheel is rejected with a bare `400 Bad Request` — after uploading to 100%,
|
|
7
|
+
# with no reason in the body. `twine check` does not catch it either, because it
|
|
8
|
+
# only renders the README.
|
|
9
|
+
#
|
|
10
|
+
# Measured, not guessed: 1.26.3 -> 2.3 1.27.0 -> 2.4 1.32.4 -> 2.5
|
|
11
|
+
#
|
|
12
|
+
# Raise this bound when PyPI announces 2.5 support, not before. scripts/publish.sh
|
|
13
|
+
# now asserts the built metadata version, so the pin and the check have to be
|
|
14
|
+
# wrong together for this to recur.
|
|
15
|
+
requires = ["hatchling>=1.24,<1.28"]
|
|
16
|
+
build-backend = "hatchling.build"
|
|
17
|
+
|
|
18
|
+
[project]
|
|
19
|
+
# The DISTRIBUTION name, which is not the import name.
|
|
20
|
+
#
|
|
21
|
+
# `pdfcraft` is unavailable: PyPI compares names with every non-alphanumeric
|
|
22
|
+
# character stripped, so the existing `pdf-craft` (a scanned-book to Markdown
|
|
23
|
+
# converter, unrelated to this) and `pdfcraft` both reduce to `pdfcraft` and the
|
|
24
|
+
# upload is refused as "too similar to an existing project" — a bare 400 with no
|
|
25
|
+
# reason in the body.
|
|
26
|
+
#
|
|
27
|
+
# The package still imports as `pdfcraft`; only `pip install` differs, the same
|
|
28
|
+
# split as python-dateutil/dateutil. `pdfcraft-dev` matches the npm scope
|
|
29
|
+
# (@pdfcraft-dev/pdf) so the two registries tell one story.
|
|
30
|
+
name = "pdfcraft-dev"
|
|
31
|
+
dynamic = ["version"]
|
|
32
|
+
description = "HTML to PDF, and PDF to structured JSON, in one call. The official PDFCraft SDK."
|
|
33
|
+
readme = "README.md"
|
|
34
|
+
license = { file = "LICENSE" }
|
|
35
|
+
requires-python = ">=3.9"
|
|
36
|
+
authors = [{ name = "PDFCraft" }]
|
|
37
|
+
# Zero runtime dependencies, on purpose: the whole client is urllib.request
|
|
38
|
+
# plus a retry loop. Dragging requests and its transitive tree into a
|
|
39
|
+
# customer's lockfile to save forty lines is a bad trade, especially for
|
|
40
|
+
# anyone installing into a Lambda or a slim container.
|
|
41
|
+
dependencies = []
|
|
42
|
+
keywords = [
|
|
43
|
+
"pdf", "pdf-api", "html-to-pdf", "html-to-pdf-api", "url-to-pdf",
|
|
44
|
+
"pdf-generation", "headless-chrome", "chromium", "invoice",
|
|
45
|
+
"pdf-extraction", "pdf-to-json", "extract-tables-from-pdf",
|
|
46
|
+
"pdf-table-extraction", "pdf-accessibility",
|
|
47
|
+
]
|
|
48
|
+
classifiers = [
|
|
49
|
+
"Development Status :: 4 - Beta",
|
|
50
|
+
"Intended Audience :: Developers",
|
|
51
|
+
"License :: OSI Approved :: MIT License",
|
|
52
|
+
"Programming Language :: Python :: 3",
|
|
53
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
54
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
55
|
+
"Topic :: Printing",
|
|
56
|
+
"Typing :: Typed",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
[project.urls]
|
|
60
|
+
Homepage = "https://pdfcraft.dev"
|
|
61
|
+
Documentation = "https://pdfcraft.dev/sdk/python/"
|
|
62
|
+
Source = "https://github.com/igaurav-dev/pdfcraft-python"
|
|
63
|
+
Issues = "https://github.com/igaurav-dev/pdfcraft-python/issues"
|
|
64
|
+
|
|
65
|
+
[tool.hatch.version]
|
|
66
|
+
path = "src/pdfcraft/_version.py"
|
|
67
|
+
|
|
68
|
+
[tool.hatch.build.targets.wheel]
|
|
69
|
+
packages = ["src/pdfcraft"]
|
|
70
|
+
|
|
71
|
+
[tool.pytest.ini_options]
|
|
72
|
+
# src-layout means the package is not importable from the repo root, so a bare
|
|
73
|
+
# `pytest` fails with ModuleNotFoundError on a fresh clone. Setting pythonpath
|
|
74
|
+
# here rather than expecting everyone to remember PYTHONPATH=src — including
|
|
75
|
+
# scripts/publish.sh, which got this wrong once.
|
|
76
|
+
pythonpath = ["src"]
|
|
77
|
+
testpaths = ["tests"]
|