pdfcraft-dev 1.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pdfcraft/__init__.py +33 -0
- pdfcraft/_client.py +266 -0
- pdfcraft/_contract.py +72 -0
- pdfcraft/_errors.py +66 -0
- pdfcraft/_version.py +6 -0
- pdfcraft/py.typed +0 -0
- pdfcraft_dev-1.4.0.dist-info/METADATA +241 -0
- pdfcraft_dev-1.4.0.dist-info/RECORD +10 -0
- pdfcraft_dev-1.4.0.dist-info/WHEEL +4 -0
- pdfcraft_dev-1.4.0.dist-info/licenses/LICENSE +21 -0
pdfcraft/__init__.py
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""PDFCraft — HTML to PDF, and PDF to structured JSON, in one call.
|
|
2
|
+
|
|
3
|
+
from pdfcraft import PDFCraft
|
|
4
|
+
|
|
5
|
+
client = PDFCraft("sk_live_...")
|
|
6
|
+
pdf = client.render(html="<h1>hello</h1>")
|
|
7
|
+
data = client.extract_pdf(pdf)
|
|
8
|
+
|
|
9
|
+
Zero dependencies. Every failure raises `PDFCraftError`; branch on `.code`.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from ._client import DEFAULT_BASE_URL, PDFCraft
|
|
13
|
+
from ._contract import (
|
|
14
|
+
A11Y_SEVERITIES,
|
|
15
|
+
ERROR_CODES,
|
|
16
|
+
FINDING_LAYERS,
|
|
17
|
+
PAGE_FORMATS,
|
|
18
|
+
SCAN_STATUSES,
|
|
19
|
+
)
|
|
20
|
+
from ._errors import PDFCraftError
|
|
21
|
+
from ._version import __version__
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"PDFCraft",
|
|
25
|
+
"PDFCraftError",
|
|
26
|
+
"ERROR_CODES",
|
|
27
|
+
"PAGE_FORMATS",
|
|
28
|
+
"A11Y_SEVERITIES",
|
|
29
|
+
"SCAN_STATUSES",
|
|
30
|
+
"FINDING_LAYERS",
|
|
31
|
+
"DEFAULT_BASE_URL",
|
|
32
|
+
"__version__",
|
|
33
|
+
]
|
pdfcraft/_client.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""The PDFCraft client.
|
|
2
|
+
|
|
3
|
+
Mirrors the TypeScript SDK's surface method for method, so the docs can show
|
|
4
|
+
the same call in both languages and mean it. Where the two differ it is
|
|
5
|
+
because Python idiom demands it: snake_case, keyword arguments, and bytes
|
|
6
|
+
rather than ``Uint8Array``.
|
|
7
|
+
|
|
8
|
+
Zero dependencies, on purpose. The whole client is ``urllib.request`` plus
|
|
9
|
+
about forty lines of retry logic, and an SDK that drags ``requests`` and its
|
|
10
|
+
transitive tree into a customer's lockfile to save those forty lines is a bad
|
|
11
|
+
trade — especially for anyone installing into a Lambda or a slim container.
|
|
12
|
+
The TypeScript SDK makes the same choice with ``fetch``.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import base64
|
|
18
|
+
import json
|
|
19
|
+
import random
|
|
20
|
+
import time
|
|
21
|
+
import urllib.error
|
|
22
|
+
import urllib.request
|
|
23
|
+
from typing import Any, Mapping, Sequence
|
|
24
|
+
|
|
25
|
+
from ._errors import PDFCraftError, to_error
|
|
26
|
+
from ._version import __version__
|
|
27
|
+
|
|
28
|
+
DEFAULT_BASE_URL = "https://api.pdfcraft.dev"
|
|
29
|
+
DEFAULT_MAX_RETRIES = 3
|
|
30
|
+
# Just past the API's own 120s ceiling, so a server-side timeout surfaces as
|
|
31
|
+
# the API's `render_timeout` — which tells you what happened — rather than as a
|
|
32
|
+
# client-side socket timeout, which tells you nothing.
|
|
33
|
+
DEFAULT_TIMEOUT = 130.0
|
|
34
|
+
|
|
35
|
+
JsonDict = dict[str, Any]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class PDFCraft:
|
|
39
|
+
"""A PDFCraft API client.
|
|
40
|
+
|
|
41
|
+
>>> pdf = PDFCraft("sk_live_...").render(html="<h1>hello</h1>")
|
|
42
|
+
>>> pdf[:4]
|
|
43
|
+
b'%PDF'
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
def __init__(
|
|
47
|
+
self,
|
|
48
|
+
api_key: str,
|
|
49
|
+
*,
|
|
50
|
+
base_url: str = DEFAULT_BASE_URL,
|
|
51
|
+
max_retries: int = DEFAULT_MAX_RETRIES,
|
|
52
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
53
|
+
opener: urllib.request.OpenerDirector | None = None,
|
|
54
|
+
) -> None:
|
|
55
|
+
if not api_key:
|
|
56
|
+
raise PDFCraftError("invalid_api_key", "An API key is required.", 401)
|
|
57
|
+
self._api_key = api_key
|
|
58
|
+
self._base_url = base_url.rstrip("/")
|
|
59
|
+
self._max_retries = max_retries
|
|
60
|
+
self._timeout = timeout
|
|
61
|
+
# Injectable so the tests can run without a network, and so anyone
|
|
62
|
+
# behind a corporate proxy can hand in their own opener.
|
|
63
|
+
self._opener = opener or urllib.request.build_opener()
|
|
64
|
+
|
|
65
|
+
# ── rendering ─────────────────────────────────────────────────────────
|
|
66
|
+
|
|
67
|
+
def render(self, *, idempotency_key: str | None = None, **input: Any) -> bytes:
|
|
68
|
+
"""Render HTML or a URL and return the PDF bytes.
|
|
69
|
+
|
|
70
|
+
The method decides the output mode, so passing ``output`` yourself
|
|
71
|
+
could only contradict it — it is rejected rather than ignored, because
|
|
72
|
+
``render(output="url")`` quietly returning JSON is the kind of surprise
|
|
73
|
+
that costs an afternoon.
|
|
74
|
+
"""
|
|
75
|
+
body = _with_output(input, "binary")
|
|
76
|
+
return self._request("POST", "/v1/render", body, idempotency_key).read()
|
|
77
|
+
|
|
78
|
+
def render_to_url(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
|
|
79
|
+
"""Store the PDF and return a signed link instead of the bytes."""
|
|
80
|
+
body = _with_output(input, "url")
|
|
81
|
+
return _json(self._request("POST", "/v1/render", body, idempotency_key))
|
|
82
|
+
|
|
83
|
+
def render_async(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
|
|
84
|
+
"""Queue a render; the webhook fires when it settles."""
|
|
85
|
+
return _json(self._request("POST", "/v1/render/async", dict(input), idempotency_key))
|
|
86
|
+
|
|
87
|
+
def get_render(self, render_id: str) -> JsonDict:
|
|
88
|
+
return _json(self._request("GET", f"/v1/renders/{_quote(render_id)}"))
|
|
89
|
+
|
|
90
|
+
def usage(self) -> JsonDict:
|
|
91
|
+
"""Current period usage and quota."""
|
|
92
|
+
return _json(self._request("GET", "/v1/usage"))
|
|
93
|
+
|
|
94
|
+
# ── extraction ────────────────────────────────────────────────────────
|
|
95
|
+
|
|
96
|
+
def extract(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
|
|
97
|
+
"""A PDF in, its tables and labelled fields out, each with a bounding box.
|
|
98
|
+
|
|
99
|
+
``file`` takes base64 PDF bytes or an https URL to one; pass ``url`` or
|
|
100
|
+
``html`` instead and the page is rendered first, then extracted — one
|
|
101
|
+
call, one charge. Billed per page read, so ``options={"pages": ...}``
|
|
102
|
+
narrows the bill as well as the work.
|
|
103
|
+
|
|
104
|
+
There is no OCR. A scan has no text layer and comes back as
|
|
105
|
+
``extraction_failed``, unbilled. Nothing is guessed by a model, so the
|
|
106
|
+
same document always produces the same answer.
|
|
107
|
+
"""
|
|
108
|
+
body = _with_output(input, "inline")
|
|
109
|
+
return _json(self._request("POST", "/v1/extract", body, idempotency_key))
|
|
110
|
+
|
|
111
|
+
def extract_to_url(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
|
|
112
|
+
body = _with_output(input, "url")
|
|
113
|
+
return _json(self._request("POST", "/v1/extract", body, idempotency_key))
|
|
114
|
+
|
|
115
|
+
def extract_async(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
|
|
116
|
+
return _json(self._request("POST", "/v1/extract/async", dict(input), idempotency_key))
|
|
117
|
+
|
|
118
|
+
def get_extraction(self, extraction_id: str) -> JsonDict:
|
|
119
|
+
"""Poll one extraction.
|
|
120
|
+
|
|
121
|
+
An extraction id is not a render id: ``get_render`` will 404 on one and
|
|
122
|
+
this will 404 on a render id, deliberately.
|
|
123
|
+
"""
|
|
124
|
+
return _json(self._request("GET", f"/v1/extractions/{_quote(extraction_id)}"))
|
|
125
|
+
|
|
126
|
+
def extract_pdf(
|
|
127
|
+
self,
|
|
128
|
+
pdf: bytes,
|
|
129
|
+
*,
|
|
130
|
+
idempotency_key: str | None = None,
|
|
131
|
+
**input: Any,
|
|
132
|
+
) -> JsonDict:
|
|
133
|
+
"""Convenience for the common case: hand it PDF bytes, it does the base64.
|
|
134
|
+
|
|
135
|
+
Kept out of ``extract`` itself so that call stays a plain dict you can
|
|
136
|
+
log, diff or replay.
|
|
137
|
+
"""
|
|
138
|
+
return self.extract(
|
|
139
|
+
file=base64.b64encode(pdf).decode("ascii"),
|
|
140
|
+
idempotency_key=idempotency_key,
|
|
141
|
+
**input,
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
# ── accessibility ─────────────────────────────────────────────────────
|
|
145
|
+
|
|
146
|
+
def scan(
|
|
147
|
+
self,
|
|
148
|
+
*,
|
|
149
|
+
domain: str | None = None,
|
|
150
|
+
sitemap: str | None = None,
|
|
151
|
+
urls: Sequence[str] | None = None,
|
|
152
|
+
**options: Any,
|
|
153
|
+
) -> JsonDict:
|
|
154
|
+
"""Start an accessibility scan. Returns at once with an id to poll.
|
|
155
|
+
|
|
156
|
+
Exactly one source. A scan of a thousand documents at one request per
|
|
157
|
+
second per host has a floor measured in minutes, so there is nothing to
|
|
158
|
+
return but an id and somewhere to look.
|
|
159
|
+
|
|
160
|
+
scan = client.scan(domain="example.gov", max_documents=500)
|
|
161
|
+
while scan["status"] not in ("succeeded", "failed"):
|
|
162
|
+
time.sleep(10)
|
|
163
|
+
scan = client.get_scan(scan["id"])
|
|
164
|
+
|
|
165
|
+
``max_documents`` is clamped to your plan rather than refused, and the
|
|
166
|
+
gap between what was found and what was checked is reported back as
|
|
167
|
+
``discovered`` minus ``checked``.
|
|
168
|
+
"""
|
|
169
|
+
source = {
|
|
170
|
+
key: value
|
|
171
|
+
for key, value in (("domain", domain), ("sitemap", sitemap), ("urls", list(urls) if urls else None))
|
|
172
|
+
if value is not None
|
|
173
|
+
}
|
|
174
|
+
if len(source) != 1:
|
|
175
|
+
raise PDFCraftError(
|
|
176
|
+
"invalid_request",
|
|
177
|
+
"exactly one of domain, sitemap or urls is required",
|
|
178
|
+
400,
|
|
179
|
+
)
|
|
180
|
+
body: JsonDict = {"source": source}
|
|
181
|
+
if options:
|
|
182
|
+
body["options"] = dict(options)
|
|
183
|
+
return _json(self._request("POST", "/v1/a11y/scan", body))
|
|
184
|
+
|
|
185
|
+
def get_scan(self, scan_id: str) -> JsonDict:
|
|
186
|
+
"""Poll one scan.
|
|
187
|
+
|
|
188
|
+
Returns progress while it runs and the full result — every document,
|
|
189
|
+
ranked, with findings and cost — once it succeeds. ``report_url`` is a
|
|
190
|
+
share token: anyone with it can read the report, no account needed.
|
|
191
|
+
"""
|
|
192
|
+
return _json(self._request("GET", f"/v1/a11y/scans/{_quote(scan_id)}"))
|
|
193
|
+
|
|
194
|
+
# ── transport ─────────────────────────────────────────────────────────
|
|
195
|
+
|
|
196
|
+
def _request(
|
|
197
|
+
self,
|
|
198
|
+
method: str,
|
|
199
|
+
path: str,
|
|
200
|
+
body: Mapping[str, Any] | None = None,
|
|
201
|
+
idempotency_key: str | None = None,
|
|
202
|
+
) -> Any:
|
|
203
|
+
payload = json.dumps(body).encode("utf-8") if body is not None else None
|
|
204
|
+
headers = {
|
|
205
|
+
"authorization": f"Bearer {self._api_key}",
|
|
206
|
+
"user-agent": f"pdfcraft-sdk-python/{__version__}",
|
|
207
|
+
}
|
|
208
|
+
if payload is not None:
|
|
209
|
+
headers["content-type"] = "application/json"
|
|
210
|
+
if idempotency_key:
|
|
211
|
+
headers["idempotency-key"] = idempotency_key
|
|
212
|
+
|
|
213
|
+
last: PDFCraftError | None = None
|
|
214
|
+
for attempt in range(self._max_retries + 1):
|
|
215
|
+
request = urllib.request.Request(
|
|
216
|
+
f"{self._base_url}{path}", data=payload, headers=headers, method=method
|
|
217
|
+
)
|
|
218
|
+
try:
|
|
219
|
+
return self._opener.open(request, timeout=self._timeout)
|
|
220
|
+
except urllib.error.HTTPError as response:
|
|
221
|
+
last = to_error(response.code, response.read())
|
|
222
|
+
if not last.retryable or attempt == self._max_retries:
|
|
223
|
+
raise last from None
|
|
224
|
+
time.sleep(_retry_after(response) or _backoff(attempt))
|
|
225
|
+
except (urllib.error.URLError, TimeoutError, OSError) as cause:
|
|
226
|
+
last = PDFCraftError("network_error", f"Could not reach PDFCraft: {cause}", 0)
|
|
227
|
+
if attempt == self._max_retries:
|
|
228
|
+
raise last from None
|
|
229
|
+
time.sleep(_backoff(attempt))
|
|
230
|
+
|
|
231
|
+
raise last or PDFCraftError("internal_error", "Request failed.", 500)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _with_output(input: Mapping[str, Any], mode: str) -> JsonDict:
|
|
235
|
+
if "output" in input:
|
|
236
|
+
raise PDFCraftError(
|
|
237
|
+
"invalid_request",
|
|
238
|
+
"Do not pass `output`; the method you call decides it.",
|
|
239
|
+
400,
|
|
240
|
+
)
|
|
241
|
+
return {**input, "output": mode}
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _json(response: Any) -> JsonDict:
|
|
245
|
+
return json.loads(response.read().decode("utf-8"))
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _quote(value: str) -> str:
|
|
249
|
+
from urllib.parse import quote
|
|
250
|
+
|
|
251
|
+
return quote(value, safe="")
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _backoff(attempt: int) -> float:
|
|
255
|
+
"""Exponential with jitter, matching the TypeScript SDK's curve exactly."""
|
|
256
|
+
return 0.5 * (2**attempt) * (0.75 + random.random() * 0.5)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _retry_after(response: Any) -> float | None:
|
|
260
|
+
header = response.headers.get("retry-after") if hasattr(response, "headers") else None
|
|
261
|
+
if not header:
|
|
262
|
+
return None
|
|
263
|
+
try:
|
|
264
|
+
return max(0.0, float(header))
|
|
265
|
+
except (TypeError, ValueError):
|
|
266
|
+
return None
|
pdfcraft/_contract.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# GENERATED FILE — do not edit.
|
|
2
|
+
# Written by packages/contract/scripts/sync-polyglot.mjs from the same
|
|
3
|
+
# definitions the PDFCraft API validates requests against.
|
|
4
|
+
# Editing it by hand will be silently overwritten on the next sync.
|
|
5
|
+
|
|
6
|
+
CONTRACT_HASH = "f5af303a2509e0d11d0c6efdc069e9acb80fddcd43a337a4789fa8f538f2588a"
|
|
7
|
+
|
|
8
|
+
#: Stable machine-readable codes. Branch on these, never on the message.
|
|
9
|
+
ERROR_CODES: tuple[str, ...] = (
|
|
10
|
+
"invalid_request",
|
|
11
|
+
"invalid_api_key",
|
|
12
|
+
"payment_required",
|
|
13
|
+
"not_found",
|
|
14
|
+
"render_timeout",
|
|
15
|
+
"render_failed",
|
|
16
|
+
"rate_limited",
|
|
17
|
+
"quota_exceeded",
|
|
18
|
+
"demo_busy",
|
|
19
|
+
"unsupported_file",
|
|
20
|
+
"extraction_failed",
|
|
21
|
+
"internal_error",
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
PAGE_FORMATS: tuple[str, ...] = (
|
|
25
|
+
"A4",
|
|
26
|
+
"A3",
|
|
27
|
+
"A5",
|
|
28
|
+
"Letter",
|
|
29
|
+
"Legal",
|
|
30
|
+
"Tabloid",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
EMULATE_MEDIA: tuple[str, ...] = (
|
|
34
|
+
"print",
|
|
35
|
+
"screen",
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
OUTPUT_MODES: tuple[str, ...] = (
|
|
39
|
+
"binary",
|
|
40
|
+
"url",
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
EXTRACT_OUTPUT_MODES: tuple[str, ...] = (
|
|
44
|
+
"inline",
|
|
45
|
+
"url",
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
MAX_TIMEOUT_MS = 120000
|
|
49
|
+
DEFAULT_TIMEOUT_MS = 30000
|
|
50
|
+
|
|
51
|
+
#: Accessibility severity, worst first. Ordered, not merely enumerated.
|
|
52
|
+
A11Y_SEVERITIES: tuple[str, ...] = (
|
|
53
|
+
"blocker",
|
|
54
|
+
"major",
|
|
55
|
+
"minor",
|
|
56
|
+
"pass",
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
#: Scan lifecycle. Poll until the status is terminal.
|
|
60
|
+
SCAN_STATUSES: tuple[str, ...] = (
|
|
61
|
+
"queued",
|
|
62
|
+
"crawling",
|
|
63
|
+
"checking",
|
|
64
|
+
"succeeded",
|
|
65
|
+
"failed",
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
#: Which layer found a finding. "geometric" is the half a conformance validator cannot do.
|
|
69
|
+
FINDING_LAYERS: tuple[str, ...] = (
|
|
70
|
+
"machine",
|
|
71
|
+
"geometric",
|
|
72
|
+
)
|
pdfcraft/_errors.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Every failure from the API arrives as one exception type. Branch on ``.code``.
|
|
2
|
+
|
|
3
|
+
Mirrors ``src/errors.ts`` in the TypeScript SDK deliberately, down to which
|
|
4
|
+
statuses count as retryable, so a bug report against one client can be read by
|
|
5
|
+
someone holding the other.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class PDFCraftError(Exception):
|
|
15
|
+
"""A PDFCraft API failure.
|
|
16
|
+
|
|
17
|
+
``code`` is the stable machine-readable string from the API's error
|
|
18
|
+
envelope — ``quota_exceeded``, ``render_timeout`` and so on — or
|
|
19
|
+
``network_error`` when the request never reached us. Branch on it rather
|
|
20
|
+
than on the message, which is written for humans and may be reworded.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
__slots__ = ("code", "status", "docs_url", "retryable")
|
|
24
|
+
|
|
25
|
+
def __init__(
|
|
26
|
+
self,
|
|
27
|
+
code: str,
|
|
28
|
+
message: str,
|
|
29
|
+
status: int,
|
|
30
|
+
docs_url: str | None = None,
|
|
31
|
+
) -> None:
|
|
32
|
+
super().__init__(message)
|
|
33
|
+
self.code = code
|
|
34
|
+
self.status = status
|
|
35
|
+
self.docs_url = docs_url
|
|
36
|
+
# A 4xx other than 429 will fail identically however many times we ask.
|
|
37
|
+
self.retryable = code == "network_error" or status == 429 or status >= 500
|
|
38
|
+
|
|
39
|
+
def __repr__(self) -> str: # pragma: no cover - debugging aid
|
|
40
|
+
return f"PDFCraftError(code={self.code!r}, status={self.status})"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def to_error(status: int, body: bytes) -> PDFCraftError:
|
|
44
|
+
"""Turn an error response into a typed exception.
|
|
45
|
+
|
|
46
|
+
Tolerates a non-JSON body on purpose: a proxy or load balancer in front of
|
|
47
|
+
the API returns an HTML error page, and a client that raises
|
|
48
|
+
``JSONDecodeError`` there hides the status code that would have explained
|
|
49
|
+
the problem.
|
|
50
|
+
"""
|
|
51
|
+
payload: Any = None
|
|
52
|
+
try:
|
|
53
|
+
payload = json.loads(body.decode("utf-8", "replace"))
|
|
54
|
+
except (ValueError, AttributeError):
|
|
55
|
+
payload = None
|
|
56
|
+
|
|
57
|
+
error = payload.get("error") if isinstance(payload, dict) else None
|
|
58
|
+
if not isinstance(error, dict):
|
|
59
|
+
error = {}
|
|
60
|
+
|
|
61
|
+
return PDFCraftError(
|
|
62
|
+
str(error.get("code") or "internal_error"),
|
|
63
|
+
str(error.get("message") or f"PDFCraft responded {status}"),
|
|
64
|
+
status,
|
|
65
|
+
error.get("docs_url") or None,
|
|
66
|
+
)
|
pdfcraft/_version.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
# src/pdfcraft/_version.py
|
|
2
|
+
#
|
|
3
|
+
# One place, read by the package metadata AND by the User-Agent header. Not
|
|
4
|
+
# read back out of the installed distribution: that fails when the package is
|
|
5
|
+
# run from a source checkout, which is exactly when someone is debugging.
|
|
6
|
+
__version__ = "1.4.0"
|
pdfcraft/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pdfcraft-dev
|
|
3
|
+
Version: 1.4.0
|
|
4
|
+
Summary: HTML to PDF, and PDF to structured JSON, in one call. The official PDFCraft SDK.
|
|
5
|
+
Project-URL: Homepage, https://pdfcraft.dev
|
|
6
|
+
Project-URL: Documentation, https://pdfcraft.dev/sdk/python/
|
|
7
|
+
Project-URL: Source, https://github.com/igaurav-dev/pdfcraft-python
|
|
8
|
+
Project-URL: Issues, https://github.com/igaurav-dev/pdfcraft-python/issues
|
|
9
|
+
Author: PDFCraft
|
|
10
|
+
License: MIT License
|
|
11
|
+
|
|
12
|
+
Copyright (c) 2026 Gaurav Singh
|
|
13
|
+
|
|
14
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
15
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
16
|
+
in the Software without restriction, including without limitation the rights
|
|
17
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
18
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
19
|
+
furnished to do so, subject to the following conditions:
|
|
20
|
+
|
|
21
|
+
The above copyright notice and this permission notice shall be included in all
|
|
22
|
+
copies or substantial portions of the Software.
|
|
23
|
+
|
|
24
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
25
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
26
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
27
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
28
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
29
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
30
|
+
SOFTWARE.
|
|
31
|
+
License-File: LICENSE
|
|
32
|
+
Keywords: chromium,extract-tables-from-pdf,headless-chrome,html-to-pdf,html-to-pdf-api,invoice,pdf,pdf-accessibility,pdf-api,pdf-extraction,pdf-generation,pdf-table-extraction,pdf-to-json,url-to-pdf
|
|
33
|
+
Classifier: Development Status :: 4 - Beta
|
|
34
|
+
Classifier: Intended Audience :: Developers
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Programming Language :: Python :: 3
|
|
37
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
38
|
+
Classifier: Topic :: Printing
|
|
39
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
40
|
+
Classifier: Typing :: Typed
|
|
41
|
+
Requires-Python: >=3.9
|
|
42
|
+
Description-Content-Type: text/markdown
|
|
43
|
+
|
|
44
|
+
# PDFCraft for Python
|
|
45
|
+
|
|
46
|
+
HTML to PDF, PDF to structured JSON, and accessibility triage for a whole document estate.
|
|
47
|
+
The official Python client for [PDFCraft](https://pdfcraft.dev).
|
|
48
|
+
|
|
49
|
+
**Zero dependencies.** The whole client is `urllib.request` plus a retry loop, so it installs
|
|
50
|
+
into a Lambda or a slim container without dragging a transitive tree behind it.
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
pip install pdfcraft-dev
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Installs as `pdfcraft-dev`, imports as `pdfcraft` — the same split as
|
|
57
|
+
`python-dateutil`/`dateutil`. The plain name was taken on PyPI by an unrelated
|
|
58
|
+
project, and the npm package is `@pdfcraft-dev/pdf`, so the two registries at
|
|
59
|
+
least agree with each other.
|
|
60
|
+
|
|
61
|
+
## Render
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from pdfcraft import PDFCraft
|
|
65
|
+
|
|
66
|
+
client = PDFCraft("sk_live_...")
|
|
67
|
+
|
|
68
|
+
pdf = client.render(html="<h1>Invoice 1042</h1>")
|
|
69
|
+
open("invoice.pdf", "wb").write(pdf)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
A URL instead of HTML, with page options:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
pdf = client.render(
|
|
76
|
+
url="https://example.com/report",
|
|
77
|
+
options={
|
|
78
|
+
"format": "A4",
|
|
79
|
+
"margin": {"top": "20mm", "bottom": "20mm"},
|
|
80
|
+
"printBackground": True,
|
|
81
|
+
"waitFor": {"selector": "#chart-ready", "networkIdle": True},
|
|
82
|
+
},
|
|
83
|
+
filename="report.pdf",
|
|
84
|
+
)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Need a link rather than bytes — for an email, or a file too big to hold in memory:
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
result = client.render_to_url(html=invoice_html)
|
|
91
|
+
print(result["url"], result["expires_at"], result["pages"])
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Extract
|
|
95
|
+
|
|
96
|
+
A PDF in, its tables and labelled fields out, each with a bounding box:
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
data = client.extract_pdf(open("statement.pdf", "rb").read())
|
|
100
|
+
|
|
101
|
+
for table in data["tables"]:
|
|
102
|
+
print(table["header"], len(table["rows"]), table["confidence"])
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Ask for specific fields by name and type:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
data = client.extract(
|
|
109
|
+
file=base64_pdf,
|
|
110
|
+
schema={"invoice_total": "currency", "due_date": "date", "po_number": "string"},
|
|
111
|
+
)
|
|
112
|
+
print(data["fields"]["invoice_total"]["value"])
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
There is **no OCR**. A scanned document has no text layer, comes back as `extraction_failed`,
|
|
116
|
+
and is not billed. Nothing is guessed by a model, so the same document always produces the
|
|
117
|
+
same answer — which is the point if you are reconciling numbers.
|
|
118
|
+
|
|
119
|
+
Extraction is billed per page read, so `options={"pages": "1-3"}` narrows the bill as well as
|
|
120
|
+
the work.
|
|
121
|
+
|
|
122
|
+
## Async
|
|
123
|
+
|
|
124
|
+
For documents slow enough that you would rather not hold the connection:
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
job = client.render_async(url="https://example.com/huge", webhookUrl="https://you/hook")
|
|
128
|
+
status = client.get_render(job["id"])
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
`get_extraction` polls extractions. An extraction id is not a render id — each endpoint 404s
|
|
132
|
+
on the other's ids, deliberately.
|
|
133
|
+
|
|
134
|
+
## Accessibility
|
|
135
|
+
|
|
136
|
+
Point it at a domain and it finds every PDF, checks each against PDF/UA and WCAG 2.1 AA, and
|
|
137
|
+
returns a report ranked by severity weighted by reach, with a remediation cost range.
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
import time
|
|
141
|
+
|
|
142
|
+
scan = client.scan(domain="example.gov", max_documents=500)
|
|
143
|
+
while scan["status"] not in ("succeeded", "failed"):
|
|
144
|
+
time.sleep(10)
|
|
145
|
+
scan = client.get_scan(scan["id"])
|
|
146
|
+
|
|
147
|
+
for doc in scan["documents"][:10]: # already ranked — this is the fix list
|
|
148
|
+
print(doc["severity"], doc["score"], doc["url"])
|
|
149
|
+
print(f" ${doc['cost_low_usd']:.0f}-${doc['cost_high_usd']:.0f} to remediate")
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Exactly one source: `domain=`, `sitemap=` or `urls=`. Passing none or two raises
|
|
153
|
+
`PDFCraftError` before anything is sent, because a round trip to be told you contradicted
|
|
154
|
+
yourself is a round trip wasted.
|
|
155
|
+
|
|
156
|
+
A scan runs for minutes — one request per second per host is a rule we do not break — so it
|
|
157
|
+
returns an id immediately and you poll. `max_documents` is **clamped to your plan rather than
|
|
158
|
+
refused**; `discovered` minus `checked` is what was found and not looked at, which is also the
|
|
159
|
+
upgrade prompt.
|
|
160
|
+
|
|
161
|
+
`report_url` on a finished scan is a share token. Anyone holding it can read the full HTML
|
|
162
|
+
report, and `GET /r/<token>/pdf` renders the same report to PDF through the render API. Treat
|
|
163
|
+
it as a credential, not an identifier.
|
|
164
|
+
|
|
165
|
+
Each finding carries `severity` (`blocker`, `major`, `minor`), the `wcag` criteria it breaks, a
|
|
166
|
+
`message` written for whoever approves the budget, and `technical_detail` for whoever does the
|
|
167
|
+
work. `occurrences` is volume, not severity — one check failing 1,535 times is one thing wrong,
|
|
168
|
+
fixed once, so never rank on it.
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from pdfcraft import A11Y_SEVERITIES, FINDING_LAYERS, SCAN_STATUSES
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
Those are generated from the same contract the API validates against, so comparing against them
|
|
175
|
+
beats comparing against a string you typed.
|
|
176
|
+
|
|
177
|
+
Accessibility is a **separate subscription** from rendering. An account can hold either, both or
|
|
178
|
+
neither, and the free tier is a real scan of 25 documents with full findings.
|
|
179
|
+
|
|
180
|
+
## Errors
|
|
181
|
+
|
|
182
|
+
Every failure raises `PDFCraftError`. Branch on `.code`, which is stable; the message is
|
|
183
|
+
written for a human and may be reworded.
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from pdfcraft import PDFCraft, PDFCraftError
|
|
187
|
+
|
|
188
|
+
try:
|
|
189
|
+
pdf = client.render(html=page)
|
|
190
|
+
except PDFCraftError as error:
|
|
191
|
+
if error.code == "quota_exceeded":
|
|
192
|
+
... # out of renders this period
|
|
193
|
+
elif error.code == "render_failed":
|
|
194
|
+
... # their HTML broke — billable, and worth logging
|
|
195
|
+
elif error.retryable:
|
|
196
|
+
... # already retried; this is after the last attempt
|
|
197
|
+
else:
|
|
198
|
+
raise
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
`error.docs_url` points at the page explaining that specific code.
|
|
202
|
+
|
|
203
|
+
### What is and is not billed
|
|
204
|
+
|
|
205
|
+
A render is billable if Chromium actually ran. Successes and `render_failed` count;
|
|
206
|
+
`render_timeout`, `internal_error` and every 4xx that never reached the browser do not.
|
|
207
|
+
|
|
208
|
+
## Retries
|
|
209
|
+
|
|
210
|
+
Retries happen automatically on `429` and `5xx` and on network failures — never on a `4xx`
|
|
211
|
+
other than `429`, because those fail identically however often you ask. `Retry-After` is
|
|
212
|
+
honoured when the API sends it, otherwise the backoff is exponential with jitter.
|
|
213
|
+
|
|
214
|
+
```python
|
|
215
|
+
client = PDFCraft("sk_live_...", max_retries=0) # off
|
|
216
|
+
client = PDFCraft("sk_live_...", timeout=200.0) # seconds
|
|
217
|
+
client = PDFCraft("sk_live_...", base_url="https://gateway.internal")
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
## Typing
|
|
221
|
+
|
|
222
|
+
Ships `py.typed`. Responses are plain dicts rather than dataclasses: the API's response shape
|
|
223
|
+
grows, and a dict that gains a key is a non-event where a frozen dataclass is a crash. The
|
|
224
|
+
option and error enumerations you might want to validate against are exported:
|
|
225
|
+
|
|
226
|
+
```python
|
|
227
|
+
from pdfcraft import ERROR_CODES, PAGE_FORMATS
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
Those are generated from the same definitions the API validates requests against, so they
|
|
231
|
+
cannot drift from the server.
|
|
232
|
+
|
|
233
|
+
## Links
|
|
234
|
+
|
|
235
|
+
- Docs — https://pdfcraft.dev/docs/
|
|
236
|
+
- Every error code, cause and fix — https://pdfcraft.dev/errors/
|
|
237
|
+
- Extraction guides on 27 real document shapes — https://pdfcraft.dev/guides/
|
|
238
|
+
- TypeScript SDK — `@pdfcraft-dev/pdf`
|
|
239
|
+
- Go SDK — `github.com/igaurav-dev/pdfcraft-go`
|
|
240
|
+
|
|
241
|
+
MIT licensed.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
pdfcraft/__init__.py,sha256=TL628Pcxuggf2SVOirPN9uZue2KuB_U16aRhnMTGB0Q,746
|
|
2
|
+
pdfcraft/_client.py,sha256=9s0PFQ56rU3fESwv6bIkauTB_xLCpKj2odhBKAmJCcE,10767
|
|
3
|
+
pdfcraft/_contract.py,sha256=H8RNxvJqYKI2CWMMn7zibTpFzlVNGKb6p0NHX0yeX4g,1539
|
|
4
|
+
pdfcraft/_errors.py,sha256=6eDr5jDcF6tLR6i9SDU5k9qRKmhlVqLF1aeCxfoIn-8,2202
|
|
5
|
+
pdfcraft/_version.py,sha256=mQUmfuD6SzNbDotSl4VdGUGeiArADo4hejfW5nDZ_eE,279
|
|
6
|
+
pdfcraft/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
pdfcraft_dev-1.4.0.dist-info/METADATA,sha256=mscBMul08Gg_M_WAE3juFgFUjj4NEQ0Mqt_rdJC0rCQ,9093
|
|
8
|
+
pdfcraft_dev-1.4.0.dist-info/WHEEL,sha256=qtCwoSJWgHk21S1Kb4ihdzI2rlJ1ZKaIurTj_ngOhyQ,87
|
|
9
|
+
pdfcraft_dev-1.4.0.dist-info/licenses/LICENSE,sha256=VvFCyi2di0rvS6eYiaa6nEK_aybkXjxeQ8iJVuafvjk,1069
|
|
10
|
+
pdfcraft_dev-1.4.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Gaurav Singh
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|