pdfcraft-dev 1.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pdfcraft/__init__.py ADDED
@@ -0,0 +1,33 @@
1
+ """PDFCraft — HTML to PDF, and PDF to structured JSON, in one call.
2
+
3
+ from pdfcraft import PDFCraft
4
+
5
+ client = PDFCraft("sk_live_...")
6
+ pdf = client.render(html="<h1>hello</h1>")
7
+ data = client.extract_pdf(pdf)
8
+
9
+ Zero dependencies. Every failure raises `PDFCraftError`; branch on `.code`.
10
+ """
11
+
12
+ from ._client import DEFAULT_BASE_URL, PDFCraft
13
+ from ._contract import (
14
+ A11Y_SEVERITIES,
15
+ ERROR_CODES,
16
+ FINDING_LAYERS,
17
+ PAGE_FORMATS,
18
+ SCAN_STATUSES,
19
+ )
20
+ from ._errors import PDFCraftError
21
+ from ._version import __version__
22
+
23
+ __all__ = [
24
+ "PDFCraft",
25
+ "PDFCraftError",
26
+ "ERROR_CODES",
27
+ "PAGE_FORMATS",
28
+ "A11Y_SEVERITIES",
29
+ "SCAN_STATUSES",
30
+ "FINDING_LAYERS",
31
+ "DEFAULT_BASE_URL",
32
+ "__version__",
33
+ ]
pdfcraft/_client.py ADDED
@@ -0,0 +1,266 @@
1
+ """The PDFCraft client.
2
+
3
+ Mirrors the TypeScript SDK's surface method for method, so the docs can show
4
+ the same call in both languages and mean it. Where the two differ it is
5
+ because Python idiom demands it: snake_case, keyword arguments, and bytes
6
+ rather than ``Uint8Array``.
7
+
8
+ Zero dependencies, on purpose. The whole client is ``urllib.request`` plus
9
+ about forty lines of retry logic, and an SDK that drags ``requests`` and its
10
+ transitive tree into a customer's lockfile to save those forty lines is a bad
11
+ trade — especially for anyone installing into a Lambda or a slim container.
12
+ The TypeScript SDK makes the same choice with ``fetch``.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import base64
18
+ import json
19
+ import random
20
+ import time
21
+ import urllib.error
22
+ import urllib.request
23
+ from typing import Any, Mapping, Sequence
24
+
25
+ from ._errors import PDFCraftError, to_error
26
+ from ._version import __version__
27
+
28
+ DEFAULT_BASE_URL = "https://api.pdfcraft.dev"
29
+ DEFAULT_MAX_RETRIES = 3
30
+ # Just past the API's own 120s ceiling, so a server-side timeout surfaces as
31
+ # the API's `render_timeout` — which tells you what happened — rather than as a
32
+ # client-side socket timeout, which tells you nothing.
33
+ DEFAULT_TIMEOUT = 130.0
34
+
35
+ JsonDict = dict[str, Any]
36
+
37
+
38
+ class PDFCraft:
39
+ """A PDFCraft API client.
40
+
41
+ >>> pdf = PDFCraft("sk_live_...").render(html="<h1>hello</h1>")
42
+ >>> pdf[:4]
43
+ b'%PDF'
44
+ """
45
+
46
+ def __init__(
47
+ self,
48
+ api_key: str,
49
+ *,
50
+ base_url: str = DEFAULT_BASE_URL,
51
+ max_retries: int = DEFAULT_MAX_RETRIES,
52
+ timeout: float = DEFAULT_TIMEOUT,
53
+ opener: urllib.request.OpenerDirector | None = None,
54
+ ) -> None:
55
+ if not api_key:
56
+ raise PDFCraftError("invalid_api_key", "An API key is required.", 401)
57
+ self._api_key = api_key
58
+ self._base_url = base_url.rstrip("/")
59
+ self._max_retries = max_retries
60
+ self._timeout = timeout
61
+ # Injectable so the tests can run without a network, and so anyone
62
+ # behind a corporate proxy can hand in their own opener.
63
+ self._opener = opener or urllib.request.build_opener()
64
+
65
+ # ── rendering ─────────────────────────────────────────────────────────
66
+
67
+ def render(self, *, idempotency_key: str | None = None, **input: Any) -> bytes:
68
+ """Render HTML or a URL and return the PDF bytes.
69
+
70
+ The method decides the output mode, so passing ``output`` yourself
71
+ could only contradict it — it is rejected rather than ignored, because
72
+ ``render(output="url")`` quietly returning JSON is the kind of surprise
73
+ that costs an afternoon.
74
+ """
75
+ body = _with_output(input, "binary")
76
+ return self._request("POST", "/v1/render", body, idempotency_key).read()
77
+
78
+ def render_to_url(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
79
+ """Store the PDF and return a signed link instead of the bytes."""
80
+ body = _with_output(input, "url")
81
+ return _json(self._request("POST", "/v1/render", body, idempotency_key))
82
+
83
+ def render_async(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
84
+ """Queue a render; the webhook fires when it settles."""
85
+ return _json(self._request("POST", "/v1/render/async", dict(input), idempotency_key))
86
+
87
+ def get_render(self, render_id: str) -> JsonDict:
88
+ return _json(self._request("GET", f"/v1/renders/{_quote(render_id)}"))
89
+
90
+ def usage(self) -> JsonDict:
91
+ """Current period usage and quota."""
92
+ return _json(self._request("GET", "/v1/usage"))
93
+
94
+ # ── extraction ────────────────────────────────────────────────────────
95
+
96
+ def extract(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
97
+ """A PDF in, its tables and labelled fields out, each with a bounding box.
98
+
99
+ ``file`` takes base64 PDF bytes or an https URL to one; pass ``url`` or
100
+ ``html`` instead and the page is rendered first, then extracted — one
101
+ call, one charge. Billed per page read, so ``options={"pages": ...}``
102
+ narrows the bill as well as the work.
103
+
104
+ There is no OCR. A scan has no text layer and comes back as
105
+ ``extraction_failed``, unbilled. Nothing is guessed by a model, so the
106
+ same document always produces the same answer.
107
+ """
108
+ body = _with_output(input, "inline")
109
+ return _json(self._request("POST", "/v1/extract", body, idempotency_key))
110
+
111
+ def extract_to_url(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
112
+ body = _with_output(input, "url")
113
+ return _json(self._request("POST", "/v1/extract", body, idempotency_key))
114
+
115
+ def extract_async(self, *, idempotency_key: str | None = None, **input: Any) -> JsonDict:
116
+ return _json(self._request("POST", "/v1/extract/async", dict(input), idempotency_key))
117
+
118
+ def get_extraction(self, extraction_id: str) -> JsonDict:
119
+ """Poll one extraction.
120
+
121
+ An extraction id is not a render id: ``get_render`` will 404 on one and
122
+ this will 404 on a render id, deliberately.
123
+ """
124
+ return _json(self._request("GET", f"/v1/extractions/{_quote(extraction_id)}"))
125
+
126
+ def extract_pdf(
127
+ self,
128
+ pdf: bytes,
129
+ *,
130
+ idempotency_key: str | None = None,
131
+ **input: Any,
132
+ ) -> JsonDict:
133
+ """Convenience for the common case: hand it PDF bytes, it does the base64.
134
+
135
+ Kept out of ``extract`` itself so that call stays a plain dict you can
136
+ log, diff or replay.
137
+ """
138
+ return self.extract(
139
+ file=base64.b64encode(pdf).decode("ascii"),
140
+ idempotency_key=idempotency_key,
141
+ **input,
142
+ )
143
+
144
+ # ── accessibility ─────────────────────────────────────────────────────
145
+
146
+ def scan(
147
+ self,
148
+ *,
149
+ domain: str | None = None,
150
+ sitemap: str | None = None,
151
+ urls: Sequence[str] | None = None,
152
+ **options: Any,
153
+ ) -> JsonDict:
154
+ """Start an accessibility scan. Returns at once with an id to poll.
155
+
156
+ Exactly one source. A scan of a thousand documents at one request per
157
+ second per host has a floor measured in minutes, so there is nothing to
158
+ return but an id and somewhere to look.
159
+
160
+ scan = client.scan(domain="example.gov", max_documents=500)
161
+ while scan["status"] not in ("succeeded", "failed"):
162
+ time.sleep(10)
163
+ scan = client.get_scan(scan["id"])
164
+
165
+ ``max_documents`` is clamped to your plan rather than refused, and the
166
+ gap between what was found and what was checked is reported back as
167
+ ``discovered`` minus ``checked``.
168
+ """
169
+ source = {
170
+ key: value
171
+ for key, value in (("domain", domain), ("sitemap", sitemap), ("urls", list(urls) if urls else None))
172
+ if value is not None
173
+ }
174
+ if len(source) != 1:
175
+ raise PDFCraftError(
176
+ "invalid_request",
177
+ "exactly one of domain, sitemap or urls is required",
178
+ 400,
179
+ )
180
+ body: JsonDict = {"source": source}
181
+ if options:
182
+ body["options"] = dict(options)
183
+ return _json(self._request("POST", "/v1/a11y/scan", body))
184
+
185
+ def get_scan(self, scan_id: str) -> JsonDict:
186
+ """Poll one scan.
187
+
188
+ Returns progress while it runs and the full result — every document,
189
+ ranked, with findings and cost — once it succeeds. ``report_url`` is a
190
+ share token: anyone with it can read the report, no account needed.
191
+ """
192
+ return _json(self._request("GET", f"/v1/a11y/scans/{_quote(scan_id)}"))
193
+
194
+ # ── transport ─────────────────────────────────────────────────────────
195
+
196
+ def _request(
197
+ self,
198
+ method: str,
199
+ path: str,
200
+ body: Mapping[str, Any] | None = None,
201
+ idempotency_key: str | None = None,
202
+ ) -> Any:
203
+ payload = json.dumps(body).encode("utf-8") if body is not None else None
204
+ headers = {
205
+ "authorization": f"Bearer {self._api_key}",
206
+ "user-agent": f"pdfcraft-sdk-python/{__version__}",
207
+ }
208
+ if payload is not None:
209
+ headers["content-type"] = "application/json"
210
+ if idempotency_key:
211
+ headers["idempotency-key"] = idempotency_key
212
+
213
+ last: PDFCraftError | None = None
214
+ for attempt in range(self._max_retries + 1):
215
+ request = urllib.request.Request(
216
+ f"{self._base_url}{path}", data=payload, headers=headers, method=method
217
+ )
218
+ try:
219
+ return self._opener.open(request, timeout=self._timeout)
220
+ except urllib.error.HTTPError as response:
221
+ last = to_error(response.code, response.read())
222
+ if not last.retryable or attempt == self._max_retries:
223
+ raise last from None
224
+ time.sleep(_retry_after(response) or _backoff(attempt))
225
+ except (urllib.error.URLError, TimeoutError, OSError) as cause:
226
+ last = PDFCraftError("network_error", f"Could not reach PDFCraft: {cause}", 0)
227
+ if attempt == self._max_retries:
228
+ raise last from None
229
+ time.sleep(_backoff(attempt))
230
+
231
+ raise last or PDFCraftError("internal_error", "Request failed.", 500)
232
+
233
+
234
+ def _with_output(input: Mapping[str, Any], mode: str) -> JsonDict:
235
+ if "output" in input:
236
+ raise PDFCraftError(
237
+ "invalid_request",
238
+ "Do not pass `output`; the method you call decides it.",
239
+ 400,
240
+ )
241
+ return {**input, "output": mode}
242
+
243
+
244
+ def _json(response: Any) -> JsonDict:
245
+ return json.loads(response.read().decode("utf-8"))
246
+
247
+
248
+ def _quote(value: str) -> str:
249
+ from urllib.parse import quote
250
+
251
+ return quote(value, safe="")
252
+
253
+
254
+ def _backoff(attempt: int) -> float:
255
+ """Exponential with jitter, matching the TypeScript SDK's curve exactly."""
256
+ return 0.5 * (2**attempt) * (0.75 + random.random() * 0.5)
257
+
258
+
259
+ def _retry_after(response: Any) -> float | None:
260
+ header = response.headers.get("retry-after") if hasattr(response, "headers") else None
261
+ if not header:
262
+ return None
263
+ try:
264
+ return max(0.0, float(header))
265
+ except (TypeError, ValueError):
266
+ return None
pdfcraft/_contract.py ADDED
@@ -0,0 +1,72 @@
1
+ # GENERATED FILE — do not edit.
2
+ # Written by packages/contract/scripts/sync-polyglot.mjs from the same
3
+ # definitions the PDFCraft API validates requests against.
4
+ # Editing it by hand will be silently overwritten on the next sync.
5
+
6
+ CONTRACT_HASH = "f5af303a2509e0d11d0c6efdc069e9acb80fddcd43a337a4789fa8f538f2588a"
7
+
8
+ #: Stable machine-readable codes. Branch on these, never on the message.
9
+ ERROR_CODES: tuple[str, ...] = (
10
+ "invalid_request",
11
+ "invalid_api_key",
12
+ "payment_required",
13
+ "not_found",
14
+ "render_timeout",
15
+ "render_failed",
16
+ "rate_limited",
17
+ "quota_exceeded",
18
+ "demo_busy",
19
+ "unsupported_file",
20
+ "extraction_failed",
21
+ "internal_error",
22
+ )
23
+
24
+ PAGE_FORMATS: tuple[str, ...] = (
25
+ "A4",
26
+ "A3",
27
+ "A5",
28
+ "Letter",
29
+ "Legal",
30
+ "Tabloid",
31
+ )
32
+
33
+ EMULATE_MEDIA: tuple[str, ...] = (
34
+ "print",
35
+ "screen",
36
+ )
37
+
38
+ OUTPUT_MODES: tuple[str, ...] = (
39
+ "binary",
40
+ "url",
41
+ )
42
+
43
+ EXTRACT_OUTPUT_MODES: tuple[str, ...] = (
44
+ "inline",
45
+ "url",
46
+ )
47
+
48
+ MAX_TIMEOUT_MS = 120000
49
+ DEFAULT_TIMEOUT_MS = 30000
50
+
51
+ #: Accessibility severity, worst first. Ordered, not merely enumerated.
52
+ A11Y_SEVERITIES: tuple[str, ...] = (
53
+ "blocker",
54
+ "major",
55
+ "minor",
56
+ "pass",
57
+ )
58
+
59
+ #: Scan lifecycle. Poll until the status is terminal.
60
+ SCAN_STATUSES: tuple[str, ...] = (
61
+ "queued",
62
+ "crawling",
63
+ "checking",
64
+ "succeeded",
65
+ "failed",
66
+ )
67
+
68
+ #: Which layer found a finding. "geometric" is the half a conformance validator cannot do.
69
+ FINDING_LAYERS: tuple[str, ...] = (
70
+ "machine",
71
+ "geometric",
72
+ )
pdfcraft/_errors.py ADDED
@@ -0,0 +1,66 @@
1
+ """Every failure from the API arrives as one exception type. Branch on ``.code``.
2
+
3
+ Mirrors ``src/errors.ts`` in the TypeScript SDK deliberately, down to which
4
+ statuses count as retryable, so a bug report against one client can be read by
5
+ someone holding the other.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from typing import Any
12
+
13
+
14
+ class PDFCraftError(Exception):
15
+ """A PDFCraft API failure.
16
+
17
+ ``code`` is the stable machine-readable string from the API's error
18
+ envelope — ``quota_exceeded``, ``render_timeout`` and so on — or
19
+ ``network_error`` when the request never reached us. Branch on it rather
20
+ than on the message, which is written for humans and may be reworded.
21
+ """
22
+
23
+ __slots__ = ("code", "status", "docs_url", "retryable")
24
+
25
+ def __init__(
26
+ self,
27
+ code: str,
28
+ message: str,
29
+ status: int,
30
+ docs_url: str | None = None,
31
+ ) -> None:
32
+ super().__init__(message)
33
+ self.code = code
34
+ self.status = status
35
+ self.docs_url = docs_url
36
+ # A 4xx other than 429 will fail identically however many times we ask.
37
+ self.retryable = code == "network_error" or status == 429 or status >= 500
38
+
39
+ def __repr__(self) -> str: # pragma: no cover - debugging aid
40
+ return f"PDFCraftError(code={self.code!r}, status={self.status})"
41
+
42
+
43
+ def to_error(status: int, body: bytes) -> PDFCraftError:
44
+ """Turn an error response into a typed exception.
45
+
46
+ Tolerates a non-JSON body on purpose: a proxy or load balancer in front of
47
+ the API returns an HTML error page, and a client that raises
48
+ ``JSONDecodeError`` there hides the status code that would have explained
49
+ the problem.
50
+ """
51
+ payload: Any = None
52
+ try:
53
+ payload = json.loads(body.decode("utf-8", "replace"))
54
+ except (ValueError, AttributeError):
55
+ payload = None
56
+
57
+ error = payload.get("error") if isinstance(payload, dict) else None
58
+ if not isinstance(error, dict):
59
+ error = {}
60
+
61
+ return PDFCraftError(
62
+ str(error.get("code") or "internal_error"),
63
+ str(error.get("message") or f"PDFCraft responded {status}"),
64
+ status,
65
+ error.get("docs_url") or None,
66
+ )
pdfcraft/_version.py ADDED
@@ -0,0 +1,6 @@
1
+ # src/pdfcraft/_version.py
2
+ #
3
+ # One place, read by the package metadata AND by the User-Agent header. Not
4
+ # read back out of the installed distribution: that fails when the package is
5
+ # run from a source checkout, which is exactly when someone is debugging.
6
+ __version__ = "1.4.0"
pdfcraft/py.typed ADDED
File without changes
@@ -0,0 +1,241 @@
1
+ Metadata-Version: 2.4
2
+ Name: pdfcraft-dev
3
+ Version: 1.4.0
4
+ Summary: HTML to PDF, and PDF to structured JSON, in one call. The official PDFCraft SDK.
5
+ Project-URL: Homepage, https://pdfcraft.dev
6
+ Project-URL: Documentation, https://pdfcraft.dev/sdk/python/
7
+ Project-URL: Source, https://github.com/igaurav-dev/pdfcraft-python
8
+ Project-URL: Issues, https://github.com/igaurav-dev/pdfcraft-python/issues
9
+ Author: PDFCraft
10
+ License: MIT License
11
+
12
+ Copyright (c) 2026 Gaurav Singh
13
+
14
+ Permission is hereby granted, free of charge, to any person obtaining a copy
15
+ of this software and associated documentation files (the "Software"), to deal
16
+ in the Software without restriction, including without limitation the rights
17
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
18
+ copies of the Software, and to permit persons to whom the Software is
19
+ furnished to do so, subject to the following conditions:
20
+
21
+ The above copyright notice and this permission notice shall be included in all
22
+ copies or substantial portions of the Software.
23
+
24
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
25
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
26
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
27
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
28
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
29
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
30
+ SOFTWARE.
31
+ License-File: LICENSE
32
+ Keywords: chromium,extract-tables-from-pdf,headless-chrome,html-to-pdf,html-to-pdf-api,invoice,pdf,pdf-accessibility,pdf-api,pdf-extraction,pdf-generation,pdf-table-extraction,pdf-to-json,url-to-pdf
33
+ Classifier: Development Status :: 4 - Beta
34
+ Classifier: Intended Audience :: Developers
35
+ Classifier: License :: OSI Approved :: MIT License
36
+ Classifier: Programming Language :: Python :: 3
37
+ Classifier: Programming Language :: Python :: 3 :: Only
38
+ Classifier: Topic :: Printing
39
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
40
+ Classifier: Typing :: Typed
41
+ Requires-Python: >=3.9
42
+ Description-Content-Type: text/markdown
43
+
44
+ # PDFCraft for Python
45
+
46
+ HTML to PDF, PDF to structured JSON, and accessibility triage for a whole document estate.
47
+ The official Python client for [PDFCraft](https://pdfcraft.dev).
48
+
49
+ **Zero dependencies.** The whole client is `urllib.request` plus a retry loop, so it installs
50
+ into a Lambda or a slim container without dragging a transitive tree behind it.
51
+
52
+ ```bash
53
+ pip install pdfcraft-dev
54
+ ```
55
+
56
+ Installs as `pdfcraft-dev`, imports as `pdfcraft` — the same split as
57
+ `python-dateutil`/`dateutil`. The plain name was taken on PyPI by an unrelated
58
+ project, and the npm package is `@pdfcraft-dev/pdf`, so the two registries at
59
+ least agree with each other.
60
+
61
+ ## Render
62
+
63
+ ```python
64
+ from pdfcraft import PDFCraft
65
+
66
+ client = PDFCraft("sk_live_...")
67
+
68
+ pdf = client.render(html="<h1>Invoice 1042</h1>")
69
+ open("invoice.pdf", "wb").write(pdf)
70
+ ```
71
+
72
+ A URL instead of HTML, with page options:
73
+
74
+ ```python
75
+ pdf = client.render(
76
+ url="https://example.com/report",
77
+ options={
78
+ "format": "A4",
79
+ "margin": {"top": "20mm", "bottom": "20mm"},
80
+ "printBackground": True,
81
+ "waitFor": {"selector": "#chart-ready", "networkIdle": True},
82
+ },
83
+ filename="report.pdf",
84
+ )
85
+ ```
86
+
87
+ Need a link rather than bytes — for an email, or a file too big to hold in memory:
88
+
89
+ ```python
90
+ result = client.render_to_url(html=invoice_html)
91
+ print(result["url"], result["expires_at"], result["pages"])
92
+ ```
93
+
94
+ ## Extract
95
+
96
+ A PDF in, its tables and labelled fields out, each with a bounding box:
97
+
98
+ ```python
99
+ data = client.extract_pdf(open("statement.pdf", "rb").read())
100
+
101
+ for table in data["tables"]:
102
+ print(table["header"], len(table["rows"]), table["confidence"])
103
+ ```
104
+
105
+ Ask for specific fields by name and type:
106
+
107
+ ```python
108
+ data = client.extract(
109
+ file=base64_pdf,
110
+ schema={"invoice_total": "currency", "due_date": "date", "po_number": "string"},
111
+ )
112
+ print(data["fields"]["invoice_total"]["value"])
113
+ ```
114
+
115
+ There is **no OCR**. A scanned document has no text layer, comes back as `extraction_failed`,
116
+ and is not billed. Nothing is guessed by a model, so the same document always produces the
117
+ same answer — which is the point if you are reconciling numbers.
118
+
119
+ Extraction is billed per page read, so `options={"pages": "1-3"}` narrows the bill as well as
120
+ the work.
121
+
122
+ ## Async
123
+
124
+ For documents slow enough that you would rather not hold the connection:
125
+
126
+ ```python
127
+ job = client.render_async(url="https://example.com/huge", webhookUrl="https://you/hook")
128
+ status = client.get_render(job["id"])
129
+ ```
130
+
131
+ `get_extraction` polls extractions. An extraction id is not a render id — each endpoint 404s
132
+ on the other's ids, deliberately.
133
+
134
+ ## Accessibility
135
+
136
+ Point it at a domain and it finds every PDF, checks each against PDF/UA and WCAG 2.1 AA, and
137
+ returns a report ranked by severity weighted by reach, with a remediation cost range.
138
+
139
+ ```python
140
+ import time
141
+
142
+ scan = client.scan(domain="example.gov", max_documents=500)
143
+ while scan["status"] not in ("succeeded", "failed"):
144
+ time.sleep(10)
145
+ scan = client.get_scan(scan["id"])
146
+
147
+ for doc in scan["documents"][:10]: # already ranked — this is the fix list
148
+ print(doc["severity"], doc["score"], doc["url"])
149
+ print(f" ${doc['cost_low_usd']:.0f}-${doc['cost_high_usd']:.0f} to remediate")
150
+ ```
151
+
152
+ Exactly one source: `domain=`, `sitemap=` or `urls=`. Passing none or two raises
153
+ `PDFCraftError` before anything is sent, because a round trip to be told you contradicted
154
+ yourself is a round trip wasted.
155
+
156
+ A scan runs for minutes — one request per second per host is a rule we do not break — so it
157
+ returns an id immediately and you poll. `max_documents` is **clamped to your plan rather than
158
+ refused**; `discovered` minus `checked` is what was found and not looked at, which is also the
159
+ upgrade prompt.
160
+
161
+ `report_url` on a finished scan is a share token. Anyone holding it can read the full HTML
162
+ report, and `GET /r/<token>/pdf` renders the same report to PDF through the render API. Treat
163
+ it as a credential, not an identifier.
164
+
165
+ Each finding carries `severity` (`blocker`, `major`, `minor`), the `wcag` criteria it breaks, a
166
+ `message` written for whoever approves the budget, and `technical_detail` for whoever does the
167
+ work. `occurrences` is volume, not severity — one check failing 1,535 times is one thing wrong,
168
+ fixed once, so never rank on it.
169
+
170
+ ```python
171
+ from pdfcraft import A11Y_SEVERITIES, FINDING_LAYERS, SCAN_STATUSES
172
+ ```
173
+
174
+ Those are generated from the same contract the API validates against, so comparing against them
175
+ beats comparing against a string you typed.
176
+
177
+ Accessibility is a **separate subscription** from rendering. An account can hold either, both or
178
+ neither, and the free tier is a real scan of 25 documents with full findings.
179
+
180
+ ## Errors
181
+
182
+ Every failure raises `PDFCraftError`. Branch on `.code`, which is stable; the message is
183
+ written for a human and may be reworded.
184
+
185
+ ```python
186
+ from pdfcraft import PDFCraft, PDFCraftError
187
+
188
+ try:
189
+ pdf = client.render(html=page)
190
+ except PDFCraftError as error:
191
+ if error.code == "quota_exceeded":
192
+ ... # out of renders this period
193
+ elif error.code == "render_failed":
194
+ ... # their HTML broke — billable, and worth logging
195
+ elif error.retryable:
196
+ ... # already retried; this is after the last attempt
197
+ else:
198
+ raise
199
+ ```
200
+
201
+ `error.docs_url` points at the page explaining that specific code.
202
+
203
+ ### What is and is not billed
204
+
205
+ A render is billable if Chromium actually ran. Successes and `render_failed` count;
206
+ `render_timeout`, `internal_error` and every 4xx that never reached the browser do not.
207
+
208
+ ## Retries
209
+
210
+ Retries happen automatically on `429` and `5xx` and on network failures — never on a `4xx`
211
+ other than `429`, because those fail identically however often you ask. `Retry-After` is
212
+ honoured when the API sends it, otherwise the backoff is exponential with jitter.
213
+
214
+ ```python
215
+ client = PDFCraft("sk_live_...", max_retries=0) # off
216
+ client = PDFCraft("sk_live_...", timeout=200.0) # seconds
217
+ client = PDFCraft("sk_live_...", base_url="https://gateway.internal")
218
+ ```
219
+
220
+ ## Typing
221
+
222
+ Ships `py.typed`. Responses are plain dicts rather than dataclasses: the API's response shape
223
+ grows, and a dict that gains a key is a non-event where a frozen dataclass is a crash. The
224
+ option and error enumerations you might want to validate against are exported:
225
+
226
+ ```python
227
+ from pdfcraft import ERROR_CODES, PAGE_FORMATS
228
+ ```
229
+
230
+ Those are generated from the same definitions the API validates requests against, so they
231
+ cannot drift from the server.
232
+
233
+ ## Links
234
+
235
+ - Docs — https://pdfcraft.dev/docs/
236
+ - Every error code, cause and fix — https://pdfcraft.dev/errors/
237
+ - Extraction guides on 27 real document shapes — https://pdfcraft.dev/guides/
238
+ - TypeScript SDK — `@pdfcraft-dev/pdf`
239
+ - Go SDK — `github.com/igaurav-dev/pdfcraft-go`
240
+
241
+ MIT licensed.
@@ -0,0 +1,10 @@
1
+ pdfcraft/__init__.py,sha256=TL628Pcxuggf2SVOirPN9uZue2KuB_U16aRhnMTGB0Q,746
2
+ pdfcraft/_client.py,sha256=9s0PFQ56rU3fESwv6bIkauTB_xLCpKj2odhBKAmJCcE,10767
3
+ pdfcraft/_contract.py,sha256=H8RNxvJqYKI2CWMMn7zibTpFzlVNGKb6p0NHX0yeX4g,1539
4
+ pdfcraft/_errors.py,sha256=6eDr5jDcF6tLR6i9SDU5k9qRKmhlVqLF1aeCxfoIn-8,2202
5
+ pdfcraft/_version.py,sha256=mQUmfuD6SzNbDotSl4VdGUGeiArADo4hejfW5nDZ_eE,279
6
+ pdfcraft/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ pdfcraft_dev-1.4.0.dist-info/METADATA,sha256=mscBMul08Gg_M_WAE3juFgFUjj4NEQ0Mqt_rdJC0rCQ,9093
8
+ pdfcraft_dev-1.4.0.dist-info/WHEEL,sha256=qtCwoSJWgHk21S1Kb4ihdzI2rlJ1ZKaIurTj_ngOhyQ,87
9
+ pdfcraft_dev-1.4.0.dist-info/licenses/LICENSE,sha256=VvFCyi2di0rvS6eYiaa6nEK_aybkXjxeQ8iJVuafvjk,1069
10
+ pdfcraft_dev-1.4.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.27.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Gaurav Singh
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.