gowkhtmltopdf 0.2.5__py3-none-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,134 @@
1
+ """gowkhtmltopdf: in-process Python bindings for the gowkhtmltopdf engine.
2
+
3
+ Two usage styles, both backed by a ctypes-loaded c-shared library:
4
+
5
+ from gowkhtmltopdf import Document, Page, Content
6
+
7
+ doc = Document(
8
+ pages=[Page(source=Content(html=b"<html><body><h1>Invoice</h1></body></html>"))],
9
+ page_size="A4",
10
+ )
11
+ pdf_bytes = doc.pdf()
12
+
13
+ Or the flat helper:
14
+
15
+ from gowkhtmltopdf import convert_html_to_pdf, PDFOptions
16
+
17
+ pdf_bytes = convert_html_to_pdf(
18
+ b"<html><body><h1>Invoice #42</h1></body></html>",
19
+ options=PDFOptions(page_size="A4"),
20
+ )
21
+
22
+ The shared library is located and loaded only when a conversion runs;
23
+ building model objects never touches it.
24
+ """
25
+
26
+ from .exceptions import (
27
+ ConversionError,
28
+ ConversionTimeoutError,
29
+ ErrEmptyContent,
30
+ ErrInvalidContent,
31
+ ErrInvalidOrientation,
32
+ ErrInvalidPDFProfile,
33
+ ErrInvalidPageSize,
34
+ ErrInvalidPDFVersion,
35
+ ErrMissingOutput,
36
+ ErrNoPageObjects,
37
+ GowkhtmltopdfError,
38
+ InternalEngineError,
39
+ InvalidArgumentError,
40
+ LoadDeniedError,
41
+ RenderError,
42
+ ResourceLimitError,
43
+ error_from_status,
44
+ sniff_sentinel,
45
+ )
46
+ from .document import (
47
+ Content,
48
+ Crop,
49
+ Document,
50
+ HeaderFooter,
51
+ ImageDocument,
52
+ ImageOptions,
53
+ Margin,
54
+ NetworkPolicy,
55
+ Page,
56
+ PDFOptions,
57
+ TOC,
58
+ compatible_network_policy,
59
+ restricted_network_policy,
60
+ )
61
+ from .api import (
62
+ convert_file_to_pdf,
63
+ convert_html_to_image,
64
+ convert_html_to_pdf,
65
+ convert_url_to_pdf,
66
+ )
67
+
68
+ __version__ = "0.2.5"
69
+
70
+ #: Upstream settings-surface identifier (api.go LibraryVersion), distinct
71
+ #: from the project release in __version__.
72
+ library_version = "0.12.7-dev"
73
+
74
+
75
+ def abi_version():
76
+ # type: () -> int
77
+ """Return the ABI revision of the loaded shared library (always 1 today).
78
+
79
+ Raises ImportError when the library is missing or built for another ABI.
80
+ """
81
+ from ._lib import abi_version as _abi_version
82
+
83
+ return _abi_version()
84
+
85
+
86
+ def library_version_string():
87
+ # type: () -> str
88
+ """Return the runtime version string reported by the shared library."""
89
+ from ._lib import library_version_string as _lvs
90
+
91
+ return _lvs()
92
+
93
+
94
+ __all__ = [
95
+ "GowkhtmltopdfError",
96
+ "ConversionError",
97
+ "InvalidArgumentError",
98
+ "LoadDeniedError",
99
+ "RenderError",
100
+ "ConversionTimeoutError",
101
+ "ResourceLimitError",
102
+ "InternalEngineError",
103
+ "ErrEmptyContent",
104
+ "ErrInvalidContent",
105
+ "ErrNoPageObjects",
106
+ "ErrInvalidPageSize",
107
+ "ErrInvalidOrientation",
108
+ "ErrInvalidPDFVersion",
109
+ "ErrInvalidPDFProfile",
110
+ "ErrMissingOutput",
111
+ "error_from_status",
112
+ "sniff_sentinel",
113
+ "Content",
114
+ "Page",
115
+ "Margin",
116
+ "HeaderFooter",
117
+ "TOC",
118
+ "Crop",
119
+ "NetworkPolicy",
120
+ "compatible_network_policy",
121
+ "restricted_network_policy",
122
+ "PDFOptions",
123
+ "ImageOptions",
124
+ "Document",
125
+ "ImageDocument",
126
+ "convert_html_to_pdf",
127
+ "convert_file_to_pdf",
128
+ "convert_url_to_pdf",
129
+ "convert_html_to_image",
130
+ "__version__",
131
+ "library_version",
132
+ "abi_version",
133
+ "library_version_string",
134
+ ]
gowkhtmltopdf/_lib.py ADDED
@@ -0,0 +1,338 @@
1
+ """ctypes loader for libgowkhtmltopdf and the frozen C ABI structs.
2
+
3
+ Search order for the shared library:
4
+
5
+ 1. ``GOWKHTMLTOPDF_LIBRARY_PATH`` environment variable (exact path).
6
+ 2. ``libgowkhtmltopdf.{so,dylib,dll}`` next to this package (wheel layout).
7
+ 3. ``<repo root>/dist/libgowkhtmltopdf.<ext>`` for in-tree builds.
8
+
9
+ The first existing candidate wins; when none exists ``find_library_path``
10
+ raises ``FileNotFoundError`` listing every path tried.
11
+
12
+ Memory ownership follows the committed header
13
+ ``bindings/c/include/gowkhtmltopdf.h``: output bytes and error strings are
14
+ allocated by the library and must be released through
15
+ ``gowkhtmltopdf_free`` / ``gowkhtmltopdf_free_string`` after Python copies
16
+ them. Input buffers are borrowed for the duration of a call only.
17
+
18
+ Engine calls are not documented as thread-affine, so every foreign call is
19
+ serialized through a module-wide lock. ctypes releases the GIL around each
20
+ CDLL call, which lets other Python threads progress while a long render
21
+ runs.
22
+ """
23
+
24
+ import ctypes
25
+ import os
26
+ import sys
27
+ import threading
28
+ from pathlib import Path
29
+
30
+ from .exceptions import error_from_status
31
+
32
+ #: ABI revision this binding is compiled against. Must match the header.
33
+ ABI_VERSION = 1
34
+
35
+ _STATUS_OK = 0
36
+
37
+ _LOAD_LOCK = threading.Lock()
38
+ _CALL_LOCK = threading.Lock()
39
+
40
+ _LOADED_LIBRARY = None # type: ctypes.CDLL
41
+
42
+
43
+ class GwkPdfOptions(ctypes.Structure):
44
+ """Mirror of GwkPdfOptions from include/gowkhtmltopdf.h.
45
+
46
+ Field order is pinned by the ABI contract; do not reorder or insert.
47
+ """
48
+
49
+ _fields_ = [
50
+ ("abi_version", ctypes.c_int32),
51
+ ("struct_size", ctypes.c_int32),
52
+ ("page_size", ctypes.c_char_p),
53
+ ("orientation", ctypes.c_char_p),
54
+ ("title", ctypes.c_char_p),
55
+ ("pdf_version", ctypes.c_char_p),
56
+ ("pdf_profile", ctypes.c_char_p),
57
+ ("base_url", ctypes.c_char_p),
58
+ ("allow", ctypes.POINTER(ctypes.c_char_p)),
59
+ ("allow_len", ctypes.c_size_t),
60
+ ("width_mm", ctypes.c_double),
61
+ ("height_mm", ctypes.c_double),
62
+ ("margin_top", ctypes.c_double),
63
+ ("margin_right", ctypes.c_double),
64
+ ("margin_bottom", ctypes.c_double),
65
+ ("margin_left", ctypes.c_double),
66
+ ("copies", ctypes.c_int32),
67
+ ("grayscale", ctypes.c_int32),
68
+ ("enable_local_file_access", ctypes.c_int32),
69
+ ("network_policy", ctypes.c_int32),
70
+ ("timeout_ms", ctypes.c_int32),
71
+ ]
72
+
73
+ @classmethod
74
+ def create(cls):
75
+ # type: () -> GwkPdfOptions
76
+ """Return a zeroed struct with the size gate fields filled."""
77
+ instance = cls()
78
+ instance.abi_version = ABI_VERSION
79
+ instance.struct_size = ctypes.sizeof(cls)
80
+ return instance
81
+
82
+
83
+ class GwkImageOptions(ctypes.Structure):
84
+ """Mirror of GwkImageOptions from include/gowkhtmltopdf.h."""
85
+
86
+ _fields_ = [
87
+ ("abi_version", ctypes.c_int32),
88
+ ("struct_size", ctypes.c_int32),
89
+ ("format", ctypes.c_char_p),
90
+ ("base_url", ctypes.c_char_p),
91
+ ("allow", ctypes.POINTER(ctypes.c_char_p)),
92
+ ("allow_len", ctypes.c_size_t),
93
+ ("width", ctypes.c_int32),
94
+ ("height", ctypes.c_int32),
95
+ ("quality", ctypes.c_int32),
96
+ ("smart_width", ctypes.c_int32),
97
+ ("transparent", ctypes.c_int32),
98
+ ("crop_left", ctypes.c_int32),
99
+ ("crop_top", ctypes.c_int32),
100
+ ("crop_width", ctypes.c_int32),
101
+ ("crop_height", ctypes.c_int32),
102
+ ("zoom", ctypes.c_double),
103
+ ("enable_local_file_access", ctypes.c_int32),
104
+ ("network_policy", ctypes.c_int32),
105
+ ("timeout_ms", ctypes.c_int32),
106
+ ]
107
+
108
+ @classmethod
109
+ def create(cls):
110
+ # type: () -> GwkImageOptions
111
+ """Return a zeroed struct with the size gate fields filled."""
112
+ instance = cls()
113
+ instance.abi_version = ABI_VERSION
114
+ instance.struct_size = ctypes.sizeof(cls)
115
+ return instance
116
+
117
+
118
+ def _library_filename():
119
+ # type: () -> str
120
+ if sys.platform == "darwin":
121
+ return "libgowkhtmltopdf.dylib"
122
+ if sys.platform == "win32":
123
+ return "libgowkhtmltopdf.dll"
124
+ return "libgowkhtmltopdf.so"
125
+
126
+
127
+ def candidate_paths():
128
+ # type: () -> list
129
+ """Return the loader's search candidates, highest priority first."""
130
+ paths = []
131
+ env_path = os.environ.get("GOWKHTMLTOPDF_LIBRARY_PATH")
132
+ if env_path:
133
+ paths.append(Path(env_path))
134
+ filename = _library_filename()
135
+ package_dir = Path(__file__).resolve().parent
136
+ paths.append(package_dir / filename)
137
+ try:
138
+ # _lib.py sits at <root>/bindings/python/src/gowkhtmltopdf/, so
139
+ # parents[4] is the repository root.
140
+ repo_root = Path(__file__).resolve().parents[4]
141
+ paths.append(repo_root / "dist" / filename)
142
+ except IndexError: # installed outside any repo-like tree
143
+ pass
144
+ return paths
145
+
146
+
147
+ def find_library_path():
148
+ # type: () -> Path
149
+ """Return the first existing shared-library candidate.
150
+
151
+ Raises:
152
+ FileNotFoundError: When no candidate exists on disk.
153
+ """
154
+ tried = []
155
+ for path in candidate_paths():
156
+ tried.append(str(path))
157
+ if path.is_file():
158
+ return path
159
+ raise FileNotFoundError(
160
+ "libgowkhtmltopdf not found; build it with"
161
+ " 'CGO_ENABLED=1 go build -buildmode=c-shared -o dist/{0} ./bindings/c'"
162
+ " or set GOWKHTMLTOPDF_LIBRARY_PATH. Tried: {1}".format(
163
+ _library_filename(), ", ".join(tried)
164
+ )
165
+ )
166
+
167
+
168
+ def _bind_prototypes(lib):
169
+ # type: (ctypes.CDLL) -> None
170
+ ubyte_pp = ctypes.POINTER(ctypes.POINTER(ctypes.c_ubyte))
171
+ size_p = ctypes.POINTER(ctypes.c_size_t)
172
+ char_pp = ctypes.POINTER(ctypes.c_char_p)
173
+
174
+ fn = lib.gowkhtmltopdf_html_to_pdf
175
+ fn.restype = ctypes.c_int
176
+ fn.argtypes = [
177
+ ctypes.c_char_p,
178
+ ctypes.c_size_t,
179
+ ctypes.POINTER(GwkPdfOptions),
180
+ ubyte_pp,
181
+ size_p,
182
+ char_pp,
183
+ ]
184
+
185
+ fn = lib.gowkhtmltopdf_html_to_image
186
+ fn.restype = ctypes.c_int
187
+ fn.argtypes = [
188
+ ctypes.c_char_p,
189
+ ctypes.c_size_t,
190
+ ctypes.POINTER(GwkImageOptions),
191
+ ubyte_pp,
192
+ size_p,
193
+ char_pp,
194
+ ]
195
+
196
+ fn = lib.gowkhtmltopdf_free
197
+ fn.restype = None
198
+ fn.argtypes = [ctypes.c_void_p]
199
+
200
+ fn = lib.gowkhtmltopdf_free_string
201
+ fn.restype = None
202
+ fn.argtypes = [ctypes.c_char_p]
203
+
204
+ fn = lib.gowkhtmltopdf_abi_version
205
+ fn.restype = ctypes.c_int32
206
+ fn.argtypes = []
207
+
208
+ # Declared c_void_p instead of c_char_p so the raw pointer survives;
209
+ # the header requires releasing it with gowkhtmltopdf_free_string.
210
+ fn = lib.gowkhtmltopdf_version
211
+ fn.restype = ctypes.c_void_p
212
+ fn.argtypes = []
213
+
214
+ fn = lib.gowkhtmltopdf_last_error_length
215
+ fn.restype = ctypes.c_int32
216
+ fn.argtypes = []
217
+
218
+ fn = lib.gowkhtmltopdf_last_error
219
+ fn.restype = ctypes.c_int32
220
+ fn.argtypes = [ctypes.c_char_p, ctypes.c_int32]
221
+
222
+
223
+ def load_library():
224
+ # type: () -> ctypes.CDLL
225
+ """Load, prototype-bind, and ABI-check the shared library once.
226
+
227
+ Raises:
228
+ ImportError: When the library is missing or reports a foreign ABI.
229
+ """
230
+ global _LOADED_LIBRARY
231
+ with _LOAD_LOCK:
232
+ if _LOADED_LIBRARY is not None:
233
+ return _LOADED_LIBRARY
234
+ path = find_library_path()
235
+ # CDLL uses RTLD_LOCAL by default, keeping Go runtime symbols out
236
+ # of the global namespace.
237
+ lib = ctypes.CDLL(str(path))
238
+ _bind_prototypes(lib)
239
+ reported = int(lib.gowkhtmltopdf_abi_version())
240
+ if reported != ABI_VERSION:
241
+ raise ImportError(
242
+ "ABI mismatch: library {0}, binding expects {1}".format(
243
+ reported, ABI_VERSION
244
+ )
245
+ )
246
+ _LOADED_LIBRARY = lib
247
+ return _LOADED_LIBRARY
248
+
249
+
250
+ def abi_version():
251
+ # type: () -> int
252
+ """Return the ABI revision reported by the loaded library."""
253
+ return int(load_library().gowkhtmltopdf_abi_version())
254
+
255
+
256
+ def library_version_string():
257
+ # type: () -> str
258
+ """Return the runtime version string, freeing the library allocation."""
259
+ lib = load_library()
260
+ ptr = 0
261
+ try:
262
+ with _CALL_LOCK:
263
+ ptr = int(lib.gowkhtmltopdf_version() or 0)
264
+ if not ptr:
265
+ return ""
266
+ text = ctypes.cast(ptr, ctypes.c_char_p).value
267
+ return (text or b"").decode("utf-8", "replace")
268
+ finally:
269
+ if ptr:
270
+ lib.gowkhtmltopdf_free_string(ctypes.cast(ptr, ctypes.c_char_p))
271
+
272
+
273
+ def _take_error_message(lib, err_ptr, status):
274
+ # type: (ctypes.CDLL, ctypes.POINTER(ctypes.c_char_p), int) -> str
275
+ """Copy and free the out_err string, falling back to the last-error slot."""
276
+ text = b""
277
+ if err_ptr:
278
+ text = err_ptr.value or b""
279
+ lib.gowkhtmltopdf_free_string(err_ptr)
280
+ if not text:
281
+ length = int(lib.gowkhtmltopdf_last_error_length())
282
+ if length > 0:
283
+ buf = ctypes.create_string_buffer(length + 1)
284
+ lib.gowkhtmltopdf_last_error(buf, length + 1)
285
+ text = buf.value
286
+ message = text.decode("utf-8", "replace")
287
+ return message or "conversion failed with status {0}".format(status)
288
+
289
+
290
+ def convert_html_to_pdf(html, opts=None):
291
+ # type: (bytes, GwkPdfOptions) -> bytes
292
+ """Run one PDF conversion and return owned bytes.
293
+
294
+ Raises the mapped ConversionError subclass on any non-zero status.
295
+ """
296
+ if not isinstance(html, (bytes, bytearray)):
297
+ raise TypeError("html must be bytes")
298
+ html = bytes(html)
299
+ lib = load_library()
300
+ out_data = ctypes.POINTER(ctypes.c_ubyte)()
301
+ out_len = ctypes.c_size_t(0)
302
+ out_err = ctypes.c_char_p()
303
+ with _CALL_LOCK:
304
+ opts_ptr = ctypes.byref(opts) if opts is not None else None
305
+ status = lib.gowkhtmltopdf_html_to_pdf(
306
+ html, len(html), opts_ptr, ctypes.byref(out_data),
307
+ ctypes.byref(out_len), ctypes.byref(out_err),
308
+ )
309
+ if status != _STATUS_OK:
310
+ raise error_from_status(status, _take_error_message(lib, out_err, status))
311
+ try:
312
+ return ctypes.string_at(out_data, out_len.value)
313
+ finally:
314
+ lib.gowkhtmltopdf_free(ctypes.cast(out_data, ctypes.c_void_p))
315
+
316
+
317
+ def convert_html_to_image(html, opts=None):
318
+ # type: (bytes, GwkImageOptions) -> bytes
319
+ """Run one image conversion and return owned bytes."""
320
+ if not isinstance(html, (bytes, bytearray)):
321
+ raise TypeError("html must be bytes")
322
+ html = bytes(html)
323
+ lib = load_library()
324
+ out_data = ctypes.POINTER(ctypes.c_ubyte)()
325
+ out_len = ctypes.c_size_t(0)
326
+ out_err = ctypes.c_char_p()
327
+ with _CALL_LOCK:
328
+ opts_ptr = ctypes.byref(opts) if opts is not None else None
329
+ status = lib.gowkhtmltopdf_html_to_image(
330
+ html, len(html), opts_ptr, ctypes.byref(out_data),
331
+ ctypes.byref(out_len), ctypes.byref(out_err),
332
+ )
333
+ if status != _STATUS_OK:
334
+ raise error_from_status(status, _take_error_message(lib, out_err, status))
335
+ try:
336
+ return ctypes.string_at(out_data, out_len.value)
337
+ finally:
338
+ lib.gowkhtmltopdf_free(ctypes.cast(out_data, ctypes.c_void_p))
gowkhtmltopdf/api.py ADDED
@@ -0,0 +1,114 @@
1
+ """High-level one-call helpers mirroring the issue contract.
2
+
3
+ ``convert_html_to_pdf`` and ``convert_html_to_image`` are sugar over the
4
+ Document / ImageDocument models; they accept either a prebuilt options
5
+ object or per-call keyword overrides.
6
+ """
7
+
8
+ import dataclasses
9
+ from typing import Optional
10
+
11
+ from .document import (
12
+ Content,
13
+ Document,
14
+ ImageDocument,
15
+ ImageOptions,
16
+ Page,
17
+ PDFOptions,
18
+ )
19
+
20
+
21
+ def _as_html_bytes(html):
22
+ # type: (object) -> bytes
23
+ if isinstance(html, str):
24
+ return html.encode("utf-8")
25
+ if isinstance(html, (bytes, bytearray)):
26
+ return bytes(html)
27
+ raise TypeError("html must be str or bytes")
28
+
29
+
30
+ def convert_html_to_pdf(
31
+ html,
32
+ options=None, # type: Optional[PDFOptions]
33
+ **overrides # type: object
34
+ ):
35
+ # type: (...) -> bytes
36
+ """Convert inline HTML to PDF bytes in one call.
37
+
38
+ ``options`` may be omitted for engine defaults. Keyword overrides are
39
+ applied onto a copy of the options before serialization; ``timeout``
40
+ (seconds) is accepted as an override too. Note that local file
41
+ references inside ``html`` resolve relative to the current working
42
+ directory because the content becomes an inline document.
43
+ """
44
+ timeout = overrides.pop("timeout", None)
45
+ resolved = (options if options is not None else PDFOptions()).update(
46
+ **overrides
47
+ )
48
+ content = Content.from_html(_as_html_bytes(html))
49
+ kwargs = resolved.to_kwargs()
50
+ option_base = kwargs.pop("base_url", None)
51
+ if option_base:
52
+ content.base = option_base
53
+ doc = Document(pages=[Page(source=content)], **kwargs)
54
+ return doc.pdf(timeout=timeout)
55
+
56
+
57
+ def convert_file_to_pdf(source, out_path=None, **options):
58
+ # type: (str, Optional[str], object) -> bytes
59
+ """Read an HTML file, convert it, and optionally write a PDF file.
60
+
61
+ The top-level file is read by this helper; local resources it
62
+ references (linked CSS, images) resolve relative to the source file's
63
+ directory. ``allow_local_files`` defaults to True here unless
64
+ overridden. Extra keyword arguments go to the Document constructor.
65
+ """
66
+ from pathlib import Path
67
+
68
+ timeout = options.pop("timeout", None)
69
+ base_url = options.pop("base_url", None)
70
+ options.setdefault("allow_local_files", True)
71
+ src_path = Path(source).resolve()
72
+ if base_url is None:
73
+ base_url = src_path.parent.as_uri() + "/"
74
+ with open(src_path, "rb") as handle:
75
+ data = handle.read()
76
+ content = Content.from_html(data, base=base_url)
77
+ doc = Document(pages=[Page(source=content)], **options)
78
+ pdf_bytes = doc.pdf(timeout=timeout)
79
+ if out_path is not None:
80
+ with open(out_path, "wb") as sink:
81
+ sink.write(pdf_bytes)
82
+ return pdf_bytes
83
+
84
+
85
+ def convert_url_to_pdf(url, **options):
86
+ # type: (str, object) -> bytes
87
+ """Not implemented: URL sources need the handle-based ABI.
88
+
89
+ Use the gowkhtmltopdf CLI for URL input today.
90
+ """
91
+ raise NotImplementedError(
92
+ "URL source lands with the handle-based ABI;"
93
+ " use the gowkhtmltopdf CLI for URL input today"
94
+ )
95
+
96
+
97
+ def convert_html_to_image(
98
+ html,
99
+ options=None, # type: Optional[ImageOptions]
100
+ **overrides # type: object
101
+ ):
102
+ # type: (...) -> bytes
103
+ """Convert inline HTML to an encoded image (PNG by default)."""
104
+ timeout = overrides.pop("timeout", None)
105
+ resolved = (options if options is not None else ImageOptions()).update(
106
+ **overrides
107
+ )
108
+ content = Content.from_html(_as_html_bytes(html))
109
+ image_doc = ImageDocument(source=content, **{
110
+ field.name: getattr(resolved, field.name)
111
+ for field in dataclasses.fields(resolved)
112
+ if field.name != "timeout_ms"
113
+ })
114
+ return image_doc.image(timeout=timeout)