pdfform 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pdfform/__init__.py ADDED
@@ -0,0 +1,63 @@
1
+ """Derive a JSON Schema from a PDF form, and fill the form from JSON.
2
+
3
+ >>> from pdfform import extract_form, build_schema, fill_form
4
+ >>> info = extract_form("form.pdf")
5
+ >>> schema = build_schema(info)
6
+ >>> fill_form("form.pdf", {"Name": "Doe", "Agreed": True}, "filled.pdf")
7
+
8
+ The command line equivalent is ``pdfform schema`` and ``pdfform fill``.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from pdfform.extract import extract_form, extract_form_and_objects, get_acroform, open_pdf
14
+ from pdfform.fill import fill_form, flatten_values, strip_xfa_layer
15
+ from pdfform.flatten import flatten_widgets
16
+ from pdfform.model import (
17
+ DynamicXfaError,
18
+ FieldKind,
19
+ FieldValueError,
20
+ FormField,
21
+ FormInfo,
22
+ PdfFormError,
23
+ UnknownFieldError,
24
+ Widget,
25
+ XfaKind,
26
+ )
27
+ from pdfform.schema import build_schema, current_values, field_schema, validate_values
28
+ from pdfform.xfa import detect_xfa, strip_xfa, xfa_packets
29
+
30
+ try: # pragma: no cover - only missing when running from a source tree
31
+ from importlib.metadata import version
32
+
33
+ __version__ = version("pdfform")
34
+ except Exception: # pragma: no cover
35
+ __version__ = "0.2.0"
36
+
37
+ __all__ = [
38
+ "DynamicXfaError",
39
+ "FieldKind",
40
+ "FieldValueError",
41
+ "FormField",
42
+ "FormInfo",
43
+ "PdfFormError",
44
+ "UnknownFieldError",
45
+ "Widget",
46
+ "XfaKind",
47
+ "build_schema",
48
+ "current_values",
49
+ "detect_xfa",
50
+ "extract_form",
51
+ "extract_form_and_objects",
52
+ "field_schema",
53
+ "fill_form",
54
+ "flatten_values",
55
+ "flatten_widgets",
56
+ "get_acroform",
57
+ "open_pdf",
58
+ "strip_xfa",
59
+ "strip_xfa_layer",
60
+ "validate_values",
61
+ "xfa_packets",
62
+ "__version__",
63
+ ]
pdfform/cli.py ADDED
@@ -0,0 +1,376 @@
1
+ """Command line interface for :mod:`pdfform`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import logging
7
+ import sys
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ import click
12
+
13
+ from pdfform.extract import extract_form, get_acroform, open_pdf
14
+ from pdfform.fill import fill_form, flatten_values, strip_xfa_layer
15
+ from pdfform.model import OFF_STATE, FieldKind, FormInfo, PdfFormError, XfaKind
16
+ from pdfform.schema import build_schema, current_values, validate_values
17
+ from pdfform.xfa import xfa_packets
18
+
19
+ logger = logging.getLogger("pdfform")
20
+
21
+ PDF_ARG = click.Path(exists=True, dir_okay=False, readable=True, path_type=Path)
22
+ OUT_OPT = click.Path(dir_okay=False, writable=True, path_type=Path)
23
+
24
+
25
+ def _configure_logging(verbose: int) -> None:
26
+ level = logging.WARNING - min(verbose, 2) * 10
27
+ logging.basicConfig(level=level, format="%(levelname)s: %(message)s")
28
+
29
+
30
+ def _emit(payload: Any, output: Path | None, indent: int) -> None:
31
+ text = json.dumps(payload, indent=indent or None, ensure_ascii=False, sort_keys=False)
32
+ if output is None:
33
+ click.echo(text)
34
+ else:
35
+ output.write_text(text + "\n", encoding="utf-8")
36
+ click.echo(f"Wrote {output}", err=True)
37
+
38
+
39
+ def _warn_xfa(info: FormInfo) -> None:
40
+ if info.xfa is XfaKind.HYBRID:
41
+ click.echo(
42
+ "warning: this form carries a static XFA layer. Acrobat may prefer the XFA "
43
+ "data over the values written here; `pdfform fill` drops that layer by default.",
44
+ err=True,
45
+ )
46
+
47
+
48
+ class _FormError(click.ClickException):
49
+ """A library error, reported without a traceback."""
50
+
51
+ exit_code = 2
52
+
53
+
54
+ class _Group(click.Group):
55
+ """Turns :class:`PdfFormError` into a clean message instead of a traceback."""
56
+
57
+ def invoke(self, ctx: click.Context) -> Any:
58
+ try:
59
+ return super().invoke(ctx)
60
+ except PdfFormError as exc:
61
+ raise _FormError(str(exc)) from exc
62
+
63
+
64
+ @click.group(cls=_Group, context_settings={"help_option_names": ["-h", "--help"]})
65
+ @click.version_option(package_name="pdfform")
66
+ @click.option("-v", "--verbose", count=True, help="Increase log verbosity. Repeatable.")
67
+ def main_cli(verbose: int) -> None:
68
+ """Inspect, describe and fill PDF forms.
69
+
70
+ \b
71
+ Typical session:
72
+ pdfform fields form.pdf # see what is in there
73
+ pdfform schema form.pdf -o schema.json # derive a JSON Schema
74
+ pdfform values form.pdf -o data.json # start from the current values
75
+ pdfform fill form.pdf -d data.json -o out.pdf
76
+ """
77
+ _configure_logging(verbose)
78
+
79
+
80
+ @main_cli.command()
81
+ @click.argument("pdf", type=PDF_ARG)
82
+ @click.option("-o", "--output", type=OUT_OPT, help="Write the schema here instead of to stdout.")
83
+ @click.option("--nested/--flat", default=False, help="Split dot-separated field names into nested objects.")
84
+ @click.option("--include-read-only", is_flag=True, help="Also describe read-only fields.")
85
+ @click.option("--include-signatures", is_flag=True, help="Also describe signature fields.")
86
+ @click.option(
87
+ "--infer-labels",
88
+ is_flag=True,
89
+ help="Guess a description for fields without a tooltip from the text next to them.",
90
+ )
91
+ @click.option("--indent", type=int, default=2, show_default=True, help="JSON indentation; 0 for one line.")
92
+ def schema(
93
+ pdf: Path,
94
+ output: Path | None,
95
+ nested: bool,
96
+ include_read_only: bool,
97
+ include_signatures: bool,
98
+ infer_labels: bool,
99
+ indent: int,
100
+ ) -> None:
101
+ """Derive a JSON Schema from the form fields of PDF."""
102
+ info = extract_form(pdf, infer_labels=infer_labels)
103
+ _warn_xfa(info)
104
+ document = build_schema(
105
+ info,
106
+ nested=nested,
107
+ include_read_only=include_read_only,
108
+ include_signatures=include_signatures,
109
+ )
110
+ _emit(document, output, indent)
111
+
112
+
113
+ @main_cli.command()
114
+ @click.argument("pdf", type=PDF_ARG)
115
+ @click.option("--json", "as_json", is_flag=True, help="Emit the inventory as JSON instead of a table.")
116
+ @click.option("--infer-labels", is_flag=True, help="Guess a label for each field from the text next to it.")
117
+ @click.option("--all", "show_all", is_flag=True, help="Include push buttons and read-only fields.")
118
+ def fields(pdf: Path, as_json: bool, infer_labels: bool, show_all: bool) -> None:
119
+ """List the form fields of PDF with their type, options and current value."""
120
+ info = extract_form(pdf, infer_labels=infer_labels)
121
+ _warn_xfa(info)
122
+ selected = [f for f in info.fields if show_all or (f.kind.holds_value and not f.read_only)]
123
+
124
+ if as_json:
125
+ payload = [
126
+ {
127
+ "name": f.name,
128
+ "kind": f.kind.value,
129
+ "pages": f.pages,
130
+ "value": f.value,
131
+ "options": f.options or f.on_states,
132
+ "optionLabels": f.option_labels,
133
+ "maxLength": f.max_length,
134
+ "required": f.required,
135
+ "readOnly": f.read_only,
136
+ "description": f.describe(),
137
+ }
138
+ for f in selected
139
+ ]
140
+ _emit(payload, None, 2)
141
+ return
142
+
143
+ if not selected:
144
+ click.echo("No fillable form fields found.", err=True)
145
+ if info.xfa is XfaKind.DYNAMIC:
146
+ click.echo("This is a dynamic XFA form; there is no static field list.", err=True)
147
+ return
148
+
149
+ name_width = min(max(len(f.name) for f in selected), 52)
150
+ click.echo(f"{'FIELD'.ljust(name_width)} {'KIND'.ljust(10)} PG DETAIL")
151
+ for field in selected:
152
+ detail_parts: list[str] = []
153
+ if field.kind is FieldKind.CHECKBOX:
154
+ detail_parts.append(f"on={field.on_states[0] if field.on_states else 'Yes'}")
155
+ options = field.options or field.on_states
156
+ if options:
157
+ detail_parts.append("[" + ", ".join(options[:6]) + ("…" if len(options) > 6 else "") + "]")
158
+ if field.max_length:
159
+ detail_parts.append(f"max {field.max_length}")
160
+ if field.required:
161
+ detail_parts.append("required")
162
+ if field.read_only:
163
+ detail_parts.append("read-only")
164
+ if field.value and field.value != OFF_STATE:
165
+ detail_parts.append(f"= {field.value!r}")
166
+ described = field.describe()
167
+ if described:
168
+ detail_parts.append(f"“{described}”")
169
+ pages = ",".join(str(p + 1) for p in field.pages) or "-"
170
+ click.echo(
171
+ f"{_clip(field.name, name_width).ljust(name_width)} {field.kind.value.ljust(10)} {pages.rjust(2)} "
172
+ + " ".join(detail_parts)
173
+ )
174
+ click.echo(f"\n{len(selected)} field(s) over {info.n_pages} page(s).", err=True)
175
+
176
+
177
+ def _clip(text: str, width: int) -> str:
178
+ return text if len(text) <= width else text[: width - 1] + "…"
179
+
180
+
181
+ @main_cli.command()
182
+ @click.argument("pdf", type=PDF_ARG)
183
+ @click.option("-o", "--output", type=OUT_OPT, help="Write the values here instead of to stdout.")
184
+ @click.option("--include-empty", is_flag=True, help="Also emit fields that are currently empty.")
185
+ @click.option("--include-read-only", is_flag=True, help="Also emit read-only fields.")
186
+ @click.option("--indent", type=int, default=2, show_default=True, help="JSON indentation; 0 for one line.")
187
+ def values(pdf: Path, output: Path | None, include_empty: bool, include_read_only: bool, indent: int) -> None:
188
+ """Dump the values currently stored in PDF as JSON.
189
+
190
+ The result is accepted by `pdfform fill`, so this is the quickest way to get
191
+ a template to edit.
192
+ """
193
+ info = extract_form(pdf)
194
+ payload = current_values(info, include_empty=include_empty, include_read_only=include_read_only)
195
+ _emit(payload, output, indent)
196
+
197
+
198
+ @main_cli.command()
199
+ @click.argument("pdf", type=PDF_ARG)
200
+ @click.option(
201
+ "-d",
202
+ "--data",
203
+ type=click.Path(dir_okay=False, allow_dash=True, path_type=Path),
204
+ help="JSON file with the field values. Use - to read from stdin.",
205
+ )
206
+ @click.option("--set", "assignments", multiple=True, metavar="NAME=VALUE", help="Set one field. Repeatable.")
207
+ @click.option("-o", "--output", type=OUT_OPT, required=True, help="Where to write the filled PDF.")
208
+ @click.option("--flatten", is_flag=True, help="Bake the values into the page and drop the form.")
209
+ @click.option(
210
+ "--strict/--no-strict",
211
+ default=True,
212
+ show_default=True,
213
+ help="Fail on unknown field names and values that do not fit their field.",
214
+ )
215
+ @click.option(
216
+ "--need-appearances/--no-need-appearances",
217
+ default=True,
218
+ show_default=True,
219
+ help="Ask viewers to regenerate the field appearances.",
220
+ )
221
+ @click.option(
222
+ "--strip-xfa/--keep-xfa",
223
+ "strip_xfa",
224
+ default=None,
225
+ help="Drop the XFA layer. The default drops it only for static XFA forms.",
226
+ )
227
+ @click.option("--validate", "check", is_flag=True, help="Validate the values against the derived schema first.")
228
+ def fill(
229
+ pdf: Path,
230
+ data: Path | None,
231
+ assignments: tuple[str, ...],
232
+ output: Path,
233
+ flatten: bool,
234
+ strict: bool,
235
+ need_appearances: bool,
236
+ strip_xfa: bool | None,
237
+ check: bool,
238
+ ) -> None:
239
+ """Fill the form in PDF and write the result to --output.
240
+
241
+ \b
242
+ Values come from a JSON file, from --set, or both; --set wins.
243
+ pdfform fill form.pdf --set Name=Doe --set Agreed=true -o out.pdf
244
+ pdfform fill form.pdf -d values.json -o out.pdf
245
+ cat values.json | pdfform fill form.pdf -d - -o out.pdf
246
+ """
247
+ payload: dict[str, Any] = _load_values(data) if data is not None else {}
248
+ for assignment in assignments:
249
+ name, separator, value = assignment.partition("=")
250
+ if not separator:
251
+ raise click.ClickException(f"--set expects NAME=VALUE, got {assignment!r}")
252
+ payload[name] = _parse_scalar(value)
253
+
254
+ if not payload:
255
+ raise click.ClickException("No values given. Use --data and/or --set.")
256
+
257
+ if check:
258
+ problems = validate_values(build_schema(extract_form(pdf)), flatten_values(payload))
259
+ if problems:
260
+ for problem in problems:
261
+ click.echo(problem, err=True)
262
+ raise click.ClickException(f"{len(problems)} value(s) do not match the schema; nothing was written.")
263
+
264
+ fill_form(
265
+ pdf,
266
+ payload,
267
+ output,
268
+ flatten=flatten,
269
+ need_appearances=need_appearances,
270
+ strict=strict,
271
+ strip_xfa=strip_xfa,
272
+ )
273
+ click.echo(f"Wrote {output}", err=True)
274
+
275
+
276
+ def _parse_scalar(text: str) -> Any:
277
+ """Interpret a --set value, keeping anything unrecognised as a string."""
278
+ lowered = text.strip().lower()
279
+ if lowered in ("true", "false"):
280
+ return lowered == "true"
281
+ if lowered == "null":
282
+ return None
283
+ return text
284
+
285
+
286
+ @main_cli.command()
287
+ @click.argument("pdf", type=PDF_ARG)
288
+ @click.option(
289
+ "-d",
290
+ "--data",
291
+ required=True,
292
+ type=click.Path(dir_okay=False, allow_dash=True, path_type=Path),
293
+ help="JSON file with the field values. Use - to read from stdin.",
294
+ )
295
+ def validate(pdf: Path, data: Path) -> None:
296
+ """Check a values file against the schema derived from PDF.
297
+
298
+ Reports unknown field names, values outside an enum, strings over
299
+ `maxLength` and missing required fields, then exits non-zero if anything is
300
+ wrong. Filling is more permissive than this on purpose, so a clean run here
301
+ is a stronger guarantee than a successful fill.
302
+ """
303
+ payload = _load_values(data)
304
+ schema = build_schema(extract_form(pdf))
305
+ problems = validate_values(schema, flatten_values(payload))
306
+ if not problems:
307
+ click.echo(f"{len(payload)} value(s) validate against the derived schema.", err=True)
308
+ return
309
+ for problem in problems:
310
+ click.echo(problem, err=True)
311
+ raise click.exceptions.Exit(1)
312
+
313
+
314
+ def _load_values(data: Path) -> dict[str, Any]:
315
+ raw = sys.stdin.read() if str(data) == "-" else data.read_text(encoding="utf-8")
316
+ try:
317
+ payload = json.loads(raw)
318
+ except json.JSONDecodeError as exc:
319
+ raise click.ClickException(f"Could not parse the values as JSON: {exc}") from exc
320
+ if not isinstance(payload, dict):
321
+ raise click.ClickException("The values file must contain a JSON object.")
322
+ return payload
323
+
324
+
325
+ @main_cli.command("strip-xfa")
326
+ @click.argument("pdf", type=PDF_ARG)
327
+ @click.option("-o", "--output", type=OUT_OPT, required=True, help="Where to write the result.")
328
+ def strip_xfa_command(pdf: Path, output: Path) -> None:
329
+ """Remove the XFA layer of PDF, leaving a plain AcroForm.
330
+
331
+ The equivalent of `pdftk in.pdf output out.pdf drop_xfa`. Useful for static
332
+ XFA forms, whose XFA data would otherwise override the AcroForm values.
333
+ """
334
+ info = extract_form(pdf)
335
+ if info.xfa is XfaKind.NONE:
336
+ click.echo("No XFA layer to remove; copying the document unchanged.", err=True)
337
+ elif info.xfa is XfaKind.DYNAMIC:
338
+ click.echo(
339
+ "warning: this is a dynamic XFA form. Removing the XFA layer leaves an empty "
340
+ "document, because the AcroForm side is only a stub.",
341
+ err=True,
342
+ )
343
+ strip_xfa_layer(pdf, output)
344
+ click.echo(f"Wrote {output}", err=True)
345
+
346
+
347
+ @main_cli.command()
348
+ @click.argument("pdf", type=PDF_ARG)
349
+ @click.option("--packet", help="Dump this XFA packet, for example template or datasets.")
350
+ @click.option("-o", "--output", type=OUT_OPT, help="Write the packet here instead of to stdout.")
351
+ def xfa(pdf: Path, packet: str | None, output: Path | None) -> None:
352
+ """Inspect the XFA layer of PDF.
353
+
354
+ Without --packet, lists the packets and their sizes. The `template` packet
355
+ holds the real form definition, `datasets` holds the values.
356
+ """
357
+ packets = xfa_packets(get_acroform(open_pdf(pdf)))
358
+ if not packets:
359
+ click.echo("No XFA layer in this document.", err=True)
360
+ return
361
+ if packet is None:
362
+ for name, blob in packets.items():
363
+ click.echo(f"{name:<12} {len(blob):>9} bytes")
364
+ return
365
+ if packet not in packets:
366
+ raise click.ClickException(f"No packet named {packet!r}. Available: " + ", ".join(packets))
367
+ blob = packets[packet]
368
+ if output is None:
369
+ click.echo(blob.decode("utf-8", errors="replace"))
370
+ else:
371
+ output.write_bytes(blob)
372
+ click.echo(f"Wrote {output}", err=True)
373
+
374
+
375
+ if __name__ == "__main__": # pragma: no cover
376
+ main_cli()