word2pdf 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
word2pdf/__init__.py ADDED
@@ -0,0 +1,18 @@
1
+ """Word2PDF - convert .docx to PDF in pure Python.
2
+
3
+ No Microsoft Word, no LibreOffice, no external binaries: the package reads
4
+ WordprocessingML directly, lays the document out itself and draws the result
5
+ with ReportLab, so it behaves the same on Windows, macOS and Linux.
6
+
7
+ from word2pdf import convert
8
+
9
+ convert('report.docx', 'report.pdf')
10
+ pdf_bytes = convert('report.docx')
11
+ convert('invoice.docx', 'invoice.pdf', context={'customer': 'Example Ltd'})
12
+ """
13
+ from .api import Converter, Result, convert, convert_many, convert_to_bytes
14
+ from .templates import DocxTemplate, render_docx
15
+
16
+ __version__ = '0.1.0'
17
+ __all__ = ['convert', 'convert_to_bytes', 'convert_many', 'Converter', 'Result',
18
+ 'DocxTemplate', 'render_docx', '__version__']
word2pdf/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from .cli import main
2
+
3
+ if __name__ == '__main__':
4
+ raise SystemExit(main())
word2pdf/api.py ADDED
@@ -0,0 +1,143 @@
1
+ """The public conversion API."""
2
+ from __future__ import annotations
3
+
4
+ import io
5
+ import os
6
+ from dataclasses import dataclass, field
7
+ from typing import List, Optional
8
+
9
+ from .document import parse_document
10
+ from .fonts import FontManager
11
+ from .layout import Flow, Layouter
12
+ from .oxml import Package
13
+ from .render import Renderer
14
+
15
+
16
+ @dataclass
17
+ class Result:
18
+ """What a conversion produced."""
19
+ pages: int = 0
20
+ warnings: List[str] = field(default_factory=list)
21
+ missing_fonts: List[str] = field(default_factory=list)
22
+ output: Optional[str] = None
23
+
24
+ def __bool__(self):
25
+ return self.pages > 0
26
+
27
+
28
+ class Converter:
29
+ """Reusable converter; keeps the font index warm between documents."""
30
+
31
+ def __init__(self, font_dirs=None, default_font='Calibri', font_cache=True,
32
+ **options):
33
+ self.options = dict(options)
34
+ self.options.setdefault('default_font', default_font)
35
+ self.options.setdefault('font_dirs', font_dirs)
36
+ self.options.setdefault('font_cache', font_cache)
37
+ self.fonts = FontManager(extra_dirs=font_dirs, use_cache=font_cache,
38
+ default_family=default_font)
39
+
40
+ def convert(self, source, target=None, context=None, **overrides):
41
+ """Convert a .docx to PDF.
42
+
43
+ *source* is a path, bytes or file-like object; *target* a path or
44
+ file-like object (omit it to get the PDF back as bytes). Pass
45
+ *context* to fill ``{{ placeholders }}`` in the document first.
46
+ """
47
+ options = dict(self.options)
48
+ options.update(overrides)
49
+ if context is not None:
50
+ from .templates import render_docx
51
+ source = render_docx(source, context, jinja=options.get('jinja', True))
52
+ package = source if isinstance(source, Package) else Package(source)
53
+ document = parse_document(package, options)
54
+ self.fonts.strategy = options.get('font_fallback', self.fonts.strategy)
55
+ self.fonts.set_document_default(_document_font(document, options))
56
+ layouter = Layouter(document, fonts=self.fonts, options=options)
57
+ pages = Flow(layouter, options).run()
58
+ renderer = Renderer(pages, document, options)
59
+
60
+ result = Result(pages=len(pages))
61
+ if target is None:
62
+ buffer = io.BytesIO()
63
+ renderer.write(buffer)
64
+ data = buffer.getvalue()
65
+ else:
66
+ if hasattr(target, 'write'):
67
+ renderer.write(target)
68
+ data = None
69
+ else:
70
+ target = os.fspath(target)
71
+ directory = os.path.dirname(os.path.abspath(target))
72
+ if directory:
73
+ os.makedirs(directory, exist_ok=True)
74
+ renderer.write(target)
75
+ result.output = target
76
+ data = None
77
+ result.warnings = list(document.warnings) + list(layouter.warnings) \
78
+ + list(renderer.warnings)
79
+ result.missing_fonts = sorted(self.fonts.missing_families)
80
+ self.last_result = result
81
+ if target is None:
82
+ return data
83
+ return result
84
+
85
+
86
+ def _document_font(document, options) -> str:
87
+ """The font Word would fall back to for this document."""
88
+ override = options.get('default_font_override')
89
+ if override:
90
+ return override
91
+ styles = document.styles
92
+ if styles is not None:
93
+ family = styles.doc_rpr.font_ascii
94
+ if not family and styles.doc_rpr.theme_ascii and styles.theme is not None:
95
+ family = styles.theme.font(styles.doc_rpr.theme_ascii)
96
+ normal = styles.get(styles.default_para_style)
97
+ if not family and normal is not None:
98
+ family = normal.rpr.font_ascii
99
+ if not family and styles.theme is not None:
100
+ family = styles.theme.minor_font
101
+ if family:
102
+ return family
103
+ return options.get('default_font', 'Calibri')
104
+
105
+
106
+ _default: Optional[Converter] = None
107
+
108
+
109
+ def _converter(**options) -> Converter:
110
+ global _default
111
+ keys = ('font_dirs', 'default_font', 'font_cache')
112
+ if _default is None or any(options.get(k) not in (None, _default.options.get(k))
113
+ for k in keys if k in options):
114
+ _default = Converter(font_dirs=options.get('font_dirs'),
115
+ default_font=options.get('default_font', 'Calibri'),
116
+ font_cache=options.get('font_cache', True))
117
+ return _default
118
+
119
+
120
+ def convert(source, target=None, context=None, **options):
121
+ """Convert a .docx file to PDF without Word or LibreOffice.
122
+
123
+ >>> convert('report.docx', 'report.pdf')
124
+ >>> pdf_bytes = convert('report.docx')
125
+ >>> convert('invoice.docx', 'invoice.pdf', context={'total': '12.00'})
126
+ """
127
+ converter = _converter(**options)
128
+ return converter.convert(source, target, context=context, **options)
129
+
130
+
131
+ def convert_to_bytes(source, context=None, **options) -> bytes:
132
+ return convert(source, None, context=context, **options)
133
+
134
+
135
+ def convert_many(sources, out_dir='.', **options):
136
+ """Convert several documents, reusing one warm font index."""
137
+ converter = _converter(**options)
138
+ results = []
139
+ for source in sources:
140
+ name = os.path.splitext(os.path.basename(os.fspath(source)))[0] + '.pdf'
141
+ results.append(converter.convert(source, os.path.join(out_dir, name),
142
+ **options))
143
+ return results
word2pdf/cli.py ADDED
@@ -0,0 +1,155 @@
1
+ """Command line interface: ``word2pdf in.docx [out.pdf]``."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import glob
6
+ import json
7
+ import os
8
+ import sys
9
+ import time
10
+
11
+ from . import __version__
12
+ from .api import Converter
13
+
14
+
15
+ def build_parser() -> argparse.ArgumentParser:
16
+ parser = argparse.ArgumentParser(
17
+ prog='word2pdf',
18
+ description='Convert .docx to PDF in pure Python - no Word, no '
19
+ 'LibreOffice, no external binaries.')
20
+ parser.add_argument('inputs', nargs='+', metavar='DOCX',
21
+ help='input .docx files (globs allowed)')
22
+ parser.add_argument('-o', '--output', metavar='PDF',
23
+ help='output file (single input) or directory')
24
+ parser.add_argument('-d', '--out-dir', metavar='DIR',
25
+ help='write the PDFs into this directory')
26
+ parser.add_argument('--data', metavar='JSON',
27
+ help='JSON file (or inline JSON) of template values')
28
+ parser.add_argument('--set', action='append', default=[], metavar='KEY=VALUE',
29
+ help='set one template value; repeatable')
30
+ parser.add_argument('--no-jinja', action='store_true',
31
+ help='use the built-in template engine even if Jinja2 '
32
+ 'is installed')
33
+ parser.add_argument('--font-dir', action='append', default=[], metavar='DIR',
34
+ help='extra directory to search for fonts; repeatable')
35
+ parser.add_argument('--default-font', default=None, metavar='NAME',
36
+ help='fallback font family (default: the document default)')
37
+ parser.add_argument('--font-fallback', choices=('word', 'closest'),
38
+ default='word',
39
+ help="how to replace missing fonts: 'word' mimics Word "
40
+ "(document default), 'closest' matches width/style")
41
+ parser.add_argument('--no-device-grid', action='store_true',
42
+ help='lay out in exact points instead of matching '
43
+ "Word's 600 dpi rounding")
44
+ parser.add_argument('--no-compress', action='store_true',
45
+ help='write an uncompressed PDF')
46
+ parser.add_argument('--image-quality', type=int, metavar='1-100',
47
+ help='re-encode photos as JPEG at this quality to '
48
+ 'shrink the PDF (default: keep them untouched)')
49
+ parser.add_argument('--image-max-dpi', type=int, metavar='DPI',
50
+ help='downsample images above this resolution')
51
+ parser.add_argument('--title', help='PDF title metadata')
52
+ parser.add_argument('--author', help='PDF author metadata')
53
+ parser.add_argument('-q', '--quiet', action='store_true')
54
+ parser.add_argument('-v', '--verbose', action='store_true',
55
+ help='report warnings and substituted fonts')
56
+ parser.add_argument('--version', action='version',
57
+ version='Word2PDF %s' % __version__)
58
+ return parser
59
+
60
+
61
+ def load_context(args) -> dict:
62
+ context: dict = {}
63
+ if args.data:
64
+ raw = args.data
65
+ if os.path.exists(raw):
66
+ with open(raw, 'r', encoding='utf-8') as fh:
67
+ context.update(json.load(fh))
68
+ else:
69
+ context.update(json.loads(raw))
70
+ for item in args.set:
71
+ key, _, value = item.partition('=')
72
+ if not key:
73
+ continue
74
+ try:
75
+ context[key.strip()] = json.loads(value)
76
+ except ValueError:
77
+ context[key.strip()] = value
78
+ return context
79
+
80
+
81
+ def expand(patterns):
82
+ files = []
83
+ for pattern in patterns:
84
+ matches = glob.glob(pattern) if any(ch in pattern for ch in '*?[') \
85
+ else [pattern]
86
+ if not matches:
87
+ print('no such file: %s' % pattern, file=sys.stderr)
88
+ files.extend(matches)
89
+ return files
90
+
91
+
92
+ def target_for(source, args, many: bool):
93
+ if args.out_dir:
94
+ base = os.path.splitext(os.path.basename(source))[0] + '.pdf'
95
+ return os.path.join(args.out_dir, base)
96
+ if args.output and not many:
97
+ if os.path.isdir(args.output):
98
+ base = os.path.splitext(os.path.basename(source))[0] + '.pdf'
99
+ return os.path.join(args.output, base)
100
+ return args.output
101
+ if args.output and many:
102
+ base = os.path.splitext(os.path.basename(source))[0] + '.pdf'
103
+ return os.path.join(args.output, base)
104
+ return os.path.splitext(source)[0] + '.pdf'
105
+
106
+
107
+ def main(argv=None) -> int:
108
+ args = build_parser().parse_args(argv)
109
+ sources = expand(args.inputs)
110
+ if not sources:
111
+ return 1
112
+ context = load_context(args) or None
113
+
114
+ options = {'jinja': not args.no_jinja,
115
+ 'font_fallback': args.font_fallback,
116
+ 'compress': not args.no_compress}
117
+ if args.no_device_grid:
118
+ options['device_grid'] = 0
119
+ if args.image_quality:
120
+ options['image_quality'] = args.image_quality
121
+ if args.image_max_dpi:
122
+ options['image_max_dpi'] = args.image_max_dpi
123
+ if args.title:
124
+ options['title'] = args.title
125
+ if args.author:
126
+ options['author'] = args.author
127
+
128
+ converter = Converter(font_dirs=args.font_dir or None,
129
+ default_font=args.default_font or 'Calibri',
130
+ **options)
131
+ failures = 0
132
+ for source in sources:
133
+ target = target_for(source, args, len(sources) > 1)
134
+ started = time.time()
135
+ try:
136
+ result = converter.convert(source, target, context=context)
137
+ except Exception as exc: # pragma: no cover
138
+ failures += 1
139
+ print('%s: %s' % (source, exc), file=sys.stderr)
140
+ continue
141
+ if not args.quiet:
142
+ print('%s -> %s (%d page%s, %.2fs)'
143
+ % (source, target, result.pages, '' if result.pages == 1 else 's',
144
+ time.time() - started))
145
+ if args.verbose:
146
+ for warning in result.warnings:
147
+ print(' warning: %s' % warning, file=sys.stderr)
148
+ if result.missing_fonts:
149
+ print(' substituted fonts: %s' % ', '.join(result.missing_fonts),
150
+ file=sys.stderr)
151
+ return 1 if failures else 0
152
+
153
+
154
+ if __name__ == '__main__':
155
+ raise SystemExit(main())
@@ -0,0 +1,24 @@
1
+ """The document: WordprocessingML in, a style-resolved model out.
2
+
3
+ `props` holds the typed property bags and their merge rules, `styles` and
4
+ `numbering` the two cascades that feed them, `model` the shapes the rest of the
5
+ engine works with, and `parser` does the reading. Nothing here knows about
6
+ pages or fonts: a Paragraph carries resolved properties, not positions.
7
+ """
8
+ from .model import (Block, BreakRun, Cell, Document, FieldRun, FloatingObject,
9
+ HeaderFooter, ImageRun, Inline, Link, NoteRun, NumberLabel,
10
+ Paragraph, Row, Section, ShapeRun, TabRun, Table, TextRun)
11
+ from .numbering import SYMBOL_BULLETS, Numbering, format_number
12
+ from .parser import parse_document, parse_field
13
+ from .props import (Border, CellProps, ParaProps, RowProps, RunProps, Shading,
14
+ TabStop, TableProps, merge_tabs)
15
+ from .styles import Style, StyleSheet, Theme, apply_tint_shade
16
+
17
+ __all__ = ['parse_document', 'parse_field', 'Document', 'Section', 'Block',
18
+ 'Paragraph', 'Table', 'Row', 'Cell', 'HeaderFooter',
19
+ 'FloatingObject', 'Inline', 'TextRun', 'ImageRun', 'BreakRun',
20
+ 'FieldRun', 'NoteRun', 'ShapeRun', 'TabRun', 'Link', 'NumberLabel',
21
+ 'RunProps', 'ParaProps', 'CellProps', 'RowProps', 'TableProps',
22
+ 'Border', 'Shading', 'TabStop', 'merge_tabs', 'StyleSheet', 'Style',
23
+ 'Theme', 'apply_tint_shade', 'Numbering', 'format_number',
24
+ 'SYMBOL_BULLETS']
@@ -0,0 +1,235 @@
1
+ """The intermediate document model produced by the parser.
2
+
3
+ Everything here is already style-resolved: the layout engine never has to look
4
+ at a style sheet again.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from dataclasses import dataclass, field
9
+ from typing import List, Optional
10
+
11
+ from .props import CellProps, ParaProps, RowProps, RunProps, TableProps
12
+
13
+
14
+ @dataclass
15
+ class Link:
16
+ url: Optional[str] = None
17
+ anchor: Optional[str] = None
18
+ tooltip: Optional[str] = None
19
+
20
+
21
+ @dataclass
22
+ class Inline:
23
+ props: RunProps = field(default_factory=RunProps)
24
+ link: Optional[Link] = None
25
+
26
+
27
+ @dataclass
28
+ class TextRun(Inline):
29
+ text: str = ''
30
+
31
+
32
+ @dataclass
33
+ class TabRun(Inline):
34
+ pass
35
+
36
+
37
+ @dataclass
38
+ class BreakRun(Inline):
39
+ kind: str = 'line' # line | page | column
40
+ clear: Optional[str] = None
41
+
42
+
43
+ @dataclass
44
+ class ImageRun(Inline):
45
+ data: Optional[bytes] = None
46
+ width: float = 0.0 # points
47
+ height: float = 0.0
48
+ alt: str = ''
49
+ rotation: float = 0.0
50
+ fmt: str = ''
51
+ crop: Optional[tuple] = None # (left, top, right, bottom) fractions
52
+
53
+
54
+ @dataclass
55
+ class ShapeRun(Inline):
56
+ """A drawing we cannot render faithfully; kept so its text is not lost."""
57
+ width: float = 0.0
58
+ height: float = 0.0
59
+ fill: Optional[str] = None
60
+ stroke: Optional[str] = None
61
+ blocks: list = field(default_factory=list)
62
+
63
+
64
+ @dataclass
65
+ class FieldRun(Inline):
66
+ kind: str = '' # PAGE | NUMPAGES | DATE | TIME | REF | ...
67
+ text: str = '' # cached result from the document
68
+ argument: str = ''
69
+ fmt: str = ''
70
+
71
+
72
+ @dataclass
73
+ class NoteRun(Inline):
74
+ """A footnote or endnote reference."""
75
+ note_id: str = ''
76
+ kind: str = 'footnote'
77
+ mark: str = ''
78
+
79
+
80
+ @dataclass
81
+ class NumberLabel:
82
+ text: str = ''
83
+ props: RunProps = field(default_factory=RunProps)
84
+ suffix: str = 'tab'
85
+ align: str = 'left'
86
+ font_family: Optional[str] = None
87
+
88
+
89
+ @dataclass
90
+ class FloatingObject:
91
+ """An anchored drawing: positioned absolutely, outside the text flow."""
92
+ image: Optional[ImageRun] = None
93
+ blocks: list = field(default_factory=list)
94
+ width: float = 0.0
95
+ height: float = 0.0
96
+ h_relative: str = 'column'
97
+ v_relative: str = 'paragraph'
98
+ h_offset: float = 0.0
99
+ v_offset: float = 0.0
100
+ h_align: Optional[str] = None
101
+ v_align: Optional[str] = None
102
+ behind: bool = False
103
+ wrap: str = 'none'
104
+ z_index: int = 0
105
+ fill: Optional[str] = None
106
+ stroke: Optional[str] = None
107
+ stroke_width: float = 0.0
108
+ rotation: float = 0.0
109
+ alt: str = ''
110
+
111
+
112
+ @dataclass
113
+ class Block:
114
+ pass
115
+
116
+
117
+ @dataclass
118
+ class Paragraph(Block):
119
+ props: ParaProps = field(default_factory=ParaProps)
120
+ items: List[Inline] = field(default_factory=list)
121
+ mark_props: RunProps = field(default_factory=RunProps)
122
+ number: Optional[NumberLabel] = None
123
+ style_id: Optional[str] = None
124
+ bookmarks: List[str] = field(default_factory=list)
125
+ in_table: bool = False
126
+ floats: List[FloatingObject] = field(default_factory=list)
127
+
128
+ @property
129
+ def text(self) -> str:
130
+ return ''.join(i.text for i in self.items if isinstance(i, TextRun))
131
+
132
+
133
+ @dataclass
134
+ class Cell:
135
+ blocks: List[Block] = field(default_factory=list)
136
+ props: CellProps = field(default_factory=CellProps)
137
+ grid_span: int = 1
138
+ v_merge: Optional[str] = None
139
+ width: Optional[float] = None
140
+ column: int = 0
141
+
142
+
143
+ @dataclass
144
+ class Row:
145
+ cells: List[Cell] = field(default_factory=list)
146
+ props: RowProps = field(default_factory=RowProps)
147
+ header: bool = False
148
+
149
+
150
+ @dataclass
151
+ class Table(Block):
152
+ rows: List[Row] = field(default_factory=list)
153
+ grid: List[float] = field(default_factory=list)
154
+ props: TableProps = field(default_factory=TableProps)
155
+ style_id: Optional[str] = None
156
+ look: dict = field(default_factory=dict)
157
+ nesting: int = 0
158
+
159
+
160
+ @dataclass
161
+ class HeaderFooter:
162
+ blocks: List[Block] = field(default_factory=list)
163
+ part: str = ''
164
+
165
+
166
+ @dataclass
167
+ class Section:
168
+ width: float = 612.0
169
+ height: float = 792.0
170
+ margin_top: float = 72.0
171
+ margin_right: float = 72.0
172
+ margin_bottom: float = 72.0
173
+ margin_left: float = 72.0
174
+ header_distance: float = 36.0
175
+ footer_distance: float = 36.0
176
+ gutter: float = 0.0
177
+ landscape: bool = False
178
+ columns: int = 1
179
+ column_space: float = 36.0
180
+ column_widths: Optional[list] = None
181
+ column_separator: bool = False
182
+ page_num_start: Optional[int] = None
183
+ page_num_format: Optional[str] = None
184
+ title_page: bool = False
185
+ vertical_align: str = 'top'
186
+ break_type: str = 'nextPage'
187
+ blocks: List[Block] = field(default_factory=list)
188
+ headers: dict = field(default_factory=dict) # default|first|even -> HeaderFooter
189
+ footers: dict = field(default_factory=dict)
190
+ borders: Optional[dict] = None
191
+
192
+ @property
193
+ def content_width(self) -> float:
194
+ return max(1.0, self.width - self.margin_left - self.margin_right - self.gutter)
195
+
196
+ @property
197
+ def content_height(self) -> float:
198
+ return max(1.0, self.height - self.margin_top - self.margin_bottom)
199
+
200
+ def column_layout(self):
201
+ """(x_offset, width) for each column, relative to the text area."""
202
+ if self.column_widths:
203
+ out = []
204
+ x = 0.0
205
+ for width, space in self.column_widths:
206
+ out.append((x, width))
207
+ x += width + space
208
+ return out
209
+ count = max(1, self.columns)
210
+ total = self.content_width
211
+ width = (total - self.column_space * (count - 1)) / count
212
+ return [(i * (width + self.column_space), width) for i in range(count)]
213
+
214
+
215
+ @dataclass
216
+ class Document:
217
+ sections: List[Section] = field(default_factory=list)
218
+ styles: object = None
219
+ numbering: object = None
220
+ theme: object = None
221
+ package: object = None
222
+ settings: dict = field(default_factory=dict)
223
+ footnotes: dict = field(default_factory=dict)
224
+ endnotes: dict = field(default_factory=dict)
225
+ properties: dict = field(default_factory=dict)
226
+ warnings: List[str] = field(default_factory=list)
227
+
228
+ @property
229
+ def default_tab(self) -> float:
230
+ return self.settings.get('default_tab', 36.0)
231
+
232
+ def iter_blocks(self):
233
+ for section in self.sections:
234
+ for block in section.blocks:
235
+ yield section, block