dsw-templating 4.35.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dsw/templating/__init__.py +32 -0
- dsw/templating/build_info.py +17 -0
- dsw/templating/consts.py +40 -0
- dsw/templating/context.py +55 -0
- dsw/templating/conversions.py +132 -0
- dsw/templating/documents.py +147 -0
- dsw/templating/docx_pagebreak.py +91 -0
- dsw/templating/exceptions.py +71 -0
- dsw/templating/extraction.py +278 -0
- dsw/templating/filters.py +212 -0
- dsw/templating/formats.py +78 -0
- dsw/templating/http.py +110 -0
- dsw/templating/locales.py +47 -0
- dsw/templating/plugins/__init__.py +5 -0
- dsw/templating/plugins/manager.py +28 -0
- dsw/templating/plugins/specs.py +77 -0
- dsw/templating/pot.py +109 -0
- dsw/templating/py.typed +0 -0
- dsw/templating/resources/pandoc/filters/docx-landscape.lua +36 -0
- dsw/templating/resources/pandoc/filters/docx-pagebreak.lua +35 -0
- dsw/templating/resources/pandoc/filters/docx-toc.lua +52 -0
- dsw/templating/settings.py +72 -0
- dsw/templating/steps/__init__.py +16 -0
- dsw/templating/steps/archive.py +160 -0
- dsw/templating/steps/base.py +84 -0
- dsw/templating/steps/conversion.py +262 -0
- dsw/templating/steps/excel.py +956 -0
- dsw/templating/steps/template.py +233 -0
- dsw/templating/steps/word.py +104 -0
- dsw/templating/template.py +250 -0
- dsw/templating/tests.py +25 -0
- dsw/templating/urls.py +141 -0
- dsw/templating/utils.py +30 -0
- dsw_templating-4.35.0.dist-info/METADATA +126 -0
- dsw_templating-4.35.0.dist-info/RECORD +38 -0
- dsw_templating-4.35.0.dist-info/WHEEL +4 -0
- dsw_templating-4.35.0.dist-info/entry_points.txt +2 -0
- dsw_templating-4.35.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Rendering of DSW document templates, without any services behind it."""
|
|
2
|
+
from .consts import VERSION
|
|
3
|
+
from .context import ContextDefaults, enrich_context_config
|
|
4
|
+
from .documents import DocumentFile, FileFormat, FileFormats
|
|
5
|
+
from .exceptions import MissingExtraError, TemplateError, TemplateTriggeredError
|
|
6
|
+
from .formats import Format
|
|
7
|
+
from .locales import RenderContext, TemplateLocale
|
|
8
|
+
from .plugins import create_manager, hookimpl, hookspec, register_plugin_steps
|
|
9
|
+
from .settings import (
|
|
10
|
+
PandocSettings,
|
|
11
|
+
RenderSettings,
|
|
12
|
+
RequestsSettings,
|
|
13
|
+
SecuritySettings,
|
|
14
|
+
TemplateSettings,
|
|
15
|
+
)
|
|
16
|
+
from .steps import FormatStepError, Step
|
|
17
|
+
from .steps.base import register_step
|
|
18
|
+
from .template import Asset, AssetMetadata, ProjectFileResolver, Template
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
'VERSION',
|
|
23
|
+
'ContextDefaults', 'enrich_context_config',
|
|
24
|
+
'Asset', 'AssetMetadata', 'ProjectFileResolver', 'Template',
|
|
25
|
+
'DocumentFile', 'FileFormat', 'FileFormats',
|
|
26
|
+
'Format', 'FormatStepError', 'Step', 'register_step',
|
|
27
|
+
'MissingExtraError', 'TemplateError', 'TemplateTriggeredError',
|
|
28
|
+
'RenderContext', 'TemplateLocale',
|
|
29
|
+
'PandocSettings', 'RenderSettings', 'RequestsSettings', 'SecuritySettings',
|
|
30
|
+
'TemplateSettings',
|
|
31
|
+
'create_manager', 'hookimpl', 'hookspec', 'register_plugin_steps',
|
|
32
|
+
]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# Generated file
|
|
2
|
+
# - do not overwrite
|
|
3
|
+
# - do not include in git
|
|
4
|
+
from collections import namedtuple
|
|
5
|
+
|
|
6
|
+
BuildInfo = namedtuple(
|
|
7
|
+
'BuildInfo',
|
|
8
|
+
['version', 'built_at', 'sha', 'branch', 'tag'],
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
BUILD_INFO = BuildInfo(
|
|
12
|
+
version='v4.35.0~a55653b',
|
|
13
|
+
built_at='2026-10-06 09:23:39Z',
|
|
14
|
+
sha='a55653b2611764140b5c31f0aed558e618d3a4ab',
|
|
15
|
+
branch='HEAD',
|
|
16
|
+
tag='v4.35.0',
|
|
17
|
+
)
|
dsw/templating/consts.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
DEFAULT_ENCODING = 'utf-8'
|
|
7
|
+
EXIT_SUCCESS = 0
|
|
8
|
+
PACKAGE_NAME = 'dsw-templating'
|
|
9
|
+
PLUGINS_ENTRYPOINT = 'dsw_templating_plugins'
|
|
10
|
+
|
|
11
|
+
JINJA_EXTENSIONS = ('jinja2.ext.do', 'jinja2.ext.loopcontrols')
|
|
12
|
+
JINJA_FILE_EXTENSIONS = ('.j2', '.jinja', '.jinja2', '.jnj')
|
|
13
|
+
|
|
14
|
+
# Rendering and POT extraction must agree on this: the msgid of a {% trans %}
|
|
15
|
+
# block depends on it, and the POT file is per document template while a step
|
|
16
|
+
# option would be per format.
|
|
17
|
+
JINJA_I18N_TRIMMED = True
|
|
18
|
+
|
|
19
|
+
DEFAULT_LANGUAGE = 'en'
|
|
20
|
+
DEFAULT_LOCALE_DOMAIN = 'default'
|
|
21
|
+
|
|
22
|
+
TEMPLATE_JSON_FILE_NAME = 'template.json'
|
|
23
|
+
PROJECT_FILES_DIR = 'project-files'
|
|
24
|
+
|
|
25
|
+
try:
|
|
26
|
+
__version__ = version(PACKAGE_NAME)
|
|
27
|
+
except PackageNotFoundError:
|
|
28
|
+
__version__ = '0.0.0'
|
|
29
|
+
VERSION = __version__
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class FormatField:
|
|
33
|
+
UUID = 'uuid'
|
|
34
|
+
NAME = 'name'
|
|
35
|
+
STEPS = 'steps'
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class StepField:
|
|
39
|
+
NAME = 'name'
|
|
40
|
+
OPTIONS = 'options'
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Values the document worker adds to the document context before rendering."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import dataclasses
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclasses.dataclass
|
|
8
|
+
class ContextDefaults:
|
|
9
|
+
"""Service information and fallbacks of the instance branding (``ctx.config``).
|
|
10
|
+
|
|
11
|
+
The defaults are those of the document worker's configuration.
|
|
12
|
+
"""
|
|
13
|
+
service_name: str = 'Data Stewardship Wizard'
|
|
14
|
+
service_name_short: str = 'DSW'
|
|
15
|
+
service_url: str = 'https://ds-wizard.org'
|
|
16
|
+
service_domain_name: str = 'ds-wizard.org'
|
|
17
|
+
default_primary_color: str = '#0033aa'
|
|
18
|
+
default_illustrations_color: str = '#0033aa'
|
|
19
|
+
default_logo_url: str = '{{clientUrl}}/assets/logo.svg'
|
|
20
|
+
default_app_title: str = 'DS Wizard'
|
|
21
|
+
default_app_title_short: str = 'DS Wizard'
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def enrich_context_config(context: dict, defaults: ContextDefaults):
|
|
25
|
+
"""Fill ``context['config']`` as the document worker does.
|
|
26
|
+
|
|
27
|
+
The service fields are always set from `defaults`; the branding of the
|
|
28
|
+
instance is kept when present and falls back to `defaults` otherwise.
|
|
29
|
+
"""
|
|
30
|
+
old = context.get('config', {})
|
|
31
|
+
|
|
32
|
+
client_url = old.get('clientUrl', '').rstrip('/')
|
|
33
|
+
app_title = (old.get('appTitle', None) or
|
|
34
|
+
defaults.default_app_title)
|
|
35
|
+
app_title_short = (old.get('appTitleShort', None) or
|
|
36
|
+
defaults.default_app_title_short)
|
|
37
|
+
primary_color = (old.get('primaryColor', None) or
|
|
38
|
+
defaults.default_primary_color)
|
|
39
|
+
illustrations_color = (old.get('illustrationsColor', None) or
|
|
40
|
+
defaults.default_illustrations_color)
|
|
41
|
+
logo_url_template = (old.get('logoUrl', None) or
|
|
42
|
+
defaults.default_logo_url)
|
|
43
|
+
logo_url = logo_url_template.replace('{{clientUrl}}', client_url)
|
|
44
|
+
|
|
45
|
+
context['config'].update({
|
|
46
|
+
'serviceName': defaults.service_name,
|
|
47
|
+
'serviceNameShort': defaults.service_name_short,
|
|
48
|
+
'serviceUrl': defaults.service_url,
|
|
49
|
+
'serviceDomainName': defaults.service_domain_name,
|
|
50
|
+
'appTitle': app_title,
|
|
51
|
+
'appTitleShort': app_title_short,
|
|
52
|
+
'primaryColor': primary_color,
|
|
53
|
+
'illustrationsColor': illustrations_color,
|
|
54
|
+
'logoUrl': logo_url,
|
|
55
|
+
})
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import shlex
|
|
5
|
+
import subprocess
|
|
6
|
+
import typing
|
|
7
|
+
|
|
8
|
+
from . import consts
|
|
9
|
+
from .documents import FileFormat, FileFormats
|
|
10
|
+
from .exceptions import require
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
if typing.TYPE_CHECKING:
|
|
14
|
+
from .settings import PandocSettings
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
LOG = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def run_conversion(*, args: list, workdir: str, input_data: bytes, name: str,
|
|
21
|
+
source_format: FileFormat, target_format: FileFormat, timeout=None) -> bytes:
|
|
22
|
+
command = ' '.join(args)
|
|
23
|
+
LOG.info('Calling "%s" to convert from %s to %s',
|
|
24
|
+
command, source_format, target_format)
|
|
25
|
+
try:
|
|
26
|
+
proc = subprocess.Popen(args, cwd=workdir, stdin=subprocess.PIPE,
|
|
27
|
+
stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
|
28
|
+
except FileNotFoundError as e:
|
|
29
|
+
raise FormatConversionError(
|
|
30
|
+
name, source_format, target_format,
|
|
31
|
+
f'Executable "{args[0]}" not found, is it installed?',
|
|
32
|
+
) from e
|
|
33
|
+
with proc:
|
|
34
|
+
stdout, stderr = proc.communicate(input=input_data, timeout=timeout)
|
|
35
|
+
exit_code = proc.returncode
|
|
36
|
+
if exit_code != consts.EXIT_SUCCESS:
|
|
37
|
+
raise FormatConversionError(
|
|
38
|
+
name, source_format, target_format,
|
|
39
|
+
f'Failed to execute (exit code: {exit_code}): '
|
|
40
|
+
f'{stderr.decode(consts.DEFAULT_ENCODING)}',
|
|
41
|
+
)
|
|
42
|
+
return stdout
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class FormatConversionError(Exception):
|
|
46
|
+
|
|
47
|
+
def __init__(self, convertor, source_format, target_format, message):
|
|
48
|
+
self.convertor = convertor
|
|
49
|
+
self.source_format = source_format
|
|
50
|
+
self.target_format = target_format
|
|
51
|
+
self.message = message
|
|
52
|
+
|
|
53
|
+
def __str__(self):
|
|
54
|
+
return f'{self.convertor} failed to convert {self.source_format}' \
|
|
55
|
+
f' to {self.target_format} - {self.message}'
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Pandoc:
|
|
59
|
+
|
|
60
|
+
def __init__(self, settings: PandocSettings, filter_names: list[str],
|
|
61
|
+
template_name: str | None):
|
|
62
|
+
self.settings = settings
|
|
63
|
+
self.filter_names = filter_names
|
|
64
|
+
self.template_name = template_name
|
|
65
|
+
self._check_filters()
|
|
66
|
+
self._check_template()
|
|
67
|
+
|
|
68
|
+
def _check_filters(self):
|
|
69
|
+
for name in self.filter_names:
|
|
70
|
+
if self.settings.filter_path(name) is None:
|
|
71
|
+
raise RuntimeError(f'Pandoc filter "{name}" not found')
|
|
72
|
+
|
|
73
|
+
def _check_template(self):
|
|
74
|
+
if self.template_name and self.settings.template_path(self.template_name) is None:
|
|
75
|
+
raise RuntimeError(f'Pandoc template "{self.template_name}" not found')
|
|
76
|
+
|
|
77
|
+
def _extra_args(self) -> list[str]:
|
|
78
|
+
# paths are passed as they are (not re-split), they may contain spaces
|
|
79
|
+
args: list[str] = []
|
|
80
|
+
if self.template_name:
|
|
81
|
+
args.extend(['--template', str(self.settings.template_path(self.template_name))])
|
|
82
|
+
for filter_name in self.filter_names:
|
|
83
|
+
option = '--lua-filter' if filter_name.endswith('.lua') else '--filter'
|
|
84
|
+
args.extend([option, str(self.settings.filter_path(filter_name))])
|
|
85
|
+
return args
|
|
86
|
+
|
|
87
|
+
def __call__(self, *, source_format: FileFormat, target_format: FileFormat,
|
|
88
|
+
data: bytes, metadata: dict, workdir: str) -> bytes:
|
|
89
|
+
args = ['-f', source_format.name, '-t', target_format.name, '-o', '-']
|
|
90
|
+
template_args = self.extract_template_args(metadata)
|
|
91
|
+
extra_args = self._extra_args()
|
|
92
|
+
command = self.settings.command + template_args + extra_args + args
|
|
93
|
+
return run_conversion(
|
|
94
|
+
args=command,
|
|
95
|
+
workdir=workdir,
|
|
96
|
+
input_data=data,
|
|
97
|
+
name=type(self).__name__,
|
|
98
|
+
source_format=source_format,
|
|
99
|
+
target_format=target_format,
|
|
100
|
+
timeout=self.settings.timeout,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
@staticmethod
|
|
104
|
+
def extract_template_args(metadata: dict):
|
|
105
|
+
return shlex.split(metadata.get('args', ''))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class RdfLibConvert:
|
|
109
|
+
|
|
110
|
+
FORMATS = {
|
|
111
|
+
FileFormats.RDF_XML: 'xml',
|
|
112
|
+
FileFormats.N3: 'n3',
|
|
113
|
+
FileFormats.NTRIPLES: 'ntriples',
|
|
114
|
+
FileFormats.TURTLE: 'turtle',
|
|
115
|
+
FileFormats.TRIG: 'trig',
|
|
116
|
+
FileFormats.JSONLD: 'json-ld',
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
def __init__(self):
|
|
120
|
+
self.rdflib = require('rdflib', 'rdf')
|
|
121
|
+
|
|
122
|
+
def __call__(self, *, source_format: FileFormat, target_format: FileFormat,
|
|
123
|
+
data: bytes, metadata: dict) -> bytes:
|
|
124
|
+
g = self.rdflib.Dataset()
|
|
125
|
+
g.parse(
|
|
126
|
+
data=data.decode(consts.DEFAULT_ENCODING),
|
|
127
|
+
format=self.FORMATS.get(source_format) or 'turtle',
|
|
128
|
+
)
|
|
129
|
+
return g.serialize(
|
|
130
|
+
format=self.FORMATS.get(target_format) or 'turtle',
|
|
131
|
+
encoding=consts.DEFAULT_ENCODING,
|
|
132
|
+
)
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import pathlib
|
|
4
|
+
|
|
5
|
+
from . import consts
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class FileFormat:
|
|
9
|
+
|
|
10
|
+
def __init__(self, name: str, content_type: str, file_extension: str):
|
|
11
|
+
self.name = name
|
|
12
|
+
self.content_type = content_type
|
|
13
|
+
self.file_extension = file_extension
|
|
14
|
+
|
|
15
|
+
def __eq__(self, other):
|
|
16
|
+
return isinstance(other, FileFormat) and other.name == self.name
|
|
17
|
+
|
|
18
|
+
def __hash__(self):
|
|
19
|
+
return hash(self.name)
|
|
20
|
+
|
|
21
|
+
def __str__(self):
|
|
22
|
+
return self.name
|
|
23
|
+
|
|
24
|
+
def __repr__(self):
|
|
25
|
+
return f'Format[{self.name}]'
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class FileFormats:
|
|
29
|
+
JSON = FileFormat('json', 'application/json', 'json')
|
|
30
|
+
HTML = FileFormat('html', 'text/html', 'html')
|
|
31
|
+
PDF = FileFormat('pdf', 'application/pdf', 'pdf')
|
|
32
|
+
DOCX = FileFormat(
|
|
33
|
+
'docx',
|
|
34
|
+
'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
|
|
35
|
+
'docx',
|
|
36
|
+
)
|
|
37
|
+
Markdown = FileFormat('markdown', 'text/markdown', 'md')
|
|
38
|
+
ODT = FileFormat('odt', 'application/vnd.oasis.opendocument.text', 'odt')
|
|
39
|
+
RST = FileFormat('rst', 'text/x-rst', 'rst')
|
|
40
|
+
LaTeX = FileFormat('latex', 'application/x-tex', 'tex')
|
|
41
|
+
EPUB = FileFormat('epub', 'application/epub+zip', 'epub')
|
|
42
|
+
DocBook4 = FileFormat('docbook4', 'application/docbook+xml', 'dbk')
|
|
43
|
+
DocBook5 = FileFormat('docbook5', 'application/docbook+xml', 'dbk')
|
|
44
|
+
PPTX = FileFormat(
|
|
45
|
+
'pptx',
|
|
46
|
+
'application/vnd.openxmlformats-officedocument.presentationml.presentation',
|
|
47
|
+
'pptx',
|
|
48
|
+
)
|
|
49
|
+
RTF = FileFormat('rtf', 'application/rtf', 'rtf')
|
|
50
|
+
ADoc = FileFormat('asciidoc', 'text/asciidoc', 'adoc')
|
|
51
|
+
RDF_XML = FileFormat('rdf', 'application/rdf+xml', 'rdf')
|
|
52
|
+
N3 = FileFormat('n3', 'text/n3', 'n3')
|
|
53
|
+
NTRIPLES = FileFormat('nt', 'application/n-triples', 'nt')
|
|
54
|
+
TURTLE = FileFormat('ttl', 'text/turtle', 'ttl')
|
|
55
|
+
TRIG = FileFormat('trig', 'application/trig', 'trig')
|
|
56
|
+
JSONLD = FileFormat('jsonld', 'application/ld+json', 'jsonld')
|
|
57
|
+
ZIP = FileFormat('zip', 'application/zip', 'zip')
|
|
58
|
+
TAR = FileFormat('tar', 'application/x-tar', 'tar')
|
|
59
|
+
TAR_GZIP = FileFormat('gzip', 'application/gzip', 'tar.gz')
|
|
60
|
+
TAR_BZIP2 = FileFormat('bzip2', 'application/x-bzip2', 'tar.bz2')
|
|
61
|
+
TAR_LZMA = FileFormat('lzma', 'application/x-lzma', 'tar.xz')
|
|
62
|
+
XLSX = FileFormat(
|
|
63
|
+
'xlsx',
|
|
64
|
+
'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
|
|
65
|
+
'xlsx',
|
|
66
|
+
)
|
|
67
|
+
XLSM = FileFormat(
|
|
68
|
+
'xlsm',
|
|
69
|
+
'application/vnd.ms-excel.sheet.macroEnabled.12',
|
|
70
|
+
'xlsm',
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
@staticmethod
|
|
74
|
+
def get(name: str):
|
|
75
|
+
known_formats = {
|
|
76
|
+
'html': FileFormats.HTML,
|
|
77
|
+
'pdf': FileFormats.PDF,
|
|
78
|
+
'docx': FileFormats.DOCX,
|
|
79
|
+
'markdown': FileFormats.Markdown,
|
|
80
|
+
'odt': FileFormats.ODT,
|
|
81
|
+
'rst': FileFormats.RST,
|
|
82
|
+
'latex': FileFormats.LaTeX,
|
|
83
|
+
'json': FileFormats.JSON,
|
|
84
|
+
'epub': FileFormats.EPUB,
|
|
85
|
+
'docbook4': FileFormats.DocBook4,
|
|
86
|
+
'docbook5': FileFormats.DocBook5,
|
|
87
|
+
'pptx': FileFormats.PPTX,
|
|
88
|
+
'rtf': FileFormats.RTF,
|
|
89
|
+
'asciidoc': FileFormats.ADoc,
|
|
90
|
+
'rdf': FileFormats.RDF_XML,
|
|
91
|
+
'rdf/xml': FileFormats.RDF_XML,
|
|
92
|
+
'turtle': FileFormats.TURTLE,
|
|
93
|
+
'ttl': FileFormats.TURTLE,
|
|
94
|
+
'n3': FileFormats.N3,
|
|
95
|
+
'ntriples': FileFormats.NTRIPLES,
|
|
96
|
+
'n-triples': FileFormats.NTRIPLES,
|
|
97
|
+
'trig': FileFormats.TRIG,
|
|
98
|
+
'json-ld': FileFormats.JSONLD,
|
|
99
|
+
'jsonld': FileFormats.JSONLD,
|
|
100
|
+
'zip': FileFormats.ZIP,
|
|
101
|
+
'tar': FileFormats.TAR,
|
|
102
|
+
'gzip': FileFormats.TAR_GZIP,
|
|
103
|
+
'bzip2': FileFormats.TAR_BZIP2,
|
|
104
|
+
'lzma': FileFormats.TAR_LZMA,
|
|
105
|
+
'xlsx': FileFormats.XLSX,
|
|
106
|
+
'xlsm': FileFormats.XLSM,
|
|
107
|
+
}
|
|
108
|
+
return known_formats.get(name)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class DocumentFile:
|
|
112
|
+
|
|
113
|
+
def __init__(self, file_format: FileFormat, content: bytes,
|
|
114
|
+
encoding: str | None = None):
|
|
115
|
+
self.file_format = file_format
|
|
116
|
+
self._content = content
|
|
117
|
+
self.byte_size = len(content)
|
|
118
|
+
self.encoding = encoding
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def content_type(self) -> str:
|
|
122
|
+
return self.file_format.content_type
|
|
123
|
+
|
|
124
|
+
@property
|
|
125
|
+
def safe_encoding(self) -> str:
|
|
126
|
+
return self.encoding or consts.DEFAULT_ENCODING
|
|
127
|
+
|
|
128
|
+
@property
|
|
129
|
+
def content(self) -> bytes:
|
|
130
|
+
return self._content
|
|
131
|
+
|
|
132
|
+
@content.setter
|
|
133
|
+
def content(self, content: bytes):
|
|
134
|
+
self._content = content
|
|
135
|
+
self.byte_size = len(content)
|
|
136
|
+
|
|
137
|
+
def filename(self, name: str) -> str:
|
|
138
|
+
return f'{name}.{self.file_format.file_extension}'
|
|
139
|
+
|
|
140
|
+
def store(self, name: str):
|
|
141
|
+
pathlib.Path(self.filename(name)).write_bytes(self.content)
|
|
142
|
+
|
|
143
|
+
@property
|
|
144
|
+
def object_content_type(self) -> str:
|
|
145
|
+
if self.encoding is not None:
|
|
146
|
+
return f'{self.content_type}; charset={self.encoding}'
|
|
147
|
+
return self.content_type
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Pandoc filter `pandoc-docx-pagebreakpy`: page breaks and TOC as OpenXML raw blocks (DOCX only).
|
|
2
|
+
|
|
3
|
+
`\\newpage` becomes a page break and `\\toc` (`\\toc1` to `\\toc6` for the depth) a table of
|
|
4
|
+
contents. Installed as the `pandoc-docx-pagebreakpy` command with the `docx` extra, for
|
|
5
|
+
`--filter=pandoc-docx-pagebreakpy` in the `args` of the `pandoc` step. Deprecated in favour of the
|
|
6
|
+
`docx-pagebreak.lua` and `docx-toc.lua` filters.
|
|
7
|
+
|
|
8
|
+
Based on https://github.com/pandocker/pandoc-docx-pagebreak-py (formerly the
|
|
9
|
+
`pandoc-docx-pagebreak` addon of the document worker):
|
|
10
|
+
"""
|
|
11
|
+
# MIT License
|
|
12
|
+
#
|
|
13
|
+
# Copyright (c) 2018 pandocker
|
|
14
|
+
#
|
|
15
|
+
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
16
|
+
# of this software and associated documentation files (the "Software"), to deal
|
|
17
|
+
# in the Software without restriction, including without limitation the rights
|
|
18
|
+
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
19
|
+
# copies of the Software, and to permit persons to whom the Software is
|
|
20
|
+
# furnished to do so, subject to the following conditions:
|
|
21
|
+
#
|
|
22
|
+
# The above copyright notice and this permission notice shall be included in all
|
|
23
|
+
# copies or substantial portions of the Software.
|
|
24
|
+
#
|
|
25
|
+
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
26
|
+
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
27
|
+
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
28
|
+
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
29
|
+
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
30
|
+
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
31
|
+
# SOFTWARE.
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import sys
|
|
35
|
+
|
|
36
|
+
from .exceptions import MissingExtraError, require
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
TOC_DEPTHS = {r'\toc1': 1, r'\toc2': 2, r'\toc4': 4, r'\toc5': 5, r'\toc6': 6}
|
|
40
|
+
DEFAULT_TOC_DEPTH = 3
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class DocxPagebreak:
|
|
44
|
+
|
|
45
|
+
def __init__(self):
|
|
46
|
+
self.pf = require('panflute', 'docx')
|
|
47
|
+
|
|
48
|
+
def _make_pagebreak(self):
|
|
49
|
+
return self.pf.RawBlock('<w:p><w:r><w:br w:type="page" /></w:r></w:p>', format='openxml')
|
|
50
|
+
|
|
51
|
+
def _make_toc(self, instr: str):
|
|
52
|
+
depth = TOC_DEPTHS.get(instr, DEFAULT_TOC_DEPTH)
|
|
53
|
+
toc_lines = [
|
|
54
|
+
r'<w:sdt>',
|
|
55
|
+
r'<w:sdtContent xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">',
|
|
56
|
+
r'<w:p><w:r>',
|
|
57
|
+
r'<w:fldChar w:fldCharType="begin" w:dirty="true" />',
|
|
58
|
+
rf'<w:instrText xml:space="preserve">TOC \o "1-{depth}" \h \z \u</w:instrText>',
|
|
59
|
+
r'<w:fldChar w:fldCharType="separate" />',
|
|
60
|
+
r'<w:fldChar w:fldCharType="end" />',
|
|
61
|
+
r'</w:r></w:p>',
|
|
62
|
+
r'</w:sdtContent>',
|
|
63
|
+
r'</w:sdt>',
|
|
64
|
+
]
|
|
65
|
+
return self.pf.RawBlock('\n'.join(toc_lines), format='openxml')
|
|
66
|
+
|
|
67
|
+
def action(self, elem, doc):
|
|
68
|
+
pf = self.pf
|
|
69
|
+
if doc.format != 'docx':
|
|
70
|
+
return elem
|
|
71
|
+
if isinstance(elem, (pf.Para, pf.Plain)):
|
|
72
|
+
for child in elem.content:
|
|
73
|
+
if isinstance(child, pf.Str) and child.text == r'\newpage':
|
|
74
|
+
elem = self._make_pagebreak()
|
|
75
|
+
elif isinstance(child, pf.Str) and child.text.startswith(r'\toc'):
|
|
76
|
+
elem = self._make_toc(child.text)
|
|
77
|
+
if isinstance(elem, pf.RawBlock):
|
|
78
|
+
if elem.text == r'\newpage':
|
|
79
|
+
elem = self._make_pagebreak()
|
|
80
|
+
elif elem.text.startswith(r'\toc'):
|
|
81
|
+
elem = self._make_toc(elem.text)
|
|
82
|
+
return elem
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def main(doc=None):
|
|
86
|
+
try:
|
|
87
|
+
dp = DocxPagebreak()
|
|
88
|
+
except MissingExtraError as e:
|
|
89
|
+
# run by pandoc: a message on stderr rather than a traceback
|
|
90
|
+
sys.exit(f'pandoc-docx-pagebreakpy: {e}')
|
|
91
|
+
return dp.pf.run_filter(dp.action, doc=doc)
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import typing
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
if typing.TYPE_CHECKING:
|
|
8
|
+
import types
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class TemplateError(Exception):
|
|
12
|
+
|
|
13
|
+
def __init__(self, template_uuid: str, message: str):
|
|
14
|
+
self.template_uuid = template_uuid
|
|
15
|
+
self.message = message
|
|
16
|
+
|
|
17
|
+
def __str__(self):
|
|
18
|
+
return f'Error in template "{self.template_uuid}"\n' \
|
|
19
|
+
f'- {self.message}'
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class TemplateTriggeredError(Exception):
|
|
23
|
+
"""Error invoked from a template to report a problem to a user (not system)."""
|
|
24
|
+
|
|
25
|
+
def __init__(self, title, message):
|
|
26
|
+
self.title = title
|
|
27
|
+
self.message = message
|
|
28
|
+
self.msg = f'{title}\n\n{message}'
|
|
29
|
+
super().__init__(self.msg)
|
|
30
|
+
|
|
31
|
+
def __str__(self):
|
|
32
|
+
return self.msg
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class MissingExtraError(ImportError):
|
|
36
|
+
"""An optional dependency of a step is not installed."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, module: str, extra: str):
|
|
39
|
+
self.module = module
|
|
40
|
+
self.extra = extra
|
|
41
|
+
super().__init__(
|
|
42
|
+
f'Module "{module}" is not installed, '
|
|
43
|
+
f'install dsw-templating[{extra}] to use it',
|
|
44
|
+
name=module,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def require(module: str, extra: str) -> types.ModuleType:
|
|
49
|
+
"""Import an optional dependency, failing with the extra to install."""
|
|
50
|
+
try:
|
|
51
|
+
return importlib.import_module(module)
|
|
52
|
+
except ImportError as e:
|
|
53
|
+
raise MissingExtraError(module, extra) from e
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class MissingModule:
|
|
57
|
+
"""Stands in for an optional module; fails as soon as it is used."""
|
|
58
|
+
|
|
59
|
+
def __init__(self, module: str, extra: str):
|
|
60
|
+
self._module = module
|
|
61
|
+
self._extra = extra
|
|
62
|
+
|
|
63
|
+
def __getattr__(self, name: str):
|
|
64
|
+
raise MissingExtraError(self._module, self._extra)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def optional(module: str, extra: str) -> types.ModuleType | MissingModule:
|
|
68
|
+
try:
|
|
69
|
+
return importlib.import_module(module)
|
|
70
|
+
except ImportError:
|
|
71
|
+
return MissingModule(module, extra)
|