qparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qmd_cli.py +262 -0
- qparse/__init__.py +16 -0
- qparse/docx_render/__init__.py +3 -0
- qparse/docx_render/docx_render.py +68 -0
- qparse/docx_render/write_buffer.py +412 -0
- qparse/markdwon_document/__init__.py +28 -0
- qparse/markdwon_document/analysis_document.py +40 -0
- qparse/markdwon_document/base_document.py +87 -0
- qparse/markdwon_document/markdwon_document.py +16 -0
- qparse/markdwon_document/nodes/__init__.py +5 -0
- qparse/markdwon_document/nodes/answer_node.py +97 -0
- qparse/markdwon_document/nodes/base_node.py +51 -0
- qparse/markdwon_document/nodes/img_node.py +31 -0
- qparse/markdwon_document/nodes/text_node.py +139 -0
- qparse/markdwon_document/question_document.py +17 -0
- qparse/markdwon_document/stem_document.py +64 -0
- qparse/markdwon_document/table_document.py +126 -0
- qparse/markdwon_loader.py +189 -0
- qparse/markdwon_render.py +123 -0
- qparse/qmd_packer.py +475 -0
- qparse/qmd_unpacker.py +169 -0
- qparse/render_option/__init__.py +40 -0
- qparse/render_option/default_render_option.py +86 -0
- qparse/render_option/default_render_template.md +21 -0
- qparse/render_option/referance.docx +0 -0
- qparse/render_option/theme.json +245 -0
- qparse/utils/__init__.py +3 -0
- qparse/utils/html_full_protector.py +154 -0
- qparse-0.1.0.dist-info/METADATA +454 -0
- qparse-0.1.0.dist-info/RECORD +34 -0
- qparse-0.1.0.dist-info/WHEEL +5 -0
- qparse-0.1.0.dist-info/entry_points.txt +2 -0
- qparse-0.1.0.dist-info/licenses/LICENSE +21 -0
- qparse-0.1.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from enum import Enum
|
|
3
|
+
import json,copy
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from .render_option import get_default_render_option,get_default_theme
|
|
6
|
+
from .markdwon_loader import MarkdownLoader
|
|
7
|
+
from .docx_render import DocxRender
|
|
8
|
+
from .markdwon_document import MarkdwonDocument
|
|
9
|
+
from bs4 import BeautifulSoup,element
|
|
10
|
+
from typing import Union
|
|
11
|
+
|
|
12
|
+
class RenderToEnum(Enum):
|
|
13
|
+
docx = 'docx'
|
|
14
|
+
markdwon = 'markdwon'
|
|
15
|
+
|
|
16
|
+
class MarkdwonRender():
|
|
17
|
+
|
|
18
|
+
def __init__(self,src:str):
|
|
19
|
+
self._src = src
|
|
20
|
+
self.__initialize__()
|
|
21
|
+
|
|
22
|
+
def __initialize__(self):
|
|
23
|
+
self._src = Path(self._src).resolve()
|
|
24
|
+
self._render_soup = MarkdownLoader(str(self._src)).get_soup().select_one('render')
|
|
25
|
+
self._mdsrc = self._resolve_path(self._render_soup.attrs['src'])
|
|
26
|
+
# 初始化主题
|
|
27
|
+
theme_src = self._render_soup.attrs.get("theme", None)
|
|
28
|
+
if theme_src:
|
|
29
|
+
theme_path = self._resolve_path(theme_src)
|
|
30
|
+
self._theme = json.load(theme_path.open(mode='r', encoding='utf-8'))
|
|
31
|
+
else:
|
|
32
|
+
self._theme = get_default_theme()
|
|
33
|
+
#初始化loader
|
|
34
|
+
self._markdwon_loader = MarkdownLoader(str(self._mdsrc))
|
|
35
|
+
self._markdwon_soup = self._markdwon_loader.get_soup()
|
|
36
|
+
|
|
37
|
+
def _resolve_path(self, raw: str) -> Path:
|
|
38
|
+
path = Path(raw)
|
|
39
|
+
if not path.is_absolute():
|
|
40
|
+
path = self._src.parent / path
|
|
41
|
+
return path.resolve()
|
|
42
|
+
@property
|
|
43
|
+
def soup(self):
|
|
44
|
+
return copy.deepcopy(self._markdwon_soup)
|
|
45
|
+
def _parse_file_node(self,node:element.Tag):
|
|
46
|
+
source_soup = self.soup
|
|
47
|
+
output_soup = BeautifulSoup(
|
|
48
|
+
f'<markdwon src="{self._mdsrc.resolve()}"></markdwon>',
|
|
49
|
+
'html.parser'
|
|
50
|
+
)
|
|
51
|
+
output = output_soup.markdwon
|
|
52
|
+
doc_node = node.find('doc', recursive=False)
|
|
53
|
+
|
|
54
|
+
before = doc_node.find('before', recursive=False) if doc_node else None
|
|
55
|
+
after = doc_node.find('after', recursive=False) if doc_node else None
|
|
56
|
+
|
|
57
|
+
if before is not None:
|
|
58
|
+
output.extend(self._translate_content(before))
|
|
59
|
+
|
|
60
|
+
output.extend(list(source_soup.contents))
|
|
61
|
+
|
|
62
|
+
if after is not None:
|
|
63
|
+
output.extend(self._translate_content(after))
|
|
64
|
+
|
|
65
|
+
return output
|
|
66
|
+
|
|
67
|
+
def _translate_content(self, container:element.Tag):
|
|
68
|
+
contents = []
|
|
69
|
+
protector = self._markdwon_loader._protector
|
|
70
|
+
|
|
71
|
+
for child in container.contents:
|
|
72
|
+
if isinstance(child, element.Tag) and child.name == 'reference-answer':
|
|
73
|
+
reference_answer = (
|
|
74
|
+
self._markdwon_loader.get_reference_answer_markdwon()
|
|
75
|
+
)
|
|
76
|
+
encoded = protector.encode(reference_answer)
|
|
77
|
+
decoded = protector.decode_to_html_entity(encoded)
|
|
78
|
+
contents.extend(
|
|
79
|
+
BeautifulSoup(decoded, 'html.parser').contents
|
|
80
|
+
)
|
|
81
|
+
else:
|
|
82
|
+
contents.append(copy.deepcopy(child))
|
|
83
|
+
|
|
84
|
+
return contents
|
|
85
|
+
|
|
86
|
+
def render(self):
|
|
87
|
+
outputs = []
|
|
88
|
+
for file_node in self._render_soup.select('file'):
|
|
89
|
+
doc_node = file_node.find('doc', recursive=False)
|
|
90
|
+
if doc_node is None:
|
|
91
|
+
continue
|
|
92
|
+
|
|
93
|
+
theme_name = doc_node.get('theme')
|
|
94
|
+
render_option = copy.deepcopy(
|
|
95
|
+
self._theme.get(theme_name, self._theme)
|
|
96
|
+
if theme_name else self._theme
|
|
97
|
+
)
|
|
98
|
+
markdown_node = self._parse_file_node(file_node)
|
|
99
|
+
markdown_doc = MarkdwonDocument(
|
|
100
|
+
node=markdown_node,
|
|
101
|
+
render_option=render_option
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
dist = file_node.get('dist')
|
|
105
|
+
if dist:
|
|
106
|
+
dist_path = Path(dist)
|
|
107
|
+
if not dist_path.is_absolute():
|
|
108
|
+
dist_path = Path(self._src).resolve().parent / dist_path
|
|
109
|
+
else:
|
|
110
|
+
dist_path = self._mdsrc.with_suffix('.docx')
|
|
111
|
+
|
|
112
|
+
with DocxRender(MdDoc=markdown_doc) as doc_render:
|
|
113
|
+
outputs.append(doc_render.render(dist=str(dist_path)))
|
|
114
|
+
|
|
115
|
+
return outputs
|
|
116
|
+
|
|
117
|
+
def dev(self):
|
|
118
|
+
return [
|
|
119
|
+
self._parse_file_node(filenode)
|
|
120
|
+
for filenode in self._render_soup.select('file')
|
|
121
|
+
]
|
|
122
|
+
|
|
123
|
+
|
qparse/qmd_packer.py
ADDED
|
@@ -0,0 +1,475 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""将 workspace 中的教案资源打包为 .qmd 文件。"""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import hashlib
|
|
7
|
+
import json
|
|
8
|
+
import re
|
|
9
|
+
import shutil
|
|
10
|
+
import tempfile
|
|
11
|
+
import zipfile
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Dict, List, Optional, Set, Tuple
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class QmdPacker:
|
|
17
|
+
"""
|
|
18
|
+
根据 code-workspace 与入口 render.md 收集 Markdown、图片、主题和模板,
|
|
19
|
+
将二进制资源按 SHA-256 去重后写入 resources/,并打包成 .qmd。
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
IMAGE_PATTERN = re.compile(r"!\[([^\]]*)\]\(([^)]+)\)")
|
|
23
|
+
IMPORT_PATTERN = re.compile(
|
|
24
|
+
r'^\s*@import\s+["\']([^"\']+)["\']\s*$',
|
|
25
|
+
re.MULTILINE,
|
|
26
|
+
)
|
|
27
|
+
HTML_IMG_PATTERN = re.compile(
|
|
28
|
+
r'<img\b[^>]*\bsrc\s*=\s*["\']([^"\']+)["\'][^>]*>',
|
|
29
|
+
re.IGNORECASE,
|
|
30
|
+
)
|
|
31
|
+
RENDER_SRC_PATTERN = re.compile(
|
|
32
|
+
r'(<render\b[^>]*\bsrc\s*=\s*["\'])([^"\']+)(["\'])',
|
|
33
|
+
re.IGNORECASE,
|
|
34
|
+
)
|
|
35
|
+
THEME_SRC_PATTERN = re.compile(
|
|
36
|
+
r'(theme\s*=\s*["\'])([^"\']+)(["\'])',
|
|
37
|
+
re.IGNORECASE,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
TEXT_SUFFIXES = {".md", ".markdown", ".json", ".txt", ".html", ".htm"}
|
|
41
|
+
BINARY_SUFFIXES = {
|
|
42
|
+
".png",
|
|
43
|
+
".jpg",
|
|
44
|
+
".jpeg",
|
|
45
|
+
".gif",
|
|
46
|
+
".bmp",
|
|
47
|
+
".webp",
|
|
48
|
+
".svg",
|
|
49
|
+
".docx",
|
|
50
|
+
".doc",
|
|
51
|
+
".pdf",
|
|
52
|
+
".ttf",
|
|
53
|
+
".otf",
|
|
54
|
+
".woff",
|
|
55
|
+
".woff2",
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
def __init__(self, workspace: Optional[str] = None):
|
|
59
|
+
default_workspace = (
|
|
60
|
+
Path(__file__).resolve().parents[1] / "QParse.code-workspace"
|
|
61
|
+
)
|
|
62
|
+
self.workspace_path = Path(workspace or default_workspace).resolve()
|
|
63
|
+
if not self.workspace_path.exists():
|
|
64
|
+
raise FileNotFoundError(f"workspace 不存在: {self.workspace_path}")
|
|
65
|
+
|
|
66
|
+
self.workspace_root = self.workspace_path.parent
|
|
67
|
+
self.folders = self._load_workspace_folders()
|
|
68
|
+
|
|
69
|
+
self._resources: Dict[str, dict] = {}
|
|
70
|
+
self._documents: Dict[str, str] = {}
|
|
71
|
+
self._document_assets: Dict[str, List[str]] = {}
|
|
72
|
+
self._path_to_package: Dict[Path, str] = {}
|
|
73
|
+
self._sha_to_path: Dict[str, Path] = {}
|
|
74
|
+
|
|
75
|
+
def pack(
|
|
76
|
+
self,
|
|
77
|
+
entry: Optional[str] = None,
|
|
78
|
+
output: Optional[str] = None,
|
|
79
|
+
) -> Path:
|
|
80
|
+
"""
|
|
81
|
+
打包入口。
|
|
82
|
+
|
|
83
|
+
:param entry: render.md 路径,默认使用 workspace 根目录下的 render.md
|
|
84
|
+
:param output: 输出 .qmd 路径,默认与 entry 同名
|
|
85
|
+
"""
|
|
86
|
+
entry_path = Path(entry or (self.workspace_root / "render.md")).resolve()
|
|
87
|
+
if not entry_path.exists():
|
|
88
|
+
raise FileNotFoundError(f"入口文件不存在: {entry_path}")
|
|
89
|
+
|
|
90
|
+
output_path = Path(
|
|
91
|
+
output or entry_path.with_suffix(".qmd")
|
|
92
|
+
).resolve()
|
|
93
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
94
|
+
|
|
95
|
+
self._resources.clear()
|
|
96
|
+
self._documents.clear()
|
|
97
|
+
self._document_assets.clear()
|
|
98
|
+
self._path_to_package.clear()
|
|
99
|
+
self._sha_to_path.clear()
|
|
100
|
+
|
|
101
|
+
with tempfile.TemporaryDirectory(prefix="qmd_pack_") as temp_dir:
|
|
102
|
+
package_root = Path(temp_dir)
|
|
103
|
+
(package_root / "documents").mkdir(parents=True, exist_ok=True)
|
|
104
|
+
(package_root / "resources").mkdir(parents=True, exist_ok=True)
|
|
105
|
+
(package_root / "themes").mkdir(parents=True, exist_ok=True)
|
|
106
|
+
(package_root / "templates").mkdir(parents=True, exist_ok=True)
|
|
107
|
+
|
|
108
|
+
entry_text = entry_path.read_text(encoding="utf-8")
|
|
109
|
+
document_src = self._extract_render_src(entry_text, entry_path)
|
|
110
|
+
theme_src = self._extract_theme_src(entry_text, entry_path)
|
|
111
|
+
|
|
112
|
+
packaged_docs = self._collect_markdown_tree(document_src)
|
|
113
|
+
for md_path in packaged_docs:
|
|
114
|
+
self._package_markdown(md_path, package_root)
|
|
115
|
+
|
|
116
|
+
theme_package_path = None
|
|
117
|
+
if theme_src is not None and theme_src.exists():
|
|
118
|
+
theme_package_path = self._package_theme(theme_src, package_root)
|
|
119
|
+
|
|
120
|
+
template_package_path = self._package_default_template(package_root)
|
|
121
|
+
|
|
122
|
+
packaged_entry = self._rewrite_entry(
|
|
123
|
+
entry_text=entry_text,
|
|
124
|
+
entry_path=entry_path,
|
|
125
|
+
document_src=document_src,
|
|
126
|
+
theme_package_path=theme_package_path,
|
|
127
|
+
)
|
|
128
|
+
(package_root / "render.md").write_text(
|
|
129
|
+
packaged_entry,
|
|
130
|
+
encoding="utf-8",
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
manifest = {
|
|
134
|
+
"version": 1,
|
|
135
|
+
"entry": "render.md",
|
|
136
|
+
"workspace": {
|
|
137
|
+
"path": self._to_posix(self.workspace_path),
|
|
138
|
+
"folders": [
|
|
139
|
+
{
|
|
140
|
+
"name": folder["name"],
|
|
141
|
+
"path": self._to_posix(folder["path"]),
|
|
142
|
+
}
|
|
143
|
+
for folder in self.folders
|
|
144
|
+
],
|
|
145
|
+
},
|
|
146
|
+
"documents": self._documents,
|
|
147
|
+
"document_assets": self._document_assets,
|
|
148
|
+
"resources": self._resources,
|
|
149
|
+
"theme": theme_package_path,
|
|
150
|
+
"template": template_package_path,
|
|
151
|
+
}
|
|
152
|
+
(package_root / "manifest.json").write_text(
|
|
153
|
+
json.dumps(manifest, ensure_ascii=False, indent=2),
|
|
154
|
+
encoding="utf-8",
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
self._zip_directory(package_root, output_path)
|
|
158
|
+
|
|
159
|
+
return output_path
|
|
160
|
+
|
|
161
|
+
def _load_workspace_folders(self) -> List[dict]:
|
|
162
|
+
data = json.loads(self.workspace_path.read_text(encoding="utf-8"))
|
|
163
|
+
folders = []
|
|
164
|
+
for index, item in enumerate(data.get("folders", [])):
|
|
165
|
+
raw = item.get("path", ".")
|
|
166
|
+
folder_path = Path(raw)
|
|
167
|
+
if not folder_path.is_absolute():
|
|
168
|
+
folder_path = (self.workspace_root / folder_path).resolve()
|
|
169
|
+
else:
|
|
170
|
+
folder_path = folder_path.resolve()
|
|
171
|
+
name = item.get("name") or folder_path.name or f"folder-{index}"
|
|
172
|
+
folders.append(
|
|
173
|
+
{
|
|
174
|
+
"name": name,
|
|
175
|
+
"path": folder_path,
|
|
176
|
+
"index": index,
|
|
177
|
+
}
|
|
178
|
+
)
|
|
179
|
+
if not folders:
|
|
180
|
+
folders.append(
|
|
181
|
+
{
|
|
182
|
+
"name": self.workspace_root.name,
|
|
183
|
+
"path": self.workspace_root,
|
|
184
|
+
"index": 0,
|
|
185
|
+
}
|
|
186
|
+
)
|
|
187
|
+
return folders
|
|
188
|
+
|
|
189
|
+
def _extract_render_src(self, text: str, base: Path) -> Path:
|
|
190
|
+
match = re.search(
|
|
191
|
+
r'<render\b[^>]*\bsrc\s*=\s*["\']([^"\']+)["\']',
|
|
192
|
+
text,
|
|
193
|
+
re.IGNORECASE,
|
|
194
|
+
)
|
|
195
|
+
if not match:
|
|
196
|
+
raise ValueError(f"入口文件缺少 <render src=...>: {base}")
|
|
197
|
+
return self._resolve_path(match.group(1), base.parent)
|
|
198
|
+
|
|
199
|
+
def _extract_theme_src(self, text: str, base: Path) -> Optional[Path]:
|
|
200
|
+
match = re.search(
|
|
201
|
+
r'<render\b[^>]*\btheme\s*=\s*["\']([^"\']+)["\']',
|
|
202
|
+
text,
|
|
203
|
+
re.IGNORECASE,
|
|
204
|
+
)
|
|
205
|
+
if not match:
|
|
206
|
+
return None
|
|
207
|
+
return self._resolve_path(match.group(1), base.parent)
|
|
208
|
+
|
|
209
|
+
def _collect_markdown_tree(self, root: Path) -> List[Path]:
|
|
210
|
+
result: List[Path] = []
|
|
211
|
+
stack = [root.resolve()]
|
|
212
|
+
seen: Set[Path] = set()
|
|
213
|
+
|
|
214
|
+
while stack:
|
|
215
|
+
current = stack.pop()
|
|
216
|
+
if current in seen:
|
|
217
|
+
continue
|
|
218
|
+
seen.add(current)
|
|
219
|
+
if not current.exists():
|
|
220
|
+
continue
|
|
221
|
+
result.append(current)
|
|
222
|
+
text = current.read_text(encoding="utf-8")
|
|
223
|
+
for raw in self.IMPORT_PATTERN.findall(text):
|
|
224
|
+
imported = self._resolve_path(raw, current.parent)
|
|
225
|
+
if imported.suffix.lower() in {".md", ".markdown"}:
|
|
226
|
+
stack.append(imported)
|
|
227
|
+
return result
|
|
228
|
+
|
|
229
|
+
def _package_markdown(self, md_path: Path, package_root: Path) -> str:
|
|
230
|
+
package_rel = self._document_package_path(md_path)
|
|
231
|
+
target = package_root / package_rel
|
|
232
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
233
|
+
|
|
234
|
+
text = md_path.read_text(encoding="utf-8")
|
|
235
|
+
asset_shas: List[str] = []
|
|
236
|
+
|
|
237
|
+
def replace_image(match: re.Match) -> str:
|
|
238
|
+
alt = match.group(1)
|
|
239
|
+
raw_url = match.group(2).strip().strip('"').strip("'")
|
|
240
|
+
title = ""
|
|
241
|
+
url = raw_url
|
|
242
|
+
if " " in raw_url:
|
|
243
|
+
url, title_part = raw_url.split(" ", 1)
|
|
244
|
+
title = " " + title_part
|
|
245
|
+
if self._is_remote_url(url):
|
|
246
|
+
return match.group(0)
|
|
247
|
+
|
|
248
|
+
resource_path = self._resolve_path(url, md_path.parent)
|
|
249
|
+
if not resource_path.exists():
|
|
250
|
+
return match.group(0)
|
|
251
|
+
|
|
252
|
+
sha, package_resource = self._add_resource(resource_path, package_root)
|
|
253
|
+
if sha not in asset_shas:
|
|
254
|
+
asset_shas.append(sha)
|
|
255
|
+
asset_name = Path(package_resource).name
|
|
256
|
+
return f""
|
|
257
|
+
|
|
258
|
+
def replace_html_img(match: re.Match) -> str:
|
|
259
|
+
raw = match.group(0)
|
|
260
|
+
url = match.group(1)
|
|
261
|
+
if self._is_remote_url(url):
|
|
262
|
+
return raw
|
|
263
|
+
resource_path = self._resolve_path(url, md_path.parent)
|
|
264
|
+
if not resource_path.exists():
|
|
265
|
+
return raw
|
|
266
|
+
sha, package_resource = self._add_resource(resource_path, package_root)
|
|
267
|
+
if sha not in asset_shas:
|
|
268
|
+
asset_shas.append(sha)
|
|
269
|
+
asset_name = Path(package_resource).name
|
|
270
|
+
return re.sub(
|
|
271
|
+
r'(src\s*=\s*["\'])([^"\']+)(["\'])',
|
|
272
|
+
rf'\1./assets/{asset_name}\3',
|
|
273
|
+
raw,
|
|
274
|
+
count=1,
|
|
275
|
+
flags=re.IGNORECASE,
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
def replace_import(match: re.Match) -> str:
|
|
279
|
+
raw = match.group(1)
|
|
280
|
+
imported = self._resolve_path(raw, md_path.parent)
|
|
281
|
+
if not imported.exists():
|
|
282
|
+
return match.group(0)
|
|
283
|
+
package_import = self._document_package_path(imported)
|
|
284
|
+
relative = self._relative_posix(
|
|
285
|
+
Path(package_rel).parent,
|
|
286
|
+
Path(package_import),
|
|
287
|
+
)
|
|
288
|
+
quote = '"' if '"' in match.group(0) else "'"
|
|
289
|
+
# 保留原引号风格
|
|
290
|
+
original = match.group(0)
|
|
291
|
+
if "'" in original and '"' not in original:
|
|
292
|
+
quote = "'"
|
|
293
|
+
else:
|
|
294
|
+
quote = '"'
|
|
295
|
+
return f"@import {quote}{relative}{quote}"
|
|
296
|
+
|
|
297
|
+
text = self.IMAGE_PATTERN.sub(replace_image, text)
|
|
298
|
+
text = self.HTML_IMG_PATTERN.sub(replace_html_img, text)
|
|
299
|
+
text = self.IMPORT_PATTERN.sub(replace_import, text)
|
|
300
|
+
|
|
301
|
+
target.write_text(text, encoding="utf-8")
|
|
302
|
+
original_key = self._to_posix(md_path)
|
|
303
|
+
self._documents[original_key] = package_rel
|
|
304
|
+
self._document_assets[package_rel] = asset_shas
|
|
305
|
+
self._path_to_package[md_path.resolve()] = package_rel
|
|
306
|
+
return package_rel
|
|
307
|
+
|
|
308
|
+
def _package_theme(self, theme_path: Path, package_root: Path) -> str:
|
|
309
|
+
package_rel = f"themes/{theme_path.name}"
|
|
310
|
+
target = package_root / package_rel
|
|
311
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
312
|
+
shutil.copy2(theme_path, target)
|
|
313
|
+
return package_rel
|
|
314
|
+
|
|
315
|
+
def _package_default_template(self, package_root: Path) -> Optional[str]:
|
|
316
|
+
template = (
|
|
317
|
+
Path(__file__).resolve().parent
|
|
318
|
+
/ "render_option"
|
|
319
|
+
/ "referance.docx"
|
|
320
|
+
)
|
|
321
|
+
if not template.exists():
|
|
322
|
+
return None
|
|
323
|
+
sha, package_rel = self._add_resource(
|
|
324
|
+
template,
|
|
325
|
+
package_root,
|
|
326
|
+
preferred_dir="templates",
|
|
327
|
+
preferred_name="referance.docx",
|
|
328
|
+
)
|
|
329
|
+
return package_rel
|
|
330
|
+
|
|
331
|
+
def _add_resource(
|
|
332
|
+
self,
|
|
333
|
+
source: Path,
|
|
334
|
+
package_root: Path,
|
|
335
|
+
preferred_dir: str = "resources",
|
|
336
|
+
preferred_name: Optional[str] = None,
|
|
337
|
+
) -> Tuple[str, str]:
|
|
338
|
+
source = source.resolve()
|
|
339
|
+
data = source.read_bytes()
|
|
340
|
+
sha = hashlib.sha256(data).hexdigest()
|
|
341
|
+
if sha in self._resources:
|
|
342
|
+
return sha, self._resources[sha]["path"]
|
|
343
|
+
|
|
344
|
+
suffix = source.suffix.lower() or ""
|
|
345
|
+
file_name = preferred_name or f"{sha}{suffix}"
|
|
346
|
+
if preferred_dir == "resources" and preferred_name is None:
|
|
347
|
+
file_name = f"{sha}{suffix}"
|
|
348
|
+
|
|
349
|
+
package_rel = f"{preferred_dir}/{file_name}"
|
|
350
|
+
target = package_root / package_rel
|
|
351
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
352
|
+
if not target.exists():
|
|
353
|
+
target.write_bytes(data)
|
|
354
|
+
|
|
355
|
+
self._resources[sha] = {
|
|
356
|
+
"path": package_rel,
|
|
357
|
+
"sha256": sha,
|
|
358
|
+
"name": source.name,
|
|
359
|
+
"original": self._to_posix(source),
|
|
360
|
+
}
|
|
361
|
+
self._sha_to_path[sha] = source
|
|
362
|
+
return sha, package_rel
|
|
363
|
+
|
|
364
|
+
def _document_package_path(self, path: Path) -> str:
|
|
365
|
+
path = path.resolve()
|
|
366
|
+
if path in self._path_to_package:
|
|
367
|
+
return self._path_to_package[path]
|
|
368
|
+
|
|
369
|
+
folder = self._find_workspace_folder(path)
|
|
370
|
+
if folder is not None:
|
|
371
|
+
relative = path.relative_to(folder["path"])
|
|
372
|
+
package_rel = self._to_posix(
|
|
373
|
+
Path("documents")
|
|
374
|
+
/ f"{folder['index']}-{folder['name']}"
|
|
375
|
+
/ relative
|
|
376
|
+
)
|
|
377
|
+
else:
|
|
378
|
+
digest = hashlib.sha256(self._to_posix(path).encode("utf-8")).hexdigest()[:12]
|
|
379
|
+
package_rel = self._to_posix(
|
|
380
|
+
Path("documents") / "_external" / digest / path.name
|
|
381
|
+
)
|
|
382
|
+
self._path_to_package[path] = package_rel
|
|
383
|
+
return package_rel
|
|
384
|
+
|
|
385
|
+
def _find_workspace_folder(self, path: Path) -> Optional[dict]:
|
|
386
|
+
path = path.resolve()
|
|
387
|
+
matched = None
|
|
388
|
+
for folder in self.folders:
|
|
389
|
+
root = folder["path"]
|
|
390
|
+
try:
|
|
391
|
+
path.relative_to(root)
|
|
392
|
+
except ValueError:
|
|
393
|
+
continue
|
|
394
|
+
if matched is None or len(str(root)) > len(str(matched["path"])):
|
|
395
|
+
matched = folder
|
|
396
|
+
return matched
|
|
397
|
+
|
|
398
|
+
def _rewrite_entry(
|
|
399
|
+
self,
|
|
400
|
+
entry_text: str,
|
|
401
|
+
entry_path: Path,
|
|
402
|
+
document_src: Path,
|
|
403
|
+
theme_package_path: Optional[str],
|
|
404
|
+
) -> str:
|
|
405
|
+
document_package = self._document_package_path(document_src)
|
|
406
|
+
|
|
407
|
+
def replace_src(match: re.Match) -> str:
|
|
408
|
+
return f"{match.group(1)}{document_package}{match.group(3)}"
|
|
409
|
+
|
|
410
|
+
text = self.RENDER_SRC_PATTERN.sub(replace_src, entry_text, count=1)
|
|
411
|
+
|
|
412
|
+
if theme_package_path:
|
|
413
|
+
|
|
414
|
+
def replace_theme(match: re.Match) -> str:
|
|
415
|
+
return f"{match.group(1)}{theme_package_path}{match.group(3)}"
|
|
416
|
+
|
|
417
|
+
if re.search(r"\btheme\s*=", text, re.IGNORECASE):
|
|
418
|
+
text = self.THEME_SRC_PATTERN.sub(replace_theme, text, count=1)
|
|
419
|
+
else:
|
|
420
|
+
text = re.sub(
|
|
421
|
+
r"(<render\b)",
|
|
422
|
+
rf'\1 theme="{theme_package_path}"',
|
|
423
|
+
text,
|
|
424
|
+
count=1,
|
|
425
|
+
flags=re.IGNORECASE,
|
|
426
|
+
)
|
|
427
|
+
return text
|
|
428
|
+
|
|
429
|
+
def _resolve_path(self, raw: str, base_dir: Path) -> Path:
|
|
430
|
+
value = raw.strip().strip('"').strip("'")
|
|
431
|
+
path = Path(value)
|
|
432
|
+
if path.is_absolute():
|
|
433
|
+
return path.resolve()
|
|
434
|
+
return (base_dir / path).resolve()
|
|
435
|
+
|
|
436
|
+
@staticmethod
|
|
437
|
+
def _is_remote_url(url: str) -> bool:
|
|
438
|
+
return bool(re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*://", url))
|
|
439
|
+
|
|
440
|
+
@staticmethod
|
|
441
|
+
def _to_posix(path: Path | str) -> str:
|
|
442
|
+
return Path(path).as_posix()
|
|
443
|
+
|
|
444
|
+
@staticmethod
|
|
445
|
+
def _relative_posix(from_dir: Path, to_path: Path) -> str:
|
|
446
|
+
from_parts = [part for part in Path(from_dir).as_posix().split("/") if part]
|
|
447
|
+
to_parts = [part for part in Path(to_path).as_posix().split("/") if part]
|
|
448
|
+
common = 0
|
|
449
|
+
for left, right in zip(from_parts, to_parts):
|
|
450
|
+
if left != right:
|
|
451
|
+
break
|
|
452
|
+
common += 1
|
|
453
|
+
ups = [".."] * (len(from_parts) - common)
|
|
454
|
+
downs = to_parts[common:]
|
|
455
|
+
parts = ups + downs
|
|
456
|
+
if not parts:
|
|
457
|
+
return "./" + Path(to_path).name
|
|
458
|
+
result = "/".join(parts)
|
|
459
|
+
if not result.startswith("."):
|
|
460
|
+
result = "./" + result
|
|
461
|
+
return result
|
|
462
|
+
|
|
463
|
+
@staticmethod
|
|
464
|
+
def _zip_directory(source_dir: Path, output_path: Path) -> None:
|
|
465
|
+
if output_path.exists():
|
|
466
|
+
output_path.unlink()
|
|
467
|
+
with zipfile.ZipFile(
|
|
468
|
+
output_path,
|
|
469
|
+
mode="w",
|
|
470
|
+
compression=zipfile.ZIP_DEFLATED,
|
|
471
|
+
) as zf:
|
|
472
|
+
for file_path in sorted(source_dir.rglob("*")):
|
|
473
|
+
if file_path.is_file():
|
|
474
|
+
arcname = file_path.relative_to(source_dir).as_posix()
|
|
475
|
+
zf.write(file_path, arcname)
|