qparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qmd_cli.py +262 -0
- qparse/__init__.py +16 -0
- qparse/docx_render/__init__.py +3 -0
- qparse/docx_render/docx_render.py +68 -0
- qparse/docx_render/write_buffer.py +412 -0
- qparse/markdwon_document/__init__.py +28 -0
- qparse/markdwon_document/analysis_document.py +40 -0
- qparse/markdwon_document/base_document.py +87 -0
- qparse/markdwon_document/markdwon_document.py +16 -0
- qparse/markdwon_document/nodes/__init__.py +5 -0
- qparse/markdwon_document/nodes/answer_node.py +97 -0
- qparse/markdwon_document/nodes/base_node.py +51 -0
- qparse/markdwon_document/nodes/img_node.py +31 -0
- qparse/markdwon_document/nodes/text_node.py +139 -0
- qparse/markdwon_document/question_document.py +17 -0
- qparse/markdwon_document/stem_document.py +64 -0
- qparse/markdwon_document/table_document.py +126 -0
- qparse/markdwon_loader.py +189 -0
- qparse/markdwon_render.py +123 -0
- qparse/qmd_packer.py +475 -0
- qparse/qmd_unpacker.py +169 -0
- qparse/render_option/__init__.py +40 -0
- qparse/render_option/default_render_option.py +86 -0
- qparse/render_option/default_render_template.md +21 -0
- qparse/render_option/referance.docx +0 -0
- qparse/render_option/theme.json +245 -0
- qparse/utils/__init__.py +3 -0
- qparse/utils/html_full_protector.py +154 -0
- qparse-0.1.0.dist-info/METADATA +454 -0
- qparse-0.1.0.dist-info/RECORD +34 -0
- qparse-0.1.0.dist-info/WHEEL +5 -0
- qparse-0.1.0.dist-info/entry_points.txt +2 -0
- qparse-0.1.0.dist-info/licenses/LICENSE +21 -0
- qparse-0.1.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
from typing import List,Union
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import re
|
|
6
|
+
import pypandoc
|
|
7
|
+
from docxcompose.composer import Composer
|
|
8
|
+
from bs4 import BeautifulSoup
|
|
9
|
+
from docx import Document
|
|
10
|
+
from docx.enum.table import WD_CELL_VERTICAL_ALIGNMENT
|
|
11
|
+
from docx.oxml import OxmlElement
|
|
12
|
+
from docx.oxml.ns import qn
|
|
13
|
+
from docx.enum.text import WD_ALIGN_PARAGRAPH
|
|
14
|
+
from docx.shared import Inches
|
|
15
|
+
|
|
16
|
+
class BaseBuffer():
|
|
17
|
+
|
|
18
|
+
def __init__(self,render):
|
|
19
|
+
self._render = render
|
|
20
|
+
@property
|
|
21
|
+
def referance_path(self)->Path:
|
|
22
|
+
return self._render.referance_path
|
|
23
|
+
def validate_concatenate(self,value:BaseBuffer):
|
|
24
|
+
if type(value) is self.__class__:
|
|
25
|
+
return True
|
|
26
|
+
def create_temp_file_path(self,fext:str)->Path:
|
|
27
|
+
return self._render.create_temp_file_path(fext)
|
|
28
|
+
|
|
29
|
+
def create_docx_file_path(self):
|
|
30
|
+
return self.create_temp_file_path(f".{id(self)}.docx")
|
|
31
|
+
|
|
32
|
+
def write_to_text_file(self):
|
|
33
|
+
fpath= self.create_temp_file_path(".md")
|
|
34
|
+
fpath.write_text(self.text,encoding='utf-8')
|
|
35
|
+
return fpath
|
|
36
|
+
def _restore_outer_blank_lines(self, docx_path):
|
|
37
|
+
lines = self.text.splitlines()
|
|
38
|
+
leading_count = next(
|
|
39
|
+
(index for index, line in enumerate(lines) if line.strip()),
|
|
40
|
+
len(lines)
|
|
41
|
+
)
|
|
42
|
+
trailing_count = next(
|
|
43
|
+
(index for index, line in enumerate(reversed(lines)) if line.strip()),
|
|
44
|
+
len(lines)
|
|
45
|
+
)
|
|
46
|
+
if not leading_count and not trailing_count:
|
|
47
|
+
return
|
|
48
|
+
|
|
49
|
+
document = Document(docx_path)
|
|
50
|
+
# first_paragraph = document.paragraphs[0] if document.paragraphs else None
|
|
51
|
+
# for _ in range(leading_count):
|
|
52
|
+
# paragraph = document.add_paragraph()
|
|
53
|
+
# if first_paragraph is not None:
|
|
54
|
+
# first_paragraph._p.addprevious(paragraph._p)
|
|
55
|
+
for _ in range(trailing_count):
|
|
56
|
+
document.add_paragraph()
|
|
57
|
+
document.save(docx_path)
|
|
58
|
+
|
|
59
|
+
def write_to_docx_file(self):
|
|
60
|
+
temp_md_file_path = self.write_to_text_file()
|
|
61
|
+
fpath= self.create_temp_file_path(".docx")
|
|
62
|
+
pypandoc.convert_file(
|
|
63
|
+
source_file=temp_md_file_path.as_posix(),
|
|
64
|
+
to="docx",
|
|
65
|
+
format="markdown+tex_math_dollars+hard_line_breaks",
|
|
66
|
+
outputfile=fpath.as_posix(),
|
|
67
|
+
extra_args=[
|
|
68
|
+
"--mathjax",
|
|
69
|
+
"--no-highlight",
|
|
70
|
+
f"--reference-doc={Path(self.referance_path).as_posix()}"
|
|
71
|
+
]
|
|
72
|
+
)
|
|
73
|
+
self._restore_outer_blank_lines(fpath)
|
|
74
|
+
return fpath
|
|
75
|
+
class MarkdwonTextBuffer(BaseBuffer):
|
|
76
|
+
_allow_type = "markdwonText"
|
|
77
|
+
def __init__(self, render):
|
|
78
|
+
super().__init__(render)
|
|
79
|
+
self._text_list:List[str] = []
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def text_list(self):
|
|
84
|
+
return [*self._text_list]
|
|
85
|
+
@property
|
|
86
|
+
def type(self):
|
|
87
|
+
return 'markdwonText'
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def text(self):
|
|
91
|
+
return "".join(self._text_list)
|
|
92
|
+
|
|
93
|
+
def _adjust_tables(self, docx_path):
|
|
94
|
+
document = Document(docx_path)
|
|
95
|
+
options = self._render._mdoc.render_option.get("docx", {}).get(
|
|
96
|
+
"markdwon-table", {}
|
|
97
|
+
)
|
|
98
|
+
width_percent = float(options.get("width_percent", 100))
|
|
99
|
+
style = options.get("style")
|
|
100
|
+
header_bold = options.get("header_bold", True)
|
|
101
|
+
header_shading = options.get("header_shading")
|
|
102
|
+
border_color = options.get("border_color")
|
|
103
|
+
alignment = options.get("cell_alignment", "center")
|
|
104
|
+
|
|
105
|
+
for table in document.tables:
|
|
106
|
+
if style:
|
|
107
|
+
try:
|
|
108
|
+
table.style = style
|
|
109
|
+
except KeyError:
|
|
110
|
+
pass
|
|
111
|
+
table.autofit = False
|
|
112
|
+
if border_color:
|
|
113
|
+
table_properties = table._tbl.tblPr
|
|
114
|
+
borders = table_properties.find(qn("w:tblBorders"))
|
|
115
|
+
if borders is None:
|
|
116
|
+
borders = OxmlElement("w:tblBorders")
|
|
117
|
+
table_properties.append(borders)
|
|
118
|
+
for border_name in ("top", "left", "bottom", "right", "insideH", "insideV"):
|
|
119
|
+
border = borders.find(qn(f"w:{border_name}"))
|
|
120
|
+
if border is None:
|
|
121
|
+
border = OxmlElement(f"w:{border_name}")
|
|
122
|
+
borders.append(border)
|
|
123
|
+
border.set(qn("w:val"), "single")
|
|
124
|
+
border.set(qn("w:sz"), "4")
|
|
125
|
+
border.set(qn("w:space"), "0")
|
|
126
|
+
border.set(qn("w:color"), border_color)
|
|
127
|
+
table_width = sum(
|
|
128
|
+
section.page_width - section.left_margin - section.right_margin
|
|
129
|
+
for section in document.sections
|
|
130
|
+
) / max(1, len(document.sections))
|
|
131
|
+
table_width = int(table_width * width_percent / 100)
|
|
132
|
+
for row_index, row in enumerate(table.rows):
|
|
133
|
+
row_width = max(1, len(row.cells))
|
|
134
|
+
for cell in row.cells:
|
|
135
|
+
cell.width = Inches(table_width / 914400 / row_width)
|
|
136
|
+
cell.vertical_alignment = WD_CELL_VERTICAL_ALIGNMENT.CENTER
|
|
137
|
+
if alignment == "center":
|
|
138
|
+
for paragraph in cell.paragraphs:
|
|
139
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
140
|
+
if row_index == 0:
|
|
141
|
+
for paragraph in cell.paragraphs:
|
|
142
|
+
for run in paragraph.runs:
|
|
143
|
+
run.bold = header_bold
|
|
144
|
+
if header_shading:
|
|
145
|
+
properties = cell._tc.get_or_add_tcPr()
|
|
146
|
+
shading = properties.find(qn("w:shd"))
|
|
147
|
+
if shading is None:
|
|
148
|
+
shading = OxmlElement("w:shd")
|
|
149
|
+
properties.append(shading)
|
|
150
|
+
shading.set(qn("w:fill"), header_shading)
|
|
151
|
+
|
|
152
|
+
document.save(docx_path)
|
|
153
|
+
|
|
154
|
+
def write_to_docx_file(self):
|
|
155
|
+
temp_md_file_path = self.write_to_text_file()
|
|
156
|
+
fpath= self.create_temp_file_path(".docx")
|
|
157
|
+
pypandoc.convert_file(
|
|
158
|
+
source_file=temp_md_file_path.as_posix(),
|
|
159
|
+
to="docx",
|
|
160
|
+
format="markdown+tex_math_dollars+hard_line_breaks",
|
|
161
|
+
outputfile=fpath.as_posix(),
|
|
162
|
+
extra_args=[
|
|
163
|
+
"--mathjax",
|
|
164
|
+
"--no-highlight",
|
|
165
|
+
f"--reference-doc={Path(self.referance_path).as_posix()}"
|
|
166
|
+
]
|
|
167
|
+
)
|
|
168
|
+
self._restore_outer_blank_lines(fpath)
|
|
169
|
+
self._adjust_tables(fpath)
|
|
170
|
+
return fpath
|
|
171
|
+
|
|
172
|
+
def create(self,type:str,text:str,*kws):
|
|
173
|
+
if not type==self._allow_type:
|
|
174
|
+
raise TypeError(detail={"meg":"type={} not allow".format(type)})
|
|
175
|
+
self._text_list.append(text)
|
|
176
|
+
return self
|
|
177
|
+
|
|
178
|
+
def concatenate(self, value:MarkdwonTextBuffer):
|
|
179
|
+
if super().validate_concatenate(value):
|
|
180
|
+
self._text_list.extend(value.text_list)
|
|
181
|
+
return self
|
|
182
|
+
return False
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class HtmlTextBuffer(MarkdwonTextBuffer):
|
|
187
|
+
_allow_type = "htmlText"
|
|
188
|
+
def __init__(self, render):
|
|
189
|
+
super().__init__(render)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def append(self, text):
|
|
193
|
+
if len(self._text_list)==0:
|
|
194
|
+
return super().append(text)
|
|
195
|
+
raise "{}已经存在数据".format(self)
|
|
196
|
+
|
|
197
|
+
def write_to_text_file(self):
|
|
198
|
+
fpath= self.create_temp_file_path(".html")
|
|
199
|
+
fpath.write_text(self.text,encoding='utf-8')
|
|
200
|
+
return fpath
|
|
201
|
+
|
|
202
|
+
def write_to_text_file(self):
|
|
203
|
+
fpath= self.create_temp_file_path(".html")
|
|
204
|
+
fpath.write_text(self.text,encoding='utf-8')
|
|
205
|
+
return fpath
|
|
206
|
+
|
|
207
|
+
def write_to_docx_file(self):
|
|
208
|
+
temp_html_file_path = self.write_to_text_file()
|
|
209
|
+
fpath = self.create_docx_file_path()
|
|
210
|
+
if fpath.exists():
|
|
211
|
+
fpath.unlink()
|
|
212
|
+
pypandoc.convert_file(
|
|
213
|
+
source_file=temp_html_file_path.as_posix(),
|
|
214
|
+
to="docx",
|
|
215
|
+
format="html+tex_math_dollars",
|
|
216
|
+
outputfile=fpath.as_posix(),
|
|
217
|
+
extra_args=["--mathjax", "--no-highlight"]
|
|
218
|
+
)
|
|
219
|
+
return fpath
|
|
220
|
+
class TableHtmlTextBuffer(HtmlTextBuffer):
|
|
221
|
+
_allow_type = "tableHtmlText"
|
|
222
|
+
|
|
223
|
+
def __init__(self, render):
|
|
224
|
+
super().__init__(render)
|
|
225
|
+
|
|
226
|
+
def concatenate(self, value):
|
|
227
|
+
# 禁止合并
|
|
228
|
+
return False
|
|
229
|
+
|
|
230
|
+
def _column_widths(self):
|
|
231
|
+
soup = BeautifulSoup(self.text, "html.parser")
|
|
232
|
+
table = soup.find("table")
|
|
233
|
+
if table is None:
|
|
234
|
+
return []
|
|
235
|
+
|
|
236
|
+
rows = table.find_all("tr")
|
|
237
|
+
column_count = max(
|
|
238
|
+
(len(row.find_all(["td", "th"], recursive=False)) for row in rows),
|
|
239
|
+
default=0
|
|
240
|
+
)
|
|
241
|
+
if column_count == 0:
|
|
242
|
+
return []
|
|
243
|
+
|
|
244
|
+
widths = [None] * column_count
|
|
245
|
+
for row in rows:
|
|
246
|
+
cells = row.find_all(["td", "th"], recursive=False)
|
|
247
|
+
for index, cell in enumerate(cells):
|
|
248
|
+
if widths[index] is None and cell.has_attr("width"):
|
|
249
|
+
value = re.search(r"[0-9]+(?:[.][0-9]+)?", cell["width"])
|
|
250
|
+
if value:
|
|
251
|
+
widths[index] = float(value.group())
|
|
252
|
+
|
|
253
|
+
missing = [index for index, width in enumerate(widths) if width is None]
|
|
254
|
+
remaining = max(0.0, 100.0 - sum(width or 0.0 for width in widths))
|
|
255
|
+
default_width = remaining / len(missing) if missing else 0.0
|
|
256
|
+
return [width if width is not None else default_width for width in widths]
|
|
257
|
+
|
|
258
|
+
def _adjust_docx(self, docx_path):
|
|
259
|
+
document = Document(docx_path)
|
|
260
|
+
widths = self._column_widths()
|
|
261
|
+
for table in document.tables:
|
|
262
|
+
if not table.columns:
|
|
263
|
+
continue
|
|
264
|
+
table.autofit = False
|
|
265
|
+
table_width = sum(
|
|
266
|
+
section.page_width - section.left_margin - section.right_margin
|
|
267
|
+
for section in document.sections
|
|
268
|
+
) / max(1, len(document.sections))
|
|
269
|
+
for index, column in enumerate(table.columns):
|
|
270
|
+
width = widths[index] if index < len(widths) else 100 / len(table.columns)
|
|
271
|
+
column_width = Inches(table_width / 914400 * width / 100)
|
|
272
|
+
for cell in column.cells:
|
|
273
|
+
cell.width = column_width
|
|
274
|
+
cell.vertical_alignment = WD_CELL_VERTICAL_ALIGNMENT.CENTER
|
|
275
|
+
for paragraph in cell.paragraphs:
|
|
276
|
+
for run in paragraph.runs:
|
|
277
|
+
for shape in run._element.xpath(".//wp:inline"):
|
|
278
|
+
extent = shape.find("{http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing}extent")
|
|
279
|
+
if extent is not None:
|
|
280
|
+
extent.set("cx", str(int(column_width.inches * 914400 * 0.95)))
|
|
281
|
+
extent.set("cy", str(int(int(extent.get("cy")) * 0.95)))
|
|
282
|
+
if any(run._element.xpath(".//wp:inline") for run in paragraph.runs):
|
|
283
|
+
paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
|
284
|
+
document.save(docx_path)
|
|
285
|
+
|
|
286
|
+
def write_to_docx_file(self):
|
|
287
|
+
temp_html_file_path = self.write_to_text_file()
|
|
288
|
+
fpath = self.create_docx_file_path()
|
|
289
|
+
pypandoc.convert_file(
|
|
290
|
+
source_file=temp_html_file_path.as_posix(),
|
|
291
|
+
to="docx",
|
|
292
|
+
format="html+tex_math_dollars",
|
|
293
|
+
outputfile=fpath.as_posix(),
|
|
294
|
+
extra_args=[
|
|
295
|
+
"--mathjax",
|
|
296
|
+
"--no-highlight",
|
|
297
|
+
f"--reference-doc={Path(self.referance_path).as_posix()}"
|
|
298
|
+
]
|
|
299
|
+
)
|
|
300
|
+
self._adjust_docx(fpath)
|
|
301
|
+
return fpath
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
class DocumentBuffer(BaseBuffer):
|
|
306
|
+
_allow_type = "document"
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def __init__(self, render):
|
|
312
|
+
super().__init__(render)
|
|
313
|
+
|
|
314
|
+
self._handles = {
|
|
315
|
+
MarkdwonTextBuffer._allow_type:MarkdwonTextBuffer,
|
|
316
|
+
HtmlTextBuffer._allow_type:MarkdwonTextBuffer,
|
|
317
|
+
TableHtmlTextBuffer._allow_type:TableHtmlTextBuffer,
|
|
318
|
+
DocumentBuffer._allow_type:DocumentBuffer,
|
|
319
|
+
}
|
|
320
|
+
self.keep_together = False
|
|
321
|
+
self._children:List[Union[BaseBuffer,HtmlTextBuffer,MarkdwonTextBuffer,TableHtmlTextBuffer]] = []
|
|
322
|
+
@property
|
|
323
|
+
def last_node(self)->BaseBuffer:
|
|
324
|
+
|
|
325
|
+
if len(self._children)>0:
|
|
326
|
+
return self._children[-1]
|
|
327
|
+
return None
|
|
328
|
+
@property
|
|
329
|
+
def children(self):
|
|
330
|
+
return [*self._children]
|
|
331
|
+
|
|
332
|
+
def validate_concatenate(self, value:DocumentBuffer):
|
|
333
|
+
return super().validate_concatenate(value) and value.keep_together==self.keep_together
|
|
334
|
+
|
|
335
|
+
def concatenate(self,value:DocumentBuffer):
|
|
336
|
+
if self.validate_concatenate(value):
|
|
337
|
+
for c in value.children:
|
|
338
|
+
self.append(c)
|
|
339
|
+
return self
|
|
340
|
+
return None
|
|
341
|
+
|
|
342
|
+
def append(self,value:Union[DocumentBuffer,HtmlTextBuffer,TableHtmlTextBuffer,BaseBuffer]):
|
|
343
|
+
if self.last_node:
|
|
344
|
+
if self.last_node.concatenate(value=value):
|
|
345
|
+
return None
|
|
346
|
+
self._children.append(value)
|
|
347
|
+
|
|
348
|
+
def create(self,type:str,keep_together:bool,children:List[dict]):
|
|
349
|
+
if not self._allow_type==type:
|
|
350
|
+
raise TypeError()
|
|
351
|
+
self.keep_together = keep_together
|
|
352
|
+
|
|
353
|
+
for child in children:
|
|
354
|
+
node_type = child['type']
|
|
355
|
+
if handle := self._handles.get(node_type,None):
|
|
356
|
+
node = handle(self._render).create(**child)
|
|
357
|
+
self.append(node)
|
|
358
|
+
return self
|
|
359
|
+
@property
|
|
360
|
+
def text(self):
|
|
361
|
+
|
|
362
|
+
return "".join([c.text for c in self.children])
|
|
363
|
+
def _keep_content_together(self, docx_path):
|
|
364
|
+
document = Document(docx_path)
|
|
365
|
+
paragraphs = document.paragraphs
|
|
366
|
+
for paragraph in paragraphs[:-1]:
|
|
367
|
+
paragraph.paragraph_format.keep_with_next = True
|
|
368
|
+
|
|
369
|
+
for table in document.tables:
|
|
370
|
+
for row in table.rows:
|
|
371
|
+
row_properties = row._tr.get_or_add_trPr()
|
|
372
|
+
cant_split = row_properties.find(
|
|
373
|
+
"{http://schemas.openxmlformats.org/wordprocessingml/2006/main}cantSplit"
|
|
374
|
+
)
|
|
375
|
+
if cant_split is None:
|
|
376
|
+
row_properties.append(
|
|
377
|
+
OxmlElement("w:cantSplit")
|
|
378
|
+
)
|
|
379
|
+
for cell in row.cells:
|
|
380
|
+
for paragraph in cell.paragraphs:
|
|
381
|
+
paragraph.paragraph_format.keep_with_next = True
|
|
382
|
+
|
|
383
|
+
document.save(docx_path)
|
|
384
|
+
|
|
385
|
+
def write_to_docx_file(self):
|
|
386
|
+
docx_paths = [c.write_to_docx_file() for c in self.children]
|
|
387
|
+
if not docx_paths:
|
|
388
|
+
return None
|
|
389
|
+
output_path = self.create_temp_file_path(".docx")
|
|
390
|
+
|
|
391
|
+
composer = Composer(Document(docx_paths[0]))
|
|
392
|
+
for docx_path in docx_paths[1:]:
|
|
393
|
+
composer.append(Document(docx_path))
|
|
394
|
+
composer.save(output_path)
|
|
395
|
+
if self.keep_together:
|
|
396
|
+
self._keep_content_together(output_path)
|
|
397
|
+
return output_path
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from .nodes import *
|
|
3
|
+
|
|
4
|
+
from .base_document import BaseDocument
|
|
5
|
+
from .markdwon_document import MarkdwonDocument
|
|
6
|
+
from .question_document import QuestionDocument
|
|
7
|
+
from .analysis_document import AnalysisDocument
|
|
8
|
+
from .stem_document import StemDocument
|
|
9
|
+
from .table_document import TableDocument
|
|
10
|
+
BaseDocument.register_node_parser('markdwon',MarkdwonDocument)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
BaseDocument.register_node_parser('choise',ChioseNode)
|
|
15
|
+
BaseDocument.register_node_parser('multiple-choice',MultipleChoiceNode)
|
|
16
|
+
BaseDocument.register_node_parser('blank',BlankNode)
|
|
17
|
+
BaseDocument.register_node_parser('answer',AnswerNode)
|
|
18
|
+
BaseDocument.register_node_parser('text',TextNode)
|
|
19
|
+
BaseDocument.register_node_parser('br',BrNode)
|
|
20
|
+
BaseDocument.register_node_parser('first-text-visible',FirstTextVisibleNode)
|
|
21
|
+
BaseDocument.register_node_parser('img',ImgNode)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
BaseDocument.register_node_parser('question',QuestionDocument)
|
|
25
|
+
BaseDocument.register_node_parser('analysis',AnalysisDocument)
|
|
26
|
+
BaseDocument.register_node_parser('stem',StemDocument)
|
|
27
|
+
|
|
28
|
+
BaseDocument.register_node_parser('table',TableDocument)
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Dict,Union
|
|
4
|
+
from .base_document import BaseDocument
|
|
5
|
+
from qparse.utils.html_full_protector import HtmlFullProtector
|
|
6
|
+
from qparse.render_option import SafeGetRenderOption
|
|
7
|
+
class AnalysisDocument(BaseDocument):
|
|
8
|
+
def __init__(self, node, render_option):
|
|
9
|
+
self.show = SafeGetRenderOption(option=render_option,path=['question','analysis','show'] ,default=False)
|
|
10
|
+
super().__init__(node, render_option)
|
|
11
|
+
htmlParser = HtmlFullProtector()
|
|
12
|
+
@classmethod
|
|
13
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
14
|
+
|
|
15
|
+
if node.name=="analysis":
|
|
16
|
+
# return cls(node=node,render_option=render_option)
|
|
17
|
+
for n in cls._iter_nodes(node):
|
|
18
|
+
if isinstance(n,element.NavigableString):
|
|
19
|
+
text = cls.htmlParser.decode_to_unicode(n.get_text()).strip()
|
|
20
|
+
if text:
|
|
21
|
+
# s = element.NavigableString()
|
|
22
|
+
tag = element.Tag(name="first-text-visible")
|
|
23
|
+
tag.append(n.get_text())
|
|
24
|
+
tag.attrs['include-by']= "analysis"
|
|
25
|
+
n.replace_with(tag)
|
|
26
|
+
break
|
|
27
|
+
return cls(node=node,render_option=render_option)
|
|
28
|
+
return None
|
|
29
|
+
|
|
30
|
+
def dump_markdwon(self):
|
|
31
|
+
if self.show:
|
|
32
|
+
text= super().dump_markdwon()
|
|
33
|
+
|
|
34
|
+
return text.strip("\n")+"\n"
|
|
35
|
+
return ''
|
|
36
|
+
def dump_docx_buffer(self):
|
|
37
|
+
if self.show:
|
|
38
|
+
return super().dump_docx_buffer()
|
|
39
|
+
return []
|
|
40
|
+
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Dict,Union,List
|
|
4
|
+
from qparse.utils.html_full_protector import HtmlFullProtector
|
|
5
|
+
import copy
|
|
6
|
+
class BaseDocument():
|
|
7
|
+
_registry: Dict[str, type["BaseDocument"]] = {}
|
|
8
|
+
htmlParser = HtmlFullProtector()
|
|
9
|
+
def __init__(self,node:element.Tag,render_option:dict):
|
|
10
|
+
|
|
11
|
+
self._soup = node
|
|
12
|
+
self._render_option = render_option
|
|
13
|
+
self.children= []
|
|
14
|
+
self.create_document()
|
|
15
|
+
@classmethod
|
|
16
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
17
|
+
return None
|
|
18
|
+
|
|
19
|
+
@classmethod
|
|
20
|
+
def register_node_parser(cls,name:str,sub_cls):
|
|
21
|
+
cls._registry[name] = sub_cls
|
|
22
|
+
|
|
23
|
+
def create_document(self):
|
|
24
|
+
stack = [*self._soup]
|
|
25
|
+
|
|
26
|
+
while len(stack):
|
|
27
|
+
node = stack.pop(0)
|
|
28
|
+
|
|
29
|
+
parsed = False
|
|
30
|
+
for name,sub_cls in self._registry.items():
|
|
31
|
+
if mDoc := sub_cls.create_by_node(node=node,render_option=self._render_option):
|
|
32
|
+
self.children.append(mDoc)
|
|
33
|
+
# print(mDoc,name,self.__class__.__name__)
|
|
34
|
+
parsed = True
|
|
35
|
+
break
|
|
36
|
+
|
|
37
|
+
if not parsed:
|
|
38
|
+
|
|
39
|
+
if isinstance(node,element.Tag):
|
|
40
|
+
stack = [*node.children,*stack]
|
|
41
|
+
|
|
42
|
+
elif isinstance(node,element.NavigableString):
|
|
43
|
+
pass
|
|
44
|
+
# else:
|
|
45
|
+
# if isinstance(node,element.Tag):
|
|
46
|
+
# print(node.select_one(".blank"))
|
|
47
|
+
|
|
48
|
+
def dump_markdwon(self):
|
|
49
|
+
nodes = []
|
|
50
|
+
for c in self.children:
|
|
51
|
+
# text = self.htmlParser. c.dump_markdwon()
|
|
52
|
+
# if len(text.strip()) ==0:
|
|
53
|
+
text = c.dump_markdwon()
|
|
54
|
+
nodes.append(text)
|
|
55
|
+
return "".join(nodes)
|
|
56
|
+
def dump_docx_buffer(self):
|
|
57
|
+
nodes = []
|
|
58
|
+
|
|
59
|
+
for c in self.children:
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
nodes.extend(c.dump_docx_buffer())
|
|
63
|
+
|
|
64
|
+
return nodes
|
|
65
|
+
@classmethod
|
|
66
|
+
def _iter_nodes(cls,soup:element.Tag):
|
|
67
|
+
stack = [soup]
|
|
68
|
+
while len(stack):
|
|
69
|
+
node = stack.pop(0)
|
|
70
|
+
yield node
|
|
71
|
+
if isinstance(node,element.Tag):
|
|
72
|
+
stack = [*node,*stack]
|
|
73
|
+
def run(self):
|
|
74
|
+
mark = self.dump_markdwon()
|
|
75
|
+
@property
|
|
76
|
+
def attrs(self):
|
|
77
|
+
return getattr(self.soup,'attrs',{})
|
|
78
|
+
@property
|
|
79
|
+
def soup(self):
|
|
80
|
+
return copy.deepcopy(self._soup)
|
|
81
|
+
@property
|
|
82
|
+
def render_option(self):
|
|
83
|
+
return copy.deepcopy(self._render_option)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Dict,Union
|
|
4
|
+
from .base_document import BaseDocument
|
|
5
|
+
|
|
6
|
+
class MarkdwonDocument(BaseDocument):
|
|
7
|
+
def __init__(self, node, render_option):
|
|
8
|
+
super().__init__(node, render_option)
|
|
9
|
+
@classmethod
|
|
10
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
11
|
+
if node.name=="markdwon":
|
|
12
|
+
return cls(node=node,render_option=render_option)
|
|
13
|
+
return None
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Union
|
|
4
|
+
from qparse.render_option import SafeGetRenderOption,render_template
|
|
5
|
+
from .text_node import TextNode
|
|
6
|
+
|
|
7
|
+
class AnswerNode(TextNode):
|
|
8
|
+
|
|
9
|
+
def __init__(self, node, render_option):
|
|
10
|
+
super().__init__(node, render_option)
|
|
11
|
+
self._show_answer = SafeGetRenderOption(
|
|
12
|
+
self._render_option,path=['question','show_answer'],default=False
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
@property
|
|
16
|
+
def template_key(self):
|
|
17
|
+
k = ':show_answer' if self._show_answer else ":not_show_answer"
|
|
18
|
+
|
|
19
|
+
return k
|
|
20
|
+
@property
|
|
21
|
+
def template(self):
|
|
22
|
+
|
|
23
|
+
return SafeGetRenderOption(
|
|
24
|
+
self._render_option,path=['question','.answer',self.template_key,'template'],default=""
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
@classmethod
|
|
28
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
29
|
+
if isinstance(node,element.Tag) and "answer" in node.attrs.get('class',[]):
|
|
30
|
+
return cls(node=node,render_option=render_option)
|
|
31
|
+
return None
|
|
32
|
+
def dump_markdwon(self):
|
|
33
|
+
|
|
34
|
+
text = super().dump_markdwon()
|
|
35
|
+
|
|
36
|
+
text= render_template(self.template,{"text":text})
|
|
37
|
+
|
|
38
|
+
return text
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ChioseNode(AnswerNode):
|
|
42
|
+
def __init__(self, node, render_option):
|
|
43
|
+
super().__init__(node, render_option)
|
|
44
|
+
@property
|
|
45
|
+
def template(self):
|
|
46
|
+
|
|
47
|
+
return SafeGetRenderOption(
|
|
48
|
+
self._render_option,path=['question','.choise',self.template_key,'template'],default=r"[%text%]"
|
|
49
|
+
)
|
|
50
|
+
@classmethod
|
|
51
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
52
|
+
if isinstance(node,element.Tag) and "chiose" in node.attrs.get('class',[]):
|
|
53
|
+
return cls(node=node,render_option=render_option)
|
|
54
|
+
return None
|
|
55
|
+
|
|
56
|
+
class MultipleChoiceNode(ChioseNode):
|
|
57
|
+
def __init__(self, node, render_option):
|
|
58
|
+
super().__init__(node, render_option)
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def template(self):
|
|
62
|
+
|
|
63
|
+
return SafeGetRenderOption(
|
|
64
|
+
self._render_option,path=['question','.multiple-choice',self.template_key,'template'],default=r"[% text %]"
|
|
65
|
+
)
|
|
66
|
+
@classmethod
|
|
67
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
68
|
+
if isinstance(node,element.Tag) and "multiple-choice" in node.attrs.get('class',[]):
|
|
69
|
+
return cls(node=node,render_option=render_option)
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
class BlankNode(AnswerNode):
|
|
73
|
+
def __init__(self, node, render_option):
|
|
74
|
+
super().__init__(node, render_option)
|
|
75
|
+
|
|
76
|
+
@classmethod
|
|
77
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
78
|
+
|
|
79
|
+
if isinstance(node,element.Tag) and "blank" in node.attrs.get('class',[]):
|
|
80
|
+
|
|
81
|
+
return cls(node=node,render_option=render_option)
|
|
82
|
+
return None
|
|
83
|
+
@property
|
|
84
|
+
def template(self):
|
|
85
|
+
|
|
86
|
+
return SafeGetRenderOption(
|
|
87
|
+
self._render_option,path=['question','.blank',self.template_key,'template'],default=r"[% text %]"
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
def dump_markdwon(self):
|
|
91
|
+
text = super().dump_markdwon()
|
|
92
|
+
text= render_template(self.template,{'text':text})
|
|
93
|
+
return text
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
|