qparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qmd_cli.py +262 -0
- qparse/__init__.py +16 -0
- qparse/docx_render/__init__.py +3 -0
- qparse/docx_render/docx_render.py +68 -0
- qparse/docx_render/write_buffer.py +412 -0
- qparse/markdwon_document/__init__.py +28 -0
- qparse/markdwon_document/analysis_document.py +40 -0
- qparse/markdwon_document/base_document.py +87 -0
- qparse/markdwon_document/markdwon_document.py +16 -0
- qparse/markdwon_document/nodes/__init__.py +5 -0
- qparse/markdwon_document/nodes/answer_node.py +97 -0
- qparse/markdwon_document/nodes/base_node.py +51 -0
- qparse/markdwon_document/nodes/img_node.py +31 -0
- qparse/markdwon_document/nodes/text_node.py +139 -0
- qparse/markdwon_document/question_document.py +17 -0
- qparse/markdwon_document/stem_document.py +64 -0
- qparse/markdwon_document/table_document.py +126 -0
- qparse/markdwon_loader.py +189 -0
- qparse/markdwon_render.py +123 -0
- qparse/qmd_packer.py +475 -0
- qparse/qmd_unpacker.py +169 -0
- qparse/render_option/__init__.py +40 -0
- qparse/render_option/default_render_option.py +86 -0
- qparse/render_option/default_render_template.md +21 -0
- qparse/render_option/referance.docx +0 -0
- qparse/render_option/theme.json +245 -0
- qparse/utils/__init__.py +3 -0
- qparse/utils/html_full_protector.py +154 -0
- qparse-0.1.0.dist-info/METADATA +454 -0
- qparse-0.1.0.dist-info/RECORD +34 -0
- qparse-0.1.0.dist-info/WHEEL +5 -0
- qparse-0.1.0.dist-info/entry_points.txt +2 -0
- qparse-0.1.0.dist-info/licenses/LICENSE +21 -0
- qparse-0.1.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Union
|
|
4
|
+
import copy
|
|
5
|
+
class BaseNode():
|
|
6
|
+
def __init__(self,node:Union[element.Tag,element.NavigableString],render_option:dict):
|
|
7
|
+
self._soup = node
|
|
8
|
+
self._render_option = render_option
|
|
9
|
+
|
|
10
|
+
@classmethod
|
|
11
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
12
|
+
|
|
13
|
+
return None
|
|
14
|
+
def dump_markdwon(self):
|
|
15
|
+
return ''
|
|
16
|
+
@property
|
|
17
|
+
def ParentMarkdown(self):
|
|
18
|
+
return self._soup.find_parent("markdwon")
|
|
19
|
+
@property
|
|
20
|
+
def QuestionSoup(self):
|
|
21
|
+
return self._soup.find_parent("question")
|
|
22
|
+
@property
|
|
23
|
+
def depth(self)->int:
|
|
24
|
+
if p:=self.ParentMarkdown:
|
|
25
|
+
|
|
26
|
+
return int( p.attrs.get('depth',0))
|
|
27
|
+
|
|
28
|
+
return 0
|
|
29
|
+
def dump_soup(self):
|
|
30
|
+
pass
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def attrs(self):
|
|
34
|
+
return copy.deepcopy(getattr(self._soup,'attrs',{}))
|
|
35
|
+
def dump_docx_buffer(self):
|
|
36
|
+
return []
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Union
|
|
4
|
+
import copy
|
|
5
|
+
from .base_node import BaseNode
|
|
6
|
+
|
|
7
|
+
class ImgNode(BaseNode):
|
|
8
|
+
def __init__(self, node, render_option):
|
|
9
|
+
super().__init__(node, render_option)
|
|
10
|
+
@classmethod
|
|
11
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
12
|
+
if isinstance(node,element.Tag) and node.name=="img":
|
|
13
|
+
return cls(node=node,render_option=render_option)
|
|
14
|
+
return None
|
|
15
|
+
|
|
16
|
+
def dump_soup(self):
|
|
17
|
+
return copy.deepcopy(self._soup)
|
|
18
|
+
def dump_markdwon(self):
|
|
19
|
+
alt = self._soup.attrs.get("alt", "")
|
|
20
|
+
src = self._soup.attrs.get("src", "")
|
|
21
|
+
title = self._soup.attrs.get("title")
|
|
22
|
+
|
|
23
|
+
if not src:
|
|
24
|
+
return ""
|
|
25
|
+
|
|
26
|
+
if title:
|
|
27
|
+
return f''
|
|
28
|
+
return f''
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from mistletoe import Document
|
|
3
|
+
from mistletoe.html_renderer import HtmlRenderer
|
|
4
|
+
from mistletoe.markdown_renderer import MarkdownRenderer
|
|
5
|
+
from mistletoe.block_token import Heading
|
|
6
|
+
from mistletoe.span_token import RawText
|
|
7
|
+
from bs4 import BeautifulSoup,element
|
|
8
|
+
from typing import Union
|
|
9
|
+
from .base_node import BaseNode
|
|
10
|
+
from qparse.utils.html_full_protector import HtmlFullProtector
|
|
11
|
+
from qparse.render_option import SafeGetRenderOption,render_template
|
|
12
|
+
class TextNode(BaseNode):
|
|
13
|
+
htmlParser = HtmlFullProtector()
|
|
14
|
+
def __init__(self, node, render_option):
|
|
15
|
+
super().__init__(node, render_option)
|
|
16
|
+
|
|
17
|
+
@classmethod
|
|
18
|
+
def create_by_node(cls,node:element.NavigableString,render_option:dict):
|
|
19
|
+
text = node.get_text(strip=True)
|
|
20
|
+
|
|
21
|
+
if isinstance(node,element.NavigableString):
|
|
22
|
+
return cls(node=node,render_option=render_option)
|
|
23
|
+
return None
|
|
24
|
+
|
|
25
|
+
def __repr__(self):
|
|
26
|
+
return '<{} text="{}...">'.format(self.__class__.__name__,self._soup.get_text(strip=True)[:10])
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def text(self):
|
|
30
|
+
return self._soup.get_text()
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def unicode_text(self):
|
|
34
|
+
return self.htmlParser.decode_to_unicode(self.text)
|
|
35
|
+
|
|
36
|
+
def dump_markdwon(self):
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
doc: Document = Document(self.unicode_text)
|
|
40
|
+
stack = [doc]
|
|
41
|
+
while len(stack):
|
|
42
|
+
block = stack.pop(0)
|
|
43
|
+
if isinstance(block, Heading):
|
|
44
|
+
block.level = self.depth + block.level
|
|
45
|
+
if children:=getattr(block,'children'):
|
|
46
|
+
stack = [ *children,*stack]
|
|
47
|
+
|
|
48
|
+
with MarkdownRenderer() as r:
|
|
49
|
+
text=r.render(doc)
|
|
50
|
+
text = text.replace("\n",'')
|
|
51
|
+
return self.htmlParser.decode_to_unicode(text)
|
|
52
|
+
def dump_soup(self):
|
|
53
|
+
markdwon_text = self.dump_markdwon()
|
|
54
|
+
doc: Document = Document(markdwon_text)
|
|
55
|
+
with HtmlRenderer() as r:
|
|
56
|
+
soup_text = r.render(doc)
|
|
57
|
+
return BeautifulSoup(soup_text,features='html.parser')
|
|
58
|
+
|
|
59
|
+
def dump_docx_buffer(self):
|
|
60
|
+
return [{'type':'markdwonText','text':self.dump_markdwon()}]
|
|
61
|
+
|
|
62
|
+
class BrNode(BaseNode):
|
|
63
|
+
@classmethod
|
|
64
|
+
def create_by_node(cls, node: element.Tag, render_option: dict):
|
|
65
|
+
if isinstance(node, element.Tag) and node.name == "br":
|
|
66
|
+
return cls(node=node, render_option=render_option)
|
|
67
|
+
return None
|
|
68
|
+
|
|
69
|
+
def dump_markdwon(self):
|
|
70
|
+
return "\n"
|
|
71
|
+
|
|
72
|
+
def dump_soup(self):
|
|
73
|
+
return element.Tag(name="br")
|
|
74
|
+
|
|
75
|
+
def dump_docx_buffer(self):
|
|
76
|
+
|
|
77
|
+
return [{'type':'markdwonText','text':self.dump_markdwon()}]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class FirstTextVisibleNode(TextNode):
|
|
81
|
+
def __init__(self, node, render_option):
|
|
82
|
+
super().__init__(node, render_option)
|
|
83
|
+
|
|
84
|
+
@classmethod
|
|
85
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
86
|
+
|
|
87
|
+
if isinstance(node,element.Tag) and node.name=="first-text-visible":
|
|
88
|
+
|
|
89
|
+
return cls(node=node.children.__next__(),render_option=render_option)
|
|
90
|
+
return None
|
|
91
|
+
@property
|
|
92
|
+
def template(self):
|
|
93
|
+
return SafeGetRenderOption(
|
|
94
|
+
self._render_option,
|
|
95
|
+
[
|
|
96
|
+
"question",
|
|
97
|
+
self._soup.parent.attrs['include-by'],
|
|
98
|
+
"::first-text-visible",
|
|
99
|
+
'template'
|
|
100
|
+
],
|
|
101
|
+
r"[% text %]"
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def dump_markdwon(self):
|
|
107
|
+
text = super().dump_markdwon()
|
|
108
|
+
# print(text)
|
|
109
|
+
doc: Document = Document(text)
|
|
110
|
+
stack = [doc]
|
|
111
|
+
text_index = 0
|
|
112
|
+
while len(stack):
|
|
113
|
+
block = stack.pop(0)
|
|
114
|
+
# print(type(block) )
|
|
115
|
+
if children:=getattr(block,'children'):
|
|
116
|
+
stack = [ *children,*stack]
|
|
117
|
+
|
|
118
|
+
if isinstance(block,RawText) and text_index==0:
|
|
119
|
+
|
|
120
|
+
text_index +=1
|
|
121
|
+
block.content = render_template(template_text=self.template,context={
|
|
122
|
+
'text':block.content,
|
|
123
|
+
**self.QuestionSoup.attrs
|
|
124
|
+
})
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
with MarkdownRenderer() as r:
|
|
128
|
+
text= r.render(doc)
|
|
129
|
+
text = text.replace("\n",'')
|
|
130
|
+
return self.htmlParser.decode_to_unicode(text)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Dict,Union
|
|
4
|
+
from .base_document import BaseDocument
|
|
5
|
+
|
|
6
|
+
class QuestionDocument(BaseDocument):
|
|
7
|
+
def __init__(self, node, render_option):
|
|
8
|
+
super().__init__(node, render_option)
|
|
9
|
+
|
|
10
|
+
@classmethod
|
|
11
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
12
|
+
if node.name=="question":
|
|
13
|
+
return cls(node=node,render_option=render_option)
|
|
14
|
+
return None
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Dict,Union
|
|
4
|
+
from .base_document import BaseDocument
|
|
5
|
+
from qparse.utils.html_full_protector import HtmlFullProtector
|
|
6
|
+
from qparse.render_option import SafeGetRenderOption, render_template
|
|
7
|
+
from markupsafe import Markup
|
|
8
|
+
class StemDocument(BaseDocument):
|
|
9
|
+
htmlParser = HtmlFullProtector()
|
|
10
|
+
def __init__(self, node, render_option):
|
|
11
|
+
self.show = SafeGetRenderOption(option=render_option,path=['question','stem','show'] ,default=False)
|
|
12
|
+
|
|
13
|
+
super().__init__(node, render_option)
|
|
14
|
+
@classmethod
|
|
15
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
16
|
+
|
|
17
|
+
if node.name=="stem":
|
|
18
|
+
# return cls(node=node,render_option=render_option)
|
|
19
|
+
for n in cls._iter_nodes(node):
|
|
20
|
+
if isinstance(n,element.NavigableString):
|
|
21
|
+
text = cls.htmlParser.decode_to_unicode(n.get_text())
|
|
22
|
+
if text.strip():
|
|
23
|
+
|
|
24
|
+
# s = element.NavigableString()
|
|
25
|
+
tag = element.Tag(name="first-text-visible")
|
|
26
|
+
tag.append(n.get_text())
|
|
27
|
+
tag.attrs['include-by']= "stem"
|
|
28
|
+
n.replace_with(tag)
|
|
29
|
+
|
|
30
|
+
break
|
|
31
|
+
return cls(node=node,render_option=render_option)
|
|
32
|
+
return None
|
|
33
|
+
def dump_markdwon(self):
|
|
34
|
+
if not self.show:
|
|
35
|
+
return ''
|
|
36
|
+
|
|
37
|
+
text = super().dump_markdwon()
|
|
38
|
+
question = self._soup.find_parent('question')
|
|
39
|
+
answer_types = question.attrs.get('answer_types', '') if question else ''
|
|
40
|
+
template = SafeGetRenderOption(
|
|
41
|
+
self._render_option,
|
|
42
|
+
['question', 'stem', f'[answer_types="{answer_types}"]', 'template'],
|
|
43
|
+
default='[%text%]'
|
|
44
|
+
)
|
|
45
|
+
return render_template(
|
|
46
|
+
template_text=template,
|
|
47
|
+
context={
|
|
48
|
+
'text': Markup(text),
|
|
49
|
+
**(question.attrs if question else {})
|
|
50
|
+
}
|
|
51
|
+
)
|
|
52
|
+
def dump_docx_buffer(self):
|
|
53
|
+
if not self.show:
|
|
54
|
+
return []
|
|
55
|
+
keep_together= SafeGetRenderOption(self.render_option,['question', 'stem']).get('keep_together',True)
|
|
56
|
+
node = {'type':"document",'keep_together':True,'children':[]}
|
|
57
|
+
if not keep_together:
|
|
58
|
+
return super().dump_docx_buffer()
|
|
59
|
+
for c in self.children:
|
|
60
|
+
node['children'].extend(c.dump_docx_buffer())
|
|
61
|
+
return [node]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from bs4 import BeautifulSoup,element
|
|
3
|
+
from typing import Dict,Union
|
|
4
|
+
from .base_document import BaseDocument
|
|
5
|
+
import copy
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class TdDocument(BaseDocument):
|
|
9
|
+
def __init__(self, node, render_option):
|
|
10
|
+
super().__init__(node, render_option)
|
|
11
|
+
|
|
12
|
+
@classmethod
|
|
13
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
14
|
+
if node.name=="td":
|
|
15
|
+
return cls(node=node,render_option=render_option)
|
|
16
|
+
return None
|
|
17
|
+
def dump_soup(self):
|
|
18
|
+
tag = element.Tag(name='td',attrs=copy.deepcopy(self._soup.attrs) )
|
|
19
|
+
|
|
20
|
+
for child in self.children:
|
|
21
|
+
sub_soup = child.dump_soup()
|
|
22
|
+
if sub_soup.name!="br":
|
|
23
|
+
tag.append(child.dump_soup())
|
|
24
|
+
return tag
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class ThDocument(BaseDocument):
|
|
28
|
+
def __init__(self, node, render_option):
|
|
29
|
+
super().__init__(node, render_option)
|
|
30
|
+
|
|
31
|
+
@classmethod
|
|
32
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
33
|
+
if node.name=="th":
|
|
34
|
+
return cls(node=node,render_option=render_option)
|
|
35
|
+
return None
|
|
36
|
+
class TrDocument(BaseDocument):
|
|
37
|
+
def __init__(self, node, render_option):
|
|
38
|
+
super().__init__(node, render_option)
|
|
39
|
+
|
|
40
|
+
@classmethod
|
|
41
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
42
|
+
if node.name=="tr":
|
|
43
|
+
return cls(node=node,render_option=render_option)
|
|
44
|
+
return None
|
|
45
|
+
def dump_soup(self):
|
|
46
|
+
|
|
47
|
+
tag = element.Tag(name='tr',attrs=copy.deepcopy(self._soup.attrs) )
|
|
48
|
+
|
|
49
|
+
for child in self.children:
|
|
50
|
+
if isinstance(child,TdDocument):
|
|
51
|
+
tag.append(child.dump_soup())
|
|
52
|
+
return tag
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class TBodyDocument(BaseDocument):
|
|
57
|
+
def __init__(self, node, render_option):
|
|
58
|
+
super().__init__(node, render_option)
|
|
59
|
+
|
|
60
|
+
@classmethod
|
|
61
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
62
|
+
if node.name=="tbody":
|
|
63
|
+
return cls(node=node,render_option=render_option)
|
|
64
|
+
return None
|
|
65
|
+
def dump_soup(self):
|
|
66
|
+
tag = element.Tag(name='tbody',attrs=copy.deepcopy(self._soup.attrs))
|
|
67
|
+
for child in self.children:
|
|
68
|
+
if isinstance(child,TrDocument):
|
|
69
|
+
tag.append(child.dump_soup())
|
|
70
|
+
return tag
|
|
71
|
+
|
|
72
|
+
def dump_markdwon(self):
|
|
73
|
+
soup = self.dump_soup()
|
|
74
|
+
return soup.prettify()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class THeadDocument(BaseDocument):
|
|
78
|
+
def __init__(self, node, render_option):
|
|
79
|
+
super().__init__(node, render_option)
|
|
80
|
+
|
|
81
|
+
@classmethod
|
|
82
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
83
|
+
if node.name=="thead":
|
|
84
|
+
return cls(node=node,render_option=render_option)
|
|
85
|
+
return None
|
|
86
|
+
def dump_soup(self):
|
|
87
|
+
tag = element.Tag(name='thead',attrs=copy.deepcopy(self._soup.attrs))
|
|
88
|
+
for child in self.children:
|
|
89
|
+
if isinstance(child,TrDocument):
|
|
90
|
+
tag.append(child.dump_soup())
|
|
91
|
+
return tag
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class TableDocument(TBodyDocument):
|
|
96
|
+
def __init__(self, node, render_option):
|
|
97
|
+
super().__init__(node, render_option)
|
|
98
|
+
|
|
99
|
+
@classmethod
|
|
100
|
+
def create_by_node(cls,node:element.Tag,render_option:dict):
|
|
101
|
+
if node.name=="table":
|
|
102
|
+
return cls(node=node,render_option=render_option)
|
|
103
|
+
return None
|
|
104
|
+
def dump_soup(self):
|
|
105
|
+
tag = element.Tag(name='table',attrs=copy.deepcopy(self._soup.attrs))
|
|
106
|
+
for child in self.children:
|
|
107
|
+
|
|
108
|
+
if isinstance(child,TBodyDocument) or isinstance(child,TrDocument):
|
|
109
|
+
if sub_soup:= child.dump_soup() :
|
|
110
|
+
if sub_soup.name!="br":
|
|
111
|
+
tag.append(sub_soup)
|
|
112
|
+
return tag
|
|
113
|
+
|
|
114
|
+
def dump_markdwon(self):
|
|
115
|
+
soup = self.dump_soup()
|
|
116
|
+
# print(soup)
|
|
117
|
+
# return ''
|
|
118
|
+
return "\n"+soup.prettify()+"\n"
|
|
119
|
+
def dump_docx_buffer(self):
|
|
120
|
+
return [{"type":'tableHtmlText',"text":self.dump_markdwon()}]
|
|
121
|
+
|
|
122
|
+
THeadDocument.register_node_parser('tr',TrDocument)
|
|
123
|
+
TBodyDocument.register_node_parser('tr',TrDocument)
|
|
124
|
+
|
|
125
|
+
TrDocument.register_node_parser('td',TdDocument)
|
|
126
|
+
TrDocument.register_node_parser('th',ThDocument)
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from cgitb import text
|
|
5
|
+
import re,copy
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from bs4 import BeautifulSoup, element
|
|
9
|
+
|
|
10
|
+
from .utils import HtmlFullProtector
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class MarkdownLoader:
|
|
14
|
+
def __init__(self, src: str):
|
|
15
|
+
self._src = Path(src)
|
|
16
|
+
self._protector = HtmlFullProtector()
|
|
17
|
+
self._reference_answer = BeautifulSoup("<div></div>", "html.parser").div
|
|
18
|
+
|
|
19
|
+
def _read_text(self, path: Path) -> str:
|
|
20
|
+
return path.read_text(encoding="utf-8")
|
|
21
|
+
|
|
22
|
+
def _to_absolute_url(self, raw_url: str, base_dir: Path) -> str:
|
|
23
|
+
url = raw_url.strip().strip('"').strip("'")
|
|
24
|
+
if re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*://", url) or url.startswith("/"):
|
|
25
|
+
return url
|
|
26
|
+
return str((base_dir / url).resolve())
|
|
27
|
+
|
|
28
|
+
def _replace_images(self, text: str, base_dir: Path) -> str:
|
|
29
|
+
pattern = re.compile(r'!\[([^\]]*)\]\(([^)]+)\)')
|
|
30
|
+
|
|
31
|
+
def repl(match: re.Match) -> str:
|
|
32
|
+
alt = match.group(1)
|
|
33
|
+
url = self._to_absolute_url(match.group(2), base_dir)
|
|
34
|
+
return f'<img alt="{alt}" src="{url}" />'
|
|
35
|
+
|
|
36
|
+
return pattern.sub(repl, text)
|
|
37
|
+
|
|
38
|
+
def _replace_imports(self, text: str, base_dir: Path) -> str:
|
|
39
|
+
pattern = re.compile(r'^\s*@import\s+["\']([^"\']+)["\']\s*$', re.MULTILINE)
|
|
40
|
+
|
|
41
|
+
def repl(match: re.Match) -> str:
|
|
42
|
+
url = self._to_absolute_url(match.group(1), base_dir)
|
|
43
|
+
return f'<markdwon src="{url}"></markdwon>'
|
|
44
|
+
|
|
45
|
+
return pattern.sub(repl, text)
|
|
46
|
+
|
|
47
|
+
def _load_markdown(self, path: Path) -> str:
|
|
48
|
+
base_dir = path.resolve().parent
|
|
49
|
+
text = self._read_text(path.resolve())
|
|
50
|
+
text = self._replace_images(text, base_dir)
|
|
51
|
+
text = self._replace_imports(text, base_dir)
|
|
52
|
+
return text
|
|
53
|
+
|
|
54
|
+
def _expand_imports(self, soup: BeautifulSoup, current_path: Path, stack: set[Path] | None = None, depth: int = 0) -> None:
|
|
55
|
+
stack = set() if stack is None else stack
|
|
56
|
+
current_resolved = current_path.resolve()
|
|
57
|
+
if current_resolved in stack:
|
|
58
|
+
return
|
|
59
|
+
stack.add(current_resolved)
|
|
60
|
+
|
|
61
|
+
# 给根 soup 加上 depth 属性!如果是根的话 depth 是 0!
|
|
62
|
+
if hasattr(soup, "markdwon_depth"):
|
|
63
|
+
soup.markdwon_depth = depth
|
|
64
|
+
else:
|
|
65
|
+
# 对于 BeautifulSoup 根,我们给根的标签(如果是 markdwon 的话)加 depth
|
|
66
|
+
if soup.contents and hasattr(soup.contents[0], "name") and soup.contents[0].name == "markdwon":
|
|
67
|
+
soup.contents[0]["depth"] = depth
|
|
68
|
+
|
|
69
|
+
for node in list(soup.find_all("markdwon")):
|
|
70
|
+
src = node.get("src")
|
|
71
|
+
if not src:
|
|
72
|
+
continue
|
|
73
|
+
|
|
74
|
+
imported_path = Path(src).resolve()
|
|
75
|
+
if imported_path in stack:
|
|
76
|
+
node["error"] = "circular import"
|
|
77
|
+
continue
|
|
78
|
+
|
|
79
|
+
if not imported_path.exists():
|
|
80
|
+
node["error"] = "import not found"
|
|
81
|
+
continue
|
|
82
|
+
|
|
83
|
+
# 这个 markdwon 标签应该是 depth + 1
|
|
84
|
+
node["depth"] = depth + 1
|
|
85
|
+
imported_markdown = self._load_markdown(imported_path)
|
|
86
|
+
protected = self._protector.encode(imported_markdown)
|
|
87
|
+
decoded = self._protector.decode_to_html_entity(protected)
|
|
88
|
+
imported_soup = BeautifulSoup(decoded, "html.parser")
|
|
89
|
+
self._expand_imports(imported_soup, imported_path, stack, depth + 1)
|
|
90
|
+
node.clear()
|
|
91
|
+
node.extend(list(imported_soup.contents))
|
|
92
|
+
|
|
93
|
+
stack.remove(current_resolved)
|
|
94
|
+
|
|
95
|
+
def _decorate_questions(self, soup: BeautifulSoup) -> None:
|
|
96
|
+
answer_class_map = {
|
|
97
|
+
"chiose": "choice",
|
|
98
|
+
"multiple-choice": "multiple-choice",
|
|
99
|
+
"multiple-chiose": "multiple-choice",
|
|
100
|
+
"blank": "blank",
|
|
101
|
+
"answer": "answer",
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
for question_id, question in enumerate(soup.find_all("question"), start=1):
|
|
105
|
+
answer_types = []
|
|
106
|
+
question["question_id"] = str(question_id)
|
|
107
|
+
|
|
108
|
+
for span in question.find_all("span"):
|
|
109
|
+
class_list = span.get("class", [])
|
|
110
|
+
for class_name in class_list:
|
|
111
|
+
answer_type = answer_class_map.get(class_name)
|
|
112
|
+
if not answer_type:
|
|
113
|
+
continue
|
|
114
|
+
span["type"] = "answer"
|
|
115
|
+
if answer_type not in answer_types:
|
|
116
|
+
answer_types.append(answer_type)
|
|
117
|
+
|
|
118
|
+
if answer_types:
|
|
119
|
+
question["answer_types"] = ",".join(answer_types)
|
|
120
|
+
|
|
121
|
+
def get_soup(self) -> BeautifulSoup:
|
|
122
|
+
markdown = self._load_markdown(self._src)
|
|
123
|
+
protected = self._protector.encode(markdown)
|
|
124
|
+
decoded = self._protector.decode_to_html_entity(protected)
|
|
125
|
+
root_src = str(self._src.resolve())
|
|
126
|
+
wrapped = f'<markdwon src="{root_src}" depth="0">{decoded}</markdwon>'
|
|
127
|
+
soup = BeautifulSoup(wrapped, "html.parser")
|
|
128
|
+
self._expand_imports(soup, self._src, set(), depth=0)
|
|
129
|
+
self._decorate_questions(soup)
|
|
130
|
+
|
|
131
|
+
return copy.deepcopy(soup.select_one('markdwon'))
|
|
132
|
+
|
|
133
|
+
def get_reference_answer_markdwon(self) -> str:
|
|
134
|
+
lines = []
|
|
135
|
+
table_header = []
|
|
136
|
+
table_answers = []
|
|
137
|
+
|
|
138
|
+
def flush_table():
|
|
139
|
+
nonlocal table_header, table_answers
|
|
140
|
+
if not table_header:
|
|
141
|
+
return
|
|
142
|
+
lines.append("| " + " | ".join(table_header) + " |")
|
|
143
|
+
lines.append("| " + " | ".join(["---"] * len(table_header)) + " |")
|
|
144
|
+
lines.append("| " + " | ".join(table_answers) + " |")
|
|
145
|
+
lines.append("")
|
|
146
|
+
table_header = []
|
|
147
|
+
table_answers = []
|
|
148
|
+
|
|
149
|
+
soup = self.get_soup()
|
|
150
|
+
|
|
151
|
+
for question in soup.find_all("question"):
|
|
152
|
+
answer_nodes = question.find_all(
|
|
153
|
+
"span",
|
|
154
|
+
attrs={"type": "answer"}
|
|
155
|
+
)
|
|
156
|
+
if not answer_nodes:
|
|
157
|
+
continue
|
|
158
|
+
|
|
159
|
+
answer = "".join(
|
|
160
|
+
answer_node.get_text()
|
|
161
|
+
for answer_node in answer_nodes
|
|
162
|
+
)
|
|
163
|
+
answer_question = answer_nodes[0].find_parent("question")
|
|
164
|
+
if answer_question is None:
|
|
165
|
+
continue
|
|
166
|
+
|
|
167
|
+
question_id = answer_question.get("question_id", "")
|
|
168
|
+
answer_types = set(
|
|
169
|
+
answer_type.strip()
|
|
170
|
+
for answer_type in answer_question.get(
|
|
171
|
+
"answer_types",
|
|
172
|
+
""
|
|
173
|
+
).split(",")
|
|
174
|
+
if answer_type.strip()
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
if "answer" in answer_types:
|
|
178
|
+
flush_table()
|
|
179
|
+
lines.append(f"{question_id}、{answer}")
|
|
180
|
+
lines.append("")
|
|
181
|
+
continue
|
|
182
|
+
|
|
183
|
+
table_header.append(question_id)
|
|
184
|
+
table_answers.append(answer)
|
|
185
|
+
|
|
186
|
+
flush_table()
|
|
187
|
+
return "\n".join(lines).rstrip() + "\n"
|
|
188
|
+
|
|
189
|
+
|