qparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,51 @@
1
+ # -*- coding: utf-8 -*-
2
+ from bs4 import BeautifulSoup,element
3
+ from typing import Union
4
+ import copy
5
+ class BaseNode():
6
+ def __init__(self,node:Union[element.Tag,element.NavigableString],render_option:dict):
7
+ self._soup = node
8
+ self._render_option = render_option
9
+
10
+ @classmethod
11
+ def create_by_node(cls,node:element.Tag,render_option:dict):
12
+
13
+ return None
14
+ def dump_markdwon(self):
15
+ return ''
16
+ @property
17
+ def ParentMarkdown(self):
18
+ return self._soup.find_parent("markdwon")
19
+ @property
20
+ def QuestionSoup(self):
21
+ return self._soup.find_parent("question")
22
+ @property
23
+ def depth(self)->int:
24
+ if p:=self.ParentMarkdown:
25
+
26
+ return int( p.attrs.get('depth',0))
27
+
28
+ return 0
29
+ def dump_soup(self):
30
+ pass
31
+
32
+ @property
33
+ def attrs(self):
34
+ return copy.deepcopy(getattr(self._soup,'attrs',{}))
35
+ def dump_docx_buffer(self):
36
+ return []
37
+
38
+
39
+
40
+
41
+
42
+
43
+
44
+
45
+
46
+
47
+
48
+
49
+
50
+
51
+
@@ -0,0 +1,31 @@
1
+ # -*- coding: utf-8 -*-
2
+ from bs4 import BeautifulSoup,element
3
+ from typing import Union
4
+ import copy
5
+ from .base_node import BaseNode
6
+
7
+ class ImgNode(BaseNode):
8
+ def __init__(self, node, render_option):
9
+ super().__init__(node, render_option)
10
+ @classmethod
11
+ def create_by_node(cls,node:element.Tag,render_option:dict):
12
+ if isinstance(node,element.Tag) and node.name=="img":
13
+ return cls(node=node,render_option=render_option)
14
+ return None
15
+
16
+ def dump_soup(self):
17
+ return copy.deepcopy(self._soup)
18
+ def dump_markdwon(self):
19
+ alt = self._soup.attrs.get("alt", "")
20
+ src = self._soup.attrs.get("src", "")
21
+ title = self._soup.attrs.get("title")
22
+
23
+ if not src:
24
+ return ""
25
+
26
+ if title:
27
+ return f'![{alt}]({src} "{title}")'
28
+ return f'![{alt}]({src})'
29
+
30
+
31
+
@@ -0,0 +1,139 @@
1
+ # -*- coding: utf-8 -*-
2
+ from mistletoe import Document
3
+ from mistletoe.html_renderer import HtmlRenderer
4
+ from mistletoe.markdown_renderer import MarkdownRenderer
5
+ from mistletoe.block_token import Heading
6
+ from mistletoe.span_token import RawText
7
+ from bs4 import BeautifulSoup,element
8
+ from typing import Union
9
+ from .base_node import BaseNode
10
+ from qparse.utils.html_full_protector import HtmlFullProtector
11
+ from qparse.render_option import SafeGetRenderOption,render_template
12
+ class TextNode(BaseNode):
13
+ htmlParser = HtmlFullProtector()
14
+ def __init__(self, node, render_option):
15
+ super().__init__(node, render_option)
16
+
17
+ @classmethod
18
+ def create_by_node(cls,node:element.NavigableString,render_option:dict):
19
+ text = node.get_text(strip=True)
20
+
21
+ if isinstance(node,element.NavigableString):
22
+ return cls(node=node,render_option=render_option)
23
+ return None
24
+
25
+ def __repr__(self):
26
+ return '<{} text="{}...">'.format(self.__class__.__name__,self._soup.get_text(strip=True)[:10])
27
+
28
+ @property
29
+ def text(self):
30
+ return self._soup.get_text()
31
+
32
+ @property
33
+ def unicode_text(self):
34
+ return self.htmlParser.decode_to_unicode(self.text)
35
+
36
+ def dump_markdwon(self):
37
+
38
+
39
+ doc: Document = Document(self.unicode_text)
40
+ stack = [doc]
41
+ while len(stack):
42
+ block = stack.pop(0)
43
+ if isinstance(block, Heading):
44
+ block.level = self.depth + block.level
45
+ if children:=getattr(block,'children'):
46
+ stack = [ *children,*stack]
47
+
48
+ with MarkdownRenderer() as r:
49
+ text=r.render(doc)
50
+ text = text.replace("\n",'')
51
+ return self.htmlParser.decode_to_unicode(text)
52
+ def dump_soup(self):
53
+ markdwon_text = self.dump_markdwon()
54
+ doc: Document = Document(markdwon_text)
55
+ with HtmlRenderer() as r:
56
+ soup_text = r.render(doc)
57
+ return BeautifulSoup(soup_text,features='html.parser')
58
+
59
+ def dump_docx_buffer(self):
60
+ return [{'type':'markdwonText','text':self.dump_markdwon()}]
61
+
62
+ class BrNode(BaseNode):
63
+ @classmethod
64
+ def create_by_node(cls, node: element.Tag, render_option: dict):
65
+ if isinstance(node, element.Tag) and node.name == "br":
66
+ return cls(node=node, render_option=render_option)
67
+ return None
68
+
69
+ def dump_markdwon(self):
70
+ return "\n"
71
+
72
+ def dump_soup(self):
73
+ return element.Tag(name="br")
74
+
75
+ def dump_docx_buffer(self):
76
+
77
+ return [{'type':'markdwonText','text':self.dump_markdwon()}]
78
+
79
+
80
+ class FirstTextVisibleNode(TextNode):
81
+ def __init__(self, node, render_option):
82
+ super().__init__(node, render_option)
83
+
84
+ @classmethod
85
+ def create_by_node(cls,node:element.Tag,render_option:dict):
86
+
87
+ if isinstance(node,element.Tag) and node.name=="first-text-visible":
88
+
89
+ return cls(node=node.children.__next__(),render_option=render_option)
90
+ return None
91
+ @property
92
+ def template(self):
93
+ return SafeGetRenderOption(
94
+ self._render_option,
95
+ [
96
+ "question",
97
+ self._soup.parent.attrs['include-by'],
98
+ "::first-text-visible",
99
+ 'template'
100
+ ],
101
+ r"[% text %]"
102
+ )
103
+
104
+
105
+
106
+ def dump_markdwon(self):
107
+ text = super().dump_markdwon()
108
+ # print(text)
109
+ doc: Document = Document(text)
110
+ stack = [doc]
111
+ text_index = 0
112
+ while len(stack):
113
+ block = stack.pop(0)
114
+ # print(type(block) )
115
+ if children:=getattr(block,'children'):
116
+ stack = [ *children,*stack]
117
+
118
+ if isinstance(block,RawText) and text_index==0:
119
+
120
+ text_index +=1
121
+ block.content = render_template(template_text=self.template,context={
122
+ 'text':block.content,
123
+ **self.QuestionSoup.attrs
124
+ })
125
+
126
+
127
+ with MarkdownRenderer() as r:
128
+ text= r.render(doc)
129
+ text = text.replace("\n",'')
130
+ return self.htmlParser.decode_to_unicode(text)
131
+
132
+
133
+
134
+
135
+
136
+
137
+
138
+
139
+
@@ -0,0 +1,17 @@
1
+ # -*- coding: utf-8 -*-
2
+ from bs4 import BeautifulSoup,element
3
+ from typing import Dict,Union
4
+ from .base_document import BaseDocument
5
+
6
+ class QuestionDocument(BaseDocument):
7
+ def __init__(self, node, render_option):
8
+ super().__init__(node, render_option)
9
+
10
+ @classmethod
11
+ def create_by_node(cls,node:element.Tag,render_option:dict):
12
+ if node.name=="question":
13
+ return cls(node=node,render_option=render_option)
14
+ return None
15
+
16
+
17
+
@@ -0,0 +1,64 @@
1
+ # -*- coding: utf-8 -*-
2
+ from bs4 import BeautifulSoup,element
3
+ from typing import Dict,Union
4
+ from .base_document import BaseDocument
5
+ from qparse.utils.html_full_protector import HtmlFullProtector
6
+ from qparse.render_option import SafeGetRenderOption, render_template
7
+ from markupsafe import Markup
8
+ class StemDocument(BaseDocument):
9
+ htmlParser = HtmlFullProtector()
10
+ def __init__(self, node, render_option):
11
+ self.show = SafeGetRenderOption(option=render_option,path=['question','stem','show'] ,default=False)
12
+
13
+ super().__init__(node, render_option)
14
+ @classmethod
15
+ def create_by_node(cls,node:element.Tag,render_option:dict):
16
+
17
+ if node.name=="stem":
18
+ # return cls(node=node,render_option=render_option)
19
+ for n in cls._iter_nodes(node):
20
+ if isinstance(n,element.NavigableString):
21
+ text = cls.htmlParser.decode_to_unicode(n.get_text())
22
+ if text.strip():
23
+
24
+ # s = element.NavigableString()
25
+ tag = element.Tag(name="first-text-visible")
26
+ tag.append(n.get_text())
27
+ tag.attrs['include-by']= "stem"
28
+ n.replace_with(tag)
29
+
30
+ break
31
+ return cls(node=node,render_option=render_option)
32
+ return None
33
+ def dump_markdwon(self):
34
+ if not self.show:
35
+ return ''
36
+
37
+ text = super().dump_markdwon()
38
+ question = self._soup.find_parent('question')
39
+ answer_types = question.attrs.get('answer_types', '') if question else ''
40
+ template = SafeGetRenderOption(
41
+ self._render_option,
42
+ ['question', 'stem', f'[answer_types="{answer_types}"]', 'template'],
43
+ default='[%text%]'
44
+ )
45
+ return render_template(
46
+ template_text=template,
47
+ context={
48
+ 'text': Markup(text),
49
+ **(question.attrs if question else {})
50
+ }
51
+ )
52
+ def dump_docx_buffer(self):
53
+ if not self.show:
54
+ return []
55
+ keep_together= SafeGetRenderOption(self.render_option,['question', 'stem']).get('keep_together',True)
56
+ node = {'type':"document",'keep_together':True,'children':[]}
57
+ if not keep_together:
58
+ return super().dump_docx_buffer()
59
+ for c in self.children:
60
+ node['children'].extend(c.dump_docx_buffer())
61
+ return [node]
62
+
63
+
64
+
@@ -0,0 +1,126 @@
1
+ # -*- coding: utf-8 -*-
2
+ from bs4 import BeautifulSoup,element
3
+ from typing import Dict,Union
4
+ from .base_document import BaseDocument
5
+ import copy
6
+
7
+
8
+ class TdDocument(BaseDocument):
9
+ def __init__(self, node, render_option):
10
+ super().__init__(node, render_option)
11
+
12
+ @classmethod
13
+ def create_by_node(cls,node:element.Tag,render_option:dict):
14
+ if node.name=="td":
15
+ return cls(node=node,render_option=render_option)
16
+ return None
17
+ def dump_soup(self):
18
+ tag = element.Tag(name='td',attrs=copy.deepcopy(self._soup.attrs) )
19
+
20
+ for child in self.children:
21
+ sub_soup = child.dump_soup()
22
+ if sub_soup.name!="br":
23
+ tag.append(child.dump_soup())
24
+ return tag
25
+
26
+
27
+ class ThDocument(BaseDocument):
28
+ def __init__(self, node, render_option):
29
+ super().__init__(node, render_option)
30
+
31
+ @classmethod
32
+ def create_by_node(cls,node:element.Tag,render_option:dict):
33
+ if node.name=="th":
34
+ return cls(node=node,render_option=render_option)
35
+ return None
36
+ class TrDocument(BaseDocument):
37
+ def __init__(self, node, render_option):
38
+ super().__init__(node, render_option)
39
+
40
+ @classmethod
41
+ def create_by_node(cls,node:element.Tag,render_option:dict):
42
+ if node.name=="tr":
43
+ return cls(node=node,render_option=render_option)
44
+ return None
45
+ def dump_soup(self):
46
+
47
+ tag = element.Tag(name='tr',attrs=copy.deepcopy(self._soup.attrs) )
48
+
49
+ for child in self.children:
50
+ if isinstance(child,TdDocument):
51
+ tag.append(child.dump_soup())
52
+ return tag
53
+
54
+
55
+
56
+ class TBodyDocument(BaseDocument):
57
+ def __init__(self, node, render_option):
58
+ super().__init__(node, render_option)
59
+
60
+ @classmethod
61
+ def create_by_node(cls,node:element.Tag,render_option:dict):
62
+ if node.name=="tbody":
63
+ return cls(node=node,render_option=render_option)
64
+ return None
65
+ def dump_soup(self):
66
+ tag = element.Tag(name='tbody',attrs=copy.deepcopy(self._soup.attrs))
67
+ for child in self.children:
68
+ if isinstance(child,TrDocument):
69
+ tag.append(child.dump_soup())
70
+ return tag
71
+
72
+ def dump_markdwon(self):
73
+ soup = self.dump_soup()
74
+ return soup.prettify()
75
+
76
+
77
+ class THeadDocument(BaseDocument):
78
+ def __init__(self, node, render_option):
79
+ super().__init__(node, render_option)
80
+
81
+ @classmethod
82
+ def create_by_node(cls,node:element.Tag,render_option:dict):
83
+ if node.name=="thead":
84
+ return cls(node=node,render_option=render_option)
85
+ return None
86
+ def dump_soup(self):
87
+ tag = element.Tag(name='thead',attrs=copy.deepcopy(self._soup.attrs))
88
+ for child in self.children:
89
+ if isinstance(child,TrDocument):
90
+ tag.append(child.dump_soup())
91
+ return tag
92
+
93
+
94
+
95
+ class TableDocument(TBodyDocument):
96
+ def __init__(self, node, render_option):
97
+ super().__init__(node, render_option)
98
+
99
+ @classmethod
100
+ def create_by_node(cls,node:element.Tag,render_option:dict):
101
+ if node.name=="table":
102
+ return cls(node=node,render_option=render_option)
103
+ return None
104
+ def dump_soup(self):
105
+ tag = element.Tag(name='table',attrs=copy.deepcopy(self._soup.attrs))
106
+ for child in self.children:
107
+
108
+ if isinstance(child,TBodyDocument) or isinstance(child,TrDocument):
109
+ if sub_soup:= child.dump_soup() :
110
+ if sub_soup.name!="br":
111
+ tag.append(sub_soup)
112
+ return tag
113
+
114
+ def dump_markdwon(self):
115
+ soup = self.dump_soup()
116
+ # print(soup)
117
+ # return ''
118
+ return "\n"+soup.prettify()+"\n"
119
+ def dump_docx_buffer(self):
120
+ return [{"type":'tableHtmlText',"text":self.dump_markdwon()}]
121
+
122
+ THeadDocument.register_node_parser('tr',TrDocument)
123
+ TBodyDocument.register_node_parser('tr',TrDocument)
124
+
125
+ TrDocument.register_node_parser('td',TdDocument)
126
+ TrDocument.register_node_parser('th',ThDocument)
@@ -0,0 +1,189 @@
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ from cgitb import text
5
+ import re,copy
6
+ from pathlib import Path
7
+
8
+ from bs4 import BeautifulSoup, element
9
+
10
+ from .utils import HtmlFullProtector
11
+
12
+
13
+ class MarkdownLoader:
14
+ def __init__(self, src: str):
15
+ self._src = Path(src)
16
+ self._protector = HtmlFullProtector()
17
+ self._reference_answer = BeautifulSoup("<div></div>", "html.parser").div
18
+
19
+ def _read_text(self, path: Path) -> str:
20
+ return path.read_text(encoding="utf-8")
21
+
22
+ def _to_absolute_url(self, raw_url: str, base_dir: Path) -> str:
23
+ url = raw_url.strip().strip('"').strip("'")
24
+ if re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*://", url) or url.startswith("/"):
25
+ return url
26
+ return str((base_dir / url).resolve())
27
+
28
+ def _replace_images(self, text: str, base_dir: Path) -> str:
29
+ pattern = re.compile(r'!\[([^\]]*)\]\(([^)]+)\)')
30
+
31
+ def repl(match: re.Match) -> str:
32
+ alt = match.group(1)
33
+ url = self._to_absolute_url(match.group(2), base_dir)
34
+ return f'<img alt="{alt}" src="{url}" />'
35
+
36
+ return pattern.sub(repl, text)
37
+
38
+ def _replace_imports(self, text: str, base_dir: Path) -> str:
39
+ pattern = re.compile(r'^\s*@import\s+["\']([^"\']+)["\']\s*$', re.MULTILINE)
40
+
41
+ def repl(match: re.Match) -> str:
42
+ url = self._to_absolute_url(match.group(1), base_dir)
43
+ return f'<markdwon src="{url}"></markdwon>'
44
+
45
+ return pattern.sub(repl, text)
46
+
47
+ def _load_markdown(self, path: Path) -> str:
48
+ base_dir = path.resolve().parent
49
+ text = self._read_text(path.resolve())
50
+ text = self._replace_images(text, base_dir)
51
+ text = self._replace_imports(text, base_dir)
52
+ return text
53
+
54
+ def _expand_imports(self, soup: BeautifulSoup, current_path: Path, stack: set[Path] | None = None, depth: int = 0) -> None:
55
+ stack = set() if stack is None else stack
56
+ current_resolved = current_path.resolve()
57
+ if current_resolved in stack:
58
+ return
59
+ stack.add(current_resolved)
60
+
61
+ # 给根 soup 加上 depth 属性!如果是根的话 depth 是 0!
62
+ if hasattr(soup, "markdwon_depth"):
63
+ soup.markdwon_depth = depth
64
+ else:
65
+ # 对于 BeautifulSoup 根,我们给根的标签(如果是 markdwon 的话)加 depth
66
+ if soup.contents and hasattr(soup.contents[0], "name") and soup.contents[0].name == "markdwon":
67
+ soup.contents[0]["depth"] = depth
68
+
69
+ for node in list(soup.find_all("markdwon")):
70
+ src = node.get("src")
71
+ if not src:
72
+ continue
73
+
74
+ imported_path = Path(src).resolve()
75
+ if imported_path in stack:
76
+ node["error"] = "circular import"
77
+ continue
78
+
79
+ if not imported_path.exists():
80
+ node["error"] = "import not found"
81
+ continue
82
+
83
+ # 这个 markdwon 标签应该是 depth + 1
84
+ node["depth"] = depth + 1
85
+ imported_markdown = self._load_markdown(imported_path)
86
+ protected = self._protector.encode(imported_markdown)
87
+ decoded = self._protector.decode_to_html_entity(protected)
88
+ imported_soup = BeautifulSoup(decoded, "html.parser")
89
+ self._expand_imports(imported_soup, imported_path, stack, depth + 1)
90
+ node.clear()
91
+ node.extend(list(imported_soup.contents))
92
+
93
+ stack.remove(current_resolved)
94
+
95
+ def _decorate_questions(self, soup: BeautifulSoup) -> None:
96
+ answer_class_map = {
97
+ "chiose": "choice",
98
+ "multiple-choice": "multiple-choice",
99
+ "multiple-chiose": "multiple-choice",
100
+ "blank": "blank",
101
+ "answer": "answer",
102
+ }
103
+
104
+ for question_id, question in enumerate(soup.find_all("question"), start=1):
105
+ answer_types = []
106
+ question["question_id"] = str(question_id)
107
+
108
+ for span in question.find_all("span"):
109
+ class_list = span.get("class", [])
110
+ for class_name in class_list:
111
+ answer_type = answer_class_map.get(class_name)
112
+ if not answer_type:
113
+ continue
114
+ span["type"] = "answer"
115
+ if answer_type not in answer_types:
116
+ answer_types.append(answer_type)
117
+
118
+ if answer_types:
119
+ question["answer_types"] = ",".join(answer_types)
120
+
121
+ def get_soup(self) -> BeautifulSoup:
122
+ markdown = self._load_markdown(self._src)
123
+ protected = self._protector.encode(markdown)
124
+ decoded = self._protector.decode_to_html_entity(protected)
125
+ root_src = str(self._src.resolve())
126
+ wrapped = f'<markdwon src="{root_src}" depth="0">{decoded}</markdwon>'
127
+ soup = BeautifulSoup(wrapped, "html.parser")
128
+ self._expand_imports(soup, self._src, set(), depth=0)
129
+ self._decorate_questions(soup)
130
+
131
+ return copy.deepcopy(soup.select_one('markdwon'))
132
+
133
+ def get_reference_answer_markdwon(self) -> str:
134
+ lines = []
135
+ table_header = []
136
+ table_answers = []
137
+
138
+ def flush_table():
139
+ nonlocal table_header, table_answers
140
+ if not table_header:
141
+ return
142
+ lines.append("| " + " | ".join(table_header) + " |")
143
+ lines.append("| " + " | ".join(["---"] * len(table_header)) + " |")
144
+ lines.append("| " + " | ".join(table_answers) + " |")
145
+ lines.append("")
146
+ table_header = []
147
+ table_answers = []
148
+
149
+ soup = self.get_soup()
150
+
151
+ for question in soup.find_all("question"):
152
+ answer_nodes = question.find_all(
153
+ "span",
154
+ attrs={"type": "answer"}
155
+ )
156
+ if not answer_nodes:
157
+ continue
158
+
159
+ answer = "".join(
160
+ answer_node.get_text()
161
+ for answer_node in answer_nodes
162
+ )
163
+ answer_question = answer_nodes[0].find_parent("question")
164
+ if answer_question is None:
165
+ continue
166
+
167
+ question_id = answer_question.get("question_id", "")
168
+ answer_types = set(
169
+ answer_type.strip()
170
+ for answer_type in answer_question.get(
171
+ "answer_types",
172
+ ""
173
+ ).split(",")
174
+ if answer_type.strip()
175
+ )
176
+
177
+ if "answer" in answer_types:
178
+ flush_table()
179
+ lines.append(f"{question_id}、{answer}")
180
+ lines.append("")
181
+ continue
182
+
183
+ table_header.append(question_id)
184
+ table_answers.append(answer)
185
+
186
+ flush_table()
187
+ return "\n".join(lines).rstrip() + "\n"
188
+
189
+