docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,763 @@
1
+ """把 DocVortex HTML v1 固定 DOM 解析为无资源副作用的 typed plan。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from copy import deepcopy
6
+ from typing import cast
7
+ from urllib.parse import unquote
8
+
9
+ from lxml import etree # type: ignore[reportMissingImports]
10
+
11
+ from ...schema import PAGE_BLOCK_TYPES, RAW_ALGORITHM, BlockType, VISUAL_TYPE_MAPPING
12
+ from docvortex.content.markup import extract_formula
13
+ from docvortex.content.markup.projector import BLOCK_TAGS, local_name
14
+ from .contracts import (
15
+ AnnotationWireSpec,
16
+ CodeBodyWireSpec,
17
+ EquationWireSpec,
18
+ FlowchartBodyWireSpec,
19
+ IndexBlockWireSpec,
20
+ IndexLeafWireSpec,
21
+ IndexWireSpec,
22
+ ListBlockWireSpec,
23
+ ListLeafWireSpec,
24
+ ListWireSpec,
25
+ DOCVORTEX_HTML_VERSION,
26
+ DocVortexHtmlWirePlan,
27
+ PageWireSpec,
28
+ RichVisualBodyWireSpec,
29
+ TableBodyWireSpec,
30
+ TextWireSpec,
31
+ VisualBodyWireSpec,
32
+ VisualWireSpec,
33
+ WireFallbackReason,
34
+ WireRenderMode,
35
+ WIRE_BLOCK_CLASS,
36
+ WIRE_DOCUMENT_CLASS,
37
+ WIRE_INDEX_CLASS,
38
+ WIRE_LIST_CONTENT_CLASS,
39
+ WIRE_LIST_MARKER_CLASS,
40
+ WIRE_PAGE_BREAK_CLASS,
41
+ WIRE_PAGE_CLASS,
42
+ WIRE_VISUAL_BODY_CLASS,
43
+ )
44
+
45
+
46
+ _SIMPLE_TEXT_TYPES = frozenset(
47
+ {
48
+ BlockType.TEXT,
49
+ BlockType.REF_TEXT,
50
+ BlockType.DOC_TITLE,
51
+ BlockType.PARAGRAPH_TITLE,
52
+ BlockType.HEADER,
53
+ BlockType.FOOTER,
54
+ BlockType.PAGE_NUMBER,
55
+ BlockType.ASIDE_TEXT,
56
+ BlockType.PAGE_FOOTNOTE,
57
+ }
58
+ )
59
+ _PAGE_AUXILIARY_TYPES = frozenset({BlockType.HEADER, BlockType.FOOTER, BlockType.PAGE_NUMBER, BlockType.ASIDE_TEXT})
60
+ _LIST_LEAF_TYPES = frozenset({BlockType.TEXT, BlockType.REF_TEXT})
61
+ _INDEX_LEAF_TYPES = frozenset({BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE})
62
+ _OWNED_VISUAL_IMAGE_TOKENS = frozenset(
63
+ {"docvortex-chart-image", "docvortex-flowchart-fallback", "docvortex-image", "docvortex-table-image"}
64
+ )
65
+
66
+
67
+ class NonCanonicalWire(ValueError):
68
+ """表示当前 DOM 不是 renderer 能生成的 canonical v1 wire。"""
69
+
70
+
71
+ def parse_docvortex_html_wire(body: etree._Element) -> tuple[DocVortexHtmlWirePlan | None, WireFallbackReason | None]:
72
+ """发现并解析 canonical v1 wire,非法结构只返回统一回退原因。"""
73
+ roots = [
74
+ element
75
+ for element in body.iter()
76
+ if isinstance(element.tag, str) and element.get("data-docvortex-html-version") is not None
77
+ ]
78
+ if not roots:
79
+ return None, None
80
+ if len(roots) != 1:
81
+ return None, "non_canonical_wire"
82
+ root = roots[0]
83
+ if (root.get("data-docvortex-html-version") or "").strip() != DOCVORTEX_HTML_VERSION:
84
+ return None, "unsupported_version"
85
+ try:
86
+ _validate_wire_root_ownership(body, root)
87
+ return _parse_wire_root(root), None
88
+ except NonCanonicalWire:
89
+ return None, "non_canonical_wire"
90
+
91
+
92
+ def _parse_wire_root(root: etree._Element) -> DocVortexHtmlWirePlan:
93
+ """解析根、渲染模式、页面容器和全部顶层 block。"""
94
+ if local_name(root) != "article" or _class_tokens(root) != {
95
+ WIRE_DOCUMENT_CLASS,
96
+ f"{WIRE_DOCUMENT_CLASS}--{(root.get('data-render-mode') or '').strip()}",
97
+ }:
98
+ raise NonCanonicalWire
99
+ _validate_structural_text(root)
100
+ mode_value = (root.get("data-render-mode") or "").strip()
101
+ if mode_value not in {"default", "full"}:
102
+ raise NonCanonicalWire
103
+ mode = cast(WireRenderMode, mode_value)
104
+ wrappers: list[tuple[etree._Element, int | None]] = []
105
+ if mode == "default":
106
+ for child in _element_children(root):
107
+ if local_name(child) != "div" or _class_tokens(child) != {WIRE_BLOCK_CLASS}:
108
+ raise NonCanonicalWire
109
+ wrappers.append((child, None))
110
+ else:
111
+ for child in _element_children(root):
112
+ if local_name(child) == "hr" and _class_tokens(child) == {WIRE_PAGE_BREAK_CLASS}:
113
+ continue
114
+ if local_name(child) != "section" or _class_tokens(child) != {WIRE_PAGE_CLASS}:
115
+ raise NonCanonicalWire
116
+ _validate_structural_text(child)
117
+ page_idx = _non_negative_integer(child, "data-page-idx", required=True)
118
+ for wrapper in _element_children(child):
119
+ if local_name(wrapper) != "div" or _class_tokens(wrapper) != {WIRE_BLOCK_CLASS}:
120
+ raise NonCanonicalWire
121
+ wrappers.append((wrapper, page_idx))
122
+ nested_wrappers = [
123
+ element
124
+ for element in root.iterdescendants()
125
+ if isinstance(element.tag, str) and WIRE_BLOCK_CLASS in _class_tokens(element)
126
+ ]
127
+ if len(nested_wrappers) != len(wrappers):
128
+ raise NonCanonicalWire
129
+ target_ids = _collect_anchor_target_ids(wrappers)
130
+ blocks = tuple(_parse_top_block(wrapper, section_page_idx, target_ids) for wrapper, section_page_idx in wrappers)
131
+ return DocVortexHtmlWirePlan(root, mode, blocks)
132
+
133
+
134
+ def _validate_wire_root_ownership(body: etree._Element, root: etree._Element) -> None:
135
+ """要求 canonical wire 根独占从自身到 body 的可见内容路径。"""
136
+ current = root
137
+ while current is not body:
138
+ parent = current.getparent()
139
+ if parent is None or (parent.text or "").strip():
140
+ raise NonCanonicalWire
141
+ for sibling in parent:
142
+ if sibling is current:
143
+ if (sibling.tail or "").strip():
144
+ raise NonCanonicalWire
145
+ continue
146
+ if isinstance(sibling.tag, str) or (sibling.tail or "").strip():
147
+ raise NonCanonicalWire
148
+ current = parent
149
+
150
+
151
+ def _parse_top_block(
152
+ wrapper: etree._Element,
153
+ section_page_idx: int | None,
154
+ target_ids: dict[str, frozenset[str]],
155
+ ) -> PageWireSpec:
156
+ """解析一个顶层 wrapper,并构造与 block 家族匹配的 typed spec。"""
157
+ _validate_structural_text(wrapper)
158
+ block_type = _page_block_type(wrapper)
159
+ page_idx = _non_negative_integer(wrapper, "data-page-idx", required=True)
160
+ if section_page_idx is not None and page_idx != section_page_idx:
161
+ raise NonCanonicalWire
162
+ block_index = _non_negative_integer(wrapper, "data-block-index", required=False)
163
+ _validate_top_metadata(wrapper, block_type)
164
+ roots = _element_children(wrapper)
165
+ if len(roots) != 1:
166
+ raise NonCanonicalWire
167
+ content_root = roots[0]
168
+ if block_type in _SIMPLE_TEXT_TYPES:
169
+ _validate_simple_content(content_root, block_type)
170
+ return TextWireSpec(wrapper, content_root, block_type, page_idx, block_index)
171
+ if block_type == BlockType.EQUATION:
172
+ _validate_equation_content(content_root)
173
+ return EquationWireSpec(wrapper, content_root, page_idx, block_index)
174
+ if block_type == BlockType.LIST:
175
+ root = _parse_list_container(content_root, top_wrapper=wrapper)
176
+ return ListBlockWireSpec(wrapper, page_idx, block_index, root)
177
+ if block_type == BlockType.INDEX:
178
+ root = _parse_index_root(content_root, wrapper, target_ids)
179
+ return IndexBlockWireSpec(wrapper, page_idx, block_index, root)
180
+ if block_type in VISUAL_TYPE_MAPPING:
181
+ return _parse_visual_content(wrapper, content_root, block_type, page_idx, block_index)
182
+ raise NonCanonicalWire
183
+
184
+
185
+ def _validate_top_metadata(wrapper: etree._Element, block_type: BlockType) -> None:
186
+ """校验 subtype、语言、anchor 和 level 只出现在 renderer 定义的位置。"""
187
+ sub_type = (wrapper.get("data-block-sub-type") or "").strip()
188
+ guess_lang = (wrapper.get("data-guess-lang") or "").strip()
189
+ anchor = (wrapper.get("data-anchor") or "").strip()
190
+ level = _optional_integer(wrapper, "data-level")
191
+ if block_type == BlockType.CODE:
192
+ if sub_type not in {BlockType.CODE, RAW_ALGORITHM}:
193
+ raise NonCanonicalWire
194
+ if sub_type == BlockType.CODE and not guess_lang:
195
+ raise NonCanonicalWire
196
+ if sub_type == RAW_ALGORITHM and guess_lang:
197
+ raise NonCanonicalWire
198
+ elif block_type == BlockType.LIST:
199
+ if sub_type and sub_type not in _LIST_LEAF_TYPES:
200
+ raise NonCanonicalWire
201
+ if guess_lang:
202
+ raise NonCanonicalWire
203
+ elif block_type in {BlockType.IMAGE, BlockType.CHART}:
204
+ if guess_lang:
205
+ raise NonCanonicalWire
206
+ elif sub_type or guess_lang:
207
+ raise NonCanonicalWire
208
+ if block_type == BlockType.DOC_TITLE:
209
+ if level != 1:
210
+ raise NonCanonicalWire
211
+ elif block_type == BlockType.PARAGRAPH_TITLE:
212
+ if level is None or not 2 <= level <= 6:
213
+ raise NonCanonicalWire
214
+ elif level is not None:
215
+ raise NonCanonicalWire
216
+ if anchor and block_type not in {
217
+ BlockType.TEXT,
218
+ BlockType.DOC_TITLE,
219
+ BlockType.PARAGRAPH_TITLE,
220
+ BlockType.PAGE_FOOTNOTE,
221
+ }:
222
+ raise NonCanonicalWire
223
+
224
+
225
+ def _validate_simple_content(content_root: etree._Element, block_type: BlockType) -> None:
226
+ """校验文本、标题、脚注和页面辅助 block 的固定行内容器。"""
227
+ expected_tags: dict[BlockType, frozenset[str]] = {
228
+ BlockType.TEXT: frozenset({"p"}),
229
+ BlockType.REF_TEXT: frozenset({"p"}),
230
+ BlockType.DOC_TITLE: frozenset({"h1"}),
231
+ BlockType.PARAGRAPH_TITLE: frozenset({"h2", "h3", "h4", "h5", "h6"}),
232
+ BlockType.PAGE_FOOTNOTE: frozenset({"div"}),
233
+ **{value: frozenset({"div"}) for value in _PAGE_AUXILIARY_TYPES},
234
+ }
235
+ if local_name(content_root) not in expected_tags[block_type]:
236
+ raise NonCanonicalWire
237
+ _validate_inline_region(content_root)
238
+
239
+
240
+ def _validate_equation_content(content_root: etree._Element) -> None:
241
+ """校验行间公式为 renderer math carrier 或公式图片。"""
242
+ if local_name(content_root) == "math":
243
+ _validate_formula_carrier(content_root, expected_display="block")
244
+ return
245
+ if local_name(content_root) != "img" or _class_tokens(content_root) != {"docvortex-equation-image"}:
246
+ raise NonCanonicalWire
247
+
248
+
249
+ def _validate_formula_carrier(element: etree._Element, *, expected_display: str) -> None:
250
+ """校验公式 carrier 的类型、显示模式和非空 LaTeX。"""
251
+ if local_name(element) != "math" or (element.get("data-block-type") or "").strip() != BlockType.EQUATION:
252
+ raise NonCanonicalWire
253
+ if (element.get("data-formula-display") or "").strip() != expected_display:
254
+ raise NonCanonicalWire
255
+ formula = extract_formula(element)
256
+ if formula is None or not formula.latex:
257
+ raise NonCanonicalWire
258
+
259
+
260
+ def _parse_visual_content(
261
+ wrapper: etree._Element,
262
+ content_root: etree._Element,
263
+ parent_type: BlockType,
264
+ page_idx: int,
265
+ block_index: int | None,
266
+ ) -> VisualWireSpec:
267
+ """解析 figure 下唯一 body 与有序 annotation 子节点。"""
268
+ if local_name(content_root) != "figure":
269
+ raise NonCanonicalWire
270
+ _validate_structural_text(content_root)
271
+ mapping = VISUAL_TYPE_MAPPING[parent_type]
272
+ sub_type = (wrapper.get("data-block-sub-type") or "").strip()
273
+ body_type = BlockType.ALGORITHM_BODY if parent_type == BlockType.CODE and sub_type == RAW_ALGORITHM else mapping["body"]
274
+ allowed_types = frozenset({body_type, mapping["caption"], mapping["footnote"]})
275
+ children = _element_children(content_root)
276
+ child_types = [(child.get("data-block-type") or "").strip() for child in children]
277
+ if child_types.count(body_type) != 1 or any(value not in allowed_types for value in child_types):
278
+ raise NonCanonicalWire
279
+ guess_lang = (wrapper.get("data-guess-lang") or "").strip()
280
+ parsed_children: list[VisualBodyWireSpec | AnnotationWireSpec] = []
281
+ for child, child_type in zip(children, child_types, strict=True):
282
+ child_index = _non_negative_integer(child, "data-block-index", required=False)
283
+ if child_type == body_type:
284
+ if local_name(child) != "div" or (block_index is not None and child_index != block_index):
285
+ raise NonCanonicalWire
286
+ parsed_children.append(_parse_visual_body(child, parent_type, sub_type))
287
+ continue
288
+ if local_name(child) != "p":
289
+ raise NonCanonicalWire
290
+ _validate_inline_region(child)
291
+ parsed_children.append(AnnotationWireSpec(child, BlockType(child_type)))
292
+ return VisualWireSpec(
293
+ wrapper,
294
+ content_root,
295
+ parent_type,
296
+ page_idx,
297
+ block_index,
298
+ sub_type,
299
+ guess_lang,
300
+ tuple(parsed_children),
301
+ )
302
+
303
+
304
+ def _parse_visual_body(body: etree._Element, parent_type: BlockType, sub_type: str) -> VisualBodyWireSpec:
305
+ """按父 visual 类型解析唯一 canonical body 载荷。"""
306
+ expected_class = f"{WIRE_VISUAL_BODY_CLASS}--{'image' if parent_type == BlockType.IMAGE else str(parent_type)}"
307
+ if _class_tokens(body) != {WIRE_VISUAL_BODY_CLASS, expected_class}:
308
+ raise NonCanonicalWire
309
+ if parent_type == BlockType.CODE:
310
+ return _parse_code_body(body, sub_type)
311
+ if parent_type == BlockType.TABLE:
312
+ return _parse_table_body(body)
313
+ if parent_type == BlockType.IMAGE and _looks_like_flowchart_body(body):
314
+ return _parse_flowchart_body(body)
315
+ return _parse_rich_visual_body(body, parent_type, sub_type)
316
+
317
+
318
+ def _parse_code_body(body: etree._Element, sub_type: str) -> CodeBodyWireSpec:
319
+ """解析普通代码或 algorithm 的固定内容载体。"""
320
+ _validate_structural_text(body)
321
+ children = _element_children(body)
322
+ if sub_type == BlockType.CODE:
323
+ if len(children) != 1 or local_name(children[0]) != "pre":
324
+ raise NonCanonicalWire
325
+ pre = children[0]
326
+ _validate_structural_text(pre)
327
+ code_children = _element_children(pre)
328
+ if len(code_children) != 1 or local_name(code_children[0]) != "code" or _element_children(code_children[0]):
329
+ raise NonCanonicalWire
330
+ return CodeBodyWireSpec(body, "code", code_children[0])
331
+ if sub_type != RAW_ALGORITHM:
332
+ raise NonCanonicalWire
333
+ if not children:
334
+ empty = etree.Element("div")
335
+ return CodeBodyWireSpec(body, "algorithm", empty)
336
+ if len(children) != 1 or local_name(children[0]) != "div" or _class_tokens(children[0]) != {"docvortex-algorithm"}:
337
+ raise NonCanonicalWire
338
+ _validate_inline_region(children[0])
339
+ return CodeBodyWireSpec(body, "algorithm", _clone_fragment(children[0]))
340
+
341
+
342
+ def _parse_table_body(body: etree._Element) -> TableBodyWireSpec:
343
+ """解析结构表格、空间文本、图片或空 table body。"""
344
+ _validate_structural_text(body)
345
+ children = _element_children(body)
346
+ if not children:
347
+ return TableBodyWireSpec(body, "empty", None)
348
+ if len(children) != 1:
349
+ raise NonCanonicalWire
350
+ child = children[0]
351
+ name = local_name(child)
352
+ classes = _class_tokens(child)
353
+ if name == "table":
354
+ return TableBodyWireSpec(body, "html", child)
355
+ if name == "pre" and classes in ({"docvortex-table-text"}, {"docvortex-raw-fallback"}) and not _element_children(child):
356
+ return TableBodyWireSpec(body, "text", child)
357
+ if name == "img" and classes == {"docvortex-table-image"}:
358
+ return TableBodyWireSpec(body, "image", child)
359
+ raise NonCanonicalWire
360
+
361
+
362
+ def _looks_like_flowchart_body(body: etree._Element) -> bool:
363
+ """判断 body 是否使用 renderer 的 flowchart 固定外壳。"""
364
+ children = _element_children(body)
365
+ return bool(children and local_name(children[0]) == "div" and "docvortex-flowchart" in _class_tokens(children[0]))
366
+
367
+
368
+ def _parse_flowchart_body(body: etree._Element) -> FlowchartBodyWireSpec:
369
+ """解析 flowchart canvas、可选 raster 和源码 details。"""
370
+ _validate_structural_text(body)
371
+ children = _element_children(body)
372
+ if len(children) != 2:
373
+ raise NonCanonicalWire
374
+ display, details = children
375
+ display_classes = _class_tokens(display)
376
+ if local_name(display) != "div" or "docvortex-flowchart" not in display_classes:
377
+ raise NonCanonicalWire
378
+ _validate_structural_text(display)
379
+ display_children = _element_children(display)
380
+ if not 1 <= len(display_children) <= 2:
381
+ raise NonCanonicalWire
382
+ canvas = display_children[0]
383
+ if local_name(canvas) != "div" or _class_tokens(canvas) != {"docvortex-flowchart-canvas"}:
384
+ raise NonCanonicalWire
385
+ _validate_structural_text(canvas)
386
+ if _element_children(canvas):
387
+ raise NonCanonicalWire
388
+ fallback_image = None
389
+ if len(display_children) == 2:
390
+ fallback_image = display_children[1]
391
+ if local_name(fallback_image) != "img" or _class_tokens(fallback_image) != {"docvortex-flowchart-fallback"}:
392
+ raise NonCanonicalWire
393
+ if local_name(details) != "details" or _class_tokens(details) != {"docvortex-details", "docvortex-flowchart-details"}:
394
+ raise NonCanonicalWire
395
+ _validate_structural_text(details)
396
+ details_children = _element_children(details)
397
+ if len(details_children) != 2:
398
+ raise NonCanonicalWire
399
+ summary, source = details_children
400
+ if (
401
+ local_name(summary) != "summary"
402
+ or _element_children(summary)
403
+ or " ".join(summary.itertext()).strip() != "flowchart source"
404
+ ):
405
+ raise NonCanonicalWire
406
+ if local_name(source) != "pre" or _class_tokens(source) != {"docvortex-flowchart-source"}:
407
+ raise NonCanonicalWire
408
+ _validate_structural_text(source)
409
+ code_children = _element_children(source)
410
+ if len(code_children) != 1 or local_name(code_children[0]) != "code" or _element_children(code_children[0]):
411
+ raise NonCanonicalWire
412
+ return FlowchartBodyWireSpec(body, code_children[0], fallback_image)
413
+
414
+
415
+ def _parse_rich_visual_body(body: etree._Element, parent_type: BlockType, sub_type: str) -> RichVisualBodyWireSpec:
416
+ """区分 renderer-owned 主图与开放但受 sanitizer 约束的富内容 carrier。"""
417
+ allowed_token = "docvortex-image" if parent_type == BlockType.IMAGE else "docvortex-chart-image"
418
+ children = _element_children(body)
419
+ primary_image = (
420
+ children[0] if children and local_name(children[0]) == "img" and allowed_token in _class_tokens(children[0]) else None
421
+ )
422
+ if primary_image is not None and _class_tokens(primary_image) != {allowed_token}:
423
+ raise NonCanonicalWire
424
+ for element in body.iterdescendants():
425
+ if not isinstance(element.tag, str) or element is primary_image:
426
+ continue
427
+ if _class_tokens(element) & _OWNED_VISUAL_IMAGE_TOKENS:
428
+ raise NonCanonicalWire
429
+ if primary_image is None:
430
+ return RichVisualBodyWireSpec(body, parent_type, sub_type, None, _clone_fragment(body))
431
+ if (body.text or "").strip() or (primary_image.tail or "").strip():
432
+ raise NonCanonicalWire
433
+ remaining = children[1:]
434
+ if not remaining:
435
+ return RichVisualBodyWireSpec(body, parent_type, sub_type, primary_image, None)
436
+ if len(remaining) != 1:
437
+ raise NonCanonicalWire
438
+ details = remaining[0]
439
+ if local_name(details) != "details" or _class_tokens(details) != {"docvortex-details"}:
440
+ raise NonCanonicalWire
441
+ if (details.text or "").strip() or (details.tail or "").strip():
442
+ raise NonCanonicalWire
443
+ details_children = _element_children(details)
444
+ if not details_children or local_name(details_children[0]) != "summary" or _element_children(details_children[0]):
445
+ raise NonCanonicalWire
446
+ expected_summary = sub_type or ("image content" if parent_type == BlockType.IMAGE else "chart content")
447
+ if " ".join(details_children[0].itertext()).strip() != expected_summary:
448
+ raise NonCanonicalWire
449
+ fragment = _clone_fragment(details, after_child=details_children[0])
450
+ return RichVisualBodyWireSpec(body, parent_type, sub_type, primary_image, fragment)
451
+
452
+
453
+ def _parse_list_container(container: etree._Element, *, top_wrapper: etree._Element | None = None) -> ListWireSpec:
454
+ """递归解析 renderer 生成的列表 carrier、叶子和嵌套列表。"""
455
+ if local_name(container) not in {"ol", "ul"} or (container.get("data-block-type") or "").strip() != BlockType.LIST:
456
+ raise NonCanonicalWire
457
+ _validate_structural_text(container)
458
+ block_index = _non_negative_integer(container, "data-block-index", required=False)
459
+ sub_type = (container.get("data-block-sub-type") or "").strip()
460
+ if sub_type and sub_type not in _LIST_LEAF_TYPES:
461
+ raise NonCanonicalWire
462
+ if top_wrapper is not None and sub_type != (top_wrapper.get("data-block-sub-type") or "").strip():
463
+ raise NonCanonicalWire
464
+ classes = _class_tokens(container)
465
+ if "docvortex-list" not in classes or len(classes) != 2:
466
+ raise NonCanonicalWire
467
+ children: list[ListLeafWireSpec | ListWireSpec] = []
468
+ for item in _element_children(container):
469
+ if local_name(item) != "li":
470
+ raise NonCanonicalWire
471
+ item_type = (item.get("data-block-type") or "").strip()
472
+ nested_lists = [child for child in _element_children(item) if local_name(child) in {"ol", "ul"}]
473
+ if item_type:
474
+ if item_type not in _LIST_LEAF_TYPES:
475
+ raise NonCanonicalWire
476
+ leaf = _parse_list_leaf(item, BlockType(item_type), nested_lists)
477
+ children.append(leaf)
478
+ elif any(child not in nested_lists for child in _element_children(item)) or not nested_lists:
479
+ raise NonCanonicalWire
480
+ children.extend(_parse_list_container(nested) for nested in nested_lists)
481
+ start = _canonical_list_start(container)
482
+ return ListWireSpec(container, block_index, local_name(container) == "ol", start, sub_type, classes, tuple(children))
483
+
484
+
485
+ def _parse_list_leaf(
486
+ item: etree._Element,
487
+ block_type: BlockType,
488
+ nested_lists: list[etree._Element],
489
+ ) -> ListLeafWireSpec:
490
+ """解析一个列表叶子的唯一 marker/content carrier。"""
491
+ block_index = _non_negative_integer(item, "data-block-index", required=False)
492
+ candidates = [child for child in _element_children(item) if child not in nested_lists]
493
+ content_carriers = [child for child in candidates if _class_tokens(child) == {WIRE_LIST_CONTENT_CLASS}]
494
+ marker_carriers = [child for child in candidates if _class_tokens(child) == {WIRE_LIST_MARKER_CLASS}]
495
+ if len(content_carriers) > 1 or len(marker_carriers) > 1:
496
+ raise NonCanonicalWire
497
+ allowed = [*content_carriers, *marker_carriers, *nested_lists]
498
+ if any(child not in allowed for child in _element_children(item)):
499
+ raise NonCanonicalWire
500
+ if content_carriers:
501
+ _validate_structural_text(item)
502
+ content_element = content_carriers[0]
503
+ if local_name(content_element) != "span":
504
+ raise NonCanonicalWire
505
+ _validate_inline_region(content_element)
506
+ else:
507
+ if marker_carriers or (item.text or "").strip() or any((child.tail or "").strip() for child in item):
508
+ raise NonCanonicalWire
509
+ content_element = None
510
+ marker = ""
511
+ if marker_carriers:
512
+ marker_element = marker_carriers[0]
513
+ if local_name(marker_element) != "span" or _element_children(marker_element):
514
+ raise NonCanonicalWire
515
+ marker = "".join(marker_element.itertext()).strip()
516
+ return ListLeafWireSpec(block_type, block_index, content_element, marker)
517
+
518
+
519
+ def _parse_index_root(
520
+ content_root: etree._Element,
521
+ wrapper: etree._Element,
522
+ target_ids: dict[str, frozenset[str]],
523
+ ) -> IndexWireSpec:
524
+ """解析目录根与唯一直属 ul。"""
525
+ if local_name(content_root) != "nav" or _class_tokens(content_root) != {WIRE_INDEX_CLASS}:
526
+ raise NonCanonicalWire
527
+ if (content_root.get("data-block-type") or "").strip() != BlockType.INDEX:
528
+ raise NonCanonicalWire
529
+ _validate_structural_text(content_root)
530
+ root_index = _non_negative_integer(content_root, "data-block-index", required=False)
531
+ wrapper_index = _non_negative_integer(wrapper, "data-block-index", required=False)
532
+ if wrapper_index is not None and root_index != wrapper_index:
533
+ raise NonCanonicalWire
534
+ lists = _element_children(content_root)
535
+ if len(lists) != 1 or local_name(lists[0]) != "ul":
536
+ raise NonCanonicalWire
537
+ return _parse_index_list(lists[0], nested=False, target_ids=target_ids, block_index=root_index)
538
+
539
+
540
+ def _parse_index_list(
541
+ container: etree._Element,
542
+ *,
543
+ nested: bool,
544
+ target_ids: dict[str, frozenset[str]],
545
+ block_index: int | None = None,
546
+ ) -> IndexWireSpec:
547
+ """递归解析目录叶子、linked carrier 和嵌套 IndexBlock。"""
548
+ _validate_structural_text(container)
549
+ if nested:
550
+ if (container.get("data-block-type") or "").strip() != BlockType.INDEX:
551
+ raise NonCanonicalWire
552
+ block_index = _non_negative_integer(container, "data-block-index", required=False)
553
+ children: list[IndexLeafWireSpec | IndexWireSpec] = []
554
+ for item in _element_children(container):
555
+ if local_name(item) != "li":
556
+ raise NonCanonicalWire
557
+ item_type_value = (item.get("data-block-type") or "").strip()
558
+ nested_lists = [child for child in _element_children(item) if local_name(child) == "ul"]
559
+ if item_type_value:
560
+ try:
561
+ item_type = BlockType(item_type_value)
562
+ except ValueError as exc:
563
+ raise NonCanonicalWire from exc
564
+ if item_type not in _INDEX_LEAF_TYPES:
565
+ raise NonCanonicalWire
566
+ children.append(_parse_index_leaf(item, item_type, nested_lists, target_ids))
567
+ elif any(child not in nested_lists for child in _element_children(item)) or not nested_lists:
568
+ raise NonCanonicalWire
569
+ children.extend(_parse_index_list(nested_list, nested=True, target_ids=target_ids) for nested_list in nested_lists)
570
+ return IndexWireSpec(container, block_index, tuple(children))
571
+
572
+
573
+ def _parse_index_leaf(
574
+ item: etree._Element,
575
+ item_type: BlockType,
576
+ nested_lists: list[etree._Element],
577
+ target_ids: dict[str, frozenset[str]],
578
+ ) -> IndexLeafWireSpec:
579
+ """解析 linked/unlinked 目录叶子并封闭 anchor 外结构。"""
580
+ block_index = _non_negative_integer(item, "data-block-index", required=False)
581
+ anchor = (item.get("data-anchor") or "").strip()
582
+ level = _optional_integer(item, "data-level")
583
+ _validate_index_leaf_metadata(item_type, anchor, level)
584
+ direct_content = [child for child in _element_children(item) if child not in nested_lists]
585
+ linked = [
586
+ child
587
+ for child in direct_content
588
+ if _is_canonical_index_link(child, item_type=item_type, anchor=anchor, target_ids=target_ids)
589
+ ]
590
+ if linked:
591
+ if len(linked) != 1 or len(direct_content) != 1:
592
+ raise NonCanonicalWire
593
+ _validate_structural_text(item)
594
+ content_element = linked[0]
595
+ _validate_inline_region(content_element)
596
+ else:
597
+ content_element = _clone_fragment(item, excluded_children=nested_lists)
598
+ _validate_inline_region(content_element)
599
+ return IndexLeafWireSpec(item_type, block_index, content_element, anchor, level)
600
+
601
+
602
+ def _collect_anchor_target_ids(
603
+ wrappers: list[tuple[etree._Element, int | None]],
604
+ ) -> dict[str, frozenset[str]]:
605
+ """预收集 renderer 正文和标题 id,供目录 linked carrier 做确定性判定。"""
606
+ collected: dict[str, set[str]] = {}
607
+ for wrapper, _ in wrappers:
608
+ block_type = (wrapper.get("data-block-type") or "").strip()
609
+ if block_type not in {BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE}:
610
+ continue
611
+ anchor = (wrapper.get("data-anchor") or "").strip()
612
+ children = _element_children(wrapper)
613
+ if not anchor or len(children) != 1:
614
+ continue
615
+ identities = {
616
+ identity
617
+ for element in [children[0], *children[0].iterdescendants()]
618
+ if (identity := (element.get("id") or "").strip())
619
+ }
620
+ if identities:
621
+ collected.setdefault(anchor, set()).update(identities)
622
+ return {anchor: frozenset(identities) for anchor, identities in collected.items()}
623
+
624
+
625
+ def _is_canonical_index_link(
626
+ element: etree._Element,
627
+ *,
628
+ item_type: BlockType,
629
+ anchor: str,
630
+ target_ids: dict[str, frozenset[str]],
631
+ ) -> bool:
632
+ """判断直属 anchor 是否为 renderer 生成的目录目标外壳。"""
633
+ if item_type not in {BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE} or local_name(element) != "a":
634
+ return False
635
+ href = (element.get("href") or "").strip()
636
+ if not href.startswith("#"):
637
+ return False
638
+ return unquote(href[1:]).strip() in target_ids.get(anchor, frozenset())
639
+
640
+
641
+ def _validate_index_leaf_metadata(item_type: BlockType, anchor: str, level: int | None) -> None:
642
+ """校验目录叶子的 anchor 与标题 level 组合。"""
643
+ if item_type == BlockType.TEXT:
644
+ if level is not None:
645
+ raise NonCanonicalWire
646
+ return
647
+ if not anchor:
648
+ raise NonCanonicalWire
649
+ if item_type == BlockType.DOC_TITLE and level != 1:
650
+ raise NonCanonicalWire
651
+ if item_type == BlockType.PARAGRAPH_TITLE and (level is None or not 2 <= level <= 6):
652
+ raise NonCanonicalWire
653
+
654
+
655
+ def _validate_inline_region(element: etree._Element) -> None:
656
+ """校验 canonical 行内区域只含行内节点和可信公式 carrier。"""
657
+ for candidate in element.iterdescendants():
658
+ if not isinstance(candidate.tag, str):
659
+ continue
660
+ name = local_name(candidate)
661
+ marker_type = (candidate.get("data-block-type") or "").strip()
662
+ if name == "math" and marker_type == BlockType.EQUATION:
663
+ _validate_formula_carrier(candidate, expected_display="inline")
664
+ continue
665
+ if name in BLOCK_TAGS or name in {"image", "img"}:
666
+ raise NonCanonicalWire
667
+
668
+
669
+ def _clone_fragment(
670
+ element: etree._Element,
671
+ *,
672
+ after_child: etree._Element | None = None,
673
+ excluded_children: list[etree._Element] | None = None,
674
+ ) -> etree._Element:
675
+ """复制一个不含 renderer 外壳的富内容片段供 materializer 使用。"""
676
+ fragment = etree.Element("div")
677
+ excluded = excluded_children or []
678
+ children = _element_children(element)
679
+ start = 0
680
+ if after_child is None:
681
+ fragment.text = element.text
682
+ else:
683
+ try:
684
+ start = children.index(after_child) + 1
685
+ except ValueError as exc:
686
+ raise NonCanonicalWire from exc
687
+ fragment.text = after_child.tail
688
+ for child in children[start:]:
689
+ if child in excluded:
690
+ continue
691
+ fragment.append(deepcopy(child))
692
+ return fragment
693
+
694
+
695
+ def _validate_structural_text(element: etree._Element) -> None:
696
+ """拒绝 renderer 结构容器直属的非空文本和 tail。"""
697
+ if (element.text or "").strip() or any((child.tail or "").strip() for child in element):
698
+ raise NonCanonicalWire
699
+
700
+
701
+ def _page_block_type(wrapper: etree._Element) -> BlockType:
702
+ """把顶层 data-block-type 转换为公开 PageBlock 类型。"""
703
+ try:
704
+ block_type = BlockType((wrapper.get("data-block-type") or "").strip())
705
+ except ValueError as exc:
706
+ raise NonCanonicalWire from exc
707
+ if block_type not in PAGE_BLOCK_TYPES:
708
+ raise NonCanonicalWire
709
+ return block_type
710
+
711
+
712
+ def _element_children(element: etree._Element) -> list[etree._Element]:
713
+ """返回元素直属的真实标签子节点。"""
714
+ return [child for child in element if isinstance(child.tag, str)]
715
+
716
+
717
+ def _class_tokens(element: etree._Element) -> frozenset[str]:
718
+ """按 HTML class 空白边界返回完整小写 token。"""
719
+ return frozenset((element.get("class") or "").casefold().split())
720
+
721
+
722
+ def _non_negative_integer(element: etree._Element, name: str, *, required: bool) -> int | None:
723
+ """读取 canonical 非负整数 data 属性。"""
724
+ value = element.get(name)
725
+ if value is None:
726
+ if required:
727
+ raise NonCanonicalWire
728
+ return None
729
+ try:
730
+ parsed = int(value)
731
+ except (TypeError, ValueError) as exc:
732
+ raise NonCanonicalWire from exc
733
+ if parsed < 0 or str(parsed) != value.strip():
734
+ raise NonCanonicalWire
735
+ return parsed
736
+
737
+
738
+ def _optional_integer(element: etree._Element, name: str) -> int | None:
739
+ """读取可选整数属性并拒绝非法文本。"""
740
+ value = element.get(name)
741
+ if value is None:
742
+ return None
743
+ try:
744
+ return int(value)
745
+ except (TypeError, ValueError) as exc:
746
+ raise NonCanonicalWire from exc
747
+
748
+
749
+ def _canonical_list_start(element: etree._Element) -> int:
750
+ """读取 renderer 生成的合法有序列表起始值。"""
751
+ value = element.get("start")
752
+ if value is None:
753
+ return 1
754
+ try:
755
+ parsed = int(value)
756
+ except ValueError as exc:
757
+ raise NonCanonicalWire from exc
758
+ if parsed < 0 or str(parsed) != value.strip():
759
+ raise NonCanonicalWire
760
+ return parsed
761
+
762
+
763
+ __all__ = ["NonCanonicalWire", "parse_docvortex_html_wire"]