docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1237 @@
1
+ """把静态 XHTML/HTML DOM 投影为 DocVortex raw blocks。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import html
6
+ import re
7
+ from dataclasses import dataclass
8
+ from typing import Protocol, TypeAlias
9
+
10
+ from lxml import etree # type: ignore[reportMissingImports]
11
+
12
+ from docvortex.schema import RAW_ALGORITHM, BlockType, VISUAL_TYPE_MAPPING
13
+ from docvortex.content.spans import (
14
+ append_code_span,
15
+ append_equation_span,
16
+ append_hyperlink_span,
17
+ append_text_span,
18
+ extend_inline_spans,
19
+ inline_span_plain_text,
20
+ strip_span_dicts,
21
+ text_spans,
22
+ )
23
+ from docvortex.foundation.xml_names import local_name
24
+ from docvortex.content.markup.formula import FormulaExtraction, extract_formula
25
+ from docvortex.content.markup.styles import MarkupStylesheet, TextStyle
26
+ from docvortex.foundation.type_identity import preserve_type_module
27
+
28
+
29
+ BLOCK_TAGS = frozenset(
30
+ {
31
+ "address",
32
+ "article",
33
+ "aside",
34
+ "blockquote",
35
+ "body",
36
+ "dd",
37
+ "details",
38
+ "div",
39
+ "dl",
40
+ "dt",
41
+ "figcaption",
42
+ "figure",
43
+ "footer",
44
+ "h1",
45
+ "h2",
46
+ "h3",
47
+ "h4",
48
+ "h5",
49
+ "h6",
50
+ "header",
51
+ "hr",
52
+ "main",
53
+ "math",
54
+ "nav",
55
+ "ol",
56
+ "p",
57
+ "pre",
58
+ "section",
59
+ "summary",
60
+ "svg",
61
+ "table",
62
+ "ul",
63
+ }
64
+ )
65
+ SKIPPED_TAGS = frozenset(
66
+ {
67
+ "audio",
68
+ "button",
69
+ "canvas",
70
+ "embed",
71
+ "form",
72
+ "head",
73
+ "iframe",
74
+ "input",
75
+ "noscript",
76
+ "object",
77
+ "script",
78
+ "select",
79
+ "style",
80
+ "template",
81
+ "textarea",
82
+ "video",
83
+ }
84
+ )
85
+ _WHITESPACE_RE = re.compile(r"[\t\r\n\f ]+")
86
+ _XLINK_HREF = "{http://www.w3.org/1999/xlink}href"
87
+ _MAX_TABLE_SPAN = 1_000
88
+ _CAPTION_TOKENS = frozenset(
89
+ {
90
+ "caption",
91
+ "figure-caption",
92
+ "image-caption",
93
+ "table-caption",
94
+ "chart-caption",
95
+ "code-caption",
96
+ "docvortex-caption",
97
+ }
98
+ )
99
+ _FOOTNOTE_TOKENS = frozenset(
100
+ {
101
+ "footnote",
102
+ "figure-footnote",
103
+ "image-footnote",
104
+ "table-footnote",
105
+ "chart-footnote",
106
+ "code-footnote",
107
+ "docvortex-footnote",
108
+ }
109
+ )
110
+ _VISUAL_ELEMENT_TAGS = frozenset({"img", "image", "pre", "svg", "table"})
111
+ _LIST_PAGE_BLOCK_TAGS = frozenset({"figure", "image", "img", "math", "pre", "svg", "table"})
112
+ _InlineSpanDict: TypeAlias = dict[str, object]
113
+ _InlineProjectionSegment: TypeAlias = list[_InlineSpanDict] | dict[str, object]
114
+
115
+
116
+ def clean_text_node(value: str | None) -> str:
117
+ """折叠普通标记文档文本节点中的排版空白。"""
118
+ return _WHITESPACE_RE.sub(" ", value) if value else ""
119
+
120
+
121
+ def visible_text(element: etree._Element) -> str:
122
+ """提取元素折叠空白后的可见纯文本。"""
123
+ return _WHITESPACE_RE.sub(" ", html.unescape("".join(element.itertext()))).strip()
124
+
125
+
126
+ def _semantic_tokens(element: etree._Element) -> frozenset[str]:
127
+ """按 class/id 的完整空白 token 返回小写集合,不执行任意 substring 匹配。"""
128
+ value = f"{element.get('class') or ''} {element.get('id') or ''}".casefold()
129
+ return frozenset(value.split())
130
+
131
+
132
+ def _raw_visual_type(value: object) -> BlockType | None:
133
+ """把 raw visual 主体或 algorithm 规范为统一父块类型。"""
134
+ if value in {BlockType.IMAGE, BlockType.TABLE, BlockType.CHART}:
135
+ return BlockType(value)
136
+ if value in {BlockType.CODE, RAW_ALGORITHM}:
137
+ return BlockType.CODE
138
+ return None
139
+
140
+
141
+ def _append_inline_segment(
142
+ segments: list[_InlineProjectionSegment],
143
+ segment: _InlineProjectionSegment,
144
+ ) -> None:
145
+ """追加行内投影片段,并合并相邻 Span 组以保持稳定 block 粒度。"""
146
+ if isinstance(segment, list) and segments and isinstance(segments[-1], list):
147
+ extend_inline_spans(segments[-1], segment)
148
+ elif not isinstance(segment, list) or segment:
149
+ segments.append(segment)
150
+
151
+
152
+ def _append_list_block_content(parts: list[_InlineSpanDict], rendered: list[_InlineSpanDict]) -> None:
153
+ """用换行包围列表项内的块级正文,避免相邻段落静默粘连。"""
154
+ if not rendered:
155
+ return
156
+ last_visible = inline_span_plain_text(parts)
157
+ if last_visible and not last_visible.endswith("\n"):
158
+ append_text_span(parts, "\n")
159
+ extend_inline_spans(parts, rendered)
160
+ append_text_span(parts, "\n")
161
+
162
+
163
+ def entity_text(element: etree._Element) -> str:
164
+ """把 lxml 保留的安全命名实体恢复为可见文本。"""
165
+ name = getattr(element, "name", "")
166
+ return html.unescape(f"&{name};") if name else ""
167
+
168
+
169
+ def bounded_table_span(value: str) -> str | None:
170
+ """规范化有界表格跨度,避免异常整数放大渲染网格。"""
171
+ if not value.isdigit():
172
+ return None
173
+ normalized = value.lstrip("0")
174
+ if not normalized or len(normalized) > len(str(_MAX_TABLE_SPAN)):
175
+ return None
176
+ span = int(normalized)
177
+ return str(span) if span <= _MAX_TABLE_SPAN else None
178
+
179
+
180
+ def visible_raw_text_with_style(
181
+ element: etree._Element,
182
+ stylesheet: MarkupStylesheet,
183
+ style: TextStyle,
184
+ visibility_hidden: bool,
185
+ ) -> str:
186
+ """递归提取遵守整树隐藏和继承 visibility 的原始文本。"""
187
+ parts: list[str] = [] if visibility_hidden else [element.text or ""]
188
+ for child in element:
189
+ if isinstance(child.tag, str):
190
+ resolved = stylesheet.resolve(child, style, visibility_hidden)
191
+ if not resolved.subtree_hidden:
192
+ parts.append(
193
+ visible_raw_text_with_style(
194
+ child,
195
+ stylesheet,
196
+ resolved.text,
197
+ resolved.visibility_hidden,
198
+ )
199
+ )
200
+ elif not visibility_hidden:
201
+ parts.append(entity_text(child))
202
+ if not visibility_hidden:
203
+ parts.append(child.tail or "")
204
+ return "".join(parts)
205
+
206
+
207
+ @dataclass(frozen=True, slots=True)
208
+ class ResolvedMarkupImage:
209
+ """保存标记文档图片解析后的互斥载荷和说明文本。"""
210
+
211
+ image_base64: str | None = None
212
+ image_url: str | None = None
213
+ alt: str = ""
214
+
215
+
216
+ class MarkupContext(Protocol):
217
+ """定义 projector 向具体容器请求链接、图片和 anchor 的边界。"""
218
+
219
+ def resolve_link(self, href: str) -> str | None:
220
+ """解析一个安全链接目标。"""
221
+
222
+ def resolve_image(self, source: str, *, alt: str = "") -> ResolvedMarkupImage | None:
223
+ """解析图片为 data URI、远程 URL 或可见降级文本。"""
224
+
225
+ def heading_anchor(self, heading: etree._Element) -> str | None:
226
+ """返回标题对应的规范 anchor。"""
227
+
228
+ def heading_label(self, anchor: str) -> str | None:
229
+ """返回规范 anchor 对应的标题标签。"""
230
+
231
+ def note_anchor(self, note: etree._Element) -> str | None:
232
+ """返回脚注节点对应的规范 anchor。"""
233
+
234
+
235
+ class MarkupProjector:
236
+ """按 DOM 顺序把一个静态内容根节点投影为统一 raw blocks。"""
237
+
238
+ def __init__(
239
+ self,
240
+ root: etree._Element,
241
+ context: MarkupContext,
242
+ stylesheet: MarkupStylesheet,
243
+ *,
244
+ single_document_title: bool = False,
245
+ document_title_emitted: bool = False,
246
+ ) -> None:
247
+ """绑定 DOM、格式适配器、有限 CSS 和标题策略。"""
248
+ self.root = root
249
+ self.context = context
250
+ self.stylesheet = stylesheet
251
+ self.single_document_title = single_document_title
252
+ self.document_title_emitted = document_title_emitted
253
+
254
+ def convert(self) -> list[dict[str, object]]:
255
+ """转换内容根节点的子树并返回按 DOM 顺序排列的 raw blocks。"""
256
+ resolved = self.stylesheet.resolve(self.root, TextStyle())
257
+ if resolved.subtree_hidden:
258
+ return []
259
+ name = local_name(self.root)
260
+ if name == "figure" or (name in {"aside", "div", "section"} and self._has_contextual_visual_annotation(self.root)):
261
+ return self._parse_figure(self.root, resolved.text, resolved.visibility_hidden)
262
+ return self._parse_container_contents(self.root, resolved.text, resolved.visibility_hidden)
263
+
264
+ def convert_svg(self) -> list[dict[str, object]]:
265
+ """把 standalone SVG 根节点尽力转换为文本和静态图片。"""
266
+ resolved = self.stylesheet.resolve(self.root, TextStyle())
267
+ return [] if resolved.subtree_hidden else self._parse_svg(self.root, resolved.text, resolved.visibility_hidden)
268
+
269
+ def project_block(self, element: etree._Element) -> list[dict[str, object]]:
270
+ """把一个已知块元素按默认继承样式投影,供版本化 HTML 解码复用。"""
271
+ return self._parse_block(element, TextStyle())
272
+
273
+ def project_inline_content(self, element: etree._Element) -> list[_InlineSpanDict]:
274
+ """把一个已知行内容器恢复为结构化 Span。"""
275
+ resolved = self.stylesheet.resolve(element, TextStyle())
276
+ if resolved.subtree_hidden:
277
+ return []
278
+ content, extras = self._render_inline_children(element, resolved.text, resolved.visibility_hidden)
279
+ if extras:
280
+ raise ValueError("inline projection produced unexpected block content")
281
+ return strip_span_dicts(content)
282
+
283
+ def _parse_container_contents(
284
+ self,
285
+ element: etree._Element,
286
+ style: TextStyle,
287
+ visibility_hidden: bool = False,
288
+ ) -> list[dict[str, object]]:
289
+ """把连续行内内容和块级子元素按源顺序拆成 raw blocks。"""
290
+ blocks: list[dict[str, object]] = []
291
+ inline_parts: list[_InlineSpanDict] = []
292
+ if not visibility_hidden:
293
+ extend_inline_spans(inline_parts, self._render_text(element.text, style))
294
+
295
+ def flush_inline() -> None:
296
+ """把当前连续行内片段写为普通正文 block。"""
297
+ content = strip_span_dicts(inline_parts)
298
+ inline_parts.clear()
299
+ if content:
300
+ blocks.append({"type": BlockType.TEXT, "content": content})
301
+
302
+ for child in element:
303
+ if not isinstance(child.tag, str):
304
+ if not visibility_hidden:
305
+ extend_inline_spans(inline_parts, self._render_text(entity_text(child), style))
306
+ extend_inline_spans(inline_parts, self._render_text(child.tail, style))
307
+ continue
308
+ name = local_name(child)
309
+ if name in BLOCK_TAGS:
310
+ flush_inline()
311
+ blocks.extend(self._parse_block(child, style, visibility_hidden))
312
+ else:
313
+ for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
314
+ if isinstance(segment, list):
315
+ extend_inline_spans(inline_parts, segment)
316
+ else:
317
+ flush_inline()
318
+ blocks.append(segment)
319
+ if not visibility_hidden:
320
+ extend_inline_spans(inline_parts, self._render_text(child.tail, style))
321
+ flush_inline()
322
+ return blocks
323
+
324
+ def _parse_block(
325
+ self,
326
+ element: etree._Element,
327
+ inherited: TextStyle,
328
+ inherited_visibility_hidden: bool = False,
329
+ ) -> list[dict[str, object]]:
330
+ """把一个块级元素分派到对应 raw block 转换逻辑。"""
331
+ resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
332
+ if resolved.subtree_hidden:
333
+ return []
334
+ name = local_name(element)
335
+ if name in SKIPPED_TAGS or name == "hr":
336
+ return []
337
+ if self.context.note_anchor(element) is not None:
338
+ return self._parse_note_element(element, resolved.text, resolved.visibility_hidden)
339
+ if name in {"h1", "h2", "h3", "h4", "h5", "h6", "p"}:
340
+ return self._parse_textual_block(element, name, resolved.text, resolved.visibility_hidden)
341
+ if name in {"ul", "ol"}:
342
+ list_block, extras = self._parse_list(element, resolved.text, resolved.visibility_hidden)
343
+ return ([list_block] if list_block is not None else []) + extras
344
+ if name == "table":
345
+ return self._parse_table(element, resolved.text, resolved.visibility_hidden)
346
+ if name == "pre":
347
+ content = self._visible_raw_text(element, resolved.text, resolved.visibility_hidden)
348
+ language = self._code_language_hint(element)
349
+ block: dict[str, object] = {"type": BlockType.CODE, "content": content}
350
+ if language:
351
+ block["guess_lang"] = language
352
+ return [block] if content.strip() else []
353
+ if name == "math":
354
+ if resolved.visibility_hidden:
355
+ return []
356
+ formula = self._formula_extraction(element)
357
+ if formula is not None:
358
+ return [{"type": BlockType.EQUATION, "content": formula.latex}]
359
+ fallback = self._visible_plain_text(element, resolved.text, resolved.visibility_hidden)
360
+ return [{"type": BlockType.TEXT, "content": text_spans(fallback)}] if fallback else []
361
+ if name == "figure":
362
+ return self._parse_figure(element, resolved.text, resolved.visibility_hidden)
363
+ if name in {"aside", "div", "section"} and self._has_contextual_visual_annotation(element):
364
+ return self._parse_figure(element, resolved.text, resolved.visibility_hidden)
365
+ if name == "svg":
366
+ return self._parse_svg(element, resolved.text, resolved.visibility_hidden)
367
+ return self._parse_container_contents(element, resolved.text, resolved.visibility_hidden)
368
+
369
+ def _parse_textual_block(
370
+ self,
371
+ element: etree._Element,
372
+ name: str,
373
+ style: TextStyle,
374
+ visibility_hidden: bool,
375
+ ) -> list[dict[str, object]]:
376
+ """转换标题或段落,并旁路其中的视觉 blocks。"""
377
+ blocks: list[dict[str, object]] = []
378
+ text_emitted = False
379
+ for segment in self._render_inline_children_ordered(element, style, visibility_hidden):
380
+ if not isinstance(segment, list):
381
+ blocks.append(segment)
382
+ continue
383
+ content = strip_span_dicts(segment)
384
+ if not content:
385
+ continue
386
+ if text_emitted:
387
+ blocks.append({"type": BlockType.TEXT, "content": content})
388
+ continue
389
+ if name == "h1" and (not self.single_document_title or not self.document_title_emitted):
390
+ block: dict[str, object] = {"type": BlockType.DOC_TITLE, "level": 1, "content": content}
391
+ self.document_title_emitted = True
392
+ elif name.startswith("h"):
393
+ level = min(max(int(name[1:]), 2), 6)
394
+ block = {
395
+ "type": BlockType.PARAGRAPH_TITLE,
396
+ "level": level,
397
+ "is_numbered_style": False,
398
+ "content": content,
399
+ }
400
+ else:
401
+ block = {"type": BlockType.TEXT, "content": content}
402
+ if name.startswith("h") and (anchor := self.context.heading_anchor(element)):
403
+ block["anchor"] = anchor
404
+ blocks.append(block)
405
+ text_emitted = True
406
+ return blocks
407
+
408
+ def _parse_note_element(
409
+ self,
410
+ element: etree._Element,
411
+ style: TextStyle,
412
+ visibility_hidden: bool = False,
413
+ ) -> list[dict[str, object]]:
414
+ """逐块转换单条脚注,并只给首个文本脚注挂载 anchor。"""
415
+ blocks = self._parse_container_contents(element, style, visibility_hidden)
416
+ anchor = self.context.note_anchor(element)
417
+ anchor_attached = False
418
+ for block in blocks:
419
+ content = block.get("content")
420
+ if block.get("type") != BlockType.TEXT or not isinstance(content, list) or not content:
421
+ continue
422
+ block["type"] = BlockType.PAGE_FOOTNOTE
423
+ if anchor is not None and not anchor_attached:
424
+ block["anchor"] = anchor
425
+ anchor_attached = True
426
+ return blocks
427
+
428
+ def _render_inline_children(
429
+ self,
430
+ element: etree._Element,
431
+ style: TextStyle,
432
+ visibility_hidden: bool = False,
433
+ ) -> tuple[list[_InlineSpanDict], list[dict[str, object]]]:
434
+ """渲染元素的连续行内 Span,并旁路其中的视觉 blocks。"""
435
+ segments = self._render_inline_children_ordered(element, style, visibility_hidden)
436
+ content: list[_InlineSpanDict] = []
437
+ for segment in segments:
438
+ if isinstance(segment, list):
439
+ extend_inline_spans(content, segment)
440
+ return (
441
+ content,
442
+ [segment for segment in segments if not isinstance(segment, list)],
443
+ )
444
+
445
+ def _render_inline_children_ordered(
446
+ self,
447
+ element: etree._Element,
448
+ style: TextStyle,
449
+ visibility_hidden: bool = False,
450
+ ) -> list[_InlineProjectionSegment]:
451
+ """按 DOM 顺序返回连续文本与旁路 block,保留 inline visual 前后边界。"""
452
+ segments: list[_InlineProjectionSegment] = []
453
+ if not visibility_hidden:
454
+ _append_inline_segment(segments, self._render_text(element.text, style))
455
+ for child in element:
456
+ if not isinstance(child.tag, str):
457
+ if not visibility_hidden:
458
+ _append_inline_segment(segments, self._render_text(entity_text(child), style))
459
+ _append_inline_segment(segments, self._render_text(child.tail, style))
460
+ continue
461
+ for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
462
+ _append_inline_segment(segments, segment)
463
+ if not visibility_hidden:
464
+ _append_inline_segment(segments, self._render_text(child.tail, style))
465
+ return segments
466
+
467
+ def _render_inline_element(
468
+ self,
469
+ element: etree._Element,
470
+ inherited: TextStyle,
471
+ inherited_visibility_hidden: bool = False,
472
+ ) -> tuple[list[_InlineSpanDict], list[dict[str, object]]]:
473
+ """把一个行内元素转换为结构化 Span 和可选视觉块。"""
474
+ segments = self._render_inline_element_ordered(element, inherited, inherited_visibility_hidden)
475
+ content: list[_InlineSpanDict] = []
476
+ for segment in segments:
477
+ if isinstance(segment, list):
478
+ extend_inline_spans(content, segment)
479
+ return (
480
+ content,
481
+ [segment for segment in segments if not isinstance(segment, list)],
482
+ )
483
+
484
+ def _render_inline_element_ordered(
485
+ self,
486
+ element: etree._Element,
487
+ inherited: TextStyle,
488
+ inherited_visibility_hidden: bool = False,
489
+ ) -> list[_InlineProjectionSegment]:
490
+ """递归投影单个行内元素,并在嵌套 visual 位置保留顺序分段。"""
491
+ resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
492
+ if resolved.subtree_hidden:
493
+ return []
494
+ name = local_name(element)
495
+ if name in SKIPPED_TAGS:
496
+ return []
497
+ if name == "br":
498
+ return [] if resolved.visibility_hidden else [text_spans("\n")]
499
+ if name in {"img", "image"}:
500
+ return [] if resolved.visibility_hidden else self._image_blocks(element)
501
+ if name == "math":
502
+ if resolved.visibility_hidden:
503
+ return []
504
+ formula = self._formula_extraction(element)
505
+ if formula is not None:
506
+ if formula.display == "block":
507
+ return [{"type": BlockType.EQUATION, "content": formula.latex}]
508
+ spans: list[_InlineSpanDict] = []
509
+ append_equation_span(spans, formula.latex)
510
+ return [spans]
511
+ fallback = self._visible_plain_text(element, resolved.text, resolved.visibility_hidden)
512
+ return [text_spans(fallback)] if fallback else []
513
+ if name == "code":
514
+ if resolved.visibility_hidden:
515
+ return []
516
+ code = self._visible_raw_text(element, resolved.text, resolved.visibility_hidden)
517
+ spans = []
518
+ append_code_span(spans, code)
519
+ return [spans] if spans else []
520
+ if name in BLOCK_TAGS:
521
+ return self._parse_block(element, inherited, inherited_visibility_hidden)
522
+ segments = self._render_inline_children_ordered(element, resolved.text, resolved.visibility_hidden)
523
+ if name == "a":
524
+ href = element.get("href") or element.get(_XLINK_HREF) or ""
525
+ target = self.context.resolve_link(href)
526
+ if target:
527
+ linked: list[_InlineProjectionSegment] = []
528
+ for segment in segments:
529
+ if not isinstance(segment, list) or not segment:
530
+ linked.append(segment)
531
+ continue
532
+ wrapped: list[_InlineSpanDict] = []
533
+ append_hyperlink_span(wrapped, segment, target)
534
+ linked.append(wrapped)
535
+ return linked
536
+ return segments
537
+
538
+ @staticmethod
539
+ def _render_text(value: str | None, style: TextStyle) -> list[_InlineSpanDict]:
540
+ """折叠文本节点并直接投影为带样式 TextSpan。"""
541
+ text = clean_text_node(value)
542
+ if not text:
543
+ return []
544
+ return text_spans(text, style.names())
545
+
546
+ def _visible_raw_text(
547
+ self,
548
+ element: etree._Element,
549
+ style: TextStyle,
550
+ visibility_hidden: bool = False,
551
+ ) -> str:
552
+ """递归提取可见原始文本,并允许后代显式恢复 visibility。"""
553
+ return visible_raw_text_with_style(element, self.stylesheet, style, visibility_hidden)
554
+
555
+ def _visible_plain_text(
556
+ self,
557
+ element: etree._Element,
558
+ style: TextStyle,
559
+ visibility_hidden: bool = False,
560
+ ) -> str:
561
+ """返回折叠空白并还原实体后的可见纯文本。"""
562
+ value = self._visible_raw_text(element, style, visibility_hidden)
563
+ return _WHITESPACE_RE.sub(" ", html.unescape(value)).strip()
564
+
565
+ def _image_blocks(
566
+ self,
567
+ element: etree._Element,
568
+ *,
569
+ caption: str | None = None,
570
+ emit_alt_caption: bool = True,
571
+ ) -> list[dict[str, object]]:
572
+ """把可解析图片转换为 image block,并用 caption/alt 补说明。"""
573
+ source = element.get("src") or element.get("href") or element.get(_XLINK_HREF) or ""
574
+ requested_alt = (caption or element.get("alt") or element.get("title") or "").strip()
575
+ resolved = self.context.resolve_image(source, alt=requested_alt)
576
+ alt = (resolved.alt if resolved is not None else requested_alt).strip()
577
+ if resolved is None or not (resolved.image_base64 or resolved.image_url):
578
+ return [{"type": BlockType.TEXT, "content": text_spans(alt)}] if alt else []
579
+ block: dict[str, object] = {"type": BlockType.IMAGE, "content": ""}
580
+ if resolved.image_base64:
581
+ block["image_base64"] = resolved.image_base64
582
+ if resolved.image_url:
583
+ block["image_url"] = resolved.image_url
584
+ blocks: list[dict[str, object]] = [block]
585
+ annotation = (caption or (alt if emit_alt_caption else "")).strip()
586
+ if annotation:
587
+ blocks.append({"type": BlockType.IMAGE_CAPTION, "content": text_spans(annotation)})
588
+ return blocks
589
+
590
+ def _parse_figure(
591
+ self,
592
+ element: etree._Element,
593
+ style: TextStyle,
594
+ visibility_hidden: bool = False,
595
+ ) -> list[dict[str, object]]:
596
+ """按标准标签或完整 token 解析 visual 主体、caption 与 footnote。"""
597
+ annotations = [
598
+ (child, kind)
599
+ for child in element
600
+ if isinstance(child.tag, str) and (kind := self._visual_annotation_kind(child)) is not None
601
+ ]
602
+ annotation_elements = {child for child, _ in annotations}
603
+ docvortex_figure = "docvortex-figure" in (element.get("class") or "").casefold().split()
604
+ blocks, visual_blocks_by_child = self._parse_figure_contents(
605
+ element,
606
+ style,
607
+ visibility_hidden,
608
+ annotation_elements=annotation_elements,
609
+ emit_alt_caption=not docvortex_figure and not annotations,
610
+ )
611
+ annotation_targets = self._figure_annotation_targets(
612
+ element,
613
+ annotation_elements,
614
+ visual_blocks_by_child,
615
+ )
616
+
617
+ annotations_by_visual: dict[int, list[dict[str, object]]] = {}
618
+ unbound_annotations: list[dict[str, object]] = []
619
+ for annotation, kind in annotations:
620
+ resolved = self.stylesheet.resolve(annotation, style, visibility_hidden)
621
+ if resolved.subtree_hidden:
622
+ continue
623
+ target = annotation_targets.get(annotation)
624
+ visual_type = _raw_visual_type(target.get("type")) if target is not None else None
625
+ annotation_type = VISUAL_TYPE_MAPPING[visual_type][kind] if visual_type is not None else BlockType.TEXT
626
+ annotation_blocks: list[dict[str, object]] = []
627
+ for segment in self._render_inline_children_ordered(annotation, resolved.text, resolved.visibility_hidden):
628
+ if isinstance(segment, list):
629
+ if content := strip_span_dicts(segment):
630
+ annotation_blocks.append({"type": annotation_type, "content": content})
631
+ continue
632
+ if visual_type is not None and segment.get("type") == BlockType.TEXT:
633
+ segment = {**segment, "type": annotation_type}
634
+ annotation_blocks.append(segment)
635
+ if target is not None and visual_type is not None:
636
+ annotations_by_visual.setdefault(id(target), []).extend(annotation_blocks)
637
+ else:
638
+ unbound_annotations.extend(annotation_blocks)
639
+
640
+ output: list[dict[str, object]] = []
641
+ for block in blocks:
642
+ output.append(block)
643
+ output.extend(annotations_by_visual.get(id(block), ()))
644
+ output.extend(unbound_annotations)
645
+ return output
646
+
647
+ def _parse_figure_contents(
648
+ self,
649
+ element: etree._Element,
650
+ style: TextStyle,
651
+ visibility_hidden: bool,
652
+ *,
653
+ annotation_elements: set[etree._Element],
654
+ emit_alt_caption: bool,
655
+ ) -> tuple[list[dict[str, object]], dict[etree._Element, list[dict[str, object]]]]:
656
+ """按 DOM 顺序缓冲 figure 文本,并在 visual extras 前后切分正文 block。"""
657
+ blocks: list[dict[str, object]] = []
658
+ visual_blocks_by_child: dict[etree._Element, list[dict[str, object]]] = {}
659
+ inline_parts: list[_InlineSpanDict] = []
660
+ if not visibility_hidden:
661
+ extend_inline_spans(inline_parts, self._render_text(element.text, style))
662
+
663
+ def flush_inline() -> None:
664
+ """把 figure 当前连续文本写为普通正文 block。"""
665
+ content = strip_span_dicts(inline_parts)
666
+ inline_parts.clear()
667
+ if content:
668
+ blocks.append({"type": BlockType.TEXT, "content": content})
669
+
670
+ for child in element:
671
+ if not isinstance(child.tag, str):
672
+ if not visibility_hidden:
673
+ extend_inline_spans(inline_parts, self._render_text(entity_text(child), style))
674
+ extend_inline_spans(inline_parts, self._render_text(child.tail, style))
675
+ continue
676
+ if child in annotation_elements:
677
+ if not visibility_hidden:
678
+ extend_inline_spans(inline_parts, self._render_text(child.tail, style))
679
+ continue
680
+
681
+ first_child_block = len(blocks)
682
+ name = local_name(child)
683
+ if name in {"img", "image"}:
684
+ flush_inline()
685
+ child_style = self.stylesheet.resolve(child, style, visibility_hidden)
686
+ if not child_style.subtree_hidden and not child_style.visibility_hidden:
687
+ blocks.extend(self._image_blocks(child, emit_alt_caption=emit_alt_caption))
688
+ elif name in BLOCK_TAGS:
689
+ flush_inline()
690
+ blocks.extend(self._parse_block(child, style, visibility_hidden))
691
+ else:
692
+ for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
693
+ if isinstance(segment, list):
694
+ extend_inline_spans(inline_parts, segment)
695
+ else:
696
+ flush_inline()
697
+ blocks.append(segment)
698
+ if not visibility_hidden:
699
+ extend_inline_spans(inline_parts, self._render_text(child.tail, style))
700
+ child_visuals = [block for block in blocks[first_child_block:] if _raw_visual_type(block.get("type")) is not None]
701
+ if child_visuals:
702
+ visual_blocks_by_child[child] = child_visuals
703
+ flush_inline()
704
+ return blocks, visual_blocks_by_child
705
+
706
+ @staticmethod
707
+ def _figure_annotation_targets(
708
+ figure: etree._Element,
709
+ annotations: set[etree._Element],
710
+ visual_blocks_by_child: dict[etree._Element, list[dict[str, object]]],
711
+ ) -> dict[etree._Element, dict[str, object] | None]:
712
+ """用双向线性扫描绑定全部 annotation,优先最近前序 visual。"""
713
+ children = [child for child in figure if isinstance(child.tag, str)]
714
+ targets: dict[etree._Element, dict[str, object] | None] = {}
715
+ previous_visual: dict[str, object] | None = None
716
+ for child in children:
717
+ if visuals := visual_blocks_by_child.get(child):
718
+ previous_visual = visuals[-1]
719
+ if child in annotations:
720
+ targets[child] = previous_visual
721
+
722
+ next_visual: dict[str, object] | None = None
723
+ for child in reversed(children):
724
+ if visuals := visual_blocks_by_child.get(child):
725
+ next_visual = visuals[0]
726
+ if child in annotations and targets[child] is None:
727
+ targets[child] = next_visual
728
+ return targets
729
+
730
+ def _has_contextual_visual_annotation(self, element: etree._Element) -> bool:
731
+ """仅在直属完整 token annotation 与 visual 后代并存时启用非标准容器解析。"""
732
+ children = [child for child in element if isinstance(child.tag, str)]
733
+ if not any(self._visual_annotation_kind(child) is not None for child in children):
734
+ return False
735
+ return any(
736
+ local_name(candidate) in _VISUAL_ELEMENT_TAGS
737
+ for child in children
738
+ if self._visual_annotation_kind(child) is None
739
+ for candidate in [child, *child.iterdescendants()]
740
+ if isinstance(candidate.tag, str)
741
+ )
742
+
743
+ @staticmethod
744
+ def _visual_annotation_kind(element: etree._Element) -> str | None:
745
+ """用标准标签、role 或完整 class/id token 返回 caption/footnote 角色。"""
746
+ if local_name(element) == "figcaption":
747
+ return "caption"
748
+ tokens = _semantic_tokens(element)
749
+ roles = frozenset((element.get("role") or "").casefold().split())
750
+ if tokens & _CAPTION_TOKENS or roles & {"caption", "doc-subtitle"}:
751
+ return "caption"
752
+ if tokens & _FOOTNOTE_TOKENS or roles & {"doc-footnote", "note"}:
753
+ return "footnote"
754
+ return None
755
+
756
+ def _parse_svg(
757
+ self,
758
+ element: etree._Element,
759
+ style: TextStyle,
760
+ visibility_hidden: bool = False,
761
+ ) -> list[dict[str, object]]:
762
+ """从 SVG 尽力提取 title/desc/text 和静态 image。"""
763
+ blocks: list[dict[str, object]] = []
764
+ texts: list[str] = []
765
+
766
+ def visit(parent: etree._Element, inherited: TextStyle, inherited_visibility_hidden: bool) -> None:
767
+ """按 SVG 树顺序访问候选节点,并允许可见后代恢复输出。"""
768
+ for child in parent:
769
+ if not isinstance(child.tag, str):
770
+ continue
771
+ resolved = self.stylesheet.resolve(child, inherited, inherited_visibility_hidden)
772
+ if resolved.subtree_hidden:
773
+ continue
774
+ name = local_name(child)
775
+ if name in {"title", "desc", "text"}:
776
+ value = self._visible_plain_text(child, resolved.text, resolved.visibility_hidden)
777
+ if value and value not in texts:
778
+ texts.append(value)
779
+ elif name == "image":
780
+ if not resolved.visibility_hidden:
781
+ blocks.extend(self._image_blocks(child))
782
+ else:
783
+ visit(child, resolved.text, resolved.visibility_hidden)
784
+
785
+ visit(element, style, visibility_hidden)
786
+ if texts:
787
+ blocks.insert(0, {"type": BlockType.TEXT, "content": text_spans("\n".join(texts))})
788
+ return blocks
789
+
790
+ def _parse_table(
791
+ self,
792
+ table: etree._Element,
793
+ style: TextStyle,
794
+ visibility_hidden: bool = False,
795
+ ) -> list[dict[str, object]]:
796
+ """重建白名单化 HTML 表格,并把 caption 投影为表格说明。"""
797
+ markup = self._serialize_table_node(table, style, visibility_hidden)
798
+ if not markup:
799
+ return []
800
+ blocks: list[dict[str, object]] = [{"type": BlockType.TABLE, "content": markup}]
801
+ caption_element = next(
802
+ (child for child in table if isinstance(child.tag, str) and local_name(child) == "caption"),
803
+ None,
804
+ )
805
+ if caption_element is not None:
806
+ caption_style = self.stylesheet.resolve(caption_element, style, visibility_hidden)
807
+ if not caption_style.subtree_hidden:
808
+ caption = self._visible_plain_text(caption_element, caption_style.text, caption_style.visibility_hidden)
809
+ if caption:
810
+ blocks.append({"type": BlockType.TABLE_CAPTION, "content": text_spans(caption)})
811
+ return blocks
812
+
813
+ def _serialize_table_node(
814
+ self,
815
+ element: etree._Element,
816
+ inherited: TextStyle,
817
+ inherited_visibility_hidden: bool = False,
818
+ *,
819
+ row_link_target: str | None = None,
820
+ ) -> str:
821
+ """递归序列化安全表格结构、行内样式、链接、公式和图片。"""
822
+ resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
823
+ if resolved.subtree_hidden:
824
+ return ""
825
+ name = local_name(element)
826
+ if name == "caption" or name in SKIPPED_TAGS:
827
+ return ""
828
+ allowed = {
829
+ "a",
830
+ "b",
831
+ "br",
832
+ "code",
833
+ "col",
834
+ "colgroup",
835
+ "em",
836
+ "i",
837
+ "img",
838
+ "math",
839
+ "p",
840
+ "s",
841
+ "span",
842
+ "strong",
843
+ "sub",
844
+ "sup",
845
+ "table",
846
+ "tbody",
847
+ "td",
848
+ "tfoot",
849
+ "th",
850
+ "thead",
851
+ "tr",
852
+ "u",
853
+ }
854
+ if name not in allowed:
855
+ return self._serialize_table_children(
856
+ element,
857
+ resolved.text,
858
+ resolved.visibility_hidden,
859
+ row_link_target=row_link_target,
860
+ )
861
+ if name == "br":
862
+ return "" if resolved.visibility_hidden else "<br>"
863
+ if name == "math":
864
+ if resolved.visibility_hidden:
865
+ return ""
866
+ formula = self._formula_extraction(element)
867
+ if formula is not None:
868
+ return f"<eq>{html.escape(formula.latex, quote=False)}</eq>"
869
+ fallback = self._visible_plain_text(element, resolved.text, resolved.visibility_hidden)
870
+ return html.escape(fallback, quote=False)
871
+ if name == "img":
872
+ if resolved.visibility_hidden:
873
+ return ""
874
+ source = element.get("src") or ""
875
+ alt_text = (element.get("alt") or "").strip()
876
+ image = self.context.resolve_image(source, alt=alt_text)
877
+ if image is None:
878
+ return html.escape(alt_text, quote=False)
879
+ image_source = image.image_base64 or image.image_url
880
+ alt = html.escape(image.alt or alt_text, quote=True)
881
+ return f'<img src="{html.escape(image_source, quote=True)}" alt="{alt}">' if image_source else alt
882
+ if name == "tr":
883
+ row_link_target = self._toc_table_row_target(element)
884
+ attributes: list[str] = []
885
+ if name in {"td", "th"}:
886
+ for attribute in ("colspan", "rowspan", "scope"):
887
+ value = (element.get(attribute) or "").strip()
888
+ if attribute == "scope" and value in {"col", "colgroup", "row", "rowgroup"}:
889
+ attributes.append(f'{attribute}="{value}"')
890
+ elif span := bounded_table_span(value):
891
+ attributes.append(f'{attribute}="{span}"')
892
+ elif name in {"col", "colgroup"}:
893
+ if span := bounded_table_span((element.get("span") or "").strip()):
894
+ attributes.append(f'span="{span}"')
895
+ if name == "a":
896
+ target = self.context.resolve_link(element.get("href") or "")
897
+ if target:
898
+ attributes.append(f'href="{html.escape(target, quote=True)}"')
899
+ inner = self._serialize_table_children(
900
+ element,
901
+ resolved.text,
902
+ resolved.visibility_hidden,
903
+ row_link_target=row_link_target,
904
+ )
905
+ if resolved.visibility_hidden and not inner:
906
+ return ""
907
+ if name in {"td", "th"} and row_link_target and self._table_cell_can_inherit_toc_link(element):
908
+ inner = f'<a href="{html.escape(row_link_target, quote=True)}">{inner}</a>'
909
+ attrs = f" {' '.join(attributes)}" if attributes else ""
910
+ return f"<{name}{attrs}>{inner}</{name}>"
911
+
912
+ def _serialize_table_children(
913
+ self,
914
+ element: etree._Element,
915
+ style: TextStyle,
916
+ visibility_hidden: bool = False,
917
+ *,
918
+ row_link_target: str | None = None,
919
+ ) -> str:
920
+ """序列化表格节点的文本、子元素和 tail。"""
921
+ parts = [] if visibility_hidden else [self._render_table_text(element.text, style)]
922
+ for child in element:
923
+ if isinstance(child.tag, str):
924
+ parts.append(self._serialize_table_node(child, style, visibility_hidden, row_link_target=row_link_target))
925
+ elif not visibility_hidden:
926
+ parts.append(self._render_table_text(entity_text(child), style))
927
+ if not visibility_hidden:
928
+ parts.append(self._render_table_text(child.tail, style))
929
+ return "".join(parts)
930
+
931
+ def _toc_table_row_target(self, row: etree._Element) -> str | None:
932
+ """为严格匹配单一目标标题的目录表格行返回内部链接。"""
933
+ links = [
934
+ element
935
+ for element in row.iter()
936
+ if isinstance(element.tag, str) and local_name(element) == "a" and (element.get("href") or "").strip()
937
+ ]
938
+ if not links:
939
+ return None
940
+ resolved_targets: list[str] = []
941
+ for link in links:
942
+ target = self.context.resolve_link(link.get("href") or "")
943
+ if target is None or not target.startswith("#"):
944
+ return None
945
+ resolved_targets.append(target)
946
+ if len(set(resolved_targets)) != 1:
947
+ return None
948
+ target = resolved_targets[0]
949
+ title = self.context.heading_label(target[1:])
950
+ if title is None:
951
+ return None
952
+ cells = [child for child in row if isinstance(child.tag, str) and local_name(child) in {"td", "th"}]
953
+ row_label = " ".join(value for cell in cells if (value := visible_text(cell)))
954
+ normalized_row = _WHITESPACE_RE.sub(" ", html.unescape(row_label)).strip().casefold()
955
+ normalized_title = _WHITESPACE_RE.sub(" ", html.unescape(title)).strip().casefold()
956
+ return target if normalized_row and normalized_row == normalized_title else None
957
+
958
+ @staticmethod
959
+ def _table_cell_can_inherit_toc_link(cell: etree._Element) -> bool:
960
+ """只允许纯文本与行内样式单元格继承目录行的唯一内部链接。"""
961
+ if not visible_text(cell):
962
+ return False
963
+ allowed_inline = {"b", "br", "code", "em", "i", "s", "span", "strong", "sub", "sup", "u"}
964
+ return all(isinstance(child.tag, str) and local_name(child) in allowed_inline for child in cell.iterdescendants())
965
+
966
+ @staticmethod
967
+ def _render_table_text(value: str | None, style: TextStyle) -> str:
968
+ """把表格文字转义后包装为 renderer 支持的安全 HTML 样式标签。"""
969
+ rendered = html.escape(clean_text_node(value), quote=False)
970
+ if not rendered:
971
+ return ""
972
+ for enabled, tag in (
973
+ (style.bold, "strong"),
974
+ (style.italic, "em"),
975
+ (style.underline, "u"),
976
+ (style.strikethrough, "s"),
977
+ (style.superscript, "sup"),
978
+ (style.subscript, "sub"),
979
+ ):
980
+ if enabled:
981
+ rendered = f"<{tag}>{rendered}</{tag}>"
982
+ return rendered
983
+
984
+ def _parse_list(
985
+ self,
986
+ element: etree._Element,
987
+ style: TextStyle,
988
+ visibility_hidden: bool = False,
989
+ ) -> tuple[dict[str, object] | None, list[dict[str, object]]]:
990
+ """解析有序/无序列表,并投影为连续阿拉伯编号结构。"""
991
+ if self._list_contains_page_blocks(element):
992
+ return self._parse_list_with_page_blocks(element, style, visibility_hidden)
993
+ ordered = local_name(element) == "ol"
994
+ items = [child for child in element if isinstance(child.tag, str) and local_name(child) == "li"]
995
+ if not items:
996
+ return None, []
997
+ children: list[dict[str, object]] = []
998
+ extras: list[dict[str, object]] = []
999
+ for item in items:
1000
+ item_style = self.stylesheet.resolve(item, style, visibility_hidden)
1001
+ if item_style.subtree_hidden:
1002
+ continue
1003
+ if self.context.note_anchor(item) is not None:
1004
+ extras.extend(self._parse_note_element(item, item_style.text, item_style.visibility_hidden))
1005
+ continue
1006
+ content_parts: list[_InlineSpanDict] = []
1007
+ if not item_style.visibility_hidden:
1008
+ extend_inline_spans(content_parts, self._render_text(item.text, item_style.text))
1009
+ nested_lists: list[dict[str, object]] = []
1010
+ for child in item:
1011
+ if not isinstance(child.tag, str):
1012
+ if not item_style.visibility_hidden:
1013
+ extend_inline_spans(content_parts, self._render_text(entity_text(child), item_style.text))
1014
+ extend_inline_spans(content_parts, self._render_text(child.tail, item_style.text))
1015
+ continue
1016
+ name = local_name(child)
1017
+ if name in {"ul", "ol"}:
1018
+ nested_style = self.stylesheet.resolve(child, item_style.text, item_style.visibility_hidden)
1019
+ if not nested_style.subtree_hidden:
1020
+ nested, nested_extras = self._parse_list(child, nested_style.text, nested_style.visibility_hidden)
1021
+ if nested is not None:
1022
+ nested_lists.append(nested)
1023
+ extras.extend(nested_extras)
1024
+ elif name in {"table", "figure", "svg"}:
1025
+ extras.extend(self._parse_block(child, item_style.text, item_style.visibility_hidden))
1026
+ elif name in BLOCK_TAGS:
1027
+ child_style = self.stylesheet.resolve(child, item_style.text, item_style.visibility_hidden)
1028
+ if not child_style.subtree_hidden:
1029
+ if self.context.note_anchor(child) is not None:
1030
+ extras.extend(self._parse_note_element(child, child_style.text, child_style.visibility_hidden))
1031
+ else:
1032
+ rendered, child_extras = self._render_inline_children(
1033
+ child,
1034
+ child_style.text,
1035
+ child_style.visibility_hidden,
1036
+ )
1037
+ _append_list_block_content(content_parts, rendered)
1038
+ extras.extend(child_extras)
1039
+ else:
1040
+ rendered, child_extras = self._render_inline_element(child, item_style.text, item_style.visibility_hidden)
1041
+ extend_inline_spans(content_parts, rendered)
1042
+ extras.extend(child_extras)
1043
+ if not item_style.visibility_hidden:
1044
+ extend_inline_spans(content_parts, self._render_text(child.tail, item_style.text))
1045
+ content = strip_span_dicts(content_parts)
1046
+ if content:
1047
+ children.append({"type": BlockType.TEXT, "content": content})
1048
+ children.extend(nested_lists)
1049
+ if not children:
1050
+ return None, extras
1051
+ block: dict[str, object] = {
1052
+ "type": BlockType.LIST,
1053
+ "attribute": "ordered" if ordered else "unordered",
1054
+ "content": children,
1055
+ }
1056
+ if ordered:
1057
+ block["start"] = self._ordered_list_start(element)
1058
+ return block, extras
1059
+
1060
+ @staticmethod
1061
+ def _list_contains_page_blocks(element: etree._Element) -> bool:
1062
+ """判断列表是否含可能提升为页面兄弟的 visual、code 或公式子树。"""
1063
+ return any(
1064
+ isinstance(candidate.tag, str) and local_name(candidate) in _LIST_PAGE_BLOCK_TAGS
1065
+ for candidate in element.iterdescendants()
1066
+ )
1067
+
1068
+ def _parse_list_with_page_blocks(
1069
+ self,
1070
+ element: etree._Element,
1071
+ style: TextStyle,
1072
+ visibility_hidden: bool,
1073
+ ) -> tuple[dict[str, object] | None, list[dict[str, object]]]:
1074
+ """把含 visual 的列表切成有序 list/text/page block 片段,保持 DOM 阅读顺序。"""
1075
+ ordered = local_name(element) == "ol"
1076
+ list_start = self._ordered_list_start(element) if ordered else 1
1077
+ items = [child for child in element if isinstance(child.tag, str) and local_name(child) == "li"]
1078
+ pending_children: list[dict[str, object]] = []
1079
+ pending_start = list_start
1080
+ output: list[dict[str, object]] = []
1081
+ visible_item_ordinal = 0
1082
+ has_page_blocks = False
1083
+
1084
+ def flush_pending() -> None:
1085
+ """把当前连续列表项写为一个顶层 list block。"""
1086
+ nonlocal pending_children
1087
+ if not pending_children:
1088
+ return
1089
+ output.append(self._build_raw_list_block(pending_children, ordered=ordered, start=pending_start))
1090
+ pending_children = []
1091
+
1092
+ for item in items:
1093
+ item_style = self.stylesheet.resolve(item, style, visibility_hidden)
1094
+ if item_style.subtree_hidden:
1095
+ continue
1096
+ if self.context.note_anchor(item) is not None:
1097
+ flush_pending()
1098
+ output.extend(self._parse_note_element(item, item_style.text, item_style.visibility_hidden))
1099
+ has_page_blocks = True
1100
+ continue
1101
+
1102
+ segments = self._normalize_list_item_segments(
1103
+ self._render_inline_children_ordered(item, item_style.text, item_style.visibility_hidden)
1104
+ )
1105
+ page_positions = [
1106
+ index
1107
+ for index, segment in enumerate(segments)
1108
+ if not isinstance(segment, list) and segment.get("type") != BlockType.LIST
1109
+ ]
1110
+ if not page_positions:
1111
+ item_children = self._list_item_children(segments)
1112
+ if item_children:
1113
+ if not pending_children:
1114
+ pending_start = list_start + visible_item_ordinal
1115
+ pending_children.extend(item_children)
1116
+ visible_item_ordinal += 1
1117
+ continue
1118
+
1119
+ first_page_position = page_positions[0]
1120
+ prefix_children = self._list_item_children(segments[:first_page_position])
1121
+ if not pending_children:
1122
+ pending_start = list_start + visible_item_ordinal
1123
+ pending_children.extend(prefix_children or [{"type": BlockType.TEXT, "content": []}])
1124
+ flush_pending()
1125
+
1126
+ for segment in segments[first_page_position:]:
1127
+ if isinstance(segment, list):
1128
+ content = strip_span_dicts(segment)
1129
+ if content:
1130
+ output.append({"type": BlockType.TEXT, "content": content})
1131
+ else:
1132
+ output.append(segment)
1133
+ visible_item_ordinal += 1
1134
+ has_page_blocks = True
1135
+
1136
+ if not has_page_blocks:
1137
+ return (
1138
+ self._build_raw_list_block(pending_children, ordered=ordered, start=list_start) if pending_children else None,
1139
+ [],
1140
+ )
1141
+ flush_pending()
1142
+ return None, output
1143
+
1144
+ @staticmethod
1145
+ def _normalize_list_item_segments(
1146
+ segments: list[_InlineProjectionSegment],
1147
+ ) -> list[_InlineProjectionSegment]:
1148
+ """把列表内部普通 text block 还原为文本片段,保留 visual/list 页面边界。"""
1149
+ normalized: list[_InlineProjectionSegment] = []
1150
+ for segment in segments:
1151
+ if isinstance(segment, dict) and segment.get("type") == BlockType.TEXT:
1152
+ content = segment.get("content")
1153
+ _append_inline_segment(normalized, content if isinstance(content, list) else [])
1154
+ else:
1155
+ _append_inline_segment(normalized, segment)
1156
+ return normalized
1157
+
1158
+ @staticmethod
1159
+ def _list_item_children(segments: list[_InlineProjectionSegment]) -> list[dict[str, object]]:
1160
+ """把无页面 visual 的列表片段收敛为一个文本叶子及其嵌套列表。"""
1161
+ content: list[_InlineSpanDict] = []
1162
+ for segment in segments:
1163
+ if isinstance(segment, list):
1164
+ extend_inline_spans(content, segment)
1165
+ content = strip_span_dicts(content)
1166
+ children = [{"type": BlockType.TEXT, "content": content}] if content else []
1167
+ children.extend(segment for segment in segments if isinstance(segment, dict) and segment.get("type") == BlockType.LIST)
1168
+ return children
1169
+
1170
+ @staticmethod
1171
+ def _build_raw_list_block(
1172
+ children: list[dict[str, object]],
1173
+ *,
1174
+ ordered: bool,
1175
+ start: int,
1176
+ ) -> dict[str, object]:
1177
+ """构造一段可由既有无坐标后处理编号的 raw list block。"""
1178
+ block: dict[str, object] = {
1179
+ "type": BlockType.LIST,
1180
+ "attribute": "ordered" if ordered else "unordered",
1181
+ "content": children,
1182
+ }
1183
+ if ordered:
1184
+ block["start"] = start
1185
+ return block
1186
+
1187
+ @staticmethod
1188
+ def _ordered_list_start(element: etree._Element) -> int:
1189
+ """读取有序列表唯一通用起始值,非法或负值统一回退为一。"""
1190
+ try:
1191
+ start = int(element.get("start") or 1)
1192
+ except ValueError:
1193
+ return 1
1194
+ return start if start >= 0 else 1
1195
+
1196
+ @staticmethod
1197
+ def _formula_extraction(element: etree._Element) -> FormulaExtraction | None:
1198
+ """调用共享公式优先级,返回裸 LaTeX 及来源信息。"""
1199
+ return extract_formula(element)
1200
+
1201
+ @staticmethod
1202
+ def _code_language_hint(element: etree._Element) -> str | None:
1203
+ """从 pre/code 的标准 class 或 data 属性提取安全语言提示。"""
1204
+ candidates = [element, *[child for child in element if isinstance(child.tag, str) and local_name(child) == "code"]]
1205
+ for candidate in candidates:
1206
+ for attribute in ("data-language", "data-lang"):
1207
+ value = (candidate.get(attribute) or "").strip()
1208
+ if re.fullmatch(r"[A-Za-z0-9_.+#-]+", value):
1209
+ return value
1210
+ for token in (candidate.get("class") or "").split():
1211
+ normalized = token.casefold()
1212
+ for prefix in ("language-", "lang-"):
1213
+ if normalized.startswith(prefix):
1214
+ value = token[len(prefix) :]
1215
+ if re.fullmatch(r"[A-Za-z0-9_.+#-]+", value):
1216
+ return value
1217
+ return None
1218
+
1219
+
1220
+ __all__ = [
1221
+ "BLOCK_TAGS",
1222
+ "MarkupContext",
1223
+ "MarkupProjector",
1224
+ "ResolvedMarkupImage",
1225
+ "SKIPPED_TAGS",
1226
+ "bounded_table_span",
1227
+ "clean_text_node",
1228
+ "entity_text",
1229
+ "local_name",
1230
+ "visible_raw_text_with_style",
1231
+ "visible_text",
1232
+ ]
1233
+
1234
+ # 保持既有公开类型的 pickle 路径,所有旧、新入口指向同一个类。
1235
+ preserve_type_module(ResolvedMarkupImage, "docvortex.analyzers.native._shared.markup.projector")
1236
+ preserve_type_module(MarkupContext, "docvortex.analyzers.native._shared.markup.projector")
1237
+ preserve_type_module(MarkupProjector, "docvortex.analyzers.native._shared.markup.projector")