docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,34 @@
1
+ """HTML wire 物化依赖的资源协议,不依赖具体输入解析器。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Protocol
6
+ from lxml import etree
7
+
8
+ from ...content.markup import MarkupContext
9
+
10
+
11
+ class WireAnchorResolver(Protocol):
12
+ """描述精确 HTML 文档提供的锚点解析能力。"""
13
+
14
+ def resolve_fragment(self, fragment: str) -> str | None:
15
+ """将源 fragment 解析为规范内部链接。"""
16
+
17
+ def heading_anchor(self, heading: etree._Element) -> str | None:
18
+ """返回标题的规范锚点。"""
19
+
20
+ def heading_label(self, anchor: str) -> str | None:
21
+ """返回标题锚点对应的文字。"""
22
+
23
+ def note_anchor(self, note: etree._Element) -> str | None:
24
+ """返回页面脚注的规范锚点。"""
25
+
26
+
27
+ class WireResourceContext(MarkupContext, Protocol):
28
+ """在共享 markup 能力上增加绑定精确 wire 锚点的操作。"""
29
+
30
+ def bind_anchors(self, anchors: WireAnchorResolver) -> None:
31
+ """绑定经过整棵 wire 验证的锚点解析器。"""
32
+
33
+
34
+ __all__ = ["WireAnchorResolver", "WireResourceContext"]
@@ -0,0 +1,17 @@
1
+ """共享文档协议的显式读取入口。"""
2
+
3
+ from typing import Any
4
+ from ..schema import MiddleJson, ModelJson
5
+
6
+
7
+ def load_model(value: dict[str, Any]) -> ModelJson:
8
+ """读取新版分析文档,不执行解析、补写来源或历史迁移。"""
9
+ return ModelJson.from_dict(value)
10
+
11
+
12
+ def load_middle(value: dict[str, Any]) -> MiddleJson:
13
+ """读取新版语义文档,不访问源文件和外部素材。"""
14
+ return MiddleJson.from_dict(value)
15
+
16
+
17
+ __all__ = ["load_model", "load_middle"]
@@ -0,0 +1,5 @@
1
+ """共享内容操作及显式的 PDF 输出文字清洗。"""
2
+
3
+ from .normalization import normalize_pdf_model_text
4
+
5
+ __all__ = ["normalize_pdf_model_text"]
@@ -0,0 +1,248 @@
1
+ """Middle JSON 2.0 行内 Span 的规范化、可见文本与段落边界操作。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from copy import deepcopy
7
+ from typing import Callable, Iterable
8
+
9
+ from ..schema import CodeInlineSpan, EquationInlineSpan, HyperlinkSpan, InlineSpan, TextSpan, parse_inline_spans
10
+ from ..foundation.language import detect_lang
11
+ from ..foundation.text import CJK_LANGS, resolve_text_line_boundary
12
+
13
+ _CJK_RE = re.compile(r"[\u3040-\u30ff\u3400-\u9fff\uac00-\ud7af]")
14
+
15
+
16
+ def normalize_inline_spans(spans: Iterable[InlineSpan | dict[str, object]]) -> list[InlineSpan]:
17
+ """严格解析、递归规范化并合并相邻等样式文字 Span。"""
18
+ parsed = parse_inline_spans(list(spans))
19
+ normalized: list[InlineSpan] = []
20
+ for span in parsed:
21
+ current: InlineSpan
22
+ if isinstance(span, HyperlinkSpan):
23
+ children = normalize_inline_spans(span.content)
24
+ non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
25
+ if not non_link_children:
26
+ continue
27
+ current = span.model_copy(update={"content": non_link_children})
28
+ else:
29
+ current = span.model_copy(deep=True)
30
+ if (
31
+ normalized
32
+ and isinstance(normalized[-1], TextSpan)
33
+ and isinstance(current, TextSpan)
34
+ and normalized[-1].styles == current.styles
35
+ ):
36
+ normalized[-1].content += current.content
37
+ continue
38
+ if (
39
+ normalized
40
+ and isinstance(normalized[-1], HyperlinkSpan)
41
+ and isinstance(current, HyperlinkSpan)
42
+ and normalized[-1].url == current.url
43
+ ):
44
+ merged_children = normalize_inline_spans([*normalized[-1].content, *current.content])
45
+ normalized[-1].content = [child for child in merged_children if not isinstance(child, HyperlinkSpan)]
46
+ continue
47
+ normalized.append(current)
48
+ return normalized
49
+
50
+
51
+ def inline_plain_text(spans: Iterable[InlineSpan]) -> str:
52
+ """提取 Span 列表的完整可见文字,供排序、合并和标题判断使用。"""
53
+ parts: list[str] = []
54
+ for span in spans:
55
+ if isinstance(span, (TextSpan, CodeInlineSpan, EquationInlineSpan)):
56
+ parts.append(span.content)
57
+ elif isinstance(span, HyperlinkSpan):
58
+ parts.append(inline_plain_text(span.content))
59
+ return "".join(parts)
60
+
61
+
62
+ def join_inline_spans(contents: Iterable[Iterable[InlineSpan]]) -> list[InlineSpan]:
63
+ """按物理段落边界规则合并多组 Span,并保持结构化语义。"""
64
+ merged: list[InlineSpan] = []
65
+ for content in contents:
66
+ current = normalize_inline_spans(list(content))
67
+ if not current:
68
+ continue
69
+ # 边界裁剪最多清空最后一个根节点;保留其前驱以重新合并新相邻的 Span。
70
+ boundary_start = max(0, len(merged) - 2)
71
+ if merged:
72
+ _join_inline_span_sequences(merged, current)
73
+ merged.extend(current)
74
+ merged[boundary_start:] = normalize_inline_spans(_drop_empty_text_spans(merged[boundary_start:]))
75
+ return merged
76
+
77
+
78
+ def strip_inline_spans(spans: Iterable[InlineSpan]) -> list[InlineSpan]:
79
+ """删除行内内容首尾空白,同时保留内部 Span 边界和样式。"""
80
+ normalized = normalize_inline_spans(deepcopy(list(spans)))
81
+ first = _first_text_span(normalized)
82
+ last = _last_text_span(normalized)
83
+ if first is not None:
84
+ object.__setattr__(first, "content", first.content.lstrip())
85
+ if last is not None:
86
+ object.__setattr__(last, "content", last.content.rstrip())
87
+ return _drop_empty_text_spans(normalized)
88
+
89
+
90
+ def replace_inline_text(spans: Iterable[InlineSpan], content: str) -> list[InlineSpan]:
91
+ """把纯文本回填为单一 TextSpan,供确实丢弃原样式的规则使用。"""
92
+ if not content:
93
+ return []
94
+ return [TextSpan(type="text", content=content)]
95
+
96
+
97
+ def slice_inline_spans(spans: Iterable[InlineSpan], start: int = 0, end: int | None = None) -> list[InlineSpan]:
98
+ """按可见字符偏移裁剪 Span,并保留覆盖范围内的样式和链接。"""
99
+ normalized = normalize_inline_spans(deepcopy(list(spans)))
100
+ visible_length = len(inline_plain_text(normalized))
101
+ resolved_start = min(max(start, 0), visible_length)
102
+ resolved_end = visible_length if end is None else min(max(end, resolved_start), visible_length)
103
+ output: list[InlineSpan] = []
104
+ cursor = 0
105
+ for span in normalized:
106
+ span_length = len(inline_plain_text([span]))
107
+ span_end = cursor + span_length
108
+ overlap_start = max(resolved_start, cursor)
109
+ overlap_end = min(resolved_end, span_end)
110
+ if overlap_start < overlap_end:
111
+ local_start = overlap_start - cursor
112
+ local_end = overlap_end - cursor
113
+ sliced = _slice_inline_span(span, local_start, local_end)
114
+ if sliced is not None:
115
+ output.append(sliced)
116
+ cursor = span_end
117
+ if cursor >= resolved_end:
118
+ break
119
+ return normalize_inline_spans(output)
120
+
121
+
122
+ def map_text_span_content(spans: Iterable[InlineSpan], transform: Callable[[str], str]) -> list[InlineSpan]:
123
+ """递归转换 TextSpan 正文,同时保留其它 Span 语义。"""
124
+ output: list[InlineSpan] = []
125
+ for span in normalize_inline_spans(deepcopy(list(spans))):
126
+ if isinstance(span, TextSpan):
127
+ content = transform(span.content)
128
+ if content:
129
+ output.append(span.model_copy(update={"content": content}))
130
+ elif isinstance(span, HyperlinkSpan):
131
+ children = map_text_span_content(span.content, transform)
132
+ non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
133
+ if non_link_children:
134
+ output.append(span.model_copy(update={"content": non_link_children}))
135
+ else:
136
+ output.append(span)
137
+ return normalize_inline_spans(output)
138
+
139
+
140
+ def _join_inline_span_sequences(previous: list[InlineSpan], current: list[InlineSpan]) -> None:
141
+ """在两组 Span 之间应用语言相关的换行拼接规则。"""
142
+ previous_visible = inline_plain_text(previous).rstrip()
143
+ current_visible = inline_plain_text(current).lstrip()
144
+ if not previous_visible or not current_visible:
145
+ return
146
+
147
+ last_text = _last_text_span(previous)
148
+ first_text = _first_text_span(current)
149
+ if last_text is not None:
150
+ object.__setattr__(last_text, "content", last_text.content.rstrip())
151
+ if first_text is not None:
152
+ object.__setattr__(first_text, "content", first_text.content.lstrip())
153
+
154
+ language = _detect_boundary_language(f"{previous_visible}{current_visible}")
155
+ if last_text is not None:
156
+ processed, separator = resolve_text_line_boundary(
157
+ last_text.content,
158
+ block_language=language,
159
+ next_content=current_visible,
160
+ )
161
+ object.__setattr__(last_text, "content", processed)
162
+ else:
163
+ separator = "" if language in CJK_LANGS else " "
164
+ if separator:
165
+ previous.append(TextSpan(type="text", content=separator))
166
+
167
+
168
+ def _detect_boundary_language(content: str) -> str:
169
+ """检测段落边界语言,短 CJK 文本优先使用字符范围兜底。"""
170
+ if _CJK_RE.search(content):
171
+ return "zh"
172
+ try:
173
+ return detect_lang(content)
174
+ except Exception:
175
+ return ""
176
+
177
+
178
+ def _first_text_span(spans: list[InlineSpan]) -> TextSpan | None:
179
+ """返回首个可见叶子为文字时对应的 TextSpan。"""
180
+ for span in spans:
181
+ if not inline_plain_text([span]):
182
+ continue
183
+ if isinstance(span, TextSpan):
184
+ return span
185
+ if isinstance(span, HyperlinkSpan):
186
+ return _first_text_span(list(span.content))
187
+ return None
188
+ return None
189
+
190
+
191
+ def _last_text_span(spans: list[InlineSpan]) -> TextSpan | None:
192
+ """返回末个可见叶子为文字时对应的 TextSpan。"""
193
+ for span in reversed(spans):
194
+ if not inline_plain_text([span]):
195
+ continue
196
+ if isinstance(span, TextSpan):
197
+ return span
198
+ if isinstance(span, HyperlinkSpan):
199
+ return _last_text_span(list(span.content))
200
+ return None
201
+ return None
202
+
203
+
204
+ def _drop_empty_text_spans(spans: list[InlineSpan]) -> list[InlineSpan]:
205
+ """删除裁剪后为空的 TextSpan,并递归清理空链接。"""
206
+ result: list[InlineSpan] = []
207
+ for span in spans:
208
+ if isinstance(span, TextSpan) and not span.content:
209
+ continue
210
+ if isinstance(span, HyperlinkSpan):
211
+ children = _drop_empty_text_spans(list(span.content))
212
+ non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
213
+ if not non_link_children:
214
+ continue
215
+ span.content = non_link_children
216
+ result.append(span)
217
+ return result
218
+
219
+
220
+ def _slice_inline_span(span: InlineSpan, start: int, end: int) -> InlineSpan | None:
221
+ """裁剪单个 Span 的局部可见区间。"""
222
+ if start >= end:
223
+ return None
224
+ if isinstance(span, TextSpan):
225
+ content = span.content[start:end]
226
+ return span.model_copy(update={"content": content}) if content else None
227
+ if isinstance(span, EquationInlineSpan):
228
+ content = span.content[start:end]
229
+ return span.model_copy(update={"content": content}) if content.strip() else None
230
+ if isinstance(span, CodeInlineSpan):
231
+ content = span.content[start:end]
232
+ return span.model_copy(update={"content": content}) if content else None
233
+ if isinstance(span, HyperlinkSpan):
234
+ children = slice_inline_spans(span.content, start, end)
235
+ non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
236
+ return span.model_copy(update={"content": non_link_children}) if non_link_children else None
237
+ return None
238
+
239
+
240
+ __all__ = [
241
+ "inline_plain_text",
242
+ "join_inline_spans",
243
+ "map_text_span_content",
244
+ "normalize_inline_spans",
245
+ "replace_inline_text",
246
+ "slice_inline_spans",
247
+ "strip_inline_spans",
248
+ ]
@@ -0,0 +1,44 @@
1
+ """Flash EPUB 与 HTML 共用的静态标记文档投影能力。"""
2
+
3
+ from docvortex.content.markup.anchors import (
4
+ AnchorTextNormalization,
5
+ AnchorVisibilityScope,
6
+ MarkupAnchorDocument,
7
+ MarkupAnchorPolicy,
8
+ MarkupAnchorRegistry,
9
+ canonical_anchor,
10
+ element_id,
11
+ visible_element_text,
12
+ )
13
+ from docvortex.content.markup.formula import (
14
+ FormulaDisplay,
15
+ FormulaExtraction,
16
+ FormulaSourceKind,
17
+ extract_formula,
18
+ strip_formula_delimiters,
19
+ )
20
+ from docvortex.content.markup.projector import MarkupContext, MarkupProjector, ResolvedMarkupImage
21
+ from docvortex.content.markup.styles import ElementStyle, MarkupStylesheet, TextStyle, TextStyleDelta
22
+
23
+ __all__ = [
24
+ "AnchorTextNormalization",
25
+ "AnchorVisibilityScope",
26
+ "ElementStyle",
27
+ "FormulaDisplay",
28
+ "FormulaExtraction",
29
+ "FormulaSourceKind",
30
+ "MarkupAnchorDocument",
31
+ "MarkupAnchorPolicy",
32
+ "MarkupAnchorRegistry",
33
+ "MarkupContext",
34
+ "MarkupProjector",
35
+ "MarkupStylesheet",
36
+ "ResolvedMarkupImage",
37
+ "TextStyle",
38
+ "TextStyleDelta",
39
+ "canonical_anchor",
40
+ "element_id",
41
+ "extract_formula",
42
+ "strip_formula_delimiters",
43
+ "visible_element_text",
44
+ ]
@@ -0,0 +1,188 @@
1
+ """集中建立静态 HTML/XHTML 标题、脚注与 fragment anchor 索引。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import hashlib
7
+ import html
8
+ import re
9
+ from typing import Literal, Protocol, TypeAlias
10
+
11
+ from lxml import etree # type: ignore[reportMissingImports]
12
+
13
+ from docvortex.content.markup.projector import local_name, visible_raw_text_with_style
14
+ from docvortex.content.markup.styles import MarkupStylesheet, TextStyle
15
+ from docvortex.foundation.type_identity import preserve_type_module
16
+
17
+
18
+ AnchorVisibilityScope: TypeAlias = Literal["all_ancestors", "nearest_body"]
19
+ AnchorTextNormalization: TypeAlias = Literal["unicode_whitespace", "xhtml_whitespace"]
20
+
21
+ _HEADING_TAGS = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"})
22
+ _XHTML_WHITESPACE_RE = re.compile(r"[\t\r\n\f ]+")
23
+ _XML_ID = "{http://www.w3.org/XML/1998/namespace}id"
24
+
25
+
26
+ @dataclass(frozen=True, slots=True)
27
+ class MarkupAnchorDocument:
28
+ """描述一个待建立 anchor 索引的 DOM、样式表与兼容可见性规则。"""
29
+
30
+ key: str
31
+ root: etree._Element
32
+ stylesheet: MarkupStylesheet
33
+ visibility_scope: AnchorVisibilityScope = "all_ancestors"
34
+ text_normalization: AnchorTextNormalization = "unicode_whitespace"
35
+
36
+
37
+ class MarkupAnchorPolicy(Protocol):
38
+ """定义格式适配器生成标题、脚注 anchor 所需的稳定策略。"""
39
+
40
+ anchor_prefix: str
41
+ register_document_start: bool
42
+
43
+ def heading_identity(self, element: etree._Element, ordinal: int) -> str:
44
+ """返回当前标题参与稳定摘要的格式专属 identity。"""
45
+
46
+ def is_materializable_note(self, element: etree._Element, document: MarkupAnchorDocument) -> bool:
47
+ """判断当前元素是否是能够兑现文本 anchor 的格式专属脚注。"""
48
+
49
+ def note_identity(self, element: etree._Element, ordinal: int) -> str:
50
+ """返回当前脚注参与稳定摘要的格式专属 identity。"""
51
+
52
+
53
+ def element_id(element: etree._Element) -> str | None:
54
+ """返回元素去除首尾空白后的 HTML id 或 xml:id。"""
55
+ value = (element.get("id") or element.get(_XML_ID) or "").strip()
56
+ return value or None
57
+
58
+
59
+ def visible_element_text(element: etree._Element, document: MarkupAnchorDocument) -> str:
60
+ """按文档兼容配置解析祖先样式链,并返回最终可输出的纯文本。"""
61
+ inherited = TextStyle()
62
+ visibility_hidden = False
63
+ chain = [ancestor for ancestor in reversed(list(element.iterancestors())) if isinstance(ancestor.tag, str)]
64
+ if document.visibility_scope == "nearest_body":
65
+ body_index = next((index for index, ancestor in enumerate(chain) if local_name(ancestor) == "body"), None)
66
+ if body_index is not None:
67
+ chain = chain[body_index:]
68
+ chain.append(element)
69
+ for current in chain:
70
+ resolved = document.stylesheet.resolve(current, inherited, visibility_hidden)
71
+ if resolved.subtree_hidden:
72
+ return ""
73
+ inherited = resolved.text
74
+ visibility_hidden = resolved.visibility_hidden
75
+ value = visible_raw_text_with_style(element, document.stylesheet, inherited, visibility_hidden)
76
+ if document.text_normalization == "xhtml_whitespace":
77
+ return _XHTML_WHITESPACE_RE.sub(" ", html.unescape(value)).strip()
78
+ return " ".join(value.split())
79
+
80
+
81
+ def canonical_anchor(prefix: str, document_key: str, identity: str) -> str:
82
+ """按格式前缀、文档 key 与 identity 生成稳定的二十位摘要 anchor。"""
83
+ digest = hashlib.sha256(f"{document_key}#{identity}".encode()).hexdigest()[:20]
84
+ return f"{prefix}-{digest}"
85
+
86
+
87
+ class MarkupAnchorRegistry:
88
+ """统一登记多文档标题、脚注及源 fragment 到实际输出 anchor 的映射。"""
89
+
90
+ def __init__(self, documents: list[MarkupAnchorDocument], policy: MarkupAnchorPolicy) -> None:
91
+ """按调用方文档顺序建立稳定索引,并保留格式专属 identity 规则。"""
92
+ self._policy = policy
93
+ self._heading_anchors: dict[etree._Element, str] = {}
94
+ self._note_anchors: dict[etree._Element, str] = {}
95
+ self._targets: dict[tuple[str, str | None], str] = {}
96
+ self._heading_labels: dict[str, str] = {}
97
+ for document in documents:
98
+ self._register_document(document)
99
+
100
+ def _register_document(self, document: MarkupAnchorDocument) -> None:
101
+ """登记单个 DOM 的标题、脚注及全部可解析 fragment 别名。"""
102
+ headings: list[tuple[etree._Element, str]] = []
103
+ for element in document.root.iter():
104
+ if not isinstance(element.tag, str) or local_name(element) not in _HEADING_TAGS:
105
+ continue
106
+ if label := visible_element_text(element, document):
107
+ headings.append((element, label))
108
+ for ordinal, (heading, label) in enumerate(headings):
109
+ identity = self._policy.heading_identity(heading, ordinal)
110
+ anchor = canonical_anchor(self._policy.anchor_prefix, document.key, identity)
111
+ self._heading_anchors[heading] = anchor
112
+ self._heading_labels[anchor] = label
113
+
114
+ notes = [
115
+ element
116
+ for element in document.root.iter()
117
+ if isinstance(element.tag, str) and self._policy.is_materializable_note(element, document)
118
+ ]
119
+ for ordinal, note in enumerate(notes):
120
+ identity = self._policy.note_identity(note, ordinal)
121
+ self._note_anchors[note] = canonical_anchor(self._policy.anchor_prefix, document.key, identity)
122
+
123
+ if self._policy.register_document_start and headings:
124
+ self._targets[(document.key, None)] = self._heading_anchors[headings[0][0]]
125
+ for element in document.root.iter():
126
+ if not isinstance(element.tag, str) or not (fragment := element_id(element)):
127
+ continue
128
+ target_key = (document.key, fragment)
129
+ if target_key in self._targets:
130
+ continue
131
+ if anchor := self._target_anchor(element):
132
+ self._targets[target_key] = anchor
133
+
134
+ def _target_anchor(self, element: etree._Element) -> str | None:
135
+ """把任意 fragment 元素映射到自身、最近祖先或首个后代输出目标。"""
136
+ direct = self._heading_anchors.get(element) or self._note_anchors.get(element)
137
+ if direct is not None:
138
+ return direct
139
+ ancestor = next(
140
+ (parent for parent in element.iterancestors() if parent in self._heading_anchors or parent in self._note_anchors),
141
+ None,
142
+ )
143
+ if ancestor is not None:
144
+ return self._heading_anchors.get(ancestor) or self._note_anchors.get(ancestor)
145
+ descendant = next(
146
+ (
147
+ child
148
+ for child in element.iterdescendants()
149
+ if isinstance(child.tag, str) and (child in self._heading_anchors or child in self._note_anchors)
150
+ ),
151
+ None,
152
+ )
153
+ if descendant is None:
154
+ return None
155
+ return self._heading_anchors.get(descendant) or self._note_anchors.get(descendant)
156
+
157
+ def heading_anchor(self, heading: etree._Element) -> str | None:
158
+ """返回一个已登记标题的规范 anchor。"""
159
+ return self._heading_anchors.get(heading)
160
+
161
+ def heading_label(self, anchor: str) -> str | None:
162
+ """返回规范标题 anchor 对应的可见标题文本。"""
163
+ return self._heading_labels.get(anchor)
164
+
165
+ def note_anchor(self, note: etree._Element) -> str | None:
166
+ """返回一个已登记脚注的规范 anchor。"""
167
+ return self._note_anchors.get(note)
168
+
169
+ def resolve_target(self, document_key: str, fragment: str | None) -> str | None:
170
+ """按文档 key 与可选源 fragment 返回不带井号的规范 anchor。"""
171
+ return self._targets.get((document_key, fragment))
172
+
173
+
174
+ __all__ = [
175
+ "AnchorTextNormalization",
176
+ "AnchorVisibilityScope",
177
+ "MarkupAnchorDocument",
178
+ "MarkupAnchorPolicy",
179
+ "MarkupAnchorRegistry",
180
+ "canonical_anchor",
181
+ "element_id",
182
+ "visible_element_text",
183
+ ]
184
+
185
+ # 保持既有公开类型的 pickle 路径,所有旧、新入口指向同一个类。
186
+ preserve_type_module(MarkupAnchorDocument, "docvortex.analyzers.native._shared.markup.anchors")
187
+ preserve_type_module(MarkupAnchorPolicy, "docvortex.analyzers.native._shared.markup.anchors")
188
+ preserve_type_module(MarkupAnchorRegistry, "docvortex.analyzers.native._shared.markup.anchors")