docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,23 @@
1
+ from enum import Enum
2
+
3
+ from pydantic import BaseModel
4
+
5
+
6
+ class Script(str, Enum):
7
+ """Text script position."""
8
+
9
+ BASELINE = "baseline"
10
+ SUB = "sub"
11
+ SUPER = "super"
12
+
13
+
14
+ class Formatting(BaseModel):
15
+ """Formatting."""
16
+
17
+ bold: bool = False
18
+ italic: bool = False
19
+ underline: bool = False
20
+ underline_style: str = ""
21
+ emphasis: bool = False
22
+ strikethrough: bool = False
23
+ script: Script = Script.BASELINE
@@ -0,0 +1,50 @@
1
+ from typing import Any, BinaryIO
2
+
3
+ from ... import DocxModel
4
+
5
+
6
+ def convert_path(file_path: str) -> list[list[dict[str, Any]]]:
7
+ """从 DOCX 文件路径调用统一模型入口。"""
8
+
9
+ with open(file_path, "rb") as fh:
10
+ return convert_binary(fh)
11
+
12
+
13
+ def convert_binary(file_binary: BinaryIO) -> list[list[dict[str, Any]]]:
14
+ """兼容旧二进制转换函数,并转发给 DocxModel。"""
15
+
16
+ return DocxModel().predict(file_binary)
17
+
18
+
19
+ if __name__ == "__main__":
20
+ # provide a more robust command-line interface and resolve the demo
21
+ # document path relative to the project root instead of depending on
22
+ # the current working directory.
23
+ from pathlib import Path
24
+ import argparse
25
+
26
+ # climb up until we find pyproject.toml or reach a reasonable depth
27
+ def find_project_root(start: Path) -> Path:
28
+ current = start
29
+ for _ in range(6): # avoid infinite loops
30
+ if (current / "pyproject.toml").exists() or (current / "README.md").exists():
31
+ return current
32
+ if current.parent == current:
33
+ break
34
+ current = current.parent
35
+ return start
36
+
37
+ script_path = Path(__file__).resolve()
38
+ project_root = find_project_root(script_path.parent)
39
+ default_docx = project_root / "demo" / "docx" / "demo1.docx"
40
+
41
+ parser = argparse.ArgumentParser(description="Convert a DOCX file to internal JSON representation")
42
+ parser.add_argument(
43
+ "docx",
44
+ nargs="?",
45
+ default=str(default_docx),
46
+ help="path to the .docx file to convert (defaults to demo/docx/demo1.docx)",
47
+ )
48
+ args = parser.parse_args()
49
+
50
+ print(convert_path(args.docx))
@@ -0,0 +1,491 @@
1
+ """DOCX 列表与编号处理;共享当前 Converter 的单文档状态。"""
2
+
3
+ from typing import Optional
4
+ from docx.oxml.xmlchemy import BaseOxmlElement
5
+ from docx.text.paragraph import Paragraph
6
+ from loguru import logger
7
+ from .....schema import BlockType
8
+
9
+ from .context import _DocxConstants
10
+
11
+
12
+ class _DocxNumbering:
13
+ """集中维护列表与编号,不自行创建文档或持有跨文档缓存。"""
14
+
15
+ def _get_numId_and_ilvl(self, paragraph: Paragraph) -> tuple[Optional[int], Optional[int]]:
16
+ """
17
+ 获取段落的列表编号ID和层级。
18
+
19
+ Args:
20
+ paragraph: 段落对象
21
+
22
+ Returns:
23
+ tuple[Optional[int], Optional[int]]: (numId, ilvl) 元组
24
+ """
25
+ numPr = self._get_effective_numPr(paragraph)
26
+
27
+ if numPr is not None:
28
+ # 获取 numId 元素并提取值
29
+ namespaces = getattr(numPr, "nsmap", None) or _DocxConstants._BLIP_NAMESPACES
30
+ numId_elem = numPr.find("w:numId", namespaces=namespaces)
31
+ ilvl_elem = numPr.find("w:ilvl", namespaces=namespaces)
32
+ numId = numId_elem.get(self.XML_KEY) if numId_elem is not None else None
33
+ ilvl = ilvl_elem.get(self.XML_KEY) if ilvl_elem is not None else None
34
+
35
+ numId_int = self._str_to_int(numId, None)
36
+ ilvl_int = self._str_to_int(ilvl, None)
37
+ if numId_int == 0:
38
+ # numId=0 是 Word 中显式取消编号的信号,不能继续从样式继承编号层级。
39
+ return numId_int, ilvl_int
40
+ if numId_int is not None and ilvl_int is None:
41
+ ilvl_int = self._infer_numbering_ilvl_from_style(numId_int, paragraph)
42
+
43
+ return numId_int, ilvl_int
44
+
45
+ return None, None # 如果段落不是列表的一部分
46
+
47
+ def _get_numbering_num_element(self, numId: int) -> Optional[BaseOxmlElement]:
48
+ """根据 numId 获取 word/numbering.xml 中的 num 定义。"""
49
+ numbering_root = self._get_numbering_root()
50
+ if numbering_root is None:
51
+ return None
52
+
53
+ namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
54
+ return numbering_root.find(
55
+ f".//w:num[@w:numId='{numId}']",
56
+ namespaces=namespaces,
57
+ )
58
+
59
+ def _get_abstract_numbering_element(self, numId: int) -> Optional[BaseOxmlElement]:
60
+ """根据 numId 获取对应的 abstractNum 定义,用于复用编号层级解析逻辑。"""
61
+ numbering_root = self._get_numbering_root()
62
+ namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
63
+ if numbering_root is None:
64
+ return None
65
+
66
+ num_element = self._get_numbering_num_element(numId)
67
+ if num_element is None:
68
+ return None
69
+
70
+ abstract_num_id_elem = num_element.find(".//w:abstractNumId", namespaces=namespaces)
71
+ if abstract_num_id_elem is None:
72
+ return None
73
+
74
+ abstract_num_id = abstract_num_id_elem.get(self.XML_KEY)
75
+ if abstract_num_id is None:
76
+ return None
77
+
78
+ abstract_num_xpath = f".//w:abstractNum[@w:abstractNumId='{abstract_num_id}']"
79
+ return numbering_root.find(abstract_num_xpath, namespaces=namespaces)
80
+
81
+ def _infer_numbering_ilvl_from_style(self, numId: int, paragraph: Paragraph) -> Optional[int]:
82
+ """当 numPr 只有 numId 时,根据 numbering.xml 中的 pStyle 反查编号层级。"""
83
+ abstract_num_element = self._get_abstract_numbering_element(numId)
84
+ if abstract_num_element is None:
85
+ return None
86
+
87
+ style_ids = {
88
+ str(getattr(style, "style_id", "") or "") for style in self._iter_style_chain(self._get_paragraph_style(paragraph))
89
+ }
90
+ style_ids.discard("")
91
+ if not style_ids:
92
+ return None
93
+
94
+ namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
95
+ ilvl_attr = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}ilvl"
96
+ for lvl_element in abstract_num_element.findall(".//w:lvl", namespaces=namespaces):
97
+ p_style = lvl_element.find("w:pStyle", namespaces=namespaces)
98
+ if p_style is None:
99
+ continue
100
+ if p_style.get(self.XML_KEY) in style_ids:
101
+ return self._str_to_int(lvl_element.get(ilvl_attr), None)
102
+ return None
103
+
104
+ def _get_numbering_root(self) -> Optional[BaseOxmlElement]:
105
+ """Load and cache word/numbering.xml once per conversion."""
106
+ if self._numbering_root_loaded:
107
+ return self._numbering_root
108
+
109
+ self._numbering_root_loaded = True
110
+
111
+ if not hasattr(self.docx_obj, "part") or not hasattr(self.docx_obj.part, "package"):
112
+ return None
113
+
114
+ for part in self.docx_obj.part.package.parts:
115
+ if "numbering" in part.partname:
116
+ self._numbering_root = part.element
117
+ break
118
+
119
+ return self._numbering_root
120
+
121
+ def _get_numbering_level_definition(self, numId: int, ilvl: int) -> Optional[BaseOxmlElement]:
122
+ """Resolve and cache the numbering level definition for a numId/ilvl pair."""
123
+ cache_key = (numId, ilvl)
124
+ if cache_key in self._numbering_level_cache:
125
+ return self._numbering_level_cache[cache_key]
126
+
127
+ numbering_root = self._get_numbering_root()
128
+ namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
129
+ lvl_element: Optional[BaseOxmlElement] = None
130
+
131
+ abstract_num_element = self._get_abstract_numbering_element(numId)
132
+ if numbering_root is not None and abstract_num_element is not None:
133
+ lvl_xpath = f".//w:lvl[@w:ilvl='{ilvl}']"
134
+ lvl_element = abstract_num_element.find(lvl_xpath, namespaces=namespaces)
135
+
136
+ self._numbering_level_cache[cache_key] = lvl_element
137
+ return lvl_element
138
+
139
+ def _get_numbering_level_start(self, numId: int, ilvl: int) -> int:
140
+ """解析编号层级的起始值,优先使用 num/lvlOverride,其次使用 abstractNum/lvl/start。"""
141
+ cache_key = (numId, ilvl)
142
+ if cache_key in self._numbering_start_cache:
143
+ return self._numbering_start_cache[cache_key]
144
+
145
+ namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
146
+ start = 1
147
+ num_element = self._get_numbering_num_element(numId)
148
+ if num_element is not None:
149
+ override = num_element.find(
150
+ f"w:lvlOverride[@w:ilvl='{ilvl}']",
151
+ namespaces=namespaces,
152
+ )
153
+ if override is not None:
154
+ start_override = override.find(
155
+ "w:startOverride",
156
+ namespaces=namespaces,
157
+ )
158
+ if start_override is not None:
159
+ start = self._str_to_int(start_override.get(self.XML_KEY), start)
160
+ self._numbering_start_cache[cache_key] = start
161
+ return start
162
+
163
+ lvl_element = self._get_numbering_level_definition(numId, ilvl)
164
+ if lvl_element is not None:
165
+ start_element = lvl_element.find("w:start", namespaces=namespaces)
166
+ if start_element is not None:
167
+ start = self._str_to_int(start_element.get(self.XML_KEY), start)
168
+
169
+ self._numbering_start_cache[cache_key] = start
170
+ return start
171
+
172
+ def _advance_list_counter(self, numId: int, ilvl: int) -> int:
173
+ """推进 Word 编号计数,并返回当前列表项应显示的真实序号。"""
174
+ counter_key = (numId, ilvl)
175
+ if counter_key not in self.list_counters:
176
+ current_number = self._get_numbering_level_start(numId, ilvl)
177
+ else:
178
+ current_number = self.list_counters[counter_key] + 1
179
+ self.list_counters[counter_key] = current_number
180
+
181
+ # 父级编号前进后,子级编号应在下次出现时重新从定义的起始值开始。
182
+ for key in list(self.list_counters.keys()):
183
+ counter_num_id, counter_ilevel = key
184
+ if counter_num_id == numId and counter_ilevel > ilvl:
185
+ self.list_counters.pop(key, None)
186
+
187
+ return current_number
188
+
189
+ def _is_numbered_list(self, numId: int, ilvl: int) -> bool:
190
+ """
191
+ 根据 numFmt 值检查列表是否为编号列表。
192
+
193
+ Args:
194
+ numId: 列表编号ID
195
+ ilvl: 列表层级
196
+
197
+ Returns:
198
+ bool: 如果是编号列表返回 True,否则返回 False
199
+ """
200
+ try:
201
+ lvl_element = self._get_numbering_level_definition(numId, ilvl)
202
+ if lvl_element is None:
203
+ return False
204
+ namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
205
+
206
+ # 获取 numFmt 元素
207
+ num_fmt_element = lvl_element.find(".//w:numFmt", namespaces=namespaces)
208
+ if num_fmt_element is None:
209
+ return False
210
+
211
+ num_fmt = num_fmt_element.get("{http://schemas.openxmlformats.org/wordprocessingml/2006/main}val")
212
+
213
+ # 编号格式包括: decimal, lowerRoman, upperRoman, lowerLetter, upperLetter
214
+ # 项目符号格式包括: bullet
215
+ numbered_formats = {
216
+ "decimal",
217
+ "lowerRoman",
218
+ "upperRoman",
219
+ "lowerLetter",
220
+ "upperLetter",
221
+ "decimalZero",
222
+ }
223
+
224
+ return num_fmt in numbered_formats
225
+
226
+ except Exception as e:
227
+ logger.debug(f"Error determining if list is numbered: {e}")
228
+ return False
229
+
230
+ def _add_list_item(
231
+ self,
232
+ *,
233
+ numid: int,
234
+ ilevel: int,
235
+ elements: list,
236
+ is_numbered: bool = False,
237
+ text: str = "",
238
+ equations: list = None,
239
+ ) -> list:
240
+ """
241
+ 添加列表项。
242
+
243
+ 生成的列表结构:
244
+ {
245
+ "type": "list",
246
+ "attribute": "ordered" / "unordered",
247
+ "ilevel": 0,
248
+ "content": [
249
+ {"type": "text", "content": "列表项文本"},
250
+ {"type": "list", "attribute": "...", "ilevel": 1, "content": [...]},
251
+ {"type": "text", "content": "另一个列表项"}
252
+ ]
253
+ }
254
+
255
+ Args:
256
+ numid: 列表ID
257
+ ilevel: 缩进等级
258
+ elements: 元素列表
259
+ is_numbered: 是否编号
260
+ text: 处理后的文本(包含公式标记)
261
+ equations: 公式列表
262
+
263
+ Returns:
264
+ list[RefItem]: 元素引用列表
265
+ """
266
+ if equations is None:
267
+ equations = []
268
+ if not elements:
269
+ return None
270
+
271
+ # 构建 content_text,处理行内公式和超链接
272
+ content_text = self._build_text_with_equations_and_hyperlinks(elements, text, equations)
273
+ content_text = self._normalize_text_block_content(content_text)
274
+ if not content_text:
275
+ return None
276
+
277
+ # 确定列表属性
278
+ list_attribute = "ordered" if is_numbered else "unordered"
279
+ list_start = self._advance_list_counter(numid, ilevel) if is_numbered else None
280
+
281
+ # 情况 1: 不存在上一个列表ID,或遇到了不同 numId 的新列表,创建新的顶层列表
282
+ if self.pre_num_id == -1 or self.pre_num_id != numid:
283
+ # 切换到不同的列表时,先重置旧列表状态
284
+ if self.pre_num_id != -1:
285
+ self._close_active_list()
286
+
287
+ list_block = {
288
+ "type": BlockType.LIST,
289
+ "attribute": list_attribute,
290
+ "content": [],
291
+ "ilevel": ilevel,
292
+ }
293
+ if list_start is not None:
294
+ list_block["start"] = list_start
295
+ self.cur_page.append(list_block)
296
+ # 入栈, 记录当前的列表块
297
+ self.list_block_stack.append(list_block)
298
+
299
+ list_item = {
300
+ "type": BlockType.TEXT,
301
+ "content": content_text,
302
+ }
303
+
304
+ list_block["content"].append(list_item)
305
+ self.pre_num_id = numid
306
+ self.pre_ilevel = ilevel
307
+
308
+ # 情况 2: 增加缩进,打开子列表
309
+ elif (
310
+ self.pre_num_id == numid # 同一个列表
311
+ and self.pre_ilevel != -1 # 上一个缩进级别已知
312
+ and self.pre_ilevel < ilevel # 当前层级比之前更缩进
313
+ ):
314
+ # 创建新的子列表块
315
+ child_list_block = {
316
+ "type": BlockType.LIST,
317
+ "attribute": list_attribute,
318
+ "content": [],
319
+ "ilevel": ilevel,
320
+ }
321
+ if list_start is not None:
322
+ child_list_block["start"] = list_start
323
+
324
+ if not self.list_block_stack:
325
+ logger.warning(
326
+ f"Missing DOCX list parent for increased indent; numid={numid}, ilevel={ilevel}. Starting a new list block."
327
+ )
328
+ self.cur_page.append(child_list_block)
329
+ self.list_block_stack.append(child_list_block)
330
+ child_list_block["content"].append(
331
+ {
332
+ "type": BlockType.TEXT,
333
+ "content": content_text,
334
+ }
335
+ )
336
+ self.pre_ilevel = ilevel
337
+ return None
338
+
339
+ # 获取栈顶的列表块,将子列表直接添加到其content中
340
+ parent_list_block = self.list_block_stack[-1]
341
+ parent_list_block["content"].append(child_list_block)
342
+
343
+ # 入栈, 记录当前的列表块
344
+ self.list_block_stack.append(child_list_block)
345
+
346
+ # 添加当前列表项到子列表
347
+ list_item = {
348
+ "type": BlockType.TEXT,
349
+ "content": content_text,
350
+ }
351
+ child_list_block["content"].append(list_item)
352
+
353
+ # 更新目前缩进
354
+ self.pre_ilevel = ilevel
355
+
356
+ # 情况3: 减少缩进,关闭子列表
357
+ elif (
358
+ self.pre_num_id == numid # 同一个列表
359
+ and self.pre_ilevel != -1 # 上一个缩进级别已知
360
+ and ilevel < self.pre_ilevel # 当前层级比之前更少缩进
361
+ ):
362
+ # 出栈,直到找到匹配的 ilevel
363
+ while self.list_block_stack:
364
+ top_list_block = self.list_block_stack[-1]
365
+ if top_list_block["ilevel"] == ilevel:
366
+ break
367
+ self.list_block_stack.pop()
368
+ if not self.list_block_stack:
369
+ logger.warning(f"Malformed DOCX list nesting; numid={numid}, ilevel={ilevel}. Starting a new list block.")
370
+ list_block = {
371
+ "type": BlockType.LIST,
372
+ "attribute": list_attribute,
373
+ "content": [],
374
+ "ilevel": ilevel,
375
+ }
376
+ if list_start is not None:
377
+ list_block["start"] = list_start
378
+ self.cur_page.append(list_block)
379
+ self.list_block_stack.append(list_block)
380
+ else:
381
+ list_block = self.list_block_stack[-1]
382
+
383
+ list_item = {
384
+ "type": BlockType.TEXT,
385
+ "content": content_text,
386
+ }
387
+ list_block["content"].append(list_item)
388
+ self.pre_ilevel = ilevel
389
+
390
+ # 情况 4: 同级列表项(相同缩进)
391
+ elif self.pre_num_id == numid and self.pre_ilevel == ilevel:
392
+ if not self.list_block_stack:
393
+ logger.warning(
394
+ f"Missing DOCX list block for same indent; numid={numid}, ilevel={ilevel}. Starting a new list block."
395
+ )
396
+ list_block = {
397
+ "type": BlockType.LIST,
398
+ "attribute": list_attribute,
399
+ "content": [],
400
+ "ilevel": ilevel,
401
+ }
402
+ if list_start is not None:
403
+ list_block["start"] = list_start
404
+ self.cur_page.append(list_block)
405
+ self.list_block_stack.append(list_block)
406
+ else:
407
+ # 获取栈顶的列表块
408
+ list_block = self.list_block_stack[-1]
409
+
410
+ list_item = {
411
+ "type": BlockType.TEXT,
412
+ "content": content_text,
413
+ }
414
+ list_block["content"].append(list_item)
415
+
416
+ else:
417
+ logger.warning(
418
+ "Unexpected DOCX list state in _add_list_item: "
419
+ f"pre_num_id={self.pre_num_id}, numid={numid}, "
420
+ f"pre_ilevel={self.pre_ilevel}, ilevel={ilevel}, "
421
+ f"stack_depth={len(self.list_block_stack)}. "
422
+ )
423
+
424
+ def _detect_heading_list_numids(self) -> set:
425
+ """
426
+ 预扫描文档,检测用作章节标题的列表numId。
427
+
428
+ 判断依据(需同时满足两个条件):
429
+ 1. 该numId的列表项之间穿插了非列表的正文内容(段落/表格等);
430
+ 2. 该numId的列表项出现在**多个不同的缩进层级**(ilevel > 1种),
431
+ 即为真正的多级列表结构,而非普通的单级内容条目列表。
432
+
433
+ 这样可以避免将"多段内容条目之间穿插了小标签"的单级列表误判为标题列表。
434
+
435
+ Returns:
436
+ set: 应当转换为标题块的列表numId集合
437
+ """
438
+ heading_numids = set()
439
+ # 收集文档元素序列:("list", numid, ilevel) 或 ("content",)
440
+ items = []
441
+ # 记录每个numId出现过的所有ilevel,用于判断是否为真正的多级列表
442
+ numid_ilvels: dict[int, set] = {}
443
+
444
+ for element in self.docx_obj.element.body:
445
+ tag_name = self._local_name(element)
446
+ if tag_name is None:
447
+ continue
448
+ if tag_name == "p":
449
+ try:
450
+ paragraph = Paragraph(element, self.docx_obj)
451
+ p_style_id, _ = self._get_label_and_level(paragraph)
452
+ numid, ilevel = self._get_numId_and_ilvl(paragraph)
453
+ if numid == 0:
454
+ numid = None
455
+ text = self._get_paragraph_text(paragraph).strip()
456
+ except Exception:
457
+ continue
458
+
459
+ if numid is not None and ilevel is not None and p_style_id not in ["Title", "Heading"] and text:
460
+ items.append(("list", numid, ilevel))
461
+ if numid not in numid_ilvels:
462
+ numid_ilvels[numid] = set()
463
+ numid_ilvels[numid].add(ilevel)
464
+ elif p_style_id not in ["Title", "Heading"] and text:
465
+ items.append(("content", None, None))
466
+ elif tag_name == "tbl":
467
+ items.append(("content", None, None))
468
+
469
+ # 对每个numId,检测其列表项之间是否有正文内容穿插
470
+ # seen_numids[numid] = True 表示该numId的最后一个列表项之后出现了正文内容
471
+ seen_numids: dict[int, bool] = {}
472
+
473
+ for item_type, numid, ilevel in items:
474
+ if item_type == "list":
475
+ if numid in seen_numids and seen_numids[numid]:
476
+ # 上次列表项之后出现了正文内容,满足条件1
477
+ heading_numids.add(numid)
478
+ seen_numids[numid] = False # 重置:记录该numId出现了新列表项
479
+ elif item_type == "content":
480
+ # 将所有已见numId标记为"之后出现了正文内容"
481
+ for nid in seen_numids:
482
+ seen_numids[nid] = True
483
+
484
+ # 条件2:只保留真正的多级列表(出现过多于1种ilevel的numId)
485
+ # 单级列表(如只有ilevel=0的内容条目列表)即使有正文段落穿插也不应转换为标题
486
+ heading_numids = {nid for nid in heading_numids if len(numid_ilvels.get(nid, set())) > 1}
487
+
488
+ if heading_numids:
489
+ logger.debug(f"Detected heading-style list numIds (will convert to title blocks): {heading_numids}")
490
+
491
+ return heading_numids
@@ -0,0 +1,57 @@
1
+ """DOCX Mammoth 兼容层使用的 Office XML 解析辅助函数。"""
2
+
3
+ import xml.dom.minidom
4
+
5
+ from mammoth.docx.xmlparser import XmlText, XmlElement
6
+ from mammoth.docx.office_xml import _collapse_alternate_content, _namespaces
7
+
8
+
9
+ def parse_xml_str(xml_str, namespace_mapping=None):
10
+ if namespace_mapping is None:
11
+ namespace_prefixes = {}
12
+ else:
13
+ namespace_prefixes = dict((uri, prefix) for prefix, uri in namespace_mapping)
14
+
15
+ document = xml.dom.minidom.parseString(xml_str)
16
+
17
+ def convert_node(node):
18
+ if node.nodeType == xml.dom.Node.ELEMENT_NODE:
19
+ return convert_element(node)
20
+ elif node.nodeType == xml.dom.Node.TEXT_NODE:
21
+ return XmlText(node.nodeValue)
22
+ else:
23
+ return None
24
+
25
+ def convert_element(element):
26
+ converted_name = convert_name(element)
27
+
28
+ converted_attributes = dict(
29
+ (convert_name(attribute), attribute.value)
30
+ for attribute in element.attributes.values()
31
+ if attribute.namespaceURI != "http://www.w3.org/2000/xmlns/"
32
+ )
33
+
34
+ converted_children = []
35
+ for child_node in element.childNodes:
36
+ converted_child_node = convert_node(child_node)
37
+ if converted_child_node is not None:
38
+ converted_children.append(converted_child_node)
39
+
40
+ return XmlElement(converted_name, converted_attributes, converted_children)
41
+
42
+ def convert_name(node):
43
+ if node.namespaceURI is None:
44
+ return node.localName
45
+ else:
46
+ prefix = namespace_prefixes.get(node.namespaceURI)
47
+ if prefix is None:
48
+ return "{%s}%s" % (node.namespaceURI, node.localName)
49
+ else:
50
+ return "%s:%s" % (prefix, node.localName)
51
+
52
+ return convert_node(document.documentElement)
53
+
54
+
55
+ def read_str(xml_str):
56
+ i = parse_xml_str(xml_str, _namespaces)
57
+ return _collapse_alternate_content(i)[0]