docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,515 @@
1
+ """DOCX 表格处理;共享当前 Converter 的单文档状态。"""
2
+
3
+ import re
4
+ from io import BytesIO
5
+ from typing import Any, Optional
6
+ from docx import Document
7
+ from docx.oxml.xmlchemy import BaseOxmlElement
8
+ from loguru import logger
9
+ from mammoth.conversion import convert_document_element_to_html
10
+ from mammoth.docx import body_xml
11
+ from .office_xml import read_str
12
+ from .....schema import BlockType
13
+
14
+ from .context import _DocxConstants
15
+
16
+
17
+ class _DocxTables:
18
+ """集中维护表格,不自行创建文档或持有跨文档缓存。"""
19
+
20
+ @staticmethod
21
+ def _mammoth_top_level_table_document(document):
22
+ """只保留 DOCX 正文顶层表格节点,避免 Mammoth 转换非表格正文。"""
23
+ from mammoth import documents as _mammoth_documents
24
+
25
+ return document.copy(children=[child for child in document.children if isinstance(child, _mammoth_documents.Table)])
26
+
27
+ def _preparse_tables_with_mammoth(self, file_bytes: bytes) -> list:
28
+ """
29
+ 使用 mammoth 在完整 DOCX 上下文中预解析所有顶层表格的 HTML。
30
+
31
+ 孤立模式下(仅传入 <w:tbl> XML 片段),mammoth 缺少编号定义
32
+ (word/numbering.xml)、样式(word/styles.xml)和关系
33
+ (word/_rels/document.xml.rels)等上下文,在遇到含列表项或图片
34
+ 的单元格时会抛出 AttributeError。这里让 mammoth 读取完整 DOCX
35
+ 包上下文,但通过 transform_document 只转换顶层表格节点,避免把
36
+ 非表格正文转换成巨大的 HTML 字符串。
37
+
38
+ 图片会被 mammoth 转换为内联 data-URI base64 格式(<img src="data:...">)。
39
+
40
+ 注意:mammoth 不支持 OMML、Equation XML 或 MTEF OLE 公式,会静默丢弃
41
+ 表格单元格内的公式。本方法在获取 mammoth HTML 后,会同步遍历原始 DOCX XML,
42
+ 将丢失的公式重新注入对应的 HTML 单元格。
43
+
44
+ Returns:
45
+ list[str | None]: 与正文顶层表格对齐的 HTML 列表;None 表示该表格
46
+ 未找到可靠 Mammoth 结果,后续走孤立 XML 回退解析
47
+ """
48
+ try:
49
+ import mammoth as _mammoth
50
+ from bs4 import BeautifulSoup as _BeautifulSoup
51
+
52
+ result = _mammoth.convert_to_html(
53
+ BytesIO(file_bytes),
54
+ transform_document=self._mammoth_top_level_table_document,
55
+ )
56
+ soup = _BeautifulSoup(result.value, "html.parser")
57
+
58
+ # 仅保留顶层表格,排除嵌套在其他表格单元格内的子表格
59
+ all_tables = soup.find_all("table")
60
+ top_level_tables = [t for t in all_tables if not t.find_parent("table")]
61
+
62
+ # 同步加载 DOCX XML,获取所有顶层表格元素,用于公式注入
63
+ docx_obj = Document(BytesIO(file_bytes))
64
+ xml_top_tables = [elem for elem in docx_obj.element.body if self._local_name(elem) == "tbl"]
65
+
66
+ logger.debug(f"Pre-parsed {len(top_level_tables)} top-level tables via filtered mammoth conversion")
67
+
68
+ result_tables = self._align_mammoth_tables_to_xml_tables(
69
+ top_level_tables,
70
+ xml_top_tables,
71
+ docx_obj.part,
72
+ )
73
+ return result_tables
74
+ except Exception as e:
75
+ logger.debug(f"Could not pre-parse tables with filtered mammoth conversion: {e}")
76
+ return []
77
+
78
+ def _align_mammoth_tables_to_xml_tables(
79
+ self,
80
+ html_tables: Any,
81
+ xml_tables: Any,
82
+ source_part: Any,
83
+ ) -> list:
84
+ """
85
+ 将 Mammoth 输出表格按正文顶层 XML 表格重新对齐。
86
+
87
+ 某些 DOCX 会在文本框、图片形状或兼容结构中包含表格,Mammoth 完整
88
+ 文档转换时可能把这些结构表格也输出为顶层 HTML table;但正文遍历
89
+ 只会在真实 body/w:tbl 上调用 _handle_tables。这里按 XML 表格的
90
+ 顺序扫描 Mammoth 候选表,跳过不属于正文顶层表格的候选,避免后续
91
+ _mammoth_table_idx 顺序消费时发生错位。
92
+ """
93
+ aligned_tables = []
94
+ html_index = 0
95
+ matched_count = 0
96
+
97
+ for xml_table in xml_tables:
98
+ matched_html_table = None
99
+ scan_index = html_index
100
+ while scan_index < len(html_tables):
101
+ candidate = html_tables[scan_index]
102
+ if self._mammoth_table_matches_xml_table(candidate, xml_table):
103
+ matched_html_table = candidate
104
+ html_index = scan_index + 1
105
+ matched_count += 1
106
+ break
107
+ scan_index += 1
108
+
109
+ if matched_html_table is None:
110
+ aligned_tables.append(None)
111
+ continue
112
+
113
+ matched_html_table = self._inject_equations_into_table(
114
+ matched_html_table,
115
+ xml_table,
116
+ source_part,
117
+ )
118
+ aligned_tables.append(str(matched_html_table))
119
+
120
+ if len(html_tables) != len(xml_tables):
121
+ logger.debug(f"Aligned {matched_count}/{len(xml_tables)} body tables from {len(html_tables)} mammoth tables")
122
+ return aligned_tables
123
+
124
+ @staticmethod
125
+ def _mammoth_table_matches_xml_table(html_table, xml_table) -> bool:
126
+ """
127
+ 判断 Mammoth HTML 表格是否对应当前正文 XML 表格。
128
+
129
+ 文本表优先比较去空白后的表格文本,避免同为 1x1 的图片/文本框表格
130
+ 误占正文表格位置;无文本表格再使用结构和图片数量兜底。
131
+ """
132
+ xml_signature = _DocxTables._xml_table_signature(xml_table)
133
+ html_signature = _DocxTables._html_table_signature(html_table)
134
+
135
+ if xml_signature["text"] or html_signature["text"]:
136
+ if not _DocxTables._table_text_matches(xml_signature["text"], html_signature["text"]):
137
+ return False
138
+ return (
139
+ xml_signature["cell_count"] == html_signature["cell_count"]
140
+ or xml_signature["row_count"] == html_signature["row_count"]
141
+ )
142
+
143
+ return (
144
+ xml_signature["row_count"] == html_signature["row_count"]
145
+ and xml_signature["cell_count"] == html_signature["cell_count"]
146
+ and xml_signature["image_count"] == html_signature["image_count"]
147
+ )
148
+
149
+ @staticmethod
150
+ def _xml_table_char_fragment(node: Any) -> str:
151
+ """按 Mammoth 规则把字符级 OOXML 元素渲染为表格签名文本。
152
+
153
+ 仅拼接 w:t 会遗漏不间断连字符、软连字符和 Symbol 字符,
154
+ 导致 XML 签名与 Mammoth HTML 签名不一致。未映射的 w:sym 在
155
+ Mammoth 中会被忽略,此处同样返回空字符串。
156
+ """
157
+ w_ns = _DocxConstants._BLIP_NAMESPACES["w"]
158
+ if node.tag == f"{{{w_ns}}}t":
159
+ return node.text or ""
160
+ if node.tag == f"{{{w_ns}}}noBreakHyphen":
161
+ # NON-BREAKING HYPHEN U+2011,与 Mammoth 的 HTML 渲染一致。
162
+ return "‑"
163
+ if node.tag == f"{{{w_ns}}}softHyphen":
164
+ # SOFT HYPHEN U+00AD,与 Mammoth 的 HTML 渲染一致。
165
+ return "­"
166
+ if node.tag == f"{{{w_ns}}}sym":
167
+ try:
168
+ from mammoth.docx.dingbats import dingbats
169
+ except ImportError:
170
+ return ""
171
+
172
+ font = node.get(f"{{{w_ns}}}font", "")
173
+ char = node.get(f"{{{w_ns}}}char", "")
174
+ try:
175
+ code = dingbats.get((font, int(char, 16)))
176
+ if code is None and re.match(r"^F0..", char):
177
+ code = dingbats.get((font, int(char[2:], 16)))
178
+ except (TypeError, ValueError):
179
+ return ""
180
+ return chr(code) if code is not None else ""
181
+ return ""
182
+
183
+ @staticmethod
184
+ def _xml_table_signature(xml_table) -> dict:
185
+ """提取与 Mammoth 渲染规则一致的 XML 表格轻量签名。
186
+
187
+ 按文档顺序收集 w:t 及特殊字符元素,并且只处理 WordprocessingML
188
+ 命名空间,以排除 Mammoth 不渲染的 OMML m:t 公式文本。
189
+ """
190
+ w_ns = _DocxConstants._BLIP_NAMESPACES["w"]
191
+ char_tags = (
192
+ f"{{{w_ns}}}t",
193
+ f"{{{w_ns}}}noBreakHyphen",
194
+ f"{{{w_ns}}}softHyphen",
195
+ f"{{{w_ns}}}sym",
196
+ )
197
+ text = "".join(_DocxTables._xml_table_char_fragment(node) for node in xml_table.iter() if node.tag in char_tags)
198
+ return {
199
+ "row_count": len(xml_table.xpath('.//*[local-name()="tr"]')),
200
+ "cell_count": len(xml_table.xpath('.//*[local-name()="tc"]')),
201
+ "image_count": len(xml_table.xpath('.//*[local-name()="blip" or local-name()="imagedata"]')),
202
+ "text": _DocxTables._normalize_table_match_text(text),
203
+ }
204
+
205
+ @staticmethod
206
+ def _html_table_signature(html_table) -> dict:
207
+ """提取 HTML 表格的轻量签名,用于过滤 Mammoth 额外生成的表格。"""
208
+ return {
209
+ "row_count": len(html_table.find_all("tr")),
210
+ "cell_count": len(html_table.find_all(["td", "th"])),
211
+ "image_count": len(html_table.find_all("img")),
212
+ "text": _DocxTables._normalize_table_match_text(html_table.get_text("", strip=True)),
213
+ }
214
+
215
+ @staticmethod
216
+ def _normalize_table_match_text(text: str) -> str:
217
+ """统一表格匹配文本,消除 Word 拆字和 Mammoth 空白差异。"""
218
+ return re.sub(r"\s+", "", text or "")
219
+
220
+ @staticmethod
221
+ def _table_text_matches(xml_text: str, html_text: str) -> bool:
222
+ """比较表格文本是否指向同一个正文表格。"""
223
+ if not xml_text or not html_text:
224
+ return False
225
+ if xml_text == html_text:
226
+ return True
227
+ return xml_text.startswith(html_text) or html_text.startswith(xml_text)
228
+
229
+ def _inject_equations_into_table(
230
+ self,
231
+ html_table: Any,
232
+ xml_table: Any,
233
+ source_part: Any,
234
+ ) -> Any:
235
+ """
236
+ 将 DOCX XML 表格中的 OMML/Equation XML/MTEF 公式注入 mammoth HTML 表格。
237
+
238
+ mammoth 会静默丢弃 OMML、Equation XML 与 MTEF OLE 公式,导致含公式
239
+ 的表格单元格在 HTML 中为空。本方法并行遍历 HTML 表格(BeautifulSoup 对象)
240
+ 和 XML 表格(lxml 元素),对含有 OMML 公式的单元格用包含公式占位符的内容
241
+ 替换原来的空内容。
242
+
243
+ Args:
244
+ html_table: BeautifulSoup 的 Tag 对象,代表 mammoth 生成的 <table> 元素
245
+ xml_table: lxml 的 Element 对象,代表原始 DOCX 中对应的 <w:tbl> 元素
246
+
247
+ Returns:
248
+ BeautifulSoup Tag: 注入公式后的 <table> 元素(原地修改并返回)
249
+ """
250
+ W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
251
+
252
+ # 快速检查:该表格是否含有任何公式
253
+ if not any(
254
+ kind in _DocxConstants._FORMULA_TOKEN_KINDS for kind, _value in self._docx_formula_tokens(xml_table, source_part)
255
+ ):
256
+ return html_table
257
+
258
+ from bs4 import BeautifulSoup
259
+
260
+ html_rows = html_table.find_all("tr")
261
+ xml_rows = xml_table.findall(f"{{{W_NS}}}tr")
262
+
263
+ if len(html_rows) != len(xml_rows):
264
+ logger.debug(f"Table row count mismatch when injecting equations: HTML {len(html_rows)} vs XML {len(xml_rows)}")
265
+ return html_table
266
+
267
+ for html_row, xml_row in zip(html_rows, xml_rows):
268
+ html_cells = html_row.find_all(["td", "th"])
269
+ xml_cells = xml_row.findall(f"{{{W_NS}}}tc")
270
+
271
+ if len(html_cells) != len(xml_cells):
272
+ continue
273
+
274
+ for html_cell, xml_cell in zip(html_cells, xml_cells):
275
+ if not any(
276
+ kind in _DocxConstants._FORMULA_TOKEN_KINDS
277
+ for kind, _value in self._docx_formula_tokens(
278
+ xml_cell,
279
+ source_part,
280
+ )
281
+ ):
282
+ continue
283
+
284
+ # 该单元格含公式,重建其 HTML 内容以保留公式
285
+ new_content = self._build_cell_html_with_equations(
286
+ xml_cell,
287
+ source_part,
288
+ )
289
+ if new_content:
290
+ html_cell.clear()
291
+ new_soup = BeautifulSoup(new_content, "html.parser")
292
+ for child in list(new_soup.children):
293
+ html_cell.append(child)
294
+
295
+ return html_table
296
+
297
+ def _build_cell_html_with_equations(
298
+ self,
299
+ xml_cell: Any,
300
+ source_part: Any,
301
+ ) -> str:
302
+ """
303
+ 为含 OMML/Equation XML/MTEF 公式的表格单元格构建 HTML 内容字符串。
304
+
305
+ 遍历单元格内的段落,将普通文本和 OMML/Equation XML/MTEF 公式
306
+ 混合在一起,生成与 mammoth 输出风格一致的 HTML 片段。
307
+
308
+ Args:
309
+ xml_cell: lxml Element,代表 DOCX 中的 <w:tc> 元素
310
+
311
+ Returns:
312
+ str: 单元格内容的 HTML 字符串,如 "<p>text<eq>latex</eq></p>";
313
+ 若单元格为空则返回空字符串
314
+ """
315
+ parts = []
316
+ for child in xml_cell:
317
+ child_tag = self._local_name(child)
318
+ if child_tag is None:
319
+ continue
320
+ if child_tag == "p":
321
+ para_html = self._build_paragraph_html_with_equations(
322
+ child,
323
+ source_part,
324
+ )
325
+ if para_html is not None:
326
+ parts.append(para_html)
327
+ # 嵌套表格暂不处理,由外层逻辑负责
328
+ return "".join(parts)
329
+
330
+ def _inject_equations_into_table_html(
331
+ self,
332
+ html: str,
333
+ xml_table: Any,
334
+ source_part: Any,
335
+ ) -> str:
336
+ """把孤立表格 HTML 包装为 Tag 后复用统一公式注入逻辑。"""
337
+
338
+ from bs4 import BeautifulSoup
339
+
340
+ soup = BeautifulSoup(html, "html.parser")
341
+ html_table = soup.find("table")
342
+ if html_table is None:
343
+ return html
344
+ return str(
345
+ self._inject_equations_into_table(
346
+ html_table,
347
+ xml_table,
348
+ source_part,
349
+ )
350
+ )
351
+
352
+ def _build_paragraph_html_with_equations(
353
+ self,
354
+ xml_para: Any,
355
+ source_part: Any,
356
+ ) -> Optional[str]:
357
+ """
358
+ 为可能含 OMML/Equation XML/MTEF 公式的段落构建 HTML 字符串。
359
+
360
+ 使用与 _handle_equations_in_text 相同的迭代逻辑:
361
+ - 普通 <w:t> 元素的文本直接收集
362
+ - <m:oMath> 元素转换为 LaTeX 并包装为公式占位符 <eq>...</eq>
363
+ - <m:t> 等 math 命名空间下的 <t> 元素因标签中含 "math" 而被跳过,
364
+ 避免在 oMath2Latex 已处理整个 oMath 子树后重复提取
365
+
366
+ Args:
367
+ xml_para: lxml Element,代表 DOCX 中的 <w:p> 元素
368
+
369
+ Returns:
370
+ str | None: 格式为 "<p>...</p>" 的 HTML 字符串;段落为空时返回 None
371
+ """
372
+ items: list[str] = []
373
+ for token_kind, value in self._docx_formula_tokens(xml_para, source_part):
374
+ if token_kind == "text":
375
+ items.append(value)
376
+ else:
377
+ items.append(self.equation_bookends.format(EQ=value))
378
+
379
+ if not items:
380
+ return None
381
+ return f"<p>{''.join(items)}</p>"
382
+
383
+ def _handle_tables(self, element: BaseOxmlElement):
384
+ """
385
+ 处理表格。
386
+
387
+ 优先使用完整文档 mammoth 转换的预解析结果(支持列表、图片、样式等
388
+ 复杂单元格内容),若预解析结果耗尽则回退到孤立 XML 解析模式。
389
+
390
+ Args:
391
+ element: 元素对象
392
+ Returns:
393
+ list[RefItem]: 元素引用列表
394
+ """
395
+ # 优先使用预解析表格(完整文档上下文,能正确处理列表/图片等)
396
+ if self._mammoth_table_idx < len(self._mammoth_tables_html):
397
+ html = self._mammoth_tables_html[self._mammoth_table_idx]
398
+ self._mammoth_table_idx += 1
399
+ if html is not None:
400
+ html = self._normalize_table_colspans(html)
401
+ table_block = {
402
+ "type": BlockType.TABLE,
403
+ "content": html,
404
+ }
405
+ self.cur_page.append(table_block)
406
+ return
407
+
408
+ # 回退:孤立 XML 解析模式(原始方案,不含文档上下文)
409
+ table = read_str(element.xml)
410
+ body_reader = body_xml.reader()
411
+ t = body_reader.read_all([table])
412
+ res = convert_document_element_to_html(t.value[0])
413
+ html = self._normalize_table_colspans(res.value)
414
+ html = self._inject_equations_into_table_html(
415
+ html,
416
+ element,
417
+ self._require_document_part(),
418
+ )
419
+ table_block = {
420
+ "type": BlockType.TABLE,
421
+ "content": html,
422
+ }
423
+ self.cur_page.append(table_block)
424
+
425
+ def _normalize_table_colspans(self, html: str) -> str:
426
+ """
427
+ 修正 HTML 表格中因无线表/少线表导致的 colspan 不一致问题。
428
+
429
+ 在无边框或少边框的 DOCX 表格中,部分行的单元格包含 w:gridSpan 值,
430
+ 该值来自 Word 内部虚拟栅格,并不反映实际视觉列数。mammoth 将这些
431
+ w:gridSpan 值直接转换为 HTML colspan 属性,导致不同行的有效列数
432
+ (所有 colspan 之和)不一致,产生行列对不齐的问题。
433
+
434
+ 本方法检测此类不一致,并将有效列数过多的行的 colspan 缩减至
435
+ 最常见的目标列数,从而恢复表格的正确结构。
436
+
437
+ 算法:
438
+ 1. 计算每行的有效列数(该行所有单元格 colspan 之和)
439
+ 2. 取最常见的列数作为目标列数
440
+ 3. 对有效列数超过目标值的行,从第一个 colspan > 1 的单元格开始缩减
441
+
442
+ Args:
443
+ html: 包含表格的 HTML 字符串
444
+
445
+ Returns:
446
+ str: 修正后的 HTML 字符串
447
+ """
448
+ try:
449
+ from collections import Counter
450
+
451
+ from bs4 import BeautifulSoup
452
+
453
+ soup = BeautifulSoup(html, "html.parser")
454
+ tables = soup.find_all("table")
455
+ modified = False
456
+
457
+ for table in tables:
458
+ rows = table.find_all("tr")
459
+ if not rows:
460
+ continue
461
+
462
+ # 若表格中存在 rowspan > 1 的单元格,各行的显式 colspan 之和
463
+ # 无法反映真实网格宽度(被 rowspan 占据的列不出现在后续行的 td
464
+ # 列表中),此时算法的假设不成立,跳过该表格以避免误修改合法的
465
+ # colspan。
466
+ all_cells = table.find_all(["td", "th"])
467
+ if any(int(c.get("rowspan", 1)) > 1 for c in all_cells):
468
+ continue
469
+
470
+ # 计算每行的有效列数(所有单元格的 colspan 之和)
471
+ row_col_counts = []
472
+ for row in rows:
473
+ cells = row.find_all(["td", "th"])
474
+ total = sum(int(c.get("colspan", 1)) for c in cells)
475
+ row_col_counts.append(total)
476
+
477
+ if not row_col_counts:
478
+ continue
479
+
480
+ # 找到目标列数(出现最多的列数)
481
+ count_freq = Counter(row_col_counts)
482
+ if len(count_freq) == 1:
483
+ continue # 各行列数已一致,无需修正
484
+
485
+ target = count_freq.most_common(1)[0][0]
486
+
487
+ # 修正有效列数超过目标值的行:缩减 colspan > 1 的单元格
488
+ for row, col_count in zip(rows, row_col_counts):
489
+ if col_count <= target:
490
+ continue
491
+
492
+ excess = col_count - target
493
+ cells = row.find_all(["td", "th"])
494
+
495
+ for cell in cells:
496
+ if excess <= 0:
497
+ break
498
+ span = int(cell.get("colspan", 1))
499
+ if span > 1:
500
+ reduce_by = min(span - 1, excess)
501
+ new_span = span - reduce_by
502
+ if new_span == 1:
503
+ if "colspan" in cell.attrs:
504
+ del cell["colspan"]
505
+ else:
506
+ cell["colspan"] = str(new_span)
507
+ excess -= reduce_by
508
+ modified = True
509
+
510
+ if modified:
511
+ return str(soup)
512
+ return html
513
+ except Exception as e:
514
+ logger.debug(f"Failed to normalize table colspans: {e}")
515
+ return html
@@ -0,0 +1,3 @@
1
+ """Office 公式格式与嵌入载体的内部解析实现。"""
2
+
3
+ __all__: list[str] = []