docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,526 @@
1
+ """DOCX 公式与图片资源处理;共享当前 Converter 的单文档状态。"""
2
+
3
+ import hashlib
4
+ import re
5
+ from typing import Any
6
+ from docx.oxml.xmlchemy import BaseOxmlElement
7
+ from docx.text.paragraph import Paragraph
8
+ from loguru import logger
9
+ from ..ooxml_chart import extract_chart_html_from_ooxml
10
+ from ..image import serialize_office_image
11
+ from ..equation.ooxml import is_mathtype_equation_prog_id
12
+ from ..equation.omml import oMath2Latex
13
+ from .....schema import BlockType
14
+
15
+ from .context import _DocxConstants
16
+
17
+
18
+ class _DocxResources:
19
+ """集中维护公式与图片资源,不自行创建文档或持有跨文档缓存。"""
20
+
21
+ def _decode_docx_ole_equation(
22
+ self,
23
+ ole_element: Any,
24
+ part: Any,
25
+ ) -> str | None:
26
+ """从当前 DOCX part 的内部 OLE relationship 解码公式对象。"""
27
+
28
+ prog_id = ole_element.get("ProgID") or ole_element.get("ProgId")
29
+ if not is_mathtype_equation_prog_id(prog_id):
30
+ return None
31
+ object_type = (ole_element.get("Type") or "Embed").strip().casefold()
32
+ if object_type != "embed":
33
+ return None
34
+ draw_aspect = (ole_element.get("DrawAspect") or "Content").strip().casefold()
35
+ if draw_aspect == "icon":
36
+ return None
37
+
38
+ relationship_id = ole_element.get("{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id")
39
+ relationships = getattr(part, "rels", None)
40
+ relationship = relationships.get(relationship_id) if relationships is not None else None
41
+ if relationship is None or getattr(relationship, "is_external", False):
42
+ return None
43
+ reltype = str(getattr(relationship, "reltype", ""))
44
+ if not reltype.rstrip("/").casefold().endswith("/oleobject"):
45
+ return None
46
+ try:
47
+ blob = relationship.target_part.blob
48
+ except (AttributeError, KeyError, ValueError):
49
+ blob = None
50
+ latex = self._ooxml_equation_decoder.decode(
51
+ blob,
52
+ prog_id=prog_id,
53
+ )
54
+ if latex is not None:
55
+ return latex
56
+
57
+ warning_key = (self._docx_part_key(part), str(relationship_id))
58
+ if warning_key not in self._mtef_warned_relations:
59
+ self._mtef_warned_relations.add(warning_key)
60
+ logger.warning(
61
+ "DOCX_MTEF_FALLBACK: part={!r}, relationship={!r} has an invalid or unsupported equation OLE object",
62
+ warning_key[0],
63
+ relationship_id,
64
+ )
65
+ return None
66
+
67
+ def _decode_docx_equationxml(
68
+ self,
69
+ shape_element: Any,
70
+ part: Any,
71
+ ) -> str | None:
72
+ """解码当前 VML shape 的 ``equationxml`` 并对失败告警去重。"""
73
+
74
+ vml_shape_tag = f"{{{_DocxConstants._BLIP_NAMESPACES['v']}}}shape"
75
+ if getattr(shape_element, "tag", None) != vml_shape_tag:
76
+ return None
77
+ equation_xml = shape_element.get("equationxml")
78
+ if equation_xml is None:
79
+ return None
80
+
81
+ latex = self._equationxml_decoder.decode(equation_xml)
82
+ if latex is not None:
83
+ return latex
84
+
85
+ digest = hashlib.sha256(equation_xml.encode("utf-8", errors="replace")).hexdigest()[:16]
86
+ warning_key = (
87
+ self._docx_part_key(part),
88
+ str(shape_element.get("id") or ""),
89
+ digest,
90
+ )
91
+ if warning_key not in self._equationxml_warned_shapes:
92
+ self._equationxml_warned_shapes.add(warning_key)
93
+ logger.warning(
94
+ "DOCX_EQUATIONXML_FALLBACK: part={!r}, shape_id={!r}, payload_sha256={!r} is malformed or unsupported",
95
+ warning_key[0],
96
+ warning_key[1],
97
+ warning_key[2],
98
+ )
99
+ return None
100
+
101
+ @staticmethod
102
+ def _docx_image_relationship_id(image: Any) -> str | None:
103
+ """读取 DrawingML/VML 图片元素的内部 relationship id。"""
104
+
105
+ relationship_id = image.get("{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed")
106
+ if not relationship_id:
107
+ relationship_id = image.get("{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id")
108
+ return str(relationship_id) if relationship_id else None
109
+
110
+ @classmethod
111
+ def _docx_image_part(cls, image: Any, part: Any) -> Any | None:
112
+ """通过当前 part 的内部关系解析图片 part。"""
113
+
114
+ relationship_id = cls._docx_image_relationship_id(image)
115
+ relationships = getattr(part, "rels", None)
116
+ if not relationship_id or relationships is None:
117
+ return None
118
+ relationship = relationships.get(relationship_id)
119
+ if relationship is None or getattr(relationship, "is_external", False):
120
+ return None
121
+ try:
122
+ return relationship.target_part
123
+ except (AttributeError, KeyError, ValueError):
124
+ return None
125
+
126
+ def _decode_docx_image_equation(
127
+ self,
128
+ image: Any,
129
+ part: Any,
130
+ ) -> str | None:
131
+ """从当前图片 part 的 WMF/GIF comment 解码 MTEF。"""
132
+
133
+ image_part = self._docx_image_part(image, part)
134
+ if image_part is None:
135
+ return None
136
+ try:
137
+ blob = image_part.blob
138
+ except (AttributeError, KeyError, ValueError):
139
+ return None
140
+ return self._image_equation_decoder.decode(
141
+ blob,
142
+ part_name=getattr(image_part, "partname", None),
143
+ content_type=getattr(image_part, "content_type", None),
144
+ )
145
+
146
+ @staticmethod
147
+ def _select_docx_compatibility_tokens(
148
+ token_groups: list[list[tuple[str, str]]],
149
+ ) -> list[tuple[str, str]]:
150
+ """按 OMML、Equation XML、OLE MTEF、图片 MTEF 选择兼容分支。"""
151
+
152
+ for token_kind in _DocxConstants._FORMULA_SOURCE_PRIORITY:
153
+ for tokens in token_groups:
154
+ if any(kind == token_kind for kind, _value in tokens):
155
+ return tokens
156
+ return token_groups[0] if token_groups else []
157
+
158
+ def _docx_formula_tokens(
159
+ self,
160
+ element: Any,
161
+ part: Any,
162
+ ) -> list[tuple[str, str]]:
163
+ """按文档顺序提取文本、OMML、Equation XML 和 MTEF。"""
164
+
165
+ tag_name = self._local_name(element)
166
+ if tag_name is None:
167
+ return []
168
+ tag = str(getattr(element, "tag", ""))
169
+ word_namespace = _DocxConstants._BLIP_NAMESPACES["w"]
170
+
171
+ if tag_name == "AlternateContent":
172
+ branch_tokens = [
173
+ self._docx_formula_tokens(child, part) for child in element if self._local_name(child) in {"Choice", "Fallback"}
174
+ ]
175
+ return self._select_docx_compatibility_tokens(branch_tokens)
176
+
177
+ if tag_name == "object" and tag == f"{{{_DocxConstants._BLIP_NAMESPACES['w']}}}object":
178
+ child_tokens = [self._docx_formula_tokens(child, part) for child in element]
179
+ return self._select_docx_compatibility_tokens(child_tokens)
180
+
181
+ if tag_name == "txbxContent":
182
+ # 外层段落会单独遍历文本框内容,避免在此重复提取公式和文字。
183
+ return []
184
+
185
+ if tag_name == "oMath" and "officeDocument/2006/math" in tag:
186
+ try:
187
+ latex = str(oMath2Latex(element)).strip()
188
+ except Exception as exc:
189
+ logger.debug(f"Failed to convert DOCX OMML equation to LaTeX: {exc}")
190
+ return []
191
+ return [("omml", latex)] if latex else []
192
+
193
+ if tag_name == "shape" and tag == f"{{{_DocxConstants._BLIP_NAMESPACES['v']}}}shape":
194
+ latex = self._decode_docx_equationxml(element, part)
195
+ if latex is not None:
196
+ return [("equationxml", latex)]
197
+
198
+ if tag_name == "OLEObject":
199
+ latex = self._decode_docx_ole_equation(element, part)
200
+ return [("mtef", latex)] if latex else []
201
+
202
+ if (tag_name == "blip" and tag == "{http://schemas.openxmlformats.org/drawingml/2006/main}blip") or (
203
+ tag_name == "imagedata" and tag == f"{{{_DocxConstants._BLIP_NAMESPACES['v']}}}imagedata"
204
+ ):
205
+ latex = self._decode_docx_image_equation(element, part)
206
+ return [("image_mtef", latex)] if latex else []
207
+
208
+ if tag_name == "t" and "officeDocument/2006/math" not in tag:
209
+ return [("text", element.text)] if isinstance(element.text, str) else []
210
+ if tag in {f"{{{word_namespace}}}tab", f"{{{word_namespace}}}ptab"}:
211
+ return [("text", "\t")]
212
+ if tag == f"{{{word_namespace}}}cr":
213
+ return [("text", "\n")]
214
+ if tag == f"{{{word_namespace}}}br":
215
+ break_type = element.get(f"{{{word_namespace}}}type")
216
+ return [("text", "\n")] if break_type in {None, "textWrapping"} else []
217
+ if tag == f"{{{word_namespace}}}noBreakHyphen":
218
+ return [("text", "-")]
219
+
220
+ tokens: list[tuple[str, str]] = []
221
+ for child in element:
222
+ tokens.extend(self._docx_formula_tokens(child, part))
223
+ return tokens
224
+
225
+ def _picture_is_equation_preview(self, image: Any, part: Any) -> bool:
226
+ """判断图片是否属于同一容器中已恢复的公式预览。"""
227
+
228
+ if self._decode_docx_image_equation(image, part):
229
+ return True
230
+
231
+ for ancestor in image.iterancestors():
232
+ ancestor_name = self._local_name(ancestor)
233
+ if ancestor_name == "shape":
234
+ if self._decode_docx_equationxml(ancestor, part):
235
+ return True
236
+ continue
237
+ if ancestor_name in {"object", "AlternateContent"}:
238
+ tokens = self._docx_formula_tokens(ancestor, part)
239
+ if any(kind in _DocxConstants._FORMULA_TOKEN_KINDS for kind, _value in tokens):
240
+ return True
241
+ continue
242
+ return False
243
+
244
+ def _handle_pictures(
245
+ self,
246
+ picture_refs: Any,
247
+ *,
248
+ part: Any | None = None,
249
+ ) -> None:
250
+ """
251
+ 处理图片。
252
+
253
+ Args:
254
+ picture_refs: 图片引用元素列表
255
+
256
+ Returns:
257
+
258
+ """
259
+
260
+ source_part = part or self._require_document_part()
261
+
262
+ seen_rel_ids: set[str] = set()
263
+ # 遍历所有图片引用元素,支持 DrawingML blip 和 VML imagedata。
264
+ for image in picture_refs:
265
+ if self._picture_is_equation_preview(image, source_part):
266
+ continue
267
+ rel_id = self._docx_image_relationship_id(image)
268
+ if rel_id and rel_id in seen_rel_ids:
269
+ continue
270
+ if rel_id:
271
+ seen_rel_ids.add(rel_id)
272
+ image_part = self._docx_image_part(image, source_part)
273
+ if image_part is None:
274
+ logger.warning("Warning: image cannot be found")
275
+ continue
276
+
277
+ img_base64 = serialize_office_image(
278
+ image_part.blob,
279
+ part_name=getattr(image_part, "partname", None),
280
+ content_type=getattr(image_part, "content_type", None),
281
+ )
282
+ if img_base64 is None:
283
+ continue
284
+
285
+ image_block = {
286
+ "type": BlockType.IMAGE,
287
+ "image_base64": img_base64,
288
+ }
289
+ self.cur_page.append(image_block)
290
+
291
+ def _handle_drawingml(self, elements: list[BaseOxmlElement]):
292
+ """
293
+ 处理 DrawingML 元素,目前先处理 chart 元素。
294
+
295
+ Args:
296
+ elements: 包含 DrawingML 元素的列表
297
+
298
+ Returns:
299
+
300
+ """
301
+ chart_rel_types = {
302
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/chart",
303
+ "http://purl.oclc.org/ooxml/officeDocument/relationships/chart",
304
+ }
305
+ package_rel_types = {
306
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/package",
307
+ "http://purl.oclc.org/ooxml/officeDocument/relationships/package",
308
+ }
309
+ rel_id_attr = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id"
310
+ for element in elements:
311
+ chart = element.find(".//c:chart", namespaces=_DocxConstants._BLIP_NAMESPACES)
312
+ if chart is None:
313
+ continue
314
+
315
+ chart_block = {
316
+ "type": BlockType.CHART,
317
+ "content": "",
318
+ }
319
+ self.cur_page.append(chart_block)
320
+
321
+ rel_id = chart.get(rel_id_attr)
322
+ if not rel_id:
323
+ continue
324
+
325
+ try:
326
+ chart_rel = self.docx_obj.part.rels[rel_id]
327
+ except KeyError:
328
+ continue
329
+
330
+ if chart_rel.reltype not in chart_rel_types:
331
+ continue
332
+
333
+ try:
334
+ chart_part = chart_rel.target_part
335
+ chart_xml = chart_part.blob
336
+ except Exception as e:
337
+ logger.warning(f"Warning: chart XML cannot be loaded: {e}")
338
+ continue
339
+
340
+ workbook_bytes = None
341
+ try:
342
+ for rel in chart_part.rels.values():
343
+ if rel.reltype in package_rel_types:
344
+ workbook_bytes = rel.target_part.blob
345
+ break
346
+ except Exception as e:
347
+ logger.warning(f"Warning: chart workbook cannot be loaded: {e}")
348
+
349
+ try:
350
+ chart_html = extract_chart_html_from_ooxml(chart_xml, workbook_bytes)
351
+ except Exception as e:
352
+ logger.warning(f"Warning: chart HTML cannot be extracted: {e}")
353
+ continue
354
+ if chart_html:
355
+ chart_block["content"] = chart_html
356
+
357
+ def _handle_textbox_content(
358
+ self,
359
+ textbox_elements: list,
360
+ ):
361
+ """
362
+ 处理文本框内容并将其添加到文档结构。
363
+ """
364
+ # 收集并组织段落
365
+ container_paragraphs = self._collect_textbox_paragraphs(textbox_elements)
366
+
367
+ # 处理所有段落
368
+ all_paragraphs = []
369
+
370
+ # 对每个容器内的段落进行排序,然后按容器顺序处理
371
+ for paragraphs in container_paragraphs.values():
372
+ # 按容器内的垂直位置进行排序
373
+ sorted_container_paragraphs = sorted(
374
+ paragraphs,
375
+ key=lambda x: (
376
+ x[1] is None,
377
+ x[1] if x[1] is not None else float("inf"),
378
+ ),
379
+ )
380
+
381
+ # 将排序后的段落添加到待处理列表
382
+ all_paragraphs.extend(sorted_container_paragraphs)
383
+
384
+ # 跟踪已处理段落以避免重复(相同内容和位置)
385
+ processed_paragraphs = set()
386
+
387
+ # 处理所有段落
388
+ for p, position in all_paragraphs:
389
+ # 创建 Paragraph 对象以获取文本内容
390
+ paragraph = Paragraph(p, self.docx_obj)
391
+ text_content = self._get_paragraph_text(paragraph)
392
+
393
+ # 基于内容和位置创建唯一标识
394
+ paragraph_id = (text_content, position)
395
+
396
+ # 如果该段落(相同内容和位置)已处理,则跳过
397
+ if paragraph_id in processed_paragraphs:
398
+ logger.debug(f"Skipping duplicate paragraph: content='{text_content[:50]}...', position={position}")
399
+ continue
400
+
401
+ # 将该段落标记为已处理
402
+ processed_paragraphs.add(paragraph_id)
403
+
404
+ self._handle_text_elements(p)
405
+ return
406
+
407
+ def _collect_textbox_paragraphs(self, textbox_elements):
408
+ """
409
+ 从文本框元素中收集并组织段落。
410
+ """
411
+ processed_paragraphs = []
412
+ container_paragraphs = {}
413
+
414
+ for element in textbox_elements:
415
+ element_id = id(element)
416
+ # 如果已处理相同元素,则跳过
417
+ if element_id in processed_paragraphs:
418
+ continue
419
+
420
+ tag_name = self._local_name(element)
421
+ if tag_name is None:
422
+ continue
423
+ processed_paragraphs.append(element_id)
424
+
425
+ # 处理直接找到的段落(VML 文本框)
426
+ if tag_name == "p":
427
+ # 查找包含该段落的文本框或形状元素
428
+ container_id = None
429
+ for ancestor in element.iterancestors():
430
+ if any(ns in ancestor.tag for ns in ["textbox", "shape", "txbx"]):
431
+ container_id = id(ancestor)
432
+ break
433
+
434
+ if container_id not in container_paragraphs:
435
+ container_paragraphs[container_id] = []
436
+ container_paragraphs[container_id].append((element, self._get_paragraph_position(element)))
437
+
438
+ # 处理 txbxContent 元素(Word DrawingML 文本框)
439
+ elif tag_name == "txbxContent":
440
+ paragraphs = element.findall(".//w:p", namespaces=element.nsmap)
441
+ container_id = id(element)
442
+ if container_id not in container_paragraphs:
443
+ container_paragraphs[container_id] = []
444
+
445
+ for p in paragraphs:
446
+ p_id = id(p)
447
+ if p_id not in processed_paragraphs:
448
+ processed_paragraphs.append(p_id)
449
+ container_paragraphs[container_id].append((p, self._get_paragraph_position(p)))
450
+ else:
451
+ # 尝试从未知元素中提取任何段落
452
+ paragraphs = element.findall(".//w:p", namespaces=element.nsmap)
453
+ container_id = id(element)
454
+ if container_id not in container_paragraphs:
455
+ container_paragraphs[container_id] = []
456
+
457
+ for p in paragraphs:
458
+ p_id = id(p)
459
+ if p_id not in processed_paragraphs:
460
+ processed_paragraphs.append(p_id)
461
+ container_paragraphs[container_id].append((p, self._get_paragraph_position(p)))
462
+
463
+ return container_paragraphs
464
+
465
+ def _get_paragraph_position(self, paragraph_element):
466
+ """
467
+ 从段落元素提取垂直位置信息。
468
+ """
469
+ # 先尝试直接从包含顺序相关属性的 w:p 元素获取索引
470
+ if hasattr(paragraph_element, "getparent") and paragraph_element.getparent() is not None:
471
+ parent = paragraph_element.getparent()
472
+ # 获取所有段落兄弟节点
473
+ paragraphs = [p for p in parent.getchildren() if self._local_name(p) == "p"]
474
+ # 查找当前段落在其兄弟节点中的索引
475
+ try:
476
+ paragraph_index = paragraphs.index(paragraph_element)
477
+ return paragraph_index # 使用索引作为位置以保证一致的排序
478
+ except ValueError:
479
+ pass
480
+
481
+ # 在元素及其祖先中查找位置提示属性
482
+ for elem in (*[paragraph_element], *paragraph_element.iterancestors()):
483
+ # 检查直接的位置信息属性
484
+ for attr_name in ["y", "top", "positionY", "y-position", "position"]:
485
+ value = elem.get(attr_name)
486
+ if value:
487
+ try:
488
+ # 移除任何非数字字符(如 'pt', 'px' 等)
489
+ clean_value = re.sub(r"[^0-9.]", "", value)
490
+ if clean_value:
491
+ return float(clean_value)
492
+ except (ValueError, TypeError):
493
+ pass
494
+
495
+ # 检查 transform 属性中的位移信息
496
+ transform = elem.get("transform")
497
+ if transform:
498
+ # 从 transform 矩阵中提取 translate 的第二个参数
499
+ match = re.search(r"translate\([^,]+,\s*([0-9.]+)", transform)
500
+ if match:
501
+ try:
502
+ return float(match.group(1))
503
+ except ValueError:
504
+ pass
505
+
506
+ # 检查 Word 格式中的锚点或相对位置指示器
507
+ # 'dist' 类属性可以表示相对位置
508
+ for attr_name in ["distT", "distB", "anchor", "relativeFrom"]:
509
+ if elem.get(attr_name) is not None:
510
+ return elem.sourceline # 使用 XML 源行号作为回退
511
+
512
+ # 针对 VML 形状,查找特定属性
513
+ for ns_uri in paragraph_element.nsmap.values():
514
+ if "vml" in ns_uri:
515
+ # 尝试从 style 属性提取 top 值
516
+ style = paragraph_element.get("style")
517
+ if style:
518
+ match = re.search(r"top:([0-9.]+)pt", style)
519
+ if match:
520
+ try:
521
+ return float(match.group(1))
522
+ except ValueError:
523
+ pass
524
+
525
+ # 如果没有更好的位置指示,则使用 XML 源行号作为顺序的代理
526
+ return paragraph_element.sourceline if hasattr(paragraph_element, "sourceline") else None