docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,573 @@
1
+ """把 Word 97–2003 语义模型转换为 DocVortex 分页 model-list。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from html import escape
6
+ import re
7
+ from typing import Any, BinaryIO, Iterable
8
+
9
+ from ..errors import LegacyOfficeEncryptedError
10
+ from ..image import serialize_office_image
11
+ from ..legacy.ole import BoundedOleReader
12
+ from ..streams import read_stream_bytes_from_start
13
+ from .....schema import RAW_CAPTION, BlockType
14
+ from .....content.spans import append_equation_span, extend_inline_spans, inline_span_plain_text, strip_span_dicts, text_spans
15
+ from ..rich_text import OfficeRichTextSegment, build_rich_text_from_segments, build_rich_text_html_from_segments
16
+
17
+ from .fib import parse_fib
18
+ from ..equation.mtef import read_object_pool_equations
19
+ from ..xls.embedded_chart import extract_embedded_chart_html
20
+ from .models import (
21
+ DocCharStyle,
22
+ DocChartPayload,
23
+ DocDocument,
24
+ DocElement,
25
+ DocImage,
26
+ DocImagePayload,
27
+ DocParagraph,
28
+ DocSection,
29
+ DocTable,
30
+ DocTableCell,
31
+ DocTextRun,
32
+ DocVisualPayload,
33
+ )
34
+ from .parser import parse_doc_document
35
+
36
+ _OBJECT_POOL_CHART_RE = re.compile(
37
+ r"^ObjectPool/_([0-9]+)/(?:Workbook|Book)$",
38
+ re.IGNORECASE,
39
+ )
40
+
41
+
42
+ def _read_object_pool_charts(ole: BoundedOleReader) -> dict[int, str]:
43
+ """读取 DOC ObjectPool 中带 Workbook/Book 的可编辑 chart。"""
44
+
45
+ charts: dict[int, str] = {}
46
+ for stream_name in ole.stream_names(prefix="ObjectPool/"):
47
+ match = _OBJECT_POOL_CHART_RE.match(stream_name)
48
+ if match is None:
49
+ continue
50
+ storage_id = int(match.group(1))
51
+ if storage_id in charts:
52
+ continue
53
+ content = extract_embedded_chart_html(ole.read_stream(stream_name))
54
+ if content:
55
+ charts[storage_id] = content
56
+ return charts
57
+
58
+
59
+ class DocConverter:
60
+ """将 Word 97–2003 OLE 二进制流转换为逐 section raw blocks。"""
61
+
62
+ def __init__(self) -> None:
63
+ """初始化空输出。"""
64
+
65
+ self.pages: list[list[dict[str, Any]]] = []
66
+
67
+ def convert(self, file_binary: BinaryIO) -> None:
68
+ """读取输入 OLE streams,解析 DOC 并生成 model-list。"""
69
+
70
+ file_bytes = read_stream_bytes_from_start(file_binary)
71
+ with BoundedOleReader(file_bytes) as ole:
72
+ if ole.has_stream("EncryptedPackage") or ole.has_stream("EncryptionInfo"):
73
+ raise LegacyOfficeEncryptedError("encrypted OOXML package is not a binary DOC")
74
+ word_document = ole.read_stream("WordDocument")
75
+ fib = parse_fib(word_document)
76
+ if fib.base.encrypted or fib.base.obfuscated:
77
+ raise LegacyOfficeEncryptedError("password-protected DOC is unsupported")
78
+ preferred = "1Table" if fib.base.uses_1table else "0Table"
79
+ alternate = "0Table" if preferred == "1Table" else "1Table"
80
+ table_stream = ole.read_stream(preferred, required=False)
81
+ if not table_stream:
82
+ table_stream = ole.read_stream(alternate, required=bool(fib.base.complex))
83
+ data_stream = ole.read_stream("Data", required=False)
84
+ native_equations = read_object_pool_equations(ole)
85
+ native_charts = _read_object_pool_charts(ole)
86
+ document = parse_doc_document(
87
+ word_document,
88
+ table_stream,
89
+ data_stream,
90
+ fib,
91
+ native_equations=native_equations,
92
+ native_charts=native_charts,
93
+ )
94
+ self.pages = self._document_pages(document)
95
+
96
+ @staticmethod
97
+ def _style_names(style: DocCharStyle) -> list[str]:
98
+ """把 DOC 字符属性转换为 DocVortex 富文本样式名。"""
99
+
100
+ names: list[str] = []
101
+ if style.bold:
102
+ names.append("bold")
103
+ if style.italic:
104
+ names.append("italic")
105
+ if style.underline:
106
+ names.append("underline")
107
+ if style.emphasis:
108
+ names.append("emphasis")
109
+ if style.strike:
110
+ names.append("strikethrough")
111
+ if style.superscript:
112
+ names.append("superscript")
113
+ elif style.subscript:
114
+ names.append("subscript")
115
+ return names
116
+
117
+ @classmethod
118
+ def _rich_text(cls, runs: Iterable[DocTextRun], *, trim: bool = True) -> list[dict[str, Any]]:
119
+ """把 DOC runs 直接转换为结构化 Span。"""
120
+
121
+ spans: list[dict[str, Any]] = []
122
+ segments: list[OfficeRichTextSegment] = []
123
+
124
+ def flush_segments() -> None:
125
+ """输出公式边界前累计的普通富文本。"""
126
+
127
+ if not segments:
128
+ return
129
+ extend_inline_spans(spans, build_rich_text_from_segments(segments, trim_plain_edges=trim and not spans))
130
+ segments.clear()
131
+
132
+ for run in runs:
133
+ if not run.text:
134
+ continue
135
+ if run.formula:
136
+ flush_segments()
137
+ latex = run.text.replace("<", r"\lt ").replace(">", r"\gt ")
138
+ append_equation_span(spans, latex)
139
+ continue
140
+ segments.append(
141
+ OfficeRichTextSegment(
142
+ text=run.text,
143
+ style=cls._style_names(run.style),
144
+ hyperlink=run.hyperlink,
145
+ )
146
+ )
147
+ flush_segments()
148
+ return strip_span_dicts(spans) if trim else spans
149
+
150
+ @classmethod
151
+ def _cell_rich_text_html(cls, runs: Iterable[DocTextRun], *, trim: bool = True) -> str:
152
+ """把 DOC 表格单元格 runs 序列化为安全 HTML。"""
153
+ parts: list[str] = []
154
+ segments: list[OfficeRichTextSegment] = []
155
+
156
+ def flush_segments() -> None:
157
+ """在公式边界输出累计单元格文字。"""
158
+ if not segments:
159
+ return
160
+ parts.append(build_rich_text_html_from_segments(segments, trim_plain_edges=trim and not parts))
161
+ segments.clear()
162
+
163
+ for run in runs:
164
+ if not run.text:
165
+ continue
166
+ if run.formula:
167
+ flush_segments()
168
+ latex = run.text.replace("<", r"\lt ").replace(">", r"\gt ")
169
+ parts.append(f"<eq>{escape(latex, quote=False)}</eq>")
170
+ continue
171
+ segments.append(
172
+ OfficeRichTextSegment(
173
+ text=run.text,
174
+ style=cls._style_names(run.style),
175
+ hyperlink=run.hyperlink,
176
+ )
177
+ )
178
+ flush_segments()
179
+ return "".join(parts)
180
+
181
+ @staticmethod
182
+ def _plain_text(paragraph: DocParagraph) -> str:
183
+ """返回段落不含内部标记的可见文本。"""
184
+
185
+ return "".join(run.text for run in paragraph.runs)
186
+
187
+ @staticmethod
188
+ def _serialize_image(payload: DocImagePayload) -> str | None:
189
+ """复用 Office 图片序列化与矢量占位策略。"""
190
+
191
+ return serialize_office_image(
192
+ payload.data,
193
+ part_name=f"image.{payload.extension}",
194
+ content_type=payload.content_type,
195
+ render_size_emu=payload.render_size_emu,
196
+ )
197
+
198
+ @classmethod
199
+ def _image_block(cls, payload: DocVisualPayload) -> dict[str, Any] | None:
200
+ """把图片、公式或 chart 载荷转换为对应 raw block。"""
201
+
202
+ if isinstance(payload, DocChartPayload):
203
+ block: dict[str, Any] = {
204
+ "type": BlockType.CHART,
205
+ "content": payload.content,
206
+ }
207
+ if payload.preview is not None:
208
+ image_base64 = cls._serialize_image(payload.preview)
209
+ if image_base64:
210
+ block["image_base64"] = image_base64
211
+ return block
212
+
213
+ if payload.equation_latex:
214
+ return {
215
+ "type": BlockType.EQUATION,
216
+ "content": payload.equation_latex,
217
+ }
218
+
219
+ image_base64 = cls._serialize_image(payload)
220
+ if image_base64 is None:
221
+ return None
222
+ return {"type": BlockType.IMAGE, "image_base64": image_base64}
223
+
224
+ @classmethod
225
+ def _toc_runs(cls, paragraph: DocParagraph) -> list[DocTextRun]:
226
+ """从 TOC 段落末尾移除仅用于排版的 tab/page number。"""
227
+
228
+ runs = list(paragraph.runs)
229
+ while runs and re.fullmatch(r"[\s\t]*\d+[\s\t]*", runs[-1].text):
230
+ runs.pop()
231
+ if runs:
232
+ cleaned = re.sub(r"\t+\s*\d+\s*$", "", runs[-1].text)
233
+ if cleaned != runs[-1].text:
234
+ last = runs[-1]
235
+ runs[-1] = DocTextRun(cleaned, last.style, last.hyperlink)
236
+ return runs
237
+
238
+ @classmethod
239
+ def _toc_anchor(cls, paragraph: DocParagraph) -> str | None:
240
+ """读取 TOC 超链接指向的内部书签。"""
241
+
242
+ for run in paragraph.runs:
243
+ if run.hyperlink and run.hyperlink.startswith("#") and len(run.hyperlink) > 1:
244
+ return run.hyperlink[1:]
245
+ return paragraph.anchor
246
+
247
+ @classmethod
248
+ def _append_index_item(
249
+ cls,
250
+ page: list[dict[str, Any]],
251
+ stack: list[dict[str, Any]],
252
+ paragraph: DocParagraph,
253
+ ) -> None:
254
+ """把一个 TOC 段落追加到对应层级的 index 树。"""
255
+
256
+ content = cls._rich_text(cls._toc_runs(paragraph))
257
+ for payload in paragraph.images:
258
+ if isinstance(payload, DocImagePayload) and payload.equation_latex:
259
+ append_equation_span(content, payload.equation_latex)
260
+ if not content:
261
+ return
262
+ level = min(max(paragraph.toc_level or 0, 0), 8)
263
+ while len(stack) > level + 1:
264
+ stack.pop()
265
+ while len(stack) < level + 1:
266
+ index_block: dict[str, Any] = {
267
+ "type": BlockType.INDEX,
268
+ "ilevel": len(stack),
269
+ "content": [],
270
+ }
271
+ if stack:
272
+ stack[-1]["content"].append(index_block)
273
+ else:
274
+ page.append(index_block)
275
+ stack.append(index_block)
276
+ leaf: dict[str, Any] = {"type": BlockType.TEXT, "content": content}
277
+ anchor = cls._toc_anchor(paragraph)
278
+ if anchor:
279
+ leaf["anchor"] = anchor
280
+ stack[level]["content"].append(leaf)
281
+
282
+ @classmethod
283
+ def _append_list_item(
284
+ cls,
285
+ page: list[dict[str, Any]],
286
+ stack: list[dict[str, Any]],
287
+ identity: int | None,
288
+ paragraph: DocParagraph,
289
+ ) -> int:
290
+ """把一个 DOC 列表段落追加到嵌套 raw list 树。"""
291
+
292
+ info = paragraph.list_info
293
+ if info is None:
294
+ return identity or -1
295
+ content = cls._rich_text(paragraph.runs)
296
+ for payload in paragraph.images:
297
+ if isinstance(payload, DocImagePayload) and payload.equation_latex:
298
+ append_equation_span(content, payload.equation_latex)
299
+ if not content and not paragraph.images:
300
+ return identity or info.identity
301
+ if identity != info.identity:
302
+ stack.clear()
303
+ identity = info.identity
304
+ level = min(max(info.level, 0), 8)
305
+ while len(stack) > level + 1:
306
+ stack.pop()
307
+ while len(stack) < level + 1:
308
+ list_block: dict[str, Any] = {
309
+ "type": BlockType.LIST,
310
+ "attribute": "ordered" if info.ordered else "unordered",
311
+ "ilevel": len(stack),
312
+ "content": [],
313
+ }
314
+ if info.ordered:
315
+ list_block["start"] = info.start
316
+ if stack:
317
+ stack[-1]["content"].append(list_block)
318
+ else:
319
+ page.append(list_block)
320
+ stack.append(list_block)
321
+ current = stack[level]
322
+ expected = "ordered" if info.ordered else "unordered"
323
+ if current.get("attribute") != expected:
324
+ del stack[level:]
325
+ return cls._append_list_item(page, stack, None, paragraph)
326
+ if content:
327
+ leaf: dict[str, Any] = {"type": BlockType.TEXT, "content": content}
328
+ if info.label:
329
+ leaf["list_label"] = info.label
330
+ current["content"].append(leaf)
331
+ return info.identity
332
+
333
+ @classmethod
334
+ def _paragraph_blocks(cls, paragraph: DocParagraph) -> list[dict[str, Any]]:
335
+ """把非目录、非普通列表段落投影为 raw blocks。"""
336
+
337
+ content = cls._rich_text(paragraph.runs)
338
+ blocks: list[dict[str, Any]] = []
339
+ formula_runs = [run for run in paragraph.runs if run.formula and run.text]
340
+ ordinary_text = "".join(run.text for run in paragraph.runs if not run.formula).strip()
341
+ list_info = paragraph.list_info
342
+ exact_label = list_info.label if list_info is not None and list_info.ordered else None
343
+ if exact_label and (paragraph.is_title or paragraph.heading_level is not None):
344
+ content = [*text_spans(f"{exact_label} "), *content]
345
+ if (
346
+ formula_runs
347
+ and not ordinary_text
348
+ and not (paragraph.is_title or paragraph.heading_level is not None or paragraph.is_caption or paragraph.is_code)
349
+ ):
350
+ blocks.extend({"type": BlockType.EQUATION, "content": run.text} for run in formula_runs)
351
+ elif content:
352
+ if paragraph.is_title:
353
+ block: dict[str, Any] = {
354
+ "type": BlockType.DOC_TITLE,
355
+ "level": 1,
356
+ "content": content,
357
+ }
358
+ elif paragraph.heading_level is not None:
359
+ block = {
360
+ "type": BlockType.PARAGRAPH_TITLE,
361
+ "level": min(max(paragraph.heading_level + 1, 2), 6),
362
+ "is_numbered_style": False,
363
+ "content": content,
364
+ }
365
+ elif paragraph.is_caption:
366
+ block = {"type": RAW_CAPTION, "content": content}
367
+ elif paragraph.is_code:
368
+ block = {"type": BlockType.CODE, "content": cls._plain_text(paragraph)}
369
+ else:
370
+ block = {"type": BlockType.TEXT, "content": content}
371
+ if paragraph.anchor and block["type"] in {BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE}:
372
+ block["anchor"] = paragraph.anchor
373
+ blocks.append(block)
374
+ for payload in paragraph.images:
375
+ image = cls._image_block(payload)
376
+ if image is not None:
377
+ blocks.append(image)
378
+ return blocks
379
+
380
+ @classmethod
381
+ def _cell_list_html(cls, paragraphs: list[DocParagraph]) -> str:
382
+ """把连续表格单元格列表段落序列化为嵌套 HTML list。"""
383
+
384
+ result: list[str] = []
385
+ stack: list[str] = []
386
+ for paragraph in paragraphs:
387
+ info = paragraph.list_info
388
+ if info is None:
389
+ while stack:
390
+ result.append(f"</{stack.pop()}>")
391
+ result.append(cls._cell_paragraph_html(paragraph))
392
+ continue
393
+ level = min(max(info.level, 0), 8)
394
+ tag = "ol" if info.ordered else "ul"
395
+ while len(stack) > level + 1:
396
+ result.append(f"</{stack.pop()}>")
397
+ while len(stack) < level + 1:
398
+ result.append(f"<{tag}>")
399
+ stack.append(tag)
400
+ if stack[-1] != tag:
401
+ result.append(f"</{stack.pop()}><{tag}>")
402
+ stack.append(tag)
403
+ label = f"{escape(info.label)} " if info.label and info.ordered else ""
404
+ result.append(
405
+ f"<li>{label}{cls._cell_rich_text_html(paragraph.runs)}{cls._cell_images_html(paragraph.images)}</li>"
406
+ )
407
+ while stack:
408
+ result.append(f"</{stack.pop()}>")
409
+ return "".join(result)
410
+
411
+ @classmethod
412
+ def _cell_paragraph_html(cls, paragraph: DocParagraph) -> str:
413
+ """序列化一个表格单元格段落及其内联图片。"""
414
+
415
+ parts: list[str] = []
416
+ content = cls._cell_rich_text_html(paragraph.runs, trim=False)
417
+ if content:
418
+ parts.append(f"<p>{content}</p>")
419
+ parts.append(cls._cell_images_html(paragraph.images))
420
+ return "".join(parts)
421
+
422
+ @classmethod
423
+ def _cell_images_html(
424
+ cls,
425
+ payloads: list[DocVisualPayload],
426
+ ) -> str:
427
+ """把表格段落中的图片或 comment 公式序列化为内联 HTML。"""
428
+
429
+ parts: list[str] = []
430
+ for payload in payloads:
431
+ if isinstance(payload, DocChartPayload):
432
+ if payload.preview is not None:
433
+ image = cls._serialize_image(payload.preview)
434
+ if image:
435
+ parts.append(f'<img src="{escape(image, quote=True)}"/>')
436
+ parts.append(payload.content)
437
+ continue
438
+ if payload.equation_latex:
439
+ parts.append(f"<eq>{escape(payload.equation_latex, quote=False)}</eq>")
440
+ continue
441
+ image = cls._serialize_image(payload)
442
+ if image:
443
+ parts.append(f'<img src="{escape(image, quote=True)}"/>')
444
+ return "".join(parts)
445
+
446
+ @classmethod
447
+ def _cell_html(cls, cell: DocTableCell) -> str:
448
+ """递归序列化单元格中的段落和嵌套表格。"""
449
+
450
+ parts: list[str] = []
451
+ paragraph_buffer: list[DocParagraph] = []
452
+
453
+ def flush() -> None:
454
+ """输出当前连续段落缓冲。"""
455
+
456
+ nonlocal paragraph_buffer
457
+ if paragraph_buffer:
458
+ parts.append(cls._cell_list_html(paragraph_buffer))
459
+ paragraph_buffer = []
460
+
461
+ for block in cell.blocks:
462
+ if isinstance(block, DocParagraph):
463
+ paragraph_buffer.append(block)
464
+ elif isinstance(block, DocTable):
465
+ flush()
466
+ parts.append(cls._table_html(block))
467
+ elif isinstance(block, DocImage):
468
+ flush()
469
+ if block.payload.equation_latex:
470
+ parts.append(f"<eq>{escape(block.payload.equation_latex, quote=False)}</eq>")
471
+ continue
472
+ image = cls._serialize_image(block.payload)
473
+ if image:
474
+ parts.append(f'<img src="{escape(image, quote=True)}"/>')
475
+ flush()
476
+ return "".join(parts)
477
+
478
+ @classmethod
479
+ def _table_html(cls, table: DocTable) -> str:
480
+ """序列化带 rowspan/colspan 及嵌套内容的 Word 表格。"""
481
+
482
+ rows: list[str] = []
483
+ for row in table.rows:
484
+ tag = "th" if row.header else "td"
485
+ cells: list[str] = []
486
+ for cell in row.cells:
487
+ attributes: list[str] = []
488
+ if cell.row_span > 1:
489
+ attributes.append(f'rowspan="{cell.row_span}"')
490
+ if cell.col_span > 1:
491
+ attributes.append(f'colspan="{cell.col_span}"')
492
+ suffix = f" {' '.join(attributes)}" if attributes else ""
493
+ cells.append(f"<{tag}{suffix}>{cls._cell_html(cell)}</{tag}>")
494
+ rows.append(f"<tr>{''.join(cells)}</tr>")
495
+ return f"<table>{''.join(rows)}</table>"
496
+
497
+ @classmethod
498
+ def _element_blocks(cls, element: DocElement) -> list[dict[str, Any]]:
499
+ """把正文语义元素转换为 raw block。"""
500
+
501
+ if isinstance(element, DocImage):
502
+ image = cls._image_block(element.payload)
503
+ return [image] if image is not None else []
504
+ if isinstance(element, DocTable):
505
+ return [{"type": BlockType.TABLE, "content": cls._table_html(element)}]
506
+ return cls._paragraph_blocks(element)
507
+
508
+ @classmethod
509
+ def _auxiliary_contents(cls, paragraphs: list[DocParagraph]) -> list[list[dict[str, Any]]]:
510
+ """去重页眉页脚段落并过滤纯页码。"""
511
+
512
+ result: list[list[dict[str, Any]]] = []
513
+ seen: set[str] = set()
514
+ for paragraph in paragraphs:
515
+ content = cls._rich_text(paragraph.runs)
516
+ plain = cls._plain_text(paragraph).strip()
517
+ visible = inline_span_plain_text(content)
518
+ if not content or plain.isdigit() or visible in seen:
519
+ continue
520
+ seen.add(visible)
521
+ result.append(content)
522
+ return result
523
+
524
+ @classmethod
525
+ def _section_page(cls, section: DocSection) -> list[dict[str, Any]]:
526
+ """转换一个 section,并将页面辅助文本稳定追加到末尾。"""
527
+
528
+ page: list[dict[str, Any]] = []
529
+ list_stack: list[dict[str, Any]] = []
530
+ list_identity: int | None = None
531
+ index_stack: list[dict[str, Any]] = []
532
+ for element in section.elements:
533
+ if isinstance(element, DocParagraph) and element.is_toc:
534
+ list_stack.clear()
535
+ list_identity = None
536
+ cls._append_index_item(page, index_stack, element)
537
+ for payload in element.images:
538
+ if isinstance(payload, DocImagePayload) and payload.equation_latex:
539
+ continue
540
+ image = cls._image_block(payload)
541
+ if image is not None:
542
+ page.append(image)
543
+ continue
544
+ index_stack.clear()
545
+ if (
546
+ isinstance(element, DocParagraph)
547
+ and element.list_info is not None
548
+ and not (element.is_title or element.heading_level is not None)
549
+ ):
550
+ list_identity = cls._append_list_item(page, list_stack, list_identity, element)
551
+ for payload in element.images:
552
+ if isinstance(payload, DocImagePayload) and payload.equation_latex:
553
+ continue
554
+ image = cls._image_block(payload)
555
+ if image is not None:
556
+ page.append(image)
557
+ continue
558
+ list_stack.clear()
559
+ list_identity = None
560
+ page.extend(cls._element_blocks(element))
561
+ page.extend({"type": BlockType.HEADER, "content": content} for content in cls._auxiliary_contents(section.headers))
562
+ page.extend({"type": BlockType.FOOTER, "content": content} for content in cls._auxiliary_contents(section.footers))
563
+ page.extend(
564
+ {"type": BlockType.PAGE_FOOTNOTE, "content": content} for content in cls._auxiliary_contents(section.footnotes)
565
+ )
566
+ return page
567
+
568
+ @classmethod
569
+ def _document_pages(cls, document: DocDocument) -> list[list[dict[str, Any]]]:
570
+ """转换整份 DOC,并至少保留一个空 section page。"""
571
+
572
+ pages = [cls._section_page(section) for section in document.sections]
573
+ return pages or [[]]