docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,256 @@
1
+ """解析 DOC CLX piece table 并恢复全局 UTF-16 CP 文本流。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from bisect import bisect_left
6
+ from dataclasses import dataclass
7
+ import struct
8
+
9
+ from ..errors import LegacyOfficeMalformedError
10
+ from ..legacy.binary import bounded_slice, get_u16, get_u32
11
+
12
+ from .records import DocBudget
13
+
14
+
15
+ @dataclass(frozen=True, slots=True)
16
+ class Piece:
17
+ """一个把逻辑 CP 范围映射到 WordDocument FC 的 piece。"""
18
+
19
+ cp_start: int
20
+ cp_end: int
21
+ fc: int
22
+ compressed: bool
23
+ prm: bytes = b""
24
+
25
+
26
+ @dataclass(slots=True)
27
+ class TextStream:
28
+ """字符标量及其 CP、FC 和 piece 映射。"""
29
+
30
+ chars: list[str]
31
+ cps: list[int]
32
+ fcs: list[int]
33
+ piece_indexes: list[int]
34
+
35
+ def index_of_cp(self, cp: int) -> int:
36
+ """返回首个 CP 不小于目标值的字符索引。"""
37
+
38
+ return bisect_left(self.cps, max(cp, 0))
39
+
40
+ def text_between(self, cp_start: int, cp_end: int) -> str:
41
+ """返回指定 CP 半开区间的 Unicode 文本。"""
42
+
43
+ return "".join(self.chars[self.index_of_cp(cp_start) : self.index_of_cp(cp_end)])
44
+
45
+
46
+ def _prm0_grpprl(prm: int) -> bytes:
47
+ """把 Prm0 中已支持的单个属性还原为 grpprl。"""
48
+
49
+ isprm = (prm >> 1) & 0x7F
50
+ value = (prm >> 8) & 0xFF
51
+ opcode = {
52
+ 0x0C: 0x260A, # sprmPIlvl
53
+ 0x18: 0x2416, # sprmPFInTable
54
+ 0x19: 0x2417, # sprmPFTtp
55
+ 0x55: 0x0835, # sprmCFBold
56
+ 0x56: 0x0836, # sprmCFItalic
57
+ 0x57: 0x0837, # sprmCFStrike
58
+ 0x78: 0x2640, # sprmPOutLvl
59
+ }.get(isprm)
60
+ if opcode is None:
61
+ return b""
62
+ return struct.pack("<HB", opcode, value)
63
+
64
+
65
+ def _parse_plc_pcd(plc: bytes, prcs: list[bytes], budget: DocBudget) -> list[Piece]:
66
+ """解析 Pcdt 内的 PlcPcd 并绑定 piece Prm。"""
67
+
68
+ if len(plc) < 16 or (len(plc) - 4) % 12:
69
+ raise LegacyOfficeMalformedError("DOC piece table is empty or malformed")
70
+ count = (len(plc) - 4) // 12
71
+ budget.charge(count)
72
+ cp_bytes = (count + 1) * 4
73
+ pieces: list[Piece] = []
74
+ previous_cp = -1
75
+ for index in range(count):
76
+ cp_start = get_u32(plc, index * 4)
77
+ cp_end = get_u32(plc, (index + 1) * 4)
78
+ pcd_offset = cp_bytes + index * 8
79
+ fc_raw = get_u32(plc, pcd_offset + 2)
80
+ prm = get_u16(plc, pcd_offset + 6) or 0
81
+ if cp_start is None or cp_end is None or fc_raw is None:
82
+ raise LegacyOfficeMalformedError("DOC piece table is truncated")
83
+ if cp_start < previous_cp or cp_end < cp_start:
84
+ raise LegacyOfficeMalformedError("DOC piece CP values are not ordered")
85
+ previous_cp = cp_end
86
+ compressed = bool(fc_raw & 0x4000_0000)
87
+ fc = fc_raw & 0x3FFF_FFFF
88
+ if compressed:
89
+ fc //= 2
90
+ grpprl = b""
91
+ if prm & 1:
92
+ prc_index = prm >> 1
93
+ if prc_index < len(prcs):
94
+ grpprl = prcs[prc_index]
95
+ elif prm:
96
+ grpprl = _prm0_grpprl(prm)
97
+ pieces.append(
98
+ Piece(
99
+ cp_start=int(cp_start),
100
+ cp_end=int(cp_end),
101
+ fc=int(fc),
102
+ compressed=compressed,
103
+ prm=grpprl,
104
+ )
105
+ )
106
+ return pieces
107
+
108
+
109
+ def parse_clx(table_stream: bytes, *, offset: int, size: int, budget: DocBudget) -> list[Piece]:
110
+ """解析 CLX 中的 Prc 数组和最终 Pcdt。"""
111
+
112
+ clx = bounded_slice(table_stream, offset, size)
113
+ if clx is None:
114
+ raise LegacyOfficeMalformedError("DOC CLX range is out of bounds")
115
+ prcs: list[bytes] = []
116
+ cursor = 0
117
+ while cursor < len(clx):
118
+ kind = clx[cursor]
119
+ if kind == 1:
120
+ length = get_u16(clx, cursor + 1)
121
+ if length is None:
122
+ raise LegacyOfficeMalformedError("DOC CLX Prc is truncated")
123
+ payload = bounded_slice(clx, cursor + 3, length)
124
+ if payload is None:
125
+ raise LegacyOfficeMalformedError("DOC CLX Prc exceeds its range")
126
+ budget.charge()
127
+ prcs.append(payload)
128
+ cursor += 3 + length
129
+ continue
130
+ if kind == 2:
131
+ length = get_u32(clx, cursor + 1)
132
+ if length is None:
133
+ raise LegacyOfficeMalformedError("DOC CLX Pcdt is truncated")
134
+ plc = bounded_slice(clx, cursor + 5, length)
135
+ if plc is None:
136
+ raise LegacyOfficeMalformedError("DOC PlcPcd exceeds CLX")
137
+ return _parse_plc_pcd(plc, prcs, budget)
138
+ raise LegacyOfficeMalformedError("DOC CLX contains an unknown record")
139
+ raise LegacyOfficeMalformedError("DOC CLX does not contain a Pcdt")
140
+
141
+
142
+ def legacy_single_piece(*, fc_min: int, fc_mac: int, ccp_text: int) -> list[Piece]:
143
+ """为没有 CLX 的非 complex 文档构造保守单 piece。"""
144
+
145
+ if fc_min < 0 or fc_mac <= fc_min:
146
+ return []
147
+ length = min(fc_mac - fc_min, max(ccp_text, 0))
148
+ if length <= 0:
149
+ return []
150
+ return [Piece(cp_start=0, cp_end=length, fc=fc_min, compressed=True)]
151
+
152
+
153
+ def codec_for_lid(lid: int) -> str:
154
+ """把 Word LID 映射为 Python 可用的 ANSI/DBCS codec。"""
155
+
156
+ primary = lid & 0x03FF
157
+ if primary == 0x11:
158
+ return "cp932"
159
+ if primary == 0x12:
160
+ return "cp949"
161
+ if primary == 0x04:
162
+ return "cp950" if lid in {0x0404, 0x0C04, 0x1404, 0x7C04} else "cp936"
163
+ if primary in {0x01, 0x20, 0x29}:
164
+ return "cp1256"
165
+ if primary in {0x02, 0x19, 0x22, 0x23}:
166
+ return "cp1251"
167
+ if primary in {0x05, 0x0E, 0x15, 0x18, 0x1A, 0x1B, 0x24}:
168
+ return "cp1250"
169
+ if primary == 0x08:
170
+ return "cp1253"
171
+ if primary == 0x0D:
172
+ return "cp1255"
173
+ if primary == 0x1E:
174
+ return "cp874"
175
+ if primary in {0x1F, 0x2C}:
176
+ return "cp1254"
177
+ if primary in {0x25, 0x26, 0x27}:
178
+ return "cp1257"
179
+ if primary == 0x2A:
180
+ return "cp1258"
181
+ return "cp1252"
182
+
183
+
184
+ def _lead_byte(codec: str, value: int) -> bool:
185
+ """判断一个压缩 piece 字节是否为 DBCS 首字节。"""
186
+
187
+ if codec == "cp932":
188
+ return 0x81 <= value <= 0x9F or 0xE0 <= value <= 0xFC
189
+ if codec in {"cp936", "cp949", "cp950"}:
190
+ return 0x81 <= value <= 0xFE
191
+ return False
192
+
193
+
194
+ def extract_text(
195
+ word_document: bytes,
196
+ pieces: list[Piece],
197
+ *,
198
+ total_cp: int,
199
+ codec: str,
200
+ budget: DocBudget,
201
+ ) -> TextStream:
202
+ """按 piece 顺序恢复字符,并保留字符到 CP/FC 的反向映射。"""
203
+
204
+ chars: list[str] = []
205
+ cps: list[int] = []
206
+ fcs: list[int] = []
207
+ piece_indexes: list[int] = []
208
+ for piece_index, piece in enumerate(pieces):
209
+ if piece.cp_start >= total_cp:
210
+ break
211
+ cp_cursor = piece.cp_start
212
+ logical_length = min(piece.cp_end, total_cp) - piece.cp_start
213
+ if logical_length <= 0:
214
+ continue
215
+ if piece.compressed:
216
+ payload = bounded_slice(word_document, piece.fc, logical_length)
217
+ if payload is None:
218
+ continue
219
+ cursor = 0
220
+ while cursor < len(payload):
221
+ width = 2 if _lead_byte(codec, payload[cursor]) and cursor + 1 < len(payload) else 1
222
+ decoded = payload[cursor : cursor + width].decode(codec, errors="replace")
223
+ for char in decoded:
224
+ chars.append(char)
225
+ cps.append(cp_cursor)
226
+ fcs.append(piece.fc + cursor)
227
+ piece_indexes.append(piece_index)
228
+ cp_cursor += width
229
+ cursor += width
230
+ budget.charge(width)
231
+ else:
232
+ byte_length = logical_length * 2
233
+ payload = bounded_slice(word_document, piece.fc, byte_length)
234
+ if payload is None:
235
+ continue
236
+ cursor = 0
237
+ while cursor + 2 <= len(payload):
238
+ first = int(struct.unpack_from("<H", payload, cursor)[0])
239
+ width = 2
240
+ units = [first]
241
+ if 0xD800 <= first <= 0xDBFF and cursor + 4 <= len(payload):
242
+ second = int(struct.unpack_from("<H", payload, cursor + 2)[0])
243
+ if 0xDC00 <= second <= 0xDFFF:
244
+ units.append(second)
245
+ width = 4
246
+ raw = struct.pack(f"<{len(units)}H", *units)
247
+ char = raw.decode("utf-16le", errors="replace")
248
+ chars.append(char)
249
+ cps.append(cp_cursor)
250
+ fcs.append(piece.fc + cursor)
251
+ piece_indexes.append(piece_index)
252
+ unit_count = width // 2
253
+ cp_cursor += unit_count
254
+ cursor += width
255
+ budget.charge(unit_count)
256
+ return TextStream(chars=chars, cps=cps, fcs=fcs, piece_indexes=piece_indexes)
@@ -0,0 +1,40 @@
1
+ """DOC 二进制结构使用的有界整数、PLC 和记录预算工具。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import struct
7
+
8
+ from ..errors import LegacyOfficeResourceLimitError
9
+ from ..limits import MAX_RECORDS
10
+
11
+
12
+ @dataclass(slots=True)
13
+ class DocBudget:
14
+ """限制 DOC 解析累计访问的记录和文本单元数。"""
15
+
16
+ visited: int = 0
17
+
18
+ def charge(self, amount: int = 1) -> None:
19
+ """计入本次访问量,超过统一上限时稳定失败。"""
20
+
21
+ if amount < 0 or self.visited + amount > MAX_RECORDS:
22
+ raise LegacyOfficeResourceLimitError(f"DOC records exceed max_records={MAX_RECORDS}")
23
+ self.visited += amount
24
+
25
+
26
+ def parse_plc(data: bytes, *, item_size: int, budget: DocBudget) -> tuple[list[int], list[bytes]]:
27
+ """解析由 CP 数组和定长数据项组成的通用 PLC。"""
28
+
29
+ if item_size < 0 or len(data) < 4:
30
+ return [], []
31
+ denominator = 4 + item_size
32
+ payload = len(data) - 4
33
+ if denominator <= 0 or payload % denominator:
34
+ return [], []
35
+ count = payload // denominator
36
+ budget.charge(count + 1)
37
+ cp_bytes = (count + 1) * 4
38
+ cps = [int(struct.unpack_from("<I", data, index * 4)[0]) for index in range(count + 1)]
39
+ items = [data[cp_bytes + index * item_size : cp_bytes + (index + 1) * item_size] for index in range(count)]
40
+ return cps, items
@@ -0,0 +1,269 @@
1
+ """遍历并应用 Word 二进制单属性修饰符 SPRM。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, replace
6
+ import struct
7
+ from typing import Callable
8
+
9
+ from ..legacy.binary import get_i16, get_u16, get_u32
10
+ from .models import DocCharStyle, DocTableCellFormat, DocTableFormat
11
+ from .records import DocBudget
12
+
13
+
14
+ def _operand_length(opcode: int, operand: bytes) -> int:
15
+ """根据 SPRM 的 spra 字段计算 operand 字节数。"""
16
+
17
+ spra = opcode >> 13
18
+ if spra in {0, 1}:
19
+ return 1
20
+ if spra in {2, 4, 5}:
21
+ return 2
22
+ if spra == 3:
23
+ return 4
24
+ if spra == 7:
25
+ return 3
26
+ if opcode == 0xD608:
27
+ return (get_u16(operand, 0) or -1) + 1
28
+ return (operand[0] + 1) if operand else 0
29
+
30
+
31
+ def walk_sprms(
32
+ grpprl: bytes,
33
+ callback: Callable[[int, bytes], None],
34
+ *,
35
+ budget: DocBudget | None = None,
36
+ ) -> None:
37
+ """有界顺序遍历 grpprl,截断尾部按可恢复内容处理。"""
38
+
39
+ cursor = 0
40
+ while cursor + 2 <= len(grpprl):
41
+ opcode = int(struct.unpack_from("<H", grpprl, cursor)[0])
42
+ cursor += 2
43
+ length = _operand_length(opcode, grpprl[cursor:])
44
+ if length <= 0 or cursor + length > len(grpprl):
45
+ return
46
+ if budget is not None:
47
+ budget.charge()
48
+ callback(opcode, grpprl[cursor : cursor + length])
49
+ cursor += length
50
+
51
+
52
+ def _toggle(operand: bytes, base: bool) -> bool | None:
53
+ """把 Word ToggleOperand 解析为相对样式基值。"""
54
+
55
+ if not operand:
56
+ return None
57
+ return {0: False, 1: True, 0x80: base, 0x81: not base}.get(operand[0])
58
+
59
+
60
+ def chpx_style_id(grpprl: bytes) -> int | None:
61
+ """读取 CHPX 指定的字符样式 istd。"""
62
+
63
+ result: int | None = None
64
+
65
+ def consume(opcode: int, operand: bytes) -> None:
66
+ """记录最后一个有效 sprmCIstd。"""
67
+
68
+ nonlocal result
69
+ if opcode == 0x4A30:
70
+ result = get_u16(operand, 0)
71
+
72
+ walk_sprms(grpprl, consume)
73
+ return result
74
+
75
+
76
+ def chpx_picture_location(grpprl: bytes) -> int | None:
77
+ """读取 CHPX 中的 sprmCPicLocation。"""
78
+
79
+ result: int | None = None
80
+
81
+ def consume(opcode: int, operand: bytes) -> None:
82
+ """记录最后一个有效图片偏移。"""
83
+
84
+ nonlocal result
85
+ if opcode == 0x6A03:
86
+ result = get_u32(operand, 0)
87
+
88
+ walk_sprms(grpprl, consume)
89
+ return result
90
+
91
+
92
+ def apply_character_sprms(
93
+ grpprl: bytes,
94
+ current: DocCharStyle,
95
+ style_base: DocCharStyle,
96
+ *,
97
+ budget: DocBudget | None = None,
98
+ ) -> DocCharStyle:
99
+ """按 Word 样式覆盖顺序把 CHPX 应用到字符样式。"""
100
+
101
+ style = current
102
+
103
+ def consume(opcode: int, operand: bytes) -> None:
104
+ """应用当前可表达的字符属性。"""
105
+
106
+ nonlocal style
107
+ toggle_field = {
108
+ 0x0800: "deleted", # sprmCFRMarkDel
109
+ 0x0802: "hidden", # sprmCFFldVanish
110
+ 0x0835: "bold",
111
+ 0x0836: "italic",
112
+ 0x0837: "strike",
113
+ 0x083C: "hidden", # sprmCFVanish
114
+ }.get(opcode)
115
+ if toggle_field is not None:
116
+ value = _toggle(operand, bool(getattr(style_base, toggle_field)))
117
+ if value is not None:
118
+ style = replace(style, **{toggle_field: value})
119
+ return
120
+ if opcode == 0x2A3E and operand: # sprmCKul
121
+ style = replace(style, underline=operand[0] not in {0, 5})
122
+ elif opcode == 0x2A48 and operand: # sprmCIss
123
+ style = replace(
124
+ style,
125
+ superscript=operand[0] == 1,
126
+ subscript=operand[0] == 2,
127
+ )
128
+ elif opcode == 0x2A53 and operand: # sprmCFDStrike
129
+ style = replace(style, strike=operand[0] != 0)
130
+ elif opcode == 0x2A54 and operand: # sprmCEm
131
+ style = replace(style, emphasis=operand[0] != 0)
132
+
133
+ walk_sprms(grpprl, consume, budget=budget)
134
+ return style
135
+
136
+
137
+ @dataclass(frozen=True, slots=True)
138
+ class PapDelta:
139
+ """PAPX 或段落样式对可见段落属性的增量。"""
140
+
141
+ in_table: bool | None = None
142
+ row_mark: bool | None = None
143
+ outline_level: int | None | object = None
144
+ ilfo: int | None = None
145
+ ilvl: int | None = None
146
+ table_depth: int | None = None
147
+ inner_cell: bool | None = None
148
+ inner_row: bool | None = None
149
+ table: DocTableFormat | None = None
150
+
151
+ def merge(self, over: PapDelta) -> PapDelta:
152
+ """让后应用的段落属性覆盖当前增量。"""
153
+
154
+ return PapDelta(
155
+ in_table=over.in_table if over.in_table is not None else self.in_table,
156
+ row_mark=over.row_mark if over.row_mark is not None else self.row_mark,
157
+ outline_level=(over.outline_level if over.outline_level is not None else self.outline_level),
158
+ ilfo=over.ilfo if over.ilfo is not None else self.ilfo,
159
+ ilvl=over.ilvl if over.ilvl is not None else self.ilvl,
160
+ table_depth=over.table_depth if over.table_depth is not None else self.table_depth,
161
+ inner_cell=over.inner_cell if over.inner_cell is not None else self.inner_cell,
162
+ inner_row=over.inner_row if over.inner_row is not None else self.inner_row,
163
+ table=over.table if over.table is not None else self.table,
164
+ )
165
+
166
+
167
+ def _parse_tdef_table(operand: bytes) -> DocTableFormat | None:
168
+ """解析 TDefTableOperand 中的边界和横纵向合并标志。"""
169
+
170
+ if len(operand) < 3:
171
+ return None
172
+ columns = operand[2]
173
+ if columns > 63:
174
+ return None
175
+ boundaries: list[int] = []
176
+ for index in range(columns + 1):
177
+ value = get_i16(operand, 3 + index * 2)
178
+ if value is None:
179
+ return None
180
+ boundaries.append(value)
181
+ cells: list[DocTableCellFormat] = []
182
+ tc_base = 3 + (columns + 1) * 2
183
+ for index in range(columns):
184
+ flags = get_u16(operand, tc_base + index * 20) or 0
185
+ horizontal = flags & 0x3
186
+ vertical = (flags >> 5) & 0x3
187
+ right = boundaries[index + 1]
188
+ cells.append(
189
+ DocTableCellFormat(
190
+ right=right,
191
+ horizontal_first=horizontal >= 2,
192
+ horizontal_continue=horizontal == 1,
193
+ vertical_first=vertical == 3,
194
+ vertical_continue=vertical == 1,
195
+ )
196
+ )
197
+ return DocTableFormat(boundaries=tuple(boundaries), cells=tuple(cells))
198
+
199
+
200
+ def apply_paragraph_sprms(
201
+ grpprl: bytes,
202
+ data_stream: bytes,
203
+ initial: PapDelta | None = None,
204
+ *,
205
+ budget: DocBudget | None = None,
206
+ ) -> PapDelta:
207
+ """应用 PAPX SPRM,并解析 huge PAPX 与表格行属性。"""
208
+
209
+ delta = initial or PapDelta()
210
+
211
+ def consume(opcode: int, operand: bytes) -> None:
212
+ """应用一个段落或表格属性。"""
213
+
214
+ nonlocal delta
215
+ if opcode == 0x2416 and operand:
216
+ delta = replace(delta, in_table=operand[0] != 0)
217
+ elif opcode == 0x2417 and operand:
218
+ delta = replace(delta, row_mark=operand[0] != 0)
219
+ elif opcode == 0x6646:
220
+ offset = get_u32(operand, 0)
221
+ length = get_u16(data_stream, offset) if offset is not None else None
222
+ if offset is not None and length is not None and offset + 2 + length <= len(data_stream):
223
+ delta = apply_paragraph_sprms(
224
+ data_stream[offset + 2 : offset + 2 + length],
225
+ b"",
226
+ delta,
227
+ budget=budget,
228
+ )
229
+ elif opcode == 0x2640 and operand:
230
+ delta = replace(delta, outline_level=operand[0] + 1 if operand[0] < 9 else -1)
231
+ elif opcode == 0x260A and operand:
232
+ delta = replace(delta, ilvl=int(operand[0]))
233
+ elif opcode == 0x460B:
234
+ delta = replace(delta, ilfo=get_u16(operand, 0))
235
+ elif opcode == 0x6649:
236
+ depth = get_u32(operand, 0)
237
+ if depth is not None:
238
+ delta = replace(delta, table_depth=int(depth))
239
+ elif opcode == 0x664A:
240
+ raw = get_u32(operand, 0)
241
+ if raw is not None:
242
+ signed = struct.unpack("<i", struct.pack("<I", raw))[0]
243
+ delta = replace(delta, table_depth=max(0, (delta.table_depth or 0) + signed))
244
+ elif opcode == 0x244B and operand:
245
+ delta = replace(delta, inner_cell=operand[0] != 0)
246
+ elif opcode == 0x244C and operand:
247
+ delta = replace(delta, inner_row=operand[0] != 0)
248
+ elif opcode == 0xD608:
249
+ table = _parse_tdef_table(operand)
250
+ if table is not None:
251
+ header = delta.table.header if delta.table is not None else False
252
+ delta = replace(delta, table=replace(table, header=header))
253
+ elif opcode == 0x3404 and operand:
254
+ table = delta.table or DocTableFormat()
255
+ delta = replace(delta, table=replace(table, header=operand[0] != 0))
256
+ elif opcode == 0xD62B and len(operand) >= 3 and delta.table is not None:
257
+ cell_index = operand[1]
258
+ flag = operand[2]
259
+ cells = list(delta.table.cells)
260
+ if cell_index < len(cells):
261
+ cells[cell_index] = replace(
262
+ cells[cell_index],
263
+ vertical_continue=flag == 1,
264
+ vertical_first=flag == 3,
265
+ )
266
+ delta = replace(delta, table=replace(delta.table, cells=tuple(cells)))
267
+
268
+ walk_sprms(grpprl, consume, budget=budget)
269
+ return delta