docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,201 @@
1
+ """有界读取 Excel 97–2003 BIFF 记录流。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import struct
7
+ from typing import Iterator
8
+
9
+ from loguru import logger
10
+
11
+ from ..errors import LegacyOfficeResourceLimitError
12
+ from ..limits import MAX_RECORDS
13
+
14
+ BOF = 0x0809
15
+ EOF = 0x000A
16
+ CONTINUE = 0x003C
17
+
18
+
19
+ @dataclass(frozen=True, slots=True)
20
+ class BiffRecord:
21
+ """一条已完成边界校验的 BIFF 记录。"""
22
+
23
+ offset: int
24
+ record_type: int
25
+ payload: bytes
26
+ next_offset: int
27
+
28
+
29
+ @dataclass(slots=True)
30
+ class RecordBudget:
31
+ """跨 globals、worksheet 与 OfficeArt 共享记录访问预算。"""
32
+
33
+ count: int = 0
34
+
35
+ def charge(self) -> None:
36
+ """计入一条记录并在超过固定上限时硬失败。"""
37
+
38
+ self.count += 1
39
+ if self.count > MAX_RECORDS:
40
+ raise LegacyOfficeResourceLimitError(f"workbook stream exceeds max_records={MAX_RECORDS}")
41
+
42
+
43
+ def record_at(
44
+ data: bytes,
45
+ offset: int,
46
+ *,
47
+ budget: RecordBudget | None = None,
48
+ ) -> BiffRecord | None:
49
+ """读取指定偏移的 BIFF 记录,截断 header 或 body 返回空值。"""
50
+
51
+ if offset < 0 or offset + 4 > len(data):
52
+ return None
53
+ record_type, length = struct.unpack_from("<HH", data, offset)
54
+ body_start = offset + 4
55
+ body_end = body_start + int(length)
56
+ if body_end < body_start or body_end > len(data):
57
+ return None
58
+ if budget is not None:
59
+ budget.charge()
60
+ return BiffRecord(
61
+ offset=offset,
62
+ record_type=int(record_type),
63
+ payload=data[body_start:body_end],
64
+ next_offset=body_end,
65
+ )
66
+
67
+
68
+ def iter_records(
69
+ data: bytes,
70
+ *,
71
+ start: int = 0,
72
+ stop_at_eof: bool = False,
73
+ budget: RecordBudget | None = None,
74
+ ) -> Iterator[BiffRecord]:
75
+ """顺序遍历 BIFF 记录,并在截断尾部保留已经完成的记录。"""
76
+
77
+ cursor = start
78
+ while cursor < len(data):
79
+ if data[cursor:] and not any(data[cursor:]):
80
+ return
81
+ record = record_at(data, cursor, budget=budget)
82
+ if record is None:
83
+ logger.warning(
84
+ "XLS_TRUNCATED_RECORD: workbook stream ends mid-record at byte {}",
85
+ cursor,
86
+ )
87
+ return
88
+ yield record
89
+ cursor = record.next_offset
90
+ if stop_at_eof and record.record_type == EOF:
91
+ return
92
+
93
+
94
+ def collect_continues(
95
+ data: bytes,
96
+ base: BiffRecord,
97
+ *,
98
+ budget: RecordBudget,
99
+ ) -> tuple[list[bytes], int]:
100
+ """收集紧随 base 的 CONTINUE bodies,并返回下一条非延续记录偏移。"""
101
+
102
+ segments = [base.payload]
103
+ cursor = base.next_offset
104
+ while True:
105
+ record = record_at(data, cursor)
106
+ if record is None or record.record_type != CONTINUE:
107
+ return segments, cursor
108
+ budget.charge()
109
+ segments.append(record.payload)
110
+ cursor = record.next_offset
111
+
112
+
113
+ class SegmentReader:
114
+ """在基础记录及其 CONTINUE segments 上执行有界顺序读取。"""
115
+
116
+ def __init__(self, segments: list[bytes]) -> None:
117
+ """保存 segment 列表并把游标置于首段开头。"""
118
+
119
+ self.segments = segments
120
+ self.segment_index = 0
121
+ self.offset = 0
122
+
123
+ def remaining_in_segment(self) -> int:
124
+ """返回当前 segment 尚未消费的字节数。"""
125
+
126
+ if self.segment_index >= len(self.segments):
127
+ return 0
128
+ return len(self.segments[self.segment_index]) - self.offset
129
+
130
+ def normalize(self) -> None:
131
+ """跳过已经耗尽的 segments。"""
132
+
133
+ while self.segment_index < len(self.segments) and self.offset >= len(self.segments[self.segment_index]):
134
+ self.segment_index += 1
135
+ self.offset = 0
136
+
137
+ def next_segment(self) -> bool:
138
+ """显式移动到下一 segment,若不存在则返回 False。"""
139
+
140
+ if self.segment_index + 1 >= len(self.segments):
141
+ return False
142
+ self.segment_index += 1
143
+ self.offset = 0
144
+ return True
145
+
146
+ def read(self, size: int) -> bytes | None:
147
+ """只在当前 segment 内读取固定长度字段。"""
148
+
149
+ self.normalize()
150
+ if size < 0 or self.segment_index >= len(self.segments):
151
+ return None
152
+ segment = self.segments[self.segment_index]
153
+ end = self.offset + size
154
+ if end > len(segment):
155
+ return None
156
+ output = segment[self.offset : end]
157
+ self.offset = end
158
+ return output
159
+
160
+ def read_across(self, size: int) -> bytes | None:
161
+ """跨 segment 读取普通非字符数据。"""
162
+
163
+ if size < 0:
164
+ return None
165
+ output = bytearray()
166
+ remaining = size
167
+ while remaining:
168
+ self.normalize()
169
+ available = self.remaining_in_segment()
170
+ if available <= 0:
171
+ return None
172
+ take = min(available, remaining)
173
+ chunk = self.read(take)
174
+ if chunk is None:
175
+ return None
176
+ output.extend(chunk)
177
+ remaining -= take
178
+ return bytes(output)
179
+
180
+ def skip(self, size: int) -> bool:
181
+ """跨 segments 跳过指定字节数。"""
182
+
183
+ return self.read_across(size) is not None
184
+
185
+ def u8(self) -> int | None:
186
+ """读取一个无符号 8 位整数。"""
187
+
188
+ value = self.read(1)
189
+ return int(value[0]) if value is not None else None
190
+
191
+ def u16(self) -> int | None:
192
+ """读取一个小端无符号 16 位整数。"""
193
+
194
+ value = self.read(2)
195
+ return int(struct.unpack("<H", value)[0]) if value is not None else None
196
+
197
+ def u32(self) -> int | None:
198
+ """读取一个小端无符号 32 位整数。"""
199
+
200
+ value = self.read(4)
201
+ return int(struct.unpack("<I", value)[0]) if value is not None else None
@@ -0,0 +1,205 @@
1
+ """BIFF5–BIFF8 字符串、富文本区间与 codepage 解码。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import re
7
+ import struct
8
+
9
+ from .models import XlsFontStyle, XlsRichRun, XlsRichText
10
+ from .records import SegmentReader
11
+
12
+
13
+ @dataclass(frozen=True, slots=True)
14
+ class DecodedString:
15
+ """已解码文本及以 UTF-16 code unit 表示的字体切换点。"""
16
+
17
+ text: str
18
+ font_starts: tuple[tuple[int, int], ...] = ()
19
+
20
+
21
+ def clean_text(text: str) -> str:
22
+ """规范换行、NUL 与不可见控制字符,同时保留制表符。"""
23
+
24
+ normalized = text.replace("\r\n", "\n").replace("\r", "\n").rstrip("\x00")
25
+ return re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f]", "", normalized)
26
+
27
+
28
+ def codepage_name(codepage: int) -> str:
29
+ """把 BIFF CODEPAGE 值映射为 Python codec 名称。"""
30
+
31
+ return {
32
+ 874: "cp874",
33
+ 932: "shift_jis",
34
+ 936: "gbk",
35
+ 949: "euc_kr",
36
+ 950: "big5",
37
+ 1200: "utf-16le",
38
+ 1250: "cp1250",
39
+ 1251: "cp1251",
40
+ 1252: "cp1252",
41
+ 1253: "cp1253",
42
+ 1254: "cp1254",
43
+ 1255: "cp1255",
44
+ 1256: "cp1256",
45
+ 1257: "cp1257",
46
+ 1258: "cp1258",
47
+ }.get(int(codepage), "cp1252")
48
+
49
+
50
+ def _decode_utf16_units(units: list[int]) -> str:
51
+ """容错解码 UTF-16 code units。"""
52
+
53
+ payload = struct.pack(f"<{len(units)}H", *units) if units else b""
54
+ return payload.decode("utf-16le", "replace")
55
+
56
+
57
+ def read_biff8_string(
58
+ reader: SegmentReader,
59
+ *,
60
+ short: bool,
61
+ rich: bool,
62
+ ) -> DecodedString | None:
63
+ """读取可跨 CONTINUE 且可切换压缩模式的 BIFF8 Unicode 字符串。"""
64
+
65
+ character_count = reader.u8() if short else reader.u16()
66
+ flags = reader.u8()
67
+ if character_count is None or flags is None:
68
+ return None
69
+ wide = bool(flags & 0x01)
70
+ run_count = reader.u16() if rich and flags & 0x08 else 0
71
+ extension_size = reader.u32() if rich and flags & 0x04 else 0
72
+ if run_count is None or extension_size is None:
73
+ return None
74
+
75
+ units: list[int] = []
76
+ remaining = int(character_count)
77
+ while remaining > 0:
78
+ if reader.remaining_in_segment() == 0:
79
+ if not reader.next_segment():
80
+ return None
81
+ repeated_flags = reader.u8()
82
+ if repeated_flags is None:
83
+ return None
84
+ wide = bool(repeated_flags & 0x01)
85
+ unit_size = 2 if wide else 1
86
+ take = min(reader.remaining_in_segment() // unit_size, remaining)
87
+ if take <= 0:
88
+ return None
89
+ payload = reader.read(take * unit_size)
90
+ if payload is None:
91
+ return None
92
+ if wide:
93
+ units.extend(struct.unpack(f"<{take}H", payload))
94
+ else:
95
+ units.extend(int(value) for value in payload)
96
+ remaining -= take
97
+
98
+ font_starts: list[tuple[int, int]] = []
99
+ for _ in range(int(run_count)):
100
+ run = reader.read_across(4)
101
+ if run is None:
102
+ break
103
+ character_index, font_index = struct.unpack("<HH", run)
104
+ if character_index <= character_count:
105
+ font_starts.append((int(character_index), int(font_index)))
106
+ if extension_size:
107
+ reader.skip(int(extension_size))
108
+ return DecodedString(
109
+ text=clean_text(_decode_utf16_units(units)),
110
+ font_starts=tuple(font_starts),
111
+ )
112
+
113
+
114
+ def read_byte_string(
115
+ reader: SegmentReader,
116
+ *,
117
+ short: bool,
118
+ encoding: str,
119
+ ) -> DecodedString | None:
120
+ """按 workbook CODEPAGE 读取 BIFF5/BIFF7 单字节字符串。"""
121
+
122
+ character_count = reader.u8() if short else reader.u16()
123
+ if character_count is None:
124
+ return None
125
+ payload = reader.read_across(int(character_count))
126
+ if payload is None:
127
+ return None
128
+ try:
129
+ text = payload.decode(encoding, "replace")
130
+ except LookupError:
131
+ text = payload.decode("cp1252", "replace")
132
+ return DecodedString(clean_text(text))
133
+
134
+
135
+ def utf16_unit_to_index(text: str, unit_offset: int) -> int:
136
+ """把 UTF-16 code unit 偏移转换为 Python 字符索引。"""
137
+
138
+ units = 0
139
+ for index, char in enumerate(text):
140
+ if units >= unit_offset:
141
+ return index
142
+ units += 2 if ord(char) > 0xFFFF else 1
143
+ return len(text)
144
+
145
+
146
+ def to_rich_text(
147
+ decoded: DecodedString,
148
+ fonts: list[XlsFontStyle],
149
+ ) -> XlsRichText:
150
+ """把字体切换点解析成稳定的字符区间。"""
151
+
152
+ if not decoded.font_starts or not decoded.text:
153
+ return XlsRichText(decoded.text)
154
+ starts = sorted(decoded.font_starts, key=lambda item: item[0])
155
+ runs: list[XlsRichRun] = []
156
+ for index, (unit_start, font_index) in enumerate(starts):
157
+ unit_end = starts[index + 1][0] if index + 1 < len(starts) else 0x7FFF_FFFF
158
+ start = utf16_unit_to_index(decoded.text, unit_start)
159
+ end = utf16_unit_to_index(decoded.text, unit_end)
160
+ resolved_font_index = font_index if font_index < 4 else font_index - 1
161
+ style = fonts[resolved_font_index] if 0 <= resolved_font_index < len(fonts) else XlsFontStyle()
162
+ if start < end and style != XlsFontStyle():
163
+ runs.append(XlsRichRun(start=start, end=end, style=style))
164
+ return XlsRichText(decoded.text, tuple(runs))
165
+
166
+
167
+ def read_txo_text(
168
+ base_payload: bytes,
169
+ continuation_segments: list[bytes],
170
+ fonts: list[XlsFontStyle],
171
+ ) -> XlsRichText | None:
172
+ """读取 TXO 文本和 8 字节 formatting runs。"""
173
+
174
+ if len(base_payload) < 14:
175
+ return None
176
+ character_count = int(struct.unpack_from("<H", base_payload, 10)[0])
177
+ formatting_size = int(struct.unpack_from("<H", base_payload, 12)[0])
178
+ if character_count <= 0:
179
+ return XlsRichText("")
180
+ units: list[int] = []
181
+ consumed_segments = 0
182
+ remaining = character_count
183
+ for segment in continuation_segments:
184
+ if remaining <= 0:
185
+ break
186
+ consumed_segments += 1
187
+ if not segment:
188
+ continue
189
+ wide = bool(segment[0] & 0x01)
190
+ payload = segment[1:]
191
+ if wide:
192
+ take = min(remaining, len(payload) // 2)
193
+ units.extend(struct.unpack(f"<{take}H", payload[: take * 2]))
194
+ else:
195
+ take = min(remaining, len(payload))
196
+ units.extend(int(value) for value in payload[:take])
197
+ remaining -= take
198
+ text = clean_text(_decode_utf16_units(units))
199
+ formatting = b"".join(continuation_segments[consumed_segments:])[:formatting_size]
200
+ starts: list[tuple[int, int]] = []
201
+ for offset in range(0, len(formatting) - 7, 8):
202
+ character_index, font_index = struct.unpack_from("<HH", formatting, offset)
203
+ if character_index <= character_count:
204
+ starts.append((int(character_index), int(font_index)))
205
+ return to_rich_text(DecodedString(text, tuple(starts)), fonts)