docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1145 @@
1
+ """纯 Python 解析 Excel 97–2003 Workbook BIFF stream。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ import struct
7
+ import uuid
8
+
9
+ from loguru import logger
10
+
11
+ from ..._shared.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
12
+ from ..errors import LegacyOfficeEncryptedError, LegacyOfficeMalformedError
13
+ from ..legacy.binary import get_f64, get_u16, get_u32
14
+ from ..legacy.officeart import OfficeArtShape, OfficeImagePayload, decode_bstore, extract_excel_shapes
15
+ from ..image import serialize_office_image
16
+ from ..equation.image import OfficeImageEquationDecoder
17
+
18
+ from .chart import chart_source_axes, chart_source_selection
19
+ from .models import (
20
+ XlsCell,
21
+ XlsChart,
22
+ XlsChartSheet,
23
+ XlsEquation,
24
+ XlsFontStyle,
25
+ XlsImage,
26
+ XlsRichRun,
27
+ XlsRichText,
28
+ XlsSheet,
29
+ XlsWorkbook,
30
+ )
31
+ from .number_format import builtin_number_format, format_number, format_text
32
+ from .records import BOF, CONTINUE, EOF, BiffRecord, RecordBudget, SegmentReader, collect_continues, iter_records, record_at
33
+ from .strings import DecodedString, clean_text, codepage_name, read_biff8_string, read_byte_string, read_txo_text, to_rich_text
34
+
35
+ FILEPASS = 0x002F
36
+ CODEPAGE = 0x0042
37
+ DATEMODE = 0x0022
38
+ BOUNDSHEET = 0x0085
39
+ SST = 0x00FC
40
+ FORMAT = 0x041E
41
+ XF = 0x00E0
42
+ FONT = 0x0031
43
+ ROW = 0x0208
44
+ COLINFO = 0x007D
45
+ MERGEDCELLS = 0x00E5
46
+ LABELSST = 0x00FD
47
+ LABEL = 0x0204
48
+ RSTRING = 0x00D6
49
+ NUMBER = 0x0203
50
+ RK = 0x027E
51
+ MULRK = 0x00BD
52
+ BOOLERR = 0x0205
53
+ FORMULA = 0x0006
54
+ STRING = 0x0207
55
+ MSODRAWINGGROUP = 0x00EB
56
+ MSODRAWING = 0x00EC
57
+ OBJ = 0x005D
58
+ TXO = 0x01B6
59
+ HLINK = 0x01B8
60
+ SUPBOOK = 0x01AE
61
+ EXTERNSHEET = 0x0017
62
+ WINDOW1 = 0x003D
63
+
64
+ WORKBOOK_GLOBALS_SUBSTREAM = 0x0005
65
+ WORKSHEET_SUBSTREAM = 0x0010
66
+ CHART_SUBSTREAM = 0x0020
67
+ MAX_ROWS = 65_536
68
+ MAX_COLS = 256
69
+
70
+ OBJ_CHART = 0x0005
71
+ OBJ_TEXTBOX = 0x0006
72
+ OBJ_PICTURE = 0x0008
73
+ OBJ_CHECKBOX = 0x000B
74
+
75
+
76
+ @dataclass(frozen=True, slots=True)
77
+ class _BoundSheet:
78
+ """BoundSheet8 目录项。"""
79
+
80
+ name: str
81
+ offset: int
82
+ visible: bool
83
+ sheet_type: int
84
+
85
+
86
+ @dataclass(frozen=True, slots=True)
87
+ class _CellFormat:
88
+ """XF 解析后的字体索引和数值格式代码。"""
89
+
90
+ font_index: int
91
+ format_code: str | None
92
+
93
+
94
+ @dataclass(slots=True)
95
+ class _Globals:
96
+ """Workbook Globals Substream 中供所有工作表共享的状态。"""
97
+
98
+ biff8: bool
99
+ date1904: bool = False
100
+ encoding: str = "cp1252"
101
+ sheets: list[_BoundSheet] = field(default_factory=list)
102
+ strings: list[DecodedString] = field(default_factory=list)
103
+ fonts: list[XlsFontStyle] = field(default_factory=list)
104
+ formats: list[_CellFormat] = field(default_factory=list)
105
+ images: dict[int, OfficeImagePayload] = field(default_factory=dict)
106
+ extern_sheets: list[int | None] = field(default_factory=list)
107
+ active_sheet_index: int | None = None
108
+
109
+ def read_string(
110
+ self,
111
+ reader: SegmentReader,
112
+ *,
113
+ short: bool,
114
+ rich: bool = False,
115
+ ) -> DecodedString | None:
116
+ """按当前 BIFF 版本和 codepage 读取字符串。"""
117
+
118
+ if self.biff8:
119
+ return read_biff8_string(reader, short=short, rich=rich)
120
+ return read_byte_string(reader, short=short, encoding=self.encoding)
121
+
122
+ def cell_format(self, index: int) -> _CellFormat:
123
+ """解析越界 XF 时返回 General 与无字体的稳定默认值。"""
124
+
125
+ if 0 <= index < len(self.formats):
126
+ return self.formats[index]
127
+ return _CellFormat(font_index=0, format_code=None)
128
+
129
+
130
+ @dataclass(slots=True)
131
+ class _SheetObject:
132
+ """一个 OBJ 记录及其后续 TXO 可见文本。"""
133
+
134
+ object_type: int
135
+ object_id: int
136
+ checked: bool | None = None
137
+ embedding_storage: str | None = None
138
+ text: XlsRichText | None = None
139
+ shape: OfficeArtShape | None = None
140
+
141
+
142
+ def _error_literal(code: int) -> str | None:
143
+ """把 BIFF error code 转成 Excel 可见错误文本。"""
144
+
145
+ return {
146
+ 0x00: "#NULL!",
147
+ 0x07: "#DIV/0!",
148
+ 0x0F: "#VALUE!",
149
+ 0x17: "#REF!",
150
+ 0x1D: "#NAME?",
151
+ 0x24: "#NUM!",
152
+ 0x2A: "#N/A",
153
+ 0x2B: "#GETTING_DATA",
154
+ }.get(int(code))
155
+
156
+
157
+ def _rk_number(value: int) -> float:
158
+ """解码 RK 压缩整数或截断双精度数。"""
159
+
160
+ if value & 0x02:
161
+ signed = struct.unpack("<i", struct.pack("<I", value))[0]
162
+ number = float(signed >> 2)
163
+ else:
164
+ number = struct.unpack("<d", struct.pack("<Q", (value & 0xFFFF_FFFC) << 32))[0]
165
+ return number / 100.0 if value & 0x01 else number
166
+
167
+
168
+ def _read_font(payload: bytes, *, biff8: bool) -> XlsFontStyle:
169
+ """提取 FONT 中可映射为 DocVortex 行内标签的字符属性。"""
170
+
171
+ if len(payload) < 11:
172
+ return XlsFontStyle()
173
+ flags = int(get_u16(payload, 2) or 0)
174
+ weight = int(get_u16(payload, 6) or 400)
175
+ escapement = int(get_u16(payload, 8) or 0)
176
+ underline = int(payload[10])
177
+ return XlsFontStyle(
178
+ bold=weight >= 700,
179
+ italic=bool(flags & 0x0002),
180
+ strike=bool(flags & 0x0008),
181
+ underline=underline != 0,
182
+ superscript=escapement == 1,
183
+ subscript=escapement == 2,
184
+ )
185
+
186
+
187
+ def _read_boundsheet(payload: bytes, globals_: _Globals) -> _BoundSheet | None:
188
+ """解析 sheet 偏移、可见性、类型和名称。"""
189
+
190
+ if len(payload) < 8:
191
+ return None
192
+ offset = get_u32(payload, 0)
193
+ if offset is None:
194
+ return None
195
+ reader = SegmentReader([payload[6:]])
196
+ decoded = globals_.read_string(reader, short=True)
197
+ if decoded is None:
198
+ return None
199
+ return _BoundSheet(
200
+ name=clean_text(decoded.text) or "Sheet",
201
+ offset=int(offset),
202
+ visible=(payload[4] & 0x03) == 0,
203
+ sheet_type=int(payload[5]),
204
+ )
205
+
206
+
207
+ def _read_sst(segments: list[bytes]) -> list[DecodedString]:
208
+ """读取共享字符串表,并允许损坏尾部保留已完成的 strings。"""
209
+
210
+ reader = SegmentReader(segments)
211
+ total = reader.u32()
212
+ unique = reader.u32()
213
+ if total is None or unique is None:
214
+ return []
215
+ strings: list[DecodedString] = []
216
+ while len(strings) < unique:
217
+ decoded = read_biff8_string(reader, short=False, rich=True)
218
+ if decoded is None:
219
+ logger.warning(
220
+ "XLS_SST_TRUNCATED: shared string table stopped at entry {}",
221
+ len(strings),
222
+ )
223
+ break
224
+ strings.append(decoded)
225
+ return strings
226
+
227
+
228
+ def _read_supbook(payload: bytes) -> bool:
229
+ """判断 SUPBOOK 是否表示当前工作簿内部 sheet 集合。"""
230
+
231
+ return len(payload) >= 4 and get_u16(payload, 2) == 0x0401
232
+
233
+
234
+ def _read_extern_sheets(
235
+ payload: bytes,
236
+ internal_supbooks: list[bool],
237
+ ) -> list[int | None]:
238
+ """把 XTI entries 解析为内部工作表索引。"""
239
+
240
+ count = min(int(get_u16(payload, 0) or 0), max(0, (len(payload) - 2) // 6))
241
+ result: list[int | None] = []
242
+ for index in range(count):
243
+ offset = 2 + index * 6
244
+ supbook, first_sheet, last_sheet = struct.unpack_from("<3H", payload, offset)
245
+ if (
246
+ supbook < len(internal_supbooks)
247
+ and internal_supbooks[supbook]
248
+ and first_sheet == last_sheet
249
+ and first_sheet < 0xFFFE
250
+ ):
251
+ result.append(int(first_sheet))
252
+ else:
253
+ result.append(None)
254
+ return result
255
+
256
+
257
+ def _read_globals(data: bytes, budget: RecordBudget) -> _Globals:
258
+ """解析 Workbook Globals Substream 及共享图片资源。"""
259
+
260
+ first = record_at(data, 0, budget=budget)
261
+ if first is None or first.record_type != BOF:
262
+ raise LegacyOfficeMalformedError("workbook stream does not start with a BOF record")
263
+ if get_u16(first.payload, 2) not in {WORKBOOK_GLOBALS_SUBSTREAM, None}:
264
+ raise LegacyOfficeMalformedError("first BIFF substream is not workbook globals")
265
+ version = get_u16(first.payload, 0)
266
+ if version not in {0x0500, 0x0600}:
267
+ raise LegacyOfficeMalformedError(f"unsupported BIFF version: {version!r}")
268
+ globals_ = _Globals(biff8=version == 0x0600)
269
+ raw_formats: dict[int, str] = {}
270
+ raw_xfs: list[tuple[int, int]] = []
271
+ drawing_chunks: list[bytes] = []
272
+ internal_supbooks: list[bool] = []
273
+ extern_payloads: list[bytes] = []
274
+ cursor = first.next_offset
275
+ depth = 1
276
+ while cursor < len(data):
277
+ record = record_at(data, cursor, budget=budget)
278
+ if record is None:
279
+ logger.warning("XLS_GLOBALS_TRUNCATED: globals end at byte {}", cursor)
280
+ break
281
+ cursor = record.next_offset
282
+ if record.record_type == BOF:
283
+ depth += 1
284
+ continue
285
+ if record.record_type == EOF:
286
+ depth -= 1
287
+ if depth == 0:
288
+ break
289
+ continue
290
+ if depth != 1:
291
+ continue
292
+ if record.record_type == FILEPASS:
293
+ raise LegacyOfficeEncryptedError("password-protected XLS is not supported")
294
+ if record.record_type == CODEPAGE:
295
+ globals_.encoding = codepage_name(int(get_u16(record.payload, 0) or 1252))
296
+ elif record.record_type == DATEMODE:
297
+ globals_.date1904 = get_u16(record.payload, 0) == 1
298
+ elif record.record_type == FONT:
299
+ globals_.fonts.append(_read_font(record.payload, biff8=globals_.biff8))
300
+ elif record.record_type == BOUNDSHEET:
301
+ sheet = _read_boundsheet(record.payload, globals_)
302
+ if sheet is not None and sheet.sheet_type != 0x06:
303
+ globals_.sheets.append(sheet)
304
+ elif record.record_type == FORMAT:
305
+ format_id = get_u16(record.payload, 0)
306
+ if format_id is not None:
307
+ reader = SegmentReader([record.payload[2:]])
308
+ decoded = globals_.read_string(reader, short=not globals_.biff8)
309
+ if decoded is not None:
310
+ raw_formats[int(format_id)] = decoded.text
311
+ elif record.record_type == XF:
312
+ font_index = int(get_u16(record.payload, 0) or 0)
313
+ format_id = int(get_u16(record.payload, 2) or 0)
314
+ raw_xfs.append((font_index, format_id))
315
+ elif record.record_type == SST and globals_.biff8:
316
+ segments, cursor = collect_continues(data, record, budget=budget)
317
+ globals_.strings = _read_sst(segments)
318
+ elif record.record_type == MSODRAWINGGROUP:
319
+ segments, cursor = collect_continues(data, record, budget=budget)
320
+ drawing_chunks.extend(segments)
321
+ elif record.record_type == SUPBOOK:
322
+ internal_supbooks.append(_read_supbook(record.payload))
323
+ elif record.record_type == EXTERNSHEET:
324
+ extern_payloads.append(record.payload)
325
+ elif record.record_type == WINDOW1 and len(record.payload) >= 12:
326
+ globals_.active_sheet_index = get_u16(record.payload, 10)
327
+
328
+ globals_.formats = [
329
+ _CellFormat(
330
+ font_index=font_index,
331
+ format_code=raw_formats.get(format_id) or builtin_number_format(format_id),
332
+ )
333
+ for font_index, format_id in raw_xfs
334
+ ]
335
+ for payload in extern_payloads:
336
+ globals_.extern_sheets.extend(_read_extern_sheets(payload, internal_supbooks))
337
+ if drawing_chunks:
338
+ globals_.images = decode_bstore(b"".join(drawing_chunks), charge=budget.charge)
339
+ return globals_
340
+
341
+
342
+ def _cell_ref(payload: bytes) -> tuple[int, int, int] | None:
343
+ """读取 cell header,并拒绝超出 BIFF8 网格的列。"""
344
+
345
+ if len(payload) < 6:
346
+ return None
347
+ row, col, xf_index = struct.unpack_from("<3H", payload, 0)
348
+ if row >= MAX_ROWS or col >= MAX_COLS:
349
+ return None
350
+ return int(row), int(col), int(xf_index)
351
+
352
+
353
+ def _resolved_rich_text(
354
+ decoded: DecodedString,
355
+ globals_: _Globals,
356
+ xf_index: int,
357
+ ) -> XlsRichText:
358
+ """应用 text number format,并在原文未变时保留 rich runs。"""
359
+
360
+ cell_format = globals_.cell_format(xf_index)
361
+ formatted = format_text(decoded.text, cell_format.format_code)
362
+ if formatted != decoded.text:
363
+ return XlsRichText(formatted)
364
+ return to_rich_text(decoded, globals_.fonts)
365
+
366
+
367
+ def _put_cell(
368
+ sheet: XlsSheet,
369
+ row: int,
370
+ col: int,
371
+ value: XlsRichText,
372
+ ) -> None:
373
+ """仅保存非空文本,并覆盖同坐标较早的缓存记录。"""
374
+
375
+ if not value.text:
376
+ return
377
+ sheet.cells[(row, col)] = XlsCell(row=row, col=col, value=value)
378
+
379
+
380
+ def _append_cell_text(
381
+ sheet: XlsSheet,
382
+ row: int,
383
+ col: int,
384
+ value: XlsRichText,
385
+ ) -> None:
386
+ """把 drawing/control 文本追加到 anchor 单元格且平移 rich runs。"""
387
+
388
+ if not value.text:
389
+ return
390
+ existing = sheet.cells.get((row, col))
391
+ if existing is None:
392
+ _put_cell(sheet, row, col, value)
393
+ return
394
+ separator = "\n" if existing.value.text else ""
395
+ shift = len(existing.value.text) + len(separator)
396
+ shifted = tuple(
397
+ XlsRichRun(
398
+ start=run.start + shift,
399
+ end=run.end + shift,
400
+ style=run.style,
401
+ )
402
+ for run in value.runs
403
+ )
404
+ existing.value = XlsRichText(
405
+ text=existing.value.text + separator + value.text,
406
+ runs=existing.value.runs + shifted,
407
+ )
408
+
409
+
410
+ def _read_label_string(
411
+ segments: list[bytes],
412
+ globals_: _Globals,
413
+ *,
414
+ rich_record: bool,
415
+ ) -> DecodedString | None:
416
+ """读取 LABEL/RSTRING 的字符串并恢复 RSTRING formatting runs。"""
417
+
418
+ if not segments or len(segments[0]) < 6:
419
+ return None
420
+ adjusted = [segments[0][6:], *segments[1:]]
421
+ reader = SegmentReader(adjusted)
422
+ decoded = globals_.read_string(reader, short=False)
423
+ if decoded is None or not rich_record:
424
+ return decoded
425
+ run_count = reader.u16()
426
+ if run_count is None:
427
+ return decoded
428
+ starts: list[tuple[int, int]] = []
429
+ for _ in range(run_count):
430
+ raw = reader.read_across(4)
431
+ if raw is None:
432
+ break
433
+ character_index, font_index = struct.unpack("<HH", raw)
434
+ starts.append((int(character_index), int(font_index)))
435
+ return DecodedString(decoded.text, tuple(starts))
436
+
437
+
438
+ def _pict_embedding_storage(payload: bytes, picture_flags: int | None) -> str | None:
439
+ """从 FtPictFmla 读取嵌入对象的 MBD storage 名称。"""
440
+
441
+ if picture_flags is None:
442
+ return None
443
+ # DDE、ActiveX、controls stream 与 camera picture 都不是内嵌公式 OLE 对象。
444
+ if picture_flags & (0x0002 | 0x0010 | 0x0020 | 0x0080):
445
+ return None
446
+ if len(payload) < 10:
447
+ return None
448
+ cb_fmla = int(get_u16(payload, 0) or 0)
449
+ formula_end = 2 + cb_fmla
450
+ if cb_fmla <= 0 or cb_fmla % 2 or formula_end + 4 > len(payload):
451
+ return None
452
+ formula = payload[2:formula_end]
453
+ if len(formula) < 7 or int(get_u16(formula, 0) or 0) & 0x7FFF != 5:
454
+ return None
455
+ # ObjectParsedFormula 的四字节 unused 在部分生产器中省略,因此兼容两个合法落点。
456
+ if not any(offset + 5 <= len(formula) and formula[offset] == 0x02 for offset in (6, 2)):
457
+ return None
458
+ location = get_u32(payload, formula_end)
459
+ return f"MBD{int(location):08X}" if location is not None else None
460
+
461
+
462
+ def _read_obj(payload: bytes) -> _SheetObject | None:
463
+ """解析 OBJ subrecords 中的对象类型、id、状态与嵌入 storage。"""
464
+
465
+ cursor = 0
466
+ object_type: int | None = None
467
+ object_id = 0
468
+ checked: bool | None = None
469
+ picture_flags: int | None = None
470
+ picture_formula: bytes | None = None
471
+ while cursor + 4 <= len(payload):
472
+ sub_type, length = struct.unpack_from("<HH", payload, cursor)
473
+ data_start = cursor + 4
474
+ data_end = data_start + int(length)
475
+ if data_end > len(payload):
476
+ break
477
+ body = payload[data_start:data_end]
478
+ if sub_type == 0x0015 and len(body) >= 4:
479
+ object_type, object_id = struct.unpack_from("<HH", body, 0)
480
+ elif sub_type == 0x0012 and len(body) >= 2:
481
+ state = int(get_u16(body, 0) or 0)
482
+ checked = state == 1 if state in {0, 1} else None
483
+ elif sub_type == 0x0008 and len(body) >= 2:
484
+ picture_flags = int(get_u16(body, 0) or 0)
485
+ elif sub_type == 0x0009:
486
+ picture_formula = body
487
+ if sub_type == 0:
488
+ break
489
+ cursor = data_end
490
+ if object_type is None:
491
+ return None
492
+ return _SheetObject(
493
+ int(object_type),
494
+ int(object_id),
495
+ checked=checked,
496
+ embedding_storage=(_pict_embedding_storage(picture_formula, picture_flags) if picture_formula is not None else None),
497
+ )
498
+
499
+
500
+ def _read_hyperlink_unicode(payload: bytes, cursor: int) -> tuple[str | None, int]:
501
+ """读取 Hyperlink Object 中含末尾 NUL 的 UTF-16 字符串。"""
502
+
503
+ if cursor + 4 > len(payload):
504
+ return None, len(payload)
505
+ character_count = int(struct.unpack_from("<I", payload, cursor)[0])
506
+ cursor += 4
507
+ byte_count = character_count * 2
508
+ if byte_count < 0 or cursor + byte_count > len(payload):
509
+ return None, len(payload)
510
+ text = payload[cursor : cursor + byte_count].decode("utf-16le", "replace").rstrip("\x00")
511
+ return clean_text(text), cursor + byte_count
512
+
513
+
514
+ def _read_url_moniker(payload: bytes, cursor: int) -> tuple[str | None, int]:
515
+ """读取 URL Moniker 的 UTF-16 URL,忽略可选尾部元数据。"""
516
+
517
+ if cursor + 4 > len(payload):
518
+ return None, len(payload)
519
+ byte_count = int(struct.unpack_from("<I", payload, cursor)[0])
520
+ cursor += 4
521
+ if byte_count < 0 or cursor + byte_count > len(payload):
522
+ return None, len(payload)
523
+ raw = payload[cursor : cursor + byte_count]
524
+ usable = raw[: len(raw) - (len(raw) % 2)]
525
+ text = usable.decode("utf-16le", "replace").split("\x00", 1)[0]
526
+ return clean_text(text), cursor + byte_count
527
+
528
+
529
+ def _read_file_moniker(payload: bytes, cursor: int) -> tuple[str | None, int]:
530
+ """尽力读取 File Moniker 的 ANSI 或 Unicode 路径。"""
531
+
532
+ if cursor + 6 > len(payload):
533
+ return None, len(payload)
534
+ anti_count = int(struct.unpack_from("<H", payload, cursor)[0])
535
+ ansi_length = int(struct.unpack_from("<I", payload, cursor + 2)[0])
536
+ cursor += 6
537
+ if cursor + ansi_length > len(payload):
538
+ return None, len(payload)
539
+ ansi = payload[cursor : cursor + ansi_length].split(b"\x00", 1)[0]
540
+ cursor += ansi_length
541
+ path = ("../" * anti_count) + ansi.decode("cp1252", "replace")
542
+ return clean_text(path), cursor
543
+
544
+
545
+ def _read_hyperlink_target(payload: bytes) -> str | None:
546
+ """解析 HLink 中的 Hyperlink Object 并返回经过白名单过滤的目标。"""
547
+
548
+ if len(payload) < 32:
549
+ return None
550
+ cursor = 24
551
+ version, flags = struct.unpack_from("<II", payload, cursor)
552
+ cursor += 8
553
+ if version != 2:
554
+ return None
555
+ if flags & 0x10:
556
+ _, cursor = _read_hyperlink_unicode(payload, cursor)
557
+ if flags & 0x80:
558
+ _, cursor = _read_hyperlink_unicode(payload, cursor)
559
+ target: str | None = None
560
+ blocked_local_file = False
561
+ if flags & 0x01:
562
+ if flags & 0x0100:
563
+ target, cursor = _read_hyperlink_unicode(payload, cursor)
564
+ elif cursor + 16 <= len(payload):
565
+ moniker = uuid.UUID(bytes_le=payload[cursor : cursor + 16])
566
+ cursor += 16
567
+ if moniker == uuid.UUID("79eac9e0-baf9-11ce-8c82-00aa004ba90b"):
568
+ target, cursor = _read_url_moniker(payload, cursor)
569
+ elif moniker == uuid.UUID("00000303-0000-0000-c000-000000000046"):
570
+ _, cursor = _read_file_moniker(payload, cursor)
571
+ blocked_local_file = True
572
+ location: str | None = None
573
+ if flags & 0x08:
574
+ location, cursor = _read_hyperlink_unicode(payload, cursor)
575
+ if location:
576
+ target = f"{target}#{location}" if target else f"#{location}"
577
+ if blocked_local_file:
578
+ logger.warning("XLS_HYPERLINK_BLOCKED: local File Moniker")
579
+ return None
580
+ sanitized = sanitize_hyperlink_target(
581
+ target,
582
+ allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
583
+ allow_relative=True,
584
+ allow_fragment=True,
585
+ )
586
+ if target and sanitized is None:
587
+ logger.warning("XLS_HYPERLINK_BLOCKED: target={!r}", target)
588
+ return sanitized
589
+
590
+
591
+ def _apply_hlink(
592
+ sheet: XlsSheet,
593
+ payload: bytes,
594
+ pending: dict[tuple[int, int], str],
595
+ ) -> None:
596
+ """把 HLink 范围目标暂存到所有覆盖单元格。"""
597
+
598
+ if len(payload) < 8:
599
+ return
600
+ row_first, row_last, col_first, col_last = struct.unpack_from("<4H", payload, 0)
601
+ target = _read_hyperlink_target(payload)
602
+ if target is None:
603
+ return
604
+ for row in range(min(row_first, row_last), min(max(row_first, row_last), MAX_ROWS - 1) + 1):
605
+ for col in range(min(col_first, col_last), min(max(col_first, col_last), MAX_COLS - 1) + 1):
606
+ pending[(int(row), int(col))] = target
607
+
608
+
609
+ def _shape_anchor(shape: OfficeArtShape | None) -> tuple[int, int] | None:
610
+ """返回 shape 左上角 cell anchor。"""
611
+
612
+ if shape is None or shape.anchor is None:
613
+ return None
614
+ return shape.anchor[0], shape.anchor[1]
615
+
616
+
617
+ def _serialize_payload(payload: OfficeImagePayload) -> str | None:
618
+ """使用共享 Office 图片策略序列化 BLIP。"""
619
+
620
+ return serialize_office_image(
621
+ payload.data,
622
+ part_name=f"picture.{payload.extension}",
623
+ content_type=payload.content_type,
624
+ render_size_emu=payload.render_size_emu,
625
+ )
626
+
627
+
628
+ def _bind_objects(
629
+ sheet: XlsSheet,
630
+ objects: list[_SheetObject],
631
+ drawing_data: bytes,
632
+ chart_streams: list[list[BiffRecord]],
633
+ *,
634
+ globals_: _Globals,
635
+ sheet_index: int,
636
+ native_equations: dict[str, str],
637
+ image_equation_decoder: OfficeImageEquationDecoder,
638
+ budget: RecordBudget,
639
+ ) -> None:
640
+ """按 drawing/OBJ 顺序绑定文本框、复选框、图片与嵌入图表。"""
641
+
642
+ shapes = extract_excel_shapes(drawing_data, charge=budget.charge) if drawing_data else []
643
+ if len(shapes) != len(objects):
644
+ logger.warning(
645
+ "XLS_DRAWING_OBJECT_MISMATCH: sheet={!r}, shapes={}, objects={}",
646
+ sheet.name,
647
+ len(shapes),
648
+ len(objects),
649
+ )
650
+ for object_, shape in zip(objects, shapes, strict=False):
651
+ object_.shape = shape
652
+ chart_objects = [object_ for object_ in objects if object_.object_type == OBJ_CHART]
653
+ for object_ in objects:
654
+ shape = object_.shape
655
+ if shape is None or shape.hidden:
656
+ continue
657
+ anchor = _shape_anchor(shape)
658
+ if anchor is None:
659
+ continue
660
+ row, col = anchor
661
+ equation = (
662
+ native_equations.get(object_.embedding_storage.casefold())
663
+ if object_.object_type == OBJ_PICTURE and object_.embedding_storage
664
+ else None
665
+ )
666
+ if equation:
667
+ sheet.equations.append(XlsEquation(row=row, col=col, latex=equation))
668
+ continue
669
+ if (
670
+ object_.object_type == OBJ_PICTURE
671
+ and shape.pib is not None
672
+ and (payload := globals_.images.get(int(shape.pib))) is not None
673
+ ):
674
+ image_latex = image_equation_decoder.decode(
675
+ payload.data,
676
+ part_name=f"picture.{payload.extension}",
677
+ content_type=payload.content_type,
678
+ )
679
+ if image_latex:
680
+ sheet.equations.append(XlsEquation(row=row, col=col, latex=image_latex))
681
+ continue
682
+ if object_.object_type == OBJ_TEXTBOX and object_.text is not None:
683
+ _append_cell_text(sheet, row, col, object_.text)
684
+ elif object_.object_type == OBJ_CHECKBOX and object_.checked is not None:
685
+ marker = "[x]" if object_.checked else "[ ]"
686
+ caption = object_.text.text.strip() if object_.text is not None else ""
687
+ _append_cell_text(sheet, row, col, XlsRichText(f"{marker} {caption}".rstrip()))
688
+ if object_.object_type in {OBJ_PICTURE, OBJ_CHART} or shape.pib is not None:
689
+ payload = globals_.images.get(int(shape.pib or 0))
690
+ if payload is not None:
691
+ image_base64 = _serialize_payload(payload)
692
+ if image_base64 and object_.object_type != OBJ_CHART:
693
+ sheet.images.append(XlsImage(row=row, col=col, image_base64=image_base64))
694
+
695
+ for index, object_ in enumerate(chart_objects):
696
+ if object_.shape is None or object_.shape.hidden:
697
+ continue
698
+ anchor = _shape_anchor(object_.shape)
699
+ if anchor is None:
700
+ continue
701
+ axes = (
702
+ chart_source_axes(
703
+ chart_streams[index],
704
+ current_sheet_index=sheet_index,
705
+ extern_sheets=globals_.extern_sheets,
706
+ )
707
+ if index < len(chart_streams)
708
+ else None
709
+ )
710
+ preview: str | None = None
711
+ if object_.shape.pib is not None:
712
+ payload = globals_.images.get(object_.shape.pib)
713
+ preview = _serialize_payload(payload) if payload is not None else None
714
+ if axes is None:
715
+ if preview:
716
+ sheet.charts.append(
717
+ XlsChart(
718
+ row=anchor[0],
719
+ col=anchor[1],
720
+ source_rows=(),
721
+ source_cols=(),
722
+ image_base64=preview,
723
+ )
724
+ )
725
+ else:
726
+ logger.warning(
727
+ "XLS_CHART_SOURCE_UNSUPPORTED: sheet={!r}, object_id={}",
728
+ sheet.name,
729
+ object_.object_id,
730
+ )
731
+ continue
732
+ rows, cols = axes
733
+ sheet.charts.append(
734
+ XlsChart(
735
+ row=anchor[0],
736
+ col=anchor[1],
737
+ source_rows=tuple(rows),
738
+ source_cols=tuple(cols),
739
+ image_base64=preview,
740
+ )
741
+ )
742
+
743
+
744
+ def _read_sheet(
745
+ data: bytes,
746
+ globals_: _Globals,
747
+ descriptor: _BoundSheet,
748
+ offset: int,
749
+ *,
750
+ sheet_index: int,
751
+ recovered: bool,
752
+ native_equations: dict[str, str],
753
+ image_equation_decoder: OfficeImageEquationDecoder,
754
+ budget: RecordBudget,
755
+ ) -> XlsSheet | None:
756
+ """解析一个 worksheet substream 并绑定其 drawing/chart 对象。"""
757
+
758
+ first = record_at(data, offset, budget=budget)
759
+ if first is None or first.record_type != BOF or get_u16(first.payload, 2) != WORKSHEET_SUBSTREAM:
760
+ return None
761
+ sheet = XlsSheet(
762
+ name=descriptor.name,
763
+ visible=descriptor.visible,
764
+ order=sheet_index,
765
+ recovered=recovered,
766
+ )
767
+ cursor = first.next_offset
768
+ depth = 1
769
+ active_chart: list[BiffRecord] | None = None
770
+ chart_streams: list[list[BiffRecord]] = []
771
+ drawing_chunks: list[bytes] = []
772
+ objects: list[_SheetObject] = []
773
+ pending_formula: tuple[int, int, int] | None = None
774
+ pending_links: dict[tuple[int, int], str] = {}
775
+
776
+ while cursor < len(data):
777
+ record = record_at(data, cursor, budget=budget)
778
+ if record is None:
779
+ logger.warning(
780
+ "XLS_SHEET_TRUNCATED: sheet={!r}, byte={}",
781
+ sheet.name,
782
+ cursor,
783
+ )
784
+ break
785
+ cursor = record.next_offset
786
+ if active_chart is not None:
787
+ active_chart.append(record)
788
+ if record.record_type == BOF:
789
+ depth += 1
790
+ elif record.record_type == EOF:
791
+ depth -= 1
792
+ if depth == 1:
793
+ chart_streams.append(active_chart)
794
+ active_chart = None
795
+ continue
796
+ if record.record_type == BOF:
797
+ depth += 1
798
+ if depth == 2 and get_u16(record.payload, 2) == CHART_SUBSTREAM:
799
+ active_chart = [record]
800
+ continue
801
+ if record.record_type == EOF:
802
+ depth -= 1
803
+ if depth == 0:
804
+ break
805
+ continue
806
+ if depth != 1:
807
+ continue
808
+
809
+ if record.record_type == MERGEDCELLS:
810
+ count = min(int(get_u16(record.payload, 0) or 0), max(0, (len(record.payload) - 2) // 8))
811
+ for index in range(count):
812
+ row_first, row_last, col_first, col_last = struct.unpack_from("<4H", record.payload, 2 + index * 8)
813
+ row_start, row_end = sorted((int(row_first), int(row_last)))
814
+ col_start, col_end = sorted((int(col_first), int(col_last)))
815
+ if col_start >= MAX_COLS or (row_start == row_end and col_start == col_end):
816
+ continue
817
+ sheet.merges.append(
818
+ (
819
+ row_start,
820
+ col_start,
821
+ min(row_end, MAX_ROWS - 1),
822
+ min(col_end, MAX_COLS - 1),
823
+ )
824
+ )
825
+ elif record.record_type == LABELSST:
826
+ reference = _cell_ref(record.payload)
827
+ string_index = get_u32(record.payload, 6)
828
+ if reference is not None and string_index is not None and string_index < len(globals_.strings):
829
+ row, col, xf_index = reference
830
+ _put_cell(
831
+ sheet,
832
+ row,
833
+ col,
834
+ _resolved_rich_text(globals_.strings[int(string_index)], globals_, xf_index),
835
+ )
836
+ elif record.record_type in {LABEL, RSTRING}:
837
+ segments, cursor = collect_continues(data, record, budget=budget)
838
+ reference = _cell_ref(record.payload)
839
+ decoded = _read_label_string(
840
+ segments,
841
+ globals_,
842
+ rich_record=record.record_type == RSTRING,
843
+ )
844
+ if reference is not None and decoded is not None:
845
+ row, col, xf_index = reference
846
+ _put_cell(sheet, row, col, _resolved_rich_text(decoded, globals_, xf_index))
847
+ elif record.record_type == NUMBER:
848
+ reference = _cell_ref(record.payload)
849
+ value = get_f64(record.payload, 6)
850
+ if reference is not None and value is not None:
851
+ row, col, xf_index = reference
852
+ cell_format = globals_.cell_format(xf_index)
853
+ _put_cell(
854
+ sheet,
855
+ row,
856
+ col,
857
+ XlsRichText(
858
+ format_number(
859
+ value,
860
+ cell_format.format_code,
861
+ date1904=globals_.date1904,
862
+ )
863
+ ),
864
+ )
865
+ elif record.record_type == RK:
866
+ reference = _cell_ref(record.payload)
867
+ raw_value = get_u32(record.payload, 6)
868
+ if reference is not None and raw_value is not None:
869
+ row, col, xf_index = reference
870
+ cell_format = globals_.cell_format(xf_index)
871
+ _put_cell(
872
+ sheet,
873
+ row,
874
+ col,
875
+ XlsRichText(
876
+ format_number(
877
+ _rk_number(raw_value),
878
+ cell_format.format_code,
879
+ date1904=globals_.date1904,
880
+ )
881
+ ),
882
+ )
883
+ elif record.record_type == MULRK and len(record.payload) >= 6:
884
+ row = int(get_u16(record.payload, 0) or 0)
885
+ first_col = int(get_u16(record.payload, 2) or 0)
886
+ pair_count = max(0, (len(record.payload) - 6) // 6)
887
+ for index in range(pair_count):
888
+ xf_index = int(get_u16(record.payload, 4 + index * 6) or 0)
889
+ raw_value = get_u32(record.payload, 6 + index * 6)
890
+ col = first_col + index
891
+ if raw_value is None or col >= MAX_COLS:
892
+ break
893
+ cell_format = globals_.cell_format(xf_index)
894
+ _put_cell(
895
+ sheet,
896
+ row,
897
+ col,
898
+ XlsRichText(
899
+ format_number(
900
+ _rk_number(raw_value),
901
+ cell_format.format_code,
902
+ date1904=globals_.date1904,
903
+ )
904
+ ),
905
+ )
906
+ elif record.record_type == BOOLERR:
907
+ reference = _cell_ref(record.payload)
908
+ if reference is not None and len(record.payload) >= 8:
909
+ row, col, _ = reference
910
+ value, is_error = record.payload[6], record.payload[7]
911
+ text = _error_literal(value) if is_error == 1 else ("TRUE" if value else "FALSE")
912
+ if text:
913
+ _put_cell(sheet, row, col, XlsRichText(text))
914
+ elif record.record_type == FORMULA:
915
+ reference = _cell_ref(record.payload)
916
+ if reference is not None and len(record.payload) >= 14:
917
+ row, col, xf_index = reference
918
+ cached = record.payload[6:14]
919
+ if cached[6:8] == b"\xff\xff":
920
+ kind = cached[0]
921
+ if kind == 0x00:
922
+ pending_formula = (row, col, xf_index)
923
+ elif kind == 0x01:
924
+ _put_cell(sheet, row, col, XlsRichText("TRUE" if cached[2] else "FALSE"))
925
+ elif kind == 0x02:
926
+ error = _error_literal(cached[2])
927
+ if error:
928
+ _put_cell(sheet, row, col, XlsRichText(error))
929
+ else:
930
+ value = get_f64(record.payload, 6)
931
+ if value is not None:
932
+ cell_format = globals_.cell_format(xf_index)
933
+ _put_cell(
934
+ sheet,
935
+ row,
936
+ col,
937
+ XlsRichText(
938
+ format_number(
939
+ value,
940
+ cell_format.format_code,
941
+ date1904=globals_.date1904,
942
+ )
943
+ ),
944
+ )
945
+ elif record.record_type == STRING:
946
+ segments, cursor = collect_continues(data, record, budget=budget)
947
+ if pending_formula is not None:
948
+ reader = SegmentReader(segments)
949
+ decoded = globals_.read_string(reader, short=False)
950
+ if decoded is not None:
951
+ if decoded.text.lstrip().upper().startswith(("=DISPIMG(", "=_XLFN.DISPIMG(")):
952
+ # 旧版文件无法携带现代 DISPIMG 计算语义,按 Office 回存结果稳定降级。
953
+ decoded = DecodedString("#NAME?")
954
+ row, col, xf_index = pending_formula
955
+ _put_cell(sheet, row, col, _resolved_rich_text(decoded, globals_, xf_index))
956
+ pending_formula = None
957
+ elif record.record_type == MSODRAWING:
958
+ segments, cursor = collect_continues(data, record, budget=budget)
959
+ drawing_chunks.extend(segments)
960
+ elif record.record_type == OBJ:
961
+ object_ = _read_obj(record.payload)
962
+ if object_ is not None:
963
+ objects.append(object_)
964
+ elif record.record_type == TXO:
965
+ segments, cursor = collect_continues(data, record, budget=budget)
966
+ if objects:
967
+ objects[-1].text = read_txo_text(
968
+ record.payload,
969
+ segments[1:],
970
+ globals_.fonts,
971
+ )
972
+ elif record.record_type == HLINK:
973
+ _apply_hlink(sheet, record.payload, pending_links)
974
+ elif record.record_type in {ROW, COLINFO, CONTINUE}:
975
+ # 用户明确要求隐藏行列中的内容仍参与表格重建。
976
+ pass
977
+
978
+ for coordinate, target in pending_links.items():
979
+ cell = sheet.cells.get(coordinate)
980
+ if cell is not None:
981
+ cell.hyperlink = target
982
+ _bind_objects(
983
+ sheet,
984
+ objects,
985
+ b"".join(drawing_chunks),
986
+ chart_streams,
987
+ globals_=globals_,
988
+ sheet_index=sheet_index,
989
+ native_equations=native_equations,
990
+ image_equation_decoder=image_equation_decoder,
991
+ budget=budget,
992
+ )
993
+ return sheet
994
+
995
+
996
+ def _worksheet_bof_offsets(data: bytes) -> list[int]:
997
+ """扫描所有可识别 worksheet BOF 偏移,供坏目录恢复使用。"""
998
+
999
+ return [
1000
+ record.offset
1001
+ for record in iter_records(data)
1002
+ if record.record_type == BOF and get_u16(record.payload, 2) == WORKSHEET_SUBSTREAM
1003
+ ]
1004
+
1005
+
1006
+ def _is_worksheet_offset(data: bytes, offset: int) -> bool:
1007
+ """判断 BoundSheet offset 是否精确指向 worksheet BOF。"""
1008
+
1009
+ record = record_at(data, offset)
1010
+ return bool(record is not None and record.record_type == BOF and get_u16(record.payload, 2) == WORKSHEET_SUBSTREAM)
1011
+
1012
+
1013
+ def _parse_chart_sheets(
1014
+ data: bytes,
1015
+ globals_: _Globals,
1016
+ *,
1017
+ budget: RecordBudget,
1018
+ ) -> list[XlsChartSheet]:
1019
+ """解析独立 chart sheet,并把 BRAI 引用绑定到唯一 worksheet。"""
1020
+
1021
+ worksheet_names = {
1022
+ index: descriptor.name for index, descriptor in enumerate(globals_.sheets) if descriptor.sheet_type == 0x00
1023
+ }
1024
+ chart_sheets: list[XlsChartSheet] = []
1025
+ for order, descriptor in enumerate(globals_.sheets):
1026
+ if descriptor.sheet_type != 0x02:
1027
+ continue
1028
+ first = record_at(data, descriptor.offset, budget=budget)
1029
+ selection = None
1030
+ if first is not None and first.record_type == BOF and get_u16(first.payload, 2) == CHART_SUBSTREAM:
1031
+ records = list(
1032
+ iter_records(
1033
+ data,
1034
+ start=descriptor.offset,
1035
+ stop_at_eof=True,
1036
+ budget=budget,
1037
+ )
1038
+ )
1039
+ selection = chart_source_selection(
1040
+ records,
1041
+ current_sheet_index=order,
1042
+ extern_sheets=globals_.extern_sheets,
1043
+ )
1044
+ source_name = worksheet_names.get(selection.sheet_index) if selection is not None else None
1045
+ chart_sheets.append(
1046
+ XlsChartSheet(
1047
+ name=descriptor.name,
1048
+ visible=descriptor.visible,
1049
+ order=order,
1050
+ source_sheet_name=source_name,
1051
+ source_rows=(selection.rows if selection is not None and source_name is not None else ()),
1052
+ source_cols=(selection.cols if selection is not None and source_name is not None else ()),
1053
+ )
1054
+ )
1055
+ return chart_sheets
1056
+
1057
+
1058
+ def parse_xls_workbook(
1059
+ data: bytes,
1060
+ *,
1061
+ native_equations: dict[str, str] | None = None,
1062
+ ) -> XlsWorkbook:
1063
+ """解析 Workbook/Book stream,并按目录顺序恢复 worksheets 与原生公式。"""
1064
+
1065
+ if not data:
1066
+ raise LegacyOfficeMalformedError("empty Workbook stream")
1067
+ budget = RecordBudget()
1068
+ normalized_equations = {storage.casefold(): latex for storage, latex in (native_equations or {}).items()}
1069
+ image_equation_decoder = OfficeImageEquationDecoder()
1070
+ globals_ = _read_globals(data, budget)
1071
+ candidates = _worksheet_bof_offsets(data)
1072
+ used_offsets: set[int] = set()
1073
+ sheets: list[XlsSheet] = []
1074
+
1075
+ descriptor_entries = [(index, sheet) for index, sheet in enumerate(globals_.sheets) if sheet.sheet_type == 0x00]
1076
+ if not descriptor_entries and candidates:
1077
+ descriptor_entries = [
1078
+ (
1079
+ index - 1,
1080
+ _BoundSheet(
1081
+ name=f"Recovered Sheet {index}",
1082
+ offset=offset,
1083
+ visible=True,
1084
+ sheet_type=0,
1085
+ ),
1086
+ )
1087
+ for index, offset in enumerate(candidates, start=1)
1088
+ ]
1089
+ if not descriptor_entries and not candidates:
1090
+ raise LegacyOfficeMalformedError("workbook contains no worksheet substream")
1091
+
1092
+ for sheet_index, descriptor in descriptor_entries:
1093
+ resolved_offset: int | None = None
1094
+ recovered = False
1095
+ if _is_worksheet_offset(data, descriptor.offset) and descriptor.offset not in used_offsets:
1096
+ resolved_offset = descriptor.offset
1097
+ else:
1098
+ resolved_offset = next((offset for offset in candidates if offset not in used_offsets), None)
1099
+ recovered = resolved_offset is not None
1100
+ if recovered:
1101
+ logger.warning(
1102
+ "XLS_BOUNDSHEET_RECOVERED: sheet={!r}, old_offset={}, new_offset={}",
1103
+ descriptor.name,
1104
+ descriptor.offset,
1105
+ resolved_offset,
1106
+ )
1107
+ if resolved_offset is None:
1108
+ logger.warning("XLS_SHEET_UNREADABLE: keeping empty sheet {!r}", descriptor.name)
1109
+ sheets.append(
1110
+ XlsSheet(
1111
+ name=descriptor.name,
1112
+ visible=descriptor.visible,
1113
+ order=sheet_index,
1114
+ recovered=True,
1115
+ )
1116
+ )
1117
+ continue
1118
+ used_offsets.add(resolved_offset)
1119
+ parsed = _read_sheet(
1120
+ data,
1121
+ globals_,
1122
+ descriptor,
1123
+ resolved_offset,
1124
+ sheet_index=sheet_index,
1125
+ recovered=recovered,
1126
+ native_equations=normalized_equations,
1127
+ image_equation_decoder=image_equation_decoder,
1128
+ budget=budget,
1129
+ )
1130
+ if parsed is None:
1131
+ sheets.append(
1132
+ XlsSheet(
1133
+ name=descriptor.name,
1134
+ visible=descriptor.visible,
1135
+ order=sheet_index,
1136
+ recovered=True,
1137
+ )
1138
+ )
1139
+ else:
1140
+ sheets.append(parsed)
1141
+ return XlsWorkbook(
1142
+ sheets=sheets,
1143
+ chart_sheets=_parse_chart_sheets(data, globals_, budget=budget),
1144
+ active_sheet_index=globals_.active_sheet_index,
1145
+ )