docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,716 @@
1
+ import collections
2
+ import hashlib
3
+ import posixpath
4
+ import re
5
+ import xml.etree.ElementTree as ET
6
+ import zipfile
7
+ from io import BytesIO
8
+ from typing import BinaryIO, cast
9
+
10
+ from loguru import logger
11
+ from openpyxl import load_workbook
12
+ from openpyxl.drawing.image import Image as XlsImage
13
+ from openpyxl.utils.cell import range_to_tuple
14
+ from openpyxl.worksheet.worksheet import Worksheet
15
+ from ..image import serialize_office_image
16
+ from ..equation.image import OfficeImageEquationDecoder
17
+ from ..equation.ooxml import OoxmlEquationDecoder
18
+ from ..errors import LegacyOfficeResourceLimitError
19
+ from ..limits import MAX_ENTRY_BYTES
20
+ from ..equation.omml import oMath2Latex
21
+ from ..streams import read_stream_bytes_from_start, rewind_stream
22
+ from ..spreadsheet.html import EQUATION_BOOKENDS, render_spreadsheet_table
23
+ from ..spreadsheet.models import AnchoredBlock, FormulaMap, SheetImage
24
+ from ..spreadsheet.projector import SpreadsheetProjector
25
+ from .package_normalizer import normalize_xlsx_package, strip_xlsx_ole_objects_for_openpyxl
26
+ from .ooxml_ole import (
27
+ XlsxOleEquationArtifact,
28
+ package_has_sheet_ole_objects,
29
+ read_sheet_image_artifacts,
30
+ read_sheet_equation_artifacts,
31
+ workbook_sheet_parts,
32
+ )
33
+ from .....schema import BlockType
34
+
35
+
36
+ class XlsxConverter(SpreadsheetProjector):
37
+ def __init__(
38
+ self,
39
+ treat_singleton_as_text: bool = True,
40
+ gap_tolerance: int | None = None,
41
+ include_hidden_sheets: bool = False,
42
+ ) -> None:
43
+ super().__init__(
44
+ treat_singleton_as_text=treat_singleton_as_text,
45
+ gap_tolerance=gap_tolerance,
46
+ include_hidden_sheets=include_hidden_sheets,
47
+ )
48
+ self.zf = None
49
+ self.image_map = {}
50
+ self.cell_image_map = {}
51
+ self._sheet_part_by_title: dict[str, str] = {}
52
+ self._ole_artifacts: list[XlsxOleEquationArtifact] = []
53
+ self._omml_shape_ids: set[str] = set()
54
+ self._omml_artifacts: list[tuple[int, int, str, int]] = []
55
+ self._suppressed_ole_previews: set[tuple[tuple[int, int], str]] = set()
56
+ self._ooxml_equation_decoder = OoxmlEquationDecoder()
57
+ self._image_equation_decoder = OfficeImageEquationDecoder()
58
+
59
+ def convert(
60
+ self,
61
+ file_stream: BinaryIO,
62
+ ) -> None:
63
+ if rewind_stream(file_stream):
64
+ try:
65
+ self._convert_package_stream(file_stream)
66
+ return
67
+ except Exception as exc:
68
+ file_bytes = read_stream_bytes_from_start(file_stream)
69
+ self._retry_convert_package_bytes_after_normalization(file_bytes, exc)
70
+ return
71
+
72
+ file_bytes = file_stream.read()
73
+ try:
74
+ self._convert_package_bytes(file_bytes)
75
+ except Exception as exc:
76
+ self._retry_convert_package_bytes_after_normalization(file_bytes, exc)
77
+
78
+ def _reset_state(self) -> None:
79
+ """重置解析状态,确保失败重试时不会残留上一次半解析结果。"""
80
+ if self.zf:
81
+ self.zf.close()
82
+ self._reset_projection_state()
83
+ self.zf = None
84
+ self.image_map = {}
85
+ self.cell_image_map = {}
86
+ self._sheet_part_by_title = {}
87
+ self._ole_artifacts = []
88
+ self._omml_shape_ids = set()
89
+ self._omml_artifacts = []
90
+ self._suppressed_ole_previews = set()
91
+ self._ooxml_equation_decoder = OoxmlEquationDecoder()
92
+ self._image_equation_decoder = OfficeImageEquationDecoder()
93
+
94
+ def _convert_package_bytes(self, file_bytes: bytes) -> None:
95
+ """用独立字节流解析 XLSX 包,便于原始包失败后用规范化包重试。"""
96
+ self._convert_package_stream(BytesIO(file_bytes))
97
+
98
+ def _convert_package_stream(self, file_stream: BinaryIO) -> None:
99
+ """直接使用可复位的 XLSX 流解析正常路径,避免提前复制完整包字节。"""
100
+ self._reset_state()
101
+ try:
102
+ self.zf = zipfile.ZipFile(file_stream)
103
+ self._sheet_part_by_title = workbook_sheet_parts(self.zf)
104
+ except Exception as e:
105
+ logger.warning(f"Failed to open zip file: {e}")
106
+ self.zf = None
107
+
108
+ try:
109
+ workbook_stream: BinaryIO = file_stream
110
+ if self.zf is not None and package_has_sheet_ole_objects(
111
+ self.zf,
112
+ self._sheet_part_by_title,
113
+ ):
114
+ file_bytes = read_stream_bytes_from_start(file_stream)
115
+ workbook_stream = BytesIO(strip_xlsx_ole_objects_for_openpyxl(file_bytes))
116
+ else:
117
+ rewind_stream(file_stream)
118
+ self.workbook = load_workbook(
119
+ filename=workbook_stream,
120
+ data_only=True,
121
+ rich_text=True,
122
+ )
123
+ if self.workbook is not None:
124
+ # 遍历需要参与转换的工作表,避免为隐藏表或尾部空页生成无效页面。
125
+ sheet_pages = []
126
+ for idx, sheet in enumerate(self._iter_sheets_to_convert(), start=1):
127
+ logger.debug(f"正在处理第 {idx} 个工作表:{sheet.title}")
128
+ self.cur_page = []
129
+ self._convert_sheet(sheet)
130
+ sheet_pages.append((sheet.title, self.cur_page))
131
+ if self._should_emit_sheet_titles([page for _, page in sheet_pages]):
132
+ self._prepend_sheet_titles(sheet_pages)
133
+ self.pages.extend(page for _, page in sheet_pages)
134
+ else:
135
+ logger.error("工作簿未初始化。")
136
+ finally:
137
+ if self.zf:
138
+ self.zf.close()
139
+ self.zf = None
140
+
141
+ def _retry_convert_package_bytes_after_normalization(
142
+ self,
143
+ file_bytes: bytes,
144
+ exc: Exception,
145
+ ) -> None:
146
+ """首次解析失败后,仅在包规范化确实产生变化时使用规范化字节重试。"""
147
+ normalized_bytes = normalize_xlsx_package(file_bytes)
148
+ if normalized_bytes == file_bytes:
149
+ raise exc
150
+ logger.warning(f"Retrying XLSX parsing after package normalization: {exc}")
151
+ self._convert_package_bytes(normalized_bytes)
152
+
153
+ def _prepare_sheet_assets(self, sheet: Worksheet) -> None:
154
+ """准备 XLSX 公式、图片与 OLE 素材,并保持既有预览抑制优先级。"""
155
+ self.math_map = self._map_math_formulas_to_cells(sheet)
156
+ self._ole_artifacts = self._read_ole_equation_artifacts(sheet)
157
+ self._suppressed_ole_previews = {
158
+ ((artifact.row, artifact.col), artifact.preview_base64)
159
+ for artifact in self._ole_artifacts
160
+ if artifact.shape_id in self._omml_shape_ids
161
+ and artifact.row is not None
162
+ and artifact.col is not None
163
+ and artifact.preview_base64 is not None
164
+ }
165
+ self._ole_artifacts = [artifact for artifact in self._ole_artifacts if artifact.shape_id not in self._omml_shape_ids]
166
+ for artifact in self._ole_artifacts:
167
+ if artifact.latex is None or artifact.row is None or artifact.col is None:
168
+ continue
169
+ self.math_map.setdefault((artifact.row, artifact.col), []).append(artifact.latex)
170
+
171
+ self.sheet_images = self._collect_sheet_images(sheet)
172
+ ole_previews = self._suppressed_ole_previews | {
173
+ ((artifact.row, artifact.col), artifact.preview_base64)
174
+ for artifact in self._ole_artifacts
175
+ if artifact.row is not None and artifact.col is not None and artifact.preview_base64
176
+ }
177
+ self.sheet_images = [image for image in self.sheet_images if (image.anchor, image.image_base64) not in ole_previews]
178
+ self.table_image_map = collections.defaultdict(list)
179
+ for image in self.sheet_images:
180
+ row, col = image.anchor
181
+ if row is None or col is None:
182
+ continue
183
+ if image.latex:
184
+ self.table_image_map[(row, col)].append(EQUATION_BOOKENDS.format(EQ=image.latex))
185
+ elif image.image_base64:
186
+ self.table_image_map[(row, col)].append(f'<img src="{image.image_base64}" />')
187
+ for artifact in self._ole_artifacts:
188
+ if artifact.latex is not None or artifact.preview_base64 is None or artifact.row is None or artifact.col is None:
189
+ continue
190
+ self.table_image_map[(artifact.row, artifact.col)].append(f'<img src="{artifact.preview_base64}" />')
191
+
192
+ def _find_additional_visual_artifacts(
193
+ self,
194
+ used_cells: set[tuple[int, int]],
195
+ ) -> list[AnchoredBlock]:
196
+ """输出未被表格吸收的 XLSX 公式、OLE 预览和图片公式。"""
197
+ return [
198
+ *self._find_equation_artifacts_in_sheet(used_cells),
199
+ *self._find_image_equation_artifacts_in_sheet(used_cells),
200
+ ]
201
+
202
+ def _read_ole_equation_artifacts(
203
+ self,
204
+ sheet: Worksheet,
205
+ ) -> list[XlsxOleEquationArtifact]:
206
+ """从当前 worksheet part 读取 MathType/Equation 公式和预览。"""
207
+
208
+ if self.zf is None:
209
+ return []
210
+ worksheet_part = self._sheet_part_by_title.get(sheet.title)
211
+ if worksheet_part is None:
212
+ return []
213
+ return read_sheet_equation_artifacts(
214
+ self.zf,
215
+ worksheet_part,
216
+ self._ooxml_equation_decoder,
217
+ self._image_equation_decoder,
218
+ )
219
+
220
+ def _find_equation_artifacts_in_sheet(
221
+ self,
222
+ used_cells: set[tuple[int, int]],
223
+ ) -> list[tuple[tuple[int, int], int, dict]]:
224
+ """输出未被表格吸收的 OMML/MTEF 公式或缓存预览。"""
225
+
226
+ artifacts: list[tuple[tuple[int, int], int, dict]] = []
227
+ for row, col, latex, order in self._omml_artifacts:
228
+ if (row, col) in used_cells:
229
+ continue
230
+ artifacts.append(
231
+ (
232
+ (row, col),
233
+ 15_000 + order,
234
+ {
235
+ "type": BlockType.EQUATION,
236
+ "content": latex,
237
+ },
238
+ )
239
+ )
240
+ for artifact in self._ole_artifacts:
241
+ coordinate = (
242
+ artifact.row if artifact.row is not None else 10**9,
243
+ artifact.col if artifact.col is not None else 10**9,
244
+ )
245
+ if artifact.row is not None and artifact.col is not None and (artifact.row, artifact.col) in used_cells:
246
+ continue
247
+ block = None
248
+ if artifact.latex is not None:
249
+ block = {
250
+ "type": BlockType.EQUATION,
251
+ "content": artifact.latex,
252
+ }
253
+ elif artifact.preview_base64 is not None:
254
+ block = {
255
+ "type": BlockType.IMAGE,
256
+ "image_base64": artifact.preview_base64,
257
+ }
258
+ if block is not None:
259
+ artifacts.append((coordinate, 20_000 + artifact.order, block))
260
+ return artifacts
261
+
262
+ def _find_image_equation_artifacts_in_sheet(
263
+ self,
264
+ used_cells: set[tuple[int, int]],
265
+ ) -> list[tuple[tuple[int, int], int, dict]]:
266
+ """按原始 drawing anchor 输出未被表格吸收的图片 comment 公式。"""
267
+
268
+ artifacts: list[tuple[tuple[int, int], int, dict]] = []
269
+ for image in self.sheet_images:
270
+ if not image.latex:
271
+ continue
272
+ row, col = image.anchor
273
+ if row is not None and col is not None and (row, col) in used_cells:
274
+ continue
275
+ coordinate = (
276
+ row if row is not None else 10**9,
277
+ col if col is not None else 10**9,
278
+ )
279
+ artifacts.append(
280
+ (
281
+ coordinate,
282
+ 25_000 + image.order,
283
+ {"type": BlockType.EQUATION, "content": image.latex},
284
+ )
285
+ )
286
+ return artifacts
287
+
288
+ def _read_xlsx_image_member(self, part_name: str) -> bytes | None:
289
+ """从原始 XLSX ZIP 有界读取 media member。"""
290
+
291
+ if self.zf is None:
292
+ return None
293
+ normalized = posixpath.normpath(part_name.lstrip("/"))
294
+ if normalized.startswith("../") or normalized not in self.zf.namelist():
295
+ return None
296
+ info = self.zf.getinfo(normalized)
297
+ if info.file_size > MAX_ENTRY_BYTES:
298
+ raise LegacyOfficeResourceLimitError(f"XLSX image exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
299
+ with self.zf.open(info) as stream:
300
+ payload = stream.read(MAX_ENTRY_BYTES + 1)
301
+ if len(payload) > MAX_ENTRY_BYTES:
302
+ raise LegacyOfficeResourceLimitError(f"XLSX image exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
303
+ return payload
304
+
305
+ def _raw_sheet_image(
306
+ self,
307
+ image: XlsImage,
308
+ ) -> tuple[bytes, str | None, str | None] | None:
309
+ """优先从原 ZIP 读取 openpyxl 图片的未转码原始字节。"""
310
+
311
+ part_name = str(getattr(image, "path", "") or "") or None
312
+ payload = self._read_xlsx_image_member(part_name) if part_name is not None else None
313
+ if payload is None:
314
+ try:
315
+ payload = image._data() # type: ignore[attr-defined]
316
+ except Exception:
317
+ return None
318
+ if len(payload) > MAX_ENTRY_BYTES:
319
+ raise LegacyOfficeResourceLimitError(f"XLSX image exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
320
+ image_format = str(getattr(image, "format", "") or "").casefold()
321
+ content_type = f"image/{image_format}" if image_format else None
322
+ return payload, part_name, content_type
323
+
324
+ def _collect_sheet_images(self, sheet: Worksheet) -> list[SheetImage]:
325
+ """读取当前工作表的原始图片并识别图片公式。"""
326
+ images: list[SheetImage] = []
327
+ if self.workbook is None:
328
+ return images
329
+
330
+ seen: set[tuple[tuple[int | None, int | None], bytes]] = set()
331
+ worksheet_part = self._sheet_part_by_title.get(sheet.title)
332
+ if self.zf is not None and worksheet_part is not None:
333
+ for artifact in read_sheet_image_artifacts(
334
+ self.zf,
335
+ worksheet_part,
336
+ ):
337
+ anchor = (artifact.row, artifact.col)
338
+ digest = hashlib.sha256(artifact.payload).digest()
339
+ key = (anchor, digest)
340
+ if key in seen:
341
+ continue
342
+ seen.add(key)
343
+ latex = self._image_equation_decoder.decode(
344
+ artifact.payload,
345
+ part_name=artifact.part_name,
346
+ )
347
+ image_base64 = serialize_office_image(
348
+ artifact.payload,
349
+ part_name=artifact.part_name,
350
+ content_type=None,
351
+ )
352
+ if latex is None and image_base64 is None:
353
+ continue
354
+ images.append(
355
+ SheetImage(
356
+ anchor=anchor,
357
+ image_base64=image_base64,
358
+ latex=latex,
359
+ order=artifact.order,
360
+ )
361
+ )
362
+
363
+ for image_order, item in enumerate(
364
+ getattr(sheet, "_images", []), # type: ignore[attr-defined]
365
+ start=10_000,
366
+ ):
367
+ try:
368
+ image: XlsImage = cast(XlsImage, item)
369
+ raw_image = self._raw_sheet_image(image)
370
+ if raw_image is None:
371
+ continue
372
+ payload, part_name, content_type = raw_image
373
+ anchor = self._get_anchor_pos(item.anchor)
374
+ key = (anchor, hashlib.sha256(payload).digest())
375
+ if key in seen:
376
+ continue
377
+ seen.add(key)
378
+ latex = self._image_equation_decoder.decode(
379
+ payload,
380
+ part_name=part_name,
381
+ content_type=content_type,
382
+ )
383
+ image_base64 = serialize_office_image(
384
+ payload,
385
+ part_name=part_name,
386
+ content_type=content_type,
387
+ )
388
+ if latex is None and image_base64 is None:
389
+ continue
390
+ images.append(
391
+ SheetImage(
392
+ anchor=anchor,
393
+ image_base64=image_base64,
394
+ latex=latex,
395
+ order=image_order,
396
+ )
397
+ )
398
+ except Exception as e:
399
+ logger.error(f"无法从 Excel 工作表中提取图片,错误信息:{e}")
400
+
401
+ return images
402
+
403
+ def _map_math_formulas_to_cells(self, sheet: Worksheet) -> FormulaMap:
404
+ """从 worksheet drawing 恢复按 cell anchor 分组的 OMML 公式。"""
405
+ math_map = collections.defaultdict(list)
406
+ self._omml_shape_ids = set()
407
+ self._omml_artifacts = []
408
+ if not self.zf:
409
+ return math_map
410
+
411
+ # Find drawing relation
412
+ drawing_rel = None
413
+ if hasattr(sheet, "_rels"):
414
+ for rel in sheet._rels:
415
+ if rel.Type.endswith("/relationships/drawing"):
416
+ drawing_rel = rel
417
+ break
418
+
419
+ if not drawing_rel:
420
+ return math_map
421
+
422
+ # Resolve path
423
+ # Assuming relative path from worksheets/sheetX.xml to drawings/drawingY.xml
424
+ # Usually target is like "../drawings/drawing1.xml"
425
+ target = drawing_rel.Target
426
+ if target.startswith("../"):
427
+ path = target.replace("../", "xl/") # simplistic resolution
428
+ elif target.startswith("/"):
429
+ path = target[1:]
430
+ else:
431
+ path = f"xl/worksheets/{target}" # unlikely but default relative
432
+
433
+ # Check if file exists in zip
434
+ if path not in self.zf.namelist():
435
+ # Try generic match if simplistic resolution failed
436
+ # drawing1.xml -> xl/drawings/drawing1.xml
437
+ basename = target.split("/")[-1]
438
+ path = f"xl/drawings/{basename}"
439
+ if path not in self.zf.namelist():
440
+ return math_map
441
+
442
+ try:
443
+ with self.zf.open(path) as f:
444
+ tree = ET.parse(f)
445
+ root = tree.getroot()
446
+
447
+ # Namespaces
448
+ ns = {
449
+ "xdr": "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing",
450
+ "a": "http://schemas.openxmlformats.org/drawingml/2006/main",
451
+ "m": "http://schemas.openxmlformats.org/officeDocument/2006/math",
452
+ }
453
+
454
+ # Iterate TwoCellAnchor and OneCellAnchor
455
+ for anchor_tag in ["twoCellAnchor", "oneCellAnchor"]:
456
+ for anchor in root.findall(f".//xdr:{anchor_tag}", ns):
457
+ # Get position
458
+ from_node = anchor.find("xdr:from", ns)
459
+ if from_node is None:
460
+ continue
461
+ col_node = from_node.find("xdr:col", ns)
462
+ row_node = from_node.find("xdr:row", ns)
463
+ if col_node is None or row_node is None:
464
+ continue
465
+
466
+ r = int(row_node.text)
467
+ c = int(col_node.text)
468
+
469
+ # Look for math content
470
+ # Usually in graphicalFrame -> graphic -> graphicData -> oMathPara
471
+ # But simpler to search descendant m:oMath
472
+ maths = anchor.findall(".//m:oMath", ns)
473
+ anchor_latex: list[str] = []
474
+ for math in maths:
475
+ # # Simple text extraction
476
+ # text = "".join(math.itertext())
477
+ # if text.strip():
478
+ # # Wrap in latex block indicator if needed, or just plain text
479
+ # # User asked for formula, assuming latex-like visual or text is acceptable
480
+ # # Adding simple latex-like wrapper
481
+ # math_map[(r, c)].append(f"${text}$")
482
+ latex = str(oMath2Latex(math)).strip()
483
+ if latex:
484
+ math_map[(r, c)].append(latex)
485
+ anchor_latex.append(latex)
486
+ self._omml_artifacts.append((r, c, latex, len(self._omml_artifacts)))
487
+ if anchor_latex:
488
+ for node in anchor.findall(".//xdr:cNvPr", ns):
489
+ shape_id = node.get("id")
490
+ if shape_id:
491
+ self._omml_shape_ids.add(shape_id)
492
+
493
+ except Exception as e:
494
+ logger.warning(f"Error parsing math formulas: {e}")
495
+
496
+ return math_map
497
+
498
+ def _get_anchor_pos(self, anchor):
499
+ """Helper to get (row, col) from anchor."""
500
+ if hasattr(anchor, "_from"):
501
+ return anchor._from.row, anchor._from.col
502
+ return None, None
503
+
504
+ def _extract_chart_range_formula(self, value_source) -> str | None:
505
+ if value_source is None:
506
+ return None
507
+
508
+ for attr_name in ("numRef", "strRef", "multiLvlStrRef"):
509
+ ref = getattr(value_source, attr_name, None)
510
+ formula = getattr(ref, "f", None)
511
+ if formula:
512
+ return formula
513
+
514
+ return None
515
+
516
+ def _iter_chart_reference_formulas(self, chart):
517
+ for series in getattr(chart, "ser", []):
518
+ for attr_name in ("cat", "val", "xVal", "yVal", "bubbleSize"):
519
+ formula = self._extract_chart_range_formula(getattr(series, attr_name, None))
520
+ if formula:
521
+ yield formula
522
+
523
+ tx = getattr(series, "tx", None)
524
+ tx_formula = getattr(getattr(tx, "strRef", None), "f", None)
525
+ if tx_formula:
526
+ yield tx_formula
527
+
528
+ def _parse_chart_reference_formula(self, formula: str, sheet_title: str) -> tuple[list[int], list[int]] | None:
529
+ try:
530
+ (
531
+ formula_sheet_name,
532
+ (
533
+ min_col,
534
+ min_row,
535
+ max_col,
536
+ max_row,
537
+ ),
538
+ ) = range_to_tuple(formula)
539
+ except ValueError:
540
+ logger.debug("Skip unsupported chart reference formula: {}", formula)
541
+ return None
542
+
543
+ if formula_sheet_name != sheet_title:
544
+ logger.debug(
545
+ "Skip chart reference formula from different sheet: {} != {}",
546
+ formula_sheet_name,
547
+ sheet_title,
548
+ )
549
+ return None
550
+
551
+ if not all(isinstance(bound, int) for bound in (min_col, min_row, max_col, max_row)):
552
+ logger.debug(
553
+ "Skip chart reference formula with open-ended bounds: {}",
554
+ formula,
555
+ )
556
+ return None
557
+
558
+ rows = list(range(min_row - 1, max_row))
559
+ cols = list(range(min_col - 1, max_col))
560
+ return rows, cols
561
+
562
+ def _collect_chart_source_axes(self, sheet: Worksheet, chart) -> tuple[list[int], list[int]] | None:
563
+ referenced_rows = set()
564
+ referenced_cols = set()
565
+ formulas_found = False
566
+
567
+ for formula in self._iter_chart_reference_formulas(chart):
568
+ formulas_found = True
569
+ parsed_axes = self._parse_chart_reference_formula(formula, sheet.title)
570
+ if parsed_axes is None:
571
+ return None
572
+
573
+ rows, cols = parsed_axes
574
+ referenced_rows.update(rows)
575
+ referenced_cols.update(cols)
576
+
577
+ if not formulas_found or not referenced_rows or not referenced_cols:
578
+ return None
579
+
580
+ return sorted(referenced_rows), sorted(referenced_cols)
581
+
582
+ def _find_charts_in_sheet(self, sheet: Worksheet) -> list[AnchoredBlock]:
583
+ chart_artifacts = []
584
+ for order, chart in enumerate(getattr(sheet, "_charts", [])):
585
+ axes = self._collect_chart_source_axes(sheet, chart)
586
+ if axes is None:
587
+ logger.debug(
588
+ "Skip chart on sheet '{}' because chart source ranges are unsupported",
589
+ sheet.title,
590
+ )
591
+ continue
592
+
593
+ rows, cols = axes
594
+ chart_table = self._build_synthetic_table_from_sheet_selection(
595
+ sheet,
596
+ rows,
597
+ cols,
598
+ )
599
+ anchor_row, anchor_col = self._get_anchor_pos(getattr(chart, "anchor", None))
600
+ chart_artifacts.append(
601
+ (
602
+ self._get_block_sort_anchor(anchor_row, anchor_col),
603
+ 10_000 + order,
604
+ {
605
+ "type": BlockType.CHART,
606
+ "content": render_spreadsheet_table(chart_table),
607
+ },
608
+ )
609
+ )
610
+
611
+ return chart_artifacts
612
+
613
+ def _resolve_cell_image(self, text: str) -> str:
614
+ """解析 WPS DISPIMG 单元格函数并返回图片或公式 HTML。"""
615
+ match = re.search(r'"([^"]+)"', text)
616
+ if match:
617
+ image_id = match.group(1)
618
+
619
+ else:
620
+ logger.error(f"无法从单元格文本中提取图片 ID,文本内容:{text}")
621
+ return ""
622
+
623
+ cell_image_map = self._load_cell_image_mappings()
624
+
625
+ zip_target_path = posixpath.normpath(posixpath.join("xl", cell_image_map.get(image_id, "")))
626
+ if self.zf is None or zip_target_path not in self.zf.namelist():
627
+ logger.warning(f"图片目标文件不存在,image_id={image_id}, target={zip_target_path}")
628
+ return ""
629
+
630
+ try:
631
+ image_payload = self._read_xlsx_image_member(zip_target_path)
632
+ if image_payload is None:
633
+ return ""
634
+ latex = self._image_equation_decoder.decode(
635
+ image_payload,
636
+ part_name=zip_target_path,
637
+ )
638
+ if latex:
639
+ return EQUATION_BOOKENDS.format(EQ=latex)
640
+ img_base64 = serialize_office_image(
641
+ image_payload,
642
+ part_name=zip_target_path,
643
+ content_type=None,
644
+ )
645
+ return rf'<img src="{img_base64}" />' if img_base64 is not None else ""
646
+ except Exception as e:
647
+ logger.warning(f"读取单元格图片失败,image_id={image_id}, target={zip_target_path}, error={e}")
648
+ return ""
649
+
650
+ def _load_cell_image_mappings(self):
651
+ if self.cell_image_map:
652
+ return self.cell_image_map
653
+
654
+ if self.zf is None:
655
+ return {}
656
+ cell_image_embed_to_name = {}
657
+ cellimages_path = "xl/cellimages.xml"
658
+ rels_path = "xl/_rels/cellimages.xml.rels"
659
+ if cellimages_path not in self.zf.namelist() or rels_path not in self.zf.namelist():
660
+ return {}
661
+
662
+ try:
663
+ with self.zf.open(cellimages_path) as f:
664
+ root = ET.parse(f).getroot()
665
+
666
+ ns = {
667
+ "xdr": "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing",
668
+ "a": "http://schemas.openxmlformats.org/drawingml/2006/main",
669
+ "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
670
+ "etc": "http://www.wps.cn/officeDocument/2017/etCustomData",
671
+ }
672
+
673
+ for cell_image in root.findall(".//etc:cellImage", ns):
674
+ c_nv_pr = cell_image.find(".//xdr:cNvPr", ns)
675
+ blip = cell_image.find(".//a:blip", ns)
676
+ if c_nv_pr is None or blip is None:
677
+ continue
678
+
679
+ image_name = c_nv_pr.attrib.get("name")
680
+ embed_id = blip.attrib.get(f"{{{ns['r']}}}embed")
681
+ if image_name and embed_id:
682
+ cell_image_embed_to_name[embed_id] = image_name
683
+
684
+ with self.zf.open(rels_path) as f:
685
+ rel_root = ET.parse(f).getroot()
686
+
687
+ rel_ns = {"pr": "http://schemas.openxmlformats.org/package/2006/relationships"}
688
+ for rel in rel_root.findall("pr:Relationship", rel_ns):
689
+ rel_id = rel.attrib.get("Id")
690
+ target = rel.attrib.get("Target")
691
+ if rel_id and target:
692
+ image_name = cell_image_embed_to_name.get(rel_id)
693
+ if not image_name:
694
+ logger.warning(f"跳过缺少 cellImage 名称映射的关系: {rel_id}")
695
+ continue
696
+ self.cell_image_map[image_name] = target
697
+
698
+ except Exception as e:
699
+ logger.warning(f"解析 cellimages 映射失败: {e}")
700
+ return {}
701
+
702
+ return self.cell_image_map
703
+
704
+ @staticmethod
705
+ def _get_sheet_content_layer(sheet: Worksheet):
706
+ """根据工作表的可见性返回对应的内容层。
707
+
708
+ 若工作表可见,返回 None(默认层);否则返回 INVISIBLE 层。
709
+
710
+ 参数:
711
+ sheet: 待检查的工作表。
712
+
713
+ 返回:
714
+ ContentLayer.INVISIBLE 或 None。
715
+ """
716
+ return None if sheet.sheet_state == Worksheet.SHEETSTATE_VISIBLE else "INVISIBLE"