docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,716 @@
|
|
|
1
|
+
import collections
|
|
2
|
+
import hashlib
|
|
3
|
+
import posixpath
|
|
4
|
+
import re
|
|
5
|
+
import xml.etree.ElementTree as ET
|
|
6
|
+
import zipfile
|
|
7
|
+
from io import BytesIO
|
|
8
|
+
from typing import BinaryIO, cast
|
|
9
|
+
|
|
10
|
+
from loguru import logger
|
|
11
|
+
from openpyxl import load_workbook
|
|
12
|
+
from openpyxl.drawing.image import Image as XlsImage
|
|
13
|
+
from openpyxl.utils.cell import range_to_tuple
|
|
14
|
+
from openpyxl.worksheet.worksheet import Worksheet
|
|
15
|
+
from ..image import serialize_office_image
|
|
16
|
+
from ..equation.image import OfficeImageEquationDecoder
|
|
17
|
+
from ..equation.ooxml import OoxmlEquationDecoder
|
|
18
|
+
from ..errors import LegacyOfficeResourceLimitError
|
|
19
|
+
from ..limits import MAX_ENTRY_BYTES
|
|
20
|
+
from ..equation.omml import oMath2Latex
|
|
21
|
+
from ..streams import read_stream_bytes_from_start, rewind_stream
|
|
22
|
+
from ..spreadsheet.html import EQUATION_BOOKENDS, render_spreadsheet_table
|
|
23
|
+
from ..spreadsheet.models import AnchoredBlock, FormulaMap, SheetImage
|
|
24
|
+
from ..spreadsheet.projector import SpreadsheetProjector
|
|
25
|
+
from .package_normalizer import normalize_xlsx_package, strip_xlsx_ole_objects_for_openpyxl
|
|
26
|
+
from .ooxml_ole import (
|
|
27
|
+
XlsxOleEquationArtifact,
|
|
28
|
+
package_has_sheet_ole_objects,
|
|
29
|
+
read_sheet_image_artifacts,
|
|
30
|
+
read_sheet_equation_artifacts,
|
|
31
|
+
workbook_sheet_parts,
|
|
32
|
+
)
|
|
33
|
+
from .....schema import BlockType
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class XlsxConverter(SpreadsheetProjector):
|
|
37
|
+
def __init__(
|
|
38
|
+
self,
|
|
39
|
+
treat_singleton_as_text: bool = True,
|
|
40
|
+
gap_tolerance: int | None = None,
|
|
41
|
+
include_hidden_sheets: bool = False,
|
|
42
|
+
) -> None:
|
|
43
|
+
super().__init__(
|
|
44
|
+
treat_singleton_as_text=treat_singleton_as_text,
|
|
45
|
+
gap_tolerance=gap_tolerance,
|
|
46
|
+
include_hidden_sheets=include_hidden_sheets,
|
|
47
|
+
)
|
|
48
|
+
self.zf = None
|
|
49
|
+
self.image_map = {}
|
|
50
|
+
self.cell_image_map = {}
|
|
51
|
+
self._sheet_part_by_title: dict[str, str] = {}
|
|
52
|
+
self._ole_artifacts: list[XlsxOleEquationArtifact] = []
|
|
53
|
+
self._omml_shape_ids: set[str] = set()
|
|
54
|
+
self._omml_artifacts: list[tuple[int, int, str, int]] = []
|
|
55
|
+
self._suppressed_ole_previews: set[tuple[tuple[int, int], str]] = set()
|
|
56
|
+
self._ooxml_equation_decoder = OoxmlEquationDecoder()
|
|
57
|
+
self._image_equation_decoder = OfficeImageEquationDecoder()
|
|
58
|
+
|
|
59
|
+
def convert(
|
|
60
|
+
self,
|
|
61
|
+
file_stream: BinaryIO,
|
|
62
|
+
) -> None:
|
|
63
|
+
if rewind_stream(file_stream):
|
|
64
|
+
try:
|
|
65
|
+
self._convert_package_stream(file_stream)
|
|
66
|
+
return
|
|
67
|
+
except Exception as exc:
|
|
68
|
+
file_bytes = read_stream_bytes_from_start(file_stream)
|
|
69
|
+
self._retry_convert_package_bytes_after_normalization(file_bytes, exc)
|
|
70
|
+
return
|
|
71
|
+
|
|
72
|
+
file_bytes = file_stream.read()
|
|
73
|
+
try:
|
|
74
|
+
self._convert_package_bytes(file_bytes)
|
|
75
|
+
except Exception as exc:
|
|
76
|
+
self._retry_convert_package_bytes_after_normalization(file_bytes, exc)
|
|
77
|
+
|
|
78
|
+
def _reset_state(self) -> None:
|
|
79
|
+
"""重置解析状态,确保失败重试时不会残留上一次半解析结果。"""
|
|
80
|
+
if self.zf:
|
|
81
|
+
self.zf.close()
|
|
82
|
+
self._reset_projection_state()
|
|
83
|
+
self.zf = None
|
|
84
|
+
self.image_map = {}
|
|
85
|
+
self.cell_image_map = {}
|
|
86
|
+
self._sheet_part_by_title = {}
|
|
87
|
+
self._ole_artifacts = []
|
|
88
|
+
self._omml_shape_ids = set()
|
|
89
|
+
self._omml_artifacts = []
|
|
90
|
+
self._suppressed_ole_previews = set()
|
|
91
|
+
self._ooxml_equation_decoder = OoxmlEquationDecoder()
|
|
92
|
+
self._image_equation_decoder = OfficeImageEquationDecoder()
|
|
93
|
+
|
|
94
|
+
def _convert_package_bytes(self, file_bytes: bytes) -> None:
|
|
95
|
+
"""用独立字节流解析 XLSX 包,便于原始包失败后用规范化包重试。"""
|
|
96
|
+
self._convert_package_stream(BytesIO(file_bytes))
|
|
97
|
+
|
|
98
|
+
def _convert_package_stream(self, file_stream: BinaryIO) -> None:
|
|
99
|
+
"""直接使用可复位的 XLSX 流解析正常路径,避免提前复制完整包字节。"""
|
|
100
|
+
self._reset_state()
|
|
101
|
+
try:
|
|
102
|
+
self.zf = zipfile.ZipFile(file_stream)
|
|
103
|
+
self._sheet_part_by_title = workbook_sheet_parts(self.zf)
|
|
104
|
+
except Exception as e:
|
|
105
|
+
logger.warning(f"Failed to open zip file: {e}")
|
|
106
|
+
self.zf = None
|
|
107
|
+
|
|
108
|
+
try:
|
|
109
|
+
workbook_stream: BinaryIO = file_stream
|
|
110
|
+
if self.zf is not None and package_has_sheet_ole_objects(
|
|
111
|
+
self.zf,
|
|
112
|
+
self._sheet_part_by_title,
|
|
113
|
+
):
|
|
114
|
+
file_bytes = read_stream_bytes_from_start(file_stream)
|
|
115
|
+
workbook_stream = BytesIO(strip_xlsx_ole_objects_for_openpyxl(file_bytes))
|
|
116
|
+
else:
|
|
117
|
+
rewind_stream(file_stream)
|
|
118
|
+
self.workbook = load_workbook(
|
|
119
|
+
filename=workbook_stream,
|
|
120
|
+
data_only=True,
|
|
121
|
+
rich_text=True,
|
|
122
|
+
)
|
|
123
|
+
if self.workbook is not None:
|
|
124
|
+
# 遍历需要参与转换的工作表,避免为隐藏表或尾部空页生成无效页面。
|
|
125
|
+
sheet_pages = []
|
|
126
|
+
for idx, sheet in enumerate(self._iter_sheets_to_convert(), start=1):
|
|
127
|
+
logger.debug(f"正在处理第 {idx} 个工作表:{sheet.title}")
|
|
128
|
+
self.cur_page = []
|
|
129
|
+
self._convert_sheet(sheet)
|
|
130
|
+
sheet_pages.append((sheet.title, self.cur_page))
|
|
131
|
+
if self._should_emit_sheet_titles([page for _, page in sheet_pages]):
|
|
132
|
+
self._prepend_sheet_titles(sheet_pages)
|
|
133
|
+
self.pages.extend(page for _, page in sheet_pages)
|
|
134
|
+
else:
|
|
135
|
+
logger.error("工作簿未初始化。")
|
|
136
|
+
finally:
|
|
137
|
+
if self.zf:
|
|
138
|
+
self.zf.close()
|
|
139
|
+
self.zf = None
|
|
140
|
+
|
|
141
|
+
def _retry_convert_package_bytes_after_normalization(
|
|
142
|
+
self,
|
|
143
|
+
file_bytes: bytes,
|
|
144
|
+
exc: Exception,
|
|
145
|
+
) -> None:
|
|
146
|
+
"""首次解析失败后,仅在包规范化确实产生变化时使用规范化字节重试。"""
|
|
147
|
+
normalized_bytes = normalize_xlsx_package(file_bytes)
|
|
148
|
+
if normalized_bytes == file_bytes:
|
|
149
|
+
raise exc
|
|
150
|
+
logger.warning(f"Retrying XLSX parsing after package normalization: {exc}")
|
|
151
|
+
self._convert_package_bytes(normalized_bytes)
|
|
152
|
+
|
|
153
|
+
def _prepare_sheet_assets(self, sheet: Worksheet) -> None:
|
|
154
|
+
"""准备 XLSX 公式、图片与 OLE 素材,并保持既有预览抑制优先级。"""
|
|
155
|
+
self.math_map = self._map_math_formulas_to_cells(sheet)
|
|
156
|
+
self._ole_artifacts = self._read_ole_equation_artifacts(sheet)
|
|
157
|
+
self._suppressed_ole_previews = {
|
|
158
|
+
((artifact.row, artifact.col), artifact.preview_base64)
|
|
159
|
+
for artifact in self._ole_artifacts
|
|
160
|
+
if artifact.shape_id in self._omml_shape_ids
|
|
161
|
+
and artifact.row is not None
|
|
162
|
+
and artifact.col is not None
|
|
163
|
+
and artifact.preview_base64 is not None
|
|
164
|
+
}
|
|
165
|
+
self._ole_artifacts = [artifact for artifact in self._ole_artifacts if artifact.shape_id not in self._omml_shape_ids]
|
|
166
|
+
for artifact in self._ole_artifacts:
|
|
167
|
+
if artifact.latex is None or artifact.row is None or artifact.col is None:
|
|
168
|
+
continue
|
|
169
|
+
self.math_map.setdefault((artifact.row, artifact.col), []).append(artifact.latex)
|
|
170
|
+
|
|
171
|
+
self.sheet_images = self._collect_sheet_images(sheet)
|
|
172
|
+
ole_previews = self._suppressed_ole_previews | {
|
|
173
|
+
((artifact.row, artifact.col), artifact.preview_base64)
|
|
174
|
+
for artifact in self._ole_artifacts
|
|
175
|
+
if artifact.row is not None and artifact.col is not None and artifact.preview_base64
|
|
176
|
+
}
|
|
177
|
+
self.sheet_images = [image for image in self.sheet_images if (image.anchor, image.image_base64) not in ole_previews]
|
|
178
|
+
self.table_image_map = collections.defaultdict(list)
|
|
179
|
+
for image in self.sheet_images:
|
|
180
|
+
row, col = image.anchor
|
|
181
|
+
if row is None or col is None:
|
|
182
|
+
continue
|
|
183
|
+
if image.latex:
|
|
184
|
+
self.table_image_map[(row, col)].append(EQUATION_BOOKENDS.format(EQ=image.latex))
|
|
185
|
+
elif image.image_base64:
|
|
186
|
+
self.table_image_map[(row, col)].append(f'<img src="{image.image_base64}" />')
|
|
187
|
+
for artifact in self._ole_artifacts:
|
|
188
|
+
if artifact.latex is not None or artifact.preview_base64 is None or artifact.row is None or artifact.col is None:
|
|
189
|
+
continue
|
|
190
|
+
self.table_image_map[(artifact.row, artifact.col)].append(f'<img src="{artifact.preview_base64}" />')
|
|
191
|
+
|
|
192
|
+
def _find_additional_visual_artifacts(
|
|
193
|
+
self,
|
|
194
|
+
used_cells: set[tuple[int, int]],
|
|
195
|
+
) -> list[AnchoredBlock]:
|
|
196
|
+
"""输出未被表格吸收的 XLSX 公式、OLE 预览和图片公式。"""
|
|
197
|
+
return [
|
|
198
|
+
*self._find_equation_artifacts_in_sheet(used_cells),
|
|
199
|
+
*self._find_image_equation_artifacts_in_sheet(used_cells),
|
|
200
|
+
]
|
|
201
|
+
|
|
202
|
+
def _read_ole_equation_artifacts(
|
|
203
|
+
self,
|
|
204
|
+
sheet: Worksheet,
|
|
205
|
+
) -> list[XlsxOleEquationArtifact]:
|
|
206
|
+
"""从当前 worksheet part 读取 MathType/Equation 公式和预览。"""
|
|
207
|
+
|
|
208
|
+
if self.zf is None:
|
|
209
|
+
return []
|
|
210
|
+
worksheet_part = self._sheet_part_by_title.get(sheet.title)
|
|
211
|
+
if worksheet_part is None:
|
|
212
|
+
return []
|
|
213
|
+
return read_sheet_equation_artifacts(
|
|
214
|
+
self.zf,
|
|
215
|
+
worksheet_part,
|
|
216
|
+
self._ooxml_equation_decoder,
|
|
217
|
+
self._image_equation_decoder,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
def _find_equation_artifacts_in_sheet(
|
|
221
|
+
self,
|
|
222
|
+
used_cells: set[tuple[int, int]],
|
|
223
|
+
) -> list[tuple[tuple[int, int], int, dict]]:
|
|
224
|
+
"""输出未被表格吸收的 OMML/MTEF 公式或缓存预览。"""
|
|
225
|
+
|
|
226
|
+
artifacts: list[tuple[tuple[int, int], int, dict]] = []
|
|
227
|
+
for row, col, latex, order in self._omml_artifacts:
|
|
228
|
+
if (row, col) in used_cells:
|
|
229
|
+
continue
|
|
230
|
+
artifacts.append(
|
|
231
|
+
(
|
|
232
|
+
(row, col),
|
|
233
|
+
15_000 + order,
|
|
234
|
+
{
|
|
235
|
+
"type": BlockType.EQUATION,
|
|
236
|
+
"content": latex,
|
|
237
|
+
},
|
|
238
|
+
)
|
|
239
|
+
)
|
|
240
|
+
for artifact in self._ole_artifacts:
|
|
241
|
+
coordinate = (
|
|
242
|
+
artifact.row if artifact.row is not None else 10**9,
|
|
243
|
+
artifact.col if artifact.col is not None else 10**9,
|
|
244
|
+
)
|
|
245
|
+
if artifact.row is not None and artifact.col is not None and (artifact.row, artifact.col) in used_cells:
|
|
246
|
+
continue
|
|
247
|
+
block = None
|
|
248
|
+
if artifact.latex is not None:
|
|
249
|
+
block = {
|
|
250
|
+
"type": BlockType.EQUATION,
|
|
251
|
+
"content": artifact.latex,
|
|
252
|
+
}
|
|
253
|
+
elif artifact.preview_base64 is not None:
|
|
254
|
+
block = {
|
|
255
|
+
"type": BlockType.IMAGE,
|
|
256
|
+
"image_base64": artifact.preview_base64,
|
|
257
|
+
}
|
|
258
|
+
if block is not None:
|
|
259
|
+
artifacts.append((coordinate, 20_000 + artifact.order, block))
|
|
260
|
+
return artifacts
|
|
261
|
+
|
|
262
|
+
def _find_image_equation_artifacts_in_sheet(
|
|
263
|
+
self,
|
|
264
|
+
used_cells: set[tuple[int, int]],
|
|
265
|
+
) -> list[tuple[tuple[int, int], int, dict]]:
|
|
266
|
+
"""按原始 drawing anchor 输出未被表格吸收的图片 comment 公式。"""
|
|
267
|
+
|
|
268
|
+
artifacts: list[tuple[tuple[int, int], int, dict]] = []
|
|
269
|
+
for image in self.sheet_images:
|
|
270
|
+
if not image.latex:
|
|
271
|
+
continue
|
|
272
|
+
row, col = image.anchor
|
|
273
|
+
if row is not None and col is not None and (row, col) in used_cells:
|
|
274
|
+
continue
|
|
275
|
+
coordinate = (
|
|
276
|
+
row if row is not None else 10**9,
|
|
277
|
+
col if col is not None else 10**9,
|
|
278
|
+
)
|
|
279
|
+
artifacts.append(
|
|
280
|
+
(
|
|
281
|
+
coordinate,
|
|
282
|
+
25_000 + image.order,
|
|
283
|
+
{"type": BlockType.EQUATION, "content": image.latex},
|
|
284
|
+
)
|
|
285
|
+
)
|
|
286
|
+
return artifacts
|
|
287
|
+
|
|
288
|
+
def _read_xlsx_image_member(self, part_name: str) -> bytes | None:
|
|
289
|
+
"""从原始 XLSX ZIP 有界读取 media member。"""
|
|
290
|
+
|
|
291
|
+
if self.zf is None:
|
|
292
|
+
return None
|
|
293
|
+
normalized = posixpath.normpath(part_name.lstrip("/"))
|
|
294
|
+
if normalized.startswith("../") or normalized not in self.zf.namelist():
|
|
295
|
+
return None
|
|
296
|
+
info = self.zf.getinfo(normalized)
|
|
297
|
+
if info.file_size > MAX_ENTRY_BYTES:
|
|
298
|
+
raise LegacyOfficeResourceLimitError(f"XLSX image exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
|
|
299
|
+
with self.zf.open(info) as stream:
|
|
300
|
+
payload = stream.read(MAX_ENTRY_BYTES + 1)
|
|
301
|
+
if len(payload) > MAX_ENTRY_BYTES:
|
|
302
|
+
raise LegacyOfficeResourceLimitError(f"XLSX image exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
|
|
303
|
+
return payload
|
|
304
|
+
|
|
305
|
+
def _raw_sheet_image(
|
|
306
|
+
self,
|
|
307
|
+
image: XlsImage,
|
|
308
|
+
) -> tuple[bytes, str | None, str | None] | None:
|
|
309
|
+
"""优先从原 ZIP 读取 openpyxl 图片的未转码原始字节。"""
|
|
310
|
+
|
|
311
|
+
part_name = str(getattr(image, "path", "") or "") or None
|
|
312
|
+
payload = self._read_xlsx_image_member(part_name) if part_name is not None else None
|
|
313
|
+
if payload is None:
|
|
314
|
+
try:
|
|
315
|
+
payload = image._data() # type: ignore[attr-defined]
|
|
316
|
+
except Exception:
|
|
317
|
+
return None
|
|
318
|
+
if len(payload) > MAX_ENTRY_BYTES:
|
|
319
|
+
raise LegacyOfficeResourceLimitError(f"XLSX image exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
|
|
320
|
+
image_format = str(getattr(image, "format", "") or "").casefold()
|
|
321
|
+
content_type = f"image/{image_format}" if image_format else None
|
|
322
|
+
return payload, part_name, content_type
|
|
323
|
+
|
|
324
|
+
def _collect_sheet_images(self, sheet: Worksheet) -> list[SheetImage]:
|
|
325
|
+
"""读取当前工作表的原始图片并识别图片公式。"""
|
|
326
|
+
images: list[SheetImage] = []
|
|
327
|
+
if self.workbook is None:
|
|
328
|
+
return images
|
|
329
|
+
|
|
330
|
+
seen: set[tuple[tuple[int | None, int | None], bytes]] = set()
|
|
331
|
+
worksheet_part = self._sheet_part_by_title.get(sheet.title)
|
|
332
|
+
if self.zf is not None and worksheet_part is not None:
|
|
333
|
+
for artifact in read_sheet_image_artifacts(
|
|
334
|
+
self.zf,
|
|
335
|
+
worksheet_part,
|
|
336
|
+
):
|
|
337
|
+
anchor = (artifact.row, artifact.col)
|
|
338
|
+
digest = hashlib.sha256(artifact.payload).digest()
|
|
339
|
+
key = (anchor, digest)
|
|
340
|
+
if key in seen:
|
|
341
|
+
continue
|
|
342
|
+
seen.add(key)
|
|
343
|
+
latex = self._image_equation_decoder.decode(
|
|
344
|
+
artifact.payload,
|
|
345
|
+
part_name=artifact.part_name,
|
|
346
|
+
)
|
|
347
|
+
image_base64 = serialize_office_image(
|
|
348
|
+
artifact.payload,
|
|
349
|
+
part_name=artifact.part_name,
|
|
350
|
+
content_type=None,
|
|
351
|
+
)
|
|
352
|
+
if latex is None and image_base64 is None:
|
|
353
|
+
continue
|
|
354
|
+
images.append(
|
|
355
|
+
SheetImage(
|
|
356
|
+
anchor=anchor,
|
|
357
|
+
image_base64=image_base64,
|
|
358
|
+
latex=latex,
|
|
359
|
+
order=artifact.order,
|
|
360
|
+
)
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
for image_order, item in enumerate(
|
|
364
|
+
getattr(sheet, "_images", []), # type: ignore[attr-defined]
|
|
365
|
+
start=10_000,
|
|
366
|
+
):
|
|
367
|
+
try:
|
|
368
|
+
image: XlsImage = cast(XlsImage, item)
|
|
369
|
+
raw_image = self._raw_sheet_image(image)
|
|
370
|
+
if raw_image is None:
|
|
371
|
+
continue
|
|
372
|
+
payload, part_name, content_type = raw_image
|
|
373
|
+
anchor = self._get_anchor_pos(item.anchor)
|
|
374
|
+
key = (anchor, hashlib.sha256(payload).digest())
|
|
375
|
+
if key in seen:
|
|
376
|
+
continue
|
|
377
|
+
seen.add(key)
|
|
378
|
+
latex = self._image_equation_decoder.decode(
|
|
379
|
+
payload,
|
|
380
|
+
part_name=part_name,
|
|
381
|
+
content_type=content_type,
|
|
382
|
+
)
|
|
383
|
+
image_base64 = serialize_office_image(
|
|
384
|
+
payload,
|
|
385
|
+
part_name=part_name,
|
|
386
|
+
content_type=content_type,
|
|
387
|
+
)
|
|
388
|
+
if latex is None and image_base64 is None:
|
|
389
|
+
continue
|
|
390
|
+
images.append(
|
|
391
|
+
SheetImage(
|
|
392
|
+
anchor=anchor,
|
|
393
|
+
image_base64=image_base64,
|
|
394
|
+
latex=latex,
|
|
395
|
+
order=image_order,
|
|
396
|
+
)
|
|
397
|
+
)
|
|
398
|
+
except Exception as e:
|
|
399
|
+
logger.error(f"无法从 Excel 工作表中提取图片,错误信息:{e}")
|
|
400
|
+
|
|
401
|
+
return images
|
|
402
|
+
|
|
403
|
+
def _map_math_formulas_to_cells(self, sheet: Worksheet) -> FormulaMap:
|
|
404
|
+
"""从 worksheet drawing 恢复按 cell anchor 分组的 OMML 公式。"""
|
|
405
|
+
math_map = collections.defaultdict(list)
|
|
406
|
+
self._omml_shape_ids = set()
|
|
407
|
+
self._omml_artifacts = []
|
|
408
|
+
if not self.zf:
|
|
409
|
+
return math_map
|
|
410
|
+
|
|
411
|
+
# Find drawing relation
|
|
412
|
+
drawing_rel = None
|
|
413
|
+
if hasattr(sheet, "_rels"):
|
|
414
|
+
for rel in sheet._rels:
|
|
415
|
+
if rel.Type.endswith("/relationships/drawing"):
|
|
416
|
+
drawing_rel = rel
|
|
417
|
+
break
|
|
418
|
+
|
|
419
|
+
if not drawing_rel:
|
|
420
|
+
return math_map
|
|
421
|
+
|
|
422
|
+
# Resolve path
|
|
423
|
+
# Assuming relative path from worksheets/sheetX.xml to drawings/drawingY.xml
|
|
424
|
+
# Usually target is like "../drawings/drawing1.xml"
|
|
425
|
+
target = drawing_rel.Target
|
|
426
|
+
if target.startswith("../"):
|
|
427
|
+
path = target.replace("../", "xl/") # simplistic resolution
|
|
428
|
+
elif target.startswith("/"):
|
|
429
|
+
path = target[1:]
|
|
430
|
+
else:
|
|
431
|
+
path = f"xl/worksheets/{target}" # unlikely but default relative
|
|
432
|
+
|
|
433
|
+
# Check if file exists in zip
|
|
434
|
+
if path not in self.zf.namelist():
|
|
435
|
+
# Try generic match if simplistic resolution failed
|
|
436
|
+
# drawing1.xml -> xl/drawings/drawing1.xml
|
|
437
|
+
basename = target.split("/")[-1]
|
|
438
|
+
path = f"xl/drawings/{basename}"
|
|
439
|
+
if path not in self.zf.namelist():
|
|
440
|
+
return math_map
|
|
441
|
+
|
|
442
|
+
try:
|
|
443
|
+
with self.zf.open(path) as f:
|
|
444
|
+
tree = ET.parse(f)
|
|
445
|
+
root = tree.getroot()
|
|
446
|
+
|
|
447
|
+
# Namespaces
|
|
448
|
+
ns = {
|
|
449
|
+
"xdr": "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing",
|
|
450
|
+
"a": "http://schemas.openxmlformats.org/drawingml/2006/main",
|
|
451
|
+
"m": "http://schemas.openxmlformats.org/officeDocument/2006/math",
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
# Iterate TwoCellAnchor and OneCellAnchor
|
|
455
|
+
for anchor_tag in ["twoCellAnchor", "oneCellAnchor"]:
|
|
456
|
+
for anchor in root.findall(f".//xdr:{anchor_tag}", ns):
|
|
457
|
+
# Get position
|
|
458
|
+
from_node = anchor.find("xdr:from", ns)
|
|
459
|
+
if from_node is None:
|
|
460
|
+
continue
|
|
461
|
+
col_node = from_node.find("xdr:col", ns)
|
|
462
|
+
row_node = from_node.find("xdr:row", ns)
|
|
463
|
+
if col_node is None or row_node is None:
|
|
464
|
+
continue
|
|
465
|
+
|
|
466
|
+
r = int(row_node.text)
|
|
467
|
+
c = int(col_node.text)
|
|
468
|
+
|
|
469
|
+
# Look for math content
|
|
470
|
+
# Usually in graphicalFrame -> graphic -> graphicData -> oMathPara
|
|
471
|
+
# But simpler to search descendant m:oMath
|
|
472
|
+
maths = anchor.findall(".//m:oMath", ns)
|
|
473
|
+
anchor_latex: list[str] = []
|
|
474
|
+
for math in maths:
|
|
475
|
+
# # Simple text extraction
|
|
476
|
+
# text = "".join(math.itertext())
|
|
477
|
+
# if text.strip():
|
|
478
|
+
# # Wrap in latex block indicator if needed, or just plain text
|
|
479
|
+
# # User asked for formula, assuming latex-like visual or text is acceptable
|
|
480
|
+
# # Adding simple latex-like wrapper
|
|
481
|
+
# math_map[(r, c)].append(f"${text}$")
|
|
482
|
+
latex = str(oMath2Latex(math)).strip()
|
|
483
|
+
if latex:
|
|
484
|
+
math_map[(r, c)].append(latex)
|
|
485
|
+
anchor_latex.append(latex)
|
|
486
|
+
self._omml_artifacts.append((r, c, latex, len(self._omml_artifacts)))
|
|
487
|
+
if anchor_latex:
|
|
488
|
+
for node in anchor.findall(".//xdr:cNvPr", ns):
|
|
489
|
+
shape_id = node.get("id")
|
|
490
|
+
if shape_id:
|
|
491
|
+
self._omml_shape_ids.add(shape_id)
|
|
492
|
+
|
|
493
|
+
except Exception as e:
|
|
494
|
+
logger.warning(f"Error parsing math formulas: {e}")
|
|
495
|
+
|
|
496
|
+
return math_map
|
|
497
|
+
|
|
498
|
+
def _get_anchor_pos(self, anchor):
|
|
499
|
+
"""Helper to get (row, col) from anchor."""
|
|
500
|
+
if hasattr(anchor, "_from"):
|
|
501
|
+
return anchor._from.row, anchor._from.col
|
|
502
|
+
return None, None
|
|
503
|
+
|
|
504
|
+
def _extract_chart_range_formula(self, value_source) -> str | None:
|
|
505
|
+
if value_source is None:
|
|
506
|
+
return None
|
|
507
|
+
|
|
508
|
+
for attr_name in ("numRef", "strRef", "multiLvlStrRef"):
|
|
509
|
+
ref = getattr(value_source, attr_name, None)
|
|
510
|
+
formula = getattr(ref, "f", None)
|
|
511
|
+
if formula:
|
|
512
|
+
return formula
|
|
513
|
+
|
|
514
|
+
return None
|
|
515
|
+
|
|
516
|
+
def _iter_chart_reference_formulas(self, chart):
|
|
517
|
+
for series in getattr(chart, "ser", []):
|
|
518
|
+
for attr_name in ("cat", "val", "xVal", "yVal", "bubbleSize"):
|
|
519
|
+
formula = self._extract_chart_range_formula(getattr(series, attr_name, None))
|
|
520
|
+
if formula:
|
|
521
|
+
yield formula
|
|
522
|
+
|
|
523
|
+
tx = getattr(series, "tx", None)
|
|
524
|
+
tx_formula = getattr(getattr(tx, "strRef", None), "f", None)
|
|
525
|
+
if tx_formula:
|
|
526
|
+
yield tx_formula
|
|
527
|
+
|
|
528
|
+
def _parse_chart_reference_formula(self, formula: str, sheet_title: str) -> tuple[list[int], list[int]] | None:
|
|
529
|
+
try:
|
|
530
|
+
(
|
|
531
|
+
formula_sheet_name,
|
|
532
|
+
(
|
|
533
|
+
min_col,
|
|
534
|
+
min_row,
|
|
535
|
+
max_col,
|
|
536
|
+
max_row,
|
|
537
|
+
),
|
|
538
|
+
) = range_to_tuple(formula)
|
|
539
|
+
except ValueError:
|
|
540
|
+
logger.debug("Skip unsupported chart reference formula: {}", formula)
|
|
541
|
+
return None
|
|
542
|
+
|
|
543
|
+
if formula_sheet_name != sheet_title:
|
|
544
|
+
logger.debug(
|
|
545
|
+
"Skip chart reference formula from different sheet: {} != {}",
|
|
546
|
+
formula_sheet_name,
|
|
547
|
+
sheet_title,
|
|
548
|
+
)
|
|
549
|
+
return None
|
|
550
|
+
|
|
551
|
+
if not all(isinstance(bound, int) for bound in (min_col, min_row, max_col, max_row)):
|
|
552
|
+
logger.debug(
|
|
553
|
+
"Skip chart reference formula with open-ended bounds: {}",
|
|
554
|
+
formula,
|
|
555
|
+
)
|
|
556
|
+
return None
|
|
557
|
+
|
|
558
|
+
rows = list(range(min_row - 1, max_row))
|
|
559
|
+
cols = list(range(min_col - 1, max_col))
|
|
560
|
+
return rows, cols
|
|
561
|
+
|
|
562
|
+
def _collect_chart_source_axes(self, sheet: Worksheet, chart) -> tuple[list[int], list[int]] | None:
|
|
563
|
+
referenced_rows = set()
|
|
564
|
+
referenced_cols = set()
|
|
565
|
+
formulas_found = False
|
|
566
|
+
|
|
567
|
+
for formula in self._iter_chart_reference_formulas(chart):
|
|
568
|
+
formulas_found = True
|
|
569
|
+
parsed_axes = self._parse_chart_reference_formula(formula, sheet.title)
|
|
570
|
+
if parsed_axes is None:
|
|
571
|
+
return None
|
|
572
|
+
|
|
573
|
+
rows, cols = parsed_axes
|
|
574
|
+
referenced_rows.update(rows)
|
|
575
|
+
referenced_cols.update(cols)
|
|
576
|
+
|
|
577
|
+
if not formulas_found or not referenced_rows or not referenced_cols:
|
|
578
|
+
return None
|
|
579
|
+
|
|
580
|
+
return sorted(referenced_rows), sorted(referenced_cols)
|
|
581
|
+
|
|
582
|
+
def _find_charts_in_sheet(self, sheet: Worksheet) -> list[AnchoredBlock]:
|
|
583
|
+
chart_artifacts = []
|
|
584
|
+
for order, chart in enumerate(getattr(sheet, "_charts", [])):
|
|
585
|
+
axes = self._collect_chart_source_axes(sheet, chart)
|
|
586
|
+
if axes is None:
|
|
587
|
+
logger.debug(
|
|
588
|
+
"Skip chart on sheet '{}' because chart source ranges are unsupported",
|
|
589
|
+
sheet.title,
|
|
590
|
+
)
|
|
591
|
+
continue
|
|
592
|
+
|
|
593
|
+
rows, cols = axes
|
|
594
|
+
chart_table = self._build_synthetic_table_from_sheet_selection(
|
|
595
|
+
sheet,
|
|
596
|
+
rows,
|
|
597
|
+
cols,
|
|
598
|
+
)
|
|
599
|
+
anchor_row, anchor_col = self._get_anchor_pos(getattr(chart, "anchor", None))
|
|
600
|
+
chart_artifacts.append(
|
|
601
|
+
(
|
|
602
|
+
self._get_block_sort_anchor(anchor_row, anchor_col),
|
|
603
|
+
10_000 + order,
|
|
604
|
+
{
|
|
605
|
+
"type": BlockType.CHART,
|
|
606
|
+
"content": render_spreadsheet_table(chart_table),
|
|
607
|
+
},
|
|
608
|
+
)
|
|
609
|
+
)
|
|
610
|
+
|
|
611
|
+
return chart_artifacts
|
|
612
|
+
|
|
613
|
+
def _resolve_cell_image(self, text: str) -> str:
|
|
614
|
+
"""解析 WPS DISPIMG 单元格函数并返回图片或公式 HTML。"""
|
|
615
|
+
match = re.search(r'"([^"]+)"', text)
|
|
616
|
+
if match:
|
|
617
|
+
image_id = match.group(1)
|
|
618
|
+
|
|
619
|
+
else:
|
|
620
|
+
logger.error(f"无法从单元格文本中提取图片 ID,文本内容:{text}")
|
|
621
|
+
return ""
|
|
622
|
+
|
|
623
|
+
cell_image_map = self._load_cell_image_mappings()
|
|
624
|
+
|
|
625
|
+
zip_target_path = posixpath.normpath(posixpath.join("xl", cell_image_map.get(image_id, "")))
|
|
626
|
+
if self.zf is None or zip_target_path not in self.zf.namelist():
|
|
627
|
+
logger.warning(f"图片目标文件不存在,image_id={image_id}, target={zip_target_path}")
|
|
628
|
+
return ""
|
|
629
|
+
|
|
630
|
+
try:
|
|
631
|
+
image_payload = self._read_xlsx_image_member(zip_target_path)
|
|
632
|
+
if image_payload is None:
|
|
633
|
+
return ""
|
|
634
|
+
latex = self._image_equation_decoder.decode(
|
|
635
|
+
image_payload,
|
|
636
|
+
part_name=zip_target_path,
|
|
637
|
+
)
|
|
638
|
+
if latex:
|
|
639
|
+
return EQUATION_BOOKENDS.format(EQ=latex)
|
|
640
|
+
img_base64 = serialize_office_image(
|
|
641
|
+
image_payload,
|
|
642
|
+
part_name=zip_target_path,
|
|
643
|
+
content_type=None,
|
|
644
|
+
)
|
|
645
|
+
return rf'<img src="{img_base64}" />' if img_base64 is not None else ""
|
|
646
|
+
except Exception as e:
|
|
647
|
+
logger.warning(f"读取单元格图片失败,image_id={image_id}, target={zip_target_path}, error={e}")
|
|
648
|
+
return ""
|
|
649
|
+
|
|
650
|
+
def _load_cell_image_mappings(self):
|
|
651
|
+
if self.cell_image_map:
|
|
652
|
+
return self.cell_image_map
|
|
653
|
+
|
|
654
|
+
if self.zf is None:
|
|
655
|
+
return {}
|
|
656
|
+
cell_image_embed_to_name = {}
|
|
657
|
+
cellimages_path = "xl/cellimages.xml"
|
|
658
|
+
rels_path = "xl/_rels/cellimages.xml.rels"
|
|
659
|
+
if cellimages_path not in self.zf.namelist() or rels_path not in self.zf.namelist():
|
|
660
|
+
return {}
|
|
661
|
+
|
|
662
|
+
try:
|
|
663
|
+
with self.zf.open(cellimages_path) as f:
|
|
664
|
+
root = ET.parse(f).getroot()
|
|
665
|
+
|
|
666
|
+
ns = {
|
|
667
|
+
"xdr": "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing",
|
|
668
|
+
"a": "http://schemas.openxmlformats.org/drawingml/2006/main",
|
|
669
|
+
"r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
|
|
670
|
+
"etc": "http://www.wps.cn/officeDocument/2017/etCustomData",
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
for cell_image in root.findall(".//etc:cellImage", ns):
|
|
674
|
+
c_nv_pr = cell_image.find(".//xdr:cNvPr", ns)
|
|
675
|
+
blip = cell_image.find(".//a:blip", ns)
|
|
676
|
+
if c_nv_pr is None or blip is None:
|
|
677
|
+
continue
|
|
678
|
+
|
|
679
|
+
image_name = c_nv_pr.attrib.get("name")
|
|
680
|
+
embed_id = blip.attrib.get(f"{{{ns['r']}}}embed")
|
|
681
|
+
if image_name and embed_id:
|
|
682
|
+
cell_image_embed_to_name[embed_id] = image_name
|
|
683
|
+
|
|
684
|
+
with self.zf.open(rels_path) as f:
|
|
685
|
+
rel_root = ET.parse(f).getroot()
|
|
686
|
+
|
|
687
|
+
rel_ns = {"pr": "http://schemas.openxmlformats.org/package/2006/relationships"}
|
|
688
|
+
for rel in rel_root.findall("pr:Relationship", rel_ns):
|
|
689
|
+
rel_id = rel.attrib.get("Id")
|
|
690
|
+
target = rel.attrib.get("Target")
|
|
691
|
+
if rel_id and target:
|
|
692
|
+
image_name = cell_image_embed_to_name.get(rel_id)
|
|
693
|
+
if not image_name:
|
|
694
|
+
logger.warning(f"跳过缺少 cellImage 名称映射的关系: {rel_id}")
|
|
695
|
+
continue
|
|
696
|
+
self.cell_image_map[image_name] = target
|
|
697
|
+
|
|
698
|
+
except Exception as e:
|
|
699
|
+
logger.warning(f"解析 cellimages 映射失败: {e}")
|
|
700
|
+
return {}
|
|
701
|
+
|
|
702
|
+
return self.cell_image_map
|
|
703
|
+
|
|
704
|
+
@staticmethod
|
|
705
|
+
def _get_sheet_content_layer(sheet: Worksheet):
|
|
706
|
+
"""根据工作表的可见性返回对应的内容层。
|
|
707
|
+
|
|
708
|
+
若工作表可见,返回 None(默认层);否则返回 INVISIBLE 层。
|
|
709
|
+
|
|
710
|
+
参数:
|
|
711
|
+
sheet: 待检查的工作表。
|
|
712
|
+
|
|
713
|
+
返回:
|
|
714
|
+
ContentLayer.INVISIBLE 或 None。
|
|
715
|
+
"""
|
|
716
|
+
return None if sheet.sheet_state == Worksheet.SHEETSTATE_VISIBLE else "INVISIBLE"
|