docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1145 @@
|
|
|
1
|
+
"""纯 Python 解析 Excel 97–2003 Workbook BIFF stream。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
import struct
|
|
7
|
+
import uuid
|
|
8
|
+
|
|
9
|
+
from loguru import logger
|
|
10
|
+
|
|
11
|
+
from ..._shared.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
|
|
12
|
+
from ..errors import LegacyOfficeEncryptedError, LegacyOfficeMalformedError
|
|
13
|
+
from ..legacy.binary import get_f64, get_u16, get_u32
|
|
14
|
+
from ..legacy.officeart import OfficeArtShape, OfficeImagePayload, decode_bstore, extract_excel_shapes
|
|
15
|
+
from ..image import serialize_office_image
|
|
16
|
+
from ..equation.image import OfficeImageEquationDecoder
|
|
17
|
+
|
|
18
|
+
from .chart import chart_source_axes, chart_source_selection
|
|
19
|
+
from .models import (
|
|
20
|
+
XlsCell,
|
|
21
|
+
XlsChart,
|
|
22
|
+
XlsChartSheet,
|
|
23
|
+
XlsEquation,
|
|
24
|
+
XlsFontStyle,
|
|
25
|
+
XlsImage,
|
|
26
|
+
XlsRichRun,
|
|
27
|
+
XlsRichText,
|
|
28
|
+
XlsSheet,
|
|
29
|
+
XlsWorkbook,
|
|
30
|
+
)
|
|
31
|
+
from .number_format import builtin_number_format, format_number, format_text
|
|
32
|
+
from .records import BOF, CONTINUE, EOF, BiffRecord, RecordBudget, SegmentReader, collect_continues, iter_records, record_at
|
|
33
|
+
from .strings import DecodedString, clean_text, codepage_name, read_biff8_string, read_byte_string, read_txo_text, to_rich_text
|
|
34
|
+
|
|
35
|
+
FILEPASS = 0x002F
|
|
36
|
+
CODEPAGE = 0x0042
|
|
37
|
+
DATEMODE = 0x0022
|
|
38
|
+
BOUNDSHEET = 0x0085
|
|
39
|
+
SST = 0x00FC
|
|
40
|
+
FORMAT = 0x041E
|
|
41
|
+
XF = 0x00E0
|
|
42
|
+
FONT = 0x0031
|
|
43
|
+
ROW = 0x0208
|
|
44
|
+
COLINFO = 0x007D
|
|
45
|
+
MERGEDCELLS = 0x00E5
|
|
46
|
+
LABELSST = 0x00FD
|
|
47
|
+
LABEL = 0x0204
|
|
48
|
+
RSTRING = 0x00D6
|
|
49
|
+
NUMBER = 0x0203
|
|
50
|
+
RK = 0x027E
|
|
51
|
+
MULRK = 0x00BD
|
|
52
|
+
BOOLERR = 0x0205
|
|
53
|
+
FORMULA = 0x0006
|
|
54
|
+
STRING = 0x0207
|
|
55
|
+
MSODRAWINGGROUP = 0x00EB
|
|
56
|
+
MSODRAWING = 0x00EC
|
|
57
|
+
OBJ = 0x005D
|
|
58
|
+
TXO = 0x01B6
|
|
59
|
+
HLINK = 0x01B8
|
|
60
|
+
SUPBOOK = 0x01AE
|
|
61
|
+
EXTERNSHEET = 0x0017
|
|
62
|
+
WINDOW1 = 0x003D
|
|
63
|
+
|
|
64
|
+
WORKBOOK_GLOBALS_SUBSTREAM = 0x0005
|
|
65
|
+
WORKSHEET_SUBSTREAM = 0x0010
|
|
66
|
+
CHART_SUBSTREAM = 0x0020
|
|
67
|
+
MAX_ROWS = 65_536
|
|
68
|
+
MAX_COLS = 256
|
|
69
|
+
|
|
70
|
+
OBJ_CHART = 0x0005
|
|
71
|
+
OBJ_TEXTBOX = 0x0006
|
|
72
|
+
OBJ_PICTURE = 0x0008
|
|
73
|
+
OBJ_CHECKBOX = 0x000B
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass(frozen=True, slots=True)
|
|
77
|
+
class _BoundSheet:
|
|
78
|
+
"""BoundSheet8 目录项。"""
|
|
79
|
+
|
|
80
|
+
name: str
|
|
81
|
+
offset: int
|
|
82
|
+
visible: bool
|
|
83
|
+
sheet_type: int
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(frozen=True, slots=True)
|
|
87
|
+
class _CellFormat:
|
|
88
|
+
"""XF 解析后的字体索引和数值格式代码。"""
|
|
89
|
+
|
|
90
|
+
font_index: int
|
|
91
|
+
format_code: str | None
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass(slots=True)
|
|
95
|
+
class _Globals:
|
|
96
|
+
"""Workbook Globals Substream 中供所有工作表共享的状态。"""
|
|
97
|
+
|
|
98
|
+
biff8: bool
|
|
99
|
+
date1904: bool = False
|
|
100
|
+
encoding: str = "cp1252"
|
|
101
|
+
sheets: list[_BoundSheet] = field(default_factory=list)
|
|
102
|
+
strings: list[DecodedString] = field(default_factory=list)
|
|
103
|
+
fonts: list[XlsFontStyle] = field(default_factory=list)
|
|
104
|
+
formats: list[_CellFormat] = field(default_factory=list)
|
|
105
|
+
images: dict[int, OfficeImagePayload] = field(default_factory=dict)
|
|
106
|
+
extern_sheets: list[int | None] = field(default_factory=list)
|
|
107
|
+
active_sheet_index: int | None = None
|
|
108
|
+
|
|
109
|
+
def read_string(
|
|
110
|
+
self,
|
|
111
|
+
reader: SegmentReader,
|
|
112
|
+
*,
|
|
113
|
+
short: bool,
|
|
114
|
+
rich: bool = False,
|
|
115
|
+
) -> DecodedString | None:
|
|
116
|
+
"""按当前 BIFF 版本和 codepage 读取字符串。"""
|
|
117
|
+
|
|
118
|
+
if self.biff8:
|
|
119
|
+
return read_biff8_string(reader, short=short, rich=rich)
|
|
120
|
+
return read_byte_string(reader, short=short, encoding=self.encoding)
|
|
121
|
+
|
|
122
|
+
def cell_format(self, index: int) -> _CellFormat:
|
|
123
|
+
"""解析越界 XF 时返回 General 与无字体的稳定默认值。"""
|
|
124
|
+
|
|
125
|
+
if 0 <= index < len(self.formats):
|
|
126
|
+
return self.formats[index]
|
|
127
|
+
return _CellFormat(font_index=0, format_code=None)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass(slots=True)
|
|
131
|
+
class _SheetObject:
|
|
132
|
+
"""一个 OBJ 记录及其后续 TXO 可见文本。"""
|
|
133
|
+
|
|
134
|
+
object_type: int
|
|
135
|
+
object_id: int
|
|
136
|
+
checked: bool | None = None
|
|
137
|
+
embedding_storage: str | None = None
|
|
138
|
+
text: XlsRichText | None = None
|
|
139
|
+
shape: OfficeArtShape | None = None
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _error_literal(code: int) -> str | None:
|
|
143
|
+
"""把 BIFF error code 转成 Excel 可见错误文本。"""
|
|
144
|
+
|
|
145
|
+
return {
|
|
146
|
+
0x00: "#NULL!",
|
|
147
|
+
0x07: "#DIV/0!",
|
|
148
|
+
0x0F: "#VALUE!",
|
|
149
|
+
0x17: "#REF!",
|
|
150
|
+
0x1D: "#NAME?",
|
|
151
|
+
0x24: "#NUM!",
|
|
152
|
+
0x2A: "#N/A",
|
|
153
|
+
0x2B: "#GETTING_DATA",
|
|
154
|
+
}.get(int(code))
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _rk_number(value: int) -> float:
|
|
158
|
+
"""解码 RK 压缩整数或截断双精度数。"""
|
|
159
|
+
|
|
160
|
+
if value & 0x02:
|
|
161
|
+
signed = struct.unpack("<i", struct.pack("<I", value))[0]
|
|
162
|
+
number = float(signed >> 2)
|
|
163
|
+
else:
|
|
164
|
+
number = struct.unpack("<d", struct.pack("<Q", (value & 0xFFFF_FFFC) << 32))[0]
|
|
165
|
+
return number / 100.0 if value & 0x01 else number
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _read_font(payload: bytes, *, biff8: bool) -> XlsFontStyle:
|
|
169
|
+
"""提取 FONT 中可映射为 DocVortex 行内标签的字符属性。"""
|
|
170
|
+
|
|
171
|
+
if len(payload) < 11:
|
|
172
|
+
return XlsFontStyle()
|
|
173
|
+
flags = int(get_u16(payload, 2) or 0)
|
|
174
|
+
weight = int(get_u16(payload, 6) or 400)
|
|
175
|
+
escapement = int(get_u16(payload, 8) or 0)
|
|
176
|
+
underline = int(payload[10])
|
|
177
|
+
return XlsFontStyle(
|
|
178
|
+
bold=weight >= 700,
|
|
179
|
+
italic=bool(flags & 0x0002),
|
|
180
|
+
strike=bool(flags & 0x0008),
|
|
181
|
+
underline=underline != 0,
|
|
182
|
+
superscript=escapement == 1,
|
|
183
|
+
subscript=escapement == 2,
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _read_boundsheet(payload: bytes, globals_: _Globals) -> _BoundSheet | None:
|
|
188
|
+
"""解析 sheet 偏移、可见性、类型和名称。"""
|
|
189
|
+
|
|
190
|
+
if len(payload) < 8:
|
|
191
|
+
return None
|
|
192
|
+
offset = get_u32(payload, 0)
|
|
193
|
+
if offset is None:
|
|
194
|
+
return None
|
|
195
|
+
reader = SegmentReader([payload[6:]])
|
|
196
|
+
decoded = globals_.read_string(reader, short=True)
|
|
197
|
+
if decoded is None:
|
|
198
|
+
return None
|
|
199
|
+
return _BoundSheet(
|
|
200
|
+
name=clean_text(decoded.text) or "Sheet",
|
|
201
|
+
offset=int(offset),
|
|
202
|
+
visible=(payload[4] & 0x03) == 0,
|
|
203
|
+
sheet_type=int(payload[5]),
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _read_sst(segments: list[bytes]) -> list[DecodedString]:
|
|
208
|
+
"""读取共享字符串表,并允许损坏尾部保留已完成的 strings。"""
|
|
209
|
+
|
|
210
|
+
reader = SegmentReader(segments)
|
|
211
|
+
total = reader.u32()
|
|
212
|
+
unique = reader.u32()
|
|
213
|
+
if total is None or unique is None:
|
|
214
|
+
return []
|
|
215
|
+
strings: list[DecodedString] = []
|
|
216
|
+
while len(strings) < unique:
|
|
217
|
+
decoded = read_biff8_string(reader, short=False, rich=True)
|
|
218
|
+
if decoded is None:
|
|
219
|
+
logger.warning(
|
|
220
|
+
"XLS_SST_TRUNCATED: shared string table stopped at entry {}",
|
|
221
|
+
len(strings),
|
|
222
|
+
)
|
|
223
|
+
break
|
|
224
|
+
strings.append(decoded)
|
|
225
|
+
return strings
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _read_supbook(payload: bytes) -> bool:
|
|
229
|
+
"""判断 SUPBOOK 是否表示当前工作簿内部 sheet 集合。"""
|
|
230
|
+
|
|
231
|
+
return len(payload) >= 4 and get_u16(payload, 2) == 0x0401
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _read_extern_sheets(
|
|
235
|
+
payload: bytes,
|
|
236
|
+
internal_supbooks: list[bool],
|
|
237
|
+
) -> list[int | None]:
|
|
238
|
+
"""把 XTI entries 解析为内部工作表索引。"""
|
|
239
|
+
|
|
240
|
+
count = min(int(get_u16(payload, 0) or 0), max(0, (len(payload) - 2) // 6))
|
|
241
|
+
result: list[int | None] = []
|
|
242
|
+
for index in range(count):
|
|
243
|
+
offset = 2 + index * 6
|
|
244
|
+
supbook, first_sheet, last_sheet = struct.unpack_from("<3H", payload, offset)
|
|
245
|
+
if (
|
|
246
|
+
supbook < len(internal_supbooks)
|
|
247
|
+
and internal_supbooks[supbook]
|
|
248
|
+
and first_sheet == last_sheet
|
|
249
|
+
and first_sheet < 0xFFFE
|
|
250
|
+
):
|
|
251
|
+
result.append(int(first_sheet))
|
|
252
|
+
else:
|
|
253
|
+
result.append(None)
|
|
254
|
+
return result
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _read_globals(data: bytes, budget: RecordBudget) -> _Globals:
|
|
258
|
+
"""解析 Workbook Globals Substream 及共享图片资源。"""
|
|
259
|
+
|
|
260
|
+
first = record_at(data, 0, budget=budget)
|
|
261
|
+
if first is None or first.record_type != BOF:
|
|
262
|
+
raise LegacyOfficeMalformedError("workbook stream does not start with a BOF record")
|
|
263
|
+
if get_u16(first.payload, 2) not in {WORKBOOK_GLOBALS_SUBSTREAM, None}:
|
|
264
|
+
raise LegacyOfficeMalformedError("first BIFF substream is not workbook globals")
|
|
265
|
+
version = get_u16(first.payload, 0)
|
|
266
|
+
if version not in {0x0500, 0x0600}:
|
|
267
|
+
raise LegacyOfficeMalformedError(f"unsupported BIFF version: {version!r}")
|
|
268
|
+
globals_ = _Globals(biff8=version == 0x0600)
|
|
269
|
+
raw_formats: dict[int, str] = {}
|
|
270
|
+
raw_xfs: list[tuple[int, int]] = []
|
|
271
|
+
drawing_chunks: list[bytes] = []
|
|
272
|
+
internal_supbooks: list[bool] = []
|
|
273
|
+
extern_payloads: list[bytes] = []
|
|
274
|
+
cursor = first.next_offset
|
|
275
|
+
depth = 1
|
|
276
|
+
while cursor < len(data):
|
|
277
|
+
record = record_at(data, cursor, budget=budget)
|
|
278
|
+
if record is None:
|
|
279
|
+
logger.warning("XLS_GLOBALS_TRUNCATED: globals end at byte {}", cursor)
|
|
280
|
+
break
|
|
281
|
+
cursor = record.next_offset
|
|
282
|
+
if record.record_type == BOF:
|
|
283
|
+
depth += 1
|
|
284
|
+
continue
|
|
285
|
+
if record.record_type == EOF:
|
|
286
|
+
depth -= 1
|
|
287
|
+
if depth == 0:
|
|
288
|
+
break
|
|
289
|
+
continue
|
|
290
|
+
if depth != 1:
|
|
291
|
+
continue
|
|
292
|
+
if record.record_type == FILEPASS:
|
|
293
|
+
raise LegacyOfficeEncryptedError("password-protected XLS is not supported")
|
|
294
|
+
if record.record_type == CODEPAGE:
|
|
295
|
+
globals_.encoding = codepage_name(int(get_u16(record.payload, 0) or 1252))
|
|
296
|
+
elif record.record_type == DATEMODE:
|
|
297
|
+
globals_.date1904 = get_u16(record.payload, 0) == 1
|
|
298
|
+
elif record.record_type == FONT:
|
|
299
|
+
globals_.fonts.append(_read_font(record.payload, biff8=globals_.biff8))
|
|
300
|
+
elif record.record_type == BOUNDSHEET:
|
|
301
|
+
sheet = _read_boundsheet(record.payload, globals_)
|
|
302
|
+
if sheet is not None and sheet.sheet_type != 0x06:
|
|
303
|
+
globals_.sheets.append(sheet)
|
|
304
|
+
elif record.record_type == FORMAT:
|
|
305
|
+
format_id = get_u16(record.payload, 0)
|
|
306
|
+
if format_id is not None:
|
|
307
|
+
reader = SegmentReader([record.payload[2:]])
|
|
308
|
+
decoded = globals_.read_string(reader, short=not globals_.biff8)
|
|
309
|
+
if decoded is not None:
|
|
310
|
+
raw_formats[int(format_id)] = decoded.text
|
|
311
|
+
elif record.record_type == XF:
|
|
312
|
+
font_index = int(get_u16(record.payload, 0) or 0)
|
|
313
|
+
format_id = int(get_u16(record.payload, 2) or 0)
|
|
314
|
+
raw_xfs.append((font_index, format_id))
|
|
315
|
+
elif record.record_type == SST and globals_.biff8:
|
|
316
|
+
segments, cursor = collect_continues(data, record, budget=budget)
|
|
317
|
+
globals_.strings = _read_sst(segments)
|
|
318
|
+
elif record.record_type == MSODRAWINGGROUP:
|
|
319
|
+
segments, cursor = collect_continues(data, record, budget=budget)
|
|
320
|
+
drawing_chunks.extend(segments)
|
|
321
|
+
elif record.record_type == SUPBOOK:
|
|
322
|
+
internal_supbooks.append(_read_supbook(record.payload))
|
|
323
|
+
elif record.record_type == EXTERNSHEET:
|
|
324
|
+
extern_payloads.append(record.payload)
|
|
325
|
+
elif record.record_type == WINDOW1 and len(record.payload) >= 12:
|
|
326
|
+
globals_.active_sheet_index = get_u16(record.payload, 10)
|
|
327
|
+
|
|
328
|
+
globals_.formats = [
|
|
329
|
+
_CellFormat(
|
|
330
|
+
font_index=font_index,
|
|
331
|
+
format_code=raw_formats.get(format_id) or builtin_number_format(format_id),
|
|
332
|
+
)
|
|
333
|
+
for font_index, format_id in raw_xfs
|
|
334
|
+
]
|
|
335
|
+
for payload in extern_payloads:
|
|
336
|
+
globals_.extern_sheets.extend(_read_extern_sheets(payload, internal_supbooks))
|
|
337
|
+
if drawing_chunks:
|
|
338
|
+
globals_.images = decode_bstore(b"".join(drawing_chunks), charge=budget.charge)
|
|
339
|
+
return globals_
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _cell_ref(payload: bytes) -> tuple[int, int, int] | None:
|
|
343
|
+
"""读取 cell header,并拒绝超出 BIFF8 网格的列。"""
|
|
344
|
+
|
|
345
|
+
if len(payload) < 6:
|
|
346
|
+
return None
|
|
347
|
+
row, col, xf_index = struct.unpack_from("<3H", payload, 0)
|
|
348
|
+
if row >= MAX_ROWS or col >= MAX_COLS:
|
|
349
|
+
return None
|
|
350
|
+
return int(row), int(col), int(xf_index)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _resolved_rich_text(
|
|
354
|
+
decoded: DecodedString,
|
|
355
|
+
globals_: _Globals,
|
|
356
|
+
xf_index: int,
|
|
357
|
+
) -> XlsRichText:
|
|
358
|
+
"""应用 text number format,并在原文未变时保留 rich runs。"""
|
|
359
|
+
|
|
360
|
+
cell_format = globals_.cell_format(xf_index)
|
|
361
|
+
formatted = format_text(decoded.text, cell_format.format_code)
|
|
362
|
+
if formatted != decoded.text:
|
|
363
|
+
return XlsRichText(formatted)
|
|
364
|
+
return to_rich_text(decoded, globals_.fonts)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _put_cell(
|
|
368
|
+
sheet: XlsSheet,
|
|
369
|
+
row: int,
|
|
370
|
+
col: int,
|
|
371
|
+
value: XlsRichText,
|
|
372
|
+
) -> None:
|
|
373
|
+
"""仅保存非空文本,并覆盖同坐标较早的缓存记录。"""
|
|
374
|
+
|
|
375
|
+
if not value.text:
|
|
376
|
+
return
|
|
377
|
+
sheet.cells[(row, col)] = XlsCell(row=row, col=col, value=value)
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _append_cell_text(
|
|
381
|
+
sheet: XlsSheet,
|
|
382
|
+
row: int,
|
|
383
|
+
col: int,
|
|
384
|
+
value: XlsRichText,
|
|
385
|
+
) -> None:
|
|
386
|
+
"""把 drawing/control 文本追加到 anchor 单元格且平移 rich runs。"""
|
|
387
|
+
|
|
388
|
+
if not value.text:
|
|
389
|
+
return
|
|
390
|
+
existing = sheet.cells.get((row, col))
|
|
391
|
+
if existing is None:
|
|
392
|
+
_put_cell(sheet, row, col, value)
|
|
393
|
+
return
|
|
394
|
+
separator = "\n" if existing.value.text else ""
|
|
395
|
+
shift = len(existing.value.text) + len(separator)
|
|
396
|
+
shifted = tuple(
|
|
397
|
+
XlsRichRun(
|
|
398
|
+
start=run.start + shift,
|
|
399
|
+
end=run.end + shift,
|
|
400
|
+
style=run.style,
|
|
401
|
+
)
|
|
402
|
+
for run in value.runs
|
|
403
|
+
)
|
|
404
|
+
existing.value = XlsRichText(
|
|
405
|
+
text=existing.value.text + separator + value.text,
|
|
406
|
+
runs=existing.value.runs + shifted,
|
|
407
|
+
)
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _read_label_string(
|
|
411
|
+
segments: list[bytes],
|
|
412
|
+
globals_: _Globals,
|
|
413
|
+
*,
|
|
414
|
+
rich_record: bool,
|
|
415
|
+
) -> DecodedString | None:
|
|
416
|
+
"""读取 LABEL/RSTRING 的字符串并恢复 RSTRING formatting runs。"""
|
|
417
|
+
|
|
418
|
+
if not segments or len(segments[0]) < 6:
|
|
419
|
+
return None
|
|
420
|
+
adjusted = [segments[0][6:], *segments[1:]]
|
|
421
|
+
reader = SegmentReader(adjusted)
|
|
422
|
+
decoded = globals_.read_string(reader, short=False)
|
|
423
|
+
if decoded is None or not rich_record:
|
|
424
|
+
return decoded
|
|
425
|
+
run_count = reader.u16()
|
|
426
|
+
if run_count is None:
|
|
427
|
+
return decoded
|
|
428
|
+
starts: list[tuple[int, int]] = []
|
|
429
|
+
for _ in range(run_count):
|
|
430
|
+
raw = reader.read_across(4)
|
|
431
|
+
if raw is None:
|
|
432
|
+
break
|
|
433
|
+
character_index, font_index = struct.unpack("<HH", raw)
|
|
434
|
+
starts.append((int(character_index), int(font_index)))
|
|
435
|
+
return DecodedString(decoded.text, tuple(starts))
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def _pict_embedding_storage(payload: bytes, picture_flags: int | None) -> str | None:
|
|
439
|
+
"""从 FtPictFmla 读取嵌入对象的 MBD storage 名称。"""
|
|
440
|
+
|
|
441
|
+
if picture_flags is None:
|
|
442
|
+
return None
|
|
443
|
+
# DDE、ActiveX、controls stream 与 camera picture 都不是内嵌公式 OLE 对象。
|
|
444
|
+
if picture_flags & (0x0002 | 0x0010 | 0x0020 | 0x0080):
|
|
445
|
+
return None
|
|
446
|
+
if len(payload) < 10:
|
|
447
|
+
return None
|
|
448
|
+
cb_fmla = int(get_u16(payload, 0) or 0)
|
|
449
|
+
formula_end = 2 + cb_fmla
|
|
450
|
+
if cb_fmla <= 0 or cb_fmla % 2 or formula_end + 4 > len(payload):
|
|
451
|
+
return None
|
|
452
|
+
formula = payload[2:formula_end]
|
|
453
|
+
if len(formula) < 7 or int(get_u16(formula, 0) or 0) & 0x7FFF != 5:
|
|
454
|
+
return None
|
|
455
|
+
# ObjectParsedFormula 的四字节 unused 在部分生产器中省略,因此兼容两个合法落点。
|
|
456
|
+
if not any(offset + 5 <= len(formula) and formula[offset] == 0x02 for offset in (6, 2)):
|
|
457
|
+
return None
|
|
458
|
+
location = get_u32(payload, formula_end)
|
|
459
|
+
return f"MBD{int(location):08X}" if location is not None else None
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _read_obj(payload: bytes) -> _SheetObject | None:
|
|
463
|
+
"""解析 OBJ subrecords 中的对象类型、id、状态与嵌入 storage。"""
|
|
464
|
+
|
|
465
|
+
cursor = 0
|
|
466
|
+
object_type: int | None = None
|
|
467
|
+
object_id = 0
|
|
468
|
+
checked: bool | None = None
|
|
469
|
+
picture_flags: int | None = None
|
|
470
|
+
picture_formula: bytes | None = None
|
|
471
|
+
while cursor + 4 <= len(payload):
|
|
472
|
+
sub_type, length = struct.unpack_from("<HH", payload, cursor)
|
|
473
|
+
data_start = cursor + 4
|
|
474
|
+
data_end = data_start + int(length)
|
|
475
|
+
if data_end > len(payload):
|
|
476
|
+
break
|
|
477
|
+
body = payload[data_start:data_end]
|
|
478
|
+
if sub_type == 0x0015 and len(body) >= 4:
|
|
479
|
+
object_type, object_id = struct.unpack_from("<HH", body, 0)
|
|
480
|
+
elif sub_type == 0x0012 and len(body) >= 2:
|
|
481
|
+
state = int(get_u16(body, 0) or 0)
|
|
482
|
+
checked = state == 1 if state in {0, 1} else None
|
|
483
|
+
elif sub_type == 0x0008 and len(body) >= 2:
|
|
484
|
+
picture_flags = int(get_u16(body, 0) or 0)
|
|
485
|
+
elif sub_type == 0x0009:
|
|
486
|
+
picture_formula = body
|
|
487
|
+
if sub_type == 0:
|
|
488
|
+
break
|
|
489
|
+
cursor = data_end
|
|
490
|
+
if object_type is None:
|
|
491
|
+
return None
|
|
492
|
+
return _SheetObject(
|
|
493
|
+
int(object_type),
|
|
494
|
+
int(object_id),
|
|
495
|
+
checked=checked,
|
|
496
|
+
embedding_storage=(_pict_embedding_storage(picture_formula, picture_flags) if picture_formula is not None else None),
|
|
497
|
+
)
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def _read_hyperlink_unicode(payload: bytes, cursor: int) -> tuple[str | None, int]:
|
|
501
|
+
"""读取 Hyperlink Object 中含末尾 NUL 的 UTF-16 字符串。"""
|
|
502
|
+
|
|
503
|
+
if cursor + 4 > len(payload):
|
|
504
|
+
return None, len(payload)
|
|
505
|
+
character_count = int(struct.unpack_from("<I", payload, cursor)[0])
|
|
506
|
+
cursor += 4
|
|
507
|
+
byte_count = character_count * 2
|
|
508
|
+
if byte_count < 0 or cursor + byte_count > len(payload):
|
|
509
|
+
return None, len(payload)
|
|
510
|
+
text = payload[cursor : cursor + byte_count].decode("utf-16le", "replace").rstrip("\x00")
|
|
511
|
+
return clean_text(text), cursor + byte_count
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def _read_url_moniker(payload: bytes, cursor: int) -> tuple[str | None, int]:
|
|
515
|
+
"""读取 URL Moniker 的 UTF-16 URL,忽略可选尾部元数据。"""
|
|
516
|
+
|
|
517
|
+
if cursor + 4 > len(payload):
|
|
518
|
+
return None, len(payload)
|
|
519
|
+
byte_count = int(struct.unpack_from("<I", payload, cursor)[0])
|
|
520
|
+
cursor += 4
|
|
521
|
+
if byte_count < 0 or cursor + byte_count > len(payload):
|
|
522
|
+
return None, len(payload)
|
|
523
|
+
raw = payload[cursor : cursor + byte_count]
|
|
524
|
+
usable = raw[: len(raw) - (len(raw) % 2)]
|
|
525
|
+
text = usable.decode("utf-16le", "replace").split("\x00", 1)[0]
|
|
526
|
+
return clean_text(text), cursor + byte_count
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def _read_file_moniker(payload: bytes, cursor: int) -> tuple[str | None, int]:
|
|
530
|
+
"""尽力读取 File Moniker 的 ANSI 或 Unicode 路径。"""
|
|
531
|
+
|
|
532
|
+
if cursor + 6 > len(payload):
|
|
533
|
+
return None, len(payload)
|
|
534
|
+
anti_count = int(struct.unpack_from("<H", payload, cursor)[0])
|
|
535
|
+
ansi_length = int(struct.unpack_from("<I", payload, cursor + 2)[0])
|
|
536
|
+
cursor += 6
|
|
537
|
+
if cursor + ansi_length > len(payload):
|
|
538
|
+
return None, len(payload)
|
|
539
|
+
ansi = payload[cursor : cursor + ansi_length].split(b"\x00", 1)[0]
|
|
540
|
+
cursor += ansi_length
|
|
541
|
+
path = ("../" * anti_count) + ansi.decode("cp1252", "replace")
|
|
542
|
+
return clean_text(path), cursor
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def _read_hyperlink_target(payload: bytes) -> str | None:
|
|
546
|
+
"""解析 HLink 中的 Hyperlink Object 并返回经过白名单过滤的目标。"""
|
|
547
|
+
|
|
548
|
+
if len(payload) < 32:
|
|
549
|
+
return None
|
|
550
|
+
cursor = 24
|
|
551
|
+
version, flags = struct.unpack_from("<II", payload, cursor)
|
|
552
|
+
cursor += 8
|
|
553
|
+
if version != 2:
|
|
554
|
+
return None
|
|
555
|
+
if flags & 0x10:
|
|
556
|
+
_, cursor = _read_hyperlink_unicode(payload, cursor)
|
|
557
|
+
if flags & 0x80:
|
|
558
|
+
_, cursor = _read_hyperlink_unicode(payload, cursor)
|
|
559
|
+
target: str | None = None
|
|
560
|
+
blocked_local_file = False
|
|
561
|
+
if flags & 0x01:
|
|
562
|
+
if flags & 0x0100:
|
|
563
|
+
target, cursor = _read_hyperlink_unicode(payload, cursor)
|
|
564
|
+
elif cursor + 16 <= len(payload):
|
|
565
|
+
moniker = uuid.UUID(bytes_le=payload[cursor : cursor + 16])
|
|
566
|
+
cursor += 16
|
|
567
|
+
if moniker == uuid.UUID("79eac9e0-baf9-11ce-8c82-00aa004ba90b"):
|
|
568
|
+
target, cursor = _read_url_moniker(payload, cursor)
|
|
569
|
+
elif moniker == uuid.UUID("00000303-0000-0000-c000-000000000046"):
|
|
570
|
+
_, cursor = _read_file_moniker(payload, cursor)
|
|
571
|
+
blocked_local_file = True
|
|
572
|
+
location: str | None = None
|
|
573
|
+
if flags & 0x08:
|
|
574
|
+
location, cursor = _read_hyperlink_unicode(payload, cursor)
|
|
575
|
+
if location:
|
|
576
|
+
target = f"{target}#{location}" if target else f"#{location}"
|
|
577
|
+
if blocked_local_file:
|
|
578
|
+
logger.warning("XLS_HYPERLINK_BLOCKED: local File Moniker")
|
|
579
|
+
return None
|
|
580
|
+
sanitized = sanitize_hyperlink_target(
|
|
581
|
+
target,
|
|
582
|
+
allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
|
|
583
|
+
allow_relative=True,
|
|
584
|
+
allow_fragment=True,
|
|
585
|
+
)
|
|
586
|
+
if target and sanitized is None:
|
|
587
|
+
logger.warning("XLS_HYPERLINK_BLOCKED: target={!r}", target)
|
|
588
|
+
return sanitized
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def _apply_hlink(
|
|
592
|
+
sheet: XlsSheet,
|
|
593
|
+
payload: bytes,
|
|
594
|
+
pending: dict[tuple[int, int], str],
|
|
595
|
+
) -> None:
|
|
596
|
+
"""把 HLink 范围目标暂存到所有覆盖单元格。"""
|
|
597
|
+
|
|
598
|
+
if len(payload) < 8:
|
|
599
|
+
return
|
|
600
|
+
row_first, row_last, col_first, col_last = struct.unpack_from("<4H", payload, 0)
|
|
601
|
+
target = _read_hyperlink_target(payload)
|
|
602
|
+
if target is None:
|
|
603
|
+
return
|
|
604
|
+
for row in range(min(row_first, row_last), min(max(row_first, row_last), MAX_ROWS - 1) + 1):
|
|
605
|
+
for col in range(min(col_first, col_last), min(max(col_first, col_last), MAX_COLS - 1) + 1):
|
|
606
|
+
pending[(int(row), int(col))] = target
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
def _shape_anchor(shape: OfficeArtShape | None) -> tuple[int, int] | None:
|
|
610
|
+
"""返回 shape 左上角 cell anchor。"""
|
|
611
|
+
|
|
612
|
+
if shape is None or shape.anchor is None:
|
|
613
|
+
return None
|
|
614
|
+
return shape.anchor[0], shape.anchor[1]
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
def _serialize_payload(payload: OfficeImagePayload) -> str | None:
|
|
618
|
+
"""使用共享 Office 图片策略序列化 BLIP。"""
|
|
619
|
+
|
|
620
|
+
return serialize_office_image(
|
|
621
|
+
payload.data,
|
|
622
|
+
part_name=f"picture.{payload.extension}",
|
|
623
|
+
content_type=payload.content_type,
|
|
624
|
+
render_size_emu=payload.render_size_emu,
|
|
625
|
+
)
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def _bind_objects(
|
|
629
|
+
sheet: XlsSheet,
|
|
630
|
+
objects: list[_SheetObject],
|
|
631
|
+
drawing_data: bytes,
|
|
632
|
+
chart_streams: list[list[BiffRecord]],
|
|
633
|
+
*,
|
|
634
|
+
globals_: _Globals,
|
|
635
|
+
sheet_index: int,
|
|
636
|
+
native_equations: dict[str, str],
|
|
637
|
+
image_equation_decoder: OfficeImageEquationDecoder,
|
|
638
|
+
budget: RecordBudget,
|
|
639
|
+
) -> None:
|
|
640
|
+
"""按 drawing/OBJ 顺序绑定文本框、复选框、图片与嵌入图表。"""
|
|
641
|
+
|
|
642
|
+
shapes = extract_excel_shapes(drawing_data, charge=budget.charge) if drawing_data else []
|
|
643
|
+
if len(shapes) != len(objects):
|
|
644
|
+
logger.warning(
|
|
645
|
+
"XLS_DRAWING_OBJECT_MISMATCH: sheet={!r}, shapes={}, objects={}",
|
|
646
|
+
sheet.name,
|
|
647
|
+
len(shapes),
|
|
648
|
+
len(objects),
|
|
649
|
+
)
|
|
650
|
+
for object_, shape in zip(objects, shapes, strict=False):
|
|
651
|
+
object_.shape = shape
|
|
652
|
+
chart_objects = [object_ for object_ in objects if object_.object_type == OBJ_CHART]
|
|
653
|
+
for object_ in objects:
|
|
654
|
+
shape = object_.shape
|
|
655
|
+
if shape is None or shape.hidden:
|
|
656
|
+
continue
|
|
657
|
+
anchor = _shape_anchor(shape)
|
|
658
|
+
if anchor is None:
|
|
659
|
+
continue
|
|
660
|
+
row, col = anchor
|
|
661
|
+
equation = (
|
|
662
|
+
native_equations.get(object_.embedding_storage.casefold())
|
|
663
|
+
if object_.object_type == OBJ_PICTURE and object_.embedding_storage
|
|
664
|
+
else None
|
|
665
|
+
)
|
|
666
|
+
if equation:
|
|
667
|
+
sheet.equations.append(XlsEquation(row=row, col=col, latex=equation))
|
|
668
|
+
continue
|
|
669
|
+
if (
|
|
670
|
+
object_.object_type == OBJ_PICTURE
|
|
671
|
+
and shape.pib is not None
|
|
672
|
+
and (payload := globals_.images.get(int(shape.pib))) is not None
|
|
673
|
+
):
|
|
674
|
+
image_latex = image_equation_decoder.decode(
|
|
675
|
+
payload.data,
|
|
676
|
+
part_name=f"picture.{payload.extension}",
|
|
677
|
+
content_type=payload.content_type,
|
|
678
|
+
)
|
|
679
|
+
if image_latex:
|
|
680
|
+
sheet.equations.append(XlsEquation(row=row, col=col, latex=image_latex))
|
|
681
|
+
continue
|
|
682
|
+
if object_.object_type == OBJ_TEXTBOX and object_.text is not None:
|
|
683
|
+
_append_cell_text(sheet, row, col, object_.text)
|
|
684
|
+
elif object_.object_type == OBJ_CHECKBOX and object_.checked is not None:
|
|
685
|
+
marker = "[x]" if object_.checked else "[ ]"
|
|
686
|
+
caption = object_.text.text.strip() if object_.text is not None else ""
|
|
687
|
+
_append_cell_text(sheet, row, col, XlsRichText(f"{marker} {caption}".rstrip()))
|
|
688
|
+
if object_.object_type in {OBJ_PICTURE, OBJ_CHART} or shape.pib is not None:
|
|
689
|
+
payload = globals_.images.get(int(shape.pib or 0))
|
|
690
|
+
if payload is not None:
|
|
691
|
+
image_base64 = _serialize_payload(payload)
|
|
692
|
+
if image_base64 and object_.object_type != OBJ_CHART:
|
|
693
|
+
sheet.images.append(XlsImage(row=row, col=col, image_base64=image_base64))
|
|
694
|
+
|
|
695
|
+
for index, object_ in enumerate(chart_objects):
|
|
696
|
+
if object_.shape is None or object_.shape.hidden:
|
|
697
|
+
continue
|
|
698
|
+
anchor = _shape_anchor(object_.shape)
|
|
699
|
+
if anchor is None:
|
|
700
|
+
continue
|
|
701
|
+
axes = (
|
|
702
|
+
chart_source_axes(
|
|
703
|
+
chart_streams[index],
|
|
704
|
+
current_sheet_index=sheet_index,
|
|
705
|
+
extern_sheets=globals_.extern_sheets,
|
|
706
|
+
)
|
|
707
|
+
if index < len(chart_streams)
|
|
708
|
+
else None
|
|
709
|
+
)
|
|
710
|
+
preview: str | None = None
|
|
711
|
+
if object_.shape.pib is not None:
|
|
712
|
+
payload = globals_.images.get(object_.shape.pib)
|
|
713
|
+
preview = _serialize_payload(payload) if payload is not None else None
|
|
714
|
+
if axes is None:
|
|
715
|
+
if preview:
|
|
716
|
+
sheet.charts.append(
|
|
717
|
+
XlsChart(
|
|
718
|
+
row=anchor[0],
|
|
719
|
+
col=anchor[1],
|
|
720
|
+
source_rows=(),
|
|
721
|
+
source_cols=(),
|
|
722
|
+
image_base64=preview,
|
|
723
|
+
)
|
|
724
|
+
)
|
|
725
|
+
else:
|
|
726
|
+
logger.warning(
|
|
727
|
+
"XLS_CHART_SOURCE_UNSUPPORTED: sheet={!r}, object_id={}",
|
|
728
|
+
sheet.name,
|
|
729
|
+
object_.object_id,
|
|
730
|
+
)
|
|
731
|
+
continue
|
|
732
|
+
rows, cols = axes
|
|
733
|
+
sheet.charts.append(
|
|
734
|
+
XlsChart(
|
|
735
|
+
row=anchor[0],
|
|
736
|
+
col=anchor[1],
|
|
737
|
+
source_rows=tuple(rows),
|
|
738
|
+
source_cols=tuple(cols),
|
|
739
|
+
image_base64=preview,
|
|
740
|
+
)
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
def _read_sheet(
|
|
745
|
+
data: bytes,
|
|
746
|
+
globals_: _Globals,
|
|
747
|
+
descriptor: _BoundSheet,
|
|
748
|
+
offset: int,
|
|
749
|
+
*,
|
|
750
|
+
sheet_index: int,
|
|
751
|
+
recovered: bool,
|
|
752
|
+
native_equations: dict[str, str],
|
|
753
|
+
image_equation_decoder: OfficeImageEquationDecoder,
|
|
754
|
+
budget: RecordBudget,
|
|
755
|
+
) -> XlsSheet | None:
|
|
756
|
+
"""解析一个 worksheet substream 并绑定其 drawing/chart 对象。"""
|
|
757
|
+
|
|
758
|
+
first = record_at(data, offset, budget=budget)
|
|
759
|
+
if first is None or first.record_type != BOF or get_u16(first.payload, 2) != WORKSHEET_SUBSTREAM:
|
|
760
|
+
return None
|
|
761
|
+
sheet = XlsSheet(
|
|
762
|
+
name=descriptor.name,
|
|
763
|
+
visible=descriptor.visible,
|
|
764
|
+
order=sheet_index,
|
|
765
|
+
recovered=recovered,
|
|
766
|
+
)
|
|
767
|
+
cursor = first.next_offset
|
|
768
|
+
depth = 1
|
|
769
|
+
active_chart: list[BiffRecord] | None = None
|
|
770
|
+
chart_streams: list[list[BiffRecord]] = []
|
|
771
|
+
drawing_chunks: list[bytes] = []
|
|
772
|
+
objects: list[_SheetObject] = []
|
|
773
|
+
pending_formula: tuple[int, int, int] | None = None
|
|
774
|
+
pending_links: dict[tuple[int, int], str] = {}
|
|
775
|
+
|
|
776
|
+
while cursor < len(data):
|
|
777
|
+
record = record_at(data, cursor, budget=budget)
|
|
778
|
+
if record is None:
|
|
779
|
+
logger.warning(
|
|
780
|
+
"XLS_SHEET_TRUNCATED: sheet={!r}, byte={}",
|
|
781
|
+
sheet.name,
|
|
782
|
+
cursor,
|
|
783
|
+
)
|
|
784
|
+
break
|
|
785
|
+
cursor = record.next_offset
|
|
786
|
+
if active_chart is not None:
|
|
787
|
+
active_chart.append(record)
|
|
788
|
+
if record.record_type == BOF:
|
|
789
|
+
depth += 1
|
|
790
|
+
elif record.record_type == EOF:
|
|
791
|
+
depth -= 1
|
|
792
|
+
if depth == 1:
|
|
793
|
+
chart_streams.append(active_chart)
|
|
794
|
+
active_chart = None
|
|
795
|
+
continue
|
|
796
|
+
if record.record_type == BOF:
|
|
797
|
+
depth += 1
|
|
798
|
+
if depth == 2 and get_u16(record.payload, 2) == CHART_SUBSTREAM:
|
|
799
|
+
active_chart = [record]
|
|
800
|
+
continue
|
|
801
|
+
if record.record_type == EOF:
|
|
802
|
+
depth -= 1
|
|
803
|
+
if depth == 0:
|
|
804
|
+
break
|
|
805
|
+
continue
|
|
806
|
+
if depth != 1:
|
|
807
|
+
continue
|
|
808
|
+
|
|
809
|
+
if record.record_type == MERGEDCELLS:
|
|
810
|
+
count = min(int(get_u16(record.payload, 0) or 0), max(0, (len(record.payload) - 2) // 8))
|
|
811
|
+
for index in range(count):
|
|
812
|
+
row_first, row_last, col_first, col_last = struct.unpack_from("<4H", record.payload, 2 + index * 8)
|
|
813
|
+
row_start, row_end = sorted((int(row_first), int(row_last)))
|
|
814
|
+
col_start, col_end = sorted((int(col_first), int(col_last)))
|
|
815
|
+
if col_start >= MAX_COLS or (row_start == row_end and col_start == col_end):
|
|
816
|
+
continue
|
|
817
|
+
sheet.merges.append(
|
|
818
|
+
(
|
|
819
|
+
row_start,
|
|
820
|
+
col_start,
|
|
821
|
+
min(row_end, MAX_ROWS - 1),
|
|
822
|
+
min(col_end, MAX_COLS - 1),
|
|
823
|
+
)
|
|
824
|
+
)
|
|
825
|
+
elif record.record_type == LABELSST:
|
|
826
|
+
reference = _cell_ref(record.payload)
|
|
827
|
+
string_index = get_u32(record.payload, 6)
|
|
828
|
+
if reference is not None and string_index is not None and string_index < len(globals_.strings):
|
|
829
|
+
row, col, xf_index = reference
|
|
830
|
+
_put_cell(
|
|
831
|
+
sheet,
|
|
832
|
+
row,
|
|
833
|
+
col,
|
|
834
|
+
_resolved_rich_text(globals_.strings[int(string_index)], globals_, xf_index),
|
|
835
|
+
)
|
|
836
|
+
elif record.record_type in {LABEL, RSTRING}:
|
|
837
|
+
segments, cursor = collect_continues(data, record, budget=budget)
|
|
838
|
+
reference = _cell_ref(record.payload)
|
|
839
|
+
decoded = _read_label_string(
|
|
840
|
+
segments,
|
|
841
|
+
globals_,
|
|
842
|
+
rich_record=record.record_type == RSTRING,
|
|
843
|
+
)
|
|
844
|
+
if reference is not None and decoded is not None:
|
|
845
|
+
row, col, xf_index = reference
|
|
846
|
+
_put_cell(sheet, row, col, _resolved_rich_text(decoded, globals_, xf_index))
|
|
847
|
+
elif record.record_type == NUMBER:
|
|
848
|
+
reference = _cell_ref(record.payload)
|
|
849
|
+
value = get_f64(record.payload, 6)
|
|
850
|
+
if reference is not None and value is not None:
|
|
851
|
+
row, col, xf_index = reference
|
|
852
|
+
cell_format = globals_.cell_format(xf_index)
|
|
853
|
+
_put_cell(
|
|
854
|
+
sheet,
|
|
855
|
+
row,
|
|
856
|
+
col,
|
|
857
|
+
XlsRichText(
|
|
858
|
+
format_number(
|
|
859
|
+
value,
|
|
860
|
+
cell_format.format_code,
|
|
861
|
+
date1904=globals_.date1904,
|
|
862
|
+
)
|
|
863
|
+
),
|
|
864
|
+
)
|
|
865
|
+
elif record.record_type == RK:
|
|
866
|
+
reference = _cell_ref(record.payload)
|
|
867
|
+
raw_value = get_u32(record.payload, 6)
|
|
868
|
+
if reference is not None and raw_value is not None:
|
|
869
|
+
row, col, xf_index = reference
|
|
870
|
+
cell_format = globals_.cell_format(xf_index)
|
|
871
|
+
_put_cell(
|
|
872
|
+
sheet,
|
|
873
|
+
row,
|
|
874
|
+
col,
|
|
875
|
+
XlsRichText(
|
|
876
|
+
format_number(
|
|
877
|
+
_rk_number(raw_value),
|
|
878
|
+
cell_format.format_code,
|
|
879
|
+
date1904=globals_.date1904,
|
|
880
|
+
)
|
|
881
|
+
),
|
|
882
|
+
)
|
|
883
|
+
elif record.record_type == MULRK and len(record.payload) >= 6:
|
|
884
|
+
row = int(get_u16(record.payload, 0) or 0)
|
|
885
|
+
first_col = int(get_u16(record.payload, 2) or 0)
|
|
886
|
+
pair_count = max(0, (len(record.payload) - 6) // 6)
|
|
887
|
+
for index in range(pair_count):
|
|
888
|
+
xf_index = int(get_u16(record.payload, 4 + index * 6) or 0)
|
|
889
|
+
raw_value = get_u32(record.payload, 6 + index * 6)
|
|
890
|
+
col = first_col + index
|
|
891
|
+
if raw_value is None or col >= MAX_COLS:
|
|
892
|
+
break
|
|
893
|
+
cell_format = globals_.cell_format(xf_index)
|
|
894
|
+
_put_cell(
|
|
895
|
+
sheet,
|
|
896
|
+
row,
|
|
897
|
+
col,
|
|
898
|
+
XlsRichText(
|
|
899
|
+
format_number(
|
|
900
|
+
_rk_number(raw_value),
|
|
901
|
+
cell_format.format_code,
|
|
902
|
+
date1904=globals_.date1904,
|
|
903
|
+
)
|
|
904
|
+
),
|
|
905
|
+
)
|
|
906
|
+
elif record.record_type == BOOLERR:
|
|
907
|
+
reference = _cell_ref(record.payload)
|
|
908
|
+
if reference is not None and len(record.payload) >= 8:
|
|
909
|
+
row, col, _ = reference
|
|
910
|
+
value, is_error = record.payload[6], record.payload[7]
|
|
911
|
+
text = _error_literal(value) if is_error == 1 else ("TRUE" if value else "FALSE")
|
|
912
|
+
if text:
|
|
913
|
+
_put_cell(sheet, row, col, XlsRichText(text))
|
|
914
|
+
elif record.record_type == FORMULA:
|
|
915
|
+
reference = _cell_ref(record.payload)
|
|
916
|
+
if reference is not None and len(record.payload) >= 14:
|
|
917
|
+
row, col, xf_index = reference
|
|
918
|
+
cached = record.payload[6:14]
|
|
919
|
+
if cached[6:8] == b"\xff\xff":
|
|
920
|
+
kind = cached[0]
|
|
921
|
+
if kind == 0x00:
|
|
922
|
+
pending_formula = (row, col, xf_index)
|
|
923
|
+
elif kind == 0x01:
|
|
924
|
+
_put_cell(sheet, row, col, XlsRichText("TRUE" if cached[2] else "FALSE"))
|
|
925
|
+
elif kind == 0x02:
|
|
926
|
+
error = _error_literal(cached[2])
|
|
927
|
+
if error:
|
|
928
|
+
_put_cell(sheet, row, col, XlsRichText(error))
|
|
929
|
+
else:
|
|
930
|
+
value = get_f64(record.payload, 6)
|
|
931
|
+
if value is not None:
|
|
932
|
+
cell_format = globals_.cell_format(xf_index)
|
|
933
|
+
_put_cell(
|
|
934
|
+
sheet,
|
|
935
|
+
row,
|
|
936
|
+
col,
|
|
937
|
+
XlsRichText(
|
|
938
|
+
format_number(
|
|
939
|
+
value,
|
|
940
|
+
cell_format.format_code,
|
|
941
|
+
date1904=globals_.date1904,
|
|
942
|
+
)
|
|
943
|
+
),
|
|
944
|
+
)
|
|
945
|
+
elif record.record_type == STRING:
|
|
946
|
+
segments, cursor = collect_continues(data, record, budget=budget)
|
|
947
|
+
if pending_formula is not None:
|
|
948
|
+
reader = SegmentReader(segments)
|
|
949
|
+
decoded = globals_.read_string(reader, short=False)
|
|
950
|
+
if decoded is not None:
|
|
951
|
+
if decoded.text.lstrip().upper().startswith(("=DISPIMG(", "=_XLFN.DISPIMG(")):
|
|
952
|
+
# 旧版文件无法携带现代 DISPIMG 计算语义,按 Office 回存结果稳定降级。
|
|
953
|
+
decoded = DecodedString("#NAME?")
|
|
954
|
+
row, col, xf_index = pending_formula
|
|
955
|
+
_put_cell(sheet, row, col, _resolved_rich_text(decoded, globals_, xf_index))
|
|
956
|
+
pending_formula = None
|
|
957
|
+
elif record.record_type == MSODRAWING:
|
|
958
|
+
segments, cursor = collect_continues(data, record, budget=budget)
|
|
959
|
+
drawing_chunks.extend(segments)
|
|
960
|
+
elif record.record_type == OBJ:
|
|
961
|
+
object_ = _read_obj(record.payload)
|
|
962
|
+
if object_ is not None:
|
|
963
|
+
objects.append(object_)
|
|
964
|
+
elif record.record_type == TXO:
|
|
965
|
+
segments, cursor = collect_continues(data, record, budget=budget)
|
|
966
|
+
if objects:
|
|
967
|
+
objects[-1].text = read_txo_text(
|
|
968
|
+
record.payload,
|
|
969
|
+
segments[1:],
|
|
970
|
+
globals_.fonts,
|
|
971
|
+
)
|
|
972
|
+
elif record.record_type == HLINK:
|
|
973
|
+
_apply_hlink(sheet, record.payload, pending_links)
|
|
974
|
+
elif record.record_type in {ROW, COLINFO, CONTINUE}:
|
|
975
|
+
# 用户明确要求隐藏行列中的内容仍参与表格重建。
|
|
976
|
+
pass
|
|
977
|
+
|
|
978
|
+
for coordinate, target in pending_links.items():
|
|
979
|
+
cell = sheet.cells.get(coordinate)
|
|
980
|
+
if cell is not None:
|
|
981
|
+
cell.hyperlink = target
|
|
982
|
+
_bind_objects(
|
|
983
|
+
sheet,
|
|
984
|
+
objects,
|
|
985
|
+
b"".join(drawing_chunks),
|
|
986
|
+
chart_streams,
|
|
987
|
+
globals_=globals_,
|
|
988
|
+
sheet_index=sheet_index,
|
|
989
|
+
native_equations=native_equations,
|
|
990
|
+
image_equation_decoder=image_equation_decoder,
|
|
991
|
+
budget=budget,
|
|
992
|
+
)
|
|
993
|
+
return sheet
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
def _worksheet_bof_offsets(data: bytes) -> list[int]:
|
|
997
|
+
"""扫描所有可识别 worksheet BOF 偏移,供坏目录恢复使用。"""
|
|
998
|
+
|
|
999
|
+
return [
|
|
1000
|
+
record.offset
|
|
1001
|
+
for record in iter_records(data)
|
|
1002
|
+
if record.record_type == BOF and get_u16(record.payload, 2) == WORKSHEET_SUBSTREAM
|
|
1003
|
+
]
|
|
1004
|
+
|
|
1005
|
+
|
|
1006
|
+
def _is_worksheet_offset(data: bytes, offset: int) -> bool:
|
|
1007
|
+
"""判断 BoundSheet offset 是否精确指向 worksheet BOF。"""
|
|
1008
|
+
|
|
1009
|
+
record = record_at(data, offset)
|
|
1010
|
+
return bool(record is not None and record.record_type == BOF and get_u16(record.payload, 2) == WORKSHEET_SUBSTREAM)
|
|
1011
|
+
|
|
1012
|
+
|
|
1013
|
+
def _parse_chart_sheets(
|
|
1014
|
+
data: bytes,
|
|
1015
|
+
globals_: _Globals,
|
|
1016
|
+
*,
|
|
1017
|
+
budget: RecordBudget,
|
|
1018
|
+
) -> list[XlsChartSheet]:
|
|
1019
|
+
"""解析独立 chart sheet,并把 BRAI 引用绑定到唯一 worksheet。"""
|
|
1020
|
+
|
|
1021
|
+
worksheet_names = {
|
|
1022
|
+
index: descriptor.name for index, descriptor in enumerate(globals_.sheets) if descriptor.sheet_type == 0x00
|
|
1023
|
+
}
|
|
1024
|
+
chart_sheets: list[XlsChartSheet] = []
|
|
1025
|
+
for order, descriptor in enumerate(globals_.sheets):
|
|
1026
|
+
if descriptor.sheet_type != 0x02:
|
|
1027
|
+
continue
|
|
1028
|
+
first = record_at(data, descriptor.offset, budget=budget)
|
|
1029
|
+
selection = None
|
|
1030
|
+
if first is not None and first.record_type == BOF and get_u16(first.payload, 2) == CHART_SUBSTREAM:
|
|
1031
|
+
records = list(
|
|
1032
|
+
iter_records(
|
|
1033
|
+
data,
|
|
1034
|
+
start=descriptor.offset,
|
|
1035
|
+
stop_at_eof=True,
|
|
1036
|
+
budget=budget,
|
|
1037
|
+
)
|
|
1038
|
+
)
|
|
1039
|
+
selection = chart_source_selection(
|
|
1040
|
+
records,
|
|
1041
|
+
current_sheet_index=order,
|
|
1042
|
+
extern_sheets=globals_.extern_sheets,
|
|
1043
|
+
)
|
|
1044
|
+
source_name = worksheet_names.get(selection.sheet_index) if selection is not None else None
|
|
1045
|
+
chart_sheets.append(
|
|
1046
|
+
XlsChartSheet(
|
|
1047
|
+
name=descriptor.name,
|
|
1048
|
+
visible=descriptor.visible,
|
|
1049
|
+
order=order,
|
|
1050
|
+
source_sheet_name=source_name,
|
|
1051
|
+
source_rows=(selection.rows if selection is not None and source_name is not None else ()),
|
|
1052
|
+
source_cols=(selection.cols if selection is not None and source_name is not None else ()),
|
|
1053
|
+
)
|
|
1054
|
+
)
|
|
1055
|
+
return chart_sheets
|
|
1056
|
+
|
|
1057
|
+
|
|
1058
|
+
def parse_xls_workbook(
|
|
1059
|
+
data: bytes,
|
|
1060
|
+
*,
|
|
1061
|
+
native_equations: dict[str, str] | None = None,
|
|
1062
|
+
) -> XlsWorkbook:
|
|
1063
|
+
"""解析 Workbook/Book stream,并按目录顺序恢复 worksheets 与原生公式。"""
|
|
1064
|
+
|
|
1065
|
+
if not data:
|
|
1066
|
+
raise LegacyOfficeMalformedError("empty Workbook stream")
|
|
1067
|
+
budget = RecordBudget()
|
|
1068
|
+
normalized_equations = {storage.casefold(): latex for storage, latex in (native_equations or {}).items()}
|
|
1069
|
+
image_equation_decoder = OfficeImageEquationDecoder()
|
|
1070
|
+
globals_ = _read_globals(data, budget)
|
|
1071
|
+
candidates = _worksheet_bof_offsets(data)
|
|
1072
|
+
used_offsets: set[int] = set()
|
|
1073
|
+
sheets: list[XlsSheet] = []
|
|
1074
|
+
|
|
1075
|
+
descriptor_entries = [(index, sheet) for index, sheet in enumerate(globals_.sheets) if sheet.sheet_type == 0x00]
|
|
1076
|
+
if not descriptor_entries and candidates:
|
|
1077
|
+
descriptor_entries = [
|
|
1078
|
+
(
|
|
1079
|
+
index - 1,
|
|
1080
|
+
_BoundSheet(
|
|
1081
|
+
name=f"Recovered Sheet {index}",
|
|
1082
|
+
offset=offset,
|
|
1083
|
+
visible=True,
|
|
1084
|
+
sheet_type=0,
|
|
1085
|
+
),
|
|
1086
|
+
)
|
|
1087
|
+
for index, offset in enumerate(candidates, start=1)
|
|
1088
|
+
]
|
|
1089
|
+
if not descriptor_entries and not candidates:
|
|
1090
|
+
raise LegacyOfficeMalformedError("workbook contains no worksheet substream")
|
|
1091
|
+
|
|
1092
|
+
for sheet_index, descriptor in descriptor_entries:
|
|
1093
|
+
resolved_offset: int | None = None
|
|
1094
|
+
recovered = False
|
|
1095
|
+
if _is_worksheet_offset(data, descriptor.offset) and descriptor.offset not in used_offsets:
|
|
1096
|
+
resolved_offset = descriptor.offset
|
|
1097
|
+
else:
|
|
1098
|
+
resolved_offset = next((offset for offset in candidates if offset not in used_offsets), None)
|
|
1099
|
+
recovered = resolved_offset is not None
|
|
1100
|
+
if recovered:
|
|
1101
|
+
logger.warning(
|
|
1102
|
+
"XLS_BOUNDSHEET_RECOVERED: sheet={!r}, old_offset={}, new_offset={}",
|
|
1103
|
+
descriptor.name,
|
|
1104
|
+
descriptor.offset,
|
|
1105
|
+
resolved_offset,
|
|
1106
|
+
)
|
|
1107
|
+
if resolved_offset is None:
|
|
1108
|
+
logger.warning("XLS_SHEET_UNREADABLE: keeping empty sheet {!r}", descriptor.name)
|
|
1109
|
+
sheets.append(
|
|
1110
|
+
XlsSheet(
|
|
1111
|
+
name=descriptor.name,
|
|
1112
|
+
visible=descriptor.visible,
|
|
1113
|
+
order=sheet_index,
|
|
1114
|
+
recovered=True,
|
|
1115
|
+
)
|
|
1116
|
+
)
|
|
1117
|
+
continue
|
|
1118
|
+
used_offsets.add(resolved_offset)
|
|
1119
|
+
parsed = _read_sheet(
|
|
1120
|
+
data,
|
|
1121
|
+
globals_,
|
|
1122
|
+
descriptor,
|
|
1123
|
+
resolved_offset,
|
|
1124
|
+
sheet_index=sheet_index,
|
|
1125
|
+
recovered=recovered,
|
|
1126
|
+
native_equations=normalized_equations,
|
|
1127
|
+
image_equation_decoder=image_equation_decoder,
|
|
1128
|
+
budget=budget,
|
|
1129
|
+
)
|
|
1130
|
+
if parsed is None:
|
|
1131
|
+
sheets.append(
|
|
1132
|
+
XlsSheet(
|
|
1133
|
+
name=descriptor.name,
|
|
1134
|
+
visible=descriptor.visible,
|
|
1135
|
+
order=sheet_index,
|
|
1136
|
+
recovered=True,
|
|
1137
|
+
)
|
|
1138
|
+
)
|
|
1139
|
+
else:
|
|
1140
|
+
sheets.append(parsed)
|
|
1141
|
+
return XlsWorkbook(
|
|
1142
|
+
sheets=sheets,
|
|
1143
|
+
chart_sheets=_parse_chart_sheets(data, globals_, budget=budget),
|
|
1144
|
+
active_sheet_index=globals_.active_sheet_index,
|
|
1145
|
+
)
|