docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
"""旧版 Office 二进制格式共享的 OfficeArt 记录与图片解码。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
import struct
|
|
8
|
+
import zlib
|
|
9
|
+
|
|
10
|
+
from loguru import logger
|
|
11
|
+
|
|
12
|
+
from ..image import ensure_bmp_header
|
|
13
|
+
from ..errors import LegacyOfficeResourceLimitError
|
|
14
|
+
from ..limits import MAX_ASSET_TOTAL_BYTES, MAX_ENTRY_BYTES, MAX_PICTURE_RECORDS, MAX_RECORD_DEPTH
|
|
15
|
+
|
|
16
|
+
OFFICEART_CONTAINER_VERSION = 0xF
|
|
17
|
+
OFFICEART_DGG_CONTAINER = 0xF000
|
|
18
|
+
OFFICEART_BSTORE_CONTAINER = 0xF001
|
|
19
|
+
OFFICEART_SP_CONTAINER = 0xF004
|
|
20
|
+
OFFICEART_BSE = 0xF007
|
|
21
|
+
OFFICEART_FSP = 0xF00A
|
|
22
|
+
OFFICEART_FOPT = 0xF00B
|
|
23
|
+
OFFICEART_CLIENT_ANCHOR = 0xF010
|
|
24
|
+
OFFICEART_TERTIARY_FOPT = 0xF122
|
|
25
|
+
|
|
26
|
+
FOPT_PIB = 0x0104
|
|
27
|
+
FOPT_GROUP_SHAPE = 0x03BF
|
|
28
|
+
F_HIDDEN = 0x0000_0002
|
|
29
|
+
F_USE_HIDDEN = 0x0002_0000
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True, slots=True)
|
|
33
|
+
class OfficeArtRecord:
|
|
34
|
+
"""一条已经通过长度边界校验的 OfficeArt 记录。"""
|
|
35
|
+
|
|
36
|
+
offset: int
|
|
37
|
+
version: int
|
|
38
|
+
instance: int
|
|
39
|
+
record_type: int
|
|
40
|
+
payload: bytes
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True, slots=True)
|
|
44
|
+
class OfficeImagePayload:
|
|
45
|
+
"""从 BLIP 中恢复出的原始图片及其媒体类型。"""
|
|
46
|
+
|
|
47
|
+
data: bytes
|
|
48
|
+
extension: str
|
|
49
|
+
content_type: str
|
|
50
|
+
render_size_emu: tuple[int, int] | None = None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True, slots=True)
|
|
54
|
+
class OfficeArtShape:
|
|
55
|
+
"""Excel drawing 中可绑定到 OBJ 的形状属性。"""
|
|
56
|
+
|
|
57
|
+
shape_id: int | None
|
|
58
|
+
anchor: tuple[int, int, int, int] | None
|
|
59
|
+
pib: int | None
|
|
60
|
+
hidden: bool
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def record_at(
|
|
64
|
+
data: bytes,
|
|
65
|
+
offset: int,
|
|
66
|
+
*,
|
|
67
|
+
end: int | None = None,
|
|
68
|
+
charge: Callable[[], None] | None = None,
|
|
69
|
+
) -> OfficeArtRecord | None:
|
|
70
|
+
"""从指定偏移读取一条 OfficeArt 记录,坏边界返回空值。"""
|
|
71
|
+
|
|
72
|
+
limit = len(data) if end is None else min(end, len(data))
|
|
73
|
+
if offset < 0 or offset + 8 > limit:
|
|
74
|
+
return None
|
|
75
|
+
version_instance, record_type, length = struct.unpack_from("<HHI", data, offset)
|
|
76
|
+
payload_start = offset + 8
|
|
77
|
+
payload_end = payload_start + int(length)
|
|
78
|
+
if payload_end < payload_start or payload_end > limit:
|
|
79
|
+
return None
|
|
80
|
+
if charge is not None:
|
|
81
|
+
charge()
|
|
82
|
+
return OfficeArtRecord(
|
|
83
|
+
offset=offset,
|
|
84
|
+
version=version_instance & 0xF,
|
|
85
|
+
instance=version_instance >> 4,
|
|
86
|
+
record_type=record_type,
|
|
87
|
+
payload=data[payload_start:payload_end],
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def iter_records(
|
|
92
|
+
data: bytes,
|
|
93
|
+
*,
|
|
94
|
+
start: int = 0,
|
|
95
|
+
end: int | None = None,
|
|
96
|
+
charge: Callable[[], None] | None = None,
|
|
97
|
+
) -> Iterator[OfficeArtRecord]:
|
|
98
|
+
"""顺序遍历同一 OfficeArt 容器内的直接子记录。"""
|
|
99
|
+
|
|
100
|
+
limit = len(data) if end is None else min(end, len(data))
|
|
101
|
+
cursor = start
|
|
102
|
+
while cursor < limit:
|
|
103
|
+
record = record_at(data, cursor, end=limit, charge=charge)
|
|
104
|
+
if record is None:
|
|
105
|
+
return
|
|
106
|
+
yield record
|
|
107
|
+
cursor += 8 + len(record.payload)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def iter_descendants(
|
|
111
|
+
data: bytes,
|
|
112
|
+
*,
|
|
113
|
+
charge: Callable[[], None] | None = None,
|
|
114
|
+
) -> Iterator[OfficeArtRecord]:
|
|
115
|
+
"""以显式栈深度优先遍历 OfficeArt 记录树并限制嵌套深度。"""
|
|
116
|
+
|
|
117
|
+
stack: list[Iterator[OfficeArtRecord]] = [iter_records(data, charge=charge)]
|
|
118
|
+
while stack:
|
|
119
|
+
if len(stack) > MAX_RECORD_DEPTH:
|
|
120
|
+
raise LegacyOfficeResourceLimitError(f"record nesting exceeds max_record_depth={MAX_RECORD_DEPTH}")
|
|
121
|
+
try:
|
|
122
|
+
record = next(stack[-1])
|
|
123
|
+
except StopIteration:
|
|
124
|
+
stack.pop()
|
|
125
|
+
continue
|
|
126
|
+
yield record
|
|
127
|
+
if record.version == OFFICEART_CONTAINER_VERSION:
|
|
128
|
+
stack.append(iter_records(record.payload, charge=charge))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _simple_properties(record: OfficeArtRecord) -> dict[int, int]:
|
|
132
|
+
"""读取 FOPT 简单属性并让后出现的同名属性覆盖前值。"""
|
|
133
|
+
|
|
134
|
+
properties: dict[int, int] = {}
|
|
135
|
+
for index in range(record.instance):
|
|
136
|
+
offset = index * 6
|
|
137
|
+
if offset + 6 > len(record.payload):
|
|
138
|
+
break
|
|
139
|
+
opid, value = struct.unpack_from("<HI", record.payload, offset)
|
|
140
|
+
properties[opid & 0x3FFF] = int(value)
|
|
141
|
+
return properties
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _excel_client_anchor(payload: bytes) -> tuple[int, int, int, int] | None:
|
|
145
|
+
"""把 OfficeArtClientAnchorChart 转成起止行列坐标。"""
|
|
146
|
+
|
|
147
|
+
if len(payload) < 18:
|
|
148
|
+
return None
|
|
149
|
+
start_col = struct.unpack_from("<H", payload, 2)[0]
|
|
150
|
+
start_row = struct.unpack_from("<H", payload, 6)[0]
|
|
151
|
+
end_col = struct.unpack_from("<H", payload, 10)[0]
|
|
152
|
+
end_row = struct.unpack_from("<H", payload, 14)[0]
|
|
153
|
+
return int(start_row), int(start_col), int(end_row), int(end_col)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _shape_from_container(
|
|
157
|
+
record: OfficeArtRecord,
|
|
158
|
+
*,
|
|
159
|
+
charge: Callable[[], None] | None = None,
|
|
160
|
+
) -> OfficeArtShape | None:
|
|
161
|
+
"""从单个 SpContainer 提取 shape id、anchor、pib 与隐藏状态。"""
|
|
162
|
+
|
|
163
|
+
shape_id: int | None = None
|
|
164
|
+
anchor: tuple[int, int, int, int] | None = None
|
|
165
|
+
properties: dict[int, int] = {}
|
|
166
|
+
for child in iter_records(record.payload, charge=charge):
|
|
167
|
+
if child.record_type == OFFICEART_FSP and len(child.payload) >= 4:
|
|
168
|
+
shape_id = int(struct.unpack_from("<I", child.payload, 0)[0])
|
|
169
|
+
elif child.record_type in {OFFICEART_FOPT, OFFICEART_TERTIARY_FOPT}:
|
|
170
|
+
properties.update(_simple_properties(child))
|
|
171
|
+
elif child.record_type == OFFICEART_CLIENT_ANCHOR:
|
|
172
|
+
anchor = _excel_client_anchor(child.payload)
|
|
173
|
+
hidden_flags = properties.get(FOPT_GROUP_SHAPE, 0)
|
|
174
|
+
hidden = bool(hidden_flags & F_USE_HIDDEN and hidden_flags & F_HIDDEN)
|
|
175
|
+
pib = properties.get(FOPT_PIB)
|
|
176
|
+
if anchor is None and pib is None:
|
|
177
|
+
return None
|
|
178
|
+
return OfficeArtShape(shape_id=shape_id, anchor=anchor, pib=pib, hidden=hidden)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def extract_excel_shapes(
|
|
182
|
+
data: bytes,
|
|
183
|
+
*,
|
|
184
|
+
charge: Callable[[], None] | None = None,
|
|
185
|
+
) -> list[OfficeArtShape]:
|
|
186
|
+
"""按 drawing 顺序提取可与 Excel OBJ 一一绑定的形状。"""
|
|
187
|
+
|
|
188
|
+
shapes: list[OfficeArtShape] = []
|
|
189
|
+
for record in iter_descendants(data, charge=charge):
|
|
190
|
+
if record.record_type != OFFICEART_SP_CONTAINER:
|
|
191
|
+
continue
|
|
192
|
+
shape = _shape_from_container(record, charge=charge)
|
|
193
|
+
if shape is not None:
|
|
194
|
+
shapes.append(shape)
|
|
195
|
+
return shapes
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _bitmap_payload(body: bytes, instance: int) -> bytes | None:
|
|
199
|
+
"""跳过 BLIP UID 和 tag,返回位图原始载荷。"""
|
|
200
|
+
|
|
201
|
+
doubled = instance in {0x46B, 0x6E3, 0x6E1, 0x7A9}
|
|
202
|
+
start = (32 if doubled else 16) + 1
|
|
203
|
+
return body[start:] if start < len(body) else None
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def decode_blip(record: OfficeArtRecord) -> OfficeImagePayload | None:
|
|
207
|
+
"""解码常见位图和 EMF/WMF BLIP,并限制矢量解压输出。"""
|
|
208
|
+
|
|
209
|
+
instance = record.instance
|
|
210
|
+
body = record.payload
|
|
211
|
+
if record.record_type in {0xF01D, 0xF01E, 0xF01F, 0xF029}:
|
|
212
|
+
data = _bitmap_payload(body, instance)
|
|
213
|
+
if data is None:
|
|
214
|
+
return None
|
|
215
|
+
if len(data) > MAX_ENTRY_BYTES:
|
|
216
|
+
raise LegacyOfficeResourceLimitError("bitmap BLIP exceeds max_entry_bytes")
|
|
217
|
+
if record.record_type == 0xF01D:
|
|
218
|
+
return OfficeImagePayload(data=data, extension="jpg", content_type="image/jpeg")
|
|
219
|
+
if record.record_type == 0xF01E:
|
|
220
|
+
return OfficeImagePayload(data=data, extension="png", content_type="image/png")
|
|
221
|
+
if record.record_type == 0xF029:
|
|
222
|
+
return OfficeImagePayload(data=data, extension="tiff", content_type="image/tiff")
|
|
223
|
+
return OfficeImagePayload(data=ensure_bmp_header(data), extension="bmp", content_type="image/bmp")
|
|
224
|
+
|
|
225
|
+
if record.record_type not in {0xF01A, 0xF01B}:
|
|
226
|
+
return None
|
|
227
|
+
doubled = instance in ({0x3D5} if record.record_type == 0xF01A else {0x217})
|
|
228
|
+
header_offset = 32 if doubled else 16
|
|
229
|
+
if header_offset + 34 > len(body):
|
|
230
|
+
return None
|
|
231
|
+
declared_size = int(struct.unpack_from("<I", body, header_offset)[0])
|
|
232
|
+
render_width_emu, render_height_emu = struct.unpack_from("<ii", body, header_offset + 20)
|
|
233
|
+
render_size_emu = (render_width_emu, render_height_emu) if render_width_emu > 0 and render_height_emu > 0 else None
|
|
234
|
+
compressed_size = int(struct.unpack_from("<I", body, header_offset + 28)[0])
|
|
235
|
+
compression = body[header_offset + 32]
|
|
236
|
+
payload_start = header_offset + 34
|
|
237
|
+
payload = body[payload_start : payload_start + compressed_size]
|
|
238
|
+
if declared_size > MAX_ENTRY_BYTES:
|
|
239
|
+
raise LegacyOfficeResourceLimitError("metafile BLIP exceeds max_entry_bytes")
|
|
240
|
+
if compression == 0:
|
|
241
|
+
data = b""
|
|
242
|
+
reached_eof = False
|
|
243
|
+
for window_bits in (-zlib.MAX_WBITS, zlib.MAX_WBITS):
|
|
244
|
+
try:
|
|
245
|
+
inflater = zlib.decompressobj(window_bits)
|
|
246
|
+
candidate = inflater.decompress(payload, MAX_ENTRY_BYTES + 1)
|
|
247
|
+
candidate += inflater.flush(MAX_ENTRY_BYTES + 1 - len(candidate))
|
|
248
|
+
except zlib.error:
|
|
249
|
+
continue
|
|
250
|
+
if inflater.eof:
|
|
251
|
+
data = candidate
|
|
252
|
+
reached_eof = True
|
|
253
|
+
break
|
|
254
|
+
if not reached_eof:
|
|
255
|
+
return None
|
|
256
|
+
if len(data) > MAX_ENTRY_BYTES:
|
|
257
|
+
raise LegacyOfficeResourceLimitError("metafile BLIP decompression exceeded its limit")
|
|
258
|
+
else:
|
|
259
|
+
data = payload
|
|
260
|
+
if record.record_type == 0xF01A:
|
|
261
|
+
return OfficeImagePayload(
|
|
262
|
+
data=data,
|
|
263
|
+
extension="emf",
|
|
264
|
+
content_type="image/emf",
|
|
265
|
+
render_size_emu=render_size_emu,
|
|
266
|
+
)
|
|
267
|
+
# 保留合法 placeable header,使跨平台渲染器继续获得 bbox 与 units-per-inch。
|
|
268
|
+
return OfficeImagePayload(
|
|
269
|
+
data=data,
|
|
270
|
+
extension="wmf",
|
|
271
|
+
content_type="image/wmf",
|
|
272
|
+
render_size_emu=render_size_emu,
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def first_blip(
|
|
277
|
+
data: bytes,
|
|
278
|
+
*,
|
|
279
|
+
charge: Callable[[], None] | None = None,
|
|
280
|
+
) -> OfficeImagePayload | None:
|
|
281
|
+
"""深度优先返回一段 OfficeArt 数据中的首个可支持 BLIP。"""
|
|
282
|
+
|
|
283
|
+
for record in iter_descendants(data, charge=charge):
|
|
284
|
+
if record.record_type == OFFICEART_BSE:
|
|
285
|
+
decoded = _decode_bse_body(record.payload, charge=charge)
|
|
286
|
+
if decoded is not None:
|
|
287
|
+
return decoded
|
|
288
|
+
decoded = decode_blip(record)
|
|
289
|
+
if decoded is not None:
|
|
290
|
+
return decoded
|
|
291
|
+
return None
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def extract_word_shapes(
|
|
295
|
+
data: bytes,
|
|
296
|
+
*,
|
|
297
|
+
charge: Callable[[], None] | None = None,
|
|
298
|
+
) -> list[OfficeArtShape]:
|
|
299
|
+
"""按 Word drawing 顺序提取 shape id、pib 和隐藏状态。"""
|
|
300
|
+
|
|
301
|
+
shapes: list[OfficeArtShape] = []
|
|
302
|
+
for record in iter_descendants(data, charge=charge):
|
|
303
|
+
if record.record_type != OFFICEART_SP_CONTAINER:
|
|
304
|
+
continue
|
|
305
|
+
shape = _shape_from_container(record, charge=charge)
|
|
306
|
+
if shape is not None:
|
|
307
|
+
shapes.append(shape)
|
|
308
|
+
return shapes
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _decode_bse_body(
|
|
312
|
+
body: bytes,
|
|
313
|
+
*,
|
|
314
|
+
charge: Callable[[], None] | None = None,
|
|
315
|
+
delay_stream: bytes | None = None,
|
|
316
|
+
) -> OfficeImagePayload | None:
|
|
317
|
+
"""从 FBSE body 的内嵌或延迟 BLIP 中恢复图片。"""
|
|
318
|
+
|
|
319
|
+
if len(body) < 36:
|
|
320
|
+
return None
|
|
321
|
+
inner_offset = 36 + int(body[33])
|
|
322
|
+
inner = record_at(body, inner_offset, charge=charge) if inner_offset < len(body) else None
|
|
323
|
+
if inner is None and delay_stream is not None:
|
|
324
|
+
delayed_size = int(struct.unpack_from("<I", body, 20)[0])
|
|
325
|
+
reference_count = int(struct.unpack_from("<I", body, 24)[0])
|
|
326
|
+
delayed_offset = int(struct.unpack_from("<I", body, 28)[0])
|
|
327
|
+
if reference_count and delayed_offset != 0xFFFF_FFFF:
|
|
328
|
+
delayed_end = delayed_offset + delayed_size
|
|
329
|
+
if delayed_end >= delayed_offset and delayed_end <= len(delay_stream):
|
|
330
|
+
inner = record_at(delay_stream, delayed_offset, end=delayed_end, charge=charge)
|
|
331
|
+
return decode_blip(inner) if inner is not None else None
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def decode_bstore(
|
|
335
|
+
data: bytes,
|
|
336
|
+
*,
|
|
337
|
+
charge: Callable[[], None] | None = None,
|
|
338
|
+
delay_stream: bytes | None = None,
|
|
339
|
+
) -> dict[int, OfficeImagePayload]:
|
|
340
|
+
"""按一基 BSE 序号解码 drawing group 中的图片资源。"""
|
|
341
|
+
|
|
342
|
+
bse_records = [record for record in iter_descendants(data, charge=charge) if record.record_type == OFFICEART_BSE]
|
|
343
|
+
result: dict[int, OfficeImagePayload] = {}
|
|
344
|
+
asset_total = 0
|
|
345
|
+
for index, bse in enumerate(bse_records[:MAX_PICTURE_RECORDS], start=1):
|
|
346
|
+
decoded = _decode_bse_body(
|
|
347
|
+
bse.payload,
|
|
348
|
+
charge=charge,
|
|
349
|
+
delay_stream=delay_stream,
|
|
350
|
+
)
|
|
351
|
+
if decoded is None:
|
|
352
|
+
continue
|
|
353
|
+
asset_total += len(decoded.data)
|
|
354
|
+
if asset_total > MAX_ASSET_TOTAL_BYTES:
|
|
355
|
+
raise LegacyOfficeResourceLimitError(f"embedded assets exceed max_asset_total_bytes={MAX_ASSET_TOTAL_BYTES}")
|
|
356
|
+
result[index] = decoded
|
|
357
|
+
if len(bse_records) > MAX_PICTURE_RECORDS:
|
|
358
|
+
logger.warning(
|
|
359
|
+
"LEGACY_OFFICE_PICTURE_LIMIT: ignored BSE records after {}",
|
|
360
|
+
MAX_PICTURE_RECORDS,
|
|
361
|
+
)
|
|
362
|
+
return result
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""基于 olefile 的有界 OLE2/CFB 只读包装。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from io import BytesIO
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import olefile # type: ignore[reportMissingModuleSource]
|
|
9
|
+
|
|
10
|
+
from ..errors import LegacyOfficeMalformedError, LegacyOfficeMissingPartError, LegacyOfficeResourceLimitError
|
|
11
|
+
from ..limits import MAX_ENTRY_BYTES, MAX_TOTAL_BYTES
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class BoundedOleReader:
|
|
15
|
+
"""限制输入、单流及累计读取量,并提供大小写无关的 stream 访问。"""
|
|
16
|
+
|
|
17
|
+
def __init__(self, file_bytes: bytes) -> None:
|
|
18
|
+
"""校验输入大小并打开内存中的 OLE2 容器。"""
|
|
19
|
+
|
|
20
|
+
if not isinstance(file_bytes, bytes):
|
|
21
|
+
raise TypeError("legacy Office input must be bytes")
|
|
22
|
+
if len(file_bytes) > MAX_TOTAL_BYTES:
|
|
23
|
+
raise LegacyOfficeResourceLimitError(f"input exceeds max_total_bytes={MAX_TOTAL_BYTES}")
|
|
24
|
+
try:
|
|
25
|
+
self._ole: Any = olefile.OleFileIO(BytesIO(file_bytes), raise_defects=olefile.DEFECT_FATAL)
|
|
26
|
+
except Exception as exc:
|
|
27
|
+
raise LegacyOfficeMalformedError(f"not a readable OLE2 compound file: {exc}") from exc
|
|
28
|
+
self._total_read = 0
|
|
29
|
+
self._stream_names = self._build_stream_name_map()
|
|
30
|
+
|
|
31
|
+
def _build_stream_name_map(self) -> dict[str, tuple[str, ...]]:
|
|
32
|
+
"""建立大小写无关的完整 stream 名称索引。"""
|
|
33
|
+
|
|
34
|
+
names: dict[str, tuple[str, ...]] = {}
|
|
35
|
+
try:
|
|
36
|
+
for parts in self._ole.listdir(streams=True, storages=False):
|
|
37
|
+
normalized = tuple(str(part) for part in parts)
|
|
38
|
+
names["/".join(normalized).casefold()] = normalized
|
|
39
|
+
except Exception as exc:
|
|
40
|
+
self.close()
|
|
41
|
+
raise LegacyOfficeMalformedError(f"cannot enumerate OLE streams: {exc}") from exc
|
|
42
|
+
return names
|
|
43
|
+
|
|
44
|
+
def has_stream(self, name: str) -> bool:
|
|
45
|
+
"""返回容器是否含有指定 stream。"""
|
|
46
|
+
|
|
47
|
+
return name.casefold() in self._stream_names
|
|
48
|
+
|
|
49
|
+
def stream_names(self, *, prefix: str | None = None) -> tuple[str, ...]:
|
|
50
|
+
"""返回完整 stream 名称;可按大小写无关前缀筛选。"""
|
|
51
|
+
|
|
52
|
+
normalized_prefix = prefix.casefold() if prefix is not None else None
|
|
53
|
+
names = (
|
|
54
|
+
"/".join(parts)
|
|
55
|
+
for key, parts in self._stream_names.items()
|
|
56
|
+
if normalized_prefix is None or key.startswith(normalized_prefix)
|
|
57
|
+
)
|
|
58
|
+
return tuple(sorted(names, key=str.casefold))
|
|
59
|
+
|
|
60
|
+
def read_stream(self, name: str, *, required: bool = True) -> bytes:
|
|
61
|
+
"""有界读取指定 stream;可选 stream 不存在时返回空字节。"""
|
|
62
|
+
|
|
63
|
+
parts = self._stream_names.get(name.casefold())
|
|
64
|
+
if parts is None:
|
|
65
|
+
if required:
|
|
66
|
+
raise LegacyOfficeMissingPartError(f"missing required OLE stream: {name}")
|
|
67
|
+
return b""
|
|
68
|
+
try:
|
|
69
|
+
size = int(self._ole.get_size(list(parts)))
|
|
70
|
+
except Exception as exc:
|
|
71
|
+
raise LegacyOfficeMalformedError(f"cannot read OLE stream size: {name}: {exc}") from exc
|
|
72
|
+
if size > MAX_ENTRY_BYTES:
|
|
73
|
+
raise LegacyOfficeResourceLimitError(f"stream {name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
|
|
74
|
+
if self._total_read + size > MAX_TOTAL_BYTES:
|
|
75
|
+
raise LegacyOfficeResourceLimitError(f"OLE streams exceed max_total_bytes={MAX_TOTAL_BYTES}")
|
|
76
|
+
try:
|
|
77
|
+
with self._ole.openstream(list(parts)) as stream:
|
|
78
|
+
payload = stream.read(MAX_ENTRY_BYTES + 1)
|
|
79
|
+
except Exception as exc:
|
|
80
|
+
raise LegacyOfficeMalformedError(f"cannot read OLE stream {name!r}: {exc}") from exc
|
|
81
|
+
if len(payload) > MAX_ENTRY_BYTES:
|
|
82
|
+
raise LegacyOfficeResourceLimitError(f"stream {name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}")
|
|
83
|
+
self._total_read += len(payload)
|
|
84
|
+
return payload
|
|
85
|
+
|
|
86
|
+
def metadata(self) -> Any | None:
|
|
87
|
+
"""尽力读取 SummaryInformation,失败时不影响正文解析。"""
|
|
88
|
+
|
|
89
|
+
try:
|
|
90
|
+
return self._ole.get_metadata()
|
|
91
|
+
except Exception:
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
def close(self) -> None:
|
|
95
|
+
"""关闭底层 olefile 句柄。"""
|
|
96
|
+
|
|
97
|
+
ole = getattr(self, "_ole", None)
|
|
98
|
+
if ole is not None:
|
|
99
|
+
ole.close()
|
|
100
|
+
self._ole = None
|
|
101
|
+
|
|
102
|
+
def __enter__(self) -> BoundedOleReader:
|
|
103
|
+
"""返回当前有界读取器。"""
|
|
104
|
+
|
|
105
|
+
return self
|
|
106
|
+
|
|
107
|
+
def __exit__(self, *_args: object) -> None:
|
|
108
|
+
"""离开上下文时关闭底层句柄。"""
|
|
109
|
+
|
|
110
|
+
self.close()
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Flash Office 二进制、嵌入对象与 RTF 解析共享的固定安全限制。"""
|
|
2
|
+
|
|
3
|
+
from typing import Final
|
|
4
|
+
|
|
5
|
+
MAX_ENTRY_BYTES: Final = 128 * 1024 * 1024
|
|
6
|
+
MAX_TOTAL_BYTES: Final = 512 * 1024 * 1024
|
|
7
|
+
MAX_ASSET_TOTAL_BYTES: Final = 128 * 1024 * 1024
|
|
8
|
+
MAX_GRID_SLOTS: Final = 4_000_000
|
|
9
|
+
MAX_RECORD_DEPTH: Final = 64
|
|
10
|
+
MAX_RECORDS: Final = 16_000_000
|
|
11
|
+
MAX_PICTURE_RECORDS: Final = 100_000
|
|
12
|
+
MAX_USER_EDIT_CHAIN: Final = 100
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""恢复 ODF 嵌入图表的预览与源数据表。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
|
|
7
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
8
|
+
|
|
9
|
+
from .....schema import BlockType
|
|
10
|
+
from .constants import qname
|
|
11
|
+
from .models import TableGrid
|
|
12
|
+
from .table import (
|
|
13
|
+
OdfTableExpansionBudget,
|
|
14
|
+
crop_table_grid,
|
|
15
|
+
parse_cell_range_bounds,
|
|
16
|
+
parse_table_grid,
|
|
17
|
+
table_grid_to_html,
|
|
18
|
+
union_bounds,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _chart_range_bounds(chart: etree._Element) -> tuple[int, int, int, int] | None:
|
|
23
|
+
"""收集 chart series、label 和 categories 的精确单元格引用范围。"""
|
|
24
|
+
values: list[tuple[int, int, int, int]] = []
|
|
25
|
+
attribute_names = {
|
|
26
|
+
qname("chart", "values-cell-range-address"),
|
|
27
|
+
qname("chart", "label-cell-address"),
|
|
28
|
+
qname("table", "cell-range-address"),
|
|
29
|
+
}
|
|
30
|
+
for element in chart.iter():
|
|
31
|
+
for attribute_name in attribute_names:
|
|
32
|
+
if bounds := parse_cell_range_bounds(element.get(attribute_name, "")):
|
|
33
|
+
values.append(bounds)
|
|
34
|
+
return union_bounds(values)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _single_nonempty_grid(grids: list[TableGrid]) -> TableGrid | None:
|
|
38
|
+
"""仅在对象内存在唯一非空表格时返回安全回退候选。"""
|
|
39
|
+
nonempty = [grid for grid in grids if grid.rows]
|
|
40
|
+
return nonempty[0] if len(nonempty) == 1 else None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def parse_chart_block(
|
|
44
|
+
object_root: etree._Element,
|
|
45
|
+
*,
|
|
46
|
+
render_cell: Callable[[etree._Element], str],
|
|
47
|
+
preview_data_uri: str | None,
|
|
48
|
+
table_expansion_budget: OdfTableExpansionBudget | None = None,
|
|
49
|
+
) -> dict | None:
|
|
50
|
+
"""按精确引用优先、唯一表回退的规则构造图表 raw block。"""
|
|
51
|
+
chart = next(object_root.iter(qname("chart", "chart")), None)
|
|
52
|
+
if chart is None:
|
|
53
|
+
return None
|
|
54
|
+
grids = [
|
|
55
|
+
parse_table_grid(table, render_cell, expansion_budget=table_expansion_budget)
|
|
56
|
+
for table in object_root.iter(qname("table", "table"))
|
|
57
|
+
]
|
|
58
|
+
selected: TableGrid | None = None
|
|
59
|
+
if grids and (bounds := _chart_range_bounds(chart)) is not None:
|
|
60
|
+
selected = crop_table_grid(grids[0], bounds)
|
|
61
|
+
if not selected.rows:
|
|
62
|
+
selected = None
|
|
63
|
+
if selected is None:
|
|
64
|
+
selected = _single_nonempty_grid(grids)
|
|
65
|
+
if selected is None:
|
|
66
|
+
return None
|
|
67
|
+
content = table_grid_to_html(selected)
|
|
68
|
+
if not content:
|
|
69
|
+
return None
|
|
70
|
+
block: dict = {"type": BlockType.CHART, "content": content}
|
|
71
|
+
if preview_data_uri:
|
|
72
|
+
block["image_base64"] = preview_data_uri
|
|
73
|
+
return block
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
__all__ = ["parse_chart_block"]
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""OpenDocument 命名空间、MIME 与固定资源上限。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Final, Literal, TypeAlias
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
OdfSuffix: TypeAlias = Literal["odt", "ods", "odp"]
|
|
9
|
+
|
|
10
|
+
NS: Final[dict[str, str]] = {
|
|
11
|
+
"chart": "urn:oasis:names:tc:opendocument:xmlns:chart:1.0",
|
|
12
|
+
"dc": "http://purl.org/dc/elements/1.1/",
|
|
13
|
+
"draw": "urn:oasis:names:tc:opendocument:xmlns:drawing:1.0",
|
|
14
|
+
"fo": "urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0",
|
|
15
|
+
"form": "urn:oasis:names:tc:opendocument:xmlns:form:1.0",
|
|
16
|
+
"manifest": "urn:oasis:names:tc:opendocument:xmlns:manifest:1.0",
|
|
17
|
+
"math": "http://www.w3.org/1998/Math/MathML",
|
|
18
|
+
"meta": "urn:oasis:names:tc:opendocument:xmlns:meta:1.0",
|
|
19
|
+
"number": "urn:oasis:names:tc:opendocument:xmlns:datastyle:1.0",
|
|
20
|
+
"office": "urn:oasis:names:tc:opendocument:xmlns:office:1.0",
|
|
21
|
+
"presentation": "urn:oasis:names:tc:opendocument:xmlns:presentation:1.0",
|
|
22
|
+
"style": "urn:oasis:names:tc:opendocument:xmlns:style:1.0",
|
|
23
|
+
"svg": "urn:oasis:names:tc:opendocument:xmlns:svg-compatible:1.0",
|
|
24
|
+
"table": "urn:oasis:names:tc:opendocument:xmlns:table:1.0",
|
|
25
|
+
"text": "urn:oasis:names:tc:opendocument:xmlns:text:1.0",
|
|
26
|
+
"xlink": "http://www.w3.org/1999/xlink",
|
|
27
|
+
"xml": "http://www.w3.org/XML/1998/namespace",
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
ODF_MIME_BY_SUFFIX: Final[dict[OdfSuffix, str]] = {
|
|
31
|
+
"odt": "application/vnd.oasis.opendocument.text",
|
|
32
|
+
"ods": "application/vnd.oasis.opendocument.spreadsheet",
|
|
33
|
+
"odp": "application/vnd.oasis.opendocument.presentation",
|
|
34
|
+
}
|
|
35
|
+
ODF_SUFFIX_BY_MIME: Final[dict[str, OdfSuffix]] = {mime: suffix for suffix, mime in ODF_MIME_BY_SUFFIX.items()}
|
|
36
|
+
ODF_BODY_BY_SUFFIX: Final[dict[OdfSuffix, str]] = {
|
|
37
|
+
"odt": "text",
|
|
38
|
+
"ods": "spreadsheet",
|
|
39
|
+
"odp": "presentation",
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
MAX_ENTRY_BYTES: Final = 128 * 1024 * 1024
|
|
43
|
+
MAX_TOTAL_BYTES: Final = 512 * 1024 * 1024
|
|
44
|
+
MAX_ENTRY_COUNT: Final = 100_000
|
|
45
|
+
MAX_XML_DEPTH: Final = 256
|
|
46
|
+
MAX_XML_NODES: Final = 2_000_000
|
|
47
|
+
MAX_EXPANSION_TEXT_BYTES: Final = 64 * 1024 * 1024
|
|
48
|
+
MAX_ASSET_TOTAL_BYTES: Final = 128 * 1024 * 1024
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def qname(prefix: str, local_name: str) -> str:
|
|
52
|
+
"""返回指定 ODF 命名空间下的 Clark notation 标签名。"""
|
|
53
|
+
return f"{{{NS[prefix]}}}{local_name}"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
__all__ = [
|
|
57
|
+
"MAX_ASSET_TOTAL_BYTES",
|
|
58
|
+
"MAX_ENTRY_BYTES",
|
|
59
|
+
"MAX_ENTRY_COUNT",
|
|
60
|
+
"MAX_EXPANSION_TEXT_BYTES",
|
|
61
|
+
"MAX_TOTAL_BYTES",
|
|
62
|
+
"MAX_XML_DEPTH",
|
|
63
|
+
"MAX_XML_NODES",
|
|
64
|
+
"NS",
|
|
65
|
+
"ODF_BODY_BY_SUFFIX",
|
|
66
|
+
"ODF_MIME_BY_SUFFIX",
|
|
67
|
+
"ODF_SUFFIX_BY_MIME",
|
|
68
|
+
"OdfSuffix",
|
|
69
|
+
"qname",
|
|
70
|
+
]
|