docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
"""受限读取 OpenDocument ZIP 包及 XML part。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import posixpath
|
|
6
|
+
from io import BytesIO
|
|
7
|
+
from pathlib import PurePosixPath
|
|
8
|
+
from urllib.parse import unquote, urlsplit
|
|
9
|
+
from zipfile import BadZipFile, ZipFile, ZipInfo
|
|
10
|
+
|
|
11
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
12
|
+
|
|
13
|
+
from .constants import (
|
|
14
|
+
MAX_ASSET_TOTAL_BYTES,
|
|
15
|
+
MAX_ENTRY_BYTES,
|
|
16
|
+
MAX_ENTRY_COUNT,
|
|
17
|
+
MAX_TOTAL_BYTES,
|
|
18
|
+
MAX_XML_DEPTH,
|
|
19
|
+
MAX_XML_NODES,
|
|
20
|
+
ODF_BODY_BY_SUFFIX,
|
|
21
|
+
ODF_MIME_BY_SUFFIX,
|
|
22
|
+
ODF_SUFFIX_BY_MIME,
|
|
23
|
+
OdfSuffix,
|
|
24
|
+
qname,
|
|
25
|
+
)
|
|
26
|
+
from .errors import OdfEncryptedError, OdfParseError, OdfResourceLimitError
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _xml_parser() -> etree.XMLParser:
|
|
30
|
+
"""为每次 part 解析新建禁用实体、DTD 和网络的 XML parser。"""
|
|
31
|
+
return etree.XMLParser(
|
|
32
|
+
resolve_entities=False,
|
|
33
|
+
load_dtd=False,
|
|
34
|
+
no_network=True,
|
|
35
|
+
recover=False,
|
|
36
|
+
remove_blank_text=False,
|
|
37
|
+
huge_tree=False,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class OdfPackage:
|
|
42
|
+
"""在固定资源预算内读取一个 OpenDocument ZIP 包。"""
|
|
43
|
+
|
|
44
|
+
def __init__(self, file_bytes: bytes) -> None:
|
|
45
|
+
"""打开内存包并在读取正文前完成中央目录安全校验。"""
|
|
46
|
+
try:
|
|
47
|
+
self._zip = ZipFile(BytesIO(file_bytes))
|
|
48
|
+
except (BadZipFile, OSError, ValueError) as exc:
|
|
49
|
+
raise OdfParseError(f"Malformed ODF package: {exc}") from exc
|
|
50
|
+
try:
|
|
51
|
+
self._infos = self._validate_members(self._zip.infolist())
|
|
52
|
+
except Exception:
|
|
53
|
+
self._zip.close()
|
|
54
|
+
raise
|
|
55
|
+
self._cache: dict[str, bytes] = {}
|
|
56
|
+
self._asset_bytes = 0
|
|
57
|
+
self._asset_parts: set[str] = set()
|
|
58
|
+
self._manifest_media_types: dict[str, str] | None = None
|
|
59
|
+
|
|
60
|
+
@staticmethod
|
|
61
|
+
def _validate_members(infos: list[ZipInfo]) -> dict[str, ZipInfo]:
|
|
62
|
+
"""校验成员数量、解压体积、路径与重名,避免包级歧义。"""
|
|
63
|
+
if len(infos) > MAX_ENTRY_COUNT:
|
|
64
|
+
raise OdfResourceLimitError(f"ODF resource limit exceeded: max_entry_count={MAX_ENTRY_COUNT}")
|
|
65
|
+
total_size = 0
|
|
66
|
+
members: dict[str, ZipInfo] = {}
|
|
67
|
+
for info in infos:
|
|
68
|
+
name = info.filename
|
|
69
|
+
if not OdfPackage._is_safe_member_name(name):
|
|
70
|
+
raise OdfParseError(f"Malformed ODF package: unsafe member path {name!r}")
|
|
71
|
+
if name in members:
|
|
72
|
+
raise OdfParseError(f"Malformed ODF package: duplicate member {name!r}")
|
|
73
|
+
if info.file_size > MAX_ENTRY_BYTES:
|
|
74
|
+
raise OdfResourceLimitError(
|
|
75
|
+
f"ODF resource limit exceeded: member {name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}"
|
|
76
|
+
)
|
|
77
|
+
total_size += info.file_size
|
|
78
|
+
if total_size > MAX_TOTAL_BYTES:
|
|
79
|
+
raise OdfResourceLimitError(f"ODF resource limit exceeded: max_total_bytes={MAX_TOTAL_BYTES}")
|
|
80
|
+
members[name] = info
|
|
81
|
+
return members
|
|
82
|
+
|
|
83
|
+
@staticmethod
|
|
84
|
+
def _is_safe_member_name(name: str) -> bool:
|
|
85
|
+
"""判断 ZIP 成员是否为无反斜杠、无绝对路径和上跳段的 POSIX 路径。"""
|
|
86
|
+
if not name or "\x00" in name or "\\" in name or name.startswith("/"):
|
|
87
|
+
return False
|
|
88
|
+
parts = PurePosixPath(name).parts
|
|
89
|
+
return bool(parts) and all(part not in {"", ".", ".."} for part in parts)
|
|
90
|
+
|
|
91
|
+
def has_part(self, part_name: str) -> bool:
|
|
92
|
+
"""返回包内是否存在指定规范成员。"""
|
|
93
|
+
return part_name in self._infos
|
|
94
|
+
|
|
95
|
+
def read_part(self, part_name: str, *, required: bool = False, asset: bool = False) -> bytes | None:
|
|
96
|
+
"""读取一个已校验成员,并对累计图片资源执行独立限制。"""
|
|
97
|
+
info = self._infos.get(part_name)
|
|
98
|
+
if info is None:
|
|
99
|
+
if required:
|
|
100
|
+
raise OdfParseError(f"Malformed ODF package: missing required part {part_name!r}")
|
|
101
|
+
return None
|
|
102
|
+
if part_name in self._cache:
|
|
103
|
+
data = self._cache[part_name]
|
|
104
|
+
if asset:
|
|
105
|
+
self._charge_asset(part_name, len(data))
|
|
106
|
+
return data
|
|
107
|
+
try:
|
|
108
|
+
data = self._zip.read(info)
|
|
109
|
+
except (BadZipFile, OSError, RuntimeError, ValueError) as exc:
|
|
110
|
+
if required:
|
|
111
|
+
raise OdfParseError(f"Malformed ODF package: cannot read {part_name!r}: {exc}") from exc
|
|
112
|
+
return None
|
|
113
|
+
if len(data) > MAX_ENTRY_BYTES:
|
|
114
|
+
raise OdfResourceLimitError(
|
|
115
|
+
f"ODF resource limit exceeded: member {part_name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}"
|
|
116
|
+
)
|
|
117
|
+
if asset:
|
|
118
|
+
self._charge_asset(part_name, len(data))
|
|
119
|
+
self._cache[part_name] = data
|
|
120
|
+
return data
|
|
121
|
+
|
|
122
|
+
def _charge_asset(self, part_name: str, byte_count: int) -> None:
|
|
123
|
+
"""按唯一资源成员累计保留字节,重复引用不重复计费。"""
|
|
124
|
+
if part_name in self._asset_parts:
|
|
125
|
+
return
|
|
126
|
+
self._asset_parts.add(part_name)
|
|
127
|
+
self._asset_bytes += byte_count
|
|
128
|
+
if self._asset_bytes > MAX_ASSET_TOTAL_BYTES:
|
|
129
|
+
raise OdfResourceLimitError(f"ODF resource limit exceeded: max_asset_total_bytes={MAX_ASSET_TOTAL_BYTES}")
|
|
130
|
+
|
|
131
|
+
def xml_part(self, part_name: str, *, required: bool = False) -> etree._Element | None:
|
|
132
|
+
"""禁用实体和网络后解析 XML,并校验节点数及最大深度。"""
|
|
133
|
+
data = self.read_part(part_name, required=required)
|
|
134
|
+
if data is None:
|
|
135
|
+
return None
|
|
136
|
+
try:
|
|
137
|
+
root = etree.fromstring(data, parser=_xml_parser())
|
|
138
|
+
except (etree.XMLSyntaxError, ValueError) as exc:
|
|
139
|
+
if required:
|
|
140
|
+
raise OdfParseError(f"Malformed ODF package: invalid XML part {part_name!r}: {exc}") from exc
|
|
141
|
+
return None
|
|
142
|
+
if root.getroottree().docinfo.doctype:
|
|
143
|
+
raise OdfParseError(f"Malformed ODF package: DTD is not allowed in {part_name!r}")
|
|
144
|
+
self._validate_xml_shape(root, part_name)
|
|
145
|
+
return root
|
|
146
|
+
|
|
147
|
+
@staticmethod
|
|
148
|
+
def _validate_xml_shape(root: etree._Element, part_name: str) -> None:
|
|
149
|
+
"""迭代统计 XML 节点与深度,避免深递归或超大 DOM 继续传播。"""
|
|
150
|
+
node_count = 0
|
|
151
|
+
stack: list[tuple[etree._Element, int]] = [(root, 1)]
|
|
152
|
+
while stack:
|
|
153
|
+
element, depth = stack.pop()
|
|
154
|
+
node_count += 1
|
|
155
|
+
if node_count > MAX_XML_NODES:
|
|
156
|
+
raise OdfResourceLimitError(f"ODF resource limit exceeded: {part_name!r} exceeds max_xml_nodes={MAX_XML_NODES}")
|
|
157
|
+
if depth > MAX_XML_DEPTH:
|
|
158
|
+
raise OdfResourceLimitError(f"ODF resource limit exceeded: {part_name!r} exceeds max_xml_depth={MAX_XML_DEPTH}")
|
|
159
|
+
for child in element:
|
|
160
|
+
if isinstance(child.tag, str):
|
|
161
|
+
stack.append((child, depth + 1))
|
|
162
|
+
|
|
163
|
+
def detected_suffix(self) -> OdfSuffix | None:
|
|
164
|
+
"""按 mimetype、manifest 根条目依次识别 ODF 三种包类型。"""
|
|
165
|
+
mimetype = self.read_part("mimetype")
|
|
166
|
+
if mimetype is not None:
|
|
167
|
+
try:
|
|
168
|
+
normalized = mimetype.decode("ascii", errors="strict").strip()
|
|
169
|
+
except UnicodeDecodeError:
|
|
170
|
+
normalized = ""
|
|
171
|
+
if suffix := ODF_SUFFIX_BY_MIME.get(normalized):
|
|
172
|
+
return suffix
|
|
173
|
+
return ODF_SUFFIX_BY_MIME.get(self.manifest_media_types().get("/", ""))
|
|
174
|
+
|
|
175
|
+
def validate_document(self, expected_suffix: OdfSuffix) -> etree._Element:
|
|
176
|
+
"""校验包类型、加密状态和 required 正文 body 后返回内容根节点。"""
|
|
177
|
+
detected = self.detected_suffix()
|
|
178
|
+
if detected is not None and detected != expected_suffix:
|
|
179
|
+
raise OdfParseError(
|
|
180
|
+
f"Malformed ODF package: expected {ODF_MIME_BY_SUFFIX[expected_suffix]!r}, got {ODF_MIME_BY_SUFFIX[detected]!r}"
|
|
181
|
+
)
|
|
182
|
+
if self.is_encrypted():
|
|
183
|
+
raise OdfEncryptedError("Encrypted ODF documents are not supported")
|
|
184
|
+
content_root = self.xml_part("content.xml", required=True)
|
|
185
|
+
assert content_root is not None
|
|
186
|
+
body = content_root.find(f".//{qname('office', 'body')}")
|
|
187
|
+
expected_body = ODF_BODY_BY_SUFFIX[expected_suffix]
|
|
188
|
+
if body is None or body.find(qname("office", expected_body)) is None:
|
|
189
|
+
raise OdfParseError(f"Malformed ODF package: content.xml has no office:{expected_body} body")
|
|
190
|
+
return content_root
|
|
191
|
+
|
|
192
|
+
def manifest_media_types(self) -> dict[str, str]:
|
|
193
|
+
"""读取 manifest 中规范成员路径到 MIME 的映射,损坏时安全降级为空。"""
|
|
194
|
+
if self._manifest_media_types is not None:
|
|
195
|
+
return self._manifest_media_types
|
|
196
|
+
result: dict[str, str] = {}
|
|
197
|
+
root = self.xml_part("META-INF/manifest.xml")
|
|
198
|
+
if root is not None:
|
|
199
|
+
for entry in root.iter(qname("manifest", "file-entry")):
|
|
200
|
+
path = entry.get(qname("manifest", "full-path"))
|
|
201
|
+
media_type = entry.get(qname("manifest", "media-type"))
|
|
202
|
+
if path and media_type:
|
|
203
|
+
result[path] = media_type
|
|
204
|
+
self._manifest_media_types = result
|
|
205
|
+
return result
|
|
206
|
+
|
|
207
|
+
def is_encrypted(self) -> bool:
|
|
208
|
+
"""按 manifest:encryption-data 元素判断包内是否存在加密内容。"""
|
|
209
|
+
root = self.xml_part("META-INF/manifest.xml")
|
|
210
|
+
return root is not None and next(root.iter(qname("manifest", "encryption-data")), None) is not None
|
|
211
|
+
|
|
212
|
+
def content_type_for(self, part_name: str) -> str | None:
|
|
213
|
+
"""返回 manifest 为指定成员声明的媒体类型。"""
|
|
214
|
+
return self.manifest_media_types().get(part_name)
|
|
215
|
+
|
|
216
|
+
def resolve_reference(self, href: str, *, base_part: str = "content.xml") -> str | None:
|
|
217
|
+
"""把相对 xlink 引用解析为安全包成员;绝对 URI 和上跳路径返回空。"""
|
|
218
|
+
normalized_href = unquote((href or "").strip())
|
|
219
|
+
if not normalized_href or normalized_href.startswith("#"):
|
|
220
|
+
return None
|
|
221
|
+
try:
|
|
222
|
+
split = urlsplit(normalized_href)
|
|
223
|
+
except ValueError:
|
|
224
|
+
return None
|
|
225
|
+
if split.scheme or split.netloc:
|
|
226
|
+
return None
|
|
227
|
+
candidate = split.path.replace("\\", "/")
|
|
228
|
+
base_dir = posixpath.dirname(base_part)
|
|
229
|
+
resolved = posixpath.normpath(posixpath.join(base_dir, candidate))
|
|
230
|
+
if resolved in {"", ".", ".."} or resolved.startswith("../") or resolved.startswith("/"):
|
|
231
|
+
return None
|
|
232
|
+
return resolved.removeprefix("./")
|
|
233
|
+
|
|
234
|
+
def resolve_object_content(self, href: str, *, base_part: str = "content.xml") -> str | None:
|
|
235
|
+
"""把 draw:object 目录引用解析到其 content.xml 成员。"""
|
|
236
|
+
resolved = self.resolve_reference(href, base_part=base_part)
|
|
237
|
+
if resolved is None:
|
|
238
|
+
return None
|
|
239
|
+
if resolved.endswith(".xml"):
|
|
240
|
+
return resolved
|
|
241
|
+
return f"{resolved.rstrip('/')}/content.xml"
|
|
242
|
+
|
|
243
|
+
def body_element(self, content_root: etree._Element, suffix: OdfSuffix) -> etree._Element:
|
|
244
|
+
"""返回已验证内容树中的 text、spreadsheet 或 presentation 正文节点。"""
|
|
245
|
+
body = content_root.find(f".//{qname('office', 'body')}")
|
|
246
|
+
if body is None:
|
|
247
|
+
raise OdfParseError("Malformed ODF package: content.xml has no office:body")
|
|
248
|
+
child = body.find(qname("office", ODF_BODY_BY_SUFFIX[suffix]))
|
|
249
|
+
if child is None:
|
|
250
|
+
raise OdfParseError(f"Malformed ODF package: content.xml has no {suffix} body")
|
|
251
|
+
return child
|
|
252
|
+
|
|
253
|
+
def close(self) -> None:
|
|
254
|
+
"""关闭底层 ZipFile,但不触碰调用方提供的输入流。"""
|
|
255
|
+
self._zip.close()
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def detect_odf_suffix(file_bytes: bytes) -> OdfSuffix | None:
|
|
259
|
+
"""只按 ODF 包身份识别三种后缀,任意损坏均返回空供上层继续兜底。"""
|
|
260
|
+
try:
|
|
261
|
+
package = OdfPackage(file_bytes)
|
|
262
|
+
except (OdfParseError, OdfResourceLimitError):
|
|
263
|
+
return None
|
|
264
|
+
try:
|
|
265
|
+
return package.detected_suffix()
|
|
266
|
+
finally:
|
|
267
|
+
package.close()
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
__all__ = ["OdfPackage", "detect_odf_suffix"]
|
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""解析 OpenDocument 样式继承、列表和逻辑分页属性。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from loguru import logger
|
|
8
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
9
|
+
|
|
10
|
+
from .constants import qname
|
|
11
|
+
from .models import ListLevel, TextStyle, TextStyleDelta
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True, slots=True)
|
|
15
|
+
class _StyleDefinition:
|
|
16
|
+
"""保存一个命名样式的父级、文本增量和跨格式投影属性。"""
|
|
17
|
+
|
|
18
|
+
name: str
|
|
19
|
+
family: str
|
|
20
|
+
parent: str | None
|
|
21
|
+
display_name: str | None
|
|
22
|
+
text_delta: TextStyleDelta
|
|
23
|
+
master_page_name: str | None
|
|
24
|
+
table_display: bool | None
|
|
25
|
+
drawing_page_visible: bool | None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _numeric_font_weight(value: str) -> int | None:
|
|
29
|
+
"""在整数转换前验证 CSS 数字字重的词法长度和有效范围。"""
|
|
30
|
+
normalized = value.strip()
|
|
31
|
+
if not normalized or len(normalized) > 4 or not normalized.isascii() or not normalized.isdigit():
|
|
32
|
+
return None
|
|
33
|
+
weight = int(normalized)
|
|
34
|
+
return weight if 1 <= weight <= 1_000 else None
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class OdfStyles:
|
|
38
|
+
"""合并 styles.xml 与 content.xml 中的 ODF 样式定义。"""
|
|
39
|
+
|
|
40
|
+
def __init__(self, *roots: etree._Element | None) -> None:
|
|
41
|
+
"""按传入顺序收集文档样式,使 content.xml 自动样式覆盖基础定义。"""
|
|
42
|
+
self._styles: dict[tuple[str, str], _StyleDefinition] = {}
|
|
43
|
+
self._defaults: dict[str, TextStyleDelta] = {}
|
|
44
|
+
self._list_styles: dict[str, dict[int, ListLevel]] = {}
|
|
45
|
+
self._resolved_text: dict[tuple[str, str], TextStyleDelta] = {}
|
|
46
|
+
self._resolved_table_display: dict[str, bool | None] = {}
|
|
47
|
+
self._resolved_drawing_page_visibility: dict[str, bool | None] = {}
|
|
48
|
+
self._master_pages: dict[str, etree._Element] = {}
|
|
49
|
+
for root in roots:
|
|
50
|
+
if root is not None:
|
|
51
|
+
self._collect(root)
|
|
52
|
+
|
|
53
|
+
def _collect(self, root: etree._Element) -> None:
|
|
54
|
+
"""从一棵 ODF XML 树收集默认、命名、列表和 master-page 样式。"""
|
|
55
|
+
for default in root.iter(qname("style", "default-style")):
|
|
56
|
+
family = default.get(qname("style", "family"))
|
|
57
|
+
if family:
|
|
58
|
+
self._defaults[family] = self._text_delta(default)
|
|
59
|
+
for style in root.iter(qname("style", "style")):
|
|
60
|
+
name = style.get(qname("style", "name"))
|
|
61
|
+
family = style.get(qname("style", "family"))
|
|
62
|
+
if not name or not family:
|
|
63
|
+
continue
|
|
64
|
+
self._styles[(family, name)] = _StyleDefinition(
|
|
65
|
+
name=name,
|
|
66
|
+
family=family,
|
|
67
|
+
parent=style.get(qname("style", "parent-style-name")),
|
|
68
|
+
display_name=style.get(qname("style", "display-name")),
|
|
69
|
+
text_delta=self._text_delta(style),
|
|
70
|
+
master_page_name=style.get(qname("style", "master-page-name")),
|
|
71
|
+
table_display=self._table_display(style),
|
|
72
|
+
drawing_page_visible=self._drawing_page_visibility(style),
|
|
73
|
+
)
|
|
74
|
+
for list_style in root.iter(qname("text", "list-style")):
|
|
75
|
+
name = list_style.get(qname("style", "name"))
|
|
76
|
+
if name:
|
|
77
|
+
self._list_styles[name] = self._parse_list_style(list_style)
|
|
78
|
+
for master_page in root.iter(qname("style", "master-page")):
|
|
79
|
+
name = master_page.get(qname("style", "name"))
|
|
80
|
+
if name:
|
|
81
|
+
self._master_pages[name] = master_page
|
|
82
|
+
|
|
83
|
+
@staticmethod
|
|
84
|
+
def _text_delta(style: etree._Element) -> TextStyleDelta:
|
|
85
|
+
"""把 style:text-properties 转换为可继承的三态样式增量。"""
|
|
86
|
+
properties = style.find(qname("style", "text-properties"))
|
|
87
|
+
if properties is None:
|
|
88
|
+
return TextStyleDelta()
|
|
89
|
+
weight = properties.get(qname("fo", "font-weight"))
|
|
90
|
+
bold: bool | None = None
|
|
91
|
+
if weight is not None:
|
|
92
|
+
normalized_weight = weight.strip().casefold()
|
|
93
|
+
numeric_weight = _numeric_font_weight(normalized_weight)
|
|
94
|
+
bold = numeric_weight >= 600 if numeric_weight is not None else normalized_weight == "bold"
|
|
95
|
+
font_style = properties.get(qname("fo", "font-style"))
|
|
96
|
+
italic = None if font_style is None else font_style.casefold() in {"italic", "oblique"}
|
|
97
|
+
underline_style = properties.get(qname("style", "text-underline-style"))
|
|
98
|
+
underline = None if underline_style is None else underline_style.casefold() != "none"
|
|
99
|
+
strike_style = properties.get(qname("style", "text-line-through-style"))
|
|
100
|
+
strikethrough = None if strike_style is None else strike_style.casefold() != "none"
|
|
101
|
+
position = (properties.get(qname("style", "text-position")) or "").strip().casefold()
|
|
102
|
+
superscript: bool | None = None
|
|
103
|
+
subscript: bool | None = None
|
|
104
|
+
if position:
|
|
105
|
+
superscript = position.startswith("super") or OdfStyles._position_is_positive(position)
|
|
106
|
+
subscript = position.startswith("sub") or OdfStyles._position_is_negative(position)
|
|
107
|
+
if position == "0%":
|
|
108
|
+
superscript = False
|
|
109
|
+
subscript = False
|
|
110
|
+
return TextStyleDelta(
|
|
111
|
+
bold=bold,
|
|
112
|
+
italic=italic,
|
|
113
|
+
underline=underline,
|
|
114
|
+
strikethrough=strikethrough,
|
|
115
|
+
superscript=superscript,
|
|
116
|
+
subscript=subscript,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
@staticmethod
|
|
120
|
+
def _position_is_positive(position: str) -> bool:
|
|
121
|
+
"""判断百分比形式的 text-position 是否表示上标。"""
|
|
122
|
+
try:
|
|
123
|
+
return float(position.split("%", 1)[0]) > 0
|
|
124
|
+
except ValueError:
|
|
125
|
+
return False
|
|
126
|
+
|
|
127
|
+
@staticmethod
|
|
128
|
+
def _position_is_negative(position: str) -> bool:
|
|
129
|
+
"""判断百分比形式的 text-position 是否表示下标。"""
|
|
130
|
+
try:
|
|
131
|
+
return float(position.split("%", 1)[0]) < 0
|
|
132
|
+
except ValueError:
|
|
133
|
+
return False
|
|
134
|
+
|
|
135
|
+
@staticmethod
|
|
136
|
+
def _table_display(style: etree._Element) -> bool | None:
|
|
137
|
+
"""读取表格样式的 display 开关,用于过滤隐藏工作表。"""
|
|
138
|
+
properties = style.find(qname("style", "table-properties"))
|
|
139
|
+
if properties is None:
|
|
140
|
+
return None
|
|
141
|
+
display = properties.get(qname("table", "display"))
|
|
142
|
+
if display is None:
|
|
143
|
+
return None
|
|
144
|
+
return display.casefold() != "false"
|
|
145
|
+
|
|
146
|
+
@staticmethod
|
|
147
|
+
def _drawing_page_visibility(style: etree._Element) -> bool | None:
|
|
148
|
+
"""读取 drawing-page 样式的 presentation visibility。"""
|
|
149
|
+
properties = style.find(qname("style", "drawing-page-properties"))
|
|
150
|
+
if properties is None:
|
|
151
|
+
return None
|
|
152
|
+
visibility = properties.get(qname("presentation", "visibility"))
|
|
153
|
+
if visibility is None:
|
|
154
|
+
return None
|
|
155
|
+
return visibility.casefold() != "hidden"
|
|
156
|
+
|
|
157
|
+
@staticmethod
|
|
158
|
+
def _parse_list_style(element: etree._Element) -> dict[int, ListLevel]:
|
|
159
|
+
"""解析列表样式的层级、类型和通用起始值。"""
|
|
160
|
+
levels: dict[int, ListLevel] = {}
|
|
161
|
+
for child in element:
|
|
162
|
+
if not isinstance(child.tag, str):
|
|
163
|
+
continue
|
|
164
|
+
local_name = etree.QName(child).localname
|
|
165
|
+
if local_name not in {"list-level-style-number", "list-level-style-bullet", "list-level-style-image"}:
|
|
166
|
+
continue
|
|
167
|
+
try:
|
|
168
|
+
level = max(1, int(child.get(qname("text", "level"), "1")))
|
|
169
|
+
except ValueError:
|
|
170
|
+
level = 1
|
|
171
|
+
try:
|
|
172
|
+
start = max(1, int(child.get(qname("text", "start-value"), "1")))
|
|
173
|
+
except ValueError:
|
|
174
|
+
start = 1
|
|
175
|
+
ordered = local_name == "list-level-style-number" and bool(child.get(qname("style", "num-format"), "1"))
|
|
176
|
+
levels[level - 1] = ListLevel(
|
|
177
|
+
ordered=ordered,
|
|
178
|
+
start=start,
|
|
179
|
+
)
|
|
180
|
+
return levels
|
|
181
|
+
|
|
182
|
+
def _resolved_delta(self, family: str, name: str | None) -> TextStyleDelta:
|
|
183
|
+
"""沿 parent-style-name 合并样式,并在循环处安全截断。"""
|
|
184
|
+
if not name:
|
|
185
|
+
return self._defaults.get(family, TextStyleDelta())
|
|
186
|
+
key = (family, name)
|
|
187
|
+
if key in self._resolved_text:
|
|
188
|
+
return self._resolved_text[key]
|
|
189
|
+
chain: list[_StyleDefinition] = []
|
|
190
|
+
seen: set[str] = set()
|
|
191
|
+
current = name
|
|
192
|
+
while current:
|
|
193
|
+
if current in seen:
|
|
194
|
+
logger.warning("ODF style inheritance cycle detected: family={}, style={}", family, current)
|
|
195
|
+
break
|
|
196
|
+
seen.add(current)
|
|
197
|
+
definition = self._styles.get((family, current))
|
|
198
|
+
if definition is None:
|
|
199
|
+
break
|
|
200
|
+
chain.append(definition)
|
|
201
|
+
current = definition.parent or ""
|
|
202
|
+
result = self._defaults.get(family, TextStyleDelta())
|
|
203
|
+
for definition in reversed(chain):
|
|
204
|
+
result = result.merge(definition.text_delta)
|
|
205
|
+
self._resolved_text[key] = result
|
|
206
|
+
return result
|
|
207
|
+
|
|
208
|
+
def text_style(self, style_name: str | None, *, family: str, inherited: TextStyle | None = None) -> TextStyle:
|
|
209
|
+
"""解析指定样式,并让 span 的未声明属性继承当前段落样式。"""
|
|
210
|
+
delta = self._resolved_delta(family, style_name)
|
|
211
|
+
if inherited is not None:
|
|
212
|
+
base = TextStyleDelta(
|
|
213
|
+
bold=inherited.bold,
|
|
214
|
+
italic=inherited.italic,
|
|
215
|
+
underline=inherited.underline,
|
|
216
|
+
strikethrough=inherited.strikethrough,
|
|
217
|
+
superscript=inherited.superscript,
|
|
218
|
+
subscript=inherited.subscript,
|
|
219
|
+
)
|
|
220
|
+
delta = base.merge(delta)
|
|
221
|
+
return delta.resolve()
|
|
222
|
+
|
|
223
|
+
def paragraph_master_page_name(self, style_name: str | None) -> str | None:
|
|
224
|
+
"""沿段落父样式查找首个有效 master-page 名称。"""
|
|
225
|
+
if not style_name:
|
|
226
|
+
return None
|
|
227
|
+
seen: set[str] = set()
|
|
228
|
+
current = style_name
|
|
229
|
+
while current:
|
|
230
|
+
if current in seen:
|
|
231
|
+
logger.warning("ODF style inheritance cycle detected: family=paragraph, style={}", current)
|
|
232
|
+
break
|
|
233
|
+
seen.add(current)
|
|
234
|
+
definition = self._styles.get(("paragraph", current))
|
|
235
|
+
if definition is None:
|
|
236
|
+
break
|
|
237
|
+
if definition.master_page_name is not None:
|
|
238
|
+
return definition.master_page_name
|
|
239
|
+
current = definition.parent or ""
|
|
240
|
+
return None
|
|
241
|
+
|
|
242
|
+
def is_document_title(self, style_name: str | None) -> bool:
|
|
243
|
+
"""沿段落父样式,根据名称和 display-name 判断是否继承文档标题语义。"""
|
|
244
|
+
if not style_name:
|
|
245
|
+
return False
|
|
246
|
+
seen: set[str] = set()
|
|
247
|
+
current = style_name
|
|
248
|
+
while current:
|
|
249
|
+
if current in seen:
|
|
250
|
+
logger.warning("ODF style inheritance cycle detected: family=paragraph, style={}", current)
|
|
251
|
+
break
|
|
252
|
+
seen.add(current)
|
|
253
|
+
definition = self._styles.get(("paragraph", current))
|
|
254
|
+
names = {current, definition.display_name if definition is not None else ""}
|
|
255
|
+
normalized = {name.replace("_20_", " ").replace("_", " ").strip().casefold() for name in names if name}
|
|
256
|
+
if normalized & {"title", "document title", "标题"}:
|
|
257
|
+
return True
|
|
258
|
+
if definition is None:
|
|
259
|
+
break
|
|
260
|
+
current = definition.parent or ""
|
|
261
|
+
return False
|
|
262
|
+
|
|
263
|
+
def list_level(self, style_name: str | None, depth: int) -> ListLevel:
|
|
264
|
+
"""返回指定列表深度的定义,不存在时使用无序默认值。"""
|
|
265
|
+
if not style_name:
|
|
266
|
+
return ListLevel()
|
|
267
|
+
levels = self._list_styles.get(style_name, {})
|
|
268
|
+
return levels.get(depth, ListLevel())
|
|
269
|
+
|
|
270
|
+
def table_is_visible(self, style_name: str | None) -> bool:
|
|
271
|
+
"""沿表格父样式解析 display,子样式显式值优先。"""
|
|
272
|
+
if not style_name:
|
|
273
|
+
return True
|
|
274
|
+
if style_name in self._resolved_table_display:
|
|
275
|
+
return self._resolved_table_display[style_name] is not False
|
|
276
|
+
seen: set[str] = set()
|
|
277
|
+
current = style_name
|
|
278
|
+
resolved: bool | None = None
|
|
279
|
+
while current:
|
|
280
|
+
if current in seen:
|
|
281
|
+
logger.warning("ODF style inheritance cycle detected: family=table, style={}", current)
|
|
282
|
+
break
|
|
283
|
+
seen.add(current)
|
|
284
|
+
definition = self._styles.get(("table", current))
|
|
285
|
+
if definition is None:
|
|
286
|
+
break
|
|
287
|
+
if definition.table_display is not None:
|
|
288
|
+
resolved = definition.table_display
|
|
289
|
+
break
|
|
290
|
+
current = definition.parent or ""
|
|
291
|
+
self._resolved_table_display[style_name] = resolved
|
|
292
|
+
return resolved is not False
|
|
293
|
+
|
|
294
|
+
def drawing_page_is_visible(self, page: etree._Element) -> bool:
|
|
295
|
+
"""解析 ODP 页面直接属性或 drawing-page 样式中的隐藏状态。"""
|
|
296
|
+
direct_visibility = page.get(qname("presentation", "visibility"))
|
|
297
|
+
if direct_visibility is not None:
|
|
298
|
+
return direct_visibility.casefold() != "hidden"
|
|
299
|
+
style_name = page.get(qname("draw", "style-name"))
|
|
300
|
+
if not style_name:
|
|
301
|
+
return True
|
|
302
|
+
if style_name in self._resolved_drawing_page_visibility:
|
|
303
|
+
return self._resolved_drawing_page_visibility[style_name] is not False
|
|
304
|
+
seen: set[str] = set()
|
|
305
|
+
current = style_name
|
|
306
|
+
resolved: bool | None = None
|
|
307
|
+
while current:
|
|
308
|
+
if current in seen:
|
|
309
|
+
logger.warning("ODF style inheritance cycle detected: family=drawing-page, style={}", current)
|
|
310
|
+
break
|
|
311
|
+
seen.add(current)
|
|
312
|
+
definition = self._styles.get(("drawing-page", current))
|
|
313
|
+
if definition is None:
|
|
314
|
+
break
|
|
315
|
+
if definition.drawing_page_visible is not None:
|
|
316
|
+
resolved = definition.drawing_page_visible
|
|
317
|
+
break
|
|
318
|
+
current = definition.parent or ""
|
|
319
|
+
self._resolved_drawing_page_visibility[style_name] = resolved
|
|
320
|
+
return resolved is not False
|
|
321
|
+
|
|
322
|
+
def master_page(self, name: str | None) -> etree._Element | None:
|
|
323
|
+
"""返回指定 master-page;空名称时优先使用第一个定义。"""
|
|
324
|
+
if name and name in self._master_pages:
|
|
325
|
+
return self._master_pages[name]
|
|
326
|
+
return next(iter(self._master_pages.values()), None)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
__all__ = ["OdfStyles"]
|