docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
"""把旧版 PPT 内部语义模型转换为 DocVortex 分页 model-list。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, BinaryIO
|
|
6
|
+
|
|
7
|
+
from ..legacy.ole import BoundedOleReader
|
|
8
|
+
from ..._shared.xycut import sort_entries
|
|
9
|
+
from ..streams import read_stream_bytes_from_start
|
|
10
|
+
from .....schema import BlockType
|
|
11
|
+
from ..rich_text import OfficeRichTextSegment, build_rich_text_from_segments, build_rich_text_html_from_segments
|
|
12
|
+
|
|
13
|
+
from .models import (
|
|
14
|
+
PptChartElement,
|
|
15
|
+
PptEquationElement,
|
|
16
|
+
PptImageElement,
|
|
17
|
+
PptParagraph,
|
|
18
|
+
PptPresentation,
|
|
19
|
+
PptSlide,
|
|
20
|
+
PptTableCell,
|
|
21
|
+
PptTableElement,
|
|
22
|
+
PptTextElement,
|
|
23
|
+
PptTextRun,
|
|
24
|
+
)
|
|
25
|
+
from .parser import parse_ppt_document
|
|
26
|
+
|
|
27
|
+
PPT_XYCUT_BETA = 2.0
|
|
28
|
+
PPT_XYCUT_DENSITY_THRESHOLD = 0.9
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class PptConverter:
|
|
32
|
+
"""将 PowerPoint 97–2003 二进制流转换为分页 raw blocks。"""
|
|
33
|
+
|
|
34
|
+
def __init__(self) -> None:
|
|
35
|
+
"""初始化无状态转换器输出。"""
|
|
36
|
+
|
|
37
|
+
self.pages: list[list[dict[str, Any]]] = []
|
|
38
|
+
|
|
39
|
+
def convert(self, file_binary: BinaryIO) -> None:
|
|
40
|
+
"""读取输入流、解析三个核心 OLE streams 并生成 model-list。"""
|
|
41
|
+
|
|
42
|
+
file_bytes = read_stream_bytes_from_start(file_binary)
|
|
43
|
+
with BoundedOleReader(file_bytes) as ole:
|
|
44
|
+
presentation = parse_ppt_document(
|
|
45
|
+
ole.read_stream("PowerPoint Document"),
|
|
46
|
+
current_user=ole.read_stream("Current User", required=False),
|
|
47
|
+
pictures=ole.read_stream("Pictures", required=False),
|
|
48
|
+
)
|
|
49
|
+
self.pages = self._presentation_to_pages(presentation)
|
|
50
|
+
|
|
51
|
+
@staticmethod
|
|
52
|
+
def _run_styles(run: PptTextRun) -> list[str]:
|
|
53
|
+
"""把内部字符属性转换为 DocVortex 富文本样式名。"""
|
|
54
|
+
|
|
55
|
+
styles: list[str] = []
|
|
56
|
+
if run.bold:
|
|
57
|
+
styles.append("bold")
|
|
58
|
+
if run.italic:
|
|
59
|
+
styles.append("italic")
|
|
60
|
+
if run.underline:
|
|
61
|
+
styles.append("underline")
|
|
62
|
+
if run.strike:
|
|
63
|
+
styles.append("strikethrough")
|
|
64
|
+
if run.baseline is not None and run.baseline > 0:
|
|
65
|
+
styles.append("superscript")
|
|
66
|
+
elif run.baseline is not None and run.baseline < 0:
|
|
67
|
+
styles.append("subscript")
|
|
68
|
+
return styles
|
|
69
|
+
|
|
70
|
+
@classmethod
|
|
71
|
+
def _paragraph_content(cls, paragraph: PptParagraph) -> list[dict[str, Any]]:
|
|
72
|
+
"""把段落 run 直接构建为结构化 Span。"""
|
|
73
|
+
|
|
74
|
+
segments = [
|
|
75
|
+
OfficeRichTextSegment(
|
|
76
|
+
text=run.text.replace("\n", " "),
|
|
77
|
+
style=cls._run_styles(run),
|
|
78
|
+
hyperlink=run.hyperlink,
|
|
79
|
+
)
|
|
80
|
+
for run in paragraph.runs
|
|
81
|
+
if run.text
|
|
82
|
+
]
|
|
83
|
+
return build_rich_text_from_segments(segments, trim_plain_edges=True)
|
|
84
|
+
|
|
85
|
+
@classmethod
|
|
86
|
+
def _append_list_paragraph(
|
|
87
|
+
cls,
|
|
88
|
+
root_blocks: list[dict[str, Any]],
|
|
89
|
+
stack: list[dict[str, Any]],
|
|
90
|
+
paragraph: PptParagraph,
|
|
91
|
+
) -> None:
|
|
92
|
+
"""把一个列表段落放入对应层级,并按需创建中间列表。"""
|
|
93
|
+
|
|
94
|
+
depth = max(0, int(paragraph.depth))
|
|
95
|
+
while len(stack) > depth + 1:
|
|
96
|
+
stack.pop()
|
|
97
|
+
while len(stack) < depth + 1:
|
|
98
|
+
attribute = paragraph.list_kind or "unordered"
|
|
99
|
+
new_list: dict[str, Any] = {
|
|
100
|
+
"type": BlockType.LIST,
|
|
101
|
+
"attribute": attribute,
|
|
102
|
+
"ilevel": len(stack),
|
|
103
|
+
"content": [],
|
|
104
|
+
}
|
|
105
|
+
if attribute == "ordered" and paragraph.start is not None:
|
|
106
|
+
new_list["start"] = paragraph.start
|
|
107
|
+
if stack:
|
|
108
|
+
stack[-1]["content"].append(new_list)
|
|
109
|
+
else:
|
|
110
|
+
root_blocks.append(new_list)
|
|
111
|
+
stack.append(new_list)
|
|
112
|
+
current = stack[depth]
|
|
113
|
+
expected_attribute = paragraph.list_kind or "unordered"
|
|
114
|
+
if current.get("attribute") != expected_attribute:
|
|
115
|
+
del stack[depth:]
|
|
116
|
+
cls._append_list_paragraph(root_blocks, stack, paragraph)
|
|
117
|
+
return
|
|
118
|
+
content = cls._paragraph_content(paragraph)
|
|
119
|
+
if content:
|
|
120
|
+
current["content"].append({"type": BlockType.TEXT, "content": content})
|
|
121
|
+
|
|
122
|
+
@classmethod
|
|
123
|
+
def _text_element_blocks(
|
|
124
|
+
cls,
|
|
125
|
+
element: PptTextElement,
|
|
126
|
+
*,
|
|
127
|
+
title_candidate: bool,
|
|
128
|
+
) -> list[dict[str, Any]]:
|
|
129
|
+
"""把文本形状转换为标题、正文和嵌套列表 raw blocks。"""
|
|
130
|
+
|
|
131
|
+
blocks: list[dict[str, Any]] = []
|
|
132
|
+
list_stack: list[dict[str, Any]] = []
|
|
133
|
+
title_consumed = False
|
|
134
|
+
for paragraph in element.paragraphs:
|
|
135
|
+
content = cls._paragraph_content(paragraph)
|
|
136
|
+
if not content:
|
|
137
|
+
continue
|
|
138
|
+
if title_candidate and not title_consumed and paragraph.list_kind is None:
|
|
139
|
+
blocks.append(
|
|
140
|
+
{
|
|
141
|
+
"type": BlockType.PARAGRAPH_TITLE,
|
|
142
|
+
"content": content,
|
|
143
|
+
"level": 2,
|
|
144
|
+
"_ppt_title_candidate": True,
|
|
145
|
+
}
|
|
146
|
+
)
|
|
147
|
+
title_consumed = True
|
|
148
|
+
list_stack.clear()
|
|
149
|
+
continue
|
|
150
|
+
if paragraph.list_kind is not None:
|
|
151
|
+
cls._append_list_paragraph(blocks, list_stack, paragraph)
|
|
152
|
+
continue
|
|
153
|
+
list_stack.clear()
|
|
154
|
+
blocks.append({"type": BlockType.TEXT, "content": content})
|
|
155
|
+
return blocks
|
|
156
|
+
|
|
157
|
+
@classmethod
|
|
158
|
+
def _table_cell_content(cls, cell: PptTableCell) -> str:
|
|
159
|
+
"""把表格单元格内的多个段落连接为 HTML 内容。"""
|
|
160
|
+
paragraphs: list[str] = []
|
|
161
|
+
for paragraph in cell.paragraphs:
|
|
162
|
+
segments = [
|
|
163
|
+
OfficeRichTextSegment(
|
|
164
|
+
text=run.text.replace("\n", " "),
|
|
165
|
+
style=cls._run_styles(run),
|
|
166
|
+
hyperlink=run.hyperlink,
|
|
167
|
+
)
|
|
168
|
+
for run in paragraph.runs
|
|
169
|
+
if run.text
|
|
170
|
+
]
|
|
171
|
+
if content := build_rich_text_html_from_segments(segments, trim_plain_edges=True):
|
|
172
|
+
paragraphs.append(content)
|
|
173
|
+
return "<br/>".join(paragraphs)
|
|
174
|
+
|
|
175
|
+
@classmethod
|
|
176
|
+
def _table_html(cls, table: PptTableElement) -> str:
|
|
177
|
+
"""按原点单元格生成带 rowspan/colspan 的稳定 HTML 表格。"""
|
|
178
|
+
|
|
179
|
+
origins = {(cell.row, cell.col): cell for cell in table.cells}
|
|
180
|
+
covered: set[tuple[int, int]] = set()
|
|
181
|
+
rows: list[str] = []
|
|
182
|
+
for row in range(table.rows):
|
|
183
|
+
cells: list[str] = []
|
|
184
|
+
for col in range(table.cols):
|
|
185
|
+
if (row, col) in covered:
|
|
186
|
+
continue
|
|
187
|
+
cell = origins.get((row, col))
|
|
188
|
+
if cell is None:
|
|
189
|
+
cells.append("<td></td>")
|
|
190
|
+
continue
|
|
191
|
+
attributes: list[str] = []
|
|
192
|
+
if cell.row_span > 1:
|
|
193
|
+
attributes.append(f'rowspan="{cell.row_span}"')
|
|
194
|
+
if cell.col_span > 1:
|
|
195
|
+
attributes.append(f'colspan="{cell.col_span}"')
|
|
196
|
+
attribute_text = f" {' '.join(attributes)}" if attributes else ""
|
|
197
|
+
cells.append(f"<td{attribute_text}>{cls._table_cell_content(cell)}</td>")
|
|
198
|
+
for covered_row in range(row, row + cell.row_span):
|
|
199
|
+
for covered_col in range(col, col + cell.col_span):
|
|
200
|
+
if (covered_row, covered_col) != (row, col):
|
|
201
|
+
covered.add((covered_row, covered_col))
|
|
202
|
+
rows.append(f"<tr>{''.join(cells)}</tr>")
|
|
203
|
+
return f'<table border="1">{"".join(rows)}</table>'
|
|
204
|
+
|
|
205
|
+
@classmethod
|
|
206
|
+
def _element_blocks(
|
|
207
|
+
cls,
|
|
208
|
+
element: PptTextElement | PptImageElement | PptEquationElement | PptChartElement | PptTableElement,
|
|
209
|
+
*,
|
|
210
|
+
slide_height: int,
|
|
211
|
+
is_first_text_element: bool,
|
|
212
|
+
) -> list[dict[str, Any]]:
|
|
213
|
+
"""把一个语义元素转换为 raw blocks。"""
|
|
214
|
+
|
|
215
|
+
if isinstance(element, PptImageElement):
|
|
216
|
+
return [{"type": BlockType.IMAGE, "image_base64": element.image_base64}]
|
|
217
|
+
if isinstance(element, PptEquationElement):
|
|
218
|
+
return [{"type": BlockType.EQUATION, "content": element.latex}]
|
|
219
|
+
if isinstance(element, PptChartElement):
|
|
220
|
+
block: dict[str, Any] = {
|
|
221
|
+
"type": BlockType.CHART,
|
|
222
|
+
"content": element.content,
|
|
223
|
+
}
|
|
224
|
+
if element.image_base64:
|
|
225
|
+
block["image_base64"] = element.image_base64
|
|
226
|
+
return [block]
|
|
227
|
+
if isinstance(element, PptTableElement):
|
|
228
|
+
return [{"type": BlockType.TABLE, "content": cls._table_html(element)}]
|
|
229
|
+
|
|
230
|
+
authoritative_title = element.text_type in {0, 6}
|
|
231
|
+
first_paragraph = element.paragraphs[0] if element.paragraphs else None
|
|
232
|
+
heuristic_title = bool(
|
|
233
|
+
is_first_text_element
|
|
234
|
+
and element.is_placeholder
|
|
235
|
+
and first_paragraph is not None
|
|
236
|
+
and first_paragraph.list_kind is None
|
|
237
|
+
and element.bbox[1] <= slide_height * 0.35
|
|
238
|
+
and len("".join(run.text for run in first_paragraph.runs).strip()) <= 200
|
|
239
|
+
)
|
|
240
|
+
return cls._text_element_blocks(
|
|
241
|
+
element,
|
|
242
|
+
title_candidate=authoritative_title or heuristic_title,
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
@classmethod
|
|
246
|
+
def _slide_to_page(cls, slide: PptSlide, presentation: PptPresentation) -> list[dict[str, Any]]:
|
|
247
|
+
"""按 XYCut++ 排序一张幻灯片,并把备注稳定追加到末尾。"""
|
|
248
|
+
|
|
249
|
+
entries: list[dict[str, Any]] = []
|
|
250
|
+
text_seen = False
|
|
251
|
+
for element in slide.elements:
|
|
252
|
+
is_first_text = isinstance(element, PptTextElement) and not text_seen
|
|
253
|
+
blocks = cls._element_blocks(
|
|
254
|
+
element,
|
|
255
|
+
slide_height=presentation.height,
|
|
256
|
+
is_first_text_element=is_first_text,
|
|
257
|
+
)
|
|
258
|
+
if isinstance(element, PptTextElement) and blocks:
|
|
259
|
+
text_seen = True
|
|
260
|
+
if blocks:
|
|
261
|
+
entries.append({"bbox": element.bbox, "blocks": blocks, "order": element.order})
|
|
262
|
+
ordered_entries = sort_entries(
|
|
263
|
+
entries,
|
|
264
|
+
beta=PPT_XYCUT_BETA,
|
|
265
|
+
density_threshold=PPT_XYCUT_DENSITY_THRESHOLD,
|
|
266
|
+
)
|
|
267
|
+
page = [block for entry in ordered_entries for block in entry["blocks"]]
|
|
268
|
+
for paragraph in slide.notes:
|
|
269
|
+
content = cls._paragraph_content(paragraph)
|
|
270
|
+
if content:
|
|
271
|
+
page.append({"type": BlockType.PAGE_FOOTNOTE, "content": content})
|
|
272
|
+
return page
|
|
273
|
+
|
|
274
|
+
@classmethod
|
|
275
|
+
def _presentation_to_pages(cls, presentation: PptPresentation) -> list[list[dict[str, Any]]]:
|
|
276
|
+
"""转换整份演示文稿,并把首个有效标题提升为文档标题。"""
|
|
277
|
+
|
|
278
|
+
pages = [cls._slide_to_page(slide, presentation) for slide in presentation.slides]
|
|
279
|
+
document_title_promoted = False
|
|
280
|
+
for page in pages:
|
|
281
|
+
candidates = [block for block in page if block.pop("_ppt_title_candidate", False)]
|
|
282
|
+
if not candidates:
|
|
283
|
+
continue
|
|
284
|
+
first = candidates[0]
|
|
285
|
+
if not document_title_promoted:
|
|
286
|
+
first["type"] = BlockType.DOC_TITLE
|
|
287
|
+
first["level"] = 1
|
|
288
|
+
document_title_promoted = True
|
|
289
|
+
for candidate in candidates[1:] if first["type"] == BlockType.DOC_TITLE else candidates:
|
|
290
|
+
candidate["type"] = BlockType.PARAGRAPH_TITLE
|
|
291
|
+
candidate["level"] = 2
|
|
292
|
+
return pages
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""有界读取 MS-PPT 与 OfficeArt 记录流。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import struct
|
|
7
|
+
from typing import Iterator
|
|
8
|
+
|
|
9
|
+
from ..errors import LegacyOfficeMalformedError, LegacyOfficeResourceLimitError
|
|
10
|
+
from ..limits import MAX_RECORD_DEPTH, MAX_RECORDS
|
|
11
|
+
|
|
12
|
+
CONTAINER_VERSION = 0xF
|
|
13
|
+
ROUNDTRIP_OPAQUE_MIN = 1053
|
|
14
|
+
ROUNDTRIP_OPAQUE_MAX = 1064
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class PptRecord:
|
|
19
|
+
"""一条已验证边界的 PPT/OfficeArt 记录。"""
|
|
20
|
+
|
|
21
|
+
offset: int
|
|
22
|
+
version: int
|
|
23
|
+
instance: int
|
|
24
|
+
record_type: int
|
|
25
|
+
payload: bytes
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(slots=True)
|
|
29
|
+
class RecordBudget:
|
|
30
|
+
"""跨解析阶段累计记录访问次数。"""
|
|
31
|
+
|
|
32
|
+
count: int = 0
|
|
33
|
+
|
|
34
|
+
def charge(self) -> None:
|
|
35
|
+
"""计入一条记录,超过固定上限时硬失败。"""
|
|
36
|
+
|
|
37
|
+
self.count += 1
|
|
38
|
+
if self.count > MAX_RECORDS:
|
|
39
|
+
raise LegacyOfficeResourceLimitError(f"record stream exceeds max_records={MAX_RECORDS}")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def record_at(
|
|
43
|
+
data: bytes,
|
|
44
|
+
offset: int,
|
|
45
|
+
*,
|
|
46
|
+
end: int | None = None,
|
|
47
|
+
strict: bool = False,
|
|
48
|
+
budget: RecordBudget | None = None,
|
|
49
|
+
) -> PptRecord | None:
|
|
50
|
+
"""读取指定偏移的单条记录;严格模式下把坏边界转换为稳定错误。"""
|
|
51
|
+
|
|
52
|
+
limit = len(data) if end is None else min(end, len(data))
|
|
53
|
+
if offset < 0 or offset + 8 > limit:
|
|
54
|
+
if strict:
|
|
55
|
+
raise LegacyOfficeMalformedError("PowerPoint record header is truncated")
|
|
56
|
+
return None
|
|
57
|
+
version_instance, record_type, length = struct.unpack_from("<HHI", data, offset)
|
|
58
|
+
payload_start = offset + 8
|
|
59
|
+
payload_end = payload_start + length
|
|
60
|
+
if payload_end < payload_start or payload_end > limit:
|
|
61
|
+
if strict:
|
|
62
|
+
raise LegacyOfficeMalformedError("PowerPoint record extends beyond its container")
|
|
63
|
+
return None
|
|
64
|
+
if budget is not None:
|
|
65
|
+
budget.charge()
|
|
66
|
+
return PptRecord(
|
|
67
|
+
offset=offset,
|
|
68
|
+
version=version_instance & 0xF,
|
|
69
|
+
instance=version_instance >> 4,
|
|
70
|
+
record_type=record_type,
|
|
71
|
+
payload=data[payload_start:payload_end],
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def iter_records(
|
|
76
|
+
data: bytes,
|
|
77
|
+
*,
|
|
78
|
+
start: int = 0,
|
|
79
|
+
end: int | None = None,
|
|
80
|
+
budget: RecordBudget | None = None,
|
|
81
|
+
strict_first: bool = False,
|
|
82
|
+
) -> Iterator[PptRecord]:
|
|
83
|
+
"""按顺序遍历同一容器内的记录,允许尾部生产器填充字节。"""
|
|
84
|
+
|
|
85
|
+
limit = len(data) if end is None else min(end, len(data))
|
|
86
|
+
cursor = start
|
|
87
|
+
first = True
|
|
88
|
+
while cursor < limit:
|
|
89
|
+
record = record_at(
|
|
90
|
+
data,
|
|
91
|
+
cursor,
|
|
92
|
+
end=limit,
|
|
93
|
+
strict=strict_first and first,
|
|
94
|
+
budget=budget,
|
|
95
|
+
)
|
|
96
|
+
if record is None:
|
|
97
|
+
return
|
|
98
|
+
yield record
|
|
99
|
+
cursor += 8 + len(record.payload)
|
|
100
|
+
first = False
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def iter_descendants(
|
|
104
|
+
record: PptRecord,
|
|
105
|
+
*,
|
|
106
|
+
budget: RecordBudget | None = None,
|
|
107
|
+
) -> Iterator[PptRecord]:
|
|
108
|
+
"""用显式栈深度优先遍历容器,跳过不可递归的 round-trip blob。"""
|
|
109
|
+
|
|
110
|
+
if record.version != CONTAINER_VERSION:
|
|
111
|
+
return
|
|
112
|
+
stack: list[tuple[bytes, Iterator[PptRecord]]] = [(record.payload, iter_records(record.payload, budget=budget))]
|
|
113
|
+
while stack:
|
|
114
|
+
if len(stack) > MAX_RECORD_DEPTH:
|
|
115
|
+
raise LegacyOfficeResourceLimitError(f"record nesting exceeds max_record_depth={MAX_RECORD_DEPTH}")
|
|
116
|
+
_, iterator = stack[-1]
|
|
117
|
+
try:
|
|
118
|
+
child = next(iterator)
|
|
119
|
+
except StopIteration:
|
|
120
|
+
stack.pop()
|
|
121
|
+
continue
|
|
122
|
+
yield child
|
|
123
|
+
if child.version == CONTAINER_VERSION and not ROUNDTRIP_OPAQUE_MIN <= child.record_type <= ROUNDTRIP_OPAQUE_MAX:
|
|
124
|
+
stack.append((child.payload, iter_records(child.payload, budget=budget)))
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def utf16_text(payload: bytes) -> str:
|
|
128
|
+
"""容错解码 UTF-16LE 文本并移除末尾 NUL。"""
|
|
129
|
+
|
|
130
|
+
usable = payload[: len(payload) - (len(payload) % 2)]
|
|
131
|
+
return usable.decode("utf-16le", "replace").rstrip("\x00")
|
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""解析 StyleTextPropAtom 与 TextMasterStyleAtom。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from ..legacy.binary import get_i16, get_u16, get_u32
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True, slots=True)
|
|
11
|
+
class ParagraphRun:
|
|
12
|
+
"""一段 UTF-16 文本范围对应的段落属性。"""
|
|
13
|
+
|
|
14
|
+
count: int
|
|
15
|
+
depth: int
|
|
16
|
+
bullet: bool | None = None
|
|
17
|
+
ordered: bool | None = None
|
|
18
|
+
start: int | None = None
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True, slots=True)
|
|
22
|
+
class CharacterRun:
|
|
23
|
+
"""一段 UTF-16 文本范围对应的字符属性。"""
|
|
24
|
+
|
|
25
|
+
count: int
|
|
26
|
+
bold: bool | None = None
|
|
27
|
+
italic: bool | None = None
|
|
28
|
+
underline: bool | None = None
|
|
29
|
+
strike: bool | None = None
|
|
30
|
+
baseline: int | None = None
|
|
31
|
+
pp9rt: int = 0
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True, slots=True)
|
|
35
|
+
class MasterLevel:
|
|
36
|
+
"""一个母版缩进层级的可继承默认值。"""
|
|
37
|
+
|
|
38
|
+
bullet: bool | None = None
|
|
39
|
+
bold: bool | None = None
|
|
40
|
+
italic: bool | None = None
|
|
41
|
+
underline: bool | None = None
|
|
42
|
+
baseline: int | None = None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(slots=True)
|
|
46
|
+
class StyleRuns:
|
|
47
|
+
"""同一文本形状的段落与字符属性序列。"""
|
|
48
|
+
|
|
49
|
+
paragraphs: list[ParagraphRun] = field(default_factory=list)
|
|
50
|
+
characters: list[CharacterRun] = field(default_factory=list)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True, slots=True)
|
|
54
|
+
class _CharacterStyle:
|
|
55
|
+
"""TextCFException 解出的三态字符属性。"""
|
|
56
|
+
|
|
57
|
+
bold: bool | None = None
|
|
58
|
+
italic: bool | None = None
|
|
59
|
+
underline: bool | None = None
|
|
60
|
+
strike: bool | None = None
|
|
61
|
+
baseline: int | None = None
|
|
62
|
+
pp9rt: int = 0
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _parse_paragraph_exception(
|
|
66
|
+
body: bytes,
|
|
67
|
+
position: int,
|
|
68
|
+
) -> tuple[bool | None, int] | None:
|
|
69
|
+
"""解析 TextPFException 并返回显式 bullet 状态与下一偏移。"""
|
|
70
|
+
|
|
71
|
+
mask = get_u32(body, position)
|
|
72
|
+
if mask is None:
|
|
73
|
+
return None
|
|
74
|
+
position += 4
|
|
75
|
+
bullet = None
|
|
76
|
+
if mask & 0x000F:
|
|
77
|
+
flags = get_u16(body, position)
|
|
78
|
+
if flags is None:
|
|
79
|
+
return None
|
|
80
|
+
if mask & 0x0001:
|
|
81
|
+
bullet = bool(flags & 0x0001)
|
|
82
|
+
position += 2
|
|
83
|
+
fixed_sizes = (
|
|
84
|
+
(0x0080, 2),
|
|
85
|
+
(0x0010, 2),
|
|
86
|
+
(0x0040, 2),
|
|
87
|
+
(0x0020, 4),
|
|
88
|
+
(0x0800, 2),
|
|
89
|
+
(0x1000, 2),
|
|
90
|
+
(0x2000, 2),
|
|
91
|
+
(0x4000, 2),
|
|
92
|
+
(0x0100, 2),
|
|
93
|
+
(0x0400, 2),
|
|
94
|
+
(0x8000, 2),
|
|
95
|
+
)
|
|
96
|
+
for property_mask, size in fixed_sizes:
|
|
97
|
+
if mask & property_mask:
|
|
98
|
+
position += size
|
|
99
|
+
if mask & 0x0010_0000:
|
|
100
|
+
tab_count = get_u16(body, position)
|
|
101
|
+
if tab_count is None:
|
|
102
|
+
return None
|
|
103
|
+
position += 2 + tab_count * 4
|
|
104
|
+
if mask & 0x0001_0000:
|
|
105
|
+
position += 2
|
|
106
|
+
if mask & 0x000E_0000:
|
|
107
|
+
position += 2
|
|
108
|
+
if mask & 0x0020_0000:
|
|
109
|
+
position += 2
|
|
110
|
+
if position > len(body):
|
|
111
|
+
return None
|
|
112
|
+
return bullet, position
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _parse_character_exception(
|
|
116
|
+
body: bytes,
|
|
117
|
+
position: int,
|
|
118
|
+
) -> tuple[_CharacterStyle, int] | None:
|
|
119
|
+
"""解析 TextCFException,并保留每个可继承属性的三态值。"""
|
|
120
|
+
|
|
121
|
+
mask = get_u32(body, position)
|
|
122
|
+
if mask is None:
|
|
123
|
+
return None
|
|
124
|
+
position += 4
|
|
125
|
+
bold = italic = underline = strike = None
|
|
126
|
+
pp9rt = 0
|
|
127
|
+
if mask & 0xFFFF:
|
|
128
|
+
flags = get_u16(body, position)
|
|
129
|
+
if flags is None:
|
|
130
|
+
return None
|
|
131
|
+
if mask & 0x0001:
|
|
132
|
+
bold = bool(flags & 0x0001)
|
|
133
|
+
if mask & 0x0002:
|
|
134
|
+
italic = bool(flags & 0x0002)
|
|
135
|
+
if mask & 0x0004:
|
|
136
|
+
underline = bool(flags & 0x0004)
|
|
137
|
+
# 一些生产器把删除线写入扩展 style 位;未声明时继续继承。
|
|
138
|
+
if mask & 0x0100:
|
|
139
|
+
strike = bool(flags & 0x0100)
|
|
140
|
+
# fontStyle 的 4 位 pp9rt 选择 StyleTextProp9 数组条目。
|
|
141
|
+
pp9rt = (flags >> 10) & 0xF
|
|
142
|
+
position += 2
|
|
143
|
+
for property_mask in (0x0001_0000, 0x0020_0000, 0x0040_0000, 0x0080_0000):
|
|
144
|
+
if mask & property_mask:
|
|
145
|
+
position += 2
|
|
146
|
+
if mask & 0x0002_0000:
|
|
147
|
+
position += 2
|
|
148
|
+
if mask & 0x0004_0000:
|
|
149
|
+
position += 4
|
|
150
|
+
baseline = None
|
|
151
|
+
if mask & 0x0008_0000:
|
|
152
|
+
baseline = get_i16(body, position)
|
|
153
|
+
if baseline is None:
|
|
154
|
+
return None
|
|
155
|
+
position += 2
|
|
156
|
+
if position > len(body):
|
|
157
|
+
return None
|
|
158
|
+
return (
|
|
159
|
+
_CharacterStyle(
|
|
160
|
+
bold=bold,
|
|
161
|
+
italic=italic,
|
|
162
|
+
underline=underline,
|
|
163
|
+
strike=strike,
|
|
164
|
+
baseline=baseline,
|
|
165
|
+
pp9rt=pp9rt,
|
|
166
|
+
),
|
|
167
|
+
position,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def parse_style_text(body: bytes, text_utf16_length: int) -> StyleRuns:
|
|
172
|
+
"""按 UTF-16 单元长度解析一个 StyleTextPropAtom。"""
|
|
173
|
+
|
|
174
|
+
runs = StyleRuns()
|
|
175
|
+
position = 0
|
|
176
|
+
covered = 0
|
|
177
|
+
while covered <= text_utf16_length:
|
|
178
|
+
count = get_u32(body, position)
|
|
179
|
+
depth = get_u16(body, position + 4)
|
|
180
|
+
if count is None or depth is None:
|
|
181
|
+
break
|
|
182
|
+
position += 6
|
|
183
|
+
parsed = _parse_paragraph_exception(body, position)
|
|
184
|
+
if parsed is None:
|
|
185
|
+
return runs
|
|
186
|
+
bullet, position = parsed
|
|
187
|
+
runs.paragraphs.append(ParagraphRun(count=int(count), depth=min(int(depth), 8), bullet=bullet))
|
|
188
|
+
covered += int(count)
|
|
189
|
+
if count == 0:
|
|
190
|
+
break
|
|
191
|
+
|
|
192
|
+
covered = 0
|
|
193
|
+
while covered <= text_utf16_length:
|
|
194
|
+
count = get_u32(body, position)
|
|
195
|
+
if count is None:
|
|
196
|
+
break
|
|
197
|
+
position += 4
|
|
198
|
+
parsed = _parse_character_exception(body, position)
|
|
199
|
+
if parsed is None:
|
|
200
|
+
break
|
|
201
|
+
style, position = parsed
|
|
202
|
+
runs.characters.append(
|
|
203
|
+
CharacterRun(
|
|
204
|
+
count=int(count),
|
|
205
|
+
bold=style.bold,
|
|
206
|
+
italic=style.italic,
|
|
207
|
+
underline=style.underline,
|
|
208
|
+
strike=style.strike,
|
|
209
|
+
baseline=style.baseline,
|
|
210
|
+
pp9rt=style.pp9rt,
|
|
211
|
+
)
|
|
212
|
+
)
|
|
213
|
+
covered += int(count)
|
|
214
|
+
if count == 0:
|
|
215
|
+
break
|
|
216
|
+
return runs
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def parse_master_style(body: bytes, instance: int) -> list[MasterLevel]:
|
|
220
|
+
"""解析 TextMasterStyleAtom 的逐层默认字符与列表属性。"""
|
|
221
|
+
|
|
222
|
+
level_count = get_u16(body, 0)
|
|
223
|
+
if level_count is None:
|
|
224
|
+
return []
|
|
225
|
+
position = 2
|
|
226
|
+
result: list[MasterLevel] = []
|
|
227
|
+
for _ in range(min(int(level_count), 10)):
|
|
228
|
+
if instance >= 5:
|
|
229
|
+
position += 2
|
|
230
|
+
parsed_paragraph = _parse_paragraph_exception(body, position)
|
|
231
|
+
if parsed_paragraph is None:
|
|
232
|
+
break
|
|
233
|
+
bullet, position = parsed_paragraph
|
|
234
|
+
parsed_character = _parse_character_exception(body, position)
|
|
235
|
+
if parsed_character is None:
|
|
236
|
+
break
|
|
237
|
+
style, position = parsed_character
|
|
238
|
+
result.append(
|
|
239
|
+
MasterLevel(
|
|
240
|
+
bullet=bullet,
|
|
241
|
+
bold=style.bold,
|
|
242
|
+
italic=style.italic,
|
|
243
|
+
underline=style.underline,
|
|
244
|
+
baseline=style.baseline,
|
|
245
|
+
)
|
|
246
|
+
)
|
|
247
|
+
return result
|