docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
"""解析 DOC CLX piece table 并恢复全局 UTF-16 CP 文本流。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from bisect import bisect_left
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
import struct
|
|
8
|
+
|
|
9
|
+
from ..errors import LegacyOfficeMalformedError
|
|
10
|
+
from ..legacy.binary import bounded_slice, get_u16, get_u32
|
|
11
|
+
|
|
12
|
+
from .records import DocBudget
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True, slots=True)
|
|
16
|
+
class Piece:
|
|
17
|
+
"""一个把逻辑 CP 范围映射到 WordDocument FC 的 piece。"""
|
|
18
|
+
|
|
19
|
+
cp_start: int
|
|
20
|
+
cp_end: int
|
|
21
|
+
fc: int
|
|
22
|
+
compressed: bool
|
|
23
|
+
prm: bytes = b""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(slots=True)
|
|
27
|
+
class TextStream:
|
|
28
|
+
"""字符标量及其 CP、FC 和 piece 映射。"""
|
|
29
|
+
|
|
30
|
+
chars: list[str]
|
|
31
|
+
cps: list[int]
|
|
32
|
+
fcs: list[int]
|
|
33
|
+
piece_indexes: list[int]
|
|
34
|
+
|
|
35
|
+
def index_of_cp(self, cp: int) -> int:
|
|
36
|
+
"""返回首个 CP 不小于目标值的字符索引。"""
|
|
37
|
+
|
|
38
|
+
return bisect_left(self.cps, max(cp, 0))
|
|
39
|
+
|
|
40
|
+
def text_between(self, cp_start: int, cp_end: int) -> str:
|
|
41
|
+
"""返回指定 CP 半开区间的 Unicode 文本。"""
|
|
42
|
+
|
|
43
|
+
return "".join(self.chars[self.index_of_cp(cp_start) : self.index_of_cp(cp_end)])
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _prm0_grpprl(prm: int) -> bytes:
|
|
47
|
+
"""把 Prm0 中已支持的单个属性还原为 grpprl。"""
|
|
48
|
+
|
|
49
|
+
isprm = (prm >> 1) & 0x7F
|
|
50
|
+
value = (prm >> 8) & 0xFF
|
|
51
|
+
opcode = {
|
|
52
|
+
0x0C: 0x260A, # sprmPIlvl
|
|
53
|
+
0x18: 0x2416, # sprmPFInTable
|
|
54
|
+
0x19: 0x2417, # sprmPFTtp
|
|
55
|
+
0x55: 0x0835, # sprmCFBold
|
|
56
|
+
0x56: 0x0836, # sprmCFItalic
|
|
57
|
+
0x57: 0x0837, # sprmCFStrike
|
|
58
|
+
0x78: 0x2640, # sprmPOutLvl
|
|
59
|
+
}.get(isprm)
|
|
60
|
+
if opcode is None:
|
|
61
|
+
return b""
|
|
62
|
+
return struct.pack("<HB", opcode, value)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _parse_plc_pcd(plc: bytes, prcs: list[bytes], budget: DocBudget) -> list[Piece]:
|
|
66
|
+
"""解析 Pcdt 内的 PlcPcd 并绑定 piece Prm。"""
|
|
67
|
+
|
|
68
|
+
if len(plc) < 16 or (len(plc) - 4) % 12:
|
|
69
|
+
raise LegacyOfficeMalformedError("DOC piece table is empty or malformed")
|
|
70
|
+
count = (len(plc) - 4) // 12
|
|
71
|
+
budget.charge(count)
|
|
72
|
+
cp_bytes = (count + 1) * 4
|
|
73
|
+
pieces: list[Piece] = []
|
|
74
|
+
previous_cp = -1
|
|
75
|
+
for index in range(count):
|
|
76
|
+
cp_start = get_u32(plc, index * 4)
|
|
77
|
+
cp_end = get_u32(plc, (index + 1) * 4)
|
|
78
|
+
pcd_offset = cp_bytes + index * 8
|
|
79
|
+
fc_raw = get_u32(plc, pcd_offset + 2)
|
|
80
|
+
prm = get_u16(plc, pcd_offset + 6) or 0
|
|
81
|
+
if cp_start is None or cp_end is None or fc_raw is None:
|
|
82
|
+
raise LegacyOfficeMalformedError("DOC piece table is truncated")
|
|
83
|
+
if cp_start < previous_cp or cp_end < cp_start:
|
|
84
|
+
raise LegacyOfficeMalformedError("DOC piece CP values are not ordered")
|
|
85
|
+
previous_cp = cp_end
|
|
86
|
+
compressed = bool(fc_raw & 0x4000_0000)
|
|
87
|
+
fc = fc_raw & 0x3FFF_FFFF
|
|
88
|
+
if compressed:
|
|
89
|
+
fc //= 2
|
|
90
|
+
grpprl = b""
|
|
91
|
+
if prm & 1:
|
|
92
|
+
prc_index = prm >> 1
|
|
93
|
+
if prc_index < len(prcs):
|
|
94
|
+
grpprl = prcs[prc_index]
|
|
95
|
+
elif prm:
|
|
96
|
+
grpprl = _prm0_grpprl(prm)
|
|
97
|
+
pieces.append(
|
|
98
|
+
Piece(
|
|
99
|
+
cp_start=int(cp_start),
|
|
100
|
+
cp_end=int(cp_end),
|
|
101
|
+
fc=int(fc),
|
|
102
|
+
compressed=compressed,
|
|
103
|
+
prm=grpprl,
|
|
104
|
+
)
|
|
105
|
+
)
|
|
106
|
+
return pieces
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def parse_clx(table_stream: bytes, *, offset: int, size: int, budget: DocBudget) -> list[Piece]:
|
|
110
|
+
"""解析 CLX 中的 Prc 数组和最终 Pcdt。"""
|
|
111
|
+
|
|
112
|
+
clx = bounded_slice(table_stream, offset, size)
|
|
113
|
+
if clx is None:
|
|
114
|
+
raise LegacyOfficeMalformedError("DOC CLX range is out of bounds")
|
|
115
|
+
prcs: list[bytes] = []
|
|
116
|
+
cursor = 0
|
|
117
|
+
while cursor < len(clx):
|
|
118
|
+
kind = clx[cursor]
|
|
119
|
+
if kind == 1:
|
|
120
|
+
length = get_u16(clx, cursor + 1)
|
|
121
|
+
if length is None:
|
|
122
|
+
raise LegacyOfficeMalformedError("DOC CLX Prc is truncated")
|
|
123
|
+
payload = bounded_slice(clx, cursor + 3, length)
|
|
124
|
+
if payload is None:
|
|
125
|
+
raise LegacyOfficeMalformedError("DOC CLX Prc exceeds its range")
|
|
126
|
+
budget.charge()
|
|
127
|
+
prcs.append(payload)
|
|
128
|
+
cursor += 3 + length
|
|
129
|
+
continue
|
|
130
|
+
if kind == 2:
|
|
131
|
+
length = get_u32(clx, cursor + 1)
|
|
132
|
+
if length is None:
|
|
133
|
+
raise LegacyOfficeMalformedError("DOC CLX Pcdt is truncated")
|
|
134
|
+
plc = bounded_slice(clx, cursor + 5, length)
|
|
135
|
+
if plc is None:
|
|
136
|
+
raise LegacyOfficeMalformedError("DOC PlcPcd exceeds CLX")
|
|
137
|
+
return _parse_plc_pcd(plc, prcs, budget)
|
|
138
|
+
raise LegacyOfficeMalformedError("DOC CLX contains an unknown record")
|
|
139
|
+
raise LegacyOfficeMalformedError("DOC CLX does not contain a Pcdt")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def legacy_single_piece(*, fc_min: int, fc_mac: int, ccp_text: int) -> list[Piece]:
|
|
143
|
+
"""为没有 CLX 的非 complex 文档构造保守单 piece。"""
|
|
144
|
+
|
|
145
|
+
if fc_min < 0 or fc_mac <= fc_min:
|
|
146
|
+
return []
|
|
147
|
+
length = min(fc_mac - fc_min, max(ccp_text, 0))
|
|
148
|
+
if length <= 0:
|
|
149
|
+
return []
|
|
150
|
+
return [Piece(cp_start=0, cp_end=length, fc=fc_min, compressed=True)]
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def codec_for_lid(lid: int) -> str:
|
|
154
|
+
"""把 Word LID 映射为 Python 可用的 ANSI/DBCS codec。"""
|
|
155
|
+
|
|
156
|
+
primary = lid & 0x03FF
|
|
157
|
+
if primary == 0x11:
|
|
158
|
+
return "cp932"
|
|
159
|
+
if primary == 0x12:
|
|
160
|
+
return "cp949"
|
|
161
|
+
if primary == 0x04:
|
|
162
|
+
return "cp950" if lid in {0x0404, 0x0C04, 0x1404, 0x7C04} else "cp936"
|
|
163
|
+
if primary in {0x01, 0x20, 0x29}:
|
|
164
|
+
return "cp1256"
|
|
165
|
+
if primary in {0x02, 0x19, 0x22, 0x23}:
|
|
166
|
+
return "cp1251"
|
|
167
|
+
if primary in {0x05, 0x0E, 0x15, 0x18, 0x1A, 0x1B, 0x24}:
|
|
168
|
+
return "cp1250"
|
|
169
|
+
if primary == 0x08:
|
|
170
|
+
return "cp1253"
|
|
171
|
+
if primary == 0x0D:
|
|
172
|
+
return "cp1255"
|
|
173
|
+
if primary == 0x1E:
|
|
174
|
+
return "cp874"
|
|
175
|
+
if primary in {0x1F, 0x2C}:
|
|
176
|
+
return "cp1254"
|
|
177
|
+
if primary in {0x25, 0x26, 0x27}:
|
|
178
|
+
return "cp1257"
|
|
179
|
+
if primary == 0x2A:
|
|
180
|
+
return "cp1258"
|
|
181
|
+
return "cp1252"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _lead_byte(codec: str, value: int) -> bool:
|
|
185
|
+
"""判断一个压缩 piece 字节是否为 DBCS 首字节。"""
|
|
186
|
+
|
|
187
|
+
if codec == "cp932":
|
|
188
|
+
return 0x81 <= value <= 0x9F or 0xE0 <= value <= 0xFC
|
|
189
|
+
if codec in {"cp936", "cp949", "cp950"}:
|
|
190
|
+
return 0x81 <= value <= 0xFE
|
|
191
|
+
return False
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def extract_text(
|
|
195
|
+
word_document: bytes,
|
|
196
|
+
pieces: list[Piece],
|
|
197
|
+
*,
|
|
198
|
+
total_cp: int,
|
|
199
|
+
codec: str,
|
|
200
|
+
budget: DocBudget,
|
|
201
|
+
) -> TextStream:
|
|
202
|
+
"""按 piece 顺序恢复字符,并保留字符到 CP/FC 的反向映射。"""
|
|
203
|
+
|
|
204
|
+
chars: list[str] = []
|
|
205
|
+
cps: list[int] = []
|
|
206
|
+
fcs: list[int] = []
|
|
207
|
+
piece_indexes: list[int] = []
|
|
208
|
+
for piece_index, piece in enumerate(pieces):
|
|
209
|
+
if piece.cp_start >= total_cp:
|
|
210
|
+
break
|
|
211
|
+
cp_cursor = piece.cp_start
|
|
212
|
+
logical_length = min(piece.cp_end, total_cp) - piece.cp_start
|
|
213
|
+
if logical_length <= 0:
|
|
214
|
+
continue
|
|
215
|
+
if piece.compressed:
|
|
216
|
+
payload = bounded_slice(word_document, piece.fc, logical_length)
|
|
217
|
+
if payload is None:
|
|
218
|
+
continue
|
|
219
|
+
cursor = 0
|
|
220
|
+
while cursor < len(payload):
|
|
221
|
+
width = 2 if _lead_byte(codec, payload[cursor]) and cursor + 1 < len(payload) else 1
|
|
222
|
+
decoded = payload[cursor : cursor + width].decode(codec, errors="replace")
|
|
223
|
+
for char in decoded:
|
|
224
|
+
chars.append(char)
|
|
225
|
+
cps.append(cp_cursor)
|
|
226
|
+
fcs.append(piece.fc + cursor)
|
|
227
|
+
piece_indexes.append(piece_index)
|
|
228
|
+
cp_cursor += width
|
|
229
|
+
cursor += width
|
|
230
|
+
budget.charge(width)
|
|
231
|
+
else:
|
|
232
|
+
byte_length = logical_length * 2
|
|
233
|
+
payload = bounded_slice(word_document, piece.fc, byte_length)
|
|
234
|
+
if payload is None:
|
|
235
|
+
continue
|
|
236
|
+
cursor = 0
|
|
237
|
+
while cursor + 2 <= len(payload):
|
|
238
|
+
first = int(struct.unpack_from("<H", payload, cursor)[0])
|
|
239
|
+
width = 2
|
|
240
|
+
units = [first]
|
|
241
|
+
if 0xD800 <= first <= 0xDBFF and cursor + 4 <= len(payload):
|
|
242
|
+
second = int(struct.unpack_from("<H", payload, cursor + 2)[0])
|
|
243
|
+
if 0xDC00 <= second <= 0xDFFF:
|
|
244
|
+
units.append(second)
|
|
245
|
+
width = 4
|
|
246
|
+
raw = struct.pack(f"<{len(units)}H", *units)
|
|
247
|
+
char = raw.decode("utf-16le", errors="replace")
|
|
248
|
+
chars.append(char)
|
|
249
|
+
cps.append(cp_cursor)
|
|
250
|
+
fcs.append(piece.fc + cursor)
|
|
251
|
+
piece_indexes.append(piece_index)
|
|
252
|
+
unit_count = width // 2
|
|
253
|
+
cp_cursor += unit_count
|
|
254
|
+
cursor += width
|
|
255
|
+
budget.charge(unit_count)
|
|
256
|
+
return TextStream(chars=chars, cps=cps, fcs=fcs, piece_indexes=piece_indexes)
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""DOC 二进制结构使用的有界整数、PLC 和记录预算工具。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import struct
|
|
7
|
+
|
|
8
|
+
from ..errors import LegacyOfficeResourceLimitError
|
|
9
|
+
from ..limits import MAX_RECORDS
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(slots=True)
|
|
13
|
+
class DocBudget:
|
|
14
|
+
"""限制 DOC 解析累计访问的记录和文本单元数。"""
|
|
15
|
+
|
|
16
|
+
visited: int = 0
|
|
17
|
+
|
|
18
|
+
def charge(self, amount: int = 1) -> None:
|
|
19
|
+
"""计入本次访问量,超过统一上限时稳定失败。"""
|
|
20
|
+
|
|
21
|
+
if amount < 0 or self.visited + amount > MAX_RECORDS:
|
|
22
|
+
raise LegacyOfficeResourceLimitError(f"DOC records exceed max_records={MAX_RECORDS}")
|
|
23
|
+
self.visited += amount
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def parse_plc(data: bytes, *, item_size: int, budget: DocBudget) -> tuple[list[int], list[bytes]]:
|
|
27
|
+
"""解析由 CP 数组和定长数据项组成的通用 PLC。"""
|
|
28
|
+
|
|
29
|
+
if item_size < 0 or len(data) < 4:
|
|
30
|
+
return [], []
|
|
31
|
+
denominator = 4 + item_size
|
|
32
|
+
payload = len(data) - 4
|
|
33
|
+
if denominator <= 0 or payload % denominator:
|
|
34
|
+
return [], []
|
|
35
|
+
count = payload // denominator
|
|
36
|
+
budget.charge(count + 1)
|
|
37
|
+
cp_bytes = (count + 1) * 4
|
|
38
|
+
cps = [int(struct.unpack_from("<I", data, index * 4)[0]) for index in range(count + 1)]
|
|
39
|
+
items = [data[cp_bytes + index * item_size : cp_bytes + (index + 1) * item_size] for index in range(count)]
|
|
40
|
+
return cps, items
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
"""遍历并应用 Word 二进制单属性修饰符 SPRM。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, replace
|
|
6
|
+
import struct
|
|
7
|
+
from typing import Callable
|
|
8
|
+
|
|
9
|
+
from ..legacy.binary import get_i16, get_u16, get_u32
|
|
10
|
+
from .models import DocCharStyle, DocTableCellFormat, DocTableFormat
|
|
11
|
+
from .records import DocBudget
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _operand_length(opcode: int, operand: bytes) -> int:
|
|
15
|
+
"""根据 SPRM 的 spra 字段计算 operand 字节数。"""
|
|
16
|
+
|
|
17
|
+
spra = opcode >> 13
|
|
18
|
+
if spra in {0, 1}:
|
|
19
|
+
return 1
|
|
20
|
+
if spra in {2, 4, 5}:
|
|
21
|
+
return 2
|
|
22
|
+
if spra == 3:
|
|
23
|
+
return 4
|
|
24
|
+
if spra == 7:
|
|
25
|
+
return 3
|
|
26
|
+
if opcode == 0xD608:
|
|
27
|
+
return (get_u16(operand, 0) or -1) + 1
|
|
28
|
+
return (operand[0] + 1) if operand else 0
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def walk_sprms(
|
|
32
|
+
grpprl: bytes,
|
|
33
|
+
callback: Callable[[int, bytes], None],
|
|
34
|
+
*,
|
|
35
|
+
budget: DocBudget | None = None,
|
|
36
|
+
) -> None:
|
|
37
|
+
"""有界顺序遍历 grpprl,截断尾部按可恢复内容处理。"""
|
|
38
|
+
|
|
39
|
+
cursor = 0
|
|
40
|
+
while cursor + 2 <= len(grpprl):
|
|
41
|
+
opcode = int(struct.unpack_from("<H", grpprl, cursor)[0])
|
|
42
|
+
cursor += 2
|
|
43
|
+
length = _operand_length(opcode, grpprl[cursor:])
|
|
44
|
+
if length <= 0 or cursor + length > len(grpprl):
|
|
45
|
+
return
|
|
46
|
+
if budget is not None:
|
|
47
|
+
budget.charge()
|
|
48
|
+
callback(opcode, grpprl[cursor : cursor + length])
|
|
49
|
+
cursor += length
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _toggle(operand: bytes, base: bool) -> bool | None:
|
|
53
|
+
"""把 Word ToggleOperand 解析为相对样式基值。"""
|
|
54
|
+
|
|
55
|
+
if not operand:
|
|
56
|
+
return None
|
|
57
|
+
return {0: False, 1: True, 0x80: base, 0x81: not base}.get(operand[0])
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def chpx_style_id(grpprl: bytes) -> int | None:
|
|
61
|
+
"""读取 CHPX 指定的字符样式 istd。"""
|
|
62
|
+
|
|
63
|
+
result: int | None = None
|
|
64
|
+
|
|
65
|
+
def consume(opcode: int, operand: bytes) -> None:
|
|
66
|
+
"""记录最后一个有效 sprmCIstd。"""
|
|
67
|
+
|
|
68
|
+
nonlocal result
|
|
69
|
+
if opcode == 0x4A30:
|
|
70
|
+
result = get_u16(operand, 0)
|
|
71
|
+
|
|
72
|
+
walk_sprms(grpprl, consume)
|
|
73
|
+
return result
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def chpx_picture_location(grpprl: bytes) -> int | None:
|
|
77
|
+
"""读取 CHPX 中的 sprmCPicLocation。"""
|
|
78
|
+
|
|
79
|
+
result: int | None = None
|
|
80
|
+
|
|
81
|
+
def consume(opcode: int, operand: bytes) -> None:
|
|
82
|
+
"""记录最后一个有效图片偏移。"""
|
|
83
|
+
|
|
84
|
+
nonlocal result
|
|
85
|
+
if opcode == 0x6A03:
|
|
86
|
+
result = get_u32(operand, 0)
|
|
87
|
+
|
|
88
|
+
walk_sprms(grpprl, consume)
|
|
89
|
+
return result
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def apply_character_sprms(
|
|
93
|
+
grpprl: bytes,
|
|
94
|
+
current: DocCharStyle,
|
|
95
|
+
style_base: DocCharStyle,
|
|
96
|
+
*,
|
|
97
|
+
budget: DocBudget | None = None,
|
|
98
|
+
) -> DocCharStyle:
|
|
99
|
+
"""按 Word 样式覆盖顺序把 CHPX 应用到字符样式。"""
|
|
100
|
+
|
|
101
|
+
style = current
|
|
102
|
+
|
|
103
|
+
def consume(opcode: int, operand: bytes) -> None:
|
|
104
|
+
"""应用当前可表达的字符属性。"""
|
|
105
|
+
|
|
106
|
+
nonlocal style
|
|
107
|
+
toggle_field = {
|
|
108
|
+
0x0800: "deleted", # sprmCFRMarkDel
|
|
109
|
+
0x0802: "hidden", # sprmCFFldVanish
|
|
110
|
+
0x0835: "bold",
|
|
111
|
+
0x0836: "italic",
|
|
112
|
+
0x0837: "strike",
|
|
113
|
+
0x083C: "hidden", # sprmCFVanish
|
|
114
|
+
}.get(opcode)
|
|
115
|
+
if toggle_field is not None:
|
|
116
|
+
value = _toggle(operand, bool(getattr(style_base, toggle_field)))
|
|
117
|
+
if value is not None:
|
|
118
|
+
style = replace(style, **{toggle_field: value})
|
|
119
|
+
return
|
|
120
|
+
if opcode == 0x2A3E and operand: # sprmCKul
|
|
121
|
+
style = replace(style, underline=operand[0] not in {0, 5})
|
|
122
|
+
elif opcode == 0x2A48 and operand: # sprmCIss
|
|
123
|
+
style = replace(
|
|
124
|
+
style,
|
|
125
|
+
superscript=operand[0] == 1,
|
|
126
|
+
subscript=operand[0] == 2,
|
|
127
|
+
)
|
|
128
|
+
elif opcode == 0x2A53 and operand: # sprmCFDStrike
|
|
129
|
+
style = replace(style, strike=operand[0] != 0)
|
|
130
|
+
elif opcode == 0x2A54 and operand: # sprmCEm
|
|
131
|
+
style = replace(style, emphasis=operand[0] != 0)
|
|
132
|
+
|
|
133
|
+
walk_sprms(grpprl, consume, budget=budget)
|
|
134
|
+
return style
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass(frozen=True, slots=True)
|
|
138
|
+
class PapDelta:
|
|
139
|
+
"""PAPX 或段落样式对可见段落属性的增量。"""
|
|
140
|
+
|
|
141
|
+
in_table: bool | None = None
|
|
142
|
+
row_mark: bool | None = None
|
|
143
|
+
outline_level: int | None | object = None
|
|
144
|
+
ilfo: int | None = None
|
|
145
|
+
ilvl: int | None = None
|
|
146
|
+
table_depth: int | None = None
|
|
147
|
+
inner_cell: bool | None = None
|
|
148
|
+
inner_row: bool | None = None
|
|
149
|
+
table: DocTableFormat | None = None
|
|
150
|
+
|
|
151
|
+
def merge(self, over: PapDelta) -> PapDelta:
|
|
152
|
+
"""让后应用的段落属性覆盖当前增量。"""
|
|
153
|
+
|
|
154
|
+
return PapDelta(
|
|
155
|
+
in_table=over.in_table if over.in_table is not None else self.in_table,
|
|
156
|
+
row_mark=over.row_mark if over.row_mark is not None else self.row_mark,
|
|
157
|
+
outline_level=(over.outline_level if over.outline_level is not None else self.outline_level),
|
|
158
|
+
ilfo=over.ilfo if over.ilfo is not None else self.ilfo,
|
|
159
|
+
ilvl=over.ilvl if over.ilvl is not None else self.ilvl,
|
|
160
|
+
table_depth=over.table_depth if over.table_depth is not None else self.table_depth,
|
|
161
|
+
inner_cell=over.inner_cell if over.inner_cell is not None else self.inner_cell,
|
|
162
|
+
inner_row=over.inner_row if over.inner_row is not None else self.inner_row,
|
|
163
|
+
table=over.table if over.table is not None else self.table,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _parse_tdef_table(operand: bytes) -> DocTableFormat | None:
|
|
168
|
+
"""解析 TDefTableOperand 中的边界和横纵向合并标志。"""
|
|
169
|
+
|
|
170
|
+
if len(operand) < 3:
|
|
171
|
+
return None
|
|
172
|
+
columns = operand[2]
|
|
173
|
+
if columns > 63:
|
|
174
|
+
return None
|
|
175
|
+
boundaries: list[int] = []
|
|
176
|
+
for index in range(columns + 1):
|
|
177
|
+
value = get_i16(operand, 3 + index * 2)
|
|
178
|
+
if value is None:
|
|
179
|
+
return None
|
|
180
|
+
boundaries.append(value)
|
|
181
|
+
cells: list[DocTableCellFormat] = []
|
|
182
|
+
tc_base = 3 + (columns + 1) * 2
|
|
183
|
+
for index in range(columns):
|
|
184
|
+
flags = get_u16(operand, tc_base + index * 20) or 0
|
|
185
|
+
horizontal = flags & 0x3
|
|
186
|
+
vertical = (flags >> 5) & 0x3
|
|
187
|
+
right = boundaries[index + 1]
|
|
188
|
+
cells.append(
|
|
189
|
+
DocTableCellFormat(
|
|
190
|
+
right=right,
|
|
191
|
+
horizontal_first=horizontal >= 2,
|
|
192
|
+
horizontal_continue=horizontal == 1,
|
|
193
|
+
vertical_first=vertical == 3,
|
|
194
|
+
vertical_continue=vertical == 1,
|
|
195
|
+
)
|
|
196
|
+
)
|
|
197
|
+
return DocTableFormat(boundaries=tuple(boundaries), cells=tuple(cells))
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def apply_paragraph_sprms(
|
|
201
|
+
grpprl: bytes,
|
|
202
|
+
data_stream: bytes,
|
|
203
|
+
initial: PapDelta | None = None,
|
|
204
|
+
*,
|
|
205
|
+
budget: DocBudget | None = None,
|
|
206
|
+
) -> PapDelta:
|
|
207
|
+
"""应用 PAPX SPRM,并解析 huge PAPX 与表格行属性。"""
|
|
208
|
+
|
|
209
|
+
delta = initial or PapDelta()
|
|
210
|
+
|
|
211
|
+
def consume(opcode: int, operand: bytes) -> None:
|
|
212
|
+
"""应用一个段落或表格属性。"""
|
|
213
|
+
|
|
214
|
+
nonlocal delta
|
|
215
|
+
if opcode == 0x2416 and operand:
|
|
216
|
+
delta = replace(delta, in_table=operand[0] != 0)
|
|
217
|
+
elif opcode == 0x2417 and operand:
|
|
218
|
+
delta = replace(delta, row_mark=operand[0] != 0)
|
|
219
|
+
elif opcode == 0x6646:
|
|
220
|
+
offset = get_u32(operand, 0)
|
|
221
|
+
length = get_u16(data_stream, offset) if offset is not None else None
|
|
222
|
+
if offset is not None and length is not None and offset + 2 + length <= len(data_stream):
|
|
223
|
+
delta = apply_paragraph_sprms(
|
|
224
|
+
data_stream[offset + 2 : offset + 2 + length],
|
|
225
|
+
b"",
|
|
226
|
+
delta,
|
|
227
|
+
budget=budget,
|
|
228
|
+
)
|
|
229
|
+
elif opcode == 0x2640 and operand:
|
|
230
|
+
delta = replace(delta, outline_level=operand[0] + 1 if operand[0] < 9 else -1)
|
|
231
|
+
elif opcode == 0x260A and operand:
|
|
232
|
+
delta = replace(delta, ilvl=int(operand[0]))
|
|
233
|
+
elif opcode == 0x460B:
|
|
234
|
+
delta = replace(delta, ilfo=get_u16(operand, 0))
|
|
235
|
+
elif opcode == 0x6649:
|
|
236
|
+
depth = get_u32(operand, 0)
|
|
237
|
+
if depth is not None:
|
|
238
|
+
delta = replace(delta, table_depth=int(depth))
|
|
239
|
+
elif opcode == 0x664A:
|
|
240
|
+
raw = get_u32(operand, 0)
|
|
241
|
+
if raw is not None:
|
|
242
|
+
signed = struct.unpack("<i", struct.pack("<I", raw))[0]
|
|
243
|
+
delta = replace(delta, table_depth=max(0, (delta.table_depth or 0) + signed))
|
|
244
|
+
elif opcode == 0x244B and operand:
|
|
245
|
+
delta = replace(delta, inner_cell=operand[0] != 0)
|
|
246
|
+
elif opcode == 0x244C and operand:
|
|
247
|
+
delta = replace(delta, inner_row=operand[0] != 0)
|
|
248
|
+
elif opcode == 0xD608:
|
|
249
|
+
table = _parse_tdef_table(operand)
|
|
250
|
+
if table is not None:
|
|
251
|
+
header = delta.table.header if delta.table is not None else False
|
|
252
|
+
delta = replace(delta, table=replace(table, header=header))
|
|
253
|
+
elif opcode == 0x3404 and operand:
|
|
254
|
+
table = delta.table or DocTableFormat()
|
|
255
|
+
delta = replace(delta, table=replace(table, header=operand[0] != 0))
|
|
256
|
+
elif opcode == 0xD62B and len(operand) >= 3 and delta.table is not None:
|
|
257
|
+
cell_index = operand[1]
|
|
258
|
+
flag = operand[2]
|
|
259
|
+
cells = list(delta.table.cells)
|
|
260
|
+
if cell_index < len(cells):
|
|
261
|
+
cells[cell_index] = replace(
|
|
262
|
+
cells[cell_index],
|
|
263
|
+
vertical_continue=flag == 1,
|
|
264
|
+
vertical_first=flag == 3,
|
|
265
|
+
)
|
|
266
|
+
delta = replace(delta, table=replace(delta.table, cells=tuple(cells)))
|
|
267
|
+
|
|
268
|
+
walk_sprms(grpprl, consume, budget=budget)
|
|
269
|
+
return delta
|