docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
"""读取 Word 97–2003 WordDocument stream 中的变长 FIB。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import struct
|
|
7
|
+
|
|
8
|
+
from ..errors import LegacyOfficeMalformedError
|
|
9
|
+
|
|
10
|
+
FIB_IDENT = 0xA5EC
|
|
11
|
+
MIN_WORD97_NFIB = 0x00C1
|
|
12
|
+
|
|
13
|
+
FCLCB_STSHF = 1
|
|
14
|
+
FCLCB_FOOTNOTE_REF = 2
|
|
15
|
+
FCLCB_FOOTNOTE_TEXT = 3
|
|
16
|
+
FCLCB_SECTION = 6
|
|
17
|
+
FCLCB_HEADER = 11
|
|
18
|
+
FCLCB_BTE_CHPX = 12
|
|
19
|
+
FCLCB_BTE_PAPX = 13
|
|
20
|
+
FCLCB_FIELD_MAIN = 16
|
|
21
|
+
FCLCB_FIELD_HEADER = 17
|
|
22
|
+
FCLCB_FIELD_FOOTNOTE = 18
|
|
23
|
+
FCLCB_BOOKMARK_NAMES = 21
|
|
24
|
+
FCLCB_BOOKMARK_START = 22
|
|
25
|
+
FCLCB_BOOKMARK_END = 23
|
|
26
|
+
FCLCB_DOP = 31
|
|
27
|
+
FCLCB_CLX = 33
|
|
28
|
+
FCLCB_SHAPE_MAIN = 40
|
|
29
|
+
FCLCB_SHAPE_HEADER = 41
|
|
30
|
+
FCLCB_ENDNOTE_REF = 46
|
|
31
|
+
FCLCB_ENDNOTE_TEXT = 47
|
|
32
|
+
FCLCB_FIELD_ENDNOTE = 48
|
|
33
|
+
FCLCB_DGG_INFO = 50
|
|
34
|
+
FCLCB_TEXTBOX_TEXT = 56
|
|
35
|
+
FCLCB_FIELD_TEXTBOX = 57
|
|
36
|
+
FCLCB_HEADER_TEXTBOX_TEXT = 58
|
|
37
|
+
FCLCB_FIELD_HEADER_TEXTBOX = 59
|
|
38
|
+
FCLCB_LISTS = 73
|
|
39
|
+
FCLCB_LIST_OVERRIDES = 74
|
|
40
|
+
FCLCB_TEXTBOX_BREAK = 75
|
|
41
|
+
FCLCB_HEADER_TEXTBOX_BREAK = 76
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True, slots=True)
|
|
45
|
+
class FcLcb:
|
|
46
|
+
"""FIB 中一对 stream 偏移和字节长度。"""
|
|
47
|
+
|
|
48
|
+
fc: int = 0
|
|
49
|
+
lcb: int = 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(frozen=True, slots=True)
|
|
53
|
+
class FibBase:
|
|
54
|
+
"""FIB 固定头中与解析相关的字段。"""
|
|
55
|
+
|
|
56
|
+
n_fib: int
|
|
57
|
+
lid: int
|
|
58
|
+
flags: int
|
|
59
|
+
fc_min: int
|
|
60
|
+
fc_mac: int
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def complex(self) -> bool:
|
|
64
|
+
"""返回文档是否使用 complex/fast-save piece table。"""
|
|
65
|
+
|
|
66
|
+
return bool(self.flags & 0x0004)
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def encrypted(self) -> bool:
|
|
70
|
+
"""返回文档是否设置加密标志。"""
|
|
71
|
+
|
|
72
|
+
return bool(self.flags & 0x0100)
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def uses_1table(self) -> bool:
|
|
76
|
+
"""返回 FIB 指定的首选 Table stream。"""
|
|
77
|
+
|
|
78
|
+
return bool(self.flags & 0x0200)
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def far_east(self) -> bool:
|
|
82
|
+
"""返回文档是否优先使用远东语言标识。"""
|
|
83
|
+
|
|
84
|
+
return bool(self.flags & 0x4000)
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def obfuscated(self) -> bool:
|
|
88
|
+
"""返回文档是否设置 XOR 混淆标志。"""
|
|
89
|
+
|
|
90
|
+
return bool(self.flags & 0x8000)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass(frozen=True, slots=True)
|
|
94
|
+
class FileInformationBlock:
|
|
95
|
+
"""完成边界校验的 Word 97+ FIB。"""
|
|
96
|
+
|
|
97
|
+
base: FibBase
|
|
98
|
+
rgw: tuple[int, ...]
|
|
99
|
+
rglw: tuple[int, ...]
|
|
100
|
+
pairs: tuple[FcLcb, ...]
|
|
101
|
+
csw_new: tuple[int, ...]
|
|
102
|
+
size: int
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def n_fib(self) -> int:
|
|
106
|
+
"""返回版本扩展中的有效 nFib。"""
|
|
107
|
+
|
|
108
|
+
return self.csw_new[0] if self.csw_new else self.base.n_fib
|
|
109
|
+
|
|
110
|
+
def pair(self, index: int) -> FcLcb:
|
|
111
|
+
"""读取可选 fc/lcb 对,不存在时返回零值。"""
|
|
112
|
+
|
|
113
|
+
return self.pairs[index] if 0 <= index < len(self.pairs) else FcLcb()
|
|
114
|
+
|
|
115
|
+
def story_count(self, index: int) -> int:
|
|
116
|
+
"""读取 FibRgLw97 中一个 story 的 UTF-16 CP 数。"""
|
|
117
|
+
|
|
118
|
+
return int(self.rglw[index]) if 0 <= index < len(self.rglw) else 0
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def ccp_text(self) -> int:
|
|
122
|
+
"""返回主文档 story 的 CP 数。"""
|
|
123
|
+
|
|
124
|
+
return self.story_count(3)
|
|
125
|
+
|
|
126
|
+
@property
|
|
127
|
+
def ccp_footnote(self) -> int:
|
|
128
|
+
"""返回脚注 story 的 CP 数。"""
|
|
129
|
+
|
|
130
|
+
return self.story_count(4)
|
|
131
|
+
|
|
132
|
+
@property
|
|
133
|
+
def ccp_header(self) -> int:
|
|
134
|
+
"""返回页眉页脚 story 的 CP 数。"""
|
|
135
|
+
|
|
136
|
+
return self.story_count(5)
|
|
137
|
+
|
|
138
|
+
@property
|
|
139
|
+
def ccp_macro(self) -> int:
|
|
140
|
+
"""返回宏 story 的 CP 数。"""
|
|
141
|
+
|
|
142
|
+
return self.story_count(6)
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def ccp_annotation(self) -> int:
|
|
146
|
+
"""返回批注 story 的 CP 数。"""
|
|
147
|
+
|
|
148
|
+
return self.story_count(7)
|
|
149
|
+
|
|
150
|
+
@property
|
|
151
|
+
def ccp_endnote(self) -> int:
|
|
152
|
+
"""返回尾注 story 的 CP 数。"""
|
|
153
|
+
|
|
154
|
+
return self.story_count(8)
|
|
155
|
+
|
|
156
|
+
@property
|
|
157
|
+
def ccp_textbox(self) -> int:
|
|
158
|
+
"""返回正文文本框 story 的 CP 数。"""
|
|
159
|
+
|
|
160
|
+
return self.story_count(9)
|
|
161
|
+
|
|
162
|
+
@property
|
|
163
|
+
def ccp_header_textbox(self) -> int:
|
|
164
|
+
"""返回页眉文本框 story 的 CP 数。"""
|
|
165
|
+
|
|
166
|
+
return self.story_count(10)
|
|
167
|
+
|
|
168
|
+
@property
|
|
169
|
+
def total_story_cp(self) -> int:
|
|
170
|
+
"""返回全部已知 story 的累计 CP 数。"""
|
|
171
|
+
|
|
172
|
+
return sum(self.story_count(index) for index in range(3, 11))
|
|
173
|
+
|
|
174
|
+
@property
|
|
175
|
+
def story_bases(self) -> dict[str, int]:
|
|
176
|
+
"""返回各 story 在全局 CP 空间中的起点。"""
|
|
177
|
+
|
|
178
|
+
counts = [self.story_count(index) for index in range(3, 11)]
|
|
179
|
+
names = [
|
|
180
|
+
"main",
|
|
181
|
+
"footnote",
|
|
182
|
+
"header",
|
|
183
|
+
"macro",
|
|
184
|
+
"annotation",
|
|
185
|
+
"endnote",
|
|
186
|
+
"textbox",
|
|
187
|
+
"header_textbox",
|
|
188
|
+
]
|
|
189
|
+
bases: dict[str, int] = {}
|
|
190
|
+
cursor = 0
|
|
191
|
+
for name, count in zip(names, counts, strict=True):
|
|
192
|
+
bases[name] = cursor
|
|
193
|
+
cursor += count
|
|
194
|
+
return bases
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _read_values(data: bytes, offset: int, count: int, width: int, label: str) -> tuple[tuple[int, ...], int]:
|
|
198
|
+
"""按指定宽度读取一组无符号小端整数。"""
|
|
199
|
+
|
|
200
|
+
if count < 0 or width not in {2, 4}:
|
|
201
|
+
raise LegacyOfficeMalformedError(f"invalid {label} count")
|
|
202
|
+
size = count * width
|
|
203
|
+
end = offset + size
|
|
204
|
+
if offset < 0 or end < offset or end > len(data):
|
|
205
|
+
raise LegacyOfficeMalformedError(f"truncated FIB {label}")
|
|
206
|
+
if count == 0:
|
|
207
|
+
return (), end
|
|
208
|
+
code = "H" if width == 2 else "I"
|
|
209
|
+
return tuple(int(value) for value in struct.unpack_from(f"<{count}{code}", data, offset)), end
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def parse_fib(word_document: bytes) -> FileInformationBlock:
|
|
213
|
+
"""按 MS-DOC 变长布局解析 FIB,并拒绝 Word 95 及更早版本。"""
|
|
214
|
+
|
|
215
|
+
if len(word_document) < 34:
|
|
216
|
+
raise LegacyOfficeMalformedError("WordDocument FIB is truncated")
|
|
217
|
+
ident, n_fib = struct.unpack_from("<HH", word_document, 0)
|
|
218
|
+
if ident != FIB_IDENT:
|
|
219
|
+
raise LegacyOfficeMalformedError("WordDocument FIB magic is invalid")
|
|
220
|
+
if n_fib < MIN_WORD97_NFIB:
|
|
221
|
+
raise LegacyOfficeMalformedError(f"Word 95 or earlier nFib is unsupported: 0x{n_fib:04X}")
|
|
222
|
+
lid = int(struct.unpack_from("<H", word_document, 6)[0])
|
|
223
|
+
flags = int(struct.unpack_from("<H", word_document, 10)[0])
|
|
224
|
+
fc_min = int(struct.unpack_from("<I", word_document, 24)[0])
|
|
225
|
+
fc_mac = int(struct.unpack_from("<I", word_document, 28)[0])
|
|
226
|
+
base = FibBase(n_fib=n_fib, lid=lid, flags=flags, fc_min=fc_min, fc_mac=fc_mac)
|
|
227
|
+
|
|
228
|
+
cursor = 32
|
|
229
|
+
csw = int(struct.unpack_from("<H", word_document, cursor)[0])
|
|
230
|
+
cursor += 2
|
|
231
|
+
rgw, cursor = _read_values(word_document, cursor, csw, 2, "FibRgW")
|
|
232
|
+
if cursor + 2 > len(word_document):
|
|
233
|
+
raise LegacyOfficeMalformedError("truncated FIB cslw")
|
|
234
|
+
cslw = int(struct.unpack_from("<H", word_document, cursor)[0])
|
|
235
|
+
cursor += 2
|
|
236
|
+
rglw, cursor = _read_values(word_document, cursor, cslw, 4, "FibRgLw")
|
|
237
|
+
if cursor + 2 > len(word_document):
|
|
238
|
+
raise LegacyOfficeMalformedError("truncated FIB cbRgFcLcb")
|
|
239
|
+
pair_count = int(struct.unpack_from("<H", word_document, cursor)[0])
|
|
240
|
+
cursor += 2
|
|
241
|
+
raw_pairs, cursor = _read_values(word_document, cursor, pair_count * 2, 4, "FibRgFcLcb")
|
|
242
|
+
pairs = tuple(FcLcb(raw_pairs[index], raw_pairs[index + 1]) for index in range(0, len(raw_pairs), 2))
|
|
243
|
+
|
|
244
|
+
csw_new: tuple[int, ...] = ()
|
|
245
|
+
if cursor + 2 <= len(word_document):
|
|
246
|
+
count = int(struct.unpack_from("<H", word_document, cursor)[0])
|
|
247
|
+
cursor += 2
|
|
248
|
+
csw_new, cursor = _read_values(word_document, cursor, count, 2, "FibRgCswNew")
|
|
249
|
+
if len(rglw) <= 3:
|
|
250
|
+
# 确定性最小 fixture 可能省略变长计数,但仍保留 Word 97 固定槽位;
|
|
251
|
+
# 仅在标准布局不可用时按这些公开槽位做恢复读取。
|
|
252
|
+
if len(word_document) < 0x6C:
|
|
253
|
+
raise LegacyOfficeMalformedError("FIB does not contain ccpText")
|
|
254
|
+
rglw = tuple(int(struct.unpack_from("<I", word_document, 0x40 + index * 4)[0]) for index in range(11))
|
|
255
|
+
if not pairs and len(word_document) >= 0x382:
|
|
256
|
+
pairs = tuple(FcLcb(*struct.unpack_from("<II", word_document, 0x9A + index * 8)) for index in range(93))
|
|
257
|
+
cursor = max(cursor, 0x382)
|
|
258
|
+
fib = FileInformationBlock(base=base, rgw=rgw, rglw=rglw, pairs=pairs, csw_new=csw_new, size=cursor)
|
|
259
|
+
if fib.total_story_cp < fib.ccp_text:
|
|
260
|
+
raise LegacyOfficeMalformedError("FIB story CP count overflow")
|
|
261
|
+
return fib
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""解析 DOC 字段指令并安全恢复超链接、目录和 caption 语义。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from loguru import logger
|
|
8
|
+
|
|
9
|
+
from ..._shared.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
|
|
10
|
+
from .models import DocTextRun
|
|
11
|
+
|
|
12
|
+
_TOKEN_RE = re.compile(r'"(?:\\.|[^"\\])*"|\\\S|\S+')
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def field_keyword(instruction: str) -> str:
|
|
16
|
+
"""返回字段指令的首个关键字大写形式。"""
|
|
17
|
+
|
|
18
|
+
tokens = _TOKEN_RE.findall(instruction.strip())
|
|
19
|
+
return tokens[0].strip('"').upper() if tokens else ""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_toc_field(instruction: str) -> bool:
|
|
23
|
+
"""判断字段是否为多段落 TOC。"""
|
|
24
|
+
|
|
25
|
+
return field_keyword(instruction) == "TOC"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def is_caption_field(instruction: str) -> bool:
|
|
29
|
+
"""判断字段是否为 Word SEQ caption 编号。"""
|
|
30
|
+
|
|
31
|
+
return field_keyword(instruction) == "SEQ"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def is_chart_embed_field(instruction: str) -> bool:
|
|
35
|
+
"""判断 EMBED 字段是否声明 Excel.Chart 或 MSGraph.Chart 对象。"""
|
|
36
|
+
|
|
37
|
+
tokens = _TOKEN_RE.findall(instruction.strip())
|
|
38
|
+
if len(tokens) < 2 or _unquote(tokens[0]).casefold() != "embed":
|
|
39
|
+
return False
|
|
40
|
+
prog_id = _unquote(tokens[1]).casefold()
|
|
41
|
+
return prog_id.startswith(("excel.chart", "msgraph.chart"))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _unquote(token: str) -> str:
|
|
45
|
+
"""解码字段引号内允许的反斜杠转义。"""
|
|
46
|
+
|
|
47
|
+
if len(token) >= 2 and token[0] == token[-1] == '"':
|
|
48
|
+
token = token[1:-1]
|
|
49
|
+
return token.replace(r"\"", '"').replace(r"\\", "\\")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def hyperlink_target(instruction: str) -> str | None:
|
|
53
|
+
"""从 HYPERLINK 字段读取 URL 与可选内部书签。"""
|
|
54
|
+
|
|
55
|
+
tokens = _TOKEN_RE.findall(instruction.strip())
|
|
56
|
+
if not tokens or _unquote(tokens[0]).casefold() != "hyperlink":
|
|
57
|
+
return None
|
|
58
|
+
url: str | None = None
|
|
59
|
+
anchor: str | None = None
|
|
60
|
+
index = 1
|
|
61
|
+
while index < len(tokens):
|
|
62
|
+
token = tokens[index]
|
|
63
|
+
if token.startswith("\\"):
|
|
64
|
+
switch = token[1:].casefold()
|
|
65
|
+
argument: str | None = None
|
|
66
|
+
if switch in {"l", "o", "t"} and index + 1 < len(tokens) and not tokens[index + 1].startswith("\\"):
|
|
67
|
+
index += 1
|
|
68
|
+
argument = _unquote(tokens[index]).strip()
|
|
69
|
+
if switch == "l" and argument:
|
|
70
|
+
anchor = argument
|
|
71
|
+
elif url is None:
|
|
72
|
+
url = _unquote(token).strip()
|
|
73
|
+
index += 1
|
|
74
|
+
if url and anchor:
|
|
75
|
+
candidate = f"{url}#{anchor}"
|
|
76
|
+
elif url:
|
|
77
|
+
candidate = url
|
|
78
|
+
elif anchor:
|
|
79
|
+
candidate = f"#{anchor}"
|
|
80
|
+
else:
|
|
81
|
+
return None
|
|
82
|
+
safe = sanitize_hyperlink_target(
|
|
83
|
+
candidate,
|
|
84
|
+
allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
|
|
85
|
+
allow_relative=True,
|
|
86
|
+
allow_fragment=True,
|
|
87
|
+
)
|
|
88
|
+
if safe is None:
|
|
89
|
+
logger.warning(f"DOC hyperlink target was rejected: {candidate!r}")
|
|
90
|
+
return safe
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def apply_field_result(instruction: str, runs: list[DocTextRun]) -> list[DocTextRun]:
|
|
94
|
+
"""把 HYPERLINK 目标绑定到字段结果,其他字段仅保留缓存结果。"""
|
|
95
|
+
|
|
96
|
+
if field_keyword(instruction) != "HYPERLINK":
|
|
97
|
+
return runs
|
|
98
|
+
target = hyperlink_target(instruction)
|
|
99
|
+
if target is None:
|
|
100
|
+
return runs
|
|
101
|
+
return [DocTextRun(run.text, run.style, target, run.formula) for run in runs]
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""解析 Word CHPX/PAPX FKP 页面并提供按 FC 查询的格式 run。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from bisect import bisect_right
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from loguru import logger
|
|
9
|
+
|
|
10
|
+
from ..legacy.binary import bounded_slice, get_u32
|
|
11
|
+
from .records import DocBudget
|
|
12
|
+
from .sprm import PapDelta, apply_paragraph_sprms
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True, slots=True)
|
|
16
|
+
class CharacterRun:
|
|
17
|
+
"""一个物理 FC 范围内的原始 CHPX。"""
|
|
18
|
+
|
|
19
|
+
fc_start: int
|
|
20
|
+
fc_end: int
|
|
21
|
+
grpprl: bytes
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True, slots=True)
|
|
25
|
+
class ParagraphRun:
|
|
26
|
+
"""一个物理 FC 范围内的段落样式和 PAPX。"""
|
|
27
|
+
|
|
28
|
+
fc_start: int
|
|
29
|
+
fc_end: int
|
|
30
|
+
style_id: int
|
|
31
|
+
delta: PapDelta
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class FormattingRuns:
|
|
35
|
+
"""按起始 FC 排序的 CHPX/PAPX 查询索引。"""
|
|
36
|
+
|
|
37
|
+
def __init__(self, characters: list[CharacterRun], paragraphs: list[ParagraphRun]) -> None:
|
|
38
|
+
"""排序 run 并缓存二分查询键。"""
|
|
39
|
+
|
|
40
|
+
self.characters = sorted(characters, key=lambda run: run.fc_start)
|
|
41
|
+
self.paragraphs = sorted(paragraphs, key=lambda run: run.fc_start)
|
|
42
|
+
self._character_starts = [run.fc_start for run in self.characters]
|
|
43
|
+
self._paragraph_starts = [run.fc_start for run in self.paragraphs]
|
|
44
|
+
|
|
45
|
+
def character_at(self, fc: int) -> CharacterRun | None:
|
|
46
|
+
"""返回覆盖指定 FC 的最后一个 CHPX run。"""
|
|
47
|
+
|
|
48
|
+
index = bisect_right(self._character_starts, fc) - 1
|
|
49
|
+
if index < 0:
|
|
50
|
+
return None
|
|
51
|
+
run = self.characters[index]
|
|
52
|
+
return run if fc < run.fc_end else None
|
|
53
|
+
|
|
54
|
+
def paragraph_at(self, fc: int) -> ParagraphRun | None:
|
|
55
|
+
"""返回覆盖指定 FC 的最后一个 PAPX run。"""
|
|
56
|
+
|
|
57
|
+
index = bisect_right(self._paragraph_starts, fc) - 1
|
|
58
|
+
if index < 0:
|
|
59
|
+
return None
|
|
60
|
+
run = self.paragraphs[index]
|
|
61
|
+
return run if fc < run.fc_end else None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _parse_bte_pages(table_stream: bytes, offset: int, size: int) -> list[int]:
|
|
65
|
+
"""从 PlcBteChpx/PlcBtePapx 读取 FKP page number。"""
|
|
66
|
+
|
|
67
|
+
plc = bounded_slice(table_stream, offset, size)
|
|
68
|
+
if plc is None or len(plc) < 8 or (len(plc) - 4) % 8:
|
|
69
|
+
return []
|
|
70
|
+
count = (len(plc) - 4) // 8
|
|
71
|
+
page_offset = (count + 1) * 4
|
|
72
|
+
pages: list[int] = []
|
|
73
|
+
for index in range(count):
|
|
74
|
+
raw = get_u32(plc, page_offset + index * 4)
|
|
75
|
+
if raw is not None:
|
|
76
|
+
pages.append(raw & 0x003F_FFFF)
|
|
77
|
+
return pages
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _parse_chpx_page(page: bytes, budget: DocBudget) -> list[CharacterRun]:
|
|
81
|
+
"""解析一个 512 字节 ChpxFkp。"""
|
|
82
|
+
|
|
83
|
+
count = page[511]
|
|
84
|
+
if count == 0 or (count + 1) * 4 + count > 511:
|
|
85
|
+
return []
|
|
86
|
+
result: list[CharacterRun] = []
|
|
87
|
+
offset_base = (count + 1) * 4
|
|
88
|
+
for index in range(count):
|
|
89
|
+
fc_start = get_u32(page, index * 4)
|
|
90
|
+
fc_end = get_u32(page, (index + 1) * 4)
|
|
91
|
+
if fc_start is None or fc_end is None or fc_end <= fc_start:
|
|
92
|
+
continue
|
|
93
|
+
byte_offset = page[offset_base + index]
|
|
94
|
+
grpprl = b""
|
|
95
|
+
if byte_offset:
|
|
96
|
+
payload_offset = byte_offset * 2
|
|
97
|
+
if payload_offset < 511:
|
|
98
|
+
length = page[payload_offset]
|
|
99
|
+
grpprl = page[payload_offset + 1 : payload_offset + 1 + length]
|
|
100
|
+
budget.charge()
|
|
101
|
+
result.append(CharacterRun(fc_start, fc_end, grpprl))
|
|
102
|
+
return result
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _parse_papx_page(page: bytes, data_stream: bytes, budget: DocBudget) -> list[ParagraphRun]:
|
|
106
|
+
"""解析一个 512 字节 PapxFkp。"""
|
|
107
|
+
|
|
108
|
+
count = page[511]
|
|
109
|
+
header_end = (count + 1) * 4 + count * 13
|
|
110
|
+
if count == 0 or header_end > 511:
|
|
111
|
+
return []
|
|
112
|
+
result: list[ParagraphRun] = []
|
|
113
|
+
bx_base = (count + 1) * 4
|
|
114
|
+
for index in range(count):
|
|
115
|
+
fc_start = get_u32(page, index * 4)
|
|
116
|
+
fc_end = get_u32(page, (index + 1) * 4)
|
|
117
|
+
if fc_start is None or fc_end is None or fc_end <= fc_start:
|
|
118
|
+
continue
|
|
119
|
+
byte_offset = page[bx_base + index * 13]
|
|
120
|
+
style_id = 0
|
|
121
|
+
delta = PapDelta()
|
|
122
|
+
if byte_offset:
|
|
123
|
+
payload_offset = byte_offset * 2
|
|
124
|
+
if payload_offset < 511:
|
|
125
|
+
first_length = page[payload_offset]
|
|
126
|
+
if first_length:
|
|
127
|
+
content_offset = payload_offset + 1
|
|
128
|
+
content_length = first_length * 2 - 1
|
|
129
|
+
elif payload_offset + 1 < 511:
|
|
130
|
+
content_offset = payload_offset + 2
|
|
131
|
+
content_length = page[payload_offset + 1] * 2
|
|
132
|
+
else:
|
|
133
|
+
content_offset = 511
|
|
134
|
+
content_length = 0
|
|
135
|
+
content = page[content_offset : min(content_offset + content_length, 511)]
|
|
136
|
+
if len(content) >= 2:
|
|
137
|
+
style_id = int.from_bytes(content[:2], "little")
|
|
138
|
+
delta = apply_paragraph_sprms(content[2:], data_stream, budget=budget)
|
|
139
|
+
budget.charge()
|
|
140
|
+
result.append(ParagraphRun(fc_start, fc_end, style_id, delta))
|
|
141
|
+
return result
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def parse_formatting_runs(
|
|
145
|
+
word_document: bytes,
|
|
146
|
+
table_stream: bytes,
|
|
147
|
+
data_stream: bytes,
|
|
148
|
+
*,
|
|
149
|
+
chpx_offset: int,
|
|
150
|
+
chpx_size: int,
|
|
151
|
+
papx_offset: int,
|
|
152
|
+
papx_size: int,
|
|
153
|
+
budget: DocBudget,
|
|
154
|
+
) -> FormattingRuns:
|
|
155
|
+
"""解析 FIB 指向的全部 CHPX/PAPX FKP;坏可选页仅告警跳过。"""
|
|
156
|
+
|
|
157
|
+
characters: list[CharacterRun] = []
|
|
158
|
+
paragraphs: list[ParagraphRun] = []
|
|
159
|
+
for page_number in _parse_bte_pages(table_stream, chpx_offset, chpx_size):
|
|
160
|
+
page = bounded_slice(word_document, page_number * 512, 512)
|
|
161
|
+
if page is None:
|
|
162
|
+
logger.warning(f"DOC ChpxFkp page is truncated: {page_number}")
|
|
163
|
+
continue
|
|
164
|
+
characters.extend(_parse_chpx_page(page, budget))
|
|
165
|
+
for page_number in _parse_bte_pages(table_stream, papx_offset, papx_size):
|
|
166
|
+
page = bounded_slice(word_document, page_number * 512, 512)
|
|
167
|
+
if page is None:
|
|
168
|
+
logger.warning(f"DOC PapxFkp page is truncated: {page_number}")
|
|
169
|
+
continue
|
|
170
|
+
paragraphs.extend(_parse_papx_page(page, data_stream, budget))
|
|
171
|
+
return FormattingRuns(characters, paragraphs)
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""从 DOC Data/PICF 与 Word OfficeArt drawing 中恢复图片。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
import hashlib
|
|
7
|
+
|
|
8
|
+
from loguru import logger
|
|
9
|
+
|
|
10
|
+
from ..errors import LegacyOfficeResourceLimitError
|
|
11
|
+
from ..legacy.binary import bounded_slice, get_u16, get_u32
|
|
12
|
+
from ..limits import MAX_ASSET_TOTAL_BYTES
|
|
13
|
+
from ..legacy.officeart import OfficeImagePayload, decode_bstore, extract_word_shapes, first_blip, record_at
|
|
14
|
+
from ..equation.image import OfficeImageEquationDecoder
|
|
15
|
+
|
|
16
|
+
from .models import DocImage, DocImagePayload
|
|
17
|
+
from .records import DocBudget, parse_plc
|
|
18
|
+
|
|
19
|
+
_PLACEABLE_WMF_MAGIC = b"\xd7\xcd\xc6\x9a"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(slots=True)
|
|
23
|
+
class ImageStore:
|
|
24
|
+
"""按内容去重并限制 DOC 图片累计字节数。"""
|
|
25
|
+
|
|
26
|
+
total: int = 0
|
|
27
|
+
cache: dict[tuple[bytes, tuple[int, int] | None], DocImagePayload] = field(default_factory=dict)
|
|
28
|
+
accounted_digests: set[bytes] = field(default_factory=set)
|
|
29
|
+
equation_cache: dict[bytes, str | None] = field(default_factory=dict)
|
|
30
|
+
equation_decoder: OfficeImageEquationDecoder = field(default_factory=OfficeImageEquationDecoder)
|
|
31
|
+
|
|
32
|
+
def add(self, payload: OfficeImagePayload) -> DocImagePayload:
|
|
33
|
+
"""计入一张唯一图片并返回内部载荷。"""
|
|
34
|
+
|
|
35
|
+
digest = hashlib.sha256(payload.data).digest()
|
|
36
|
+
cache_key = digest, payload.render_size_emu
|
|
37
|
+
cached = self.cache.get(cache_key)
|
|
38
|
+
if cached is not None:
|
|
39
|
+
return cached
|
|
40
|
+
is_new_payload = digest not in self.accounted_digests
|
|
41
|
+
if is_new_payload and self.total + len(payload.data) > MAX_ASSET_TOTAL_BYTES:
|
|
42
|
+
raise LegacyOfficeResourceLimitError(f"embedded assets exceed max_asset_total_bytes={MAX_ASSET_TOTAL_BYTES}")
|
|
43
|
+
if digest in self.equation_cache:
|
|
44
|
+
equation_latex = self.equation_cache[digest]
|
|
45
|
+
else:
|
|
46
|
+
equation_latex = self.equation_decoder.decode(
|
|
47
|
+
payload.data,
|
|
48
|
+
part_name=f"image.{payload.extension}",
|
|
49
|
+
content_type=payload.content_type,
|
|
50
|
+
)
|
|
51
|
+
converted = DocImagePayload(
|
|
52
|
+
payload.data,
|
|
53
|
+
payload.extension,
|
|
54
|
+
payload.content_type,
|
|
55
|
+
equation_latex,
|
|
56
|
+
render_size_emu=payload.render_size_emu,
|
|
57
|
+
)
|
|
58
|
+
if is_new_payload:
|
|
59
|
+
self.total += len(payload.data)
|
|
60
|
+
self.accounted_digests.add(digest)
|
|
61
|
+
self.equation_cache[digest] = equation_latex
|
|
62
|
+
self.cache[cache_key] = converted
|
|
63
|
+
return converted
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def inline_picture(
|
|
67
|
+
data_stream: bytes,
|
|
68
|
+
*,
|
|
69
|
+
offset: int,
|
|
70
|
+
store: ImageStore,
|
|
71
|
+
budget: DocBudget,
|
|
72
|
+
) -> DocImagePayload | None:
|
|
73
|
+
"""解析 sprmCPicLocation 指向的 PICFAndOfficeArtData。"""
|
|
74
|
+
|
|
75
|
+
total_length = get_u32(data_stream, offset)
|
|
76
|
+
header_length = get_u16(data_stream, offset + 4)
|
|
77
|
+
if total_length is None or header_length is None or total_length < header_length:
|
|
78
|
+
return None
|
|
79
|
+
picf = bounded_slice(data_stream, offset, total_length)
|
|
80
|
+
if picf is None:
|
|
81
|
+
logger.warning(f"DOC picture at Data offset {offset} is out of bounds")
|
|
82
|
+
return None
|
|
83
|
+
art = picf[min(header_length, len(picf)) :]
|
|
84
|
+
decoded = first_blip(art, charge=budget.charge)
|
|
85
|
+
if decoded is None:
|
|
86
|
+
# 少量旧文件把原始位图直接放在 PICF 尾部,按 magic 尽力保留。
|
|
87
|
+
signatures = (
|
|
88
|
+
(_PLACEABLE_WMF_MAGIC, "wmf", "image/wmf"),
|
|
89
|
+
(b"\x89PNG\r\n\x1a\n", "png", "image/png"),
|
|
90
|
+
(b"\xff\xd8\xff", "jpg", "image/jpeg"),
|
|
91
|
+
(b"GIF8", "gif", "image/gif"),
|
|
92
|
+
(b"BM", "bmp", "image/bmp"),
|
|
93
|
+
(b"II*\x00", "tiff", "image/tiff"),
|
|
94
|
+
(b"MM\x00*", "tiff", "image/tiff"),
|
|
95
|
+
)
|
|
96
|
+
for signature, extension, content_type in signatures:
|
|
97
|
+
position = art.find(signature)
|
|
98
|
+
if position >= 0:
|
|
99
|
+
decoded = OfficeImagePayload(art[position:], extension, content_type)
|
|
100
|
+
break
|
|
101
|
+
return store.add(decoded) if decoded is not None else None
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def floating_pictures(
|
|
105
|
+
table_stream: bytes,
|
|
106
|
+
*,
|
|
107
|
+
word_document: bytes,
|
|
108
|
+
shape_plc_offset: int,
|
|
109
|
+
shape_plc_size: int,
|
|
110
|
+
drawing_offset: int,
|
|
111
|
+
drawing_size: int,
|
|
112
|
+
store: ImageStore,
|
|
113
|
+
budget: DocBudget,
|
|
114
|
+
) -> list[DocImage]:
|
|
115
|
+
"""按 PlcfSpaMom anchor CP 将 floating shape 的 BStore 图片绑定到正文。"""
|
|
116
|
+
|
|
117
|
+
plc_payload = bounded_slice(table_stream, shape_plc_offset, shape_plc_size)
|
|
118
|
+
drawing = bounded_slice(table_stream, drawing_offset, drawing_size)
|
|
119
|
+
if plc_payload is None or drawing is None:
|
|
120
|
+
return []
|
|
121
|
+
cps, items = parse_plc(plc_payload, item_size=26, budget=budget)
|
|
122
|
+
if not items:
|
|
123
|
+
return []
|
|
124
|
+
assets = decode_bstore(
|
|
125
|
+
drawing,
|
|
126
|
+
charge=budget.charge,
|
|
127
|
+
delay_stream=word_document,
|
|
128
|
+
)
|
|
129
|
+
shapes = {}
|
|
130
|
+
first = record_at(drawing, 0, charge=budget.charge)
|
|
131
|
+
cursor = (8 + len(first.payload)) if first is not None else len(drawing)
|
|
132
|
+
while cursor < len(drawing):
|
|
133
|
+
# OfficeArtWordDrawing 在 DgContainer 前有一个 main/header 标签字节。
|
|
134
|
+
cursor += 1
|
|
135
|
+
container = record_at(drawing, cursor, charge=budget.charge)
|
|
136
|
+
if container is None:
|
|
137
|
+
break
|
|
138
|
+
for shape in extract_word_shapes(container.payload, charge=budget.charge):
|
|
139
|
+
if shape.shape_id is not None:
|
|
140
|
+
shapes.setdefault(shape.shape_id, shape)
|
|
141
|
+
cursor += 8 + len(container.payload)
|
|
142
|
+
anchors = [
|
|
143
|
+
(shape_id, cps[index])
|
|
144
|
+
for index, item in enumerate(items)
|
|
145
|
+
if index < len(cps) and (shape_id := get_u32(item, 0)) is not None
|
|
146
|
+
]
|
|
147
|
+
images: list[DocImage] = []
|
|
148
|
+
for shape_id, shape in sorted(shapes.items()):
|
|
149
|
+
if shape.hidden or shape.pib is None:
|
|
150
|
+
continue
|
|
151
|
+
payload = assets.get(shape.pib)
|
|
152
|
+
if payload is None:
|
|
153
|
+
continue
|
|
154
|
+
anchor_cp = next((cp for anchor_id, cp in anchors if anchor_id == shape_id), None)
|
|
155
|
+
if anchor_cp is None:
|
|
156
|
+
candidates = [(anchor_id, cp) for anchor_id, cp in anchors if anchor_id <= shape_id]
|
|
157
|
+
if candidates:
|
|
158
|
+
anchor_cp = max(candidates)[1]
|
|
159
|
+
if anchor_cp is None:
|
|
160
|
+
continue
|
|
161
|
+
images.append(DocImage(cp=anchor_cp, payload=store.add(payload)))
|
|
162
|
+
return images
|