docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,526 @@
|
|
|
1
|
+
"""DOCX 公式与图片资源处理;共享当前 Converter 的单文档状态。"""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import re
|
|
5
|
+
from typing import Any
|
|
6
|
+
from docx.oxml.xmlchemy import BaseOxmlElement
|
|
7
|
+
from docx.text.paragraph import Paragraph
|
|
8
|
+
from loguru import logger
|
|
9
|
+
from ..ooxml_chart import extract_chart_html_from_ooxml
|
|
10
|
+
from ..image import serialize_office_image
|
|
11
|
+
from ..equation.ooxml import is_mathtype_equation_prog_id
|
|
12
|
+
from ..equation.omml import oMath2Latex
|
|
13
|
+
from .....schema import BlockType
|
|
14
|
+
|
|
15
|
+
from .context import _DocxConstants
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class _DocxResources:
|
|
19
|
+
"""集中维护公式与图片资源,不自行创建文档或持有跨文档缓存。"""
|
|
20
|
+
|
|
21
|
+
def _decode_docx_ole_equation(
|
|
22
|
+
self,
|
|
23
|
+
ole_element: Any,
|
|
24
|
+
part: Any,
|
|
25
|
+
) -> str | None:
|
|
26
|
+
"""从当前 DOCX part 的内部 OLE relationship 解码公式对象。"""
|
|
27
|
+
|
|
28
|
+
prog_id = ole_element.get("ProgID") or ole_element.get("ProgId")
|
|
29
|
+
if not is_mathtype_equation_prog_id(prog_id):
|
|
30
|
+
return None
|
|
31
|
+
object_type = (ole_element.get("Type") or "Embed").strip().casefold()
|
|
32
|
+
if object_type != "embed":
|
|
33
|
+
return None
|
|
34
|
+
draw_aspect = (ole_element.get("DrawAspect") or "Content").strip().casefold()
|
|
35
|
+
if draw_aspect == "icon":
|
|
36
|
+
return None
|
|
37
|
+
|
|
38
|
+
relationship_id = ole_element.get("{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id")
|
|
39
|
+
relationships = getattr(part, "rels", None)
|
|
40
|
+
relationship = relationships.get(relationship_id) if relationships is not None else None
|
|
41
|
+
if relationship is None or getattr(relationship, "is_external", False):
|
|
42
|
+
return None
|
|
43
|
+
reltype = str(getattr(relationship, "reltype", ""))
|
|
44
|
+
if not reltype.rstrip("/").casefold().endswith("/oleobject"):
|
|
45
|
+
return None
|
|
46
|
+
try:
|
|
47
|
+
blob = relationship.target_part.blob
|
|
48
|
+
except (AttributeError, KeyError, ValueError):
|
|
49
|
+
blob = None
|
|
50
|
+
latex = self._ooxml_equation_decoder.decode(
|
|
51
|
+
blob,
|
|
52
|
+
prog_id=prog_id,
|
|
53
|
+
)
|
|
54
|
+
if latex is not None:
|
|
55
|
+
return latex
|
|
56
|
+
|
|
57
|
+
warning_key = (self._docx_part_key(part), str(relationship_id))
|
|
58
|
+
if warning_key not in self._mtef_warned_relations:
|
|
59
|
+
self._mtef_warned_relations.add(warning_key)
|
|
60
|
+
logger.warning(
|
|
61
|
+
"DOCX_MTEF_FALLBACK: part={!r}, relationship={!r} has an invalid or unsupported equation OLE object",
|
|
62
|
+
warning_key[0],
|
|
63
|
+
relationship_id,
|
|
64
|
+
)
|
|
65
|
+
return None
|
|
66
|
+
|
|
67
|
+
def _decode_docx_equationxml(
|
|
68
|
+
self,
|
|
69
|
+
shape_element: Any,
|
|
70
|
+
part: Any,
|
|
71
|
+
) -> str | None:
|
|
72
|
+
"""解码当前 VML shape 的 ``equationxml`` 并对失败告警去重。"""
|
|
73
|
+
|
|
74
|
+
vml_shape_tag = f"{{{_DocxConstants._BLIP_NAMESPACES['v']}}}shape"
|
|
75
|
+
if getattr(shape_element, "tag", None) != vml_shape_tag:
|
|
76
|
+
return None
|
|
77
|
+
equation_xml = shape_element.get("equationxml")
|
|
78
|
+
if equation_xml is None:
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
latex = self._equationxml_decoder.decode(equation_xml)
|
|
82
|
+
if latex is not None:
|
|
83
|
+
return latex
|
|
84
|
+
|
|
85
|
+
digest = hashlib.sha256(equation_xml.encode("utf-8", errors="replace")).hexdigest()[:16]
|
|
86
|
+
warning_key = (
|
|
87
|
+
self._docx_part_key(part),
|
|
88
|
+
str(shape_element.get("id") or ""),
|
|
89
|
+
digest,
|
|
90
|
+
)
|
|
91
|
+
if warning_key not in self._equationxml_warned_shapes:
|
|
92
|
+
self._equationxml_warned_shapes.add(warning_key)
|
|
93
|
+
logger.warning(
|
|
94
|
+
"DOCX_EQUATIONXML_FALLBACK: part={!r}, shape_id={!r}, payload_sha256={!r} is malformed or unsupported",
|
|
95
|
+
warning_key[0],
|
|
96
|
+
warning_key[1],
|
|
97
|
+
warning_key[2],
|
|
98
|
+
)
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
@staticmethod
|
|
102
|
+
def _docx_image_relationship_id(image: Any) -> str | None:
|
|
103
|
+
"""读取 DrawingML/VML 图片元素的内部 relationship id。"""
|
|
104
|
+
|
|
105
|
+
relationship_id = image.get("{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed")
|
|
106
|
+
if not relationship_id:
|
|
107
|
+
relationship_id = image.get("{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id")
|
|
108
|
+
return str(relationship_id) if relationship_id else None
|
|
109
|
+
|
|
110
|
+
@classmethod
|
|
111
|
+
def _docx_image_part(cls, image: Any, part: Any) -> Any | None:
|
|
112
|
+
"""通过当前 part 的内部关系解析图片 part。"""
|
|
113
|
+
|
|
114
|
+
relationship_id = cls._docx_image_relationship_id(image)
|
|
115
|
+
relationships = getattr(part, "rels", None)
|
|
116
|
+
if not relationship_id or relationships is None:
|
|
117
|
+
return None
|
|
118
|
+
relationship = relationships.get(relationship_id)
|
|
119
|
+
if relationship is None or getattr(relationship, "is_external", False):
|
|
120
|
+
return None
|
|
121
|
+
try:
|
|
122
|
+
return relationship.target_part
|
|
123
|
+
except (AttributeError, KeyError, ValueError):
|
|
124
|
+
return None
|
|
125
|
+
|
|
126
|
+
def _decode_docx_image_equation(
|
|
127
|
+
self,
|
|
128
|
+
image: Any,
|
|
129
|
+
part: Any,
|
|
130
|
+
) -> str | None:
|
|
131
|
+
"""从当前图片 part 的 WMF/GIF comment 解码 MTEF。"""
|
|
132
|
+
|
|
133
|
+
image_part = self._docx_image_part(image, part)
|
|
134
|
+
if image_part is None:
|
|
135
|
+
return None
|
|
136
|
+
try:
|
|
137
|
+
blob = image_part.blob
|
|
138
|
+
except (AttributeError, KeyError, ValueError):
|
|
139
|
+
return None
|
|
140
|
+
return self._image_equation_decoder.decode(
|
|
141
|
+
blob,
|
|
142
|
+
part_name=getattr(image_part, "partname", None),
|
|
143
|
+
content_type=getattr(image_part, "content_type", None),
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
@staticmethod
|
|
147
|
+
def _select_docx_compatibility_tokens(
|
|
148
|
+
token_groups: list[list[tuple[str, str]]],
|
|
149
|
+
) -> list[tuple[str, str]]:
|
|
150
|
+
"""按 OMML、Equation XML、OLE MTEF、图片 MTEF 选择兼容分支。"""
|
|
151
|
+
|
|
152
|
+
for token_kind in _DocxConstants._FORMULA_SOURCE_PRIORITY:
|
|
153
|
+
for tokens in token_groups:
|
|
154
|
+
if any(kind == token_kind for kind, _value in tokens):
|
|
155
|
+
return tokens
|
|
156
|
+
return token_groups[0] if token_groups else []
|
|
157
|
+
|
|
158
|
+
def _docx_formula_tokens(
|
|
159
|
+
self,
|
|
160
|
+
element: Any,
|
|
161
|
+
part: Any,
|
|
162
|
+
) -> list[tuple[str, str]]:
|
|
163
|
+
"""按文档顺序提取文本、OMML、Equation XML 和 MTEF。"""
|
|
164
|
+
|
|
165
|
+
tag_name = self._local_name(element)
|
|
166
|
+
if tag_name is None:
|
|
167
|
+
return []
|
|
168
|
+
tag = str(getattr(element, "tag", ""))
|
|
169
|
+
word_namespace = _DocxConstants._BLIP_NAMESPACES["w"]
|
|
170
|
+
|
|
171
|
+
if tag_name == "AlternateContent":
|
|
172
|
+
branch_tokens = [
|
|
173
|
+
self._docx_formula_tokens(child, part) for child in element if self._local_name(child) in {"Choice", "Fallback"}
|
|
174
|
+
]
|
|
175
|
+
return self._select_docx_compatibility_tokens(branch_tokens)
|
|
176
|
+
|
|
177
|
+
if tag_name == "object" and tag == f"{{{_DocxConstants._BLIP_NAMESPACES['w']}}}object":
|
|
178
|
+
child_tokens = [self._docx_formula_tokens(child, part) for child in element]
|
|
179
|
+
return self._select_docx_compatibility_tokens(child_tokens)
|
|
180
|
+
|
|
181
|
+
if tag_name == "txbxContent":
|
|
182
|
+
# 外层段落会单独遍历文本框内容,避免在此重复提取公式和文字。
|
|
183
|
+
return []
|
|
184
|
+
|
|
185
|
+
if tag_name == "oMath" and "officeDocument/2006/math" in tag:
|
|
186
|
+
try:
|
|
187
|
+
latex = str(oMath2Latex(element)).strip()
|
|
188
|
+
except Exception as exc:
|
|
189
|
+
logger.debug(f"Failed to convert DOCX OMML equation to LaTeX: {exc}")
|
|
190
|
+
return []
|
|
191
|
+
return [("omml", latex)] if latex else []
|
|
192
|
+
|
|
193
|
+
if tag_name == "shape" and tag == f"{{{_DocxConstants._BLIP_NAMESPACES['v']}}}shape":
|
|
194
|
+
latex = self._decode_docx_equationxml(element, part)
|
|
195
|
+
if latex is not None:
|
|
196
|
+
return [("equationxml", latex)]
|
|
197
|
+
|
|
198
|
+
if tag_name == "OLEObject":
|
|
199
|
+
latex = self._decode_docx_ole_equation(element, part)
|
|
200
|
+
return [("mtef", latex)] if latex else []
|
|
201
|
+
|
|
202
|
+
if (tag_name == "blip" and tag == "{http://schemas.openxmlformats.org/drawingml/2006/main}blip") or (
|
|
203
|
+
tag_name == "imagedata" and tag == f"{{{_DocxConstants._BLIP_NAMESPACES['v']}}}imagedata"
|
|
204
|
+
):
|
|
205
|
+
latex = self._decode_docx_image_equation(element, part)
|
|
206
|
+
return [("image_mtef", latex)] if latex else []
|
|
207
|
+
|
|
208
|
+
if tag_name == "t" and "officeDocument/2006/math" not in tag:
|
|
209
|
+
return [("text", element.text)] if isinstance(element.text, str) else []
|
|
210
|
+
if tag in {f"{{{word_namespace}}}tab", f"{{{word_namespace}}}ptab"}:
|
|
211
|
+
return [("text", "\t")]
|
|
212
|
+
if tag == f"{{{word_namespace}}}cr":
|
|
213
|
+
return [("text", "\n")]
|
|
214
|
+
if tag == f"{{{word_namespace}}}br":
|
|
215
|
+
break_type = element.get(f"{{{word_namespace}}}type")
|
|
216
|
+
return [("text", "\n")] if break_type in {None, "textWrapping"} else []
|
|
217
|
+
if tag == f"{{{word_namespace}}}noBreakHyphen":
|
|
218
|
+
return [("text", "-")]
|
|
219
|
+
|
|
220
|
+
tokens: list[tuple[str, str]] = []
|
|
221
|
+
for child in element:
|
|
222
|
+
tokens.extend(self._docx_formula_tokens(child, part))
|
|
223
|
+
return tokens
|
|
224
|
+
|
|
225
|
+
def _picture_is_equation_preview(self, image: Any, part: Any) -> bool:
|
|
226
|
+
"""判断图片是否属于同一容器中已恢复的公式预览。"""
|
|
227
|
+
|
|
228
|
+
if self._decode_docx_image_equation(image, part):
|
|
229
|
+
return True
|
|
230
|
+
|
|
231
|
+
for ancestor in image.iterancestors():
|
|
232
|
+
ancestor_name = self._local_name(ancestor)
|
|
233
|
+
if ancestor_name == "shape":
|
|
234
|
+
if self._decode_docx_equationxml(ancestor, part):
|
|
235
|
+
return True
|
|
236
|
+
continue
|
|
237
|
+
if ancestor_name in {"object", "AlternateContent"}:
|
|
238
|
+
tokens = self._docx_formula_tokens(ancestor, part)
|
|
239
|
+
if any(kind in _DocxConstants._FORMULA_TOKEN_KINDS for kind, _value in tokens):
|
|
240
|
+
return True
|
|
241
|
+
continue
|
|
242
|
+
return False
|
|
243
|
+
|
|
244
|
+
def _handle_pictures(
|
|
245
|
+
self,
|
|
246
|
+
picture_refs: Any,
|
|
247
|
+
*,
|
|
248
|
+
part: Any | None = None,
|
|
249
|
+
) -> None:
|
|
250
|
+
"""
|
|
251
|
+
处理图片。
|
|
252
|
+
|
|
253
|
+
Args:
|
|
254
|
+
picture_refs: 图片引用元素列表
|
|
255
|
+
|
|
256
|
+
Returns:
|
|
257
|
+
|
|
258
|
+
"""
|
|
259
|
+
|
|
260
|
+
source_part = part or self._require_document_part()
|
|
261
|
+
|
|
262
|
+
seen_rel_ids: set[str] = set()
|
|
263
|
+
# 遍历所有图片引用元素,支持 DrawingML blip 和 VML imagedata。
|
|
264
|
+
for image in picture_refs:
|
|
265
|
+
if self._picture_is_equation_preview(image, source_part):
|
|
266
|
+
continue
|
|
267
|
+
rel_id = self._docx_image_relationship_id(image)
|
|
268
|
+
if rel_id and rel_id in seen_rel_ids:
|
|
269
|
+
continue
|
|
270
|
+
if rel_id:
|
|
271
|
+
seen_rel_ids.add(rel_id)
|
|
272
|
+
image_part = self._docx_image_part(image, source_part)
|
|
273
|
+
if image_part is None:
|
|
274
|
+
logger.warning("Warning: image cannot be found")
|
|
275
|
+
continue
|
|
276
|
+
|
|
277
|
+
img_base64 = serialize_office_image(
|
|
278
|
+
image_part.blob,
|
|
279
|
+
part_name=getattr(image_part, "partname", None),
|
|
280
|
+
content_type=getattr(image_part, "content_type", None),
|
|
281
|
+
)
|
|
282
|
+
if img_base64 is None:
|
|
283
|
+
continue
|
|
284
|
+
|
|
285
|
+
image_block = {
|
|
286
|
+
"type": BlockType.IMAGE,
|
|
287
|
+
"image_base64": img_base64,
|
|
288
|
+
}
|
|
289
|
+
self.cur_page.append(image_block)
|
|
290
|
+
|
|
291
|
+
def _handle_drawingml(self, elements: list[BaseOxmlElement]):
|
|
292
|
+
"""
|
|
293
|
+
处理 DrawingML 元素,目前先处理 chart 元素。
|
|
294
|
+
|
|
295
|
+
Args:
|
|
296
|
+
elements: 包含 DrawingML 元素的列表
|
|
297
|
+
|
|
298
|
+
Returns:
|
|
299
|
+
|
|
300
|
+
"""
|
|
301
|
+
chart_rel_types = {
|
|
302
|
+
"http://schemas.openxmlformats.org/officeDocument/2006/relationships/chart",
|
|
303
|
+
"http://purl.oclc.org/ooxml/officeDocument/relationships/chart",
|
|
304
|
+
}
|
|
305
|
+
package_rel_types = {
|
|
306
|
+
"http://schemas.openxmlformats.org/officeDocument/2006/relationships/package",
|
|
307
|
+
"http://purl.oclc.org/ooxml/officeDocument/relationships/package",
|
|
308
|
+
}
|
|
309
|
+
rel_id_attr = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id"
|
|
310
|
+
for element in elements:
|
|
311
|
+
chart = element.find(".//c:chart", namespaces=_DocxConstants._BLIP_NAMESPACES)
|
|
312
|
+
if chart is None:
|
|
313
|
+
continue
|
|
314
|
+
|
|
315
|
+
chart_block = {
|
|
316
|
+
"type": BlockType.CHART,
|
|
317
|
+
"content": "",
|
|
318
|
+
}
|
|
319
|
+
self.cur_page.append(chart_block)
|
|
320
|
+
|
|
321
|
+
rel_id = chart.get(rel_id_attr)
|
|
322
|
+
if not rel_id:
|
|
323
|
+
continue
|
|
324
|
+
|
|
325
|
+
try:
|
|
326
|
+
chart_rel = self.docx_obj.part.rels[rel_id]
|
|
327
|
+
except KeyError:
|
|
328
|
+
continue
|
|
329
|
+
|
|
330
|
+
if chart_rel.reltype not in chart_rel_types:
|
|
331
|
+
continue
|
|
332
|
+
|
|
333
|
+
try:
|
|
334
|
+
chart_part = chart_rel.target_part
|
|
335
|
+
chart_xml = chart_part.blob
|
|
336
|
+
except Exception as e:
|
|
337
|
+
logger.warning(f"Warning: chart XML cannot be loaded: {e}")
|
|
338
|
+
continue
|
|
339
|
+
|
|
340
|
+
workbook_bytes = None
|
|
341
|
+
try:
|
|
342
|
+
for rel in chart_part.rels.values():
|
|
343
|
+
if rel.reltype in package_rel_types:
|
|
344
|
+
workbook_bytes = rel.target_part.blob
|
|
345
|
+
break
|
|
346
|
+
except Exception as e:
|
|
347
|
+
logger.warning(f"Warning: chart workbook cannot be loaded: {e}")
|
|
348
|
+
|
|
349
|
+
try:
|
|
350
|
+
chart_html = extract_chart_html_from_ooxml(chart_xml, workbook_bytes)
|
|
351
|
+
except Exception as e:
|
|
352
|
+
logger.warning(f"Warning: chart HTML cannot be extracted: {e}")
|
|
353
|
+
continue
|
|
354
|
+
if chart_html:
|
|
355
|
+
chart_block["content"] = chart_html
|
|
356
|
+
|
|
357
|
+
def _handle_textbox_content(
|
|
358
|
+
self,
|
|
359
|
+
textbox_elements: list,
|
|
360
|
+
):
|
|
361
|
+
"""
|
|
362
|
+
处理文本框内容并将其添加到文档结构。
|
|
363
|
+
"""
|
|
364
|
+
# 收集并组织段落
|
|
365
|
+
container_paragraphs = self._collect_textbox_paragraphs(textbox_elements)
|
|
366
|
+
|
|
367
|
+
# 处理所有段落
|
|
368
|
+
all_paragraphs = []
|
|
369
|
+
|
|
370
|
+
# 对每个容器内的段落进行排序,然后按容器顺序处理
|
|
371
|
+
for paragraphs in container_paragraphs.values():
|
|
372
|
+
# 按容器内的垂直位置进行排序
|
|
373
|
+
sorted_container_paragraphs = sorted(
|
|
374
|
+
paragraphs,
|
|
375
|
+
key=lambda x: (
|
|
376
|
+
x[1] is None,
|
|
377
|
+
x[1] if x[1] is not None else float("inf"),
|
|
378
|
+
),
|
|
379
|
+
)
|
|
380
|
+
|
|
381
|
+
# 将排序后的段落添加到待处理列表
|
|
382
|
+
all_paragraphs.extend(sorted_container_paragraphs)
|
|
383
|
+
|
|
384
|
+
# 跟踪已处理段落以避免重复(相同内容和位置)
|
|
385
|
+
processed_paragraphs = set()
|
|
386
|
+
|
|
387
|
+
# 处理所有段落
|
|
388
|
+
for p, position in all_paragraphs:
|
|
389
|
+
# 创建 Paragraph 对象以获取文本内容
|
|
390
|
+
paragraph = Paragraph(p, self.docx_obj)
|
|
391
|
+
text_content = self._get_paragraph_text(paragraph)
|
|
392
|
+
|
|
393
|
+
# 基于内容和位置创建唯一标识
|
|
394
|
+
paragraph_id = (text_content, position)
|
|
395
|
+
|
|
396
|
+
# 如果该段落(相同内容和位置)已处理,则跳过
|
|
397
|
+
if paragraph_id in processed_paragraphs:
|
|
398
|
+
logger.debug(f"Skipping duplicate paragraph: content='{text_content[:50]}...', position={position}")
|
|
399
|
+
continue
|
|
400
|
+
|
|
401
|
+
# 将该段落标记为已处理
|
|
402
|
+
processed_paragraphs.add(paragraph_id)
|
|
403
|
+
|
|
404
|
+
self._handle_text_elements(p)
|
|
405
|
+
return
|
|
406
|
+
|
|
407
|
+
def _collect_textbox_paragraphs(self, textbox_elements):
|
|
408
|
+
"""
|
|
409
|
+
从文本框元素中收集并组织段落。
|
|
410
|
+
"""
|
|
411
|
+
processed_paragraphs = []
|
|
412
|
+
container_paragraphs = {}
|
|
413
|
+
|
|
414
|
+
for element in textbox_elements:
|
|
415
|
+
element_id = id(element)
|
|
416
|
+
# 如果已处理相同元素,则跳过
|
|
417
|
+
if element_id in processed_paragraphs:
|
|
418
|
+
continue
|
|
419
|
+
|
|
420
|
+
tag_name = self._local_name(element)
|
|
421
|
+
if tag_name is None:
|
|
422
|
+
continue
|
|
423
|
+
processed_paragraphs.append(element_id)
|
|
424
|
+
|
|
425
|
+
# 处理直接找到的段落(VML 文本框)
|
|
426
|
+
if tag_name == "p":
|
|
427
|
+
# 查找包含该段落的文本框或形状元素
|
|
428
|
+
container_id = None
|
|
429
|
+
for ancestor in element.iterancestors():
|
|
430
|
+
if any(ns in ancestor.tag for ns in ["textbox", "shape", "txbx"]):
|
|
431
|
+
container_id = id(ancestor)
|
|
432
|
+
break
|
|
433
|
+
|
|
434
|
+
if container_id not in container_paragraphs:
|
|
435
|
+
container_paragraphs[container_id] = []
|
|
436
|
+
container_paragraphs[container_id].append((element, self._get_paragraph_position(element)))
|
|
437
|
+
|
|
438
|
+
# 处理 txbxContent 元素(Word DrawingML 文本框)
|
|
439
|
+
elif tag_name == "txbxContent":
|
|
440
|
+
paragraphs = element.findall(".//w:p", namespaces=element.nsmap)
|
|
441
|
+
container_id = id(element)
|
|
442
|
+
if container_id not in container_paragraphs:
|
|
443
|
+
container_paragraphs[container_id] = []
|
|
444
|
+
|
|
445
|
+
for p in paragraphs:
|
|
446
|
+
p_id = id(p)
|
|
447
|
+
if p_id not in processed_paragraphs:
|
|
448
|
+
processed_paragraphs.append(p_id)
|
|
449
|
+
container_paragraphs[container_id].append((p, self._get_paragraph_position(p)))
|
|
450
|
+
else:
|
|
451
|
+
# 尝试从未知元素中提取任何段落
|
|
452
|
+
paragraphs = element.findall(".//w:p", namespaces=element.nsmap)
|
|
453
|
+
container_id = id(element)
|
|
454
|
+
if container_id not in container_paragraphs:
|
|
455
|
+
container_paragraphs[container_id] = []
|
|
456
|
+
|
|
457
|
+
for p in paragraphs:
|
|
458
|
+
p_id = id(p)
|
|
459
|
+
if p_id not in processed_paragraphs:
|
|
460
|
+
processed_paragraphs.append(p_id)
|
|
461
|
+
container_paragraphs[container_id].append((p, self._get_paragraph_position(p)))
|
|
462
|
+
|
|
463
|
+
return container_paragraphs
|
|
464
|
+
|
|
465
|
+
def _get_paragraph_position(self, paragraph_element):
|
|
466
|
+
"""
|
|
467
|
+
从段落元素提取垂直位置信息。
|
|
468
|
+
"""
|
|
469
|
+
# 先尝试直接从包含顺序相关属性的 w:p 元素获取索引
|
|
470
|
+
if hasattr(paragraph_element, "getparent") and paragraph_element.getparent() is not None:
|
|
471
|
+
parent = paragraph_element.getparent()
|
|
472
|
+
# 获取所有段落兄弟节点
|
|
473
|
+
paragraphs = [p for p in parent.getchildren() if self._local_name(p) == "p"]
|
|
474
|
+
# 查找当前段落在其兄弟节点中的索引
|
|
475
|
+
try:
|
|
476
|
+
paragraph_index = paragraphs.index(paragraph_element)
|
|
477
|
+
return paragraph_index # 使用索引作为位置以保证一致的排序
|
|
478
|
+
except ValueError:
|
|
479
|
+
pass
|
|
480
|
+
|
|
481
|
+
# 在元素及其祖先中查找位置提示属性
|
|
482
|
+
for elem in (*[paragraph_element], *paragraph_element.iterancestors()):
|
|
483
|
+
# 检查直接的位置信息属性
|
|
484
|
+
for attr_name in ["y", "top", "positionY", "y-position", "position"]:
|
|
485
|
+
value = elem.get(attr_name)
|
|
486
|
+
if value:
|
|
487
|
+
try:
|
|
488
|
+
# 移除任何非数字字符(如 'pt', 'px' 等)
|
|
489
|
+
clean_value = re.sub(r"[^0-9.]", "", value)
|
|
490
|
+
if clean_value:
|
|
491
|
+
return float(clean_value)
|
|
492
|
+
except (ValueError, TypeError):
|
|
493
|
+
pass
|
|
494
|
+
|
|
495
|
+
# 检查 transform 属性中的位移信息
|
|
496
|
+
transform = elem.get("transform")
|
|
497
|
+
if transform:
|
|
498
|
+
# 从 transform 矩阵中提取 translate 的第二个参数
|
|
499
|
+
match = re.search(r"translate\([^,]+,\s*([0-9.]+)", transform)
|
|
500
|
+
if match:
|
|
501
|
+
try:
|
|
502
|
+
return float(match.group(1))
|
|
503
|
+
except ValueError:
|
|
504
|
+
pass
|
|
505
|
+
|
|
506
|
+
# 检查 Word 格式中的锚点或相对位置指示器
|
|
507
|
+
# 'dist' 类属性可以表示相对位置
|
|
508
|
+
for attr_name in ["distT", "distB", "anchor", "relativeFrom"]:
|
|
509
|
+
if elem.get(attr_name) is not None:
|
|
510
|
+
return elem.sourceline # 使用 XML 源行号作为回退
|
|
511
|
+
|
|
512
|
+
# 针对 VML 形状,查找特定属性
|
|
513
|
+
for ns_uri in paragraph_element.nsmap.values():
|
|
514
|
+
if "vml" in ns_uri:
|
|
515
|
+
# 尝试从 style 属性提取 top 值
|
|
516
|
+
style = paragraph_element.get("style")
|
|
517
|
+
if style:
|
|
518
|
+
match = re.search(r"top:([0-9.]+)pt", style)
|
|
519
|
+
if match:
|
|
520
|
+
try:
|
|
521
|
+
return float(match.group(1))
|
|
522
|
+
except ValueError:
|
|
523
|
+
pass
|
|
524
|
+
|
|
525
|
+
# 如果没有更好的位置指示,则使用 XML 源行号作为顺序的代理
|
|
526
|
+
return paragraph_element.sourceline if hasattr(paragraph_element, "sourceline") else None
|