docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,693 @@
|
|
|
1
|
+
"""DOCX 富文本与样式处理;共享当前 Converter 的单文档状态。"""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any, Iterator, Optional, Union
|
|
5
|
+
from docx.enum.style import WD_STYLE_TYPE
|
|
6
|
+
from docx.oxml.xmlchemy import BaseOxmlElement
|
|
7
|
+
from docx.text.hyperlink import Hyperlink
|
|
8
|
+
from docx.text.paragraph import Paragraph
|
|
9
|
+
from docx.text.run import Run
|
|
10
|
+
from pydantic import AnyUrl
|
|
11
|
+
from .formatting_types import Formatting, Script
|
|
12
|
+
from .....content.spans import (
|
|
13
|
+
append_equation_span,
|
|
14
|
+
extend_inline_spans,
|
|
15
|
+
inline_span_plain_text,
|
|
16
|
+
slice_span_dicts,
|
|
17
|
+
strip_span_dicts,
|
|
18
|
+
)
|
|
19
|
+
from ..rich_text import (
|
|
20
|
+
append_rich_text_element,
|
|
21
|
+
build_spans_from_elements,
|
|
22
|
+
formatting_to_style_str,
|
|
23
|
+
has_non_visible_text_style,
|
|
24
|
+
has_visible_style,
|
|
25
|
+
normalize_format_for_text,
|
|
26
|
+
should_keep_group_text,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
from .context import _DocxConstants, _ParagraphElement
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class _DocxStyles:
|
|
33
|
+
"""集中维护富文本与样式,不自行创建文档或持有跨文档缓存。"""
|
|
34
|
+
|
|
35
|
+
def _get_style_id_from_property(
|
|
36
|
+
self,
|
|
37
|
+
xml_element: Optional[BaseOxmlElement],
|
|
38
|
+
property_tag: str,
|
|
39
|
+
style_tag: str,
|
|
40
|
+
) -> Optional[str]:
|
|
41
|
+
"""从段落或 run 的直接属性节点读取样式 ID,避免触发 python-docx 样式查找。"""
|
|
42
|
+
if xml_element is None:
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
property_element = xml_element.find(
|
|
46
|
+
property_tag,
|
|
47
|
+
namespaces=_DocxConstants._BLIP_NAMESPACES,
|
|
48
|
+
)
|
|
49
|
+
if property_element is None:
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
style_element = property_element.find(
|
|
53
|
+
style_tag,
|
|
54
|
+
namespaces=_DocxConstants._BLIP_NAMESPACES,
|
|
55
|
+
)
|
|
56
|
+
if style_element is None:
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
return style_element.get(self.XML_KEY) or None
|
|
60
|
+
|
|
61
|
+
def _get_cached_docx_style(
|
|
62
|
+
self,
|
|
63
|
+
part: Any,
|
|
64
|
+
style_id: Optional[str],
|
|
65
|
+
style_type: Any,
|
|
66
|
+
) -> Any:
|
|
67
|
+
"""按 style id 和类型缓存 python-docx 样式对象,避免大 styles.xml 被反复线性扫描。"""
|
|
68
|
+
if part is None:
|
|
69
|
+
return None
|
|
70
|
+
|
|
71
|
+
cache_key = (style_type, style_id)
|
|
72
|
+
if cache_key not in self._style_lookup_cache:
|
|
73
|
+
self._style_lookup_cache[cache_key] = part.get_style(
|
|
74
|
+
style_id,
|
|
75
|
+
style_type,
|
|
76
|
+
)
|
|
77
|
+
return self._style_lookup_cache[cache_key]
|
|
78
|
+
|
|
79
|
+
def _get_paragraph_style(self, paragraph: Optional[Paragraph]) -> Any:
|
|
80
|
+
"""读取段落样式;无显式 pStyle 时缓存默认段落样式查询结果。"""
|
|
81
|
+
if paragraph is None:
|
|
82
|
+
return None
|
|
83
|
+
style_id = self._get_style_id_from_property(
|
|
84
|
+
paragraph._element,
|
|
85
|
+
"w:pPr",
|
|
86
|
+
"w:pStyle",
|
|
87
|
+
)
|
|
88
|
+
return self._get_cached_docx_style(
|
|
89
|
+
paragraph.part,
|
|
90
|
+
style_id,
|
|
91
|
+
WD_STYLE_TYPE.PARAGRAPH,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
def _get_run_style(self, run: Optional[Run]) -> Any:
|
|
95
|
+
"""读取 run 字符样式;无显式 rStyle 时缓存默认字符样式查询结果。"""
|
|
96
|
+
if run is None:
|
|
97
|
+
return None
|
|
98
|
+
style_id = self._get_style_id_from_property(
|
|
99
|
+
run._element,
|
|
100
|
+
"w:rPr",
|
|
101
|
+
"w:rStyle",
|
|
102
|
+
)
|
|
103
|
+
return self._get_cached_docx_style(
|
|
104
|
+
run.part,
|
|
105
|
+
style_id,
|
|
106
|
+
WD_STYLE_TYPE.CHARACTER,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
@staticmethod
|
|
110
|
+
def _escape_hyperlink_text(text: str) -> str:
|
|
111
|
+
"""
|
|
112
|
+
转义超链接文本中的方括号。
|
|
113
|
+
|
|
114
|
+
Args:
|
|
115
|
+
text: 要转义的文本
|
|
116
|
+
|
|
117
|
+
Returns:
|
|
118
|
+
str: 转义后的文本
|
|
119
|
+
"""
|
|
120
|
+
if not text:
|
|
121
|
+
return text
|
|
122
|
+
# 转义方括号
|
|
123
|
+
text = text.replace("[", "\\[").replace("]", "\\]")
|
|
124
|
+
return text
|
|
125
|
+
|
|
126
|
+
@staticmethod
|
|
127
|
+
def _escape_hyperlink_url(url: str) -> str:
|
|
128
|
+
"""
|
|
129
|
+
转义超链接 URL 中的括号。
|
|
130
|
+
|
|
131
|
+
Args:
|
|
132
|
+
url: 要转义的 URL
|
|
133
|
+
|
|
134
|
+
Returns:
|
|
135
|
+
str: 转义后的 URL
|
|
136
|
+
"""
|
|
137
|
+
if not url:
|
|
138
|
+
return url
|
|
139
|
+
# 对括号进行 URL 编码
|
|
140
|
+
url = url.replace("(", "%28").replace(")", "%29")
|
|
141
|
+
return url
|
|
142
|
+
|
|
143
|
+
@staticmethod
|
|
144
|
+
def _get_style_str_from_format(format_obj) -> Optional[str]:
|
|
145
|
+
"""
|
|
146
|
+
从 Formatting 对象提取样式字符串。
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
format_obj: Formatting 对象
|
|
150
|
+
|
|
151
|
+
Returns:
|
|
152
|
+
Optional[str]: 样式字符串(如 "bold,italic"),无样式时返回 None
|
|
153
|
+
"""
|
|
154
|
+
return formatting_to_style_str(format_obj)
|
|
155
|
+
|
|
156
|
+
@staticmethod
|
|
157
|
+
def _has_visible_style(format_obj) -> bool:
|
|
158
|
+
"""
|
|
159
|
+
检查格式是否包含可见样式(下划线或删除线)。
|
|
160
|
+
|
|
161
|
+
空白文本在有这些样式时仍然是可见的,应当保留。
|
|
162
|
+
|
|
163
|
+
Args:
|
|
164
|
+
format_obj: Formatting 对象
|
|
165
|
+
|
|
166
|
+
Returns:
|
|
167
|
+
bool: 是否包含可见样式
|
|
168
|
+
"""
|
|
169
|
+
return has_visible_style(format_obj)
|
|
170
|
+
|
|
171
|
+
@staticmethod
|
|
172
|
+
def _has_non_visible_text_style(format_obj) -> bool:
|
|
173
|
+
"""判断格式是否只有空白文本不可见的字形样式。"""
|
|
174
|
+
return has_non_visible_text_style(format_obj)
|
|
175
|
+
|
|
176
|
+
@classmethod
|
|
177
|
+
def _normalize_format_for_text(
|
|
178
|
+
cls,
|
|
179
|
+
format_obj: Optional[Formatting],
|
|
180
|
+
text: str,
|
|
181
|
+
*,
|
|
182
|
+
preserve_blank_non_visible_style: bool = False,
|
|
183
|
+
) -> Optional[Formatting]:
|
|
184
|
+
"""按文本内容收敛 run 格式,避免空白 run 把不可见样式传给输出。
|
|
185
|
+
|
|
186
|
+
preserve_blank_non_visible_style 用于保留同一文本片段内空白 run 的
|
|
187
|
+
bold/italic:这些样式自身不让空格可见,但可能是连续同样式文本的一部分。
|
|
188
|
+
"""
|
|
189
|
+
return normalize_format_for_text(
|
|
190
|
+
format_obj,
|
|
191
|
+
text,
|
|
192
|
+
preserve_blank_non_visible_style=preserve_blank_non_visible_style,
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
def _find_adjacent_non_blank_run_format(
|
|
196
|
+
self,
|
|
197
|
+
inline_contents: list[Any],
|
|
198
|
+
current_index: int,
|
|
199
|
+
step: int,
|
|
200
|
+
) -> Optional[Formatting]:
|
|
201
|
+
"""查找相邻方向上最近的非空白普通 run 格式,用于判断空白 run 是否属于同一段样式文本。"""
|
|
202
|
+
index = current_index + step
|
|
203
|
+
while 0 <= index < len(inline_contents):
|
|
204
|
+
content = inline_contents[index]
|
|
205
|
+
# 超链接是独立输出边界,不跨越超链接借用样式上下文。
|
|
206
|
+
if isinstance(content, Hyperlink):
|
|
207
|
+
return None
|
|
208
|
+
if not isinstance(content, Run):
|
|
209
|
+
index += step
|
|
210
|
+
continue
|
|
211
|
+
if self._is_hidden_run(content):
|
|
212
|
+
index += step
|
|
213
|
+
continue
|
|
214
|
+
text = content.text or ""
|
|
215
|
+
if text.strip():
|
|
216
|
+
return self._get_format_from_run(content)
|
|
217
|
+
index += step
|
|
218
|
+
return None
|
|
219
|
+
|
|
220
|
+
def _should_preserve_blank_non_visible_style(
|
|
221
|
+
self,
|
|
222
|
+
inline_contents: list[Any],
|
|
223
|
+
current_index: int,
|
|
224
|
+
text: str,
|
|
225
|
+
format_obj: Optional[Formatting],
|
|
226
|
+
) -> bool:
|
|
227
|
+
"""判断空白 run 的 bold/italic 是否应保留,以便连续同样式文本合并成一个 span。"""
|
|
228
|
+
if not text or text.strip():
|
|
229
|
+
return False
|
|
230
|
+
if not self._has_non_visible_text_style(format_obj):
|
|
231
|
+
return False
|
|
232
|
+
|
|
233
|
+
previous_format = self._find_adjacent_non_blank_run_format(
|
|
234
|
+
inline_contents,
|
|
235
|
+
current_index,
|
|
236
|
+
-1,
|
|
237
|
+
)
|
|
238
|
+
if format_obj == previous_format:
|
|
239
|
+
return True
|
|
240
|
+
|
|
241
|
+
next_format = self._find_adjacent_non_blank_run_format(
|
|
242
|
+
inline_contents,
|
|
243
|
+
current_index,
|
|
244
|
+
1,
|
|
245
|
+
)
|
|
246
|
+
return format_obj == next_format
|
|
247
|
+
|
|
248
|
+
@classmethod
|
|
249
|
+
def _should_keep_group_text(
|
|
250
|
+
cls,
|
|
251
|
+
text: str,
|
|
252
|
+
format_obj: Optional[Formatting],
|
|
253
|
+
*,
|
|
254
|
+
preserve_plain_blank: bool = False,
|
|
255
|
+
) -> bool:
|
|
256
|
+
"""判断当前累积 run 是否需要输出,保留夹在可见样式之间的普通空白。"""
|
|
257
|
+
return should_keep_group_text(
|
|
258
|
+
text,
|
|
259
|
+
format_obj,
|
|
260
|
+
preserve_plain_blank=preserve_plain_blank,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
@staticmethod
|
|
264
|
+
def _append_paragraph_element(
|
|
265
|
+
paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
|
|
266
|
+
text: str,
|
|
267
|
+
format_obj: Optional[Formatting],
|
|
268
|
+
hyperlink: Optional[Union[AnyUrl, Path, str]],
|
|
269
|
+
) -> None:
|
|
270
|
+
"""追加段落元素;相邻同超链接且同格式的 run 合并为一个元素。"""
|
|
271
|
+
append_rich_text_element(paragraph_elements, text, format_obj, hyperlink)
|
|
272
|
+
|
|
273
|
+
@staticmethod
|
|
274
|
+
def _normalize_hyperlink_group_boundaries(
|
|
275
|
+
paragraph_elements: list[_ParagraphElement],
|
|
276
|
+
) -> list[_ParagraphElement]:
|
|
277
|
+
"""把链接组边界空白移为普通文本,仅裁剪整段首尾并保留组内空白。"""
|
|
278
|
+
|
|
279
|
+
output: list[_ParagraphElement] = []
|
|
280
|
+
index = 0
|
|
281
|
+
while index < len(paragraph_elements):
|
|
282
|
+
element = paragraph_elements[index]
|
|
283
|
+
hyperlink = element[2]
|
|
284
|
+
if hyperlink is None:
|
|
285
|
+
output.append(element)
|
|
286
|
+
index += 1
|
|
287
|
+
continue
|
|
288
|
+
group_end = index + 1
|
|
289
|
+
while (
|
|
290
|
+
group_end < len(paragraph_elements)
|
|
291
|
+
and paragraph_elements[group_end][2] is not None
|
|
292
|
+
and str(paragraph_elements[group_end][2]) == str(hyperlink)
|
|
293
|
+
):
|
|
294
|
+
group_end += 1
|
|
295
|
+
group = list(paragraph_elements[index:group_end])
|
|
296
|
+
first_text, first_format, first_hyperlink = group[0]
|
|
297
|
+
last_text, last_format, last_hyperlink = group[-1]
|
|
298
|
+
if len(group) == 1 and not first_text.strip():
|
|
299
|
+
leading_space = ""
|
|
300
|
+
trailing_space = first_text
|
|
301
|
+
group[0] = ("", first_format, first_hyperlink)
|
|
302
|
+
else:
|
|
303
|
+
leading_length = len(first_text) - len(first_text.lstrip())
|
|
304
|
+
trailing_length = len(last_text) - len(last_text.rstrip())
|
|
305
|
+
leading_space = first_text[:leading_length]
|
|
306
|
+
trailing_space = last_text[len(last_text) - trailing_length :] if trailing_length else ""
|
|
307
|
+
if len(group) == 1:
|
|
308
|
+
group[0] = (
|
|
309
|
+
first_text[leading_length : len(first_text) - trailing_length if trailing_length else len(first_text)],
|
|
310
|
+
first_format,
|
|
311
|
+
first_hyperlink,
|
|
312
|
+
)
|
|
313
|
+
else:
|
|
314
|
+
group[0] = (
|
|
315
|
+
first_text[leading_length:],
|
|
316
|
+
first_format,
|
|
317
|
+
first_hyperlink,
|
|
318
|
+
)
|
|
319
|
+
group[-1] = (
|
|
320
|
+
last_text[: len(last_text) - trailing_length] if trailing_length else last_text,
|
|
321
|
+
last_format,
|
|
322
|
+
last_hyperlink,
|
|
323
|
+
)
|
|
324
|
+
if leading_space and output:
|
|
325
|
+
output.append((leading_space, first_format, None))
|
|
326
|
+
output.extend(item for item in group if item[0])
|
|
327
|
+
if trailing_space and group_end < len(paragraph_elements):
|
|
328
|
+
output.append((trailing_space, last_format, None))
|
|
329
|
+
index = group_end
|
|
330
|
+
return output
|
|
331
|
+
|
|
332
|
+
@staticmethod
|
|
333
|
+
def _is_hidden_run(run: Run) -> bool:
|
|
334
|
+
"""Check whether a run is marked as hidden text in Word."""
|
|
335
|
+
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
336
|
+
rpr = run._element.find(f"{{{_W}}}rPr")
|
|
337
|
+
if rpr is None:
|
|
338
|
+
return False
|
|
339
|
+
# webHidden: commonly used by TOC page-number field runs
|
|
340
|
+
if rpr.find(f"{{{_W}}}webHidden") is not None:
|
|
341
|
+
return True
|
|
342
|
+
# vanish: generic hidden text
|
|
343
|
+
if rpr.find(f"{{{_W}}}vanish") is not None:
|
|
344
|
+
return True
|
|
345
|
+
return False
|
|
346
|
+
|
|
347
|
+
@classmethod
|
|
348
|
+
def _build_spans_from_elements(
|
|
349
|
+
cls,
|
|
350
|
+
paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
|
|
351
|
+
) -> list[dict[str, Any]]:
|
|
352
|
+
"""按连续同 URL hyperlink 分组,直接生成结构化 Span。"""
|
|
353
|
+
return build_spans_from_elements(paragraph_elements)
|
|
354
|
+
|
|
355
|
+
def _build_text_from_elements(
|
|
356
|
+
self,
|
|
357
|
+
paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
|
|
358
|
+
) -> list[dict[str, Any]]:
|
|
359
|
+
"""
|
|
360
|
+
从 paragraph_elements 重组文本,应用超链接格式和字体样式。
|
|
361
|
+
|
|
362
|
+
Args:
|
|
363
|
+
paragraph_elements: 段落元素列表
|
|
364
|
+
|
|
365
|
+
Returns:
|
|
366
|
+
list[dict]: 重组后的 Span
|
|
367
|
+
"""
|
|
368
|
+
return self._build_spans_from_elements(paragraph_elements)
|
|
369
|
+
|
|
370
|
+
@staticmethod
|
|
371
|
+
def _normalize_text_block_content(content: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
372
|
+
"""
|
|
373
|
+
规范化普通文本块导出内容。
|
|
374
|
+
|
|
375
|
+
DOCX 常用段首/段尾空格模拟版式对齐,导出普通文本块前去除这些前后空白。
|
|
376
|
+
"""
|
|
377
|
+
return strip_span_dicts(content)
|
|
378
|
+
|
|
379
|
+
def _build_text_with_equations_and_hyperlinks(
|
|
380
|
+
self,
|
|
381
|
+
paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
|
|
382
|
+
text_with_equations: str,
|
|
383
|
+
equations: list[tuple[str, str]],
|
|
384
|
+
) -> list[dict[str, Any]]:
|
|
385
|
+
"""
|
|
386
|
+
构建同时包含公式、超链接和字体样式的文本。
|
|
387
|
+
|
|
388
|
+
Args:
|
|
389
|
+
paragraph_elements: 段落元素列表,包含格式和超链接信息
|
|
390
|
+
text_with_equations: 不含公式的原始可见文本
|
|
391
|
+
equations: 按源顺序排列的 text/equation token
|
|
392
|
+
|
|
393
|
+
Returns:
|
|
394
|
+
list[dict]: 包含公式、超链接和字体样式的 Span
|
|
395
|
+
"""
|
|
396
|
+
if not equations:
|
|
397
|
+
return self._build_text_from_elements(paragraph_elements)
|
|
398
|
+
styled_spans = self._build_text_from_elements(paragraph_elements)
|
|
399
|
+
styled_visible = inline_span_plain_text(styled_spans)
|
|
400
|
+
plain_visible = "".join(value for kind, value in equations if kind == "text")
|
|
401
|
+
if plain_visible != styled_visible:
|
|
402
|
+
return styled_spans
|
|
403
|
+
|
|
404
|
+
output: list[dict[str, Any]] = []
|
|
405
|
+
visible_cursor = 0
|
|
406
|
+
for kind, value in equations:
|
|
407
|
+
if kind == "equation":
|
|
408
|
+
append_equation_span(output, value)
|
|
409
|
+
continue
|
|
410
|
+
next_cursor = visible_cursor + len(value)
|
|
411
|
+
extend_inline_spans(output, slice_span_dicts(styled_spans, visible_cursor, next_cursor))
|
|
412
|
+
visible_cursor = next_cursor
|
|
413
|
+
return strip_span_dicts(output)
|
|
414
|
+
|
|
415
|
+
@staticmethod
|
|
416
|
+
def _get_paragraph_text_from_contents(
|
|
417
|
+
inner_contents: list[Union[Run, Hyperlink]],
|
|
418
|
+
) -> str:
|
|
419
|
+
"""Rebuild paragraph plain text from visible inline containers."""
|
|
420
|
+
return "".join(content.text or "" for content in inner_contents)
|
|
421
|
+
|
|
422
|
+
def _get_paragraph_text(self, paragraph: Paragraph) -> str:
|
|
423
|
+
"""Return paragraph plain text, including inline ``w:sdt`` content."""
|
|
424
|
+
return self._get_paragraph_text_from_contents(list(self._iter_paragraph_inner_content(paragraph)))
|
|
425
|
+
|
|
426
|
+
def _resolve_style_chain_bool(
|
|
427
|
+
self,
|
|
428
|
+
style_obj,
|
|
429
|
+
attr_name: str,
|
|
430
|
+
) -> Optional[bool]:
|
|
431
|
+
"""从样式继承链中解析布尔字体属性。"""
|
|
432
|
+
if style_obj is None:
|
|
433
|
+
return None
|
|
434
|
+
|
|
435
|
+
cache_key = (id(style_obj), attr_name)
|
|
436
|
+
if cache_key in self._style_bool_cache:
|
|
437
|
+
return self._style_bool_cache[cache_key]
|
|
438
|
+
|
|
439
|
+
style = style_obj
|
|
440
|
+
result = None
|
|
441
|
+
visited_styles = set()
|
|
442
|
+
while style is not None:
|
|
443
|
+
style_element = getattr(style, "_element", None)
|
|
444
|
+
style_id = getattr(style, "style_id", None)
|
|
445
|
+
style_type = getattr(style, "type", None)
|
|
446
|
+
# DOCX 可能存在 basedOn 自引用或环形引用,记录已访问样式避免继承链死循环。
|
|
447
|
+
style_marker = (
|
|
448
|
+
id(style_element) if style_element is not None else id(style),
|
|
449
|
+
style_type,
|
|
450
|
+
style_id,
|
|
451
|
+
)
|
|
452
|
+
if style_marker in visited_styles:
|
|
453
|
+
break
|
|
454
|
+
visited_styles.add(style_marker)
|
|
455
|
+
|
|
456
|
+
font = getattr(style, "font", None)
|
|
457
|
+
if font is not None:
|
|
458
|
+
if attr_name == "underline":
|
|
459
|
+
value = font.underline
|
|
460
|
+
elif attr_name == "strikethrough":
|
|
461
|
+
value = font.strike
|
|
462
|
+
else:
|
|
463
|
+
value = getattr(font, attr_name, None)
|
|
464
|
+
if value is not None:
|
|
465
|
+
result = bool(value)
|
|
466
|
+
break
|
|
467
|
+
style = getattr(style, "base_style", None)
|
|
468
|
+
self._style_bool_cache[cache_key] = result
|
|
469
|
+
return result
|
|
470
|
+
|
|
471
|
+
def _resolve_run_bool_with_inheritance(
|
|
472
|
+
self,
|
|
473
|
+
run: Run,
|
|
474
|
+
attr_name: str,
|
|
475
|
+
) -> bool:
|
|
476
|
+
"""解析 run 的字体属性,支持 run/字符样式/段落样式继承。"""
|
|
477
|
+
if attr_name == "underline":
|
|
478
|
+
direct_value = run.underline
|
|
479
|
+
elif attr_name == "strikethrough":
|
|
480
|
+
direct_value = run.font.strike
|
|
481
|
+
else:
|
|
482
|
+
direct_value = getattr(run, attr_name, None)
|
|
483
|
+
|
|
484
|
+
if direct_value is not None:
|
|
485
|
+
return bool(direct_value)
|
|
486
|
+
|
|
487
|
+
# 先看 run 级字符样式链(跳过 Hyperlink 默认字符样式,避免把默认下划线
|
|
488
|
+
# 误当作正文强调样式注入到解析结果中)
|
|
489
|
+
run_style = self._get_run_style(run)
|
|
490
|
+
run_style_id = str(getattr(run_style, "style_id", "") or "").lower()
|
|
491
|
+
run_style_name = str(getattr(run_style, "name", "") or "").lower()
|
|
492
|
+
is_hyperlink_style = run_style_id == "hyperlink" or "hyperlink" in run_style_name
|
|
493
|
+
if not is_hyperlink_style:
|
|
494
|
+
inherited = self._resolve_style_chain_bool(run_style, attr_name)
|
|
495
|
+
if inherited is not None:
|
|
496
|
+
return inherited
|
|
497
|
+
|
|
498
|
+
# 再看所在段落样式链
|
|
499
|
+
parent = getattr(run, "_parent", None)
|
|
500
|
+
inherited = self._resolve_style_chain_bool(
|
|
501
|
+
self._get_paragraph_style(parent),
|
|
502
|
+
attr_name,
|
|
503
|
+
)
|
|
504
|
+
if inherited is not None:
|
|
505
|
+
return inherited
|
|
506
|
+
|
|
507
|
+
return False
|
|
508
|
+
|
|
509
|
+
@staticmethod
|
|
510
|
+
def _get_direct_underline_style(run: Run) -> str:
|
|
511
|
+
"""读取 run 级下划线类型,用于区分 words 这类不作用于空格的下划线。"""
|
|
512
|
+
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
513
|
+
rPr = run._element.find(f"{{{_W}}}rPr")
|
|
514
|
+
if rPr is None:
|
|
515
|
+
return ""
|
|
516
|
+
underline = rPr.find(f"{{{_W}}}u")
|
|
517
|
+
if underline is None:
|
|
518
|
+
return ""
|
|
519
|
+
return underline.get(f"{{{_W}}}val", "single")
|
|
520
|
+
|
|
521
|
+
def _get_format_from_run(self, run: Run) -> Optional[Formatting]:
|
|
522
|
+
"""
|
|
523
|
+
从 Run 对象获取格式信息。
|
|
524
|
+
|
|
525
|
+
Args:
|
|
526
|
+
run: Run 对象
|
|
527
|
+
|
|
528
|
+
Returns:
|
|
529
|
+
Optional[Formatting]: 格式对象
|
|
530
|
+
"""
|
|
531
|
+
is_bold = self._resolve_run_bool_with_inheritance(run, "bold")
|
|
532
|
+
is_italic = self._resolve_run_bool_with_inheritance(run, "italic")
|
|
533
|
+
is_strikethrough = self._resolve_run_bool_with_inheritance(run, "strikethrough")
|
|
534
|
+
is_underline = self._resolve_run_bool_with_inheritance(run, "underline")
|
|
535
|
+
underline_style = self._get_direct_underline_style(run)
|
|
536
|
+
|
|
537
|
+
# 检测着重符号 (w:em):独立保留为 emphasis,避免和真实下划线混淆。
|
|
538
|
+
is_emphasis = False
|
|
539
|
+
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
540
|
+
rPr = run._element.find(f"{{{_W}}}rPr")
|
|
541
|
+
if rPr is not None:
|
|
542
|
+
em = rPr.find(f"{{{_W}}}em")
|
|
543
|
+
if em is not None:
|
|
544
|
+
em_val = em.get(f"{{{_W}}}val", "")
|
|
545
|
+
if em_val and em_val != "none":
|
|
546
|
+
is_emphasis = True
|
|
547
|
+
|
|
548
|
+
is_sub = run.font.subscript or False
|
|
549
|
+
is_sup = run.font.superscript or False
|
|
550
|
+
script = Script.SUB if is_sub else Script.SUPER if is_sup else Script.BASELINE
|
|
551
|
+
|
|
552
|
+
return Formatting(
|
|
553
|
+
bold=is_bold,
|
|
554
|
+
italic=is_italic,
|
|
555
|
+
underline=is_underline,
|
|
556
|
+
underline_style=underline_style,
|
|
557
|
+
emphasis=is_emphasis,
|
|
558
|
+
strikethrough=is_strikethrough,
|
|
559
|
+
script=script,
|
|
560
|
+
)
|
|
561
|
+
|
|
562
|
+
def _handle_equations_in_text(
|
|
563
|
+
self,
|
|
564
|
+
element: Any,
|
|
565
|
+
text: str,
|
|
566
|
+
*,
|
|
567
|
+
part: Any | None = None,
|
|
568
|
+
) -> tuple[str, list[tuple[str, str]]]:
|
|
569
|
+
"""
|
|
570
|
+
处理文本中的公式。
|
|
571
|
+
|
|
572
|
+
Args:
|
|
573
|
+
element: 元素对象
|
|
574
|
+
text: 文本内容
|
|
575
|
+
part: 当前段落所属的 OOXML part,用于解析局部 relationship
|
|
576
|
+
|
|
577
|
+
Returns:
|
|
578
|
+
tuple: (原始可见文本, 含公式时的有序 text/equation token)
|
|
579
|
+
"""
|
|
580
|
+
source_part = part or self._require_document_part()
|
|
581
|
+
only_texts: list[str] = []
|
|
582
|
+
formula_values: list[str] = []
|
|
583
|
+
tokens: list[tuple[str, str]] = []
|
|
584
|
+
for token_kind, value in self._docx_formula_tokens(element, source_part):
|
|
585
|
+
if token_kind == "text":
|
|
586
|
+
only_texts.append(value)
|
|
587
|
+
tokens.append(("text", value))
|
|
588
|
+
continue
|
|
589
|
+
formula_values.append(value)
|
|
590
|
+
tokens.append(("equation", value))
|
|
591
|
+
|
|
592
|
+
if not formula_values:
|
|
593
|
+
return text, []
|
|
594
|
+
|
|
595
|
+
if "".join(only_texts) != text:
|
|
596
|
+
# 如果我们无法重构初始原始文本
|
|
597
|
+
# 不要尝试解析公式并返回原始文本
|
|
598
|
+
return text, []
|
|
599
|
+
|
|
600
|
+
return text, tokens
|
|
601
|
+
|
|
602
|
+
def _get_label_and_level(self, paragraph: Paragraph) -> tuple[str, Optional[int]]:
|
|
603
|
+
"""
|
|
604
|
+
获取段落的标签和层级。
|
|
605
|
+
|
|
606
|
+
Args:
|
|
607
|
+
paragraph: 段落对象
|
|
608
|
+
|
|
609
|
+
Returns:
|
|
610
|
+
tuple[str, Optional[int]]: (标签, 层级) 元组
|
|
611
|
+
"""
|
|
612
|
+
paragraph_style = self._get_paragraph_style(paragraph)
|
|
613
|
+
if paragraph_style is None:
|
|
614
|
+
return "Normal", None
|
|
615
|
+
|
|
616
|
+
label = paragraph_style.style_id
|
|
617
|
+
name = paragraph_style.name
|
|
618
|
+
|
|
619
|
+
if label is None:
|
|
620
|
+
return "Normal", None
|
|
621
|
+
|
|
622
|
+
for style in self._iter_style_chain(paragraph_style):
|
|
623
|
+
style_label = getattr(style, "style_id", None)
|
|
624
|
+
style_name = getattr(style, "name", None)
|
|
625
|
+
|
|
626
|
+
if style_label and ":" in style_label:
|
|
627
|
+
parts = style_label.split(":")
|
|
628
|
+
if len(parts) == 2:
|
|
629
|
+
return parts[0], self._str_to_int(parts[1], None)
|
|
630
|
+
|
|
631
|
+
for candidate in (style_label, style_name):
|
|
632
|
+
if candidate and "heading" in candidate.lower():
|
|
633
|
+
return self._get_heading_and_level(candidate)
|
|
634
|
+
|
|
635
|
+
outline_level = self._get_effective_outline_level(paragraph)
|
|
636
|
+
if outline_level is not None:
|
|
637
|
+
return "Heading", outline_level + 1
|
|
638
|
+
|
|
639
|
+
return name or label or "Normal", None
|
|
640
|
+
|
|
641
|
+
def _iter_style_chain(self, style: Any) -> Iterator[Any]:
|
|
642
|
+
"""Yield a style and its base-style chain once each."""
|
|
643
|
+
seen: set[int] = set()
|
|
644
|
+
current = style
|
|
645
|
+
while current is not None:
|
|
646
|
+
current_id = id(current)
|
|
647
|
+
if current_id in seen:
|
|
648
|
+
break
|
|
649
|
+
seen.add(current_id)
|
|
650
|
+
yield current
|
|
651
|
+
current = getattr(current, "base_style", None)
|
|
652
|
+
|
|
653
|
+
def _get_paragraph_property_child(
|
|
654
|
+
self, xml_element: Optional[BaseOxmlElement], child_tag: str
|
|
655
|
+
) -> Optional[BaseOxmlElement]:
|
|
656
|
+
"""Read a direct child from w:pPr without matching nested descendants."""
|
|
657
|
+
if xml_element is None:
|
|
658
|
+
return None
|
|
659
|
+
|
|
660
|
+
namespaces = getattr(xml_element, "nsmap", None) or _DocxConstants._BLIP_NAMESPACES
|
|
661
|
+
pPr = xml_element.find("w:pPr", namespaces=namespaces)
|
|
662
|
+
if pPr is None:
|
|
663
|
+
return None
|
|
664
|
+
return pPr.find(child_tag, namespaces=namespaces)
|
|
665
|
+
|
|
666
|
+
def _get_effective_numPr(self, paragraph: Paragraph) -> Optional[BaseOxmlElement]:
|
|
667
|
+
"""Resolve paragraph numbering from direct properties, then style inheritance."""
|
|
668
|
+
numPr = self._get_paragraph_property_child(paragraph._element, "w:numPr")
|
|
669
|
+
if numPr is not None:
|
|
670
|
+
return numPr
|
|
671
|
+
|
|
672
|
+
for style in self._iter_style_chain(self._get_paragraph_style(paragraph)):
|
|
673
|
+
style_element = getattr(style, "element", None)
|
|
674
|
+
numPr = self._get_paragraph_property_child(style_element, "w:numPr")
|
|
675
|
+
if numPr is not None:
|
|
676
|
+
return numPr
|
|
677
|
+
|
|
678
|
+
return None
|
|
679
|
+
|
|
680
|
+
def _get_effective_outline_level(self, paragraph: Paragraph) -> Optional[int]:
|
|
681
|
+
"""Resolve outline level from paragraph properties or inherited styles."""
|
|
682
|
+
outline_lvl = self._get_paragraph_property_child(paragraph._element, "w:outlineLvl")
|
|
683
|
+
if outline_lvl is None:
|
|
684
|
+
for style in self._iter_style_chain(self._get_paragraph_style(paragraph)):
|
|
685
|
+
style_element = getattr(style, "element", None)
|
|
686
|
+
outline_lvl = self._get_paragraph_property_child(style_element, "w:outlineLvl")
|
|
687
|
+
if outline_lvl is not None:
|
|
688
|
+
break
|
|
689
|
+
|
|
690
|
+
if outline_lvl is None:
|
|
691
|
+
return None
|
|
692
|
+
|
|
693
|
+
return self._str_to_int(outline_lvl.get(self.XML_KEY), None)
|