docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
|
|
3
|
+
from pydantic import BaseModel
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Script(str, Enum):
|
|
7
|
+
"""Text script position."""
|
|
8
|
+
|
|
9
|
+
BASELINE = "baseline"
|
|
10
|
+
SUB = "sub"
|
|
11
|
+
SUPER = "super"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Formatting(BaseModel):
|
|
15
|
+
"""Formatting."""
|
|
16
|
+
|
|
17
|
+
bold: bool = False
|
|
18
|
+
italic: bool = False
|
|
19
|
+
underline: bool = False
|
|
20
|
+
underline_style: str = ""
|
|
21
|
+
emphasis: bool = False
|
|
22
|
+
strikethrough: bool = False
|
|
23
|
+
script: Script = Script.BASELINE
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
from typing import Any, BinaryIO
|
|
2
|
+
|
|
3
|
+
from ... import DocxModel
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def convert_path(file_path: str) -> list[list[dict[str, Any]]]:
|
|
7
|
+
"""从 DOCX 文件路径调用统一模型入口。"""
|
|
8
|
+
|
|
9
|
+
with open(file_path, "rb") as fh:
|
|
10
|
+
return convert_binary(fh)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def convert_binary(file_binary: BinaryIO) -> list[list[dict[str, Any]]]:
|
|
14
|
+
"""兼容旧二进制转换函数,并转发给 DocxModel。"""
|
|
15
|
+
|
|
16
|
+
return DocxModel().predict(file_binary)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
if __name__ == "__main__":
|
|
20
|
+
# provide a more robust command-line interface and resolve the demo
|
|
21
|
+
# document path relative to the project root instead of depending on
|
|
22
|
+
# the current working directory.
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
import argparse
|
|
25
|
+
|
|
26
|
+
# climb up until we find pyproject.toml or reach a reasonable depth
|
|
27
|
+
def find_project_root(start: Path) -> Path:
|
|
28
|
+
current = start
|
|
29
|
+
for _ in range(6): # avoid infinite loops
|
|
30
|
+
if (current / "pyproject.toml").exists() or (current / "README.md").exists():
|
|
31
|
+
return current
|
|
32
|
+
if current.parent == current:
|
|
33
|
+
break
|
|
34
|
+
current = current.parent
|
|
35
|
+
return start
|
|
36
|
+
|
|
37
|
+
script_path = Path(__file__).resolve()
|
|
38
|
+
project_root = find_project_root(script_path.parent)
|
|
39
|
+
default_docx = project_root / "demo" / "docx" / "demo1.docx"
|
|
40
|
+
|
|
41
|
+
parser = argparse.ArgumentParser(description="Convert a DOCX file to internal JSON representation")
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
"docx",
|
|
44
|
+
nargs="?",
|
|
45
|
+
default=str(default_docx),
|
|
46
|
+
help="path to the .docx file to convert (defaults to demo/docx/demo1.docx)",
|
|
47
|
+
)
|
|
48
|
+
args = parser.parse_args()
|
|
49
|
+
|
|
50
|
+
print(convert_path(args.docx))
|
|
@@ -0,0 +1,491 @@
|
|
|
1
|
+
"""DOCX 列表与编号处理;共享当前 Converter 的单文档状态。"""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
from docx.oxml.xmlchemy import BaseOxmlElement
|
|
5
|
+
from docx.text.paragraph import Paragraph
|
|
6
|
+
from loguru import logger
|
|
7
|
+
from .....schema import BlockType
|
|
8
|
+
|
|
9
|
+
from .context import _DocxConstants
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class _DocxNumbering:
|
|
13
|
+
"""集中维护列表与编号,不自行创建文档或持有跨文档缓存。"""
|
|
14
|
+
|
|
15
|
+
def _get_numId_and_ilvl(self, paragraph: Paragraph) -> tuple[Optional[int], Optional[int]]:
|
|
16
|
+
"""
|
|
17
|
+
获取段落的列表编号ID和层级。
|
|
18
|
+
|
|
19
|
+
Args:
|
|
20
|
+
paragraph: 段落对象
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
tuple[Optional[int], Optional[int]]: (numId, ilvl) 元组
|
|
24
|
+
"""
|
|
25
|
+
numPr = self._get_effective_numPr(paragraph)
|
|
26
|
+
|
|
27
|
+
if numPr is not None:
|
|
28
|
+
# 获取 numId 元素并提取值
|
|
29
|
+
namespaces = getattr(numPr, "nsmap", None) or _DocxConstants._BLIP_NAMESPACES
|
|
30
|
+
numId_elem = numPr.find("w:numId", namespaces=namespaces)
|
|
31
|
+
ilvl_elem = numPr.find("w:ilvl", namespaces=namespaces)
|
|
32
|
+
numId = numId_elem.get(self.XML_KEY) if numId_elem is not None else None
|
|
33
|
+
ilvl = ilvl_elem.get(self.XML_KEY) if ilvl_elem is not None else None
|
|
34
|
+
|
|
35
|
+
numId_int = self._str_to_int(numId, None)
|
|
36
|
+
ilvl_int = self._str_to_int(ilvl, None)
|
|
37
|
+
if numId_int == 0:
|
|
38
|
+
# numId=0 是 Word 中显式取消编号的信号,不能继续从样式继承编号层级。
|
|
39
|
+
return numId_int, ilvl_int
|
|
40
|
+
if numId_int is not None and ilvl_int is None:
|
|
41
|
+
ilvl_int = self._infer_numbering_ilvl_from_style(numId_int, paragraph)
|
|
42
|
+
|
|
43
|
+
return numId_int, ilvl_int
|
|
44
|
+
|
|
45
|
+
return None, None # 如果段落不是列表的一部分
|
|
46
|
+
|
|
47
|
+
def _get_numbering_num_element(self, numId: int) -> Optional[BaseOxmlElement]:
|
|
48
|
+
"""根据 numId 获取 word/numbering.xml 中的 num 定义。"""
|
|
49
|
+
numbering_root = self._get_numbering_root()
|
|
50
|
+
if numbering_root is None:
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
|
|
54
|
+
return numbering_root.find(
|
|
55
|
+
f".//w:num[@w:numId='{numId}']",
|
|
56
|
+
namespaces=namespaces,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
def _get_abstract_numbering_element(self, numId: int) -> Optional[BaseOxmlElement]:
|
|
60
|
+
"""根据 numId 获取对应的 abstractNum 定义,用于复用编号层级解析逻辑。"""
|
|
61
|
+
numbering_root = self._get_numbering_root()
|
|
62
|
+
namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
|
|
63
|
+
if numbering_root is None:
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
num_element = self._get_numbering_num_element(numId)
|
|
67
|
+
if num_element is None:
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
abstract_num_id_elem = num_element.find(".//w:abstractNumId", namespaces=namespaces)
|
|
71
|
+
if abstract_num_id_elem is None:
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
abstract_num_id = abstract_num_id_elem.get(self.XML_KEY)
|
|
75
|
+
if abstract_num_id is None:
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
abstract_num_xpath = f".//w:abstractNum[@w:abstractNumId='{abstract_num_id}']"
|
|
79
|
+
return numbering_root.find(abstract_num_xpath, namespaces=namespaces)
|
|
80
|
+
|
|
81
|
+
def _infer_numbering_ilvl_from_style(self, numId: int, paragraph: Paragraph) -> Optional[int]:
|
|
82
|
+
"""当 numPr 只有 numId 时,根据 numbering.xml 中的 pStyle 反查编号层级。"""
|
|
83
|
+
abstract_num_element = self._get_abstract_numbering_element(numId)
|
|
84
|
+
if abstract_num_element is None:
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
style_ids = {
|
|
88
|
+
str(getattr(style, "style_id", "") or "") for style in self._iter_style_chain(self._get_paragraph_style(paragraph))
|
|
89
|
+
}
|
|
90
|
+
style_ids.discard("")
|
|
91
|
+
if not style_ids:
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
|
|
95
|
+
ilvl_attr = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}ilvl"
|
|
96
|
+
for lvl_element in abstract_num_element.findall(".//w:lvl", namespaces=namespaces):
|
|
97
|
+
p_style = lvl_element.find("w:pStyle", namespaces=namespaces)
|
|
98
|
+
if p_style is None:
|
|
99
|
+
continue
|
|
100
|
+
if p_style.get(self.XML_KEY) in style_ids:
|
|
101
|
+
return self._str_to_int(lvl_element.get(ilvl_attr), None)
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
def _get_numbering_root(self) -> Optional[BaseOxmlElement]:
|
|
105
|
+
"""Load and cache word/numbering.xml once per conversion."""
|
|
106
|
+
if self._numbering_root_loaded:
|
|
107
|
+
return self._numbering_root
|
|
108
|
+
|
|
109
|
+
self._numbering_root_loaded = True
|
|
110
|
+
|
|
111
|
+
if not hasattr(self.docx_obj, "part") or not hasattr(self.docx_obj.part, "package"):
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
for part in self.docx_obj.part.package.parts:
|
|
115
|
+
if "numbering" in part.partname:
|
|
116
|
+
self._numbering_root = part.element
|
|
117
|
+
break
|
|
118
|
+
|
|
119
|
+
return self._numbering_root
|
|
120
|
+
|
|
121
|
+
def _get_numbering_level_definition(self, numId: int, ilvl: int) -> Optional[BaseOxmlElement]:
|
|
122
|
+
"""Resolve and cache the numbering level definition for a numId/ilvl pair."""
|
|
123
|
+
cache_key = (numId, ilvl)
|
|
124
|
+
if cache_key in self._numbering_level_cache:
|
|
125
|
+
return self._numbering_level_cache[cache_key]
|
|
126
|
+
|
|
127
|
+
numbering_root = self._get_numbering_root()
|
|
128
|
+
namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
|
|
129
|
+
lvl_element: Optional[BaseOxmlElement] = None
|
|
130
|
+
|
|
131
|
+
abstract_num_element = self._get_abstract_numbering_element(numId)
|
|
132
|
+
if numbering_root is not None and abstract_num_element is not None:
|
|
133
|
+
lvl_xpath = f".//w:lvl[@w:ilvl='{ilvl}']"
|
|
134
|
+
lvl_element = abstract_num_element.find(lvl_xpath, namespaces=namespaces)
|
|
135
|
+
|
|
136
|
+
self._numbering_level_cache[cache_key] = lvl_element
|
|
137
|
+
return lvl_element
|
|
138
|
+
|
|
139
|
+
def _get_numbering_level_start(self, numId: int, ilvl: int) -> int:
|
|
140
|
+
"""解析编号层级的起始值,优先使用 num/lvlOverride,其次使用 abstractNum/lvl/start。"""
|
|
141
|
+
cache_key = (numId, ilvl)
|
|
142
|
+
if cache_key in self._numbering_start_cache:
|
|
143
|
+
return self._numbering_start_cache[cache_key]
|
|
144
|
+
|
|
145
|
+
namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
|
|
146
|
+
start = 1
|
|
147
|
+
num_element = self._get_numbering_num_element(numId)
|
|
148
|
+
if num_element is not None:
|
|
149
|
+
override = num_element.find(
|
|
150
|
+
f"w:lvlOverride[@w:ilvl='{ilvl}']",
|
|
151
|
+
namespaces=namespaces,
|
|
152
|
+
)
|
|
153
|
+
if override is not None:
|
|
154
|
+
start_override = override.find(
|
|
155
|
+
"w:startOverride",
|
|
156
|
+
namespaces=namespaces,
|
|
157
|
+
)
|
|
158
|
+
if start_override is not None:
|
|
159
|
+
start = self._str_to_int(start_override.get(self.XML_KEY), start)
|
|
160
|
+
self._numbering_start_cache[cache_key] = start
|
|
161
|
+
return start
|
|
162
|
+
|
|
163
|
+
lvl_element = self._get_numbering_level_definition(numId, ilvl)
|
|
164
|
+
if lvl_element is not None:
|
|
165
|
+
start_element = lvl_element.find("w:start", namespaces=namespaces)
|
|
166
|
+
if start_element is not None:
|
|
167
|
+
start = self._str_to_int(start_element.get(self.XML_KEY), start)
|
|
168
|
+
|
|
169
|
+
self._numbering_start_cache[cache_key] = start
|
|
170
|
+
return start
|
|
171
|
+
|
|
172
|
+
def _advance_list_counter(self, numId: int, ilvl: int) -> int:
|
|
173
|
+
"""推进 Word 编号计数,并返回当前列表项应显示的真实序号。"""
|
|
174
|
+
counter_key = (numId, ilvl)
|
|
175
|
+
if counter_key not in self.list_counters:
|
|
176
|
+
current_number = self._get_numbering_level_start(numId, ilvl)
|
|
177
|
+
else:
|
|
178
|
+
current_number = self.list_counters[counter_key] + 1
|
|
179
|
+
self.list_counters[counter_key] = current_number
|
|
180
|
+
|
|
181
|
+
# 父级编号前进后,子级编号应在下次出现时重新从定义的起始值开始。
|
|
182
|
+
for key in list(self.list_counters.keys()):
|
|
183
|
+
counter_num_id, counter_ilevel = key
|
|
184
|
+
if counter_num_id == numId and counter_ilevel > ilvl:
|
|
185
|
+
self.list_counters.pop(key, None)
|
|
186
|
+
|
|
187
|
+
return current_number
|
|
188
|
+
|
|
189
|
+
def _is_numbered_list(self, numId: int, ilvl: int) -> bool:
|
|
190
|
+
"""
|
|
191
|
+
根据 numFmt 值检查列表是否为编号列表。
|
|
192
|
+
|
|
193
|
+
Args:
|
|
194
|
+
numId: 列表编号ID
|
|
195
|
+
ilvl: 列表层级
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
bool: 如果是编号列表返回 True,否则返回 False
|
|
199
|
+
"""
|
|
200
|
+
try:
|
|
201
|
+
lvl_element = self._get_numbering_level_definition(numId, ilvl)
|
|
202
|
+
if lvl_element is None:
|
|
203
|
+
return False
|
|
204
|
+
namespaces = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"}
|
|
205
|
+
|
|
206
|
+
# 获取 numFmt 元素
|
|
207
|
+
num_fmt_element = lvl_element.find(".//w:numFmt", namespaces=namespaces)
|
|
208
|
+
if num_fmt_element is None:
|
|
209
|
+
return False
|
|
210
|
+
|
|
211
|
+
num_fmt = num_fmt_element.get("{http://schemas.openxmlformats.org/wordprocessingml/2006/main}val")
|
|
212
|
+
|
|
213
|
+
# 编号格式包括: decimal, lowerRoman, upperRoman, lowerLetter, upperLetter
|
|
214
|
+
# 项目符号格式包括: bullet
|
|
215
|
+
numbered_formats = {
|
|
216
|
+
"decimal",
|
|
217
|
+
"lowerRoman",
|
|
218
|
+
"upperRoman",
|
|
219
|
+
"lowerLetter",
|
|
220
|
+
"upperLetter",
|
|
221
|
+
"decimalZero",
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
return num_fmt in numbered_formats
|
|
225
|
+
|
|
226
|
+
except Exception as e:
|
|
227
|
+
logger.debug(f"Error determining if list is numbered: {e}")
|
|
228
|
+
return False
|
|
229
|
+
|
|
230
|
+
def _add_list_item(
|
|
231
|
+
self,
|
|
232
|
+
*,
|
|
233
|
+
numid: int,
|
|
234
|
+
ilevel: int,
|
|
235
|
+
elements: list,
|
|
236
|
+
is_numbered: bool = False,
|
|
237
|
+
text: str = "",
|
|
238
|
+
equations: list = None,
|
|
239
|
+
) -> list:
|
|
240
|
+
"""
|
|
241
|
+
添加列表项。
|
|
242
|
+
|
|
243
|
+
生成的列表结构:
|
|
244
|
+
{
|
|
245
|
+
"type": "list",
|
|
246
|
+
"attribute": "ordered" / "unordered",
|
|
247
|
+
"ilevel": 0,
|
|
248
|
+
"content": [
|
|
249
|
+
{"type": "text", "content": "列表项文本"},
|
|
250
|
+
{"type": "list", "attribute": "...", "ilevel": 1, "content": [...]},
|
|
251
|
+
{"type": "text", "content": "另一个列表项"}
|
|
252
|
+
]
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
Args:
|
|
256
|
+
numid: 列表ID
|
|
257
|
+
ilevel: 缩进等级
|
|
258
|
+
elements: 元素列表
|
|
259
|
+
is_numbered: 是否编号
|
|
260
|
+
text: 处理后的文本(包含公式标记)
|
|
261
|
+
equations: 公式列表
|
|
262
|
+
|
|
263
|
+
Returns:
|
|
264
|
+
list[RefItem]: 元素引用列表
|
|
265
|
+
"""
|
|
266
|
+
if equations is None:
|
|
267
|
+
equations = []
|
|
268
|
+
if not elements:
|
|
269
|
+
return None
|
|
270
|
+
|
|
271
|
+
# 构建 content_text,处理行内公式和超链接
|
|
272
|
+
content_text = self._build_text_with_equations_and_hyperlinks(elements, text, equations)
|
|
273
|
+
content_text = self._normalize_text_block_content(content_text)
|
|
274
|
+
if not content_text:
|
|
275
|
+
return None
|
|
276
|
+
|
|
277
|
+
# 确定列表属性
|
|
278
|
+
list_attribute = "ordered" if is_numbered else "unordered"
|
|
279
|
+
list_start = self._advance_list_counter(numid, ilevel) if is_numbered else None
|
|
280
|
+
|
|
281
|
+
# 情况 1: 不存在上一个列表ID,或遇到了不同 numId 的新列表,创建新的顶层列表
|
|
282
|
+
if self.pre_num_id == -1 or self.pre_num_id != numid:
|
|
283
|
+
# 切换到不同的列表时,先重置旧列表状态
|
|
284
|
+
if self.pre_num_id != -1:
|
|
285
|
+
self._close_active_list()
|
|
286
|
+
|
|
287
|
+
list_block = {
|
|
288
|
+
"type": BlockType.LIST,
|
|
289
|
+
"attribute": list_attribute,
|
|
290
|
+
"content": [],
|
|
291
|
+
"ilevel": ilevel,
|
|
292
|
+
}
|
|
293
|
+
if list_start is not None:
|
|
294
|
+
list_block["start"] = list_start
|
|
295
|
+
self.cur_page.append(list_block)
|
|
296
|
+
# 入栈, 记录当前的列表块
|
|
297
|
+
self.list_block_stack.append(list_block)
|
|
298
|
+
|
|
299
|
+
list_item = {
|
|
300
|
+
"type": BlockType.TEXT,
|
|
301
|
+
"content": content_text,
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
list_block["content"].append(list_item)
|
|
305
|
+
self.pre_num_id = numid
|
|
306
|
+
self.pre_ilevel = ilevel
|
|
307
|
+
|
|
308
|
+
# 情况 2: 增加缩进,打开子列表
|
|
309
|
+
elif (
|
|
310
|
+
self.pre_num_id == numid # 同一个列表
|
|
311
|
+
and self.pre_ilevel != -1 # 上一个缩进级别已知
|
|
312
|
+
and self.pre_ilevel < ilevel # 当前层级比之前更缩进
|
|
313
|
+
):
|
|
314
|
+
# 创建新的子列表块
|
|
315
|
+
child_list_block = {
|
|
316
|
+
"type": BlockType.LIST,
|
|
317
|
+
"attribute": list_attribute,
|
|
318
|
+
"content": [],
|
|
319
|
+
"ilevel": ilevel,
|
|
320
|
+
}
|
|
321
|
+
if list_start is not None:
|
|
322
|
+
child_list_block["start"] = list_start
|
|
323
|
+
|
|
324
|
+
if not self.list_block_stack:
|
|
325
|
+
logger.warning(
|
|
326
|
+
f"Missing DOCX list parent for increased indent; numid={numid}, ilevel={ilevel}. Starting a new list block."
|
|
327
|
+
)
|
|
328
|
+
self.cur_page.append(child_list_block)
|
|
329
|
+
self.list_block_stack.append(child_list_block)
|
|
330
|
+
child_list_block["content"].append(
|
|
331
|
+
{
|
|
332
|
+
"type": BlockType.TEXT,
|
|
333
|
+
"content": content_text,
|
|
334
|
+
}
|
|
335
|
+
)
|
|
336
|
+
self.pre_ilevel = ilevel
|
|
337
|
+
return None
|
|
338
|
+
|
|
339
|
+
# 获取栈顶的列表块,将子列表直接添加到其content中
|
|
340
|
+
parent_list_block = self.list_block_stack[-1]
|
|
341
|
+
parent_list_block["content"].append(child_list_block)
|
|
342
|
+
|
|
343
|
+
# 入栈, 记录当前的列表块
|
|
344
|
+
self.list_block_stack.append(child_list_block)
|
|
345
|
+
|
|
346
|
+
# 添加当前列表项到子列表
|
|
347
|
+
list_item = {
|
|
348
|
+
"type": BlockType.TEXT,
|
|
349
|
+
"content": content_text,
|
|
350
|
+
}
|
|
351
|
+
child_list_block["content"].append(list_item)
|
|
352
|
+
|
|
353
|
+
# 更新目前缩进
|
|
354
|
+
self.pre_ilevel = ilevel
|
|
355
|
+
|
|
356
|
+
# 情况3: 减少缩进,关闭子列表
|
|
357
|
+
elif (
|
|
358
|
+
self.pre_num_id == numid # 同一个列表
|
|
359
|
+
and self.pre_ilevel != -1 # 上一个缩进级别已知
|
|
360
|
+
and ilevel < self.pre_ilevel # 当前层级比之前更少缩进
|
|
361
|
+
):
|
|
362
|
+
# 出栈,直到找到匹配的 ilevel
|
|
363
|
+
while self.list_block_stack:
|
|
364
|
+
top_list_block = self.list_block_stack[-1]
|
|
365
|
+
if top_list_block["ilevel"] == ilevel:
|
|
366
|
+
break
|
|
367
|
+
self.list_block_stack.pop()
|
|
368
|
+
if not self.list_block_stack:
|
|
369
|
+
logger.warning(f"Malformed DOCX list nesting; numid={numid}, ilevel={ilevel}. Starting a new list block.")
|
|
370
|
+
list_block = {
|
|
371
|
+
"type": BlockType.LIST,
|
|
372
|
+
"attribute": list_attribute,
|
|
373
|
+
"content": [],
|
|
374
|
+
"ilevel": ilevel,
|
|
375
|
+
}
|
|
376
|
+
if list_start is not None:
|
|
377
|
+
list_block["start"] = list_start
|
|
378
|
+
self.cur_page.append(list_block)
|
|
379
|
+
self.list_block_stack.append(list_block)
|
|
380
|
+
else:
|
|
381
|
+
list_block = self.list_block_stack[-1]
|
|
382
|
+
|
|
383
|
+
list_item = {
|
|
384
|
+
"type": BlockType.TEXT,
|
|
385
|
+
"content": content_text,
|
|
386
|
+
}
|
|
387
|
+
list_block["content"].append(list_item)
|
|
388
|
+
self.pre_ilevel = ilevel
|
|
389
|
+
|
|
390
|
+
# 情况 4: 同级列表项(相同缩进)
|
|
391
|
+
elif self.pre_num_id == numid and self.pre_ilevel == ilevel:
|
|
392
|
+
if not self.list_block_stack:
|
|
393
|
+
logger.warning(
|
|
394
|
+
f"Missing DOCX list block for same indent; numid={numid}, ilevel={ilevel}. Starting a new list block."
|
|
395
|
+
)
|
|
396
|
+
list_block = {
|
|
397
|
+
"type": BlockType.LIST,
|
|
398
|
+
"attribute": list_attribute,
|
|
399
|
+
"content": [],
|
|
400
|
+
"ilevel": ilevel,
|
|
401
|
+
}
|
|
402
|
+
if list_start is not None:
|
|
403
|
+
list_block["start"] = list_start
|
|
404
|
+
self.cur_page.append(list_block)
|
|
405
|
+
self.list_block_stack.append(list_block)
|
|
406
|
+
else:
|
|
407
|
+
# 获取栈顶的列表块
|
|
408
|
+
list_block = self.list_block_stack[-1]
|
|
409
|
+
|
|
410
|
+
list_item = {
|
|
411
|
+
"type": BlockType.TEXT,
|
|
412
|
+
"content": content_text,
|
|
413
|
+
}
|
|
414
|
+
list_block["content"].append(list_item)
|
|
415
|
+
|
|
416
|
+
else:
|
|
417
|
+
logger.warning(
|
|
418
|
+
"Unexpected DOCX list state in _add_list_item: "
|
|
419
|
+
f"pre_num_id={self.pre_num_id}, numid={numid}, "
|
|
420
|
+
f"pre_ilevel={self.pre_ilevel}, ilevel={ilevel}, "
|
|
421
|
+
f"stack_depth={len(self.list_block_stack)}. "
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
def _detect_heading_list_numids(self) -> set:
|
|
425
|
+
"""
|
|
426
|
+
预扫描文档,检测用作章节标题的列表numId。
|
|
427
|
+
|
|
428
|
+
判断依据(需同时满足两个条件):
|
|
429
|
+
1. 该numId的列表项之间穿插了非列表的正文内容(段落/表格等);
|
|
430
|
+
2. 该numId的列表项出现在**多个不同的缩进层级**(ilevel > 1种),
|
|
431
|
+
即为真正的多级列表结构,而非普通的单级内容条目列表。
|
|
432
|
+
|
|
433
|
+
这样可以避免将"多段内容条目之间穿插了小标签"的单级列表误判为标题列表。
|
|
434
|
+
|
|
435
|
+
Returns:
|
|
436
|
+
set: 应当转换为标题块的列表numId集合
|
|
437
|
+
"""
|
|
438
|
+
heading_numids = set()
|
|
439
|
+
# 收集文档元素序列:("list", numid, ilevel) 或 ("content",)
|
|
440
|
+
items = []
|
|
441
|
+
# 记录每个numId出现过的所有ilevel,用于判断是否为真正的多级列表
|
|
442
|
+
numid_ilvels: dict[int, set] = {}
|
|
443
|
+
|
|
444
|
+
for element in self.docx_obj.element.body:
|
|
445
|
+
tag_name = self._local_name(element)
|
|
446
|
+
if tag_name is None:
|
|
447
|
+
continue
|
|
448
|
+
if tag_name == "p":
|
|
449
|
+
try:
|
|
450
|
+
paragraph = Paragraph(element, self.docx_obj)
|
|
451
|
+
p_style_id, _ = self._get_label_and_level(paragraph)
|
|
452
|
+
numid, ilevel = self._get_numId_and_ilvl(paragraph)
|
|
453
|
+
if numid == 0:
|
|
454
|
+
numid = None
|
|
455
|
+
text = self._get_paragraph_text(paragraph).strip()
|
|
456
|
+
except Exception:
|
|
457
|
+
continue
|
|
458
|
+
|
|
459
|
+
if numid is not None and ilevel is not None and p_style_id not in ["Title", "Heading"] and text:
|
|
460
|
+
items.append(("list", numid, ilevel))
|
|
461
|
+
if numid not in numid_ilvels:
|
|
462
|
+
numid_ilvels[numid] = set()
|
|
463
|
+
numid_ilvels[numid].add(ilevel)
|
|
464
|
+
elif p_style_id not in ["Title", "Heading"] and text:
|
|
465
|
+
items.append(("content", None, None))
|
|
466
|
+
elif tag_name == "tbl":
|
|
467
|
+
items.append(("content", None, None))
|
|
468
|
+
|
|
469
|
+
# 对每个numId,检测其列表项之间是否有正文内容穿插
|
|
470
|
+
# seen_numids[numid] = True 表示该numId的最后一个列表项之后出现了正文内容
|
|
471
|
+
seen_numids: dict[int, bool] = {}
|
|
472
|
+
|
|
473
|
+
for item_type, numid, ilevel in items:
|
|
474
|
+
if item_type == "list":
|
|
475
|
+
if numid in seen_numids and seen_numids[numid]:
|
|
476
|
+
# 上次列表项之后出现了正文内容,满足条件1
|
|
477
|
+
heading_numids.add(numid)
|
|
478
|
+
seen_numids[numid] = False # 重置:记录该numId出现了新列表项
|
|
479
|
+
elif item_type == "content":
|
|
480
|
+
# 将所有已见numId标记为"之后出现了正文内容"
|
|
481
|
+
for nid in seen_numids:
|
|
482
|
+
seen_numids[nid] = True
|
|
483
|
+
|
|
484
|
+
# 条件2:只保留真正的多级列表(出现过多于1种ilevel的numId)
|
|
485
|
+
# 单级列表(如只有ilevel=0的内容条目列表)即使有正文段落穿插也不应转换为标题
|
|
486
|
+
heading_numids = {nid for nid in heading_numids if len(numid_ilvels.get(nid, set())) > 1}
|
|
487
|
+
|
|
488
|
+
if heading_numids:
|
|
489
|
+
logger.debug(f"Detected heading-style list numIds (will convert to title blocks): {heading_numids}")
|
|
490
|
+
|
|
491
|
+
return heading_numids
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""DOCX Mammoth 兼容层使用的 Office XML 解析辅助函数。"""
|
|
2
|
+
|
|
3
|
+
import xml.dom.minidom
|
|
4
|
+
|
|
5
|
+
from mammoth.docx.xmlparser import XmlText, XmlElement
|
|
6
|
+
from mammoth.docx.office_xml import _collapse_alternate_content, _namespaces
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def parse_xml_str(xml_str, namespace_mapping=None):
|
|
10
|
+
if namespace_mapping is None:
|
|
11
|
+
namespace_prefixes = {}
|
|
12
|
+
else:
|
|
13
|
+
namespace_prefixes = dict((uri, prefix) for prefix, uri in namespace_mapping)
|
|
14
|
+
|
|
15
|
+
document = xml.dom.minidom.parseString(xml_str)
|
|
16
|
+
|
|
17
|
+
def convert_node(node):
|
|
18
|
+
if node.nodeType == xml.dom.Node.ELEMENT_NODE:
|
|
19
|
+
return convert_element(node)
|
|
20
|
+
elif node.nodeType == xml.dom.Node.TEXT_NODE:
|
|
21
|
+
return XmlText(node.nodeValue)
|
|
22
|
+
else:
|
|
23
|
+
return None
|
|
24
|
+
|
|
25
|
+
def convert_element(element):
|
|
26
|
+
converted_name = convert_name(element)
|
|
27
|
+
|
|
28
|
+
converted_attributes = dict(
|
|
29
|
+
(convert_name(attribute), attribute.value)
|
|
30
|
+
for attribute in element.attributes.values()
|
|
31
|
+
if attribute.namespaceURI != "http://www.w3.org/2000/xmlns/"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
converted_children = []
|
|
35
|
+
for child_node in element.childNodes:
|
|
36
|
+
converted_child_node = convert_node(child_node)
|
|
37
|
+
if converted_child_node is not None:
|
|
38
|
+
converted_children.append(converted_child_node)
|
|
39
|
+
|
|
40
|
+
return XmlElement(converted_name, converted_attributes, converted_children)
|
|
41
|
+
|
|
42
|
+
def convert_name(node):
|
|
43
|
+
if node.namespaceURI is None:
|
|
44
|
+
return node.localName
|
|
45
|
+
else:
|
|
46
|
+
prefix = namespace_prefixes.get(node.namespaceURI)
|
|
47
|
+
if prefix is None:
|
|
48
|
+
return "{%s}%s" % (node.namespaceURI, node.localName)
|
|
49
|
+
else:
|
|
50
|
+
return "%s:%s" % (prefix, node.localName)
|
|
51
|
+
|
|
52
|
+
return convert_node(document.documentElement)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def read_str(xml_str):
|
|
56
|
+
i = parse_xml_str(xml_str, _namespaces)
|
|
57
|
+
return _collapse_alternate_content(i)[0]
|