docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1002 @@
|
|
|
1
|
+
"""把 ODF 文本、列表、表格和嵌入对象投影为 DocVortex raw blocks。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import base64
|
|
6
|
+
import html
|
|
7
|
+
import re
|
|
8
|
+
from collections.abc import Sequence
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
13
|
+
|
|
14
|
+
from .....schema import BlockType
|
|
15
|
+
from ..._shared.hyperlink import sanitize_hyperlink_target
|
|
16
|
+
from .....content.spans import (
|
|
17
|
+
append_equation_span,
|
|
18
|
+
append_text_span,
|
|
19
|
+
extend_inline_spans,
|
|
20
|
+
inline_span_plain_text,
|
|
21
|
+
strip_span_dicts,
|
|
22
|
+
)
|
|
23
|
+
from docvortex.content.mathml import mathml_to_latex
|
|
24
|
+
from docvortex.foundation.image_encoding import image_to_b64str
|
|
25
|
+
from ..image import create_text_placeholder, serialize_office_image
|
|
26
|
+
from ..rich_text import OfficeRichTextSegment, build_rich_text_from_segments
|
|
27
|
+
from .chart import parse_chart_block
|
|
28
|
+
from .constants import MAX_EXPANSION_TEXT_BYTES, qname
|
|
29
|
+
from .errors import OdfResourceLimitError
|
|
30
|
+
from .models import InlineAtom, InlineBlockGroup, InlineBreak, InlineImage, InlineMath, InlineNote, InlineText, TextStyle
|
|
31
|
+
from .package import OdfPackage
|
|
32
|
+
from .styles import OdfStyles
|
|
33
|
+
from .table import OdfTableExpansionBudget, parse_table_grid, table_grid_to_html
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
_WHITESPACE_RE = re.compile(r"[\t\r\n ]+")
|
|
37
|
+
_MAX_EXPLICIT_SPACE_COUNT = 10_000
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(slots=True)
|
|
41
|
+
class OdfTextExpansionBudget:
|
|
42
|
+
"""记录单个 ODF 文档显式文本膨胀的累计字节数。"""
|
|
43
|
+
|
|
44
|
+
used_bytes: int = 0
|
|
45
|
+
|
|
46
|
+
def charge(self, byte_count: int) -> None:
|
|
47
|
+
"""在分配膨胀文本前计费,超过固定上限时立即失败。"""
|
|
48
|
+
if byte_count < 0 or self.used_bytes > MAX_EXPANSION_TEXT_BYTES - byte_count:
|
|
49
|
+
raise OdfResourceLimitError(f"ODF resource limit exceeded: max_expansion_text_bytes={MAX_EXPANSION_TEXT_BYTES}")
|
|
50
|
+
self.used_bytes += byte_count
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True, slots=True)
|
|
54
|
+
class OdfMasterPageChange:
|
|
55
|
+
"""表示列表流中由段落样式请求的 master-page 变化。"""
|
|
56
|
+
|
|
57
|
+
master_page_name: str
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
RawFlowItem = dict[str, Any] | InlineNote
|
|
61
|
+
OdfListFlowItem = dict[str, Any] | InlineNote | OdfMasterPageChange
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _clean_xml_text(value: str | None) -> str:
|
|
65
|
+
"""折叠 XML 排版空白,显式多空格由 text:s 单独恢复。"""
|
|
66
|
+
if not value:
|
|
67
|
+
return ""
|
|
68
|
+
return _WHITESPACE_RE.sub(" ", value)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _paragraph_anchor(paragraph: etree._Element) -> str | None:
|
|
72
|
+
"""返回段落最终能够挂载到输出 block 的首个 bookmark 名称。"""
|
|
73
|
+
for tag in (qname("text", "bookmark"), qname("text", "bookmark-start")):
|
|
74
|
+
bookmark = next(paragraph.iter(tag), None)
|
|
75
|
+
if bookmark is not None and (name := bookmark.get(qname("text", "name"))):
|
|
76
|
+
return name
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def collect_emittable_anchor_targets(root: etree._Element, styles: OdfStyles) -> frozenset[str]:
|
|
81
|
+
"""收集 ODT 标题类 block 实际能够公开的 bookmark target。"""
|
|
82
|
+
targets: set[str] = set()
|
|
83
|
+
for paragraph in root.iter():
|
|
84
|
+
if paragraph.tag not in {qname("text", "p"), qname("text", "h")}:
|
|
85
|
+
continue
|
|
86
|
+
style_name = paragraph.get(qname("text", "style-name"))
|
|
87
|
+
if paragraph.tag != qname("text", "h") and not styles.is_document_title(style_name):
|
|
88
|
+
continue
|
|
89
|
+
if anchor := _paragraph_anchor(paragraph):
|
|
90
|
+
targets.add(anchor)
|
|
91
|
+
return frozenset(targets)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _style_html(text: str, style: TextStyle) -> str:
|
|
95
|
+
"""按稳定顺序把已转义文本包裹为 HTML 行内样式。"""
|
|
96
|
+
rendered = text
|
|
97
|
+
wrappers = [
|
|
98
|
+
(style.bold, "strong"),
|
|
99
|
+
(style.italic, "em"),
|
|
100
|
+
(style.underline, "u"),
|
|
101
|
+
(style.strikethrough, "s"),
|
|
102
|
+
(style.superscript, "sup"),
|
|
103
|
+
(style.subscript, "sub"),
|
|
104
|
+
]
|
|
105
|
+
for enabled, tag in wrappers:
|
|
106
|
+
if enabled:
|
|
107
|
+
rendered = f"<{tag}>{rendered}</{tag}>"
|
|
108
|
+
return rendered
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _serialize_odf_image(
|
|
112
|
+
image_bytes: bytes,
|
|
113
|
+
*,
|
|
114
|
+
part_name: str | None,
|
|
115
|
+
content_type: str | None,
|
|
116
|
+
) -> str | None:
|
|
117
|
+
"""序列化 ODF 图片;SVG、SVM 和 GDIMeta 使用安全占位图保留对象位置。"""
|
|
118
|
+
normalized_type = (content_type or "").split(";", 1)[0].strip().casefold()
|
|
119
|
+
suffix = (part_name or "").rsplit(".", 1)[-1].casefold() if "." in (part_name or "") else ""
|
|
120
|
+
if normalized_type == "image/svg+xml" or suffix == "svg":
|
|
121
|
+
placeholder = create_text_placeholder((320, 180), ["SVG image", "Preview unavailable"])
|
|
122
|
+
return image_to_b64str(placeholder, image_format="JPEG")
|
|
123
|
+
if suffix == "svm" or "gdimetafile" in normalized_type or image_bytes.startswith(b"VCLMTF"):
|
|
124
|
+
placeholder = create_text_placeholder((320, 180), ["ODF vector image", "Preview unavailable"])
|
|
125
|
+
return image_to_b64str(placeholder, image_format="JPEG")
|
|
126
|
+
return serialize_office_image(image_bytes, part_name=part_name, content_type=content_type)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def render_atoms_to_html(atoms: Sequence[InlineAtom]) -> str:
|
|
130
|
+
"""把 ODF 行内语义序列安全渲染为表格单元格 HTML。"""
|
|
131
|
+
parts: list[str] = []
|
|
132
|
+
for atom in atoms:
|
|
133
|
+
if isinstance(atom, InlineText):
|
|
134
|
+
rendered = _style_html(html.escape(atom.text), atom.style)
|
|
135
|
+
if atom.hyperlink:
|
|
136
|
+
rendered = f'<a href="{html.escape(atom.hyperlink, quote=True)}">{rendered}</a>'
|
|
137
|
+
parts.append(rendered)
|
|
138
|
+
elif isinstance(atom, InlineMath):
|
|
139
|
+
parts.append(f"<eq>{html.escape(atom.latex)}</eq>")
|
|
140
|
+
elif isinstance(atom, InlineBreak):
|
|
141
|
+
parts.append("<br/>")
|
|
142
|
+
elif isinstance(atom, InlineImage):
|
|
143
|
+
parts.append(f'<img src="{html.escape(atom.data_uri, quote=True)}" alt="{html.escape(atom.alt, quote=True)}"/>')
|
|
144
|
+
elif isinstance(atom, (InlineBlockGroup, InlineNote)):
|
|
145
|
+
continue
|
|
146
|
+
return "".join(parts)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def render_atoms_to_model(atoms: Sequence[InlineAtom], *, trim_edges: bool = False) -> list[dict[str, Any]]:
|
|
150
|
+
"""把 ODF 行内语义序列直接转换为结构化 Span。"""
|
|
151
|
+
spans: list[dict[str, Any]] = []
|
|
152
|
+
segments: list[OfficeRichTextSegment] = []
|
|
153
|
+
text_fragments: list[str] = []
|
|
154
|
+
fragment_style: tuple[str, ...] | None = None
|
|
155
|
+
fragment_hyperlink: str | None = None
|
|
156
|
+
|
|
157
|
+
def flush_text_fragments() -> None:
|
|
158
|
+
"""线性合并连续同样式文本,避免逐片段重复复制前缀。"""
|
|
159
|
+
nonlocal fragment_style, fragment_hyperlink
|
|
160
|
+
if not text_fragments:
|
|
161
|
+
return
|
|
162
|
+
segments.append(OfficeRichTextSegment("".join(text_fragments), fragment_style, fragment_hyperlink))
|
|
163
|
+
text_fragments.clear()
|
|
164
|
+
fragment_style = None
|
|
165
|
+
fragment_hyperlink = None
|
|
166
|
+
|
|
167
|
+
def flush_segments() -> None:
|
|
168
|
+
"""把连续文本片段批量写入富文本结果。"""
|
|
169
|
+
flush_text_fragments()
|
|
170
|
+
if not segments:
|
|
171
|
+
return
|
|
172
|
+
extend_inline_spans(spans, build_rich_text_from_segments(list(segments), trim_plain_edges=trim_edges and not spans))
|
|
173
|
+
segments.clear()
|
|
174
|
+
|
|
175
|
+
for atom in atoms:
|
|
176
|
+
if isinstance(atom, InlineText):
|
|
177
|
+
style_names = atom.style.names()
|
|
178
|
+
hyperlink = atom.hyperlink
|
|
179
|
+
if text_fragments and (style_names != fragment_style or hyperlink != fragment_hyperlink):
|
|
180
|
+
flush_text_fragments()
|
|
181
|
+
if not text_fragments:
|
|
182
|
+
fragment_style = style_names
|
|
183
|
+
fragment_hyperlink = hyperlink
|
|
184
|
+
text_fragments.append(atom.text)
|
|
185
|
+
continue
|
|
186
|
+
flush_segments()
|
|
187
|
+
if isinstance(atom, InlineMath):
|
|
188
|
+
append_equation_span(spans, atom.latex)
|
|
189
|
+
elif isinstance(atom, InlineBreak):
|
|
190
|
+
append_text_span(spans, "\n")
|
|
191
|
+
elif isinstance(atom, InlineImage) and atom.alt:
|
|
192
|
+
append_text_span(spans, atom.alt)
|
|
193
|
+
elif isinstance(atom, (InlineBlockGroup, InlineNote)):
|
|
194
|
+
continue
|
|
195
|
+
flush_segments()
|
|
196
|
+
return strip_span_dicts(spans) if trim_edges else spans
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class OdfBlockParser:
|
|
200
|
+
"""在单个 ODF 包上下文中解析正文、表格和嵌入资源。"""
|
|
201
|
+
|
|
202
|
+
def __init__(
|
|
203
|
+
self,
|
|
204
|
+
package: OdfPackage,
|
|
205
|
+
styles: OdfStyles,
|
|
206
|
+
*,
|
|
207
|
+
base_part: str = "content.xml",
|
|
208
|
+
shared_notes: list[str] | None = None,
|
|
209
|
+
list_counters: dict[tuple[str, int], int] | None = None,
|
|
210
|
+
list_ids: dict[str, int] | None = None,
|
|
211
|
+
collect_cell_visuals: bool = False,
|
|
212
|
+
shared_cell_visuals: list[dict[str, Any]] | None = None,
|
|
213
|
+
anchor_targets: frozenset[str] | None = None,
|
|
214
|
+
text_expansion_budget: OdfTextExpansionBudget | None = None,
|
|
215
|
+
table_expansion_budget: OdfTableExpansionBudget | None = None,
|
|
216
|
+
) -> None:
|
|
217
|
+
"""绑定单次解析包、样式、子文档路径及可跨 parser 共享的状态。"""
|
|
218
|
+
self.package = package
|
|
219
|
+
self.styles = styles
|
|
220
|
+
self.base_part = base_part
|
|
221
|
+
self.notes = shared_notes if shared_notes is not None else []
|
|
222
|
+
self._list_counters = list_counters if list_counters is not None else {}
|
|
223
|
+
self._list_ids = list_ids if list_ids is not None else {}
|
|
224
|
+
self._collect_cell_visuals = collect_cell_visuals
|
|
225
|
+
self._cell_visuals = shared_cell_visuals if shared_cell_visuals is not None else []
|
|
226
|
+
self._anchor_targets = anchor_targets or frozenset()
|
|
227
|
+
self._text_expansion_budget = text_expansion_budget or OdfTextExpansionBudget()
|
|
228
|
+
self.table_expansion_budget = table_expansion_budget or OdfTableExpansionBudget()
|
|
229
|
+
|
|
230
|
+
def _append_text_atom(
|
|
231
|
+
self,
|
|
232
|
+
atoms: list[InlineAtom],
|
|
233
|
+
value: str | None,
|
|
234
|
+
*,
|
|
235
|
+
style: TextStyle,
|
|
236
|
+
hyperlink: str | None,
|
|
237
|
+
preserve_whitespace: bool = False,
|
|
238
|
+
) -> None:
|
|
239
|
+
"""清理并追加文本节点;显式 ODF 空格可跳过普通 XML 空白折叠。"""
|
|
240
|
+
text = value if preserve_whitespace else _clean_xml_text(value)
|
|
241
|
+
if not text:
|
|
242
|
+
return
|
|
243
|
+
atoms.append(InlineText(text=text, style=style, hyperlink=hyperlink))
|
|
244
|
+
|
|
245
|
+
def _walk_inlines(
|
|
246
|
+
self,
|
|
247
|
+
element: etree._Element,
|
|
248
|
+
*,
|
|
249
|
+
style: TextStyle,
|
|
250
|
+
hyperlink: str | None,
|
|
251
|
+
atoms: list[InlineAtom],
|
|
252
|
+
) -> None:
|
|
253
|
+
"""递归遍历段落行内节点,并把 frame 视觉对象旁路为 block。"""
|
|
254
|
+
self._append_text_atom(atoms, element.text, style=style, hyperlink=hyperlink)
|
|
255
|
+
for child in element:
|
|
256
|
+
if not isinstance(child.tag, str):
|
|
257
|
+
self._append_text_atom(atoms, child.tail, style=style, hyperlink=hyperlink)
|
|
258
|
+
continue
|
|
259
|
+
if child.tag == qname("text", "span"):
|
|
260
|
+
span_style = self.styles.text_style(
|
|
261
|
+
child.get(qname("text", "style-name")),
|
|
262
|
+
family="text",
|
|
263
|
+
inherited=style,
|
|
264
|
+
)
|
|
265
|
+
self._walk_inlines(
|
|
266
|
+
child,
|
|
267
|
+
style=span_style,
|
|
268
|
+
hyperlink=hyperlink,
|
|
269
|
+
atoms=atoms,
|
|
270
|
+
)
|
|
271
|
+
elif child.tag == qname("text", "a"):
|
|
272
|
+
target = sanitize_hyperlink_target(
|
|
273
|
+
child.get(qname("xlink", "href")),
|
|
274
|
+
allow_relative=True,
|
|
275
|
+
allow_fragment=True,
|
|
276
|
+
)
|
|
277
|
+
if target is not None and target.startswith("#") and target[1:] not in self._anchor_targets:
|
|
278
|
+
target = None
|
|
279
|
+
self._walk_inlines(
|
|
280
|
+
child,
|
|
281
|
+
style=style,
|
|
282
|
+
hyperlink=target or hyperlink,
|
|
283
|
+
atoms=atoms,
|
|
284
|
+
)
|
|
285
|
+
elif child.tag == qname("text", "s"):
|
|
286
|
+
count = _positive_space_count(child.get(qname("text", "c")))
|
|
287
|
+
self._text_expansion_budget.charge(count)
|
|
288
|
+
self._append_text_atom(
|
|
289
|
+
atoms,
|
|
290
|
+
" " * count,
|
|
291
|
+
style=style,
|
|
292
|
+
hyperlink=hyperlink,
|
|
293
|
+
preserve_whitespace=True,
|
|
294
|
+
)
|
|
295
|
+
elif child.tag == qname("text", "tab"):
|
|
296
|
+
self._append_text_atom(atoms, " ", style=style, hyperlink=hyperlink)
|
|
297
|
+
elif child.tag == qname("text", "line-break"):
|
|
298
|
+
atoms.append(InlineBreak())
|
|
299
|
+
elif child.tag == qname("text", "soft-page-break"):
|
|
300
|
+
pass
|
|
301
|
+
elif child.tag == qname("text", "note"):
|
|
302
|
+
self._parse_note(child, style=style, hyperlink=hyperlink, atoms=atoms)
|
|
303
|
+
elif child.tag == qname("office", "annotation"):
|
|
304
|
+
if annotation_text := self._annotation_text(child):
|
|
305
|
+
atoms.append(InlineNote(annotation_text))
|
|
306
|
+
elif child.tag == qname("office", "annotation-end"):
|
|
307
|
+
pass
|
|
308
|
+
elif child.tag == qname("draw", "frame"):
|
|
309
|
+
inline_atom, blocks = self._parse_frame(child)
|
|
310
|
+
if inline_atom is not None:
|
|
311
|
+
atoms.append(inline_atom)
|
|
312
|
+
if blocks:
|
|
313
|
+
atoms.append(
|
|
314
|
+
InlineBlockGroup(
|
|
315
|
+
tuple(blocks),
|
|
316
|
+
inline_image_rendered=isinstance(inline_atom, InlineImage),
|
|
317
|
+
)
|
|
318
|
+
)
|
|
319
|
+
elif child.tag == qname("math", "math"):
|
|
320
|
+
if latex := mathml_to_latex(child):
|
|
321
|
+
atoms.append(InlineMath(latex))
|
|
322
|
+
elif child.tag in {
|
|
323
|
+
qname("text", "bookmark"),
|
|
324
|
+
qname("text", "bookmark-start"),
|
|
325
|
+
qname("text", "bookmark-end"),
|
|
326
|
+
}:
|
|
327
|
+
pass
|
|
328
|
+
else:
|
|
329
|
+
self._walk_inlines(
|
|
330
|
+
child,
|
|
331
|
+
style=style,
|
|
332
|
+
hyperlink=hyperlink,
|
|
333
|
+
atoms=atoms,
|
|
334
|
+
)
|
|
335
|
+
self._append_text_atom(atoms, child.tail, style=style, hyperlink=hyperlink)
|
|
336
|
+
|
|
337
|
+
def _annotation_text(self, annotation: etree._Element) -> str:
|
|
338
|
+
"""只提取 ODF annotation 的正文段落与列表,不混入作者日期元数据。"""
|
|
339
|
+
return flatten_block_text(self.parse_container(annotation)).strip()
|
|
340
|
+
|
|
341
|
+
def _parse_note(
|
|
342
|
+
self,
|
|
343
|
+
note: etree._Element,
|
|
344
|
+
*,
|
|
345
|
+
style: TextStyle,
|
|
346
|
+
hyperlink: str | None,
|
|
347
|
+
atoms: list[InlineAtom],
|
|
348
|
+
) -> None:
|
|
349
|
+
"""保留脚注标记,并把 note-body 内容排入当前逻辑页脚注队列。"""
|
|
350
|
+
citation = note.find(qname("text", "note-citation"))
|
|
351
|
+
citation_text = (
|
|
352
|
+
"".join(citation.itertext()).strip()
|
|
353
|
+
if citation is not None
|
|
354
|
+
else str(len(self.notes) + sum(isinstance(atom, InlineNote) for atom in atoms) + 1)
|
|
355
|
+
)
|
|
356
|
+
self._append_text_atom(atoms, f"[{citation_text}]", style=style, hyperlink=hyperlink)
|
|
357
|
+
body = note.find(qname("text", "note-body"))
|
|
358
|
+
if body is None:
|
|
359
|
+
return
|
|
360
|
+
blocks = self.parse_container(body)
|
|
361
|
+
visible = flatten_block_text(blocks)
|
|
362
|
+
if visible:
|
|
363
|
+
atoms.append(InlineNote(f"[{citation_text}] {visible}"))
|
|
364
|
+
|
|
365
|
+
def parse_inline_atoms(self, paragraph: etree._Element) -> list[InlineAtom]:
|
|
366
|
+
"""解析一个段落的行内语义,并用原位 marker 保留段外 block。"""
|
|
367
|
+
paragraph_style = self.styles.text_style(
|
|
368
|
+
paragraph.get(qname("text", "style-name")),
|
|
369
|
+
family="paragraph",
|
|
370
|
+
)
|
|
371
|
+
atoms: list[InlineAtom] = []
|
|
372
|
+
self._walk_inlines(
|
|
373
|
+
paragraph,
|
|
374
|
+
style=paragraph_style,
|
|
375
|
+
hyperlink=None,
|
|
376
|
+
atoms=atoms,
|
|
377
|
+
)
|
|
378
|
+
return atoms
|
|
379
|
+
|
|
380
|
+
def parse_paragraph(self, paragraph: etree._Element) -> list[RawFlowItem]:
|
|
381
|
+
"""把 text:p/text:h 转为标题、正文、公式和段外内容。"""
|
|
382
|
+
atoms = self.parse_inline_atoms(paragraph)
|
|
383
|
+
results: list[RawFlowItem] = []
|
|
384
|
+
is_heading = paragraph.tag == qname("text", "h")
|
|
385
|
+
style_name = paragraph.get(qname("text", "style-name"))
|
|
386
|
+
content_atoms = [atom for atom in atoms if not isinstance(atom, (InlineBlockGroup, InlineNote))]
|
|
387
|
+
content = render_atoms_to_model(content_atoms, trim_edges=True)
|
|
388
|
+
math_atoms = [atom for atom in content_atoms if isinstance(atom, InlineMath)]
|
|
389
|
+
visible_text = "".join(atom.text for atom in content_atoms if isinstance(atom, InlineText)).strip()
|
|
390
|
+
if content:
|
|
391
|
+
if math_atoms and not visible_text and len(math_atoms) == 1 and len(content_atoms) == 1:
|
|
392
|
+
results.append({"type": BlockType.EQUATION, "content": math_atoms[0].latex})
|
|
393
|
+
elif self.styles.is_document_title(style_name):
|
|
394
|
+
block: dict[str, Any] = {"type": BlockType.DOC_TITLE, "level": 1, "content": content}
|
|
395
|
+
if anchor := _paragraph_anchor(paragraph):
|
|
396
|
+
block["anchor"] = anchor
|
|
397
|
+
results.append(block)
|
|
398
|
+
elif is_heading:
|
|
399
|
+
try:
|
|
400
|
+
outline_level = int(paragraph.get(qname("text", "outline-level"), "1"))
|
|
401
|
+
except ValueError:
|
|
402
|
+
outline_level = 1
|
|
403
|
+
block = {
|
|
404
|
+
"type": BlockType.PARAGRAPH_TITLE,
|
|
405
|
+
"level": min(max(outline_level + 1, 2), 6),
|
|
406
|
+
"is_numbered_style": False,
|
|
407
|
+
"content": content,
|
|
408
|
+
}
|
|
409
|
+
if anchor := _paragraph_anchor(paragraph):
|
|
410
|
+
block["anchor"] = anchor
|
|
411
|
+
results.append(block)
|
|
412
|
+
else:
|
|
413
|
+
results.append({"type": BlockType.TEXT, "content": content})
|
|
414
|
+
for atom in atoms:
|
|
415
|
+
if isinstance(atom, InlineBlockGroup):
|
|
416
|
+
results.extend(atom.blocks)
|
|
417
|
+
elif isinstance(atom, InlineNote):
|
|
418
|
+
results.append(atom)
|
|
419
|
+
return results
|
|
420
|
+
|
|
421
|
+
def parse_list(
|
|
422
|
+
self,
|
|
423
|
+
element: etree._Element,
|
|
424
|
+
*,
|
|
425
|
+
depth: int = 0,
|
|
426
|
+
inherited_style: str | None = None,
|
|
427
|
+
emit_master_page_changes: bool = False,
|
|
428
|
+
) -> list[OdfListFlowItem]:
|
|
429
|
+
"""递归构造严格 LIST 分片,并把不允许嵌套的 block 提升为有序兄弟。"""
|
|
430
|
+
items = [
|
|
431
|
+
item
|
|
432
|
+
for item in element
|
|
433
|
+
if isinstance(item.tag, str) and item.tag in {qname("text", "list-item"), qname("text", "list-header")}
|
|
434
|
+
]
|
|
435
|
+
return self._parse_list_items(
|
|
436
|
+
element,
|
|
437
|
+
items,
|
|
438
|
+
depth=depth,
|
|
439
|
+
inherited_style=inherited_style,
|
|
440
|
+
emit_master_page_changes=emit_master_page_changes,
|
|
441
|
+
)
|
|
442
|
+
|
|
443
|
+
def _parse_list_items(
|
|
444
|
+
self,
|
|
445
|
+
element: etree._Element,
|
|
446
|
+
items: Sequence[etree._Element],
|
|
447
|
+
*,
|
|
448
|
+
depth: int,
|
|
449
|
+
inherited_style: str | None,
|
|
450
|
+
emit_master_page_changes: bool,
|
|
451
|
+
) -> list[OdfListFlowItem]:
|
|
452
|
+
"""按源条目构造 LIST 分片,每个条目只保留一个文本叶子和一个 marker。"""
|
|
453
|
+
style_name = element.get(qname("text", "style-name")) or inherited_style
|
|
454
|
+
level = self.styles.list_level(style_name, depth)
|
|
455
|
+
key = (style_name or "", depth)
|
|
456
|
+
start = level.start
|
|
457
|
+
continue_list = element.get(qname("text", "continue-list"))
|
|
458
|
+
if continue_list and continue_list in self._list_ids:
|
|
459
|
+
start = self._list_ids[continue_list]
|
|
460
|
+
elif element.get(qname("text", "continue-numbering")) == "true" and key in self._list_counters:
|
|
461
|
+
start = self._list_counters[key]
|
|
462
|
+
results: list[OdfListFlowItem] = []
|
|
463
|
+
content: list[dict[str, Any]] = []
|
|
464
|
+
fragment_notes: list[InlineNote] = []
|
|
465
|
+
item_count = 0
|
|
466
|
+
active_master: str | None = None
|
|
467
|
+
|
|
468
|
+
def flush_content(fragment_start: int) -> None:
|
|
469
|
+
"""把当前合法子块冻结为一个 LIST 分片。"""
|
|
470
|
+
if content:
|
|
471
|
+
block: dict[str, Any] = {
|
|
472
|
+
"type": BlockType.LIST,
|
|
473
|
+
"attribute": "ordered" if level.ordered else "unordered",
|
|
474
|
+
"ilevel": depth,
|
|
475
|
+
"content": list(content),
|
|
476
|
+
}
|
|
477
|
+
if level.ordered:
|
|
478
|
+
block["start"] = fragment_start
|
|
479
|
+
results.append(block)
|
|
480
|
+
content.clear()
|
|
481
|
+
if fragment_notes:
|
|
482
|
+
results.extend(fragment_notes)
|
|
483
|
+
fragment_notes.clear()
|
|
484
|
+
|
|
485
|
+
fragment_start = start
|
|
486
|
+
for item in items:
|
|
487
|
+
is_header = item.tag == qname("text", "list-header")
|
|
488
|
+
if not is_header:
|
|
489
|
+
if item_count == 0:
|
|
490
|
+
try:
|
|
491
|
+
item_start = int(item.get(qname("text", "start-value"), str(start)))
|
|
492
|
+
start = max(0, item_start)
|
|
493
|
+
fragment_start = start
|
|
494
|
+
except ValueError:
|
|
495
|
+
pass
|
|
496
|
+
# 统一 LIST 只支持列表级起始值,后续逐项重启按连续序号投影。
|
|
497
|
+
item_count += 1
|
|
498
|
+
first_paragraph = next(
|
|
499
|
+
(
|
|
500
|
+
child
|
|
501
|
+
for child in item
|
|
502
|
+
if isinstance(child.tag, str) and child.tag in {qname("text", "p"), qname("text", "h")}
|
|
503
|
+
),
|
|
504
|
+
None,
|
|
505
|
+
)
|
|
506
|
+
requested_master = (
|
|
507
|
+
self.styles.paragraph_master_page_name(first_paragraph.get(qname("text", "style-name")))
|
|
508
|
+
if first_paragraph is not None
|
|
509
|
+
else None
|
|
510
|
+
)
|
|
511
|
+
if emit_master_page_changes and requested_master is not None and requested_master != active_master:
|
|
512
|
+
flush_content(fragment_start)
|
|
513
|
+
results.append(OdfMasterPageChange(requested_master))
|
|
514
|
+
active_master = requested_master
|
|
515
|
+
fragment_start = start + item_count - (0 if is_header else 1)
|
|
516
|
+
text_content: list[dict[str, Any]] = []
|
|
517
|
+
nested_blocks: list[dict[str, Any]] = []
|
|
518
|
+
lifted_blocks: list[OdfListFlowItem] = []
|
|
519
|
+
|
|
520
|
+
def consume_flow(flow: Sequence[RawFlowItem]) -> None:
|
|
521
|
+
"""把段落子流投影到列表文本、嵌套列表或提升块。"""
|
|
522
|
+
for block in flow:
|
|
523
|
+
if isinstance(block, InlineNote):
|
|
524
|
+
if emit_master_page_changes:
|
|
525
|
+
fragment_notes.append(block)
|
|
526
|
+
else:
|
|
527
|
+
self.notes.append(block.content)
|
|
528
|
+
continue
|
|
529
|
+
block_type = block.get("type")
|
|
530
|
+
block_content = block.get("content")
|
|
531
|
+
if block_type in {
|
|
532
|
+
BlockType.TEXT,
|
|
533
|
+
BlockType.REF_TEXT,
|
|
534
|
+
BlockType.DOC_TITLE,
|
|
535
|
+
BlockType.PARAGRAPH_TITLE,
|
|
536
|
+
} and isinstance(block_content, list):
|
|
537
|
+
block_spans = [span for span in block_content if isinstance(span, dict)]
|
|
538
|
+
if block_spans:
|
|
539
|
+
if text_content:
|
|
540
|
+
append_text_span(text_content, "\n")
|
|
541
|
+
extend_inline_spans(text_content, block_spans)
|
|
542
|
+
elif block_type == BlockType.LIST:
|
|
543
|
+
nested_blocks.append(block)
|
|
544
|
+
else:
|
|
545
|
+
lifted_blocks.append(block)
|
|
546
|
+
|
|
547
|
+
for child in item:
|
|
548
|
+
if not isinstance(child.tag, str):
|
|
549
|
+
continue
|
|
550
|
+
if child.tag in {qname("text", "p"), qname("text", "h")}:
|
|
551
|
+
consume_flow(self.parse_paragraph(child))
|
|
552
|
+
elif child.tag == qname("text", "list"):
|
|
553
|
+
nested_flow = self.parse_list_blocks(
|
|
554
|
+
child,
|
|
555
|
+
depth=depth + 1,
|
|
556
|
+
inherited_style=style_name,
|
|
557
|
+
emit_master_page_changes=emit_master_page_changes,
|
|
558
|
+
)
|
|
559
|
+
if not any(isinstance(block, OdfMasterPageChange) for block in nested_flow) and all(
|
|
560
|
+
isinstance(block, InlineNote) or block.get("type") == BlockType.LIST for block in nested_flow
|
|
561
|
+
):
|
|
562
|
+
nested_blocks.extend(block for block in nested_flow if isinstance(block, dict))
|
|
563
|
+
fragment_notes.extend(block for block in nested_flow if isinstance(block, InlineNote))
|
|
564
|
+
else:
|
|
565
|
+
lifted_blocks.extend(nested_flow)
|
|
566
|
+
else:
|
|
567
|
+
consume_flow(self.parse_container(child))
|
|
568
|
+
if text_content:
|
|
569
|
+
content.append({"type": BlockType.TEXT, "content": text_content})
|
|
570
|
+
content.extend(nested_blocks)
|
|
571
|
+
if lifted_blocks:
|
|
572
|
+
flush_content(fragment_start)
|
|
573
|
+
results.extend(lifted_blocks)
|
|
574
|
+
fragment_start = start + item_count
|
|
575
|
+
|
|
576
|
+
flush_content(fragment_start)
|
|
577
|
+
next_value = start + item_count
|
|
578
|
+
self._list_counters[key] = next_value
|
|
579
|
+
if list_id := element.get(qname("xml", "id")):
|
|
580
|
+
self._list_ids[list_id] = next_value
|
|
581
|
+
return results
|
|
582
|
+
|
|
583
|
+
def parse_list_blocks(
|
|
584
|
+
self,
|
|
585
|
+
element: etree._Element,
|
|
586
|
+
*,
|
|
587
|
+
depth: int = 0,
|
|
588
|
+
inherited_style: str | None = None,
|
|
589
|
+
emit_master_page_changes: bool = False,
|
|
590
|
+
) -> list[OdfListFlowItem]:
|
|
591
|
+
"""把含 text:h 的编号章节提升为标题,并保留其余连续列表。"""
|
|
592
|
+
if next(element.iter(qname("text", "h")), None) is None:
|
|
593
|
+
return self.parse_list(
|
|
594
|
+
element,
|
|
595
|
+
depth=depth,
|
|
596
|
+
inherited_style=inherited_style,
|
|
597
|
+
emit_master_page_changes=emit_master_page_changes,
|
|
598
|
+
)
|
|
599
|
+
results: list[OdfListFlowItem] = []
|
|
600
|
+
pending_items: list[etree._Element] = []
|
|
601
|
+
active_master: str | None = None
|
|
602
|
+
|
|
603
|
+
def flush_pending() -> None:
|
|
604
|
+
"""把标题之间积累的普通列表项写为独立连续 LIST block。"""
|
|
605
|
+
if not pending_items:
|
|
606
|
+
return
|
|
607
|
+
blocks = self._parse_list_items(
|
|
608
|
+
element,
|
|
609
|
+
list(pending_items),
|
|
610
|
+
depth=depth,
|
|
611
|
+
inherited_style=inherited_style,
|
|
612
|
+
emit_master_page_changes=emit_master_page_changes,
|
|
613
|
+
)
|
|
614
|
+
pending_items.clear()
|
|
615
|
+
results.extend(blocks)
|
|
616
|
+
|
|
617
|
+
for item in element:
|
|
618
|
+
if not isinstance(item.tag, str) or item.tag not in {
|
|
619
|
+
qname("text", "list-item"),
|
|
620
|
+
qname("text", "list-header"),
|
|
621
|
+
}:
|
|
622
|
+
continue
|
|
623
|
+
if next(item.iter(qname("text", "h")), None) is None:
|
|
624
|
+
pending_items.append(item)
|
|
625
|
+
continue
|
|
626
|
+
flush_pending()
|
|
627
|
+
for child in item:
|
|
628
|
+
if not isinstance(child.tag, str):
|
|
629
|
+
continue
|
|
630
|
+
if child.tag in {qname("text", "p"), qname("text", "h")}:
|
|
631
|
+
requested_master = self.styles.paragraph_master_page_name(child.get(qname("text", "style-name")))
|
|
632
|
+
if emit_master_page_changes and requested_master is not None and requested_master != active_master:
|
|
633
|
+
results.append(OdfMasterPageChange(requested_master))
|
|
634
|
+
active_master = requested_master
|
|
635
|
+
for parsed in self.parse_paragraph(child):
|
|
636
|
+
if isinstance(parsed, InlineNote):
|
|
637
|
+
if emit_master_page_changes:
|
|
638
|
+
results.append(parsed)
|
|
639
|
+
else:
|
|
640
|
+
self.notes.append(parsed.content)
|
|
641
|
+
continue
|
|
642
|
+
if child.tag == qname("text", "h") and parsed.get("type") == BlockType.PARAGRAPH_TITLE:
|
|
643
|
+
parsed["is_numbered_style"] = True
|
|
644
|
+
results.append(parsed)
|
|
645
|
+
elif child.tag == qname("text", "list"):
|
|
646
|
+
results.extend(
|
|
647
|
+
self.parse_list_blocks(
|
|
648
|
+
child,
|
|
649
|
+
depth=depth + 1,
|
|
650
|
+
inherited_style=element.get(qname("text", "style-name")) or inherited_style,
|
|
651
|
+
emit_master_page_changes=emit_master_page_changes,
|
|
652
|
+
)
|
|
653
|
+
)
|
|
654
|
+
else:
|
|
655
|
+
results.extend(self.parse_element(child))
|
|
656
|
+
flush_pending()
|
|
657
|
+
return results
|
|
658
|
+
|
|
659
|
+
def _parse_index(self, element: etree._Element) -> dict[str, Any] | None:
|
|
660
|
+
"""把 ODF 已存储目录正文转换为扁平 INDEX 子项。"""
|
|
661
|
+
leaves: list[dict[str, Any]] = []
|
|
662
|
+
for paragraph in element.iter():
|
|
663
|
+
if paragraph.tag not in {qname("text", "p"), qname("text", "h")}:
|
|
664
|
+
continue
|
|
665
|
+
for block in self.parse_paragraph(paragraph):
|
|
666
|
+
if isinstance(block, InlineNote):
|
|
667
|
+
self.notes.append(block.content)
|
|
668
|
+
continue
|
|
669
|
+
if isinstance(block, dict) and block.get("content"):
|
|
670
|
+
leaves.append({"type": BlockType.TEXT, "content": block["content"]})
|
|
671
|
+
if not leaves:
|
|
672
|
+
return None
|
|
673
|
+
return {"type": BlockType.INDEX, "ilevel": 0, "content": leaves}
|
|
674
|
+
|
|
675
|
+
def parse_table(self, element: etree._Element) -> dict[str, Any] | None:
|
|
676
|
+
"""把一个 ODF table 转为包含合并语义的 TABLE raw block。"""
|
|
677
|
+
grid = parse_table_grid(element, self.render_cell_html, expansion_budget=self.table_expansion_budget)
|
|
678
|
+
content = table_grid_to_html(grid)
|
|
679
|
+
return {"type": BlockType.TABLE, "content": content} if content else None
|
|
680
|
+
|
|
681
|
+
def _load_image(self, image: etree._Element) -> tuple[str | None, str]:
|
|
682
|
+
"""读取 draw:image 的包内或内联载荷并复用 Office 图片序列化。"""
|
|
683
|
+
href = image.get(qname("xlink", "href"), "")
|
|
684
|
+
part_name = self.package.resolve_reference(href, base_part=self.base_part) if href else None
|
|
685
|
+
image_bytes: bytes | None = None
|
|
686
|
+
content_type: str | None = None
|
|
687
|
+
if part_name:
|
|
688
|
+
image_bytes = self.package.read_part(part_name, asset=True)
|
|
689
|
+
content_type = self.package.content_type_for(part_name)
|
|
690
|
+
if image_bytes is None:
|
|
691
|
+
binary = image.find(f".//{qname('office', 'binary-data')}")
|
|
692
|
+
if binary is not None and (binary.text or "").strip():
|
|
693
|
+
try:
|
|
694
|
+
image_bytes = base64.b64decode("".join((binary.text or "").split()), validate=True)
|
|
695
|
+
except (ValueError, TypeError):
|
|
696
|
+
image_bytes = None
|
|
697
|
+
alt = ""
|
|
698
|
+
parent = image.getparent()
|
|
699
|
+
if parent is not None:
|
|
700
|
+
title = parent.find(qname("svg", "title"))
|
|
701
|
+
description = parent.find(qname("svg", "desc"))
|
|
702
|
+
alt = " ".join(
|
|
703
|
+
text.strip()
|
|
704
|
+
for text in (
|
|
705
|
+
"".join(title.itertext()) if title is not None else "",
|
|
706
|
+
"".join(description.itertext()) if description is not None else "",
|
|
707
|
+
)
|
|
708
|
+
if text.strip()
|
|
709
|
+
)
|
|
710
|
+
if not image_bytes:
|
|
711
|
+
return None, alt
|
|
712
|
+
return _serialize_odf_image(image_bytes, part_name=part_name, content_type=content_type), alt
|
|
713
|
+
|
|
714
|
+
def _object_root(self, object_element: etree._Element) -> tuple[etree._Element | None, str | None]:
|
|
715
|
+
"""读取 draw:object 指向的子文档内容树和成员路径。"""
|
|
716
|
+
inline_math = next(object_element.iter(qname("math", "math")), None)
|
|
717
|
+
if inline_math is not None:
|
|
718
|
+
return inline_math, self.base_part
|
|
719
|
+
href = object_element.get(qname("xlink", "href"), "")
|
|
720
|
+
part_name = self.package.resolve_object_content(href, base_part=self.base_part)
|
|
721
|
+
if part_name is None:
|
|
722
|
+
return None, None
|
|
723
|
+
return self.package.xml_part(part_name), part_name
|
|
724
|
+
|
|
725
|
+
def _parse_frame(self, frame: etree._Element) -> tuple[InlineAtom | None, list[dict[str, Any]]]:
|
|
726
|
+
"""按公式、图表、文本框、表格、图片优先级解析一个 draw:frame。"""
|
|
727
|
+
image_element = next(frame.iter(qname("draw", "image")), None)
|
|
728
|
+
preview_uri: str | None = None
|
|
729
|
+
preview_alt = ""
|
|
730
|
+
|
|
731
|
+
def load_preview() -> tuple[str | None, str]:
|
|
732
|
+
"""只在对象需要图片回退或图表预览时读取 sibling draw:image。"""
|
|
733
|
+
nonlocal preview_uri, preview_alt
|
|
734
|
+
if image_element is not None and preview_uri is None:
|
|
735
|
+
preview_uri, preview_alt = self._load_image(image_element)
|
|
736
|
+
return preview_uri, preview_alt
|
|
737
|
+
|
|
738
|
+
object_element = next(frame.iter(qname("draw", "object")), None)
|
|
739
|
+
if object_element is not None:
|
|
740
|
+
object_root, object_part = self._object_root(object_element)
|
|
741
|
+
if object_root is not None:
|
|
742
|
+
math_element = (
|
|
743
|
+
object_root
|
|
744
|
+
if object_root.tag == qname("math", "math")
|
|
745
|
+
else next(
|
|
746
|
+
object_root.iter(qname("math", "math")),
|
|
747
|
+
None,
|
|
748
|
+
)
|
|
749
|
+
)
|
|
750
|
+
if math_element is not None and (latex := mathml_to_latex(math_element)):
|
|
751
|
+
return InlineMath(latex), []
|
|
752
|
+
load_preview()
|
|
753
|
+
object_parser = OdfBlockParser(
|
|
754
|
+
self.package,
|
|
755
|
+
self.styles,
|
|
756
|
+
base_part=object_part or self.base_part,
|
|
757
|
+
shared_notes=self.notes,
|
|
758
|
+
list_counters=self._list_counters,
|
|
759
|
+
list_ids=self._list_ids,
|
|
760
|
+
collect_cell_visuals=self._collect_cell_visuals,
|
|
761
|
+
shared_cell_visuals=self._cell_visuals,
|
|
762
|
+
anchor_targets=self._anchor_targets,
|
|
763
|
+
text_expansion_budget=self._text_expansion_budget,
|
|
764
|
+
table_expansion_budget=self.table_expansion_budget,
|
|
765
|
+
)
|
|
766
|
+
chart = parse_chart_block(
|
|
767
|
+
object_root,
|
|
768
|
+
render_cell=object_parser.render_cell_html,
|
|
769
|
+
preview_data_uri=preview_uri,
|
|
770
|
+
table_expansion_budget=self.table_expansion_budget,
|
|
771
|
+
)
|
|
772
|
+
if chart is not None:
|
|
773
|
+
return None, [chart]
|
|
774
|
+
text_box = next(frame.iter(qname("draw", "text-box")), None)
|
|
775
|
+
if text_box is not None:
|
|
776
|
+
return None, self.parse_container(text_box)
|
|
777
|
+
table = next(frame.iter(qname("table", "table")), None)
|
|
778
|
+
if table is not None and (table_block := self.parse_table(table)) is not None:
|
|
779
|
+
return None, [table_block]
|
|
780
|
+
load_preview()
|
|
781
|
+
if preview_uri:
|
|
782
|
+
return InlineImage(preview_uri, preview_alt), [{"type": BlockType.IMAGE, "image_base64": preview_uri}]
|
|
783
|
+
if preview_alt:
|
|
784
|
+
return InlineText(preview_alt), []
|
|
785
|
+
return None, []
|
|
786
|
+
|
|
787
|
+
def parse_frame_blocks(self, frame: etree._Element) -> list[dict[str, Any]]:
|
|
788
|
+
"""把 frame 的内联结果提升为页面级 block,避免正文重复图片。"""
|
|
789
|
+
inline, blocks = self._parse_frame(frame)
|
|
790
|
+
if blocks:
|
|
791
|
+
return blocks
|
|
792
|
+
if isinstance(inline, InlineMath):
|
|
793
|
+
return [{"type": BlockType.EQUATION, "content": inline.latex}]
|
|
794
|
+
if isinstance(inline, InlineImage):
|
|
795
|
+
return [{"type": BlockType.IMAGE, "image_base64": inline.data_uri}]
|
|
796
|
+
if isinstance(inline, InlineText):
|
|
797
|
+
content = render_atoms_to_model([inline], trim_edges=True)
|
|
798
|
+
return [{"type": BlockType.TEXT, "content": content}] if content else []
|
|
799
|
+
return []
|
|
800
|
+
|
|
801
|
+
def parse_element(self, element: etree._Element) -> list[dict[str, Any]]:
|
|
802
|
+
"""解析一个 ODF block 元素,不移动或修改原始 XML 节点。"""
|
|
803
|
+
if element.tag in {qname("text", "p"), qname("text", "h")}:
|
|
804
|
+
blocks: list[dict[str, Any]] = []
|
|
805
|
+
for item in self.parse_paragraph(element):
|
|
806
|
+
if isinstance(item, dict):
|
|
807
|
+
blocks.append(item)
|
|
808
|
+
elif isinstance(item, InlineNote):
|
|
809
|
+
self.notes.append(item.content)
|
|
810
|
+
return blocks
|
|
811
|
+
if element.tag == qname("text", "list"):
|
|
812
|
+
return [item for item in self.parse_list_blocks(element) if isinstance(item, dict)]
|
|
813
|
+
if element.tag == qname("office", "annotation"):
|
|
814
|
+
if annotation_text := self._annotation_text(element):
|
|
815
|
+
self.notes.append(annotation_text)
|
|
816
|
+
return []
|
|
817
|
+
if element.tag == qname("office", "annotation-end"):
|
|
818
|
+
return []
|
|
819
|
+
if element.tag == qname("table", "table"):
|
|
820
|
+
table_block = self.parse_table(element)
|
|
821
|
+
return [table_block] if table_block is not None else []
|
|
822
|
+
if element.tag == qname("draw", "frame"):
|
|
823
|
+
return self.parse_frame_blocks(element)
|
|
824
|
+
if element.tag in {
|
|
825
|
+
qname("text", "section"),
|
|
826
|
+
qname("text", "index-body"),
|
|
827
|
+
qname("text", "index-title"),
|
|
828
|
+
qname("draw", "g"),
|
|
829
|
+
qname("draw", "custom-shape"),
|
|
830
|
+
}:
|
|
831
|
+
return self.parse_container(element)
|
|
832
|
+
if element.tag in {
|
|
833
|
+
qname("text", "table-of-content"),
|
|
834
|
+
qname("text", "alphabetical-index"),
|
|
835
|
+
qname("text", "bibliography"),
|
|
836
|
+
qname("text", "illustration-index"),
|
|
837
|
+
}:
|
|
838
|
+
index = self._parse_index(element)
|
|
839
|
+
return [index] if index is not None else []
|
|
840
|
+
return []
|
|
841
|
+
|
|
842
|
+
def parse_container(self, parent: etree._Element) -> list[dict[str, Any]]:
|
|
843
|
+
"""按文档顺序解析普通 ODF block 容器,不建立页面边界。"""
|
|
844
|
+
blocks: list[dict[str, Any]] = []
|
|
845
|
+
for child in parent:
|
|
846
|
+
if isinstance(child.tag, str):
|
|
847
|
+
blocks.extend(self.parse_element(child))
|
|
848
|
+
return blocks
|
|
849
|
+
|
|
850
|
+
def _append_cell_blocks(
|
|
851
|
+
self,
|
|
852
|
+
parts: list[str],
|
|
853
|
+
blocks: Sequence[dict[str, Any]],
|
|
854
|
+
*,
|
|
855
|
+
inline_image_rendered: bool,
|
|
856
|
+
) -> None:
|
|
857
|
+
"""按单元格视觉策略收集或内联 block,并避免重复输出配对图片。"""
|
|
858
|
+
for block in blocks:
|
|
859
|
+
block_type = block.get("type")
|
|
860
|
+
if self._collect_cell_visuals and block_type in {
|
|
861
|
+
BlockType.IMAGE,
|
|
862
|
+
BlockType.CHART,
|
|
863
|
+
BlockType.EQUATION,
|
|
864
|
+
}:
|
|
865
|
+
self._cell_visuals.append(block)
|
|
866
|
+
continue
|
|
867
|
+
if inline_image_rendered and block_type == BlockType.IMAGE:
|
|
868
|
+
continue
|
|
869
|
+
if block_type in {BlockType.TABLE, BlockType.CHART} and block.get("content"):
|
|
870
|
+
parts.append(str(block["content"]))
|
|
871
|
+
elif block.get("image_base64"):
|
|
872
|
+
parts.append(f'<img src="{html.escape(str(block["image_base64"]), quote=True)}"/>')
|
|
873
|
+
|
|
874
|
+
def _queue_inline_notes(self, atoms: Sequence[InlineAtom]) -> None:
|
|
875
|
+
"""把单元格行内流中的 note marker 排入当前逻辑页队列。"""
|
|
876
|
+
self.notes.extend(atom.content for atom in atoms if isinstance(atom, InlineNote))
|
|
877
|
+
|
|
878
|
+
def render_cell_html(self, cell: etree._Element) -> str:
|
|
879
|
+
"""把表格单元格中的段落、列表、嵌套表和 frame 转为 HTML。"""
|
|
880
|
+
parts: list[str] = []
|
|
881
|
+
for child in cell:
|
|
882
|
+
if not isinstance(child.tag, str):
|
|
883
|
+
continue
|
|
884
|
+
if child.tag in {qname("text", "p"), qname("text", "h")}:
|
|
885
|
+
atoms = self.parse_inline_atoms(child)
|
|
886
|
+
self._queue_inline_notes(atoms)
|
|
887
|
+
rendered_atoms = (
|
|
888
|
+
[atom for atom in atoms if not isinstance(atom, InlineImage)] if self._collect_cell_visuals else atoms
|
|
889
|
+
)
|
|
890
|
+
parts.append(f"<p>{render_atoms_to_html(rendered_atoms)}</p>")
|
|
891
|
+
for atom in atoms:
|
|
892
|
+
if isinstance(atom, InlineBlockGroup):
|
|
893
|
+
self._append_cell_blocks(
|
|
894
|
+
parts,
|
|
895
|
+
atom.blocks,
|
|
896
|
+
inline_image_rendered=atom.inline_image_rendered and not self._collect_cell_visuals,
|
|
897
|
+
)
|
|
898
|
+
elif child.tag == qname("text", "list"):
|
|
899
|
+
parts.append(self._render_list_html(child))
|
|
900
|
+
elif child.tag == qname("table", "table"):
|
|
901
|
+
nested = parse_table_grid(child, self.render_cell_html, expansion_budget=self.table_expansion_budget)
|
|
902
|
+
parts.append(table_grid_to_html(nested))
|
|
903
|
+
elif child.tag == qname("draw", "frame"):
|
|
904
|
+
inline, blocks = self._parse_frame(child)
|
|
905
|
+
if inline is not None and not (self._collect_cell_visuals and isinstance(inline, InlineImage)):
|
|
906
|
+
parts.append(render_atoms_to_html([inline]))
|
|
907
|
+
self._append_cell_blocks(
|
|
908
|
+
parts,
|
|
909
|
+
blocks,
|
|
910
|
+
inline_image_rendered=isinstance(inline, InlineImage) and not self._collect_cell_visuals,
|
|
911
|
+
)
|
|
912
|
+
return "".join(part for part in parts if part)
|
|
913
|
+
|
|
914
|
+
def _render_list_html(self, element: etree._Element, *, depth: int = 0, inherited_style: str | None = None) -> str:
|
|
915
|
+
"""把单元格内 ODF 列表递归渲染为 ol/ul HTML。"""
|
|
916
|
+
style_name = element.get(qname("text", "style-name")) or inherited_style
|
|
917
|
+
level = self.styles.list_level(style_name, depth)
|
|
918
|
+
tag = "ol" if level.ordered else "ul"
|
|
919
|
+
start = f' start="{level.start}"' if level.ordered and level.start != 1 else ""
|
|
920
|
+
parts = [f"<{tag}{start}>"]
|
|
921
|
+
for item in element:
|
|
922
|
+
if item.tag not in {qname("text", "list-item"), qname("text", "list-header")}:
|
|
923
|
+
continue
|
|
924
|
+
if item.tag == qname("text", "list-header"):
|
|
925
|
+
for child in item:
|
|
926
|
+
if child.tag in {qname("text", "p"), qname("text", "h")}:
|
|
927
|
+
atoms = self.parse_inline_atoms(child)
|
|
928
|
+
self._queue_inline_notes(atoms)
|
|
929
|
+
parts.append(f"<li>{render_atoms_to_html(atoms)}</li>")
|
|
930
|
+
continue
|
|
931
|
+
parts.append("<li>")
|
|
932
|
+
for child in item:
|
|
933
|
+
if child.tag in {qname("text", "p"), qname("text", "h")}:
|
|
934
|
+
atoms = self.parse_inline_atoms(child)
|
|
935
|
+
self._queue_inline_notes(atoms)
|
|
936
|
+
parts.append(render_atoms_to_html(atoms))
|
|
937
|
+
elif child.tag == qname("text", "list"):
|
|
938
|
+
parts.append(self._render_list_html(child, depth=depth + 1, inherited_style=style_name))
|
|
939
|
+
parts.append("</li>")
|
|
940
|
+
parts.append(f"</{tag}>")
|
|
941
|
+
return "".join(parts)
|
|
942
|
+
|
|
943
|
+
def drain_notes(self) -> list[str]:
|
|
944
|
+
"""取出当前累计脚注并清空共享队列。"""
|
|
945
|
+
values = list(self.notes)
|
|
946
|
+
self.notes.clear()
|
|
947
|
+
return values
|
|
948
|
+
|
|
949
|
+
def drain_cell_visuals(self) -> list[dict[str, Any]]:
|
|
950
|
+
"""取出 ODS 单元格解析期间收集的视觉对象并清空队列。"""
|
|
951
|
+
values = list(self._cell_visuals)
|
|
952
|
+
self._cell_visuals.clear()
|
|
953
|
+
return values
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
def _positive_space_count(value: str | None) -> int:
|
|
957
|
+
"""在整数转换前校验并限制 text:s 重复空格数,非法值按一处理。"""
|
|
958
|
+
normalized = (value or "").strip()
|
|
959
|
+
if normalized.startswith("+"):
|
|
960
|
+
normalized = normalized[1:]
|
|
961
|
+
if not normalized or not normalized.isascii() or not normalized.isdigit():
|
|
962
|
+
return 1
|
|
963
|
+
significant = normalized.lstrip("0")
|
|
964
|
+
if not significant:
|
|
965
|
+
return 1
|
|
966
|
+
max_digits = len(str(_MAX_EXPLICIT_SPACE_COUNT))
|
|
967
|
+
if len(significant) > max_digits:
|
|
968
|
+
return _MAX_EXPLICIT_SPACE_COUNT
|
|
969
|
+
return min(max(1, int(significant)), _MAX_EXPLICIT_SPACE_COUNT)
|
|
970
|
+
|
|
971
|
+
|
|
972
|
+
def flatten_block_text(blocks: list[dict[str, Any]]) -> str:
|
|
973
|
+
"""递归提取 raw block 的可见字符串,供标题和备注聚合。"""
|
|
974
|
+
parts: list[str] = []
|
|
975
|
+
for block in blocks:
|
|
976
|
+
content = block.get("content")
|
|
977
|
+
if isinstance(content, str):
|
|
978
|
+
if content.strip():
|
|
979
|
+
visible = re.sub(r"<[^>]+>", "", content)
|
|
980
|
+
parts.append(html.unescape(visible).strip())
|
|
981
|
+
elif isinstance(content, list):
|
|
982
|
+
children = [child for child in content if isinstance(child, dict)]
|
|
983
|
+
if children and all(
|
|
984
|
+
child.get("type") in {"text", "equation_inline", "code_inline", "hyperlink"} for child in children
|
|
985
|
+
):
|
|
986
|
+
nested = inline_span_plain_text(children)
|
|
987
|
+
else:
|
|
988
|
+
nested = flatten_block_text(children)
|
|
989
|
+
if nested:
|
|
990
|
+
parts.append(nested)
|
|
991
|
+
return "\n".join(parts)
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
__all__ = [
|
|
995
|
+
"OdfBlockParser",
|
|
996
|
+
"OdfMasterPageChange",
|
|
997
|
+
"OdfTextExpansionBudget",
|
|
998
|
+
"RawFlowItem",
|
|
999
|
+
"flatten_block_text",
|
|
1000
|
+
"render_atoms_to_html",
|
|
1001
|
+
"render_atoms_to_model",
|
|
1002
|
+
]
|