docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,928 @@
|
|
|
1
|
+
"""XLS 与 XLSX 复用的中立工作表投影器。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import collections
|
|
6
|
+
import html
|
|
7
|
+
from collections.abc import Iterator
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from loguru import logger
|
|
11
|
+
from openpyxl.cell.rich_text import CellRichText
|
|
12
|
+
from openpyxl.workbook.workbook import Workbook
|
|
13
|
+
from openpyxl.worksheet.worksheet import Worksheet
|
|
14
|
+
|
|
15
|
+
from .....schema import BlockType
|
|
16
|
+
from ..._shared.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
|
|
17
|
+
from .....content.spans import text_spans
|
|
18
|
+
from .html import EQUATION_BOOKENDS, render_spreadsheet_table
|
|
19
|
+
from .models import AnchoredBlock, DataRegion, ExcelCell, ExcelTable, FormulaMap, SheetImage
|
|
20
|
+
|
|
21
|
+
AUTO_GAP_TOLERANCE_CANDIDATES = (0, 1, 2)
|
|
22
|
+
AUTO_GAP_TOLERANCE_PREFERENCE = {1: 0, 0: 1, 2: 2}
|
|
23
|
+
AUTO_GAP_TOLERANCE_PREFERENCE_MARGIN = 0.15
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class _MergedCellLookup:
|
|
27
|
+
"""按行缓存合并单元格范围,避免解析时反复扫描 openpyxl 合并区域。"""
|
|
28
|
+
|
|
29
|
+
def __init__(self, sheet: Worksheet):
|
|
30
|
+
"""从工作表合并区域构建 0-based 坐标索引。"""
|
|
31
|
+
self._merged_row_intervals: dict[int, list[tuple[int, int]]] = collections.defaultdict(list)
|
|
32
|
+
self._hidden_row_intervals: dict[int, list[tuple[int, int]]] = collections.defaultdict(list)
|
|
33
|
+
self._anchor_spans: dict[tuple[int, int], tuple[int, int]] = {}
|
|
34
|
+
|
|
35
|
+
for merged in sheet.merged_cells.ranges:
|
|
36
|
+
min_row = merged.min_row - 1
|
|
37
|
+
max_row = merged.max_row - 1
|
|
38
|
+
min_col = merged.min_col - 1
|
|
39
|
+
max_col = merged.max_col - 1
|
|
40
|
+
|
|
41
|
+
self._anchor_spans[(min_row, min_col)] = (
|
|
42
|
+
max_row - min_row + 1,
|
|
43
|
+
max_col - min_col + 1,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
for row in range(min_row, max_row + 1):
|
|
47
|
+
self._merged_row_intervals[row].append((min_col, max_col))
|
|
48
|
+
hidden_start_col = min_col + 1 if row == min_row else min_col
|
|
49
|
+
if hidden_start_col <= max_col:
|
|
50
|
+
self._hidden_row_intervals[row].append((hidden_start_col, max_col))
|
|
51
|
+
|
|
52
|
+
for intervals in self._merged_row_intervals.values():
|
|
53
|
+
intervals.sort()
|
|
54
|
+
for intervals in self._hidden_row_intervals.values():
|
|
55
|
+
intervals.sort()
|
|
56
|
+
|
|
57
|
+
@staticmethod
|
|
58
|
+
def _contains_interval(
|
|
59
|
+
row_intervals: dict[int, list[tuple[int, int]]],
|
|
60
|
+
row: int,
|
|
61
|
+
col: int,
|
|
62
|
+
) -> bool:
|
|
63
|
+
"""判断 0-based 坐标是否落入指定行的任一列区间。"""
|
|
64
|
+
for start_col, end_col in row_intervals.get(row, []):
|
|
65
|
+
if start_col <= col <= end_col:
|
|
66
|
+
return True
|
|
67
|
+
if start_col > col:
|
|
68
|
+
break
|
|
69
|
+
return False
|
|
70
|
+
|
|
71
|
+
def contains_merged_cell(self, row: int, col: int) -> bool:
|
|
72
|
+
"""判断 0-based 坐标是否属于任一合并区域。"""
|
|
73
|
+
return self._contains_interval(self._merged_row_intervals, row, col)
|
|
74
|
+
|
|
75
|
+
def is_hidden_merged_cell(self, row: int, col: int) -> bool:
|
|
76
|
+
"""判断 0-based 坐标是否为合并区域内非左上角的隐藏格。"""
|
|
77
|
+
return self._contains_interval(self._hidden_row_intervals, row, col)
|
|
78
|
+
|
|
79
|
+
def get_anchor_span(self, row: int, col: int) -> tuple[int, int]:
|
|
80
|
+
"""返回合并区域左上角坐标对应的 rowspan/colspan,非合并锚点返回 1x1。"""
|
|
81
|
+
return self._anchor_spans.get((row, col), (1, 1))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class SpreadsheetProjector:
|
|
85
|
+
"""把 openpyxl 工作表投影为稳定的分页 model-list。"""
|
|
86
|
+
|
|
87
|
+
def __init__(
|
|
88
|
+
self,
|
|
89
|
+
*,
|
|
90
|
+
treat_singleton_as_text: bool = True,
|
|
91
|
+
gap_tolerance: int | None = None,
|
|
92
|
+
include_hidden_sheets: bool = False,
|
|
93
|
+
) -> None:
|
|
94
|
+
"""保存工作表投影配置并初始化无格式专属依赖的运行状态。"""
|
|
95
|
+
self.treat_singleton_as_text = treat_singleton_as_text
|
|
96
|
+
self.gap_tolerance = gap_tolerance
|
|
97
|
+
self.include_hidden_sheets = include_hidden_sheets
|
|
98
|
+
self._reset_projection_state()
|
|
99
|
+
|
|
100
|
+
def _reset_projection_state(self) -> None:
|
|
101
|
+
"""重置工作簿、分页和逐 sheet 的共享投影状态。"""
|
|
102
|
+
self.workbook: Workbook | None = None
|
|
103
|
+
self.pages: list[list[dict[str, Any]]] = []
|
|
104
|
+
self.cur_page: list[dict[str, Any]] = []
|
|
105
|
+
self.math_map: FormulaMap = {}
|
|
106
|
+
self.sheet_images: list[SheetImage] = []
|
|
107
|
+
self.table_image_map: dict[tuple[int, int], list[str]] = collections.defaultdict(list)
|
|
108
|
+
self._merged_cell_lookup_cache: dict[int, _MergedCellLookup] = {}
|
|
109
|
+
|
|
110
|
+
def _prepare_sheet_assets(self, sheet: Worksheet) -> None:
|
|
111
|
+
"""准备普通公式和图片,并构建表格单元格使用的媒体映射。"""
|
|
112
|
+
self.math_map = self._map_math_formulas_to_cells(sheet)
|
|
113
|
+
self.sheet_images = self._collect_sheet_images(sheet)
|
|
114
|
+
self.table_image_map = collections.defaultdict(list)
|
|
115
|
+
for image in self.sheet_images:
|
|
116
|
+
row, col = image.anchor
|
|
117
|
+
if row is None or col is None:
|
|
118
|
+
continue
|
|
119
|
+
if image.latex:
|
|
120
|
+
self.table_image_map[(row, col)].append(EQUATION_BOOKENDS.format(EQ=image.latex))
|
|
121
|
+
elif image.image_base64:
|
|
122
|
+
self.table_image_map[(row, col)].append(f'<img src="{image.image_base64}" />')
|
|
123
|
+
|
|
124
|
+
def _convert_sheet(self, sheet: Worksheet) -> None:
|
|
125
|
+
"""按表格、图表、附加素材和独立图片的稳定顺序投影一个工作表。"""
|
|
126
|
+
self._prepare_sheet_assets(sheet)
|
|
127
|
+
used_cells, visual_artifacts = self._find_tables_in_sheet(sheet)
|
|
128
|
+
visual_artifacts.extend(self._find_charts_in_sheet(sheet))
|
|
129
|
+
visual_artifacts.extend(self._find_additional_visual_artifacts(used_cells))
|
|
130
|
+
for _, _, block in sorted(
|
|
131
|
+
visual_artifacts,
|
|
132
|
+
key=lambda item: (item[0][0], item[0][1], item[1]),
|
|
133
|
+
):
|
|
134
|
+
self.cur_page.append(block)
|
|
135
|
+
self._find_images_in_sheet(used_cells)
|
|
136
|
+
|
|
137
|
+
def _map_math_formulas_to_cells(self, sheet: Worksheet) -> FormulaMap:
|
|
138
|
+
"""返回当前工作表按 0-based cell anchor 分组的公式。"""
|
|
139
|
+
return {}
|
|
140
|
+
|
|
141
|
+
def _collect_sheet_images(self, sheet: Worksheet) -> list[SheetImage]:
|
|
142
|
+
"""返回当前工作表按 anchor 排序的图片或图片公式。"""
|
|
143
|
+
return []
|
|
144
|
+
|
|
145
|
+
def _find_charts_in_sheet(self, sheet: Worksheet) -> list[AnchoredBlock]:
|
|
146
|
+
"""返回当前工作表的格式专属图表 blocks。"""
|
|
147
|
+
return []
|
|
148
|
+
|
|
149
|
+
def _find_additional_visual_artifacts(
|
|
150
|
+
self,
|
|
151
|
+
used_cells: set[tuple[int, int]],
|
|
152
|
+
) -> list[AnchoredBlock]:
|
|
153
|
+
"""返回未被表格吸收的格式专属公式或图片 blocks。"""
|
|
154
|
+
return []
|
|
155
|
+
|
|
156
|
+
def _resolve_cell_image(self, raw_cell_text: str) -> str:
|
|
157
|
+
"""解析格式专属的单元格图片函数,默认不产生媒体。"""
|
|
158
|
+
return ""
|
|
159
|
+
|
|
160
|
+
def _iter_sheets_to_convert(self) -> Iterator[Worksheet]:
|
|
161
|
+
"""按工作簿顺序遍历允许输出的可见工作表。"""
|
|
162
|
+
if self.workbook is None:
|
|
163
|
+
return
|
|
164
|
+
|
|
165
|
+
for sheet in self.workbook.worksheets:
|
|
166
|
+
if not self.include_hidden_sheets and sheet.sheet_state != Worksheet.SHEETSTATE_VISIBLE:
|
|
167
|
+
logger.debug(f"跳过隐藏工作表:{sheet.title}")
|
|
168
|
+
continue
|
|
169
|
+
yield sheet
|
|
170
|
+
|
|
171
|
+
@staticmethod
|
|
172
|
+
def _build_sheet_title_block(sheet_title: str) -> dict:
|
|
173
|
+
"""构造工作表标题块,复用 Office 标题渲染链路输出 Markdown 标题。"""
|
|
174
|
+
return {
|
|
175
|
+
"type": BlockType.PARAGRAPH_TITLE,
|
|
176
|
+
"level": 2,
|
|
177
|
+
"content": text_spans(sheet_title),
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
@staticmethod
|
|
181
|
+
def _should_emit_sheet_titles(pages: list[list[dict]]) -> bool:
|
|
182
|
+
"""仅当存在多个非空输出 sheet 时才添加标题,避免单表或空表噪声。"""
|
|
183
|
+
return sum(1 for page in pages if page) > 1
|
|
184
|
+
|
|
185
|
+
def _prepend_sheet_titles(self, sheet_pages: list[tuple[str, list[dict]]]) -> None:
|
|
186
|
+
"""将 sheet 标题插入每个非空 page 开头,不参与表格/图表视觉排序。"""
|
|
187
|
+
for sheet_title, page in sheet_pages:
|
|
188
|
+
if not page:
|
|
189
|
+
continue
|
|
190
|
+
page.insert(0, self._build_sheet_title_block(sheet_title))
|
|
191
|
+
|
|
192
|
+
def _get_block_sort_anchor(self, row: int | None, col: int | None) -> tuple[int, int]:
|
|
193
|
+
"""把缺失 anchor 稳定放到全部有效工作表坐标之后。"""
|
|
194
|
+
if row is None or col is None:
|
|
195
|
+
return (10**9, 10**9)
|
|
196
|
+
return row, col
|
|
197
|
+
|
|
198
|
+
def _build_block_from_excel_table(self, excel_table: ExcelTable) -> dict:
|
|
199
|
+
"""按 singleton 规则把表格 IR 投影为文本或表格 block。"""
|
|
200
|
+
if self.treat_singleton_as_text and len(excel_table.data) == 1 and self._can_render_singleton_as_text(excel_table):
|
|
201
|
+
return {
|
|
202
|
+
"type": BlockType.TEXT,
|
|
203
|
+
"content": text_spans(excel_table.data[0].text),
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
return {
|
|
207
|
+
"type": BlockType.TABLE,
|
|
208
|
+
"content": render_spreadsheet_table(excel_table),
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
def _find_tables_in_sheet(self, sheet: Worksheet) -> tuple[set[tuple[int, int]], list[tuple[tuple[int, int], int, dict]]]:
|
|
212
|
+
"""发现当前 sheet 表格并返回已吸收 cell 与锚定 blocks。"""
|
|
213
|
+
used_cells = set()
|
|
214
|
+
visual_artifacts = []
|
|
215
|
+
if self.workbook is not None:
|
|
216
|
+
tables = self._find_data_tables(sheet) # 检测工作表中的所有数据表格
|
|
217
|
+
|
|
218
|
+
for order, excel_table in enumerate(tables):
|
|
219
|
+
# Record used cells
|
|
220
|
+
anchor_c, anchor_r = excel_table.anchor
|
|
221
|
+
for cell in excel_table.data:
|
|
222
|
+
source_row, source_col = self._resolve_excel_cell_source_position(
|
|
223
|
+
excel_table.anchor,
|
|
224
|
+
cell,
|
|
225
|
+
)
|
|
226
|
+
used_cells.add((source_row, source_col))
|
|
227
|
+
|
|
228
|
+
visual_artifacts.append(
|
|
229
|
+
(
|
|
230
|
+
self._get_block_sort_anchor(anchor_r, anchor_c),
|
|
231
|
+
order,
|
|
232
|
+
self._build_block_from_excel_table(excel_table),
|
|
233
|
+
)
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
return used_cells, visual_artifacts
|
|
237
|
+
|
|
238
|
+
def _build_excel_cell(
|
|
239
|
+
self,
|
|
240
|
+
sheet: Worksheet,
|
|
241
|
+
display_row: int,
|
|
242
|
+
display_col: int,
|
|
243
|
+
source_row: int,
|
|
244
|
+
source_col: int,
|
|
245
|
+
row_span: int = 1,
|
|
246
|
+
col_span: int = 1,
|
|
247
|
+
) -> ExcelCell:
|
|
248
|
+
"""把源工作表单元格完整物化为中立 ExcelCell。"""
|
|
249
|
+
cell = sheet.cell(row=source_row + 1, column=source_col + 1)
|
|
250
|
+
raw_cell_text = str(cell.value) if cell.value is not None else ""
|
|
251
|
+
cell_text = ""
|
|
252
|
+
text_is_html = False
|
|
253
|
+
media_content = []
|
|
254
|
+
if "DISPIMG" in raw_cell_text:
|
|
255
|
+
cell_image = self._resolve_cell_image(raw_cell_text)
|
|
256
|
+
if cell_image:
|
|
257
|
+
media_content.append(cell_image)
|
|
258
|
+
else:
|
|
259
|
+
cell_text, text_is_html = self._cell_value_to_html(cell)
|
|
260
|
+
media_content.extend(self.table_image_map.get((source_row, source_col), []))
|
|
261
|
+
|
|
262
|
+
return ExcelCell(
|
|
263
|
+
row=display_row,
|
|
264
|
+
col=display_col,
|
|
265
|
+
text=cell_text,
|
|
266
|
+
row_span=row_span,
|
|
267
|
+
col_span=col_span,
|
|
268
|
+
styles=self._extract_cell_style(cell),
|
|
269
|
+
media=media_content,
|
|
270
|
+
equations=list(self.math_map.get((source_row, source_col), [])),
|
|
271
|
+
text_is_html=text_is_html,
|
|
272
|
+
source_row=source_row,
|
|
273
|
+
source_col=source_col,
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
def _build_synthetic_table_from_sheet_selection(self, sheet: Worksheet, rows: list[int], cols: list[int]) -> ExcelTable:
|
|
277
|
+
"""把指定源行列选择物化为紧凑的表格 IR。"""
|
|
278
|
+
selected_coords = {(row, col) for row in rows for col in cols}
|
|
279
|
+
hidden_merge_cells = set()
|
|
280
|
+
merge_spans = {}
|
|
281
|
+
|
|
282
|
+
for mr in sheet.merged_cells.ranges:
|
|
283
|
+
top_left = (mr.min_row - 1, mr.min_col - 1)
|
|
284
|
+
if top_left not in selected_coords:
|
|
285
|
+
continue
|
|
286
|
+
|
|
287
|
+
selected_rows = [row for row in rows if mr.min_row - 1 <= row <= mr.max_row - 1]
|
|
288
|
+
selected_cols = [col for col in cols if mr.min_col - 1 <= col <= mr.max_col - 1]
|
|
289
|
+
if not selected_rows or not selected_cols:
|
|
290
|
+
continue
|
|
291
|
+
|
|
292
|
+
merge_spans[top_left] = (len(selected_rows), len(selected_cols))
|
|
293
|
+
for row in selected_rows:
|
|
294
|
+
for col in selected_cols:
|
|
295
|
+
if (row, col) != top_left:
|
|
296
|
+
hidden_merge_cells.add((row, col))
|
|
297
|
+
|
|
298
|
+
data = []
|
|
299
|
+
for display_row, source_row in enumerate(rows):
|
|
300
|
+
for display_col, source_col in enumerate(cols):
|
|
301
|
+
if (source_row, source_col) in hidden_merge_cells:
|
|
302
|
+
continue
|
|
303
|
+
|
|
304
|
+
row_span, col_span = merge_spans.get((source_row, source_col), (1, 1))
|
|
305
|
+
data.append(
|
|
306
|
+
self._build_excel_cell(
|
|
307
|
+
sheet,
|
|
308
|
+
display_row,
|
|
309
|
+
display_col,
|
|
310
|
+
source_row,
|
|
311
|
+
source_col,
|
|
312
|
+
row_span=row_span,
|
|
313
|
+
col_span=col_span,
|
|
314
|
+
)
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
return ExcelTable(
|
|
318
|
+
anchor=(cols[0], rows[0]),
|
|
319
|
+
num_rows=len(rows),
|
|
320
|
+
num_cols=len(cols),
|
|
321
|
+
data=data,
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
def _resolve_excel_cell_source_position(
|
|
325
|
+
self,
|
|
326
|
+
table_anchor: tuple[int, int],
|
|
327
|
+
excel_cell: ExcelCell | None,
|
|
328
|
+
row: int | None = None,
|
|
329
|
+
col: int | None = None,
|
|
330
|
+
) -> tuple[int, int]:
|
|
331
|
+
"""优先使用显式源坐标,否则通过表格 anchor 还原源坐标。"""
|
|
332
|
+
if excel_cell is not None:
|
|
333
|
+
if excel_cell.source_row is not None and excel_cell.source_col is not None:
|
|
334
|
+
return excel_cell.source_row, excel_cell.source_col
|
|
335
|
+
row = excel_cell.row
|
|
336
|
+
col = excel_cell.col
|
|
337
|
+
|
|
338
|
+
if row is None or col is None:
|
|
339
|
+
raise ValueError("row and col must be provided when excel_cell is None")
|
|
340
|
+
|
|
341
|
+
return table_anchor[1] + row, table_anchor[0] + col
|
|
342
|
+
|
|
343
|
+
def _can_render_singleton_as_text(self, excel_table: ExcelTable) -> bool:
|
|
344
|
+
"""判断单格表是否可安全降级为普通文本 block。"""
|
|
345
|
+
cell = excel_table.data[0]
|
|
346
|
+
return cell.row_span == 1 and cell.col_span == 1 and not cell.media and not cell.text_is_html and not cell.equations
|
|
347
|
+
|
|
348
|
+
def _cell_has_semantic_content(self, excel_table: ExcelTable, cell: ExcelCell) -> bool:
|
|
349
|
+
"""判断单元格是否包含文本、媒体或公式语义。"""
|
|
350
|
+
return bool(cell.text.strip() or any(media.strip() for media in cell.media) or cell.equations)
|
|
351
|
+
|
|
352
|
+
def _get_table_semantic_positions(self, excel_table: ExcelTable) -> set[tuple[int, int]]:
|
|
353
|
+
"""返回表格内具有语义内容的源工作表坐标。"""
|
|
354
|
+
semantic_positions = set()
|
|
355
|
+
for cell in excel_table.data:
|
|
356
|
+
if not self._cell_has_semantic_content(excel_table, cell):
|
|
357
|
+
continue
|
|
358
|
+
semantic_positions.add(
|
|
359
|
+
self._resolve_excel_cell_source_position(
|
|
360
|
+
excel_table.anchor,
|
|
361
|
+
excel_cell=cell,
|
|
362
|
+
)
|
|
363
|
+
)
|
|
364
|
+
return semantic_positions
|
|
365
|
+
|
|
366
|
+
def _filter_semantic_subset_tables(self, tables: list[ExcelTable]) -> list[ExcelTable]:
|
|
367
|
+
"""删除语义坐标严格包含于其它候选的重复表格。"""
|
|
368
|
+
semantic_positions = [self._get_table_semantic_positions(table) for table in tables]
|
|
369
|
+
filtered_tables = []
|
|
370
|
+
|
|
371
|
+
for table_idx, table in enumerate(tables):
|
|
372
|
+
if any(
|
|
373
|
+
semantic_positions[table_idx] < semantic_positions[other_idx]
|
|
374
|
+
for other_idx in range(len(tables))
|
|
375
|
+
if other_idx != table_idx
|
|
376
|
+
):
|
|
377
|
+
continue
|
|
378
|
+
filtered_tables.append(table)
|
|
379
|
+
|
|
380
|
+
return filtered_tables
|
|
381
|
+
|
|
382
|
+
def _build_table_content_mask(self, excel_table: ExcelTable) -> list[list[bool]]:
|
|
383
|
+
"""构造包含合并跨度的表格语义内容掩码。"""
|
|
384
|
+
mask = [[False for _ in range(excel_table.num_cols)] for _ in range(excel_table.num_rows)]
|
|
385
|
+
for cell in excel_table.data:
|
|
386
|
+
if not self._cell_has_semantic_content(excel_table, cell):
|
|
387
|
+
continue
|
|
388
|
+
for row_idx in range(cell.row, min(cell.row + cell.row_span, excel_table.num_rows)):
|
|
389
|
+
for col_idx in range(cell.col, min(cell.col + cell.col_span, excel_table.num_cols)):
|
|
390
|
+
mask[row_idx][col_idx] = True
|
|
391
|
+
return mask
|
|
392
|
+
|
|
393
|
+
@staticmethod
|
|
394
|
+
def _count_max_consecutive_true(flags: list[bool]) -> int:
|
|
395
|
+
"""返回布尔序列中最长连续真值长度。"""
|
|
396
|
+
max_count = 0
|
|
397
|
+
current = 0
|
|
398
|
+
for flag in flags:
|
|
399
|
+
if flag:
|
|
400
|
+
current += 1
|
|
401
|
+
max_count = max(max_count, current)
|
|
402
|
+
else:
|
|
403
|
+
current = 0
|
|
404
|
+
return max_count
|
|
405
|
+
|
|
406
|
+
@staticmethod
|
|
407
|
+
def _is_real_singleton_table(excel_table: ExcelTable) -> bool:
|
|
408
|
+
"""判断候选是否是单格且不可进一步拆分的真实表格。"""
|
|
409
|
+
if excel_table.num_rows != 1 or excel_table.num_cols != 1 or len(excel_table.data) != 1:
|
|
410
|
+
return False
|
|
411
|
+
cell = excel_table.data[0]
|
|
412
|
+
return cell.row_span == 1 and cell.col_span == 1
|
|
413
|
+
|
|
414
|
+
def _summarize_table_for_gap_selection(self, excel_table: ExcelTable) -> dict[str, float | int | bool]:
|
|
415
|
+
"""计算 gap 候选评分使用的单表形态指标。"""
|
|
416
|
+
table_area = excel_table.num_rows * excel_table.num_cols
|
|
417
|
+
content_mask = self._build_table_content_mask(excel_table)
|
|
418
|
+
content_area = sum(sum(1 for flag in row if flag) for row in content_mask)
|
|
419
|
+
blank_ratio = 1.0 - (content_area / max(table_area, 1))
|
|
420
|
+
|
|
421
|
+
interior_blank_rows = [not any(content_mask[row_idx]) for row_idx in range(1, max(excel_table.num_rows - 1, 1))]
|
|
422
|
+
interior_blank_cols = [
|
|
423
|
+
not any(content_mask[row_idx][col_idx] for row_idx in range(excel_table.num_rows))
|
|
424
|
+
for col_idx in range(1, max(excel_table.num_cols - 1, 1))
|
|
425
|
+
]
|
|
426
|
+
if excel_table.num_rows <= 2:
|
|
427
|
+
interior_blank_rows = []
|
|
428
|
+
if excel_table.num_cols <= 2:
|
|
429
|
+
interior_blank_cols = []
|
|
430
|
+
|
|
431
|
+
interior_blank_row_count = sum(interior_blank_rows)
|
|
432
|
+
interior_blank_col_count = sum(interior_blank_cols)
|
|
433
|
+
max_consecutive_interior_blank_lines = max(
|
|
434
|
+
self._count_max_consecutive_true(interior_blank_rows),
|
|
435
|
+
self._count_max_consecutive_true(interior_blank_cols),
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
return {
|
|
439
|
+
"table_area": table_area,
|
|
440
|
+
"content_area": content_area,
|
|
441
|
+
"blank_ratio": blank_ratio,
|
|
442
|
+
"interior_blank_row_count": interior_blank_row_count,
|
|
443
|
+
"interior_blank_col_count": interior_blank_col_count,
|
|
444
|
+
"max_consecutive_interior_blank_lines": max_consecutive_interior_blank_lines,
|
|
445
|
+
"real_singleton": self._is_real_singleton_table(excel_table),
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
def _summarize_candidate_tables(self, tables: list[ExcelTable]) -> dict[str, float | int]:
|
|
449
|
+
"""汇总一组 gap 候选表格的惩罚指标。"""
|
|
450
|
+
table_count = len(tables)
|
|
451
|
+
real_singleton_count = 0
|
|
452
|
+
severe_separator_count = 0
|
|
453
|
+
sparse_large_table_count = 0
|
|
454
|
+
total_area = 0
|
|
455
|
+
weighted_blank_numerator = 0.0
|
|
456
|
+
total_interior_blank_lines = 0
|
|
457
|
+
total_possible_interior_lines = 0
|
|
458
|
+
row_cover_count = collections.Counter()
|
|
459
|
+
|
|
460
|
+
for table in tables:
|
|
461
|
+
table_summary = self._summarize_table_for_gap_selection(table)
|
|
462
|
+
table_area = int(table_summary["table_area"])
|
|
463
|
+
blank_ratio = float(table_summary["blank_ratio"])
|
|
464
|
+
interior_blank_row_count = int(table_summary["interior_blank_row_count"])
|
|
465
|
+
interior_blank_col_count = int(table_summary["interior_blank_col_count"])
|
|
466
|
+
max_consecutive_interior_blank_lines = int(table_summary["max_consecutive_interior_blank_lines"])
|
|
467
|
+
|
|
468
|
+
total_area += table_area
|
|
469
|
+
weighted_blank_numerator += table_area * blank_ratio
|
|
470
|
+
total_interior_blank_lines += interior_blank_row_count + interior_blank_col_count
|
|
471
|
+
total_possible_interior_lines += max(table.num_rows - 2, 0) + max(table.num_cols - 2, 0)
|
|
472
|
+
for row_idx in range(table.anchor[1], table.anchor[1] + table.num_rows):
|
|
473
|
+
row_cover_count[row_idx] += 1
|
|
474
|
+
|
|
475
|
+
if bool(table_summary["real_singleton"]):
|
|
476
|
+
real_singleton_count += 1
|
|
477
|
+
if table_area >= 6 and blank_ratio > 0.35:
|
|
478
|
+
sparse_large_table_count += 1
|
|
479
|
+
if max_consecutive_interior_blank_lines >= 2:
|
|
480
|
+
severe_separator_count += 1
|
|
481
|
+
|
|
482
|
+
occupied_row_count = max(len(row_cover_count), 1)
|
|
483
|
+
row_overlap_excess_ratio = sum(max(0, count - 1) for count in row_cover_count.values()) / occupied_row_count
|
|
484
|
+
|
|
485
|
+
return {
|
|
486
|
+
"real_singleton_ratio": real_singleton_count / max(table_count, 1),
|
|
487
|
+
"weighted_blank_ratio": weighted_blank_numerator / max(total_area, 1),
|
|
488
|
+
"interior_blank_line_ratio": total_interior_blank_lines / max(total_possible_interior_lines, 1),
|
|
489
|
+
"sparse_large_table_ratio": sparse_large_table_count / max(table_count, 1),
|
|
490
|
+
"severe_separator_count": severe_separator_count,
|
|
491
|
+
"row_overlap_excess_ratio": row_overlap_excess_ratio,
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
def _select_best_gap_candidate(self, sheet: Worksheet) -> tuple[int, float, list[ExcelTable]]:
|
|
495
|
+
"""按固定候选与偏好顺序选择最稳定的 gap tolerance。"""
|
|
496
|
+
candidates = []
|
|
497
|
+
for gap_tolerance in AUTO_GAP_TOLERANCE_CANDIDATES:
|
|
498
|
+
raw_tables = self._find_data_tables_with_gap_raw(sheet, gap_tolerance)
|
|
499
|
+
summary = self._summarize_candidate_tables(raw_tables)
|
|
500
|
+
penalty = (
|
|
501
|
+
6.0 * int(summary["severe_separator_count"])
|
|
502
|
+
+ 2.5 * float(summary["interior_blank_line_ratio"])
|
|
503
|
+
+ 1.5 * float(summary["sparse_large_table_ratio"])
|
|
504
|
+
+ 1.0 * float(summary["real_singleton_ratio"])
|
|
505
|
+
+ 0.5 * float(summary["weighted_blank_ratio"])
|
|
506
|
+
+ 1.0 * float(summary["row_overlap_excess_ratio"])
|
|
507
|
+
)
|
|
508
|
+
candidates.append(
|
|
509
|
+
{
|
|
510
|
+
"gap_tolerance": gap_tolerance,
|
|
511
|
+
"penalty": penalty,
|
|
512
|
+
"tables": self._filter_semantic_subset_tables(raw_tables),
|
|
513
|
+
**summary,
|
|
514
|
+
}
|
|
515
|
+
)
|
|
516
|
+
|
|
517
|
+
min_penalty = min(float(candidate["penalty"]) for candidate in candidates)
|
|
518
|
+
near_best_candidates = [
|
|
519
|
+
candidate
|
|
520
|
+
for candidate in candidates
|
|
521
|
+
if float(candidate["penalty"]) <= (min_penalty + AUTO_GAP_TOLERANCE_PREFERENCE_MARGIN)
|
|
522
|
+
]
|
|
523
|
+
|
|
524
|
+
best_candidate = min(
|
|
525
|
+
near_best_candidates,
|
|
526
|
+
key=lambda candidate: (
|
|
527
|
+
int(candidate["severe_separator_count"]),
|
|
528
|
+
AUTO_GAP_TOLERANCE_PREFERENCE[int(candidate["gap_tolerance"])],
|
|
529
|
+
float(candidate["interior_blank_line_ratio"]),
|
|
530
|
+
float(candidate["penalty"]),
|
|
531
|
+
),
|
|
532
|
+
)
|
|
533
|
+
return (
|
|
534
|
+
int(best_candidate["gap_tolerance"]),
|
|
535
|
+
float(best_candidate["penalty"]),
|
|
536
|
+
best_candidate["tables"],
|
|
537
|
+
)
|
|
538
|
+
|
|
539
|
+
def _select_best_tables(self, sheet: Worksheet) -> list[ExcelTable]:
|
|
540
|
+
"""选择并记录当前工作表的最佳表格候选集合。"""
|
|
541
|
+
gap_tolerance, penalty, tables = self._select_best_gap_candidate(sheet)
|
|
542
|
+
logger.debug(
|
|
543
|
+
"Selected gap_tolerance={} for sheet '{}' with penalty={:.4f}",
|
|
544
|
+
gap_tolerance,
|
|
545
|
+
sheet.title,
|
|
546
|
+
penalty,
|
|
547
|
+
)
|
|
548
|
+
return tables
|
|
549
|
+
|
|
550
|
+
def _find_images_in_sheet(self, used_cells: set[tuple[int, int]] | None = None) -> None:
|
|
551
|
+
"""输出没有被表格吸收且不是公式载体的独立图片。"""
|
|
552
|
+
if self.workbook is not None:
|
|
553
|
+
for image in self.sheet_images:
|
|
554
|
+
r, c = image.anchor
|
|
555
|
+
if used_cells and r is not None and c is not None and (r, c) in used_cells:
|
|
556
|
+
continue
|
|
557
|
+
|
|
558
|
+
if image.latex:
|
|
559
|
+
continue
|
|
560
|
+
if image.image_base64:
|
|
561
|
+
self.cur_page.append(
|
|
562
|
+
{
|
|
563
|
+
"type": BlockType.IMAGE,
|
|
564
|
+
"image_base64": image.image_base64,
|
|
565
|
+
}
|
|
566
|
+
)
|
|
567
|
+
|
|
568
|
+
def _find_data_tables(self, sheet: Worksheet) -> list[ExcelTable]:
|
|
569
|
+
"""在 Excel 工作表中查找所有紧凑的矩形数据表格。
|
|
570
|
+
|
|
571
|
+
参数:
|
|
572
|
+
sheet: 待解析的 Excel 工作表。
|
|
573
|
+
|
|
574
|
+
返回:
|
|
575
|
+
表示所有数据表格的 ExcelTable 对象列表。
|
|
576
|
+
"""
|
|
577
|
+
if self.gap_tolerance is None:
|
|
578
|
+
return self._select_best_tables(sheet)
|
|
579
|
+
return self._find_data_tables_with_gap(sheet, self.gap_tolerance)
|
|
580
|
+
|
|
581
|
+
def _find_data_tables_with_gap(self, sheet: Worksheet, gap_tolerance: int) -> list[ExcelTable]:
|
|
582
|
+
"""按固定 gap 发现表格并移除语义子集候选。"""
|
|
583
|
+
return self._filter_semantic_subset_tables(self._find_data_tables_with_gap_raw(sheet, gap_tolerance))
|
|
584
|
+
|
|
585
|
+
def _find_data_tables_with_gap_raw(self, sheet: Worksheet, gap_tolerance: int) -> list[ExcelTable]:
|
|
586
|
+
"""在固定 gap_tolerance 下查找工作表中的所有数据表格。"""
|
|
587
|
+
bounds: DataRegion = self._find_true_data_bounds(sheet) # 获取真实数据边界
|
|
588
|
+
tables: list[ExcelTable] = [] # 存储已发现的表格
|
|
589
|
+
visited: set[tuple[int, int]] = set() # 记录已访问的单元格
|
|
590
|
+
|
|
591
|
+
# 仅遍历已存在且有值的单元格,避免 iter_rows 在稀疏大表上创建大量空单元格。
|
|
592
|
+
for ri, rj in self._get_non_empty_cell_positions(sheet, bounds):
|
|
593
|
+
# 跳过已访问的单元格
|
|
594
|
+
if (ri, rj) in visited:
|
|
595
|
+
continue
|
|
596
|
+
|
|
597
|
+
# 从当前单元格出发,通过洪水填充算法确定所属表格的边界
|
|
598
|
+
table_bounds, visited_cells = self._find_table_bounds(
|
|
599
|
+
sheet,
|
|
600
|
+
ri,
|
|
601
|
+
rj,
|
|
602
|
+
bounds.max_row,
|
|
603
|
+
bounds.max_col,
|
|
604
|
+
gap_tolerance,
|
|
605
|
+
)
|
|
606
|
+
visited.update(visited_cells) # 将已访问单元格加入全局记录
|
|
607
|
+
tables.append(table_bounds)
|
|
608
|
+
|
|
609
|
+
return tables
|
|
610
|
+
|
|
611
|
+
def _get_non_empty_cell_positions(
|
|
612
|
+
self,
|
|
613
|
+
sheet: Worksheet,
|
|
614
|
+
bounds: DataRegion,
|
|
615
|
+
) -> list[tuple[int, int]]:
|
|
616
|
+
"""按行列顺序返回真实边界内已有值单元格的 0-based 坐标。"""
|
|
617
|
+
positions = []
|
|
618
|
+
for cell in sheet._cells.values():
|
|
619
|
+
if cell.value is None:
|
|
620
|
+
continue
|
|
621
|
+
if not (bounds.min_row <= cell.row <= bounds.max_row and bounds.min_col <= cell.column <= bounds.max_col):
|
|
622
|
+
continue
|
|
623
|
+
positions.append((cell.row - 1, cell.column - 1))
|
|
624
|
+
return sorted(positions)
|
|
625
|
+
|
|
626
|
+
def _find_true_data_bounds(self, sheet: Worksheet) -> DataRegion:
|
|
627
|
+
"""查找工作表中真实的数据边界(最小/最大行列)。
|
|
628
|
+
|
|
629
|
+
该函数扫描所有单元格,找到包含所有非空单元格或合并单元格区域的
|
|
630
|
+
最小矩形范围,返回边界的行列索引。
|
|
631
|
+
|
|
632
|
+
参数:
|
|
633
|
+
sheet: 待分析的工作表。
|
|
634
|
+
|
|
635
|
+
返回:
|
|
636
|
+
覆盖所有数据和合并单元格的最小矩形区域 DataRegion。
|
|
637
|
+
若工作表为空,则默认返回 (1, 1, 1, 1)。
|
|
638
|
+
"""
|
|
639
|
+
min_row, min_col = None, None
|
|
640
|
+
max_row, max_col = 0, 0
|
|
641
|
+
|
|
642
|
+
# 遍历所有有值的单元格,动态更新边界
|
|
643
|
+
for cell in sheet._cells.values():
|
|
644
|
+
if cell.value is not None:
|
|
645
|
+
r, c = cell.row, cell.column
|
|
646
|
+
min_row = r if min_row is None else min(min_row, r)
|
|
647
|
+
min_col = c if min_col is None else min(min_col, c)
|
|
648
|
+
max_row = max(max_row, r)
|
|
649
|
+
max_col = max(max_col, c)
|
|
650
|
+
|
|
651
|
+
# 将合并单元格的范围也纳入边界计算
|
|
652
|
+
for merged in sheet.merged_cells.ranges:
|
|
653
|
+
min_row = merged.min_row if min_row is None else min(min_row, merged.min_row)
|
|
654
|
+
min_col = merged.min_col if min_col is None else min(min_col, merged.min_col)
|
|
655
|
+
max_row = max(max_row, merged.max_row)
|
|
656
|
+
max_col = max(max_col, merged.max_col)
|
|
657
|
+
|
|
658
|
+
# 若工作表中没有任何数据,默认返回 (1, 1, 1, 1)
|
|
659
|
+
if min_row is None or min_col is None:
|
|
660
|
+
min_row = min_col = max_row = max_col = 1
|
|
661
|
+
|
|
662
|
+
return DataRegion(min_row, max_row, min_col, max_col)
|
|
663
|
+
|
|
664
|
+
def _find_table_bounds(
|
|
665
|
+
self,
|
|
666
|
+
sheet: Worksheet,
|
|
667
|
+
start_row: int,
|
|
668
|
+
start_col: int,
|
|
669
|
+
max_row: int,
|
|
670
|
+
max_col: int,
|
|
671
|
+
gap_tolerance: int,
|
|
672
|
+
) -> tuple[ExcelTable, set[tuple[int, int]]]:
|
|
673
|
+
"""使用洪水填充(BFS)策略确定表格边界。
|
|
674
|
+
|
|
675
|
+
该方法通过广度优先搜索(BFS)算法识别 Excel 工作表中连续的非空单元格区域,
|
|
676
|
+
能够准确检测非矩形表格(如 L 形、错位列等),并支持通过间隔容忍度
|
|
677
|
+
连接相邻但不直接相连的单元格。
|
|
678
|
+
|
|
679
|
+
算法分两个阶段执行:
|
|
680
|
+
1. 洪水填充阶段:使用 BFS 从给定位置出发,找出所有相连的单元格。
|
|
681
|
+
2. 数据提取阶段:构建矩形边界框并提取单元格数据,正确处理合并单元格。
|
|
682
|
+
|
|
683
|
+
参数:
|
|
684
|
+
sheet: 待分析的 Excel 工作表。
|
|
685
|
+
start_row: 洪水填充起始行索引(从0开始)。
|
|
686
|
+
start_col: 洪水填充起始列索引(从0开始)。
|
|
687
|
+
max_row: 工作表中可考虑的最大行索引(从0开始)。
|
|
688
|
+
max_col: 工作表中可考虑的最大列索引(从0开始)。
|
|
689
|
+
gap_tolerance: 允许跨越空白单元格查找邻居的最大间隔。
|
|
690
|
+
|
|
691
|
+
返回:
|
|
692
|
+
一个元组,包含:
|
|
693
|
+
- ExcelTable:表示检测到的表格对象,含锚点位置、尺寸和单元格数据。
|
|
694
|
+
- set[tuple[int, int]]:洪水填充期间访问的所有 (行, 列) 元组集合,
|
|
695
|
+
用于防止重复扫描。
|
|
696
|
+
|
|
697
|
+
说明:
|
|
698
|
+
该方法遵循 GAP_TOLERANCE 选项,允许在容忍距离内将被空单元格隔开的
|
|
699
|
+
单元格视为同一表格的一部分。
|
|
700
|
+
"""
|
|
701
|
+
|
|
702
|
+
# BFS 队列,存储待处理的 (行, 列) 坐标
|
|
703
|
+
queue = collections.deque([(start_row, start_col)])
|
|
704
|
+
|
|
705
|
+
# 记录当前表格内已访问的单元格(避免重复加入队列)
|
|
706
|
+
# 调用方维护全局 visited 集合,防止重复启动新表格
|
|
707
|
+
table_cells: set[tuple[int, int]] = set()
|
|
708
|
+
table_cells.add((start_row, start_col))
|
|
709
|
+
|
|
710
|
+
# 动态记录当前表格的行列边界
|
|
711
|
+
min_r, max_r = start_row, start_row
|
|
712
|
+
min_c, max_c = start_col, start_col
|
|
713
|
+
merged_lookup = self._get_merged_cell_lookup(sheet)
|
|
714
|
+
|
|
715
|
+
def has_content(r: int, c: int) -> bool:
|
|
716
|
+
"""检查指定单元格(0-based索引)是否有内容(有值或属于合并区域)。"""
|
|
717
|
+
if r < 0 or c < 0 or r > max_row or c > max_col:
|
|
718
|
+
return False
|
|
719
|
+
|
|
720
|
+
# 1. 检查单元格直接值
|
|
721
|
+
cell = sheet._cells.get((r + 1, c + 1))
|
|
722
|
+
if cell is not None and cell.value is not None:
|
|
723
|
+
return True
|
|
724
|
+
|
|
725
|
+
# 2. 检查是否属于某个合并单元格区域
|
|
726
|
+
return merged_lookup.contains_merged_cell(r, c)
|
|
727
|
+
|
|
728
|
+
# --- 第一阶段:洪水填充(连通性检测)---
|
|
729
|
+
while queue:
|
|
730
|
+
curr_r, curr_c = queue.popleft()
|
|
731
|
+
|
|
732
|
+
# 动态更新表格边界
|
|
733
|
+
min_r = min(min_r, curr_r)
|
|
734
|
+
max_r = max(max_r, curr_r)
|
|
735
|
+
min_c = min(min_c, curr_c)
|
|
736
|
+
max_c = max(max_c, curr_c)
|
|
737
|
+
|
|
738
|
+
# 四个方向(上、下、左、右)的邻居检测
|
|
739
|
+
directions = [
|
|
740
|
+
(0, 1), # 右
|
|
741
|
+
(0, -1), # 左
|
|
742
|
+
(1, 0), # 下
|
|
743
|
+
(-1, 0), # 上
|
|
744
|
+
]
|
|
745
|
+
|
|
746
|
+
for dr, dc in directions:
|
|
747
|
+
# 在容忍距离范围内逐步检查邻居(优先检查最近的)
|
|
748
|
+
for step in range(1, gap_tolerance + 2):
|
|
749
|
+
nr, nc = curr_r + (dr * step), curr_c + (dc * step)
|
|
750
|
+
|
|
751
|
+
if (nr, nc) in table_cells:
|
|
752
|
+
break # 已属于当前表格,不跨越继续查找
|
|
753
|
+
|
|
754
|
+
if has_content(nr, nc):
|
|
755
|
+
table_cells.add((nr, nc))
|
|
756
|
+
queue.append((nr, nc))
|
|
757
|
+
# 在该方向找到连接点,停止扩展间隔
|
|
758
|
+
break
|
|
759
|
+
|
|
760
|
+
# --- 第二阶段:数据提取(语义网格构建)---
|
|
761
|
+
data = []
|
|
762
|
+
|
|
763
|
+
# 遍历发现区域的边界框(bbox内部的空格作为空单元格保留,维持矩形布局)
|
|
764
|
+
for ri in range(min_r, max_r + 1):
|
|
765
|
+
for rj in range(min_c, max_c + 1):
|
|
766
|
+
# 跳过被合并单元格遮蔽的单元格(非左上角)
|
|
767
|
+
if merged_lookup.is_hidden_merged_cell(ri, rj):
|
|
768
|
+
continue
|
|
769
|
+
|
|
770
|
+
# 计算合并跨度(默认为 1x1)
|
|
771
|
+
row_span, col_span = merged_lookup.get_anchor_span(ri, rj)
|
|
772
|
+
|
|
773
|
+
data.append(
|
|
774
|
+
self._build_excel_cell(
|
|
775
|
+
sheet,
|
|
776
|
+
ri - min_r, # 相对于表格起始行的偏移
|
|
777
|
+
rj - min_c, # 相对于表格起始列的偏移
|
|
778
|
+
ri,
|
|
779
|
+
rj,
|
|
780
|
+
row_span=row_span,
|
|
781
|
+
col_span=col_span,
|
|
782
|
+
)
|
|
783
|
+
)
|
|
784
|
+
|
|
785
|
+
# 返回给调用方的 visited_cells 严格为包含数据/合并的单元格,
|
|
786
|
+
# 使主循环不会重复扫描已处理的单元格。
|
|
787
|
+
return (
|
|
788
|
+
ExcelTable(
|
|
789
|
+
anchor=(min_c, min_r),
|
|
790
|
+
num_rows=max_r + 1 - min_r,
|
|
791
|
+
num_cols=max_c + 1 - min_c,
|
|
792
|
+
data=data,
|
|
793
|
+
),
|
|
794
|
+
table_cells,
|
|
795
|
+
)
|
|
796
|
+
|
|
797
|
+
def _get_merged_cell_lookup(self, sheet: Worksheet) -> _MergedCellLookup:
|
|
798
|
+
"""获取工作表合并单元格缓存,同一轮转换内每个 sheet 只构建一次。"""
|
|
799
|
+
cache_key = id(sheet)
|
|
800
|
+
lookup = self._merged_cell_lookup_cache.get(cache_key)
|
|
801
|
+
if lookup is None:
|
|
802
|
+
lookup = _MergedCellLookup(sheet)
|
|
803
|
+
self._merged_cell_lookup_cache[cache_key] = lookup
|
|
804
|
+
return lookup
|
|
805
|
+
|
|
806
|
+
@staticmethod
|
|
807
|
+
def _escape_text_with_line_breaks(text: str) -> str:
|
|
808
|
+
"""转义文本并把平台换行统一投影为 HTML 换行。"""
|
|
809
|
+
return html.escape(text).replace("\r\n", "\n").replace("\r", "\n").replace("\n", "<br>")
|
|
810
|
+
|
|
811
|
+
@staticmethod
|
|
812
|
+
def _get_cell_hyperlink_target(cell: Any) -> str:
|
|
813
|
+
"""读取单元格外链或工作簿内 location。"""
|
|
814
|
+
hyperlink = getattr(cell, "hyperlink", None)
|
|
815
|
+
if not hyperlink:
|
|
816
|
+
return ""
|
|
817
|
+
|
|
818
|
+
target = getattr(hyperlink, "target", None)
|
|
819
|
+
if target:
|
|
820
|
+
return str(target)
|
|
821
|
+
|
|
822
|
+
location = getattr(hyperlink, "location", None)
|
|
823
|
+
if location:
|
|
824
|
+
return f"#{location}"
|
|
825
|
+
|
|
826
|
+
return ""
|
|
827
|
+
|
|
828
|
+
@staticmethod
|
|
829
|
+
def _apply_inline_font_tags(text_html: str, inline_font: Any) -> str:
|
|
830
|
+
"""按 openpyxl 行内字体顺序包装可见 HTML 标签。"""
|
|
831
|
+
if not text_html or inline_font is None:
|
|
832
|
+
return text_html
|
|
833
|
+
|
|
834
|
+
wrapped = text_html
|
|
835
|
+
if getattr(inline_font, "strike", False) or getattr(inline_font, "u", None):
|
|
836
|
+
wrapped = wrapped.replace(" ", " ")
|
|
837
|
+
vert_align = getattr(inline_font, "vertAlign", None)
|
|
838
|
+
if vert_align == "superscript":
|
|
839
|
+
wrapped = f"<sup>{wrapped}</sup>"
|
|
840
|
+
elif vert_align == "subscript":
|
|
841
|
+
wrapped = f"<sub>{wrapped}</sub>"
|
|
842
|
+
|
|
843
|
+
if getattr(inline_font, "strike", False):
|
|
844
|
+
wrapped = f"<s>{wrapped}</s>"
|
|
845
|
+
if getattr(inline_font, "u", None):
|
|
846
|
+
wrapped = f"<u>{wrapped}</u>"
|
|
847
|
+
if getattr(inline_font, "i", False):
|
|
848
|
+
wrapped = f"<em>{wrapped}</em>"
|
|
849
|
+
if getattr(inline_font, "b", False):
|
|
850
|
+
wrapped = f"<strong>{wrapped}</strong>"
|
|
851
|
+
|
|
852
|
+
return wrapped
|
|
853
|
+
|
|
854
|
+
def _cell_value_to_html(self, cell: Any) -> tuple[str, bool]:
|
|
855
|
+
"""把普通或富文本单元格转换为安全 HTML 与内容类型标记。"""
|
|
856
|
+
if cell.value is None:
|
|
857
|
+
return "", False
|
|
858
|
+
|
|
859
|
+
safe_target = sanitize_hyperlink_target(
|
|
860
|
+
self._get_cell_hyperlink_target(cell),
|
|
861
|
+
allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
|
|
862
|
+
allow_relative=True,
|
|
863
|
+
allow_fragment=True,
|
|
864
|
+
)
|
|
865
|
+
link_target = html.escape(safe_target, quote=True) if safe_target else ""
|
|
866
|
+
|
|
867
|
+
if isinstance(cell.value, CellRichText):
|
|
868
|
+
html_parts = []
|
|
869
|
+
for part in cell.value:
|
|
870
|
+
if hasattr(part, "text"):
|
|
871
|
+
part_text = self._escape_text_with_line_breaks(str(getattr(part, "text", "")))
|
|
872
|
+
html_parts.append(
|
|
873
|
+
self._apply_inline_font_tags(
|
|
874
|
+
part_text,
|
|
875
|
+
getattr(part, "font", None),
|
|
876
|
+
)
|
|
877
|
+
)
|
|
878
|
+
else:
|
|
879
|
+
html_parts.append(self._escape_text_with_line_breaks(str(part)))
|
|
880
|
+
|
|
881
|
+
rich_text_html = "".join(html_parts)
|
|
882
|
+
if link_target and rich_text_html:
|
|
883
|
+
rich_text_html = f'<a href="{link_target}">{rich_text_html}</a>'
|
|
884
|
+
return rich_text_html, True
|
|
885
|
+
|
|
886
|
+
plain_text = str(cell.value)
|
|
887
|
+
if link_target and plain_text:
|
|
888
|
+
escaped_text = self._escape_text_with_line_breaks(plain_text)
|
|
889
|
+
return f'<a href="{link_target}">{escaped_text}</a>', True
|
|
890
|
+
|
|
891
|
+
return plain_text, False
|
|
892
|
+
|
|
893
|
+
def _extract_cell_style(self, cell: Any) -> dict[str, Any]:
|
|
894
|
+
"""从 openpyxl 单元格提取当前 IR 保留的可见样式。"""
|
|
895
|
+
style: dict[str, Any] = {}
|
|
896
|
+
if cell.font:
|
|
897
|
+
if cell.font.b:
|
|
898
|
+
style["font-weight"] = "bold"
|
|
899
|
+
if cell.font.i:
|
|
900
|
+
style["font-style"] = "italic"
|
|
901
|
+
if cell.font.u:
|
|
902
|
+
style["text-decoration"] = "underline"
|
|
903
|
+
if cell.font.strike:
|
|
904
|
+
style["text-decoration"] = "line-through"
|
|
905
|
+
if cell.font.color and hasattr(cell.font.color, "rgb") and cell.font.color.rgb:
|
|
906
|
+
# Color might be ARGB "FF000000"
|
|
907
|
+
color = cell.font.color.rgb
|
|
908
|
+
if isinstance(color, str) and len(color) == 8:
|
|
909
|
+
style["color"] = "#" + color[2:]
|
|
910
|
+
elif isinstance(color, str):
|
|
911
|
+
style["color"] = "#" + color
|
|
912
|
+
|
|
913
|
+
if cell.alignment:
|
|
914
|
+
if cell.alignment.horizontal:
|
|
915
|
+
style["text-align"] = cell.alignment.horizontal
|
|
916
|
+
if cell.alignment.vertical:
|
|
917
|
+
style["vertical-align"] = cell.alignment.vertical
|
|
918
|
+
|
|
919
|
+
if cell.fill and cell.fill.patternType == "solid" and cell.fill.fgColor:
|
|
920
|
+
# handle bg color
|
|
921
|
+
color = cell.fill.fgColor.rgb
|
|
922
|
+
if hasattr(cell.fill.fgColor, "type") and cell.fill.fgColor.type == "rgb" and color:
|
|
923
|
+
if isinstance(color, str) and len(color) == 8:
|
|
924
|
+
style["background-color"] = "#" + color[2:]
|
|
925
|
+
return style
|
|
926
|
+
|
|
927
|
+
|
|
928
|
+
__all__ = ["SpreadsheetProjector"]
|