docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""跨页表格 HTML 内容、行列结构和单元格语义合并。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
|
|
7
|
+
from bs4 import Tag
|
|
8
|
+
|
|
9
|
+
from ...schema import BlockType
|
|
10
|
+
|
|
11
|
+
from .blocks import _build_post_body_child_index, _build_table_state, _table_children
|
|
12
|
+
from .html import (
|
|
13
|
+
_colspan,
|
|
14
|
+
_refresh_table_state_metrics,
|
|
15
|
+
_rowspan,
|
|
16
|
+
_scan_rows,
|
|
17
|
+
_serialize_table_state_html,
|
|
18
|
+
build_visual_col_mapping,
|
|
19
|
+
calculate_row_columns,
|
|
20
|
+
calculate_visual_columns,
|
|
21
|
+
)
|
|
22
|
+
from .models import BlockDict, TableMergeState
|
|
23
|
+
from .structure import _expand_header_count_by_rowspan, can_merge_tables, check_row_columns_match, detect_table_headers
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def adjust_table_rows_colspan(
|
|
27
|
+
rows: list[Tag],
|
|
28
|
+
start_idx: int,
|
|
29
|
+
end_idx: int,
|
|
30
|
+
row_effective_cols: list[int],
|
|
31
|
+
reference_structure: list[int],
|
|
32
|
+
reference_visual_cols: int,
|
|
33
|
+
target_cols: int,
|
|
34
|
+
match_reference_row: Tag,
|
|
35
|
+
) -> None:
|
|
36
|
+
"""调整表格行的colspan属性以匹配目标列数."""
|
|
37
|
+
reference_row_copy = deepcopy(match_reference_row)
|
|
38
|
+
|
|
39
|
+
for row_idx in range(start_idx, end_idx):
|
|
40
|
+
row = rows[row_idx]
|
|
41
|
+
cells = row.find_all(["td", "th"])
|
|
42
|
+
if not cells:
|
|
43
|
+
continue
|
|
44
|
+
|
|
45
|
+
current_row_effective_cols = row_effective_cols[row_idx]
|
|
46
|
+
current_row_cols = calculate_row_columns(row)
|
|
47
|
+
|
|
48
|
+
if current_row_effective_cols >= target_cols or current_row_cols >= target_cols:
|
|
49
|
+
continue
|
|
50
|
+
|
|
51
|
+
if calculate_visual_columns(row) == reference_visual_cols and check_row_columns_match(row, reference_row_copy):
|
|
52
|
+
if len(cells) <= len(reference_structure):
|
|
53
|
+
for cell_idx, cell in enumerate(cells):
|
|
54
|
+
if cell_idx < len(reference_structure) and reference_structure[cell_idx] > 1:
|
|
55
|
+
cell["colspan"] = str(reference_structure[cell_idx])
|
|
56
|
+
else:
|
|
57
|
+
cols_diff = target_cols - current_row_effective_cols
|
|
58
|
+
if cols_diff > 0:
|
|
59
|
+
last_cell = cells[-1]
|
|
60
|
+
current_last_span = _colspan(last_cell)
|
|
61
|
+
last_cell["colspan"] = str(current_last_span + cols_diff)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _cell_has_semantic_content(cell: Tag) -> bool:
|
|
65
|
+
"""判断单元格是否仍包含用户可见的语义内容。"""
|
|
66
|
+
if cell.get_text(strip=True):
|
|
67
|
+
return True
|
|
68
|
+
|
|
69
|
+
return cell.find(["img", "svg", "math", "eq", "table", "figure", "object", "embed", "canvas"]) is not None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _row_has_semantic_content(row: Tag) -> bool:
|
|
73
|
+
"""判断整行是否仍保留未并回的语义内容。"""
|
|
74
|
+
return any(_cell_has_semantic_content(cell) for cell in row.find_all(["td", "th"]))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _insert_cell_before_visual_column(rows: list[Tag], target_row_index: int, start_vcol: int, cell: Tag) -> None:
|
|
78
|
+
"""将单元格插入到目标行中对应视觉列之前。"""
|
|
79
|
+
target_row = rows[target_row_index]
|
|
80
|
+
target_cells = target_row.find_all(["td", "th"])
|
|
81
|
+
target_vcol_map = build_visual_col_mapping(rows, target_row_index)
|
|
82
|
+
|
|
83
|
+
for idx, target_start_vcol in enumerate(target_vcol_map):
|
|
84
|
+
if target_start_vcol >= start_vcol:
|
|
85
|
+
target_cells[idx].insert_before(cell)
|
|
86
|
+
return
|
|
87
|
+
|
|
88
|
+
target_row.append(cell)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _carry_rowspan_structure_to_next_row(rows: list[Tag], row_idx: int) -> None:
|
|
92
|
+
"""下沉空白结构占位单元格,避免删除当前行后破坏后续列对齐。"""
|
|
93
|
+
next_row_idx = row_idx + 1
|
|
94
|
+
if next_row_idx >= len(rows):
|
|
95
|
+
return
|
|
96
|
+
|
|
97
|
+
current_row = rows[row_idx]
|
|
98
|
+
current_cells = current_row.find_all(["td", "th"])
|
|
99
|
+
current_vcol_map = build_visual_col_mapping(rows, row_idx)
|
|
100
|
+
carried_cells = []
|
|
101
|
+
|
|
102
|
+
for cell, start_vcol in zip(current_cells, current_vcol_map):
|
|
103
|
+
rowspan = _rowspan(cell)
|
|
104
|
+
if rowspan <= 1 or _cell_has_semantic_content(cell):
|
|
105
|
+
continue
|
|
106
|
+
|
|
107
|
+
carried_cell = deepcopy(cell)
|
|
108
|
+
new_rowspan = rowspan - 1
|
|
109
|
+
if new_rowspan > 1:
|
|
110
|
+
carried_cell["rowspan"] = str(new_rowspan)
|
|
111
|
+
else:
|
|
112
|
+
carried_cell.attrs.pop("rowspan", None)
|
|
113
|
+
carried_cells.append((start_vcol, carried_cell))
|
|
114
|
+
|
|
115
|
+
for start_vcol, carried_cell in sorted(carried_cells, key=lambda item: item[0], reverse=True):
|
|
116
|
+
_insert_cell_before_visual_column(rows, next_row_idx, start_vcol, carried_cell)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _clip_overlapped_blank_rowspan_cells(
|
|
120
|
+
rows: list[Tag],
|
|
121
|
+
initial_occupied: dict[int, set[int]],
|
|
122
|
+
) -> bool:
|
|
123
|
+
"""裁剪被上页 rowspan 覆盖的当前页空白结构占位。
|
|
124
|
+
|
|
125
|
+
跨页表格中,上一页未结束的 rowspan 会通过 initial_occupied 占住
|
|
126
|
+
当前页开头的视觉列。如果当前页表格识别又生成了同位置的空白
|
|
127
|
+
rowspan 单元格,这个单元格只是结构占位;直接拼接会把同一视觉列
|
|
128
|
+
当成两列。这里仅裁剪无语义内容的空白占位,真实内容单元格不处理。
|
|
129
|
+
"""
|
|
130
|
+
if not rows or not initial_occupied:
|
|
131
|
+
return False
|
|
132
|
+
|
|
133
|
+
cells_to_remove = []
|
|
134
|
+
cells_to_move = []
|
|
135
|
+
|
|
136
|
+
for row_idx, row in enumerate(rows):
|
|
137
|
+
cells = row.find_all(["td", "th"])
|
|
138
|
+
visual_col_map = build_visual_col_mapping(rows, row_idx)
|
|
139
|
+
for cell, start_vcol in zip(cells, visual_col_map):
|
|
140
|
+
rowspan = _rowspan(cell)
|
|
141
|
+
if rowspan <= 1 or _cell_has_semantic_content(cell):
|
|
142
|
+
continue
|
|
143
|
+
|
|
144
|
+
colspan = _colspan(cell)
|
|
145
|
+
occupied_cols = set(range(start_vcol, start_vcol + colspan))
|
|
146
|
+
if not occupied_cols:
|
|
147
|
+
continue
|
|
148
|
+
|
|
149
|
+
overlap_rows = 0
|
|
150
|
+
while overlap_rows < rowspan:
|
|
151
|
+
covered_cols = initial_occupied.get(row_idx + overlap_rows, set())
|
|
152
|
+
if not occupied_cols.issubset(covered_cols):
|
|
153
|
+
break
|
|
154
|
+
overlap_rows += 1
|
|
155
|
+
|
|
156
|
+
if overlap_rows == 0:
|
|
157
|
+
continue
|
|
158
|
+
|
|
159
|
+
remaining_rowspan = rowspan - overlap_rows
|
|
160
|
+
target_row_idx = row_idx + overlap_rows
|
|
161
|
+
if remaining_rowspan > 0 and target_row_idx >= len(rows):
|
|
162
|
+
continue
|
|
163
|
+
|
|
164
|
+
cells_to_remove.append(cell)
|
|
165
|
+
if remaining_rowspan > 0:
|
|
166
|
+
moved_cell = deepcopy(cell)
|
|
167
|
+
if remaining_rowspan > 1:
|
|
168
|
+
moved_cell["rowspan"] = str(remaining_rowspan)
|
|
169
|
+
else:
|
|
170
|
+
moved_cell.attrs.pop("rowspan", None)
|
|
171
|
+
cells_to_move.append((target_row_idx, start_vcol, moved_cell))
|
|
172
|
+
|
|
173
|
+
if not cells_to_remove:
|
|
174
|
+
return False
|
|
175
|
+
|
|
176
|
+
for cell in cells_to_remove:
|
|
177
|
+
cell.extract()
|
|
178
|
+
|
|
179
|
+
for target_row_idx, start_vcol, moved_cell in sorted(
|
|
180
|
+
cells_to_move,
|
|
181
|
+
key=lambda item: (item[0], item[1]),
|
|
182
|
+
reverse=True,
|
|
183
|
+
):
|
|
184
|
+
_insert_cell_before_visual_column(rows, target_row_idx, start_vcol, moved_cell)
|
|
185
|
+
|
|
186
|
+
return True
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _apply_cell_merge(
|
|
190
|
+
previous_state: TableMergeState,
|
|
191
|
+
current_state: TableMergeState,
|
|
192
|
+
header_count: int,
|
|
193
|
+
) -> bool:
|
|
194
|
+
"""应用 cell_merge 语义合并。
|
|
195
|
+
|
|
196
|
+
当 cell_merge 中的值为 1 时,将下表第一数据行对应单元格的内容
|
|
197
|
+
追加到上表最后一行对应单元格中。全部为 1 时删除该数据行,
|
|
198
|
+
混合时清空已合并单元格的内容但保留行。
|
|
199
|
+
|
|
200
|
+
cell_merge 按视觉列索引对齐,通过构建视觉列映射来正确匹配
|
|
201
|
+
两个表格中可能因 rowspan 而具有不同 <td> 元素数量的行。
|
|
202
|
+
元数据从当前页 table 根块读取,HTML 与列结构仍由唯一 table body 提供。
|
|
203
|
+
"""
|
|
204
|
+
current_table_block = current_state.owner_block
|
|
205
|
+
if not isinstance(current_table_block, dict):
|
|
206
|
+
return False
|
|
207
|
+
|
|
208
|
+
cell_merge = current_table_block.get("cell_merge")
|
|
209
|
+
if not isinstance(cell_merge, list) or not cell_merge:
|
|
210
|
+
return False
|
|
211
|
+
|
|
212
|
+
rows2 = current_state.rows
|
|
213
|
+
if header_count >= len(rows2):
|
|
214
|
+
return False
|
|
215
|
+
if not previous_state.rows:
|
|
216
|
+
return False
|
|
217
|
+
|
|
218
|
+
first_data_row = rows2[header_count]
|
|
219
|
+
last_row = previous_state.rows[-1]
|
|
220
|
+
|
|
221
|
+
cells1 = last_row.find_all(["td", "th"])
|
|
222
|
+
cells2 = first_data_row.find_all(["td", "th"])
|
|
223
|
+
|
|
224
|
+
# 构建视觉列到单元格索引的映射
|
|
225
|
+
last_row_idx = len(previous_state.rows) - 1
|
|
226
|
+
vcol_map1 = build_visual_col_mapping(previous_state.rows, last_row_idx)
|
|
227
|
+
current_merge_rows = rows2[header_count:]
|
|
228
|
+
vcol_map2 = build_visual_col_mapping(
|
|
229
|
+
current_merge_rows,
|
|
230
|
+
0,
|
|
231
|
+
initial_occupied=previous_state.tail_occupied,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
# 构建视觉列 -> 单元格索引的反向映射(展开 colspan)
|
|
235
|
+
vcol_to_cell1: dict[int, int] = {}
|
|
236
|
+
for ci, start_vcol in enumerate(vcol_map1):
|
|
237
|
+
colspan = int(cells1[ci].get("colspan", 1))
|
|
238
|
+
for c in range(start_vcol, start_vcol + colspan):
|
|
239
|
+
vcol_to_cell1[c] = ci
|
|
240
|
+
vcol_to_cell2: dict[int, int] = {}
|
|
241
|
+
for ci, start_vcol in enumerate(vcol_map2):
|
|
242
|
+
colspan = int(cells2[ci].get("colspan", 1))
|
|
243
|
+
for c in range(start_vcol, start_vcol + colspan):
|
|
244
|
+
vcol_to_cell2[c] = ci
|
|
245
|
+
|
|
246
|
+
# 按唯一 (src_cell_idx, dst_cell_idx) 对执行一次转移,避免 colspan 重复处理
|
|
247
|
+
transferred_pairs: set[tuple[int, int]] = set()
|
|
248
|
+
for vi, merge_flag in enumerate(cell_merge):
|
|
249
|
+
if merge_flag == 1:
|
|
250
|
+
ci1 = vcol_to_cell1.get(vi)
|
|
251
|
+
ci2 = vcol_to_cell2.get(vi)
|
|
252
|
+
if ci1 is not None and ci2 is not None:
|
|
253
|
+
pair = (ci1, ci2)
|
|
254
|
+
if pair not in transferred_pairs:
|
|
255
|
+
for child in list(cells2[ci2].children):
|
|
256
|
+
cells1[ci1].append(child.extract())
|
|
257
|
+
transferred_pairs.add(pair)
|
|
258
|
+
|
|
259
|
+
# 只清空确实成功转移过的源单元格
|
|
260
|
+
cleared_ci2: set[int] = set()
|
|
261
|
+
for vi, merge_flag in enumerate(cell_merge):
|
|
262
|
+
if merge_flag == 1:
|
|
263
|
+
ci1 = vcol_to_cell1.get(vi)
|
|
264
|
+
ci2 = vcol_to_cell2.get(vi)
|
|
265
|
+
if ci1 is not None and ci2 is not None and ci2 not in cleared_ci2:
|
|
266
|
+
cells2[ci2].clear()
|
|
267
|
+
cleared_ci2.add(ci2)
|
|
268
|
+
|
|
269
|
+
if not _row_has_semantic_content(first_data_row):
|
|
270
|
+
_carry_rowspan_structure_to_next_row(rows2, header_count)
|
|
271
|
+
first_data_row.extract()
|
|
272
|
+
if first_data_row in rows2:
|
|
273
|
+
rows2.remove(first_data_row)
|
|
274
|
+
|
|
275
|
+
return bool(transferred_pairs)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _perform_table_content_merge(
|
|
279
|
+
previous_state: TableMergeState,
|
|
280
|
+
current_state: TableMergeState,
|
|
281
|
+
previous_table_block: BlockDict,
|
|
282
|
+
current_table_block: BlockDict,
|
|
283
|
+
) -> bool:
|
|
284
|
+
"""在两个克隆表格上执行 HTML、单元格和 footnote 的内容合并。"""
|
|
285
|
+
header_count, _, _ = detect_table_headers(previous_state, current_state)
|
|
286
|
+
header_count = _expand_header_count_by_rowspan(current_state.rows, header_count)
|
|
287
|
+
|
|
288
|
+
rows1 = previous_state.rows
|
|
289
|
+
rows2 = current_state.rows
|
|
290
|
+
if not rows1 or header_count >= len(rows2):
|
|
291
|
+
return False
|
|
292
|
+
|
|
293
|
+
previous_adjusted = False
|
|
294
|
+
|
|
295
|
+
if header_count < len(rows2):
|
|
296
|
+
current_merge_rows = rows2[header_count:]
|
|
297
|
+
if _clip_overlapped_blank_rowspan_cells(current_merge_rows, previous_state.tail_occupied):
|
|
298
|
+
_refresh_table_state_metrics(current_state)
|
|
299
|
+
|
|
300
|
+
if rows1 and rows2 and header_count < len(rows2):
|
|
301
|
+
last_row1 = rows1[-1]
|
|
302
|
+
first_data_row2 = rows2[header_count]
|
|
303
|
+
table_cols1 = previous_state.total_cols
|
|
304
|
+
table_cols2 = current_state.total_cols
|
|
305
|
+
|
|
306
|
+
if table_cols1 > table_cols2:
|
|
307
|
+
reference_structure = [int(cell.get("colspan", 1)) for cell in last_row1.find_all(["td", "th"])]
|
|
308
|
+
reference_visual_cols = calculate_visual_columns(last_row1)
|
|
309
|
+
adjust_table_rows_colspan(
|
|
310
|
+
rows2,
|
|
311
|
+
header_count,
|
|
312
|
+
len(rows2),
|
|
313
|
+
current_state.row_effective_cols,
|
|
314
|
+
reference_structure,
|
|
315
|
+
reference_visual_cols,
|
|
316
|
+
table_cols1,
|
|
317
|
+
first_data_row2,
|
|
318
|
+
)
|
|
319
|
+
elif table_cols2 > table_cols1:
|
|
320
|
+
reference_structure = [int(cell.get("colspan", 1)) for cell in first_data_row2.find_all(["td", "th"])]
|
|
321
|
+
reference_visual_cols = calculate_visual_columns(first_data_row2)
|
|
322
|
+
adjust_table_rows_colspan(
|
|
323
|
+
rows1,
|
|
324
|
+
0,
|
|
325
|
+
len(rows1),
|
|
326
|
+
previous_state.row_effective_cols,
|
|
327
|
+
reference_structure,
|
|
328
|
+
reference_visual_cols,
|
|
329
|
+
table_cols2,
|
|
330
|
+
last_row1,
|
|
331
|
+
)
|
|
332
|
+
previous_adjusted = True
|
|
333
|
+
|
|
334
|
+
if previous_adjusted:
|
|
335
|
+
_refresh_table_state_metrics(previous_state)
|
|
336
|
+
|
|
337
|
+
cell_merge_applied = _apply_cell_merge(previous_state, current_state, header_count)
|
|
338
|
+
|
|
339
|
+
appended_rows = rows2[header_count:]
|
|
340
|
+
append_start_idx = len(previous_state.rows)
|
|
341
|
+
merged_rows = []
|
|
342
|
+
|
|
343
|
+
if previous_state.tbody is None or current_state.tbody is None:
|
|
344
|
+
return False
|
|
345
|
+
|
|
346
|
+
for row in appended_rows:
|
|
347
|
+
row.extract()
|
|
348
|
+
previous_state.tbody.append(row)
|
|
349
|
+
merged_rows.append(row)
|
|
350
|
+
|
|
351
|
+
if not merged_rows and not cell_merge_applied:
|
|
352
|
+
return False
|
|
353
|
+
|
|
354
|
+
previous_state.rows.extend(merged_rows)
|
|
355
|
+
|
|
356
|
+
if merged_rows:
|
|
357
|
+
appended_scan = _scan_rows(
|
|
358
|
+
merged_rows,
|
|
359
|
+
initial_occupied=previous_state.tail_occupied,
|
|
360
|
+
start_row_idx=append_start_idx,
|
|
361
|
+
)
|
|
362
|
+
previous_state.row_effective_cols.extend(appended_scan.row_effective_cols)
|
|
363
|
+
previous_state.total_cols = max(previous_state.total_cols, appended_scan.total_cols)
|
|
364
|
+
if appended_scan.last_nonempty_row_metrics is not None:
|
|
365
|
+
previous_state.last_data_row_metrics = appended_scan.last_nonempty_row_metrics
|
|
366
|
+
previous_state.tail_occupied = appended_scan.tail_occupied
|
|
367
|
+
|
|
368
|
+
previous_content = previous_table_block.get("content")
|
|
369
|
+
if not isinstance(previous_content, list):
|
|
370
|
+
return False
|
|
371
|
+
|
|
372
|
+
previous_table_block["content"] = [
|
|
373
|
+
block for block in previous_content if not isinstance(block, dict) or block.get("type") != BlockType.TABLE_FOOTNOTE
|
|
374
|
+
]
|
|
375
|
+
current_footnotes = [
|
|
376
|
+
block for block in _table_children(current_table_block) if block.get("type") == BlockType.TABLE_FOOTNOTE
|
|
377
|
+
]
|
|
378
|
+
footnote_base_index = _build_post_body_child_index(previous_table_block, 0)
|
|
379
|
+
for footnote_offset, table_footnote in enumerate(current_footnotes, start=1):
|
|
380
|
+
temp_table_footnote = deepcopy(table_footnote)
|
|
381
|
+
temp_table_footnote.pop("_cross_page", None)
|
|
382
|
+
if footnote_base_index is None:
|
|
383
|
+
temp_table_footnote["index"] = 0
|
|
384
|
+
else:
|
|
385
|
+
temp_table_footnote["index"] = footnote_base_index + footnote_offset
|
|
386
|
+
previous_table_block["content"].append(temp_table_footnote)
|
|
387
|
+
|
|
388
|
+
previous_state.dirty = True
|
|
389
|
+
return _serialize_table_state_html(previous_state)
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def merge_table_content(previous_table: BlockDict, current_table: BlockDict) -> BlockDict | None:
|
|
393
|
+
"""纯函数式合并两张跨页表格的内容,失败时返回 ``None``。
|
|
394
|
+
|
|
395
|
+
两个输入都会先深拷贝;返回块保留前表外层信息,只改克隆表体 HTML
|
|
396
|
+
并用当前表 footnote 替换前表 footnote,不会修改任何输入对象。
|
|
397
|
+
"""
|
|
398
|
+
if (
|
|
399
|
+
not isinstance(previous_table, dict)
|
|
400
|
+
or not isinstance(current_table, dict)
|
|
401
|
+
or previous_table.get("type") != BlockType.TABLE
|
|
402
|
+
or current_table.get("type") != BlockType.TABLE
|
|
403
|
+
):
|
|
404
|
+
return None
|
|
405
|
+
|
|
406
|
+
previous_clone = deepcopy(previous_table)
|
|
407
|
+
current_clone = deepcopy(current_table)
|
|
408
|
+
try:
|
|
409
|
+
previous_state = _build_table_state(previous_clone)
|
|
410
|
+
current_state = _build_table_state(current_clone)
|
|
411
|
+
if previous_state is None or current_state is None:
|
|
412
|
+
return None
|
|
413
|
+
if not can_merge_tables(current_state, previous_state):
|
|
414
|
+
return None
|
|
415
|
+
if not _perform_table_content_merge(
|
|
416
|
+
previous_state,
|
|
417
|
+
current_state,
|
|
418
|
+
previous_clone,
|
|
419
|
+
current_clone,
|
|
420
|
+
):
|
|
421
|
+
return None
|
|
422
|
+
except (AssertionError, TypeError, ValueError):
|
|
423
|
+
return None
|
|
424
|
+
|
|
425
|
+
return previous_clone
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""文档页边界上的跨页表格识别与延续标记编排。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from ...schema import MERGE_TRANSPARENT_BLOCK_TYPES, BlockType
|
|
8
|
+
|
|
9
|
+
from .blocks import _get_or_create_table_state
|
|
10
|
+
from .models import BlockDict, PageInfoDict, TableMergeState
|
|
11
|
+
from .structure import can_merge_tables
|
|
12
|
+
|
|
13
|
+
TABLE_BOUNDARY_IGNORED_TYPES = set(MERGE_TRANSPARENT_BLOCK_TYPES)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _clear_table_continuation_marker(table_block: BlockDict) -> None:
|
|
17
|
+
"""递归清除 table 根块及其子块中过期的 ``continues_prev``。"""
|
|
18
|
+
table_block.pop("continues_prev", None)
|
|
19
|
+
content = table_block.get("content")
|
|
20
|
+
if not isinstance(content, list):
|
|
21
|
+
return
|
|
22
|
+
for child in content:
|
|
23
|
+
if isinstance(child, dict):
|
|
24
|
+
child.pop("continues_prev", None)
|
|
25
|
+
_clear_nested_continuation_markers(child)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _clear_nested_continuation_markers(block: BlockDict) -> None:
|
|
29
|
+
"""清除表格子树中的旧延续标记,避免标记落到嵌套子块。"""
|
|
30
|
+
content = block.get("content")
|
|
31
|
+
if not isinstance(content, list):
|
|
32
|
+
return
|
|
33
|
+
for child in content:
|
|
34
|
+
if isinstance(child, dict):
|
|
35
|
+
child.pop("continues_prev", None)
|
|
36
|
+
_clear_nested_continuation_markers(child)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _find_boundary_table(blocks: list[Any], *, from_end: bool) -> BlockDict | None:
|
|
40
|
+
"""从页边界扫描 table;噪声块可跳过,其他语义块立即阻断。"""
|
|
41
|
+
ordered_blocks = reversed(blocks) if from_end else iter(blocks)
|
|
42
|
+
for block in ordered_blocks:
|
|
43
|
+
if not isinstance(block, dict):
|
|
44
|
+
return None
|
|
45
|
+
block_type = block.get("type")
|
|
46
|
+
if block_type in TABLE_BOUNDARY_IGNORED_TYPES:
|
|
47
|
+
continue
|
|
48
|
+
if block_type == BlockType.TABLE:
|
|
49
|
+
return block
|
|
50
|
+
return None
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _is_consecutive_page_pair(previous_page: PageInfoDict, current_page: PageInfoDict) -> bool:
|
|
55
|
+
"""按显式零基 page_idx 判断页面在文档中是否严格连续。"""
|
|
56
|
+
previous_page_idx = previous_page.get("page_idx")
|
|
57
|
+
current_page_idx = current_page.get("page_idx")
|
|
58
|
+
return type(previous_page_idx) is int and type(current_page_idx) is int and current_page_idx == previous_page_idx + 1
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def merge_table(page_info_list: list[PageInfoDict]) -> None:
|
|
62
|
+
"""倒序识别连续页边界表格,并只在后表写入延续标记。"""
|
|
63
|
+
if not isinstance(page_info_list, list):
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
for page_info in page_info_list:
|
|
67
|
+
if not isinstance(page_info, dict):
|
|
68
|
+
continue
|
|
69
|
+
blocks = page_info.get("blocks")
|
|
70
|
+
if not isinstance(blocks, list):
|
|
71
|
+
continue
|
|
72
|
+
for block in blocks:
|
|
73
|
+
if isinstance(block, dict) and block.get("type") == BlockType.TABLE:
|
|
74
|
+
_clear_table_continuation_marker(block)
|
|
75
|
+
|
|
76
|
+
state_cache: dict[int, TableMergeState] = {}
|
|
77
|
+
|
|
78
|
+
for page_position in range(len(page_info_list) - 1, 0, -1):
|
|
79
|
+
current_page = page_info_list[page_position]
|
|
80
|
+
previous_page = page_info_list[page_position - 1]
|
|
81
|
+
if not isinstance(current_page, dict) or not isinstance(previous_page, dict):
|
|
82
|
+
continue
|
|
83
|
+
if not _is_consecutive_page_pair(previous_page, current_page):
|
|
84
|
+
continue
|
|
85
|
+
|
|
86
|
+
current_blocks = current_page.get("blocks")
|
|
87
|
+
previous_blocks = previous_page.get("blocks")
|
|
88
|
+
if not isinstance(current_blocks, list) or not isinstance(previous_blocks, list):
|
|
89
|
+
continue
|
|
90
|
+
|
|
91
|
+
current_table_block = _find_boundary_table(current_blocks, from_end=False)
|
|
92
|
+
previous_table_block = _find_boundary_table(previous_blocks, from_end=True)
|
|
93
|
+
if current_table_block is None or previous_table_block is None:
|
|
94
|
+
continue
|
|
95
|
+
|
|
96
|
+
current_state = _get_or_create_table_state(current_table_block, state_cache)
|
|
97
|
+
previous_state = _get_or_create_table_state(previous_table_block, state_cache)
|
|
98
|
+
if current_state is None or previous_state is None:
|
|
99
|
+
continue
|
|
100
|
+
|
|
101
|
+
if not can_merge_tables(current_state, previous_state):
|
|
102
|
+
continue
|
|
103
|
+
|
|
104
|
+
current_table_block["continues_prev"] = True
|