docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
"""HTML 表格解析、行列扫描和结构状态缓存。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from bs4 import BeautifulSoup, Tag
|
|
8
|
+
|
|
9
|
+
from ...foundation.text import full_to_half
|
|
10
|
+
|
|
11
|
+
from .models import MAX_HEADER_ROWS, RenderedCellSegment, RowMetrics, RowScanResult, RowSignature, TableMergeState
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _colspan(cell: Any) -> int:
|
|
15
|
+
"""读取 HTML 单元格 colspan,非法值交由上层安全降级。"""
|
|
16
|
+
val = cell.get("colspan", "1")
|
|
17
|
+
assert isinstance(val, str)
|
|
18
|
+
return int(val)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _rowspan(cell: Any) -> int:
|
|
22
|
+
"""读取 HTML 单元格 rowspan,非法值交由上层安全降级。"""
|
|
23
|
+
val = cell.get("rowspan", "1")
|
|
24
|
+
assert isinstance(val, str)
|
|
25
|
+
return int(val)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _normalize_cell_text(cell: Tag) -> str:
|
|
29
|
+
"""生成表头匹配使用的半角无空白文本。"""
|
|
30
|
+
return "".join(full_to_half(cell.get_text()).split())
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _display_cell_text(cell: Tag) -> str:
|
|
34
|
+
"""生成保留内部空白的半角展示文本。"""
|
|
35
|
+
return full_to_half(cell.get_text().strip())
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _scan_rows(rows: list[Tag], initial_occupied: dict[int, set[int]] | None = None, start_row_idx: int = 0) -> RowScanResult:
|
|
39
|
+
"""单次扫描 HTML 行并缓存有效列、显式列和跨行占位指标。
|
|
40
|
+
|
|
41
|
+
``initial_occupied`` 使用相对首行的偏移记录未来行占位,从而保留跨越
|
|
42
|
+
前后表边界的 rowspan 结构。
|
|
43
|
+
"""
|
|
44
|
+
occupied: dict[int, dict[int, bool]] = {}
|
|
45
|
+
max_cols = 0
|
|
46
|
+
|
|
47
|
+
for row_offset, cols in (initial_occupied or {}).items():
|
|
48
|
+
if not cols:
|
|
49
|
+
continue
|
|
50
|
+
occupied[row_offset] = dict.fromkeys(cols, True)
|
|
51
|
+
max_cols = max(max_cols, max(cols) + 1)
|
|
52
|
+
|
|
53
|
+
row_effective_cols: list[int] = []
|
|
54
|
+
row_metrics: list[RowMetrics] = []
|
|
55
|
+
last_nonempty_row_metrics: RowMetrics | None = None
|
|
56
|
+
|
|
57
|
+
for local_idx, row in enumerate(rows):
|
|
58
|
+
occupied_row = occupied.setdefault(local_idx, {})
|
|
59
|
+
col_idx = 0
|
|
60
|
+
cells = row.find_all(["td", "th"])
|
|
61
|
+
actual_cols = 0
|
|
62
|
+
|
|
63
|
+
for cell in cells:
|
|
64
|
+
while col_idx in occupied_row:
|
|
65
|
+
col_idx += 1
|
|
66
|
+
|
|
67
|
+
colspan = _colspan(cell)
|
|
68
|
+
rowspan = _rowspan(cell)
|
|
69
|
+
actual_cols += colspan
|
|
70
|
+
|
|
71
|
+
for row_offset in range(rowspan):
|
|
72
|
+
target_idx = local_idx + row_offset
|
|
73
|
+
occupied_target = occupied.setdefault(target_idx, {})
|
|
74
|
+
for col in range(col_idx, col_idx + colspan):
|
|
75
|
+
occupied_target[col] = True
|
|
76
|
+
|
|
77
|
+
col_idx += colspan
|
|
78
|
+
max_cols = max(max_cols, col_idx)
|
|
79
|
+
|
|
80
|
+
effective_cols = max(occupied_row.keys()) + 1 if occupied_row else 0
|
|
81
|
+
row_effective_cols.append(effective_cols)
|
|
82
|
+
max_cols = max(max_cols, effective_cols)
|
|
83
|
+
|
|
84
|
+
metrics = RowMetrics(
|
|
85
|
+
row_idx=start_row_idx + local_idx,
|
|
86
|
+
effective_cols=effective_cols,
|
|
87
|
+
actual_cols=actual_cols,
|
|
88
|
+
visual_cols=len(cells),
|
|
89
|
+
)
|
|
90
|
+
row_metrics.append(metrics)
|
|
91
|
+
if cells:
|
|
92
|
+
last_nonempty_row_metrics = metrics
|
|
93
|
+
|
|
94
|
+
tail_occupied = {
|
|
95
|
+
row_idx - len(rows): set(cols.keys()) for row_idx, cols in occupied.items() if row_idx >= len(rows) and cols
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
return RowScanResult(
|
|
99
|
+
row_effective_cols=row_effective_cols,
|
|
100
|
+
row_metrics=row_metrics,
|
|
101
|
+
total_cols=max_cols,
|
|
102
|
+
last_nonempty_row_metrics=last_nonempty_row_metrics,
|
|
103
|
+
tail_occupied=tail_occupied,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _build_row_signature(row: Tag, effective_cols: int) -> RowSignature:
|
|
108
|
+
"""构建表头检测使用的行结构与文本签名。"""
|
|
109
|
+
cells = row.find_all(["td", "th"])
|
|
110
|
+
return RowSignature(
|
|
111
|
+
effective_cols=effective_cols,
|
|
112
|
+
colspans=tuple(_colspan(cell) for cell in cells),
|
|
113
|
+
rowspans=tuple(_rowspan(cell) for cell in cells),
|
|
114
|
+
normalized_texts=tuple(_normalize_cell_text(cell) for cell in cells),
|
|
115
|
+
display_texts=tuple(_display_cell_text(cell) for cell in cells),
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _build_front_cache(
|
|
120
|
+
rows: list[Tag], max_header_rows: int = MAX_HEADER_ROWS
|
|
121
|
+
) -> tuple[list[RowSignature], dict[int, RowMetrics]]:
|
|
122
|
+
"""缓存表格前部表头签名和首批数据行指标。"""
|
|
123
|
+
front_limit = min(len(rows), max_header_rows + 1)
|
|
124
|
+
front_rows = rows[:front_limit]
|
|
125
|
+
front_scan = _scan_rows(front_rows)
|
|
126
|
+
|
|
127
|
+
front_header_info = [
|
|
128
|
+
_build_row_signature(front_rows[idx], front_scan.row_effective_cols[idx])
|
|
129
|
+
for idx in range(min(len(front_rows), max_header_rows))
|
|
130
|
+
]
|
|
131
|
+
front_first_data_row_metrics = dict(enumerate(front_scan.row_metrics))
|
|
132
|
+
return front_header_info, front_first_data_row_metrics
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _refresh_table_state_metrics(state: TableMergeState) -> None:
|
|
136
|
+
"""HTML 结构调整后重新计算表格状态指标。"""
|
|
137
|
+
scan = _scan_rows(state.rows)
|
|
138
|
+
state.row_effective_cols = scan.row_effective_cols
|
|
139
|
+
state.total_cols = scan.total_cols
|
|
140
|
+
state.last_data_row_metrics = scan.last_nonempty_row_metrics
|
|
141
|
+
state.tail_occupied = scan.tail_occupied
|
|
142
|
+
state.front_header_info, state.front_first_data_row_metrics = _build_front_cache(state.rows)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def build_table_state_from_html(
|
|
146
|
+
html: str,
|
|
147
|
+
max_header_rows: int = MAX_HEADER_ROWS,
|
|
148
|
+
) -> TableMergeState | None:
|
|
149
|
+
"""从原始 HTML 构建 TableMergeState,不依赖 DocVortex block 结构。
|
|
150
|
+
|
|
151
|
+
供外部工具(如 mineru-vl-utils)调用,用于跨页表格结构检测。
|
|
152
|
+
返回的 state 供 HTML-only 结构 helper 使用,不包含 DocVortex block 所有者。
|
|
153
|
+
"""
|
|
154
|
+
if not html:
|
|
155
|
+
return None
|
|
156
|
+
|
|
157
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
158
|
+
tbody = soup.find("tbody") or soup.find("table")
|
|
159
|
+
rows = soup.find_all("tr")
|
|
160
|
+
if tbody is None or not rows:
|
|
161
|
+
return None
|
|
162
|
+
|
|
163
|
+
try:
|
|
164
|
+
scan = _scan_rows(rows)
|
|
165
|
+
front_header_info, front_first_data_row_metrics = _build_front_cache(
|
|
166
|
+
rows,
|
|
167
|
+
max_header_rows=max_header_rows,
|
|
168
|
+
)
|
|
169
|
+
except (AssertionError, TypeError, ValueError):
|
|
170
|
+
return None
|
|
171
|
+
if scan.total_cols <= 0 or scan.last_nonempty_row_metrics is None:
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
return TableMergeState(
|
|
175
|
+
owner_block=None,
|
|
176
|
+
body_block=None,
|
|
177
|
+
soup=soup,
|
|
178
|
+
tbody=tbody,
|
|
179
|
+
rows=rows,
|
|
180
|
+
total_cols=scan.total_cols,
|
|
181
|
+
front_header_info=front_header_info,
|
|
182
|
+
front_first_data_row_metrics=front_first_data_row_metrics,
|
|
183
|
+
last_data_row_metrics=scan.last_nonempty_row_metrics,
|
|
184
|
+
row_effective_cols=scan.row_effective_cols,
|
|
185
|
+
tail_occupied=scan.tail_occupied,
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _serialize_table_state_html(state: TableMergeState) -> bool:
|
|
190
|
+
"""将合并后的 BeautifulSoup 写回克隆表体,缺失表体时返回失败。"""
|
|
191
|
+
if state.body_block is None:
|
|
192
|
+
return False
|
|
193
|
+
state.body_block["content"] = str(state.soup)
|
|
194
|
+
state.dirty = False
|
|
195
|
+
return True
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def calculate_table_total_columns(soup: BeautifulSoup) -> int:
|
|
199
|
+
"""计算表格的总列数,通过分析整个表格结构来处理rowspan和colspan."""
|
|
200
|
+
rows = soup.find_all("tr")
|
|
201
|
+
return _scan_rows(rows).total_cols if rows else 0
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def build_table_occupied_matrix(soup: BeautifulSoup) -> dict[int, int]:
|
|
205
|
+
"""构建表格的占用矩阵,返回每行的有效列数."""
|
|
206
|
+
rows = soup.find_all("tr")
|
|
207
|
+
if not rows:
|
|
208
|
+
return {}
|
|
209
|
+
|
|
210
|
+
scan = _scan_rows(rows)
|
|
211
|
+
return dict(enumerate(scan.row_effective_cols))
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def calculate_row_effective_columns(soup: BeautifulSoup, row_idx: int) -> int:
|
|
215
|
+
"""计算指定行的有效列数(考虑rowspan占用)."""
|
|
216
|
+
row_effective_cols = build_table_occupied_matrix(soup)
|
|
217
|
+
return row_effective_cols.get(row_idx, 0)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def calculate_row_columns(row: Tag) -> int:
|
|
221
|
+
"""计算表格行的实际列数,考虑colspan属性."""
|
|
222
|
+
cells = row.find_all(["td", "th"])
|
|
223
|
+
column_count = 0
|
|
224
|
+
|
|
225
|
+
for cell in cells:
|
|
226
|
+
colspan = _colspan(cell)
|
|
227
|
+
column_count += colspan
|
|
228
|
+
|
|
229
|
+
return column_count
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def calculate_visual_columns(row: Tag) -> int:
|
|
233
|
+
"""计算表格行的视觉列数(实际td/th单元格数量,不考虑colspan)."""
|
|
234
|
+
cells = row.find_all(["td", "th"])
|
|
235
|
+
return len(cells)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _scan_row_visual_sources(
|
|
239
|
+
rows: list[Tag],
|
|
240
|
+
target_row_index: int,
|
|
241
|
+
initial_occupied: dict[int, set[int]] | None = None,
|
|
242
|
+
) -> tuple[dict[int, tuple[int, int]], int]:
|
|
243
|
+
"""扫描到目标行,记录每个视觉列当前由哪个源单元格占据。
|
|
244
|
+
|
|
245
|
+
initial_occupied 表示从上一页延续过来的 rowspan 占位,行号相对
|
|
246
|
+
rows[0] 计算。它只作为虚拟源单元格参与列定位,不对应当前页真实
|
|
247
|
+
<td>/<th> 元素。
|
|
248
|
+
"""
|
|
249
|
+
if target_row_index < 0:
|
|
250
|
+
target_row_index += len(rows)
|
|
251
|
+
if target_row_index < 0 or target_row_index >= len(rows):
|
|
252
|
+
return {}, 0
|
|
253
|
+
|
|
254
|
+
# occupied[row_idx][col_idx] = (source_row_idx, source_cell_idx)
|
|
255
|
+
occupied: dict[int, dict[int, tuple[int, int]]] = {}
|
|
256
|
+
total_cols = 0
|
|
257
|
+
for row_offset, cols in (initial_occupied or {}).items():
|
|
258
|
+
if not cols:
|
|
259
|
+
continue
|
|
260
|
+
occupied[row_offset] = {col: (-1, col) for col in cols}
|
|
261
|
+
total_cols = max(total_cols, max(cols) + 1)
|
|
262
|
+
|
|
263
|
+
for r_idx in range(target_row_index + 1):
|
|
264
|
+
occupied_row = occupied.setdefault(r_idx, {})
|
|
265
|
+
col_idx = 0
|
|
266
|
+
cells = rows[r_idx].find_all(["td", "th"])
|
|
267
|
+
for cell_idx, cell in enumerate(cells):
|
|
268
|
+
while col_idx in occupied_row:
|
|
269
|
+
col_idx += 1
|
|
270
|
+
colspan = _colspan(cell)
|
|
271
|
+
rowspan = _rowspan(cell)
|
|
272
|
+
source_marker = (r_idx, cell_idx)
|
|
273
|
+
for ro in range(rowspan):
|
|
274
|
+
target_idx = r_idx + ro
|
|
275
|
+
occ = occupied.setdefault(target_idx, {})
|
|
276
|
+
for c in range(col_idx, col_idx + colspan):
|
|
277
|
+
occ[c] = source_marker
|
|
278
|
+
col_idx += colspan
|
|
279
|
+
total_cols = max(total_cols, col_idx)
|
|
280
|
+
|
|
281
|
+
return occupied.get(target_row_index, {}), total_cols
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def build_visual_col_mapping(
|
|
285
|
+
rows: list[Tag],
|
|
286
|
+
target_row_index: int,
|
|
287
|
+
initial_occupied: dict[int, set[int]] | None = None,
|
|
288
|
+
) -> list[int]:
|
|
289
|
+
"""构建目标行中每个显式 <td>/<th> 元素到视觉列位置的映射。
|
|
290
|
+
|
|
291
|
+
该映射会正确考虑从前序行继承而来的 rowspan 占位。
|
|
292
|
+
initial_occupied 可额外传入上一页延续到当前切片的 rowspan 占位。
|
|
293
|
+
"""
|
|
294
|
+
if target_row_index < 0:
|
|
295
|
+
target_row_index += len(rows)
|
|
296
|
+
if target_row_index < 0 or target_row_index >= len(rows):
|
|
297
|
+
return []
|
|
298
|
+
|
|
299
|
+
target_occupied, _ = _scan_row_visual_sources(
|
|
300
|
+
rows,
|
|
301
|
+
target_row_index,
|
|
302
|
+
initial_occupied=initial_occupied,
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
col_idx = 0
|
|
306
|
+
mapping = []
|
|
307
|
+
target_cells = rows[target_row_index].find_all(["td", "th"])
|
|
308
|
+
for cell in target_cells:
|
|
309
|
+
while col_idx in target_occupied and target_occupied[col_idx][0] < target_row_index:
|
|
310
|
+
col_idx += 1
|
|
311
|
+
mapping.append(col_idx)
|
|
312
|
+
colspan = _colspan(cell)
|
|
313
|
+
col_idx += colspan
|
|
314
|
+
return mapping
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def build_row_rendered_cell_segments(
|
|
318
|
+
rows: list[Tag],
|
|
319
|
+
target_row_index: int,
|
|
320
|
+
initial_occupied: dict[int, set[int]] | None = None,
|
|
321
|
+
) -> list[RenderedCellSegment]:
|
|
322
|
+
"""构建目标行的渲染单元格段,保留每段覆盖的视觉列范围。
|
|
323
|
+
|
|
324
|
+
该函数复用表格行视觉来源扫描结果,语义与 calculate_row_rendered_segments()
|
|
325
|
+
保持一致:colspan 只算一个渲染段,rowspan 延续下来的单元格也会作为
|
|
326
|
+
目标行的渲染段返回。
|
|
327
|
+
"""
|
|
328
|
+
if target_row_index < 0:
|
|
329
|
+
target_row_index += len(rows)
|
|
330
|
+
if target_row_index < 0 or target_row_index >= len(rows):
|
|
331
|
+
return []
|
|
332
|
+
|
|
333
|
+
target_occupied, total_cols = _scan_row_visual_sources(
|
|
334
|
+
rows,
|
|
335
|
+
target_row_index,
|
|
336
|
+
initial_occupied=initial_occupied,
|
|
337
|
+
)
|
|
338
|
+
if total_cols == 0:
|
|
339
|
+
return []
|
|
340
|
+
|
|
341
|
+
segments: list[RenderedCellSegment] = []
|
|
342
|
+
current_marker: tuple[int, int] | None = None
|
|
343
|
+
current_start_col: int | None = None
|
|
344
|
+
current_text = ""
|
|
345
|
+
|
|
346
|
+
# 连续视觉列来自同一个源单元格时,合并为一个渲染段。
|
|
347
|
+
for col_idx in range(total_cols):
|
|
348
|
+
marker = target_occupied.get(col_idx)
|
|
349
|
+
if marker is None:
|
|
350
|
+
if current_marker is not None and current_start_col is not None:
|
|
351
|
+
segments.append(RenderedCellSegment(text=current_text, start_col=current_start_col, end_col=col_idx))
|
|
352
|
+
current_marker = None
|
|
353
|
+
current_start_col = None
|
|
354
|
+
current_text = ""
|
|
355
|
+
continue
|
|
356
|
+
|
|
357
|
+
if marker != current_marker:
|
|
358
|
+
if current_marker is not None and current_start_col is not None:
|
|
359
|
+
segments.append(RenderedCellSegment(text=current_text, start_col=current_start_col, end_col=col_idx))
|
|
360
|
+
current_marker = marker
|
|
361
|
+
current_start_col = col_idx
|
|
362
|
+
source_row_idx, source_cell_idx = marker
|
|
363
|
+
current_text = ""
|
|
364
|
+
if source_row_idx >= 0:
|
|
365
|
+
source_cells = rows[source_row_idx].find_all(["td", "th"])
|
|
366
|
+
if source_cell_idx < len(source_cells):
|
|
367
|
+
current_text = _display_cell_text(source_cells[source_cell_idx])
|
|
368
|
+
|
|
369
|
+
if current_marker is not None and current_start_col is not None:
|
|
370
|
+
segments.append(RenderedCellSegment(text=current_text, start_col=current_start_col, end_col=total_cols))
|
|
371
|
+
|
|
372
|
+
return segments
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def calculate_row_rendered_segments(rows: list[Tag], target_row_index: int) -> int:
|
|
376
|
+
"""计算目标行渲染后的视觉段数。
|
|
377
|
+
|
|
378
|
+
段数按“渲染出来的单元格块”统计:
|
|
379
|
+
- 当前行显式单元格各算一段,不展开 colspan
|
|
380
|
+
- 从前序行继承而来的 rowspan 占位也算段
|
|
381
|
+
- 只有连续列且来自同一个源单元格时才算同一段
|
|
382
|
+
"""
|
|
383
|
+
target_occupied, total_cols = _scan_row_visual_sources(rows, target_row_index)
|
|
384
|
+
if total_cols == 0:
|
|
385
|
+
return 0
|
|
386
|
+
|
|
387
|
+
segment_count = 0
|
|
388
|
+
previous_marker: tuple[int, int] | None = None
|
|
389
|
+
|
|
390
|
+
for col_idx in range(total_cols):
|
|
391
|
+
marker = target_occupied.get(col_idx)
|
|
392
|
+
if marker is None:
|
|
393
|
+
previous_marker = None
|
|
394
|
+
continue
|
|
395
|
+
if marker != previous_marker:
|
|
396
|
+
segment_count += 1
|
|
397
|
+
previous_marker = marker
|
|
398
|
+
|
|
399
|
+
return segment_count
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""跨页表格合并使用的内部状态模型。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any, TypeAlias
|
|
7
|
+
|
|
8
|
+
MAX_HEADER_ROWS = 5
|
|
9
|
+
|
|
10
|
+
BlockDict: TypeAlias = dict[str, Any]
|
|
11
|
+
PageInfoDict: TypeAlias = dict[str, Any]
|
|
12
|
+
CalculationBBox: TypeAlias = tuple[int, int, int, int]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class RowMetrics:
|
|
17
|
+
"""记录单行的有效列、实际列和视觉列指标。"""
|
|
18
|
+
|
|
19
|
+
row_idx: int
|
|
20
|
+
effective_cols: int
|
|
21
|
+
actual_cols: int
|
|
22
|
+
visual_cols: int
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class RowSignature:
|
|
27
|
+
"""记录表头行的列结构与规范化文本签名。"""
|
|
28
|
+
|
|
29
|
+
effective_cols: int
|
|
30
|
+
colspans: tuple[int, ...]
|
|
31
|
+
rowspans: tuple[int, ...]
|
|
32
|
+
normalized_texts: tuple[str, ...]
|
|
33
|
+
display_texts: tuple[str, ...]
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def cell_count(self) -> int:
|
|
37
|
+
"""返回签名中的显式单元格数量。"""
|
|
38
|
+
return len(self.colspans)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class RenderedCellSegment:
|
|
43
|
+
"""记录一个渲染单元格覆盖的视觉列区间。"""
|
|
44
|
+
|
|
45
|
+
text: str
|
|
46
|
+
start_col: int
|
|
47
|
+
end_col: int
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class RowScanResult:
|
|
52
|
+
"""封装一次 HTML 行扫描得到的列指标与跨行占位。"""
|
|
53
|
+
|
|
54
|
+
row_effective_cols: list[int]
|
|
55
|
+
row_metrics: list[RowMetrics]
|
|
56
|
+
total_cols: int
|
|
57
|
+
last_nonempty_row_metrics: RowMetrics | None
|
|
58
|
+
tail_occupied: dict[int, set[int]]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class TableMergeState:
|
|
63
|
+
"""缓存单张表格的 block 所有者、HTML 树和结构指标。"""
|
|
64
|
+
|
|
65
|
+
owner_block: BlockDict | None
|
|
66
|
+
body_block: BlockDict | None
|
|
67
|
+
soup: Any
|
|
68
|
+
tbody: Any
|
|
69
|
+
rows: list[Any]
|
|
70
|
+
total_cols: int
|
|
71
|
+
front_header_info: list[RowSignature]
|
|
72
|
+
front_first_data_row_metrics: dict[int, RowMetrics]
|
|
73
|
+
last_data_row_metrics: RowMetrics | None
|
|
74
|
+
row_effective_cols: list[int]
|
|
75
|
+
tail_occupied: dict[int, set[int]]
|
|
76
|
+
dirty: bool = False
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""跨页表格延续文本与 caption 的轻量判定规则。"""
|
|
2
|
+
|
|
3
|
+
from ...foundation.text import full_to_half
|
|
4
|
+
|
|
5
|
+
CONTINUATION_END_MARKERS = [
|
|
6
|
+
"(续)",
|
|
7
|
+
"(续表)",
|
|
8
|
+
"(续上表)",
|
|
9
|
+
"(continued)",
|
|
10
|
+
"(cont.)",
|
|
11
|
+
"(cont’d)",
|
|
12
|
+
"(…continued)",
|
|
13
|
+
"continued",
|
|
14
|
+
"续表",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
CONTINUATION_INLINE_MARKERS = [
|
|
18
|
+
"(continued)",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_table_continuation_text(text: str) -> bool:
|
|
23
|
+
"""判断文本是否表达续表语义,供表格归组和跨页合并共同复用。"""
|
|
24
|
+
continuation_text = full_to_half((text or "").strip()).lower()
|
|
25
|
+
if not continuation_text:
|
|
26
|
+
return False
|
|
27
|
+
|
|
28
|
+
return any(
|
|
29
|
+
_matches_continuation_end_marker(continuation_text, marker.lower()) for marker in CONTINUATION_END_MARKERS
|
|
30
|
+
) or any(marker.lower() in continuation_text for marker in CONTINUATION_INLINE_MARKERS)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _matches_continuation_end_marker(text: str, marker: str) -> bool:
|
|
34
|
+
"""判断续表后缀是否按词边界命中,避免 discontinued 误命中 continued。"""
|
|
35
|
+
if not text.endswith(marker):
|
|
36
|
+
return False
|
|
37
|
+
|
|
38
|
+
if marker == "continued":
|
|
39
|
+
marker_start = len(text) - len(marker)
|
|
40
|
+
return marker_start == 0 or not text[marker_start - 1].isalpha()
|
|
41
|
+
|
|
42
|
+
return True
|