docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1931 @@
|
|
|
1
|
+
"""基于 PDF 横竖线与矩形路径恢复原子网格和合并单元格。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
import statistics
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from .candidate import GridCellSpec, build_candidate
|
|
11
|
+
from .contracts import NativeTableCandidate, NativeTableInput, NativeTableText
|
|
12
|
+
from .geometry import (
|
|
13
|
+
bbox_area,
|
|
14
|
+
bbox_intersection,
|
|
15
|
+
clamp,
|
|
16
|
+
cluster_positions,
|
|
17
|
+
covered_interval_ratio,
|
|
18
|
+
normalize_angle,
|
|
19
|
+
normalize_bbox,
|
|
20
|
+
page_bbox_to_table_local,
|
|
21
|
+
rotate_local_bbox,
|
|
22
|
+
table_local_size,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
MAX_PRIMITIVES_PER_TABLE = 5000
|
|
26
|
+
MAX_TRACKS_PER_AXIS = 200
|
|
27
|
+
MAX_ATOMIC_CELLS = 10000
|
|
28
|
+
SEPARATOR_COVERAGE_THRESHOLD = 0.80
|
|
29
|
+
MAX_TRACK_HYPOTHESES = 8
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True, slots=True)
|
|
33
|
+
class _MergedRule:
|
|
34
|
+
"""保存吸附并连接后的局部单轴线段。"""
|
|
35
|
+
|
|
36
|
+
orientation: str
|
|
37
|
+
coordinate: float
|
|
38
|
+
start: float
|
|
39
|
+
end: float
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True, slots=True)
|
|
43
|
+
class _CanonicalTrack:
|
|
44
|
+
"""保存折叠后的规范轨道及其全部原始坐标别名。"""
|
|
45
|
+
|
|
46
|
+
coordinate: float
|
|
47
|
+
aliases: tuple[float, ...]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True, slots=True)
|
|
51
|
+
class _PhysicalRowEvidence:
|
|
52
|
+
"""保存一个原子行由 drawing 独立验证的边界可靠度。"""
|
|
53
|
+
|
|
54
|
+
row: int
|
|
55
|
+
top_coverage: float
|
|
56
|
+
bottom_coverage: float
|
|
57
|
+
left_coverage: float
|
|
58
|
+
right_coverage: float
|
|
59
|
+
height_ratio: float
|
|
60
|
+
glyph_crossing: bool
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def reliability(self) -> float:
|
|
64
|
+
"""返回该行所有强物理条件中的最小可靠度。"""
|
|
65
|
+
|
|
66
|
+
if self.glyph_crossing:
|
|
67
|
+
return 0.0
|
|
68
|
+
return min(
|
|
69
|
+
self.top_coverage,
|
|
70
|
+
self.bottom_coverage,
|
|
71
|
+
self.left_coverage,
|
|
72
|
+
self.right_coverage,
|
|
73
|
+
self.height_ratio,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def verified(self) -> bool:
|
|
78
|
+
"""判断该行能否脱离文本占用独立证明结构存在。"""
|
|
79
|
+
|
|
80
|
+
return self.reliability >= SEPARATOR_COVERAGE_THRESHOLD
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass(frozen=True, slots=True)
|
|
84
|
+
class _SingleRowEvidence:
|
|
85
|
+
"""保存单物理行网格的全部外框和纵向隔断可靠度。"""
|
|
86
|
+
|
|
87
|
+
top_coverage: float
|
|
88
|
+
bottom_coverage: float
|
|
89
|
+
vertical_coverages: tuple[float, ...]
|
|
90
|
+
height_ratio: float
|
|
91
|
+
glyph_crossing: bool
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def reliability(self) -> float:
|
|
95
|
+
"""返回单行网格所有不可替代物理证据中的最小值。"""
|
|
96
|
+
|
|
97
|
+
if self.glyph_crossing or not self.vertical_coverages:
|
|
98
|
+
return 0.0
|
|
99
|
+
return min(
|
|
100
|
+
self.top_coverage,
|
|
101
|
+
self.bottom_coverage,
|
|
102
|
+
min(self.vertical_coverages),
|
|
103
|
+
self.height_ratio,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def verified(self) -> bool:
|
|
108
|
+
"""判断单行网格是否可脱离文本对齐独立验证。"""
|
|
109
|
+
|
|
110
|
+
return self.reliability >= SEPARATOR_COVERAGE_THRESHOLD
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def confidence(self) -> float:
|
|
114
|
+
"""把通过八成物理硬门的覆盖率校准到 verified 分数区间。"""
|
|
115
|
+
|
|
116
|
+
if not self.verified:
|
|
117
|
+
return 0.0
|
|
118
|
+
return min(
|
|
119
|
+
1.0,
|
|
120
|
+
0.95 + 0.05 * (self.reliability - SEPARATOR_COVERAGE_THRESHOLD) / (1.0 - SEPARATOR_COVERAGE_THRESHOLD),
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@dataclass(frozen=True, slots=True)
|
|
125
|
+
class _SingleColumnEvidence:
|
|
126
|
+
"""保存多行单列表单的横向边界和左右外框可靠度。"""
|
|
127
|
+
|
|
128
|
+
horizontal_coverages: tuple[float, ...]
|
|
129
|
+
left_coverage: float
|
|
130
|
+
right_coverage: float
|
|
131
|
+
minimum_height_ratio: float
|
|
132
|
+
glyph_crossing: bool
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def reliability(self) -> float:
|
|
136
|
+
"""返回单列表单所有强物理条件中的最小可靠度。"""
|
|
137
|
+
|
|
138
|
+
if self.glyph_crossing or not self.horizontal_coverages:
|
|
139
|
+
return 0.0
|
|
140
|
+
return min(
|
|
141
|
+
min(self.horizontal_coverages),
|
|
142
|
+
self.left_coverage,
|
|
143
|
+
self.right_coverage,
|
|
144
|
+
self.minimum_height_ratio,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
@property
|
|
148
|
+
def verified(self) -> bool:
|
|
149
|
+
"""判断单列表单是否具备不依赖文本对齐的完整线框证据。"""
|
|
150
|
+
|
|
151
|
+
return self.reliability >= SEPARATOR_COVERAGE_THRESHOLD
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def confidence(self) -> float:
|
|
155
|
+
"""把通过八成物理硬门的可靠度校准到 verified 分数区间。"""
|
|
156
|
+
|
|
157
|
+
if not self.verified:
|
|
158
|
+
return 0.0
|
|
159
|
+
return min(
|
|
160
|
+
1.0,
|
|
161
|
+
0.95 + 0.05 * (self.reliability - SEPARATOR_COVERAGE_THRESHOLD) / (1.0 - SEPARATOR_COVERAGE_THRESHOLD),
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class _UnionFind:
|
|
166
|
+
"""维护缺失内部隔断连接的原子网格并查集。"""
|
|
167
|
+
|
|
168
|
+
def __init__(self, size: int) -> None:
|
|
169
|
+
"""为固定数量原子格初始化各自独立的集合。"""
|
|
170
|
+
|
|
171
|
+
self._parents = list(range(size))
|
|
172
|
+
|
|
173
|
+
def find(self, index: int) -> int:
|
|
174
|
+
"""返回原子格根节点并执行路径压缩。"""
|
|
175
|
+
|
|
176
|
+
parent = self._parents[index]
|
|
177
|
+
if parent != index:
|
|
178
|
+
self._parents[index] = self.find(parent)
|
|
179
|
+
return self._parents[index]
|
|
180
|
+
|
|
181
|
+
def union(self, first: int, second: int) -> None:
|
|
182
|
+
"""合并两个原子格所属集合。"""
|
|
183
|
+
|
|
184
|
+
first_root = self.find(first)
|
|
185
|
+
second_root = self.find(second)
|
|
186
|
+
if first_root != second_root:
|
|
187
|
+
self._parents[second_root] = first_root
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _drawing_bbox_to_table_local(
|
|
191
|
+
rule_bbox: tuple[float, float, float, float],
|
|
192
|
+
table_bbox: tuple[float, float, float, float],
|
|
193
|
+
angle: int,
|
|
194
|
+
evidence_halo: float,
|
|
195
|
+
) -> tuple[float, float, float, float] | None:
|
|
196
|
+
"""在不扩大字符区域的前提下,把邻近外框 drawing 转到表格局部坐标。"""
|
|
197
|
+
|
|
198
|
+
clipped = bbox_intersection(rule_bbox, table_bbox)
|
|
199
|
+
if clipped is None and evidence_halo > 0:
|
|
200
|
+
expanded_bbox = (
|
|
201
|
+
table_bbox[0] - evidence_halo,
|
|
202
|
+
table_bbox[1] - evidence_halo,
|
|
203
|
+
table_bbox[2] + evidence_halo,
|
|
204
|
+
table_bbox[3] + evidence_halo,
|
|
205
|
+
)
|
|
206
|
+
clipped = bbox_intersection(rule_bbox, expanded_bbox)
|
|
207
|
+
if clipped is None:
|
|
208
|
+
return None
|
|
209
|
+
width = table_bbox[2] - table_bbox[0]
|
|
210
|
+
height = table_bbox[3] - table_bbox[1]
|
|
211
|
+
relative = (
|
|
212
|
+
clipped[0] - table_bbox[0],
|
|
213
|
+
clipped[1] - table_bbox[1],
|
|
214
|
+
clipped[2] - table_bbox[0],
|
|
215
|
+
clipped[3] - table_bbox[1],
|
|
216
|
+
)
|
|
217
|
+
return rotate_local_bbox(relative, width, height, angle)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _local_rule_fragments(
|
|
221
|
+
table_input: NativeTableInput,
|
|
222
|
+
snap_tolerance: float,
|
|
223
|
+
*,
|
|
224
|
+
include_drawing: bool = True,
|
|
225
|
+
include_rectangles: bool = True,
|
|
226
|
+
evidence_halo: float = 0.0,
|
|
227
|
+
) -> list[_MergedRule]:
|
|
228
|
+
"""按来源裁剪 drawing/矩形,并转换为局部轴线片段。"""
|
|
229
|
+
|
|
230
|
+
table_bbox = normalize_bbox(table_input.table_bbox)
|
|
231
|
+
if table_bbox is None:
|
|
232
|
+
return []
|
|
233
|
+
angle = normalize_angle(table_input.angle)
|
|
234
|
+
fragments: list[_MergedRule] = []
|
|
235
|
+
for rule in table_input.drawing_lines if include_drawing else ():
|
|
236
|
+
rule_bbox = normalize_bbox(rule.bbox)
|
|
237
|
+
if rule_bbox is None:
|
|
238
|
+
continue
|
|
239
|
+
local_bbox = _drawing_bbox_to_table_local(
|
|
240
|
+
rule_bbox,
|
|
241
|
+
table_bbox,
|
|
242
|
+
angle,
|
|
243
|
+
evidence_halo,
|
|
244
|
+
)
|
|
245
|
+
if local_bbox is None:
|
|
246
|
+
continue
|
|
247
|
+
width = local_bbox[2] - local_bbox[0]
|
|
248
|
+
height = local_bbox[3] - local_bbox[1]
|
|
249
|
+
orientation = "horizontal" if width >= height else "vertical"
|
|
250
|
+
if orientation == "horizontal" and width >= max(1.0, 2.0 * snap_tolerance):
|
|
251
|
+
fragments.append(
|
|
252
|
+
_MergedRule(
|
|
253
|
+
orientation="horizontal",
|
|
254
|
+
coordinate=(local_bbox[1] + local_bbox[3]) / 2.0,
|
|
255
|
+
start=local_bbox[0],
|
|
256
|
+
end=local_bbox[2],
|
|
257
|
+
)
|
|
258
|
+
)
|
|
259
|
+
elif orientation == "vertical" and height >= max(1.0, 2.0 * snap_tolerance):
|
|
260
|
+
fragments.append(
|
|
261
|
+
_MergedRule(
|
|
262
|
+
orientation="vertical",
|
|
263
|
+
coordinate=(local_bbox[0] + local_bbox[2]) / 2.0,
|
|
264
|
+
start=local_bbox[1],
|
|
265
|
+
end=local_bbox[3],
|
|
266
|
+
)
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
local_width, local_height = table_local_size(table_bbox, angle)
|
|
270
|
+
table_area = local_width * local_height
|
|
271
|
+
for rectangle in table_input.rectangles if include_rectangles else ():
|
|
272
|
+
if rectangle.segment_count != 5 or not (rectangle.fill_visible or rectangle.stroke_visible):
|
|
273
|
+
continue
|
|
274
|
+
rectangle_bbox = normalize_bbox(rectangle.bbox)
|
|
275
|
+
if rectangle_bbox is None:
|
|
276
|
+
continue
|
|
277
|
+
local_bbox = page_bbox_to_table_local(rectangle_bbox, table_bbox, angle)
|
|
278
|
+
if local_bbox is None:
|
|
279
|
+
continue
|
|
280
|
+
rect_width = local_bbox[2] - local_bbox[0]
|
|
281
|
+
rect_height = local_bbox[3] - local_bbox[1]
|
|
282
|
+
area_ratio = bbox_area(local_bbox) / table_area if table_area > 0 else 0.0
|
|
283
|
+
if rect_width <= snap_tolerance or rect_height <= snap_tolerance or area_ratio >= 0.85:
|
|
284
|
+
continue
|
|
285
|
+
fragments.extend(
|
|
286
|
+
[
|
|
287
|
+
_MergedRule("horizontal", local_bbox[1], local_bbox[0], local_bbox[2]),
|
|
288
|
+
_MergedRule("horizontal", local_bbox[3], local_bbox[0], local_bbox[2]),
|
|
289
|
+
_MergedRule("vertical", local_bbox[0], local_bbox[1], local_bbox[3]),
|
|
290
|
+
_MergedRule("vertical", local_bbox[2], local_bbox[1], local_bbox[3]),
|
|
291
|
+
]
|
|
292
|
+
)
|
|
293
|
+
return fragments
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _merge_rule_fragments(
|
|
297
|
+
fragments: list[_MergedRule],
|
|
298
|
+
snap_tolerance: float,
|
|
299
|
+
join_gap: float,
|
|
300
|
+
) -> list[_MergedRule]:
|
|
301
|
+
"""按方向和轴坐标吸附线段,再连接小间隙共线片段。"""
|
|
302
|
+
|
|
303
|
+
output: list[_MergedRule] = []
|
|
304
|
+
for orientation in ("horizontal", "vertical"):
|
|
305
|
+
oriented = [fragment for fragment in fragments if fragment.orientation == orientation]
|
|
306
|
+
coordinates = cluster_positions(
|
|
307
|
+
(fragment.coordinate for fragment in oriented),
|
|
308
|
+
snap_tolerance,
|
|
309
|
+
)
|
|
310
|
+
for coordinate in coordinates:
|
|
311
|
+
intervals = sorted(
|
|
312
|
+
(
|
|
313
|
+
min(fragment.start, fragment.end),
|
|
314
|
+
max(fragment.start, fragment.end),
|
|
315
|
+
)
|
|
316
|
+
for fragment in oriented
|
|
317
|
+
if abs(fragment.coordinate - coordinate) <= snap_tolerance
|
|
318
|
+
)
|
|
319
|
+
if not intervals:
|
|
320
|
+
continue
|
|
321
|
+
current_start, current_end = intervals[0]
|
|
322
|
+
for start, end in intervals[1:]:
|
|
323
|
+
if start <= current_end + join_gap:
|
|
324
|
+
current_end = max(current_end, end)
|
|
325
|
+
continue
|
|
326
|
+
output.append(_MergedRule(orientation, coordinate, current_start, current_end))
|
|
327
|
+
current_start, current_end = start, end
|
|
328
|
+
output.append(_MergedRule(orientation, coordinate, current_start, current_end))
|
|
329
|
+
return output
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _repeated_long_rule_endpoints(
|
|
333
|
+
rules: list[_MergedRule],
|
|
334
|
+
axis_extent: float,
|
|
335
|
+
snap_tolerance: float,
|
|
336
|
+
) -> list[float]:
|
|
337
|
+
"""仅保留重复长线端点或接近表格外缘的外围轨道证据。"""
|
|
338
|
+
|
|
339
|
+
if not rules:
|
|
340
|
+
return []
|
|
341
|
+
maximum_length = max(rule.end - rule.start for rule in rules)
|
|
342
|
+
long_rules = [rule for rule in rules if rule.end - rule.start >= 0.80 * maximum_length]
|
|
343
|
+
required_support = max(2, math.ceil(0.50 * len(long_rules)))
|
|
344
|
+
positions = cluster_positions(
|
|
345
|
+
(endpoint for rule in long_rules for endpoint in (rule.start, rule.end)),
|
|
346
|
+
snap_tolerance,
|
|
347
|
+
)
|
|
348
|
+
output: list[float] = []
|
|
349
|
+
for position in positions:
|
|
350
|
+
support = sum(
|
|
351
|
+
min(
|
|
352
|
+
abs(rule.start - position),
|
|
353
|
+
abs(rule.end - position),
|
|
354
|
+
)
|
|
355
|
+
<= snap_tolerance
|
|
356
|
+
for rule in long_rules
|
|
357
|
+
)
|
|
358
|
+
if support >= required_support or position <= 2.0 * snap_tolerance or axis_extent - position <= 2.0 * snap_tolerance:
|
|
359
|
+
output.append(position)
|
|
360
|
+
return output
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _infer_grid_tracks(
|
|
364
|
+
rules: list[_MergedRule],
|
|
365
|
+
snap_tolerance: float,
|
|
366
|
+
width: float,
|
|
367
|
+
height: float,
|
|
368
|
+
*,
|
|
369
|
+
prune_unsupported_horizontal: bool = False,
|
|
370
|
+
) -> tuple[list[float], list[float], list[float]]:
|
|
371
|
+
"""融合物理轴线与受重复门约束的端点,恢复开放外框轨道。"""
|
|
372
|
+
|
|
373
|
+
horizontal = [rule for rule in rules if rule.orientation == "horizontal"]
|
|
374
|
+
vertical = [rule for rule in rules if rule.orientation == "vertical"]
|
|
375
|
+
x_values = [rule.coordinate for rule in vertical]
|
|
376
|
+
x_values.extend(
|
|
377
|
+
_repeated_long_rule_endpoints(
|
|
378
|
+
horizontal,
|
|
379
|
+
width,
|
|
380
|
+
snap_tolerance,
|
|
381
|
+
)
|
|
382
|
+
)
|
|
383
|
+
provisional_x_tracks = cluster_positions(x_values, snap_tolerance)
|
|
384
|
+
removed_horizontal_tracks: list[float] = []
|
|
385
|
+
supported_horizontal = horizontal
|
|
386
|
+
if prune_unsupported_horizontal:
|
|
387
|
+
supported_horizontal, removed_horizontal_tracks = _filter_horizontal_track_creators(
|
|
388
|
+
horizontal,
|
|
389
|
+
provisional_x_tracks,
|
|
390
|
+
width,
|
|
391
|
+
snap_tolerance,
|
|
392
|
+
)
|
|
393
|
+
y_values = [rule.coordinate for rule in supported_horizontal]
|
|
394
|
+
y_values.extend(
|
|
395
|
+
_repeated_long_rule_endpoints(
|
|
396
|
+
vertical,
|
|
397
|
+
height,
|
|
398
|
+
snap_tolerance,
|
|
399
|
+
)
|
|
400
|
+
)
|
|
401
|
+
return (
|
|
402
|
+
provisional_x_tracks,
|
|
403
|
+
cluster_positions(y_values, snap_tolerance),
|
|
404
|
+
removed_horizontal_tracks,
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _filter_horizontal_track_creators(
|
|
409
|
+
rules: list[_MergedRule],
|
|
410
|
+
x_tracks: list[float],
|
|
411
|
+
width: float,
|
|
412
|
+
snap_tolerance: float,
|
|
413
|
+
) -> tuple[list[_MergedRule], list[float]]:
|
|
414
|
+
"""剔除完全缩进在单元格内、不能形成真实横向轨道的装饰短线。"""
|
|
415
|
+
|
|
416
|
+
if len(x_tracks) < 2:
|
|
417
|
+
return rules, []
|
|
418
|
+
supported_coordinates: set[float] = set()
|
|
419
|
+
removed_coordinates: list[float] = []
|
|
420
|
+
for coordinate in cluster_positions(
|
|
421
|
+
(rule.coordinate for rule in rules),
|
|
422
|
+
snap_tolerance,
|
|
423
|
+
):
|
|
424
|
+
coordinate_rules = [rule for rule in rules if abs(rule.coordinate - coordinate) <= snap_tolerance]
|
|
425
|
+
intervals = [(rule.start, rule.end) for rule in coordinate_rules]
|
|
426
|
+
full_width = covered_interval_ratio(intervals, 0.0, width)
|
|
427
|
+
band_supported = False
|
|
428
|
+
for left, right in zip(x_tracks, x_tracks[1:]):
|
|
429
|
+
if covered_interval_ratio(intervals, left, right) < SEPARATOR_COVERAGE_THRESHOLD:
|
|
430
|
+
continue
|
|
431
|
+
touches_left = any(
|
|
432
|
+
rule.start <= left + snap_tolerance and rule.end >= left - snap_tolerance for rule in coordinate_rules
|
|
433
|
+
)
|
|
434
|
+
touches_right = any(
|
|
435
|
+
rule.start <= right + snap_tolerance and rule.end >= right - snap_tolerance for rule in coordinate_rules
|
|
436
|
+
)
|
|
437
|
+
if touches_left and touches_right:
|
|
438
|
+
band_supported = True
|
|
439
|
+
break
|
|
440
|
+
if full_width >= SEPARATOR_COVERAGE_THRESHOLD or band_supported:
|
|
441
|
+
supported_coordinates.add(coordinate)
|
|
442
|
+
else:
|
|
443
|
+
removed_coordinates.append(coordinate)
|
|
444
|
+
return (
|
|
445
|
+
[
|
|
446
|
+
rule
|
|
447
|
+
for rule in rules
|
|
448
|
+
if any(abs(rule.coordinate - coordinate) <= snap_tolerance for coordinate in supported_coordinates)
|
|
449
|
+
],
|
|
450
|
+
removed_coordinates,
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _rule_indices_for_track(
|
|
455
|
+
tracks_rules: list[_MergedRule],
|
|
456
|
+
orientation: str,
|
|
457
|
+
track: _CanonicalTrack,
|
|
458
|
+
snap_tolerance: float,
|
|
459
|
+
) -> set[int]:
|
|
460
|
+
"""返回能够归属指定轨道的全部物理线索引。"""
|
|
461
|
+
|
|
462
|
+
return {
|
|
463
|
+
index
|
|
464
|
+
for index, rule in enumerate(tracks_rules)
|
|
465
|
+
if rule.orientation == orientation and any(abs(rule.coordinate - alias) <= snap_tolerance for alias in track.aliases)
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def _collapse_outer_duplicate_tracks(
|
|
470
|
+
tracks: list[_CanonicalTrack],
|
|
471
|
+
glyph_centers: list[float],
|
|
472
|
+
threshold: float,
|
|
473
|
+
extent: float,
|
|
474
|
+
rules: list[_MergedRule] | None,
|
|
475
|
+
orientation: str,
|
|
476
|
+
separator_extent: float,
|
|
477
|
+
snap_tolerance: float,
|
|
478
|
+
collapse_leading_edge: bool,
|
|
479
|
+
) -> tuple[list[_CanonicalTrack], int, bool]:
|
|
480
|
+
"""折叠外缘同一物理描边产生的重复轨,并保留独立双边界。"""
|
|
481
|
+
|
|
482
|
+
if rules is None or not orientation or len(tracks) < 2:
|
|
483
|
+
return tracks, 0, False
|
|
484
|
+
collapsed_count = 0
|
|
485
|
+
while len(tracks) >= 2:
|
|
486
|
+
candidate_indices = [len(tracks) - 2]
|
|
487
|
+
if collapse_leading_edge:
|
|
488
|
+
candidate_indices.insert(0, 0)
|
|
489
|
+
collapse_index: int | None = None
|
|
490
|
+
for index in candidate_indices:
|
|
491
|
+
left, right = tracks[index], tracks[index + 1]
|
|
492
|
+
if right.coordinate - left.coordinate > threshold or any(
|
|
493
|
+
left.coordinate < center < right.coordinate for center in glyph_centers
|
|
494
|
+
):
|
|
495
|
+
continue
|
|
496
|
+
near_left_edge = right.coordinate <= max(
|
|
497
|
+
threshold,
|
|
498
|
+
2.0 * snap_tolerance,
|
|
499
|
+
)
|
|
500
|
+
near_right_edge = extent - left.coordinate <= max(
|
|
501
|
+
threshold,
|
|
502
|
+
2.0 * snap_tolerance,
|
|
503
|
+
)
|
|
504
|
+
if not (near_left_edge or near_right_edge):
|
|
505
|
+
continue
|
|
506
|
+
left_rules = _rule_indices_for_track(
|
|
507
|
+
rules,
|
|
508
|
+
orientation,
|
|
509
|
+
left,
|
|
510
|
+
snap_tolerance,
|
|
511
|
+
)
|
|
512
|
+
right_rules = _rule_indices_for_track(
|
|
513
|
+
rules,
|
|
514
|
+
orientation,
|
|
515
|
+
right,
|
|
516
|
+
snap_tolerance,
|
|
517
|
+
)
|
|
518
|
+
if left_rules and right_rules and left_rules.isdisjoint(right_rules):
|
|
519
|
+
continue
|
|
520
|
+
combined = _CanonicalTrack(
|
|
521
|
+
coordinate=float(
|
|
522
|
+
statistics.median(
|
|
523
|
+
(*left.aliases, *right.aliases),
|
|
524
|
+
)
|
|
525
|
+
),
|
|
526
|
+
aliases=tuple(sorted({*left.aliases, *right.aliases})),
|
|
527
|
+
)
|
|
528
|
+
if (
|
|
529
|
+
combined.aliases[-1] - combined.aliases[0] > threshold
|
|
530
|
+
or _separator_coverage_for_track(
|
|
531
|
+
rules,
|
|
532
|
+
orientation,
|
|
533
|
+
combined,
|
|
534
|
+
0.0,
|
|
535
|
+
separator_extent,
|
|
536
|
+
snap_tolerance,
|
|
537
|
+
)
|
|
538
|
+
< SEPARATOR_COVERAGE_THRESHOLD
|
|
539
|
+
):
|
|
540
|
+
continue
|
|
541
|
+
collapse_index = index
|
|
542
|
+
break
|
|
543
|
+
if collapse_index is None:
|
|
544
|
+
break
|
|
545
|
+
left, right = tracks[collapse_index : collapse_index + 2]
|
|
546
|
+
aliases = tuple(sorted({*left.aliases, *right.aliases}))
|
|
547
|
+
if aliases[-1] - aliases[0] > threshold:
|
|
548
|
+
return tracks, collapsed_count, True
|
|
549
|
+
tracks[collapse_index : collapse_index + 2] = [
|
|
550
|
+
_CanonicalTrack(
|
|
551
|
+
coordinate=float(statistics.median(aliases)),
|
|
552
|
+
aliases=aliases,
|
|
553
|
+
)
|
|
554
|
+
]
|
|
555
|
+
collapsed_count += 1
|
|
556
|
+
return tracks, collapsed_count, False
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _canonicalize_axis_tracks(
|
|
560
|
+
positions: list[float],
|
|
561
|
+
glyph_centers: list[float],
|
|
562
|
+
threshold: float,
|
|
563
|
+
extent: float,
|
|
564
|
+
evidence_halo: float,
|
|
565
|
+
minimum_track_count: int,
|
|
566
|
+
*,
|
|
567
|
+
rules: list[_MergedRule] | None = None,
|
|
568
|
+
orientation: str = "",
|
|
569
|
+
separator_extent: float = 0.0,
|
|
570
|
+
snap_tolerance: float = 0.0,
|
|
571
|
+
preserve_double_boundary: bool = False,
|
|
572
|
+
collapse_narrow_bands: bool = True,
|
|
573
|
+
collapse_leading_edge: bool = True,
|
|
574
|
+
) -> tuple[list[_CanonicalTrack], bool, int]:
|
|
575
|
+
"""吸附外缘并折叠无字形占用的窄带,同时保留全部原始别名。"""
|
|
576
|
+
|
|
577
|
+
tracks = [
|
|
578
|
+
_CanonicalTrack(
|
|
579
|
+
coordinate=coordinate,
|
|
580
|
+
aliases=(coordinate,),
|
|
581
|
+
)
|
|
582
|
+
for coordinate in positions
|
|
583
|
+
]
|
|
584
|
+
for edge in (0.0, extent):
|
|
585
|
+
halo_indices = [index for index, track in enumerate(tracks) if abs(track.coordinate - edge) <= evidence_halo]
|
|
586
|
+
if not halo_indices:
|
|
587
|
+
continue
|
|
588
|
+
closest_coordinate = min(
|
|
589
|
+
(tracks[index].coordinate for index in halo_indices),
|
|
590
|
+
key=lambda coordinate: abs(coordinate - edge),
|
|
591
|
+
)
|
|
592
|
+
edge_indices = [index for index in halo_indices if abs(tracks[index].coordinate - closest_coordinate) <= snap_tolerance]
|
|
593
|
+
aliases = tuple(
|
|
594
|
+
sorted(
|
|
595
|
+
{
|
|
596
|
+
edge,
|
|
597
|
+
*(alias for index in edge_indices for alias in tracks[index].aliases),
|
|
598
|
+
}
|
|
599
|
+
)
|
|
600
|
+
)
|
|
601
|
+
first_index = edge_indices[0]
|
|
602
|
+
tracks[first_index : edge_indices[-1] + 1] = [
|
|
603
|
+
_CanonicalTrack(
|
|
604
|
+
coordinate=edge,
|
|
605
|
+
aliases=aliases,
|
|
606
|
+
)
|
|
607
|
+
]
|
|
608
|
+
|
|
609
|
+
tracks, outer_collapse_count, outer_alias_conflict = _collapse_outer_duplicate_tracks(
|
|
610
|
+
tracks,
|
|
611
|
+
glyph_centers,
|
|
612
|
+
threshold,
|
|
613
|
+
extent,
|
|
614
|
+
rules,
|
|
615
|
+
orientation,
|
|
616
|
+
separator_extent,
|
|
617
|
+
snap_tolerance,
|
|
618
|
+
collapse_leading_edge,
|
|
619
|
+
)
|
|
620
|
+
if outer_alias_conflict:
|
|
621
|
+
return tracks, True, outer_collapse_count
|
|
622
|
+
|
|
623
|
+
while collapse_narrow_bands and len(tracks) > minimum_track_count:
|
|
624
|
+
collapse_index = next(
|
|
625
|
+
(
|
|
626
|
+
index
|
|
627
|
+
for index, (left, right) in enumerate(zip(tracks, tracks[1:]))
|
|
628
|
+
if right.coordinate - left.coordinate <= threshold
|
|
629
|
+
and not any(left.coordinate < center < right.coordinate for center in glyph_centers)
|
|
630
|
+
and not (
|
|
631
|
+
preserve_double_boundary
|
|
632
|
+
and rules is not None
|
|
633
|
+
and _separator_coverage_for_track(
|
|
634
|
+
rules,
|
|
635
|
+
orientation,
|
|
636
|
+
left,
|
|
637
|
+
0.0,
|
|
638
|
+
separator_extent,
|
|
639
|
+
snap_tolerance,
|
|
640
|
+
)
|
|
641
|
+
> 0.0
|
|
642
|
+
and _separator_coverage_for_track(
|
|
643
|
+
rules,
|
|
644
|
+
orientation,
|
|
645
|
+
right,
|
|
646
|
+
0.0,
|
|
647
|
+
separator_extent,
|
|
648
|
+
snap_tolerance,
|
|
649
|
+
)
|
|
650
|
+
> 0.0
|
|
651
|
+
)
|
|
652
|
+
),
|
|
653
|
+
None,
|
|
654
|
+
)
|
|
655
|
+
if collapse_index is None:
|
|
656
|
+
break
|
|
657
|
+
aliases = tuple(
|
|
658
|
+
sorted(
|
|
659
|
+
{
|
|
660
|
+
*tracks[collapse_index].aliases,
|
|
661
|
+
*tracks[collapse_index + 1].aliases,
|
|
662
|
+
}
|
|
663
|
+
)
|
|
664
|
+
)
|
|
665
|
+
if aliases[-1] - aliases[0] > threshold:
|
|
666
|
+
return tracks, True, outer_collapse_count
|
|
667
|
+
tracks[collapse_index : collapse_index + 2] = [
|
|
668
|
+
_CanonicalTrack(
|
|
669
|
+
coordinate=float(statistics.median(aliases)),
|
|
670
|
+
aliases=aliases,
|
|
671
|
+
)
|
|
672
|
+
]
|
|
673
|
+
return tracks, False, outer_collapse_count
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
def _canonical_track_coordinates(
|
|
677
|
+
tracks: list[_CanonicalTrack],
|
|
678
|
+
) -> list[float]:
|
|
679
|
+
"""提取规范轨道数值坐标,供网格 bbox 和稳定度计算使用。"""
|
|
680
|
+
|
|
681
|
+
return [track.coordinate for track in tracks]
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
def _canonical_tracks_are_unique(
|
|
685
|
+
tracks: list[_CanonicalTrack],
|
|
686
|
+
rules: list[_MergedRule],
|
|
687
|
+
orientation: str,
|
|
688
|
+
snap_tolerance: float,
|
|
689
|
+
) -> bool:
|
|
690
|
+
"""校验每条物理线最多只能归属一个规范轨道别名集合。"""
|
|
691
|
+
|
|
692
|
+
for rule in rules:
|
|
693
|
+
if rule.orientation != orientation:
|
|
694
|
+
continue
|
|
695
|
+
matching_tracks = sum(
|
|
696
|
+
any(abs(rule.coordinate - alias) <= snap_tolerance for alias in track.aliases) for track in tracks
|
|
697
|
+
)
|
|
698
|
+
if matching_tracks > 1:
|
|
699
|
+
return False
|
|
700
|
+
return True
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
def _separator_coverage_for_track(
|
|
704
|
+
rules: list[_MergedRule],
|
|
705
|
+
orientation: str,
|
|
706
|
+
track: _CanonicalTrack,
|
|
707
|
+
start: float,
|
|
708
|
+
end: float,
|
|
709
|
+
snap_tolerance: float,
|
|
710
|
+
) -> float:
|
|
711
|
+
"""按规范轨道全部 alias 合并计算 separator 覆盖率。"""
|
|
712
|
+
|
|
713
|
+
intervals = [
|
|
714
|
+
(rule.start, rule.end)
|
|
715
|
+
for rule in rules
|
|
716
|
+
if rule.orientation == orientation and any(abs(rule.coordinate - alias) <= snap_tolerance for alias in track.aliases)
|
|
717
|
+
]
|
|
718
|
+
return covered_interval_ratio(intervals, start, end)
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
def _rect_lattice_is_repeated(
|
|
722
|
+
rules: list[_MergedRule],
|
|
723
|
+
x_tracks: list[float],
|
|
724
|
+
y_tracks: list[float],
|
|
725
|
+
snap_tolerance: float,
|
|
726
|
+
) -> bool:
|
|
727
|
+
"""要求矩形边缘在二维晶格中重复且覆盖至少八成理论边界。"""
|
|
728
|
+
|
|
729
|
+
rows = len(y_tracks) - 1
|
|
730
|
+
cols = len(x_tracks) - 1
|
|
731
|
+
if rows < 2 or cols < 2:
|
|
732
|
+
return False
|
|
733
|
+
segment_present: list[bool] = []
|
|
734
|
+
vertical_repetitions: list[int] = []
|
|
735
|
+
for boundary_index, coordinate in enumerate(x_tracks):
|
|
736
|
+
repetitions = 0
|
|
737
|
+
for top, bottom in zip(y_tracks, y_tracks[1:]):
|
|
738
|
+
present = (
|
|
739
|
+
_separator_coverage(
|
|
740
|
+
rules,
|
|
741
|
+
"vertical",
|
|
742
|
+
coordinate,
|
|
743
|
+
top,
|
|
744
|
+
bottom,
|
|
745
|
+
snap_tolerance,
|
|
746
|
+
)
|
|
747
|
+
>= SEPARATOR_COVERAGE_THRESHOLD
|
|
748
|
+
)
|
|
749
|
+
segment_present.append(present)
|
|
750
|
+
repetitions += int(present)
|
|
751
|
+
if 0 < boundary_index < len(x_tracks) - 1:
|
|
752
|
+
vertical_repetitions.append(repetitions)
|
|
753
|
+
horizontal_repetitions: list[int] = []
|
|
754
|
+
for boundary_index, coordinate in enumerate(y_tracks):
|
|
755
|
+
repetitions = 0
|
|
756
|
+
for left, right in zip(x_tracks, x_tracks[1:]):
|
|
757
|
+
present = (
|
|
758
|
+
_separator_coverage(
|
|
759
|
+
rules,
|
|
760
|
+
"horizontal",
|
|
761
|
+
coordinate,
|
|
762
|
+
left,
|
|
763
|
+
right,
|
|
764
|
+
snap_tolerance,
|
|
765
|
+
)
|
|
766
|
+
>= SEPARATOR_COVERAGE_THRESHOLD
|
|
767
|
+
)
|
|
768
|
+
segment_present.append(present)
|
|
769
|
+
repetitions += int(present)
|
|
770
|
+
if 0 < boundary_index < len(y_tracks) - 1:
|
|
771
|
+
horizontal_repetitions.append(repetitions)
|
|
772
|
+
coverage = sum(segment_present) / len(segment_present) if segment_present else 0.0
|
|
773
|
+
return (
|
|
774
|
+
coverage >= SEPARATOR_COVERAGE_THRESHOLD
|
|
775
|
+
and any(count >= 2 for count in vertical_repetitions)
|
|
776
|
+
and any(count >= 2 for count in horizontal_repetitions)
|
|
777
|
+
)
|
|
778
|
+
|
|
779
|
+
|
|
780
|
+
def _separator_coverage(
|
|
781
|
+
rules: list[_MergedRule],
|
|
782
|
+
orientation: str,
|
|
783
|
+
coordinate: float,
|
|
784
|
+
start: float,
|
|
785
|
+
end: float,
|
|
786
|
+
snap_tolerance: float,
|
|
787
|
+
) -> float:
|
|
788
|
+
"""计算指定潜在隔断被同轴物理线段覆盖的比例。"""
|
|
789
|
+
|
|
790
|
+
intervals = [
|
|
791
|
+
(rule.start, rule.end)
|
|
792
|
+
for rule in rules
|
|
793
|
+
if rule.orientation == orientation and abs(rule.coordinate - coordinate) <= snap_tolerance
|
|
794
|
+
]
|
|
795
|
+
return covered_interval_ratio(intervals, start, end)
|
|
796
|
+
|
|
797
|
+
|
|
798
|
+
def _grid_index(row: int, col: int, cols: int) -> int:
|
|
799
|
+
"""把二维原子格坐标转换为并查集线性索引。"""
|
|
800
|
+
|
|
801
|
+
return row * cols + col
|
|
802
|
+
|
|
803
|
+
|
|
804
|
+
def _build_component_specs(
|
|
805
|
+
union_find: _UnionFind,
|
|
806
|
+
rows: int,
|
|
807
|
+
cols: int,
|
|
808
|
+
x_tracks: list[float],
|
|
809
|
+
y_tracks: list[float],
|
|
810
|
+
) -> tuple[GridCellSpec, ...] | None:
|
|
811
|
+
"""把原子格连通分量转成矩形逻辑单元格,非矩形分量整体拒绝。"""
|
|
812
|
+
|
|
813
|
+
components: dict[int, list[tuple[int, int]]] = {}
|
|
814
|
+
for row in range(rows):
|
|
815
|
+
for col in range(cols):
|
|
816
|
+
root = union_find.find(_grid_index(row, col, cols))
|
|
817
|
+
components.setdefault(root, []).append((row, col))
|
|
818
|
+
specs: list[GridCellSpec] = []
|
|
819
|
+
for positions in components.values():
|
|
820
|
+
row_values = [item[0] for item in positions]
|
|
821
|
+
col_values = [item[1] for item in positions]
|
|
822
|
+
min_row, max_row = min(row_values), max(row_values)
|
|
823
|
+
min_col, max_col = min(col_values), max(col_values)
|
|
824
|
+
expected_size = (max_row - min_row + 1) * (max_col - min_col + 1)
|
|
825
|
+
if len(positions) != expected_size:
|
|
826
|
+
return None
|
|
827
|
+
specs.append(
|
|
828
|
+
GridCellSpec(
|
|
829
|
+
row=min_row,
|
|
830
|
+
col=min_col,
|
|
831
|
+
rowspan=max_row - min_row + 1,
|
|
832
|
+
colspan=max_col - min_col + 1,
|
|
833
|
+
bbox=(
|
|
834
|
+
x_tracks[min_col],
|
|
835
|
+
y_tracks[min_row],
|
|
836
|
+
x_tracks[max_col + 1],
|
|
837
|
+
y_tracks[max_row + 1],
|
|
838
|
+
),
|
|
839
|
+
)
|
|
840
|
+
)
|
|
841
|
+
return tuple(sorted(specs, key=lambda item: (item.row, item.col)))
|
|
842
|
+
|
|
843
|
+
|
|
844
|
+
def _occupied_text_rows(
|
|
845
|
+
text: NativeTableText,
|
|
846
|
+
y_tracks: list[float],
|
|
847
|
+
) -> set[int]:
|
|
848
|
+
"""返回至少包含一个视觉文本行中心的物理行索引。"""
|
|
849
|
+
|
|
850
|
+
occupied_rows: set[int] = set()
|
|
851
|
+
for row in text.rows:
|
|
852
|
+
center_y = (row.bbox[1] + row.bbox[3]) / 2.0
|
|
853
|
+
for row_index, (top, bottom) in enumerate(zip(y_tracks, y_tracks[1:])):
|
|
854
|
+
if top <= center_y <= bottom:
|
|
855
|
+
occupied_rows.add(row_index)
|
|
856
|
+
break
|
|
857
|
+
return occupied_rows
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
def _line_grid_row_evidence(
|
|
861
|
+
rules: list[_MergedRule],
|
|
862
|
+
x_tracks: list[_CanonicalTrack],
|
|
863
|
+
y_tracks: list[_CanonicalTrack],
|
|
864
|
+
text: NativeTableText,
|
|
865
|
+
snap_tolerance: float,
|
|
866
|
+
minimum_row_height: float,
|
|
867
|
+
local_width: float,
|
|
868
|
+
) -> tuple[_PhysicalRowEvidence, ...]:
|
|
869
|
+
"""计算 line-grid 每个原子行的独立物理封闭证据。"""
|
|
870
|
+
|
|
871
|
+
left = x_tracks[0].coordinate
|
|
872
|
+
right = x_tracks[-1].coordinate
|
|
873
|
+
evidence: list[_PhysicalRowEvidence] = []
|
|
874
|
+
for row_index, (top_track, bottom_track) in enumerate(zip(y_tracks, y_tracks[1:])):
|
|
875
|
+
top = top_track.coordinate
|
|
876
|
+
bottom = bottom_track.coordinate
|
|
877
|
+
top_coverage = _separator_coverage_for_track(
|
|
878
|
+
rules,
|
|
879
|
+
"horizontal",
|
|
880
|
+
top_track,
|
|
881
|
+
left,
|
|
882
|
+
right,
|
|
883
|
+
snap_tolerance,
|
|
884
|
+
)
|
|
885
|
+
bottom_coverage = _separator_coverage_for_track(
|
|
886
|
+
rules,
|
|
887
|
+
"horizontal",
|
|
888
|
+
bottom_track,
|
|
889
|
+
left,
|
|
890
|
+
right,
|
|
891
|
+
snap_tolerance,
|
|
892
|
+
)
|
|
893
|
+
left_coverage = _separator_coverage_for_track(
|
|
894
|
+
rules,
|
|
895
|
+
"vertical",
|
|
896
|
+
x_tracks[0],
|
|
897
|
+
top,
|
|
898
|
+
bottom,
|
|
899
|
+
snap_tolerance,
|
|
900
|
+
)
|
|
901
|
+
right_coverage = _separator_coverage_for_track(
|
|
902
|
+
rules,
|
|
903
|
+
"vertical",
|
|
904
|
+
x_tracks[-1],
|
|
905
|
+
top,
|
|
906
|
+
bottom,
|
|
907
|
+
snap_tolerance,
|
|
908
|
+
)
|
|
909
|
+
endpoint_support = min(top_coverage, bottom_coverage)
|
|
910
|
+
if left <= 2.0 * snap_tolerance:
|
|
911
|
+
left_coverage = max(left_coverage, endpoint_support)
|
|
912
|
+
if local_width - right <= 2.0 * snap_tolerance:
|
|
913
|
+
right_coverage = max(right_coverage, endpoint_support)
|
|
914
|
+
glyph_crossing = any(
|
|
915
|
+
glyph.bbox[1] < bottom and glyph.bbox[3] > top and not (top <= (glyph.bbox[1] + glyph.bbox[3]) / 2.0 <= bottom)
|
|
916
|
+
for glyph in text.glyphs
|
|
917
|
+
)
|
|
918
|
+
evidence.append(
|
|
919
|
+
_PhysicalRowEvidence(
|
|
920
|
+
row=row_index,
|
|
921
|
+
top_coverage=top_coverage,
|
|
922
|
+
bottom_coverage=bottom_coverage,
|
|
923
|
+
left_coverage=left_coverage,
|
|
924
|
+
right_coverage=right_coverage,
|
|
925
|
+
height_ratio=min(
|
|
926
|
+
1.0,
|
|
927
|
+
(bottom - top) / max(minimum_row_height, 0.1),
|
|
928
|
+
),
|
|
929
|
+
glyph_crossing=glyph_crossing,
|
|
930
|
+
)
|
|
931
|
+
)
|
|
932
|
+
return tuple(evidence)
|
|
933
|
+
|
|
934
|
+
|
|
935
|
+
def _single_row_line_grid_evidence(
|
|
936
|
+
rules: list[_MergedRule],
|
|
937
|
+
x_tracks: list[_CanonicalTrack],
|
|
938
|
+
y_tracks: list[_CanonicalTrack],
|
|
939
|
+
text: NativeTableText,
|
|
940
|
+
snap_tolerance: float,
|
|
941
|
+
minimum_row_height: float,
|
|
942
|
+
) -> _SingleRowEvidence:
|
|
943
|
+
"""校验单物理行候选的上下外框及每一条纵向边界。"""
|
|
944
|
+
|
|
945
|
+
top_track, bottom_track = y_tracks
|
|
946
|
+
top = top_track.coordinate
|
|
947
|
+
bottom = bottom_track.coordinate
|
|
948
|
+
left = x_tracks[0].coordinate
|
|
949
|
+
right = x_tracks[-1].coordinate
|
|
950
|
+
top_coverage = _separator_coverage_for_track(
|
|
951
|
+
rules,
|
|
952
|
+
"horizontal",
|
|
953
|
+
top_track,
|
|
954
|
+
left,
|
|
955
|
+
right,
|
|
956
|
+
snap_tolerance,
|
|
957
|
+
)
|
|
958
|
+
bottom_coverage = _separator_coverage_for_track(
|
|
959
|
+
rules,
|
|
960
|
+
"horizontal",
|
|
961
|
+
bottom_track,
|
|
962
|
+
left,
|
|
963
|
+
right,
|
|
964
|
+
snap_tolerance,
|
|
965
|
+
)
|
|
966
|
+
vertical_coverages = tuple(
|
|
967
|
+
_separator_coverage_for_track(
|
|
968
|
+
rules,
|
|
969
|
+
"vertical",
|
|
970
|
+
track,
|
|
971
|
+
top,
|
|
972
|
+
bottom,
|
|
973
|
+
snap_tolerance,
|
|
974
|
+
)
|
|
975
|
+
for track in x_tracks
|
|
976
|
+
)
|
|
977
|
+
glyph_crossing = any(glyph.bbox[0] < track.coordinate < glyph.bbox[2] for glyph in text.glyphs for track in x_tracks[1:-1])
|
|
978
|
+
return _SingleRowEvidence(
|
|
979
|
+
top_coverage=top_coverage,
|
|
980
|
+
bottom_coverage=bottom_coverage,
|
|
981
|
+
vertical_coverages=vertical_coverages,
|
|
982
|
+
height_ratio=min(
|
|
983
|
+
1.0,
|
|
984
|
+
(bottom - top) / max(minimum_row_height, 0.1),
|
|
985
|
+
),
|
|
986
|
+
glyph_crossing=glyph_crossing,
|
|
987
|
+
)
|
|
988
|
+
|
|
989
|
+
|
|
990
|
+
def _single_column_line_grid_evidence(
|
|
991
|
+
rules: list[_MergedRule],
|
|
992
|
+
x_tracks: list[_CanonicalTrack],
|
|
993
|
+
y_tracks: list[_CanonicalTrack],
|
|
994
|
+
text: NativeTableText,
|
|
995
|
+
snap_tolerance: float,
|
|
996
|
+
minimum_row_height: float,
|
|
997
|
+
) -> _SingleColumnEvidence:
|
|
998
|
+
"""校验多行单列表单的全部横边和左右连续外框。"""
|
|
999
|
+
|
|
1000
|
+
left_track, right_track = x_tracks
|
|
1001
|
+
top = y_tracks[0].coordinate
|
|
1002
|
+
bottom = y_tracks[-1].coordinate
|
|
1003
|
+
horizontal_coverages = tuple(
|
|
1004
|
+
_separator_coverage_for_track(
|
|
1005
|
+
rules,
|
|
1006
|
+
"horizontal",
|
|
1007
|
+
track,
|
|
1008
|
+
left_track.coordinate,
|
|
1009
|
+
right_track.coordinate,
|
|
1010
|
+
snap_tolerance,
|
|
1011
|
+
)
|
|
1012
|
+
for track in y_tracks
|
|
1013
|
+
)
|
|
1014
|
+
left_coverage = _separator_coverage_for_track(
|
|
1015
|
+
rules,
|
|
1016
|
+
"vertical",
|
|
1017
|
+
left_track,
|
|
1018
|
+
top,
|
|
1019
|
+
bottom,
|
|
1020
|
+
snap_tolerance,
|
|
1021
|
+
)
|
|
1022
|
+
right_coverage = _separator_coverage_for_track(
|
|
1023
|
+
rules,
|
|
1024
|
+
"vertical",
|
|
1025
|
+
right_track,
|
|
1026
|
+
top,
|
|
1027
|
+
bottom,
|
|
1028
|
+
snap_tolerance,
|
|
1029
|
+
)
|
|
1030
|
+
minimum_height_ratio = min(
|
|
1031
|
+
(
|
|
1032
|
+
min(
|
|
1033
|
+
1.0,
|
|
1034
|
+
(current.coordinate - previous.coordinate) / max(minimum_row_height, 0.1),
|
|
1035
|
+
)
|
|
1036
|
+
for previous, current in zip(y_tracks, y_tracks[1:])
|
|
1037
|
+
),
|
|
1038
|
+
default=0.0,
|
|
1039
|
+
)
|
|
1040
|
+
glyph_crossing = any(glyph.bbox[1] < track.coordinate < glyph.bbox[3] for glyph in text.glyphs for track in y_tracks[1:-1])
|
|
1041
|
+
return _SingleColumnEvidence(
|
|
1042
|
+
horizontal_coverages=horizontal_coverages,
|
|
1043
|
+
left_coverage=left_coverage,
|
|
1044
|
+
right_coverage=right_coverage,
|
|
1045
|
+
minimum_height_ratio=minimum_height_ratio,
|
|
1046
|
+
glyph_crossing=glyph_crossing,
|
|
1047
|
+
)
|
|
1048
|
+
|
|
1049
|
+
|
|
1050
|
+
def _text_grid_stability(
|
|
1051
|
+
text: NativeTableText,
|
|
1052
|
+
x_tracks: list[float],
|
|
1053
|
+
y_tracks: list[float],
|
|
1054
|
+
physically_verified_rows: set[int] | None = None,
|
|
1055
|
+
) -> tuple[float, float]:
|
|
1056
|
+
"""衡量视觉文本行和文本项对推断行列轨道的占用稳定性。"""
|
|
1057
|
+
|
|
1058
|
+
occupied_rows = _occupied_text_rows(text, y_tracks)
|
|
1059
|
+
supported_rows = occupied_rows | (physically_verified_rows or set())
|
|
1060
|
+
occupied_cols: set[int] = set()
|
|
1061
|
+
for row in text.rows:
|
|
1062
|
+
for token in row.tokens:
|
|
1063
|
+
center_x = (token.bbox[0] + token.bbox[2]) / 2.0
|
|
1064
|
+
for col_index, (left, right) in enumerate(zip(x_tracks, x_tracks[1:])):
|
|
1065
|
+
if left <= center_x <= right:
|
|
1066
|
+
occupied_cols.add(col_index)
|
|
1067
|
+
break
|
|
1068
|
+
row_denominator = min(
|
|
1069
|
+
len(y_tracks) - 1,
|
|
1070
|
+
max(1, len(text.rows)),
|
|
1071
|
+
)
|
|
1072
|
+
col_denominator = min(
|
|
1073
|
+
len(x_tracks) - 1,
|
|
1074
|
+
max((len(row.tokens) for row in text.rows), default=1),
|
|
1075
|
+
)
|
|
1076
|
+
return (
|
|
1077
|
+
min(1.0, len(supported_rows) / row_denominator),
|
|
1078
|
+
min(1.0, len(occupied_cols) / col_denominator),
|
|
1079
|
+
)
|
|
1080
|
+
|
|
1081
|
+
|
|
1082
|
+
def _physical_row_dense_baseline_pairs(
|
|
1083
|
+
text: NativeTableText,
|
|
1084
|
+
x_tracks: list[float],
|
|
1085
|
+
y_tracks: list[float],
|
|
1086
|
+
) -> tuple[dict[str, object], ...]:
|
|
1087
|
+
"""识别同一物理行带内占用集合相同的多条稠密文本基线。"""
|
|
1088
|
+
|
|
1089
|
+
cols = len(x_tracks) - 1
|
|
1090
|
+
dense_column_count = max(2, math.ceil(0.60 * cols))
|
|
1091
|
+
rows_by_band: dict[int, list[tuple[int, tuple[int, ...]]]] = {row: [] for row in range(len(y_tracks) - 1)}
|
|
1092
|
+
for text_row in text.rows:
|
|
1093
|
+
center_y = (text_row.bbox[1] + text_row.bbox[3]) / 2.0
|
|
1094
|
+
band = next(
|
|
1095
|
+
(row for row, (top, bottom) in enumerate(zip(y_tracks, y_tracks[1:])) if top <= center_y <= bottom),
|
|
1096
|
+
None,
|
|
1097
|
+
)
|
|
1098
|
+
if band is None:
|
|
1099
|
+
continue
|
|
1100
|
+
occupied_cols: set[int] = set()
|
|
1101
|
+
for token in text_row.tokens:
|
|
1102
|
+
center_x = (token.bbox[0] + token.bbox[2]) / 2.0
|
|
1103
|
+
col = next(
|
|
1104
|
+
(index for index, (left, right) in enumerate(zip(x_tracks, x_tracks[1:])) if left <= center_x <= right),
|
|
1105
|
+
None,
|
|
1106
|
+
)
|
|
1107
|
+
if col is not None:
|
|
1108
|
+
occupied_cols.add(col)
|
|
1109
|
+
rows_by_band[band].append(
|
|
1110
|
+
(
|
|
1111
|
+
text_row.row_index,
|
|
1112
|
+
tuple(sorted(occupied_cols)),
|
|
1113
|
+
)
|
|
1114
|
+
)
|
|
1115
|
+
|
|
1116
|
+
ambiguous_pairs: list[dict[str, object]] = []
|
|
1117
|
+
for band, entries in rows_by_band.items():
|
|
1118
|
+
nonempty_entries = [entry for entry in entries if entry[1]]
|
|
1119
|
+
if (
|
|
1120
|
+
len(nonempty_entries) < 2
|
|
1121
|
+
or len(nonempty_entries[0][1]) < dense_column_count
|
|
1122
|
+
or any(entry[1] != nonempty_entries[0][1] for entry in nonempty_entries[1:])
|
|
1123
|
+
):
|
|
1124
|
+
continue
|
|
1125
|
+
for previous, current in zip(nonempty_entries, nonempty_entries[1:]):
|
|
1126
|
+
ambiguous_pairs.append(
|
|
1127
|
+
{
|
|
1128
|
+
"physical_row": band,
|
|
1129
|
+
"visual_rows": [previous[0], current[0]],
|
|
1130
|
+
"occupied_cols": list(previous[1]),
|
|
1131
|
+
}
|
|
1132
|
+
)
|
|
1133
|
+
return tuple(ambiguous_pairs)
|
|
1134
|
+
|
|
1135
|
+
|
|
1136
|
+
def _looks_like_single_column_tracks(
|
|
1137
|
+
x_tracks: list[float],
|
|
1138
|
+
local_width: float,
|
|
1139
|
+
edge_band: float,
|
|
1140
|
+
) -> bool:
|
|
1141
|
+
"""判断初始 X 轨是否全部属于单列表单的左右外缘。"""
|
|
1142
|
+
|
|
1143
|
+
return (
|
|
1144
|
+
len(x_tracks) >= 2
|
|
1145
|
+
and local_width > 0
|
|
1146
|
+
and all(min(abs(track), abs(local_width - track)) <= edge_band for track in x_tracks)
|
|
1147
|
+
)
|
|
1148
|
+
|
|
1149
|
+
|
|
1150
|
+
@dataclass(frozen=True, slots=True)
|
|
1151
|
+
class _VectorTracks:
|
|
1152
|
+
"""保存已通过别名、尺寸和物理行数校验的规范轨道。"""
|
|
1153
|
+
|
|
1154
|
+
snap_tolerance: float
|
|
1155
|
+
local_width: float
|
|
1156
|
+
rules: list[_MergedRule]
|
|
1157
|
+
canonical_x_tracks: list[_CanonicalTrack]
|
|
1158
|
+
canonical_y_tracks: list[_CanonicalTrack]
|
|
1159
|
+
x_tracks: list[float]
|
|
1160
|
+
y_tracks: list[float]
|
|
1161
|
+
narrow_empty_threshold: float
|
|
1162
|
+
is_line_grid: bool
|
|
1163
|
+
is_single_row_shape: bool
|
|
1164
|
+
is_single_column_shape: bool
|
|
1165
|
+
rows: int
|
|
1166
|
+
cols: int
|
|
1167
|
+
|
|
1168
|
+
|
|
1169
|
+
@dataclass(frozen=True, slots=True)
|
|
1170
|
+
class _VectorTopology:
|
|
1171
|
+
"""保存隔断连接后的逻辑单元格及独立物理证据。"""
|
|
1172
|
+
|
|
1173
|
+
specs: tuple[GridCellSpec, ...]
|
|
1174
|
+
separator_decisions: list[float]
|
|
1175
|
+
ambiguous_ratio: float
|
|
1176
|
+
alias_separator_recoveries: int
|
|
1177
|
+
y_alias_separator_recoveries: int
|
|
1178
|
+
alias_affected_rows: set[int]
|
|
1179
|
+
single_row_evidence: _SingleRowEvidence | None
|
|
1180
|
+
single_column_evidence: _SingleColumnEvidence | None
|
|
1181
|
+
|
|
1182
|
+
|
|
1183
|
+
def _reject_vector_candidate(diagnostics: dict[str, Any] | None, gate: str) -> None:
|
|
1184
|
+
"""记录当前假设的首个拒绝门,供各显式阶段保留统一诊断行为。"""
|
|
1185
|
+
|
|
1186
|
+
if diagnostics is not None:
|
|
1187
|
+
diagnostics["first_rejection_gate"] = gate
|
|
1188
|
+
return None
|
|
1189
|
+
|
|
1190
|
+
|
|
1191
|
+
def _build_vector_tracks(
|
|
1192
|
+
table_input: NativeTableInput,
|
|
1193
|
+
text: NativeTableText,
|
|
1194
|
+
*,
|
|
1195
|
+
include_drawing: bool,
|
|
1196
|
+
include_rectangles: bool,
|
|
1197
|
+
prune_unsupported_horizontal: bool,
|
|
1198
|
+
diagnostics: dict[str, Any] | None,
|
|
1199
|
+
) -> _VectorTracks | None:
|
|
1200
|
+
"""构造并规范化轨道,保持 halo、别名及物理行数的原有拒绝顺序。"""
|
|
1201
|
+
|
|
1202
|
+
snap_tolerance = clamp(
|
|
1203
|
+
0.08 * text.median_glyph_height,
|
|
1204
|
+
0.5,
|
|
1205
|
+
2.5,
|
|
1206
|
+
)
|
|
1207
|
+
join_gap = clamp(
|
|
1208
|
+
0.40 * text.median_glyph_height,
|
|
1209
|
+
2.0,
|
|
1210
|
+
8.0,
|
|
1211
|
+
)
|
|
1212
|
+
configured_evidence_halo = (
|
|
1213
|
+
clamp(
|
|
1214
|
+
0.25 * text.median_glyph_height,
|
|
1215
|
+
1.0,
|
|
1216
|
+
3.0,
|
|
1217
|
+
)
|
|
1218
|
+
if include_drawing and not include_rectangles
|
|
1219
|
+
else 0.0
|
|
1220
|
+
)
|
|
1221
|
+
table_bbox = normalize_bbox(table_input.table_bbox)
|
|
1222
|
+
if table_bbox is None:
|
|
1223
|
+
return _reject_vector_candidate(diagnostics, "table_geometry")
|
|
1224
|
+
local_width, local_height = table_local_size(
|
|
1225
|
+
table_bbox,
|
|
1226
|
+
normalize_angle(table_input.angle),
|
|
1227
|
+
)
|
|
1228
|
+
exact_fragments = _local_rule_fragments(
|
|
1229
|
+
table_input,
|
|
1230
|
+
snap_tolerance,
|
|
1231
|
+
include_drawing=include_drawing,
|
|
1232
|
+
include_rectangles=include_rectangles,
|
|
1233
|
+
evidence_halo=0.0,
|
|
1234
|
+
)
|
|
1235
|
+
if not exact_fragments or len(exact_fragments) > MAX_PRIMITIVES_PER_TABLE:
|
|
1236
|
+
return _reject_vector_candidate(diagnostics, "raw_fragments")
|
|
1237
|
+
exact_rules = _merge_rule_fragments(
|
|
1238
|
+
exact_fragments,
|
|
1239
|
+
snap_tolerance,
|
|
1240
|
+
join_gap,
|
|
1241
|
+
)
|
|
1242
|
+
exact_x_tracks, exact_y_tracks, _removed = _infer_grid_tracks(
|
|
1243
|
+
exact_rules,
|
|
1244
|
+
snap_tolerance,
|
|
1245
|
+
local_width,
|
|
1246
|
+
local_height,
|
|
1247
|
+
)
|
|
1248
|
+
# 多行表格保持原始 bbox 裁剪;halo 只服务可能退化为单物理行的
|
|
1249
|
+
# 边界片段,避免吸入相邻行或页外端点改变既有拓扑。
|
|
1250
|
+
single_column_halo_hint = (
|
|
1251
|
+
include_drawing
|
|
1252
|
+
and not include_rectangles
|
|
1253
|
+
and len(exact_y_tracks) >= 3
|
|
1254
|
+
and _looks_like_single_column_tracks(
|
|
1255
|
+
exact_x_tracks,
|
|
1256
|
+
local_width,
|
|
1257
|
+
max(3.0 * configured_evidence_halo, 2.0),
|
|
1258
|
+
)
|
|
1259
|
+
)
|
|
1260
|
+
evidence_halo = (
|
|
1261
|
+
configured_evidence_halo
|
|
1262
|
+
if configured_evidence_halo > 0
|
|
1263
|
+
and (
|
|
1264
|
+
single_column_halo_hint
|
|
1265
|
+
or len(exact_y_tracks) == 2
|
|
1266
|
+
or (
|
|
1267
|
+
len(exact_y_tracks) == 3
|
|
1268
|
+
and min(
|
|
1269
|
+
current - previous
|
|
1270
|
+
for previous, current in zip(
|
|
1271
|
+
exact_y_tracks,
|
|
1272
|
+
exact_y_tracks[1:],
|
|
1273
|
+
)
|
|
1274
|
+
)
|
|
1275
|
+
<= 0.75 * text.median_glyph_height
|
|
1276
|
+
)
|
|
1277
|
+
)
|
|
1278
|
+
else 0.0
|
|
1279
|
+
)
|
|
1280
|
+
if evidence_halo > 0:
|
|
1281
|
+
raw_fragments = _local_rule_fragments(
|
|
1282
|
+
table_input,
|
|
1283
|
+
snap_tolerance,
|
|
1284
|
+
include_drawing=include_drawing,
|
|
1285
|
+
include_rectangles=include_rectangles,
|
|
1286
|
+
evidence_halo=evidence_halo,
|
|
1287
|
+
)
|
|
1288
|
+
rules = _merge_rule_fragments(
|
|
1289
|
+
raw_fragments,
|
|
1290
|
+
snap_tolerance,
|
|
1291
|
+
join_gap,
|
|
1292
|
+
)
|
|
1293
|
+
else:
|
|
1294
|
+
raw_fragments = exact_fragments
|
|
1295
|
+
rules = exact_rules
|
|
1296
|
+
if diagnostics is not None:
|
|
1297
|
+
diagnostics["raw_fragment_count"] = len(raw_fragments)
|
|
1298
|
+
inferred_x_tracks, inferred_y_tracks, removed_horizontal_tracks = _infer_grid_tracks(
|
|
1299
|
+
rules,
|
|
1300
|
+
snap_tolerance,
|
|
1301
|
+
local_width,
|
|
1302
|
+
local_height,
|
|
1303
|
+
prune_unsupported_horizontal=prune_unsupported_horizontal,
|
|
1304
|
+
)
|
|
1305
|
+
if diagnostics is not None:
|
|
1306
|
+
diagnostics["track_hypothesis"] = "supported" if prune_unsupported_horizontal else "raw"
|
|
1307
|
+
diagnostics["evidence_halo"] = evidence_halo
|
|
1308
|
+
diagnostics["removed_horizontal_tracks"] = removed_horizontal_tracks
|
|
1309
|
+
diagnostics["inferred_tracks"] = {
|
|
1310
|
+
"x": len(inferred_x_tracks),
|
|
1311
|
+
"y": len(inferred_y_tracks),
|
|
1312
|
+
}
|
|
1313
|
+
if (
|
|
1314
|
+
include_rectangles
|
|
1315
|
+
and not include_drawing
|
|
1316
|
+
and not _rect_lattice_is_repeated(
|
|
1317
|
+
rules,
|
|
1318
|
+
inferred_x_tracks,
|
|
1319
|
+
inferred_y_tracks,
|
|
1320
|
+
snap_tolerance,
|
|
1321
|
+
)
|
|
1322
|
+
):
|
|
1323
|
+
return _reject_vector_candidate(diagnostics, "rect_lattice")
|
|
1324
|
+
line_widths = [rule.width for rule in table_input.drawing_lines if rule.width > 0]
|
|
1325
|
+
median_line_width = float(statistics.median(line_widths)) if line_widths else 0.0
|
|
1326
|
+
narrow_empty_threshold = max(
|
|
1327
|
+
0.75 * text.median_glyph_height,
|
|
1328
|
+
3.0 * median_line_width,
|
|
1329
|
+
)
|
|
1330
|
+
glyph_centers_x = [(glyph.bbox[0] + glyph.bbox[2]) / 2.0 for glyph in text.glyphs]
|
|
1331
|
+
glyph_centers_y = [(glyph.bbox[1] + glyph.bbox[3]) / 2.0 for glyph in text.glyphs]
|
|
1332
|
+
canonical_x_tracks, x_alias_conflict, outer_x_collapses = _canonicalize_axis_tracks(
|
|
1333
|
+
inferred_x_tracks,
|
|
1334
|
+
glyph_centers_x,
|
|
1335
|
+
narrow_empty_threshold,
|
|
1336
|
+
local_width,
|
|
1337
|
+
evidence_halo,
|
|
1338
|
+
3,
|
|
1339
|
+
rules=rules,
|
|
1340
|
+
orientation="vertical",
|
|
1341
|
+
separator_extent=local_height,
|
|
1342
|
+
snap_tolerance=snap_tolerance,
|
|
1343
|
+
)
|
|
1344
|
+
canonical_y_tracks, y_alias_conflict, outer_y_collapses = _canonicalize_axis_tracks(
|
|
1345
|
+
inferred_y_tracks,
|
|
1346
|
+
glyph_centers_y,
|
|
1347
|
+
narrow_empty_threshold,
|
|
1348
|
+
local_height,
|
|
1349
|
+
evidence_halo,
|
|
1350
|
+
2,
|
|
1351
|
+
rules=rules,
|
|
1352
|
+
orientation="horizontal",
|
|
1353
|
+
separator_extent=local_width,
|
|
1354
|
+
snap_tolerance=snap_tolerance,
|
|
1355
|
+
preserve_double_boundary=True,
|
|
1356
|
+
collapse_narrow_bands=len(exact_y_tracks) <= 3,
|
|
1357
|
+
collapse_leading_edge=False,
|
|
1358
|
+
)
|
|
1359
|
+
if (
|
|
1360
|
+
x_alias_conflict
|
|
1361
|
+
or y_alias_conflict
|
|
1362
|
+
or not _canonical_tracks_are_unique(
|
|
1363
|
+
canonical_x_tracks,
|
|
1364
|
+
rules,
|
|
1365
|
+
"vertical",
|
|
1366
|
+
snap_tolerance,
|
|
1367
|
+
)
|
|
1368
|
+
or not _canonical_tracks_are_unique(
|
|
1369
|
+
canonical_y_tracks,
|
|
1370
|
+
rules,
|
|
1371
|
+
"horizontal",
|
|
1372
|
+
snap_tolerance,
|
|
1373
|
+
)
|
|
1374
|
+
):
|
|
1375
|
+
return _reject_vector_candidate(diagnostics, "canonical_alias")
|
|
1376
|
+
x_tracks = _canonical_track_coordinates(canonical_x_tracks)
|
|
1377
|
+
y_tracks = _canonical_track_coordinates(canonical_y_tracks)
|
|
1378
|
+
if diagnostics is not None:
|
|
1379
|
+
diagnostics["canonical_tracks"] = [
|
|
1380
|
+
{
|
|
1381
|
+
"coordinate": track.coordinate,
|
|
1382
|
+
"aliases": list(track.aliases),
|
|
1383
|
+
}
|
|
1384
|
+
for track in canonical_x_tracks
|
|
1385
|
+
]
|
|
1386
|
+
diagnostics["canonical_y_tracks"] = [
|
|
1387
|
+
{
|
|
1388
|
+
"coordinate": track.coordinate,
|
|
1389
|
+
"aliases": list(track.aliases),
|
|
1390
|
+
}
|
|
1391
|
+
for track in canonical_y_tracks
|
|
1392
|
+
]
|
|
1393
|
+
diagnostics["outer_track_collapses"] = {
|
|
1394
|
+
"x": outer_x_collapses,
|
|
1395
|
+
"y": outer_y_collapses,
|
|
1396
|
+
}
|
|
1397
|
+
if any(
|
|
1398
|
+
right - left <= narrow_empty_threshold and not any(left < center < right for center in glyph_centers_x)
|
|
1399
|
+
for left, right in zip(x_tracks, x_tracks[1:])
|
|
1400
|
+
):
|
|
1401
|
+
return _reject_vector_candidate(diagnostics, "remaining_narrow_track")
|
|
1402
|
+
if any(
|
|
1403
|
+
bottom_track.coordinate - top_track.coordinate <= narrow_empty_threshold
|
|
1404
|
+
and not any(top_track.coordinate < center < bottom_track.coordinate for center in glyph_centers_y)
|
|
1405
|
+
and _separator_coverage_for_track(
|
|
1406
|
+
rules,
|
|
1407
|
+
"horizontal",
|
|
1408
|
+
top_track,
|
|
1409
|
+
x_tracks[0],
|
|
1410
|
+
x_tracks[-1],
|
|
1411
|
+
snap_tolerance,
|
|
1412
|
+
)
|
|
1413
|
+
>= SEPARATOR_COVERAGE_THRESHOLD
|
|
1414
|
+
and _separator_coverage_for_track(
|
|
1415
|
+
rules,
|
|
1416
|
+
"horizontal",
|
|
1417
|
+
bottom_track,
|
|
1418
|
+
x_tracks[0],
|
|
1419
|
+
x_tracks[-1],
|
|
1420
|
+
snap_tolerance,
|
|
1421
|
+
)
|
|
1422
|
+
>= SEPARATOR_COVERAGE_THRESHOLD
|
|
1423
|
+
for top_track, bottom_track in zip(
|
|
1424
|
+
canonical_y_tracks,
|
|
1425
|
+
canonical_y_tracks[1:],
|
|
1426
|
+
)
|
|
1427
|
+
):
|
|
1428
|
+
return _reject_vector_candidate(diagnostics, "remaining_narrow_track")
|
|
1429
|
+
is_line_grid = include_drawing and not include_rectangles
|
|
1430
|
+
is_single_row_shape = is_line_grid and len(y_tracks) == 2 and len(x_tracks) >= 3
|
|
1431
|
+
is_single_column_shape = is_line_grid and len(x_tracks) == 2 and len(y_tracks) >= 3
|
|
1432
|
+
if (
|
|
1433
|
+
len(x_tracks) < (2 if is_single_column_shape else 3)
|
|
1434
|
+
or len(y_tracks) < (2 if is_single_row_shape else 3)
|
|
1435
|
+
or len(x_tracks) > MAX_TRACKS_PER_AXIS
|
|
1436
|
+
or len(y_tracks) > MAX_TRACKS_PER_AXIS
|
|
1437
|
+
):
|
|
1438
|
+
return _reject_vector_candidate(diagnostics, "track_count")
|
|
1439
|
+
rows = len(y_tracks) - 1
|
|
1440
|
+
cols = len(x_tracks) - 1
|
|
1441
|
+
if diagnostics is not None:
|
|
1442
|
+
diagnostics["grid"] = {"rows": rows, "cols": cols}
|
|
1443
|
+
if rows * cols > MAX_ATOMIC_CELLS:
|
|
1444
|
+
return _reject_vector_candidate(diagnostics, "atomic_cell_limit")
|
|
1445
|
+
dense_baseline_pairs = (
|
|
1446
|
+
_physical_row_dense_baseline_pairs(
|
|
1447
|
+
text,
|
|
1448
|
+
x_tracks,
|
|
1449
|
+
y_tracks,
|
|
1450
|
+
)
|
|
1451
|
+
if is_line_grid
|
|
1452
|
+
else ()
|
|
1453
|
+
)
|
|
1454
|
+
if diagnostics is not None:
|
|
1455
|
+
diagnostics["physical_row_dense_baseline_pairs"] = list(dense_baseline_pairs)
|
|
1456
|
+
if dense_baseline_pairs:
|
|
1457
|
+
return _reject_vector_candidate(diagnostics, "physical_row_undercount")
|
|
1458
|
+
|
|
1459
|
+
return _VectorTracks(
|
|
1460
|
+
snap_tolerance=snap_tolerance,
|
|
1461
|
+
local_width=local_width,
|
|
1462
|
+
rules=rules,
|
|
1463
|
+
canonical_x_tracks=canonical_x_tracks,
|
|
1464
|
+
canonical_y_tracks=canonical_y_tracks,
|
|
1465
|
+
x_tracks=x_tracks,
|
|
1466
|
+
y_tracks=y_tracks,
|
|
1467
|
+
narrow_empty_threshold=narrow_empty_threshold,
|
|
1468
|
+
is_line_grid=is_line_grid,
|
|
1469
|
+
is_single_row_shape=is_single_row_shape,
|
|
1470
|
+
is_single_column_shape=is_single_column_shape,
|
|
1471
|
+
rows=rows,
|
|
1472
|
+
cols=cols,
|
|
1473
|
+
)
|
|
1474
|
+
|
|
1475
|
+
|
|
1476
|
+
def _build_vector_topology(
|
|
1477
|
+
tracks: _VectorTracks,
|
|
1478
|
+
text: NativeTableText,
|
|
1479
|
+
diagnostics: dict[str, Any] | None,
|
|
1480
|
+
) -> _VectorTopology | None:
|
|
1481
|
+
"""连接原子格并验证矩形拓扑,保留单行和单列的独立物理证据。"""
|
|
1482
|
+
|
|
1483
|
+
snap_tolerance = tracks.snap_tolerance
|
|
1484
|
+
rules = tracks.rules
|
|
1485
|
+
canonical_x_tracks = tracks.canonical_x_tracks
|
|
1486
|
+
canonical_y_tracks = tracks.canonical_y_tracks
|
|
1487
|
+
x_tracks = tracks.x_tracks
|
|
1488
|
+
y_tracks = tracks.y_tracks
|
|
1489
|
+
narrow_empty_threshold = tracks.narrow_empty_threshold
|
|
1490
|
+
is_single_row_shape = tracks.is_single_row_shape
|
|
1491
|
+
is_single_column_shape = tracks.is_single_column_shape
|
|
1492
|
+
rows = tracks.rows
|
|
1493
|
+
cols = tracks.cols
|
|
1494
|
+
|
|
1495
|
+
union_find = _UnionFind(rows * cols)
|
|
1496
|
+
separator_decisions: list[float] = []
|
|
1497
|
+
ambiguous_separator_count = 0
|
|
1498
|
+
alias_separator_recoveries = 0
|
|
1499
|
+
y_alias_separator_recoveries = 0
|
|
1500
|
+
alias_affected_rows: set[int] = set()
|
|
1501
|
+
for row in range(rows):
|
|
1502
|
+
for boundary_index in range(1, len(canonical_x_tracks) - 1):
|
|
1503
|
+
track = canonical_x_tracks[boundary_index]
|
|
1504
|
+
strict_coverage = _separator_coverage(
|
|
1505
|
+
rules,
|
|
1506
|
+
"vertical",
|
|
1507
|
+
track.coordinate,
|
|
1508
|
+
y_tracks[row],
|
|
1509
|
+
y_tracks[row + 1],
|
|
1510
|
+
snap_tolerance,
|
|
1511
|
+
)
|
|
1512
|
+
coverage = _separator_coverage_for_track(
|
|
1513
|
+
rules,
|
|
1514
|
+
"vertical",
|
|
1515
|
+
track,
|
|
1516
|
+
y_tracks[row],
|
|
1517
|
+
y_tracks[row + 1],
|
|
1518
|
+
snap_tolerance,
|
|
1519
|
+
)
|
|
1520
|
+
if (
|
|
1521
|
+
len(track.aliases) > 1
|
|
1522
|
+
and strict_coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD
|
|
1523
|
+
and coverage >= SEPARATOR_COVERAGE_THRESHOLD
|
|
1524
|
+
):
|
|
1525
|
+
alias_separator_recoveries += 1
|
|
1526
|
+
alias_affected_rows.add(row)
|
|
1527
|
+
separator_decisions.append(max(coverage, 1.0 - coverage))
|
|
1528
|
+
if coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD:
|
|
1529
|
+
union_find.union(
|
|
1530
|
+
_grid_index(row, boundary_index - 1, cols),
|
|
1531
|
+
_grid_index(row, boundary_index, cols),
|
|
1532
|
+
)
|
|
1533
|
+
elif coverage < SEPARATOR_COVERAGE_THRESHOLD:
|
|
1534
|
+
ambiguous_separator_count += 1
|
|
1535
|
+
for boundary_index in range(1, len(canonical_y_tracks) - 1):
|
|
1536
|
+
track = canonical_y_tracks[boundary_index]
|
|
1537
|
+
for col in range(cols):
|
|
1538
|
+
strict_coverage = _separator_coverage(
|
|
1539
|
+
rules,
|
|
1540
|
+
"horizontal",
|
|
1541
|
+
track.coordinate,
|
|
1542
|
+
x_tracks[col],
|
|
1543
|
+
x_tracks[col + 1],
|
|
1544
|
+
snap_tolerance,
|
|
1545
|
+
)
|
|
1546
|
+
coverage = _separator_coverage_for_track(
|
|
1547
|
+
rules,
|
|
1548
|
+
"horizontal",
|
|
1549
|
+
track,
|
|
1550
|
+
x_tracks[col],
|
|
1551
|
+
x_tracks[col + 1],
|
|
1552
|
+
snap_tolerance,
|
|
1553
|
+
)
|
|
1554
|
+
if (
|
|
1555
|
+
len(track.aliases) > 1
|
|
1556
|
+
and strict_coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD
|
|
1557
|
+
and coverage >= SEPARATOR_COVERAGE_THRESHOLD
|
|
1558
|
+
):
|
|
1559
|
+
y_alias_separator_recoveries += 1
|
|
1560
|
+
if boundary_index > 0:
|
|
1561
|
+
alias_affected_rows.add(boundary_index - 1)
|
|
1562
|
+
if boundary_index < rows:
|
|
1563
|
+
alias_affected_rows.add(boundary_index)
|
|
1564
|
+
separator_decisions.append(max(coverage, 1.0 - coverage))
|
|
1565
|
+
if coverage <= 1.0 - SEPARATOR_COVERAGE_THRESHOLD:
|
|
1566
|
+
union_find.union(
|
|
1567
|
+
_grid_index(boundary_index - 1, col, cols),
|
|
1568
|
+
_grid_index(boundary_index, col, cols),
|
|
1569
|
+
)
|
|
1570
|
+
elif coverage < SEPARATOR_COVERAGE_THRESHOLD:
|
|
1571
|
+
ambiguous_separator_count += 1
|
|
1572
|
+
|
|
1573
|
+
ambiguous_ratio = ambiguous_separator_count / len(separator_decisions) if separator_decisions else 0.0
|
|
1574
|
+
if diagnostics is not None:
|
|
1575
|
+
diagnostics["grid"] = {"rows": rows, "cols": cols}
|
|
1576
|
+
diagnostics["ambiguous_separator_ratio"] = ambiguous_ratio
|
|
1577
|
+
diagnostics["alias_separator_recoveries"] = alias_separator_recoveries
|
|
1578
|
+
diagnostics["y_alias_separator_recoveries"] = y_alias_separator_recoveries
|
|
1579
|
+
diagnostics["alias_affected_rows"] = sorted(alias_affected_rows)
|
|
1580
|
+
if ambiguous_ratio > 0.05:
|
|
1581
|
+
return _reject_vector_candidate(diagnostics, "ambiguous_separator")
|
|
1582
|
+
|
|
1583
|
+
single_row_evidence: _SingleRowEvidence | None = None
|
|
1584
|
+
if is_single_row_shape:
|
|
1585
|
+
single_row_evidence = _single_row_line_grid_evidence(
|
|
1586
|
+
rules,
|
|
1587
|
+
canonical_x_tracks,
|
|
1588
|
+
canonical_y_tracks,
|
|
1589
|
+
text,
|
|
1590
|
+
snap_tolerance,
|
|
1591
|
+
narrow_empty_threshold,
|
|
1592
|
+
)
|
|
1593
|
+
if diagnostics is not None:
|
|
1594
|
+
diagnostics["single_row_evidence"] = {
|
|
1595
|
+
"reliability": single_row_evidence.reliability,
|
|
1596
|
+
"confidence": single_row_evidence.confidence,
|
|
1597
|
+
"verified": single_row_evidence.verified,
|
|
1598
|
+
"top": single_row_evidence.top_coverage,
|
|
1599
|
+
"bottom": single_row_evidence.bottom_coverage,
|
|
1600
|
+
"vertical": list(single_row_evidence.vertical_coverages),
|
|
1601
|
+
"height_ratio": single_row_evidence.height_ratio,
|
|
1602
|
+
"glyph_crossing": single_row_evidence.glyph_crossing,
|
|
1603
|
+
}
|
|
1604
|
+
if not single_row_evidence.verified:
|
|
1605
|
+
return _reject_vector_candidate(diagnostics, "single_row_physical_evidence")
|
|
1606
|
+
|
|
1607
|
+
single_column_evidence: _SingleColumnEvidence | None = None
|
|
1608
|
+
if is_single_column_shape:
|
|
1609
|
+
single_column_evidence = _single_column_line_grid_evidence(
|
|
1610
|
+
rules,
|
|
1611
|
+
canonical_x_tracks,
|
|
1612
|
+
canonical_y_tracks,
|
|
1613
|
+
text,
|
|
1614
|
+
snap_tolerance,
|
|
1615
|
+
narrow_empty_threshold,
|
|
1616
|
+
)
|
|
1617
|
+
if diagnostics is not None:
|
|
1618
|
+
diagnostics["single_column_evidence"] = {
|
|
1619
|
+
"reliability": single_column_evidence.reliability,
|
|
1620
|
+
"confidence": single_column_evidence.confidence,
|
|
1621
|
+
"verified": single_column_evidence.verified,
|
|
1622
|
+
"horizontal": list(single_column_evidence.horizontal_coverages),
|
|
1623
|
+
"left": single_column_evidence.left_coverage,
|
|
1624
|
+
"right": single_column_evidence.right_coverage,
|
|
1625
|
+
"minimum_height_ratio": (single_column_evidence.minimum_height_ratio),
|
|
1626
|
+
"glyph_crossing": single_column_evidence.glyph_crossing,
|
|
1627
|
+
}
|
|
1628
|
+
if not single_column_evidence.verified:
|
|
1629
|
+
return _reject_vector_candidate(diagnostics, "single_column_physical_evidence")
|
|
1630
|
+
|
|
1631
|
+
specs = _build_component_specs(
|
|
1632
|
+
union_find,
|
|
1633
|
+
rows,
|
|
1634
|
+
cols,
|
|
1635
|
+
x_tracks,
|
|
1636
|
+
y_tracks,
|
|
1637
|
+
)
|
|
1638
|
+
if specs is None:
|
|
1639
|
+
return _reject_vector_candidate(diagnostics, "nonrectangular_topology")
|
|
1640
|
+
maximum_row_cells = max(sum(spec.row <= row_index < spec.row + spec.rowspan for spec in specs) for row_index in range(rows))
|
|
1641
|
+
maximum_col_cells = max(sum(spec.col <= col_index < spec.col + spec.colspan for spec in specs) for col_index in range(cols))
|
|
1642
|
+
if (maximum_row_cells < 2 and not is_single_column_shape) or (maximum_col_cells < 2 and not is_single_row_shape):
|
|
1643
|
+
return _reject_vector_candidate(diagnostics, "degenerate_grid")
|
|
1644
|
+
return _VectorTopology(
|
|
1645
|
+
specs=specs,
|
|
1646
|
+
separator_decisions=separator_decisions,
|
|
1647
|
+
ambiguous_ratio=ambiguous_ratio,
|
|
1648
|
+
alias_separator_recoveries=alias_separator_recoveries,
|
|
1649
|
+
y_alias_separator_recoveries=y_alias_separator_recoveries,
|
|
1650
|
+
alias_affected_rows=alias_affected_rows,
|
|
1651
|
+
single_row_evidence=single_row_evidence,
|
|
1652
|
+
single_column_evidence=single_column_evidence,
|
|
1653
|
+
)
|
|
1654
|
+
|
|
1655
|
+
|
|
1656
|
+
def _materialize_vector_candidate(
|
|
1657
|
+
tracks: _VectorTracks,
|
|
1658
|
+
topology: _VectorTopology,
|
|
1659
|
+
text: NativeTableText,
|
|
1660
|
+
evidence_label: str,
|
|
1661
|
+
diagnostics: dict[str, Any] | None,
|
|
1662
|
+
) -> NativeTableCandidate | None:
|
|
1663
|
+
"""将文本落格并评分,按原顺序执行完整性与空行发布门。"""
|
|
1664
|
+
|
|
1665
|
+
snap_tolerance = tracks.snap_tolerance
|
|
1666
|
+
local_width = tracks.local_width
|
|
1667
|
+
rules = tracks.rules
|
|
1668
|
+
canonical_x_tracks = tracks.canonical_x_tracks
|
|
1669
|
+
canonical_y_tracks = tracks.canonical_y_tracks
|
|
1670
|
+
x_tracks = tracks.x_tracks
|
|
1671
|
+
y_tracks = tracks.y_tracks
|
|
1672
|
+
narrow_empty_threshold = tracks.narrow_empty_threshold
|
|
1673
|
+
is_line_grid = tracks.is_line_grid
|
|
1674
|
+
is_single_row_shape = tracks.is_single_row_shape
|
|
1675
|
+
is_single_column_shape = tracks.is_single_column_shape
|
|
1676
|
+
rows = tracks.rows
|
|
1677
|
+
cols = tracks.cols
|
|
1678
|
+
|
|
1679
|
+
specs = topology.specs
|
|
1680
|
+
separator_decisions = topology.separator_decisions
|
|
1681
|
+
ambiguous_ratio = topology.ambiguous_ratio
|
|
1682
|
+
alias_separator_recoveries = topology.alias_separator_recoveries
|
|
1683
|
+
y_alias_separator_recoveries = topology.y_alias_separator_recoveries
|
|
1684
|
+
alias_affected_rows = topology.alias_affected_rows
|
|
1685
|
+
single_row_evidence = topology.single_row_evidence
|
|
1686
|
+
single_column_evidence = topology.single_column_evidence
|
|
1687
|
+
|
|
1688
|
+
decisiveness = float(statistics.mean(separator_decisions)) if separator_decisions else 1.0
|
|
1689
|
+
if single_row_evidence is not None:
|
|
1690
|
+
decisiveness = max(
|
|
1691
|
+
decisiveness,
|
|
1692
|
+
single_row_evidence.confidence,
|
|
1693
|
+
)
|
|
1694
|
+
if single_column_evidence is not None:
|
|
1695
|
+
decisiveness = max(
|
|
1696
|
+
decisiveness,
|
|
1697
|
+
single_column_evidence.confidence,
|
|
1698
|
+
)
|
|
1699
|
+
evidence_ratio = min(1.0, len(rules) / max(1, rows + cols))
|
|
1700
|
+
structure_support = min(
|
|
1701
|
+
decisiveness,
|
|
1702
|
+
evidence_ratio,
|
|
1703
|
+
1.0 - ambiguous_ratio,
|
|
1704
|
+
(single_row_evidence.confidence if single_row_evidence is not None else 1.0),
|
|
1705
|
+
(single_column_evidence.confidence if single_column_evidence is not None else 1.0),
|
|
1706
|
+
)
|
|
1707
|
+
occupied_rows = _occupied_text_rows(text, y_tracks)
|
|
1708
|
+
line_row_evidence: tuple[_PhysicalRowEvidence, ...] = ()
|
|
1709
|
+
physically_verified_rows: set[int] = set()
|
|
1710
|
+
if is_line_grid:
|
|
1711
|
+
line_row_evidence = _line_grid_row_evidence(
|
|
1712
|
+
rules,
|
|
1713
|
+
canonical_x_tracks,
|
|
1714
|
+
canonical_y_tracks,
|
|
1715
|
+
text,
|
|
1716
|
+
snap_tolerance,
|
|
1717
|
+
narrow_empty_threshold,
|
|
1718
|
+
local_width,
|
|
1719
|
+
)
|
|
1720
|
+
physically_verified_rows = {evidence.row for evidence in line_row_evidence if evidence.verified}
|
|
1721
|
+
if diagnostics is not None:
|
|
1722
|
+
diagnostics["physical_rows"] = [
|
|
1723
|
+
{
|
|
1724
|
+
"row": evidence.row,
|
|
1725
|
+
"reliability": evidence.reliability,
|
|
1726
|
+
"verified": evidence.verified,
|
|
1727
|
+
"top": evidence.top_coverage,
|
|
1728
|
+
"bottom": evidence.bottom_coverage,
|
|
1729
|
+
"left": evidence.left_coverage,
|
|
1730
|
+
"right": evidence.right_coverage,
|
|
1731
|
+
"height_ratio": evidence.height_ratio,
|
|
1732
|
+
"glyph_crossing": evidence.glyph_crossing,
|
|
1733
|
+
}
|
|
1734
|
+
for evidence in line_row_evidence
|
|
1735
|
+
]
|
|
1736
|
+
row_stability, column_stability = _text_grid_stability(
|
|
1737
|
+
text,
|
|
1738
|
+
x_tracks,
|
|
1739
|
+
y_tracks,
|
|
1740
|
+
physically_verified_rows=(physically_verified_rows if is_line_grid else None),
|
|
1741
|
+
)
|
|
1742
|
+
collapsed_track_count = sum(len(track.aliases) > 1 for track in canonical_x_tracks)
|
|
1743
|
+
collapsed_y_track_count = sum(len(track.aliases) > 1 for track in canonical_y_tracks)
|
|
1744
|
+
maximum_alias_span = max(
|
|
1745
|
+
(
|
|
1746
|
+
track.aliases[-1] - track.aliases[0]
|
|
1747
|
+
for track in (*canonical_x_tracks, *canonical_y_tracks)
|
|
1748
|
+
if len(track.aliases) > 1
|
|
1749
|
+
),
|
|
1750
|
+
default=0.0,
|
|
1751
|
+
)
|
|
1752
|
+
potential_blank_rows = sorted(set(range(rows)) - occupied_rows)
|
|
1753
|
+
candidate = build_candidate(
|
|
1754
|
+
source="vector_grid",
|
|
1755
|
+
rows=rows,
|
|
1756
|
+
cols=cols,
|
|
1757
|
+
specs=specs,
|
|
1758
|
+
text=text,
|
|
1759
|
+
structure_support=structure_support,
|
|
1760
|
+
row_stability=row_stability,
|
|
1761
|
+
column_stability=column_stability,
|
|
1762
|
+
issues=(
|
|
1763
|
+
f"evidence={evidence_label}",
|
|
1764
|
+
f"ambiguous_separator_ratio={ambiguous_ratio:.4f}",
|
|
1765
|
+
f"collapsed_x_tracks={collapsed_track_count}",
|
|
1766
|
+
f"collapsed_y_tracks={collapsed_y_track_count}",
|
|
1767
|
+
f"alias_max_span={maximum_alias_span:.4f}",
|
|
1768
|
+
f"alias_separator_recoveries={alias_separator_recoveries}",
|
|
1769
|
+
f"y_alias_separator_recoveries={y_alias_separator_recoveries}",
|
|
1770
|
+
f"single_row_line_grid={str(is_single_row_shape).lower()}",
|
|
1771
|
+
f"single_column_line_grid={str(is_single_column_shape).lower()}",
|
|
1772
|
+
"single_row_reliability="
|
|
1773
|
+
+ (f"{single_row_evidence.reliability:.4f}" if single_row_evidence is not None else "n/a"),
|
|
1774
|
+
"single_row_confidence=" + (f"{single_row_evidence.confidence:.4f}" if single_row_evidence is not None else "n/a"),
|
|
1775
|
+
"single_column_reliability="
|
|
1776
|
+
+ (f"{single_column_evidence.reliability:.4f}" if single_column_evidence is not None else "n/a"),
|
|
1777
|
+
"single_column_confidence="
|
|
1778
|
+
+ (f"{single_column_evidence.confidence:.4f}" if single_column_evidence is not None else "n/a"),
|
|
1779
|
+
"physical_blank_rows=" + ",".join(str(row) for row in potential_blank_rows if row in physically_verified_rows),
|
|
1780
|
+
),
|
|
1781
|
+
allow_single_row=is_single_row_shape,
|
|
1782
|
+
allow_single_column=is_single_column_shape,
|
|
1783
|
+
use_grid_index=True,
|
|
1784
|
+
diagnostics=diagnostics,
|
|
1785
|
+
)
|
|
1786
|
+
if candidate is None:
|
|
1787
|
+
candidate_gate = (
|
|
1788
|
+
str(diagnostics.get("candidate_rejection_gate"))
|
|
1789
|
+
if diagnostics is not None and diagnostics.get("candidate_rejection_gate")
|
|
1790
|
+
else "candidate_hard_gate"
|
|
1791
|
+
)
|
|
1792
|
+
return _reject_vector_candidate(diagnostics, candidate_gate)
|
|
1793
|
+
if is_single_row_shape and (candidate.text_capture < 1.0 or candidate.order_consistency < 1.0):
|
|
1794
|
+
return _reject_vector_candidate(diagnostics, "single_row_text_integrity")
|
|
1795
|
+
if is_single_column_shape and (candidate.text_capture < 1.0 or candidate.order_consistency < 1.0):
|
|
1796
|
+
return _reject_vector_candidate(diagnostics, "single_column_text_integrity")
|
|
1797
|
+
row_content_support = [
|
|
1798
|
+
sum(bool(cell.content.strip()) for cell in candidate.cells if cell.row <= row_index < cell.row + cell.rowspan)
|
|
1799
|
+
for row_index in range(candidate.rows)
|
|
1800
|
+
]
|
|
1801
|
+
empty_rows = {row_index for row_index, support in enumerate(row_content_support) if support == 0}
|
|
1802
|
+
if empty_rows and len(text.rows) >= 2:
|
|
1803
|
+
if (
|
|
1804
|
+
not is_line_grid
|
|
1805
|
+
or not empty_rows.isdisjoint(alias_affected_rows)
|
|
1806
|
+
or not empty_rows.issubset(physically_verified_rows)
|
|
1807
|
+
):
|
|
1808
|
+
if diagnostics is not None:
|
|
1809
|
+
diagnostics["empty_rows"] = sorted(empty_rows)
|
|
1810
|
+
return _reject_vector_candidate(diagnostics, "empty_row")
|
|
1811
|
+
if diagnostics is not None:
|
|
1812
|
+
diagnostics["first_rejection_gate"] = None
|
|
1813
|
+
diagnostics["score"] = candidate.score
|
|
1814
|
+
diagnostics["empty_rows"] = sorted(empty_rows)
|
|
1815
|
+
return candidate
|
|
1816
|
+
|
|
1817
|
+
|
|
1818
|
+
def _build_vector_candidate(
|
|
1819
|
+
table_input: NativeTableInput,
|
|
1820
|
+
text: NativeTableText,
|
|
1821
|
+
*,
|
|
1822
|
+
include_drawing: bool,
|
|
1823
|
+
include_rectangles: bool,
|
|
1824
|
+
evidence_label: str,
|
|
1825
|
+
prune_unsupported_horizontal: bool = False,
|
|
1826
|
+
diagnostics: dict[str, Any] | None = None,
|
|
1827
|
+
) -> NativeTableCandidate | None:
|
|
1828
|
+
"""按轨道、拓扑、文本落格及评分的固定顺序构造矢量候选。"""
|
|
1829
|
+
|
|
1830
|
+
if diagnostics is not None:
|
|
1831
|
+
diagnostics["evidence"] = evidence_label
|
|
1832
|
+
tracks = _build_vector_tracks(
|
|
1833
|
+
table_input,
|
|
1834
|
+
text,
|
|
1835
|
+
include_drawing=include_drawing,
|
|
1836
|
+
include_rectangles=include_rectangles,
|
|
1837
|
+
prune_unsupported_horizontal=prune_unsupported_horizontal,
|
|
1838
|
+
diagnostics=diagnostics,
|
|
1839
|
+
)
|
|
1840
|
+
if tracks is None:
|
|
1841
|
+
return None
|
|
1842
|
+
topology = _build_vector_topology(tracks, text, diagnostics)
|
|
1843
|
+
if topology is None:
|
|
1844
|
+
return None
|
|
1845
|
+
return _materialize_vector_candidate(tracks, topology, text, evidence_label, diagnostics)
|
|
1846
|
+
|
|
1847
|
+
|
|
1848
|
+
def build_vector_candidates(
|
|
1849
|
+
table_input: NativeTableInput,
|
|
1850
|
+
text: NativeTableText,
|
|
1851
|
+
diagnostics: list[dict[str, Any]] | None = None,
|
|
1852
|
+
) -> list[NativeTableCandidate]:
|
|
1853
|
+
"""分别从 drawing 中心线和矩形晶格生成矢量网格候选。"""
|
|
1854
|
+
|
|
1855
|
+
candidates: list[NativeTableCandidate] = []
|
|
1856
|
+
raw_line_diagnostics: dict[str, Any] | None = {} if diagnostics is not None else None
|
|
1857
|
+
line_candidate = _build_vector_candidate(
|
|
1858
|
+
table_input,
|
|
1859
|
+
text,
|
|
1860
|
+
include_drawing=True,
|
|
1861
|
+
include_rectangles=False,
|
|
1862
|
+
evidence_label="line_grid",
|
|
1863
|
+
diagnostics=raw_line_diagnostics,
|
|
1864
|
+
)
|
|
1865
|
+
line_hypotheses = [raw_line_diagnostics] if raw_line_diagnostics is not None else []
|
|
1866
|
+
selected_line_diagnostics = raw_line_diagnostics
|
|
1867
|
+
if line_candidate is None and len(line_hypotheses) < MAX_TRACK_HYPOTHESES:
|
|
1868
|
+
supported_line_diagnostics: dict[str, Any] | None = {} if diagnostics is not None else None
|
|
1869
|
+
supported_line_candidate = _build_vector_candidate(
|
|
1870
|
+
table_input,
|
|
1871
|
+
text,
|
|
1872
|
+
include_drawing=True,
|
|
1873
|
+
include_rectangles=False,
|
|
1874
|
+
evidence_label="line_grid",
|
|
1875
|
+
prune_unsupported_horizontal=True,
|
|
1876
|
+
diagnostics=supported_line_diagnostics,
|
|
1877
|
+
)
|
|
1878
|
+
if supported_line_diagnostics is not None:
|
|
1879
|
+
removed_tracks = supported_line_diagnostics.get(
|
|
1880
|
+
"removed_horizontal_tracks",
|
|
1881
|
+
[],
|
|
1882
|
+
)
|
|
1883
|
+
if removed_tracks:
|
|
1884
|
+
line_hypotheses.append(supported_line_diagnostics)
|
|
1885
|
+
selected_line_diagnostics = supported_line_diagnostics
|
|
1886
|
+
if supported_line_candidate is not None:
|
|
1887
|
+
line_candidate = supported_line_candidate
|
|
1888
|
+
selected_line_diagnostics = supported_line_diagnostics
|
|
1889
|
+
if diagnostics is not None and selected_line_diagnostics is not None:
|
|
1890
|
+
line_record = dict(selected_line_diagnostics)
|
|
1891
|
+
line_record["track_hypotheses"] = [dict(hypothesis) for hypothesis in line_hypotheses if hypothesis is not None]
|
|
1892
|
+
diagnostics.append(line_record)
|
|
1893
|
+
if line_candidate is not None:
|
|
1894
|
+
candidates.append(line_candidate)
|
|
1895
|
+
rect_diagnostics: dict[str, Any] | None = {} if diagnostics is not None else None
|
|
1896
|
+
rect_candidate = _build_vector_candidate(
|
|
1897
|
+
table_input,
|
|
1898
|
+
text,
|
|
1899
|
+
include_drawing=False,
|
|
1900
|
+
include_rectangles=True,
|
|
1901
|
+
evidence_label="rect_grid",
|
|
1902
|
+
diagnostics=rect_diagnostics,
|
|
1903
|
+
)
|
|
1904
|
+
if diagnostics is not None and rect_diagnostics is not None:
|
|
1905
|
+
diagnostics.append(rect_diagnostics)
|
|
1906
|
+
if rect_candidate is not None:
|
|
1907
|
+
candidates.append(rect_candidate)
|
|
1908
|
+
return candidates
|
|
1909
|
+
|
|
1910
|
+
|
|
1911
|
+
def diagnose_vector_candidate_builds(
|
|
1912
|
+
table_input: NativeTableInput,
|
|
1913
|
+
text: NativeTableText,
|
|
1914
|
+
) -> tuple[dict[str, Any], ...]:
|
|
1915
|
+
"""返回 line/rect 假设真实首个拒绝门和物理证据。"""
|
|
1916
|
+
|
|
1917
|
+
diagnostics: list[dict[str, Any]] = []
|
|
1918
|
+
build_vector_candidates(
|
|
1919
|
+
table_input,
|
|
1920
|
+
text,
|
|
1921
|
+
diagnostics=diagnostics,
|
|
1922
|
+
)
|
|
1923
|
+
return tuple(diagnostics)
|
|
1924
|
+
|
|
1925
|
+
|
|
1926
|
+
__all__ = [
|
|
1927
|
+
"MAX_ATOMIC_CELLS",
|
|
1928
|
+
"MAX_PRIMITIVES_PER_TABLE",
|
|
1929
|
+
"MAX_TRACKS_PER_AXIS",
|
|
1930
|
+
"build_vector_candidates",
|
|
1931
|
+
]
|