docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,715 @@
|
|
|
1
|
+
"""视觉主体、标题和脚注的 raw block 关联与分组。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from ..content.spans import inline_span_plain_text
|
|
9
|
+
from ..schema import BBox, BlockType, VISUAL_MAIN_TYPES, VISUAL_RELATION_IGNORED_TYPES, VISUAL_TYPE_MAPPING
|
|
10
|
+
from ..foundation.geometry import bbox_center_distance, bbox_distance
|
|
11
|
+
from ..schema import RAW_CAPTION, RAW_FOOTNOTE
|
|
12
|
+
from ..content.table.rules import is_table_continuation_text
|
|
13
|
+
|
|
14
|
+
INLINE_CAPTION_FRAGMENT_TYPES = {BlockType.TEXT, RAW_FOOTNOTE}
|
|
15
|
+
STACKED_TABLE_CAPTION_CLUSTER_TYPES = {
|
|
16
|
+
RAW_CAPTION,
|
|
17
|
+
BlockType.TEXT,
|
|
18
|
+
RAW_FOOTNOTE,
|
|
19
|
+
}
|
|
20
|
+
VISUAL_CHILD_TYPE_MAPPING: dict[str, tuple[str | None, str]] = {
|
|
21
|
+
RAW_CAPTION: (None, "caption"),
|
|
22
|
+
RAW_FOOTNOTE: (None, "footnote"),
|
|
23
|
+
BlockType.IMAGE_CAPTION: (BlockType.IMAGE, "caption"),
|
|
24
|
+
BlockType.IMAGE_FOOTNOTE: (BlockType.IMAGE, "footnote"),
|
|
25
|
+
BlockType.TABLE_CAPTION: (BlockType.TABLE, "caption"),
|
|
26
|
+
BlockType.TABLE_FOOTNOTE: (BlockType.TABLE, "footnote"),
|
|
27
|
+
BlockType.CHART_CAPTION: (BlockType.CHART, "caption"),
|
|
28
|
+
BlockType.CHART_FOOTNOTE: (BlockType.CHART, "footnote"),
|
|
29
|
+
BlockType.CODE_CAPTION: (BlockType.CODE, "caption"),
|
|
30
|
+
BlockType.CODE_FOOTNOTE: (BlockType.CODE, "footnote"),
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
BlockDict = dict[str, Any]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _get_block_field(block: BlockDict, field_name: str, default: Any = None) -> Any:
|
|
37
|
+
"""读取 raw dict block 字段。"""
|
|
38
|
+
return block.get(field_name, default)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _set_block_field(block: BlockDict, field_name: str, value: Any) -> None:
|
|
42
|
+
"""回写 raw dict block 字段。"""
|
|
43
|
+
block[field_name] = value
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _block_index(block: BlockDict) -> int:
|
|
47
|
+
"""读取 block 顺序索引,缺失时沿用原有的零值排序语义。"""
|
|
48
|
+
return int(_get_block_field(block, "index", 0) or 0)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _block_type(block: BlockDict) -> str:
|
|
52
|
+
"""读取 block 类型。"""
|
|
53
|
+
return str(_get_block_field(block, "type", "") or "")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _bbox_for_calculation(bbox: BBox) -> BBox:
|
|
57
|
+
"""返回仅用于计算的 bbox 副本,归一化坐标临时放大一千倍。"""
|
|
58
|
+
x0, y0, x1, y1 = bbox
|
|
59
|
+
if all(value <= 1 for value in (x0, y0, x1, y1)):
|
|
60
|
+
return x0 * 1000, y0 * 1000, x1 * 1000, y1 * 1000
|
|
61
|
+
return x0, y0, x1, y1
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _block_bbox(block: BlockDict) -> BBox:
|
|
65
|
+
"""读取参与视觉关系判断的 block bbox。"""
|
|
66
|
+
return _bbox_for_calculation(block["bbox"])
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _block_line_items(block: BlockDict) -> list[Any]:
|
|
70
|
+
"""读取 raw block 的临时行级元数据。"""
|
|
71
|
+
return list(block.get("lines") or [])
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def fallback_inline_caption_fragments(
|
|
75
|
+
blocks: list[BlockDict],
|
|
76
|
+
visual_main_types: dict[str, str] | set[str],
|
|
77
|
+
) -> None:
|
|
78
|
+
"""将紧贴视觉主体上方的同行 text/footnote 片段兜底为通用 caption。"""
|
|
79
|
+
if len(blocks) < 3:
|
|
80
|
+
return
|
|
81
|
+
|
|
82
|
+
main_types = set(visual_main_types)
|
|
83
|
+
ordered_blocks = sorted(blocks, key=_block_index)
|
|
84
|
+
for pos, block in enumerate(ordered_blocks):
|
|
85
|
+
if _block_type(block) not in INLINE_CAPTION_FRAGMENT_TYPES:
|
|
86
|
+
continue
|
|
87
|
+
|
|
88
|
+
previous_block = find_previous_effective_block(ordered_blocks, pos)
|
|
89
|
+
next_block = find_next_effective_block(ordered_blocks, pos)
|
|
90
|
+
if not (
|
|
91
|
+
previous_block
|
|
92
|
+
and next_block
|
|
93
|
+
and _block_type(previous_block) == RAW_CAPTION
|
|
94
|
+
and _block_type(next_block) in main_types
|
|
95
|
+
):
|
|
96
|
+
continue
|
|
97
|
+
|
|
98
|
+
if not is_inline_caption_fragment(previous_block, block, next_block):
|
|
99
|
+
continue
|
|
100
|
+
|
|
101
|
+
_set_block_field(block, "type", RAW_CAPTION)
|
|
102
|
+
|
|
103
|
+
fallback_stacked_table_caption_fragments(blocks, visual_main_types)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def fallback_no_bbox_caption_fragments(
|
|
107
|
+
blocks: list[BlockDict],
|
|
108
|
+
visual_main_types: dict[str, str],
|
|
109
|
+
) -> None:
|
|
110
|
+
"""将无坐标视觉主体后紧邻且带标题前缀的 text 兜底为 caption。"""
|
|
111
|
+
caption_prefixes = {
|
|
112
|
+
BlockType.TABLE: ("表", "table"),
|
|
113
|
+
BlockType.IMAGE: ("图", "fig"),
|
|
114
|
+
BlockType.CHART: ("图", "fig", "chart"),
|
|
115
|
+
}
|
|
116
|
+
ordered_blocks = sorted(blocks, key=_block_index)
|
|
117
|
+
for pos, block in enumerate(ordered_blocks[:-1]):
|
|
118
|
+
visual_type = visual_main_types.get(_block_type(block))
|
|
119
|
+
prefixes = caption_prefixes.get(visual_type)
|
|
120
|
+
if not prefixes:
|
|
121
|
+
continue
|
|
122
|
+
|
|
123
|
+
next_block = ordered_blocks[pos + 1]
|
|
124
|
+
if _block_type(next_block) != BlockType.TEXT:
|
|
125
|
+
continue
|
|
126
|
+
|
|
127
|
+
content = _block_text_content(next_block).strip().lower()
|
|
128
|
+
if any(content.startswith(prefix) for prefix in prefixes):
|
|
129
|
+
_set_block_field(next_block, "type", RAW_CAPTION)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def fallback_leading_table_continuation_captions(
|
|
133
|
+
blocks: list[BlockDict],
|
|
134
|
+
visual_main_types: dict[Any, Any] | set[Any],
|
|
135
|
+
) -> None:
|
|
136
|
+
"""将页首紧贴表格的续表文本兜底为通用 caption。
|
|
137
|
+
|
|
138
|
+
该规则只处理页面有效块开头的 text,避免正文中出现“续表”时被误挂。
|
|
139
|
+
后续 regroup_visual_blocks() 会根据表格主体类型将通用 caption 落成
|
|
140
|
+
table_caption 子块。
|
|
141
|
+
"""
|
|
142
|
+
table_main_types = get_table_main_types(visual_main_types)
|
|
143
|
+
if not table_main_types:
|
|
144
|
+
return
|
|
145
|
+
|
|
146
|
+
effective_blocks = [
|
|
147
|
+
block for block in sorted(blocks, key=_block_index) if _block_type(block) not in VISUAL_RELATION_IGNORED_TYPES
|
|
148
|
+
]
|
|
149
|
+
if len(effective_blocks) < 2:
|
|
150
|
+
return
|
|
151
|
+
|
|
152
|
+
leading_blocks = []
|
|
153
|
+
cursor = 0
|
|
154
|
+
while cursor < len(effective_blocks):
|
|
155
|
+
block = effective_blocks[cursor]
|
|
156
|
+
if not _is_leading_continuation_text_block(block):
|
|
157
|
+
break
|
|
158
|
+
leading_blocks.append(block)
|
|
159
|
+
cursor += 1
|
|
160
|
+
|
|
161
|
+
if not leading_blocks or cursor >= len(effective_blocks):
|
|
162
|
+
return
|
|
163
|
+
|
|
164
|
+
table_block = effective_blocks[cursor]
|
|
165
|
+
if _block_type(table_block) not in table_main_types:
|
|
166
|
+
return
|
|
167
|
+
|
|
168
|
+
if not _is_leading_continuation_cluster_near_table(leading_blocks, table_block):
|
|
169
|
+
return
|
|
170
|
+
|
|
171
|
+
for block in leading_blocks:
|
|
172
|
+
_set_block_field(block, "type", RAW_CAPTION)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _is_leading_continuation_text_block(block: BlockDict) -> bool:
|
|
176
|
+
"""判断页首候选块是否是单行续表文本。"""
|
|
177
|
+
return (
|
|
178
|
+
_block_type(block) in INLINE_CAPTION_FRAGMENT_TYPES
|
|
179
|
+
and is_single_line_caption_fragment(block)
|
|
180
|
+
and is_table_continuation_text(_block_text_content(block))
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _block_text_content(block: BlockDict) -> str:
|
|
185
|
+
"""提取视觉块中的可见文本,用于续表 marker 判断。"""
|
|
186
|
+
content = _get_block_field(block, "content", "")
|
|
187
|
+
if isinstance(content, list):
|
|
188
|
+
return inline_span_plain_text(item for item in content if isinstance(item, dict))
|
|
189
|
+
return content if isinstance(content, str) else ""
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def is_transparent_visual_relation_block(block: BlockDict) -> bool:
|
|
193
|
+
"""判断视觉关系中可忽略的结构性空块。"""
|
|
194
|
+
if _block_type(block) != BlockType.LIST:
|
|
195
|
+
return False
|
|
196
|
+
|
|
197
|
+
content = block.get("content")
|
|
198
|
+
nested_blocks = content if isinstance(content, list) else []
|
|
199
|
+
if nested_blocks:
|
|
200
|
+
return False
|
|
201
|
+
|
|
202
|
+
return not _block_text_content(block).strip()
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _is_leading_continuation_cluster_near_table(
|
|
206
|
+
leading_blocks: list[BlockDict],
|
|
207
|
+
table_block: BlockDict,
|
|
208
|
+
) -> bool:
|
|
209
|
+
"""判断页首续表文本簇是否与后续 table 在几何上相邻。"""
|
|
210
|
+
next_top = _block_bbox(table_block)[1]
|
|
211
|
+
max_child_height = 1
|
|
212
|
+
|
|
213
|
+
for block in reversed(leading_blocks):
|
|
214
|
+
if not is_horizontally_near_table(block, table_block):
|
|
215
|
+
return False
|
|
216
|
+
|
|
217
|
+
block_bbox = _block_bbox(block)
|
|
218
|
+
block_height = max(block_bbox[3] - block_bbox[1], 1)
|
|
219
|
+
vertical_gap = next_top - block_bbox[3]
|
|
220
|
+
max_gap = stacked_caption_max_gap(max(max_child_height, block_height))
|
|
221
|
+
max_overlap = max(2, block_height * 0.3)
|
|
222
|
+
if vertical_gap > max_gap or vertical_gap < -max_overlap:
|
|
223
|
+
return False
|
|
224
|
+
|
|
225
|
+
next_top = block_bbox[1]
|
|
226
|
+
max_child_height = max(max_child_height, block_height)
|
|
227
|
+
|
|
228
|
+
return True
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def fallback_stacked_table_caption_fragments(
|
|
232
|
+
blocks: list[BlockDict],
|
|
233
|
+
visual_main_types: dict[str, str] | set[str],
|
|
234
|
+
) -> None:
|
|
235
|
+
"""将 table 上方紧贴标题簇里的 text/footnote 片段兜底为 caption。"""
|
|
236
|
+
table_main_types = get_table_main_types(visual_main_types)
|
|
237
|
+
if not table_main_types:
|
|
238
|
+
return
|
|
239
|
+
|
|
240
|
+
for table_block in blocks:
|
|
241
|
+
if _block_type(table_block) not in table_main_types:
|
|
242
|
+
continue
|
|
243
|
+
|
|
244
|
+
caption_cluster = find_stacked_table_caption_cluster(table_block, blocks)
|
|
245
|
+
if not caption_cluster:
|
|
246
|
+
continue
|
|
247
|
+
|
|
248
|
+
last_caption_pos = find_last_caption_position(caption_cluster)
|
|
249
|
+
if last_caption_pos is None:
|
|
250
|
+
continue
|
|
251
|
+
|
|
252
|
+
for block in caption_cluster[last_caption_pos + 1 :]:
|
|
253
|
+
if _block_type(block) in INLINE_CAPTION_FRAGMENT_TYPES and is_single_line_caption_fragment(block):
|
|
254
|
+
_set_block_field(block, "type", RAW_CAPTION)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def get_table_main_types(visual_main_types: dict[str, str] | set[str]) -> set[str]:
|
|
258
|
+
"""根据调用方传入的视觉主体类型,找出 table 对应的主体类型。"""
|
|
259
|
+
if isinstance(visual_main_types, dict):
|
|
260
|
+
return {block_type for block_type, visual_type in visual_main_types.items() if visual_type == BlockType.TABLE}
|
|
261
|
+
|
|
262
|
+
main_types = set(visual_main_types)
|
|
263
|
+
return main_types & {BlockType.TABLE, BlockType.TABLE_BODY}
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def find_stacked_table_caption_cluster(
|
|
267
|
+
table_block: BlockDict,
|
|
268
|
+
blocks: list[BlockDict],
|
|
269
|
+
) -> list[BlockDict]:
|
|
270
|
+
"""按几何位置收集紧贴 table 上方的 caption/text/footnote 标题簇。"""
|
|
271
|
+
table_bbox = _block_bbox(table_block)
|
|
272
|
+
table_top = table_bbox[1]
|
|
273
|
+
above_candidates = [
|
|
274
|
+
block
|
|
275
|
+
for block in blocks
|
|
276
|
+
if block is not table_block
|
|
277
|
+
and _block_type(block) in STACKED_TABLE_CAPTION_CLUSTER_TYPES
|
|
278
|
+
and _block_bbox(block)[3] <= table_top
|
|
279
|
+
and is_horizontally_near_table(block, table_block)
|
|
280
|
+
]
|
|
281
|
+
if not above_candidates:
|
|
282
|
+
return []
|
|
283
|
+
|
|
284
|
+
caption_cluster = []
|
|
285
|
+
next_top = table_top
|
|
286
|
+
max_child_height = 1
|
|
287
|
+
for block in sorted(
|
|
288
|
+
above_candidates,
|
|
289
|
+
key=lambda x: (_block_bbox(x)[1], _block_index(x)),
|
|
290
|
+
reverse=True,
|
|
291
|
+
):
|
|
292
|
+
block_bbox = _block_bbox(block)
|
|
293
|
+
block_height = max(block_bbox[3] - block_bbox[1], 1)
|
|
294
|
+
max_allowed_gap = stacked_caption_max_gap(max(max_child_height, block_height))
|
|
295
|
+
vertical_gap = next_top - block_bbox[3]
|
|
296
|
+
if not 0 <= vertical_gap <= max_allowed_gap:
|
|
297
|
+
break
|
|
298
|
+
|
|
299
|
+
caption_cluster.append(block)
|
|
300
|
+
next_top = block_bbox[1]
|
|
301
|
+
max_child_height = max(max_child_height, block_height)
|
|
302
|
+
|
|
303
|
+
return list(reversed(caption_cluster))
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def find_last_caption_position(caption_cluster: list[BlockDict]) -> int | None:
|
|
307
|
+
"""定位标题簇里的最后一个 caption,避免吸收上一张表的尾注。"""
|
|
308
|
+
for pos in range(len(caption_cluster) - 1, -1, -1):
|
|
309
|
+
if _block_type(caption_cluster[pos]) == RAW_CAPTION:
|
|
310
|
+
return pos
|
|
311
|
+
return None
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def is_horizontally_near_table(block: BlockDict, table_block: BlockDict) -> bool:
|
|
315
|
+
"""判断标题簇候选块是否横向落在 table 范围附近。"""
|
|
316
|
+
table_bbox = _block_bbox(table_block)
|
|
317
|
+
block_bbox = _block_bbox(block)
|
|
318
|
+
table_width = max(table_bbox[2] - table_bbox[0], 1)
|
|
319
|
+
tolerance = max(12, table_width * 0.03)
|
|
320
|
+
return not (block_bbox[2] < table_bbox[0] - tolerance or block_bbox[0] > table_bbox[2] + tolerance)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def is_single_line_caption_fragment(block: BlockDict) -> bool:
|
|
324
|
+
"""判断待兜底片段是否是单行块,避免吞掉多行正文。"""
|
|
325
|
+
return len(_block_line_items(block) or [None]) <= 1
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def stacked_caption_max_gap(block_height: float) -> float:
|
|
329
|
+
"""计算堆叠标题簇允许的最大纵向间距。"""
|
|
330
|
+
return max(12, block_height * 1.5)
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def find_previous_effective_block(ordered_blocks: list[BlockDict], pos: int) -> BlockDict | None:
|
|
334
|
+
"""向前查找参与视觉关系判断的有效块,跳过页眉页脚等外围块。"""
|
|
335
|
+
for index in range(pos - 1, -1, -1):
|
|
336
|
+
block = ordered_blocks[index]
|
|
337
|
+
if _block_type(block) not in VISUAL_RELATION_IGNORED_TYPES:
|
|
338
|
+
return block
|
|
339
|
+
return None
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def find_next_effective_block(ordered_blocks: list[BlockDict], pos: int) -> BlockDict | None:
|
|
343
|
+
"""向后查找参与视觉关系判断的有效块,跳过页眉页脚等外围块。"""
|
|
344
|
+
for index in range(pos + 1, len(ordered_blocks)):
|
|
345
|
+
block = ordered_blocks[index]
|
|
346
|
+
if _block_type(block) not in VISUAL_RELATION_IGNORED_TYPES:
|
|
347
|
+
return block
|
|
348
|
+
return None
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def is_inline_caption_fragment(
|
|
352
|
+
previous_caption: BlockDict,
|
|
353
|
+
text_block: BlockDict,
|
|
354
|
+
next_visual: BlockDict,
|
|
355
|
+
) -> bool:
|
|
356
|
+
"""判断当前块是否是前一 caption 的同行补充片段。"""
|
|
357
|
+
caption_bbox = _block_bbox(previous_caption)
|
|
358
|
+
text_bbox = _block_bbox(text_block)
|
|
359
|
+
visual_bbox = _block_bbox(next_visual)
|
|
360
|
+
|
|
361
|
+
caption_height = max(caption_bbox[3] - caption_bbox[1], 1)
|
|
362
|
+
text_height = max(text_bbox[3] - text_bbox[1], 1)
|
|
363
|
+
min_text_height = max(min(caption_height, text_height), 1)
|
|
364
|
+
|
|
365
|
+
vertical_overlap = min(caption_bbox[3], text_bbox[3]) - max(caption_bbox[1], text_bbox[1])
|
|
366
|
+
center_y_diff = abs(((caption_bbox[1] + caption_bbox[3]) / 2) - ((text_bbox[1] + text_bbox[3]) / 2))
|
|
367
|
+
is_same_line = vertical_overlap / min_text_height >= 0.6 or center_y_diff <= max(caption_height, text_height) * 0.5
|
|
368
|
+
if not is_same_line:
|
|
369
|
+
return False
|
|
370
|
+
|
|
371
|
+
vertical_gap_to_visual = visual_bbox[1] - max(caption_bbox[3], text_bbox[3])
|
|
372
|
+
max_allowed_gap = max(12, max(caption_height, text_height) * 1.5)
|
|
373
|
+
return 0 <= vertical_gap_to_visual <= max_allowed_gap
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def regroup_visual_blocks(
|
|
377
|
+
blocks: list[BlockDict],
|
|
378
|
+
*,
|
|
379
|
+
use_bbox: bool = True,
|
|
380
|
+
) -> tuple[dict[Any, list[BlockDict]], list[BlockDict]]:
|
|
381
|
+
"""按 bbox 或纯阅读顺序将通用 caption/footnote 归入视觉主体。"""
|
|
382
|
+
ordered_blocks = sorted(blocks, key=_block_index)
|
|
383
|
+
if use_bbox:
|
|
384
|
+
visual_relation_blocks = [block for block in ordered_blocks if not is_transparent_visual_relation_block(block)]
|
|
385
|
+
else:
|
|
386
|
+
visual_relation_blocks = ordered_blocks
|
|
387
|
+
position_by_index = {_block_index(block): pos for pos, block in enumerate(visual_relation_blocks)}
|
|
388
|
+
main_blocks = [block for block in visual_relation_blocks if _block_type(block) in VISUAL_MAIN_TYPES]
|
|
389
|
+
child_blocks = [block for block in visual_relation_blocks if _block_type(block) in VISUAL_CHILD_TYPE_MAPPING]
|
|
390
|
+
|
|
391
|
+
grouped_children: dict[int, dict[str, list[BlockDict]]] = {
|
|
392
|
+
_block_index(block): {"captions": [], "footnotes": []} for block in main_blocks
|
|
393
|
+
}
|
|
394
|
+
unmatched_child_blocks: list[BlockDict] = []
|
|
395
|
+
|
|
396
|
+
for child_block in child_blocks:
|
|
397
|
+
child_visual_type, child_kind = VISUAL_CHILD_TYPE_MAPPING[_block_type(child_block)]
|
|
398
|
+
candidate_main_blocks = [
|
|
399
|
+
main_block
|
|
400
|
+
for main_block in main_blocks
|
|
401
|
+
if child_visual_type is None or VISUAL_MAIN_TYPES[_block_type(main_block)] == child_visual_type
|
|
402
|
+
]
|
|
403
|
+
parent_block = find_best_visual_parent(
|
|
404
|
+
child_block,
|
|
405
|
+
candidate_main_blocks,
|
|
406
|
+
visual_relation_blocks,
|
|
407
|
+
position_by_index,
|
|
408
|
+
use_bbox=use_bbox,
|
|
409
|
+
)
|
|
410
|
+
if parent_block is None:
|
|
411
|
+
unmatched_child_blocks.append(child_block)
|
|
412
|
+
continue
|
|
413
|
+
|
|
414
|
+
grouped_children[_block_index(parent_block)][f"{child_kind}s"].append(child_block)
|
|
415
|
+
|
|
416
|
+
grouped_blocks: dict[str, list[BlockDict]] = {
|
|
417
|
+
BlockType.IMAGE: [],
|
|
418
|
+
BlockType.TABLE: [],
|
|
419
|
+
BlockType.CHART: [],
|
|
420
|
+
BlockType.CODE: [],
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
for main_block in main_blocks:
|
|
424
|
+
main_index = _block_index(main_block)
|
|
425
|
+
visual_type = VISUAL_MAIN_TYPES[_block_type(main_block)]
|
|
426
|
+
mapping = VISUAL_TYPE_MAPPING[visual_type]
|
|
427
|
+
main_sub_type = str(_get_block_field(main_block, "sub_type", "") or "")
|
|
428
|
+
main_guess_lang = str(_get_block_field(main_block, "guess_lang", "") or "")
|
|
429
|
+
body_block = deepcopy(main_block)
|
|
430
|
+
_set_block_field(body_block, "type", mapping["body"])
|
|
431
|
+
body_block.pop("sub_type", None)
|
|
432
|
+
body_block.pop("guess_lang", None)
|
|
433
|
+
table_cell_merge_is_set = visual_type == BlockType.TABLE and "cell_merge" in body_block
|
|
434
|
+
table_cell_merge = body_block.pop("cell_merge", None) if table_cell_merge_is_set else None
|
|
435
|
+
|
|
436
|
+
captions: list[BlockDict] = []
|
|
437
|
+
for caption in sorted(
|
|
438
|
+
grouped_children[main_index]["captions"],
|
|
439
|
+
key=_block_index,
|
|
440
|
+
):
|
|
441
|
+
child_block = deepcopy(caption)
|
|
442
|
+
_set_block_field(child_block, "type", mapping["caption"])
|
|
443
|
+
child_block.pop("lines", None)
|
|
444
|
+
captions.append(child_block)
|
|
445
|
+
|
|
446
|
+
footnotes: list[BlockDict] = []
|
|
447
|
+
for footnote in sorted(
|
|
448
|
+
grouped_children[main_index]["footnotes"],
|
|
449
|
+
key=_block_index,
|
|
450
|
+
):
|
|
451
|
+
child_block = deepcopy(footnote)
|
|
452
|
+
_set_block_field(child_block, "type", mapping["footnote"])
|
|
453
|
+
child_block.pop("lines", None)
|
|
454
|
+
footnotes.append(child_block)
|
|
455
|
+
|
|
456
|
+
child_items = sorted(
|
|
457
|
+
[body_block, *captions, *footnotes],
|
|
458
|
+
key=_block_index,
|
|
459
|
+
)
|
|
460
|
+
two_layer_block: BlockDict = {
|
|
461
|
+
"index": main_index,
|
|
462
|
+
"type": visual_type,
|
|
463
|
+
"content": child_items,
|
|
464
|
+
}
|
|
465
|
+
if main_sub_type:
|
|
466
|
+
two_layer_block["sub_type"] = main_sub_type
|
|
467
|
+
if visual_type == BlockType.CODE and main_guess_lang:
|
|
468
|
+
two_layer_block["guess_lang"] = main_guess_lang
|
|
469
|
+
if table_cell_merge_is_set:
|
|
470
|
+
two_layer_block["cell_merge"] = table_cell_merge
|
|
471
|
+
if "bbox" in main_block:
|
|
472
|
+
two_layer_block["bbox"] = deepcopy(main_block["bbox"])
|
|
473
|
+
|
|
474
|
+
grouped_blocks[visual_type].append(two_layer_block)
|
|
475
|
+
|
|
476
|
+
for blocks_of_type in grouped_blocks.values():
|
|
477
|
+
blocks_of_type.sort(key=_block_index)
|
|
478
|
+
|
|
479
|
+
return grouped_blocks, unmatched_child_blocks
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def find_best_visual_parent(
|
|
483
|
+
child_block: BlockDict,
|
|
484
|
+
main_blocks: list[BlockDict],
|
|
485
|
+
ordered_blocks: list[BlockDict],
|
|
486
|
+
position_by_index: dict[int, int],
|
|
487
|
+
main_type_to_visual_type: dict[Any, Any] | None = None,
|
|
488
|
+
type_by_index: dict[int, str] | None = None,
|
|
489
|
+
use_bbox: bool = True,
|
|
490
|
+
) -> BlockDict | None:
|
|
491
|
+
"""为通用 caption/footnote 查找最合适的视觉主体。"""
|
|
492
|
+
if main_type_to_visual_type is None:
|
|
493
|
+
main_type_to_visual_type = VISUAL_MAIN_TYPES
|
|
494
|
+
candidates = []
|
|
495
|
+
for main_block in main_blocks:
|
|
496
|
+
if not is_visual_neighbor(
|
|
497
|
+
child_block,
|
|
498
|
+
main_block,
|
|
499
|
+
ordered_blocks,
|
|
500
|
+
position_by_index,
|
|
501
|
+
type_by_index=type_by_index,
|
|
502
|
+
use_bbox=use_bbox,
|
|
503
|
+
):
|
|
504
|
+
continue
|
|
505
|
+
|
|
506
|
+
candidates.append(main_block)
|
|
507
|
+
|
|
508
|
+
if not candidates:
|
|
509
|
+
return None
|
|
510
|
+
|
|
511
|
+
min_effective_index_diff = min(
|
|
512
|
+
effective_visual_index_diff(
|
|
513
|
+
child_block,
|
|
514
|
+
main_block,
|
|
515
|
+
ordered_blocks,
|
|
516
|
+
type_by_index=type_by_index,
|
|
517
|
+
)
|
|
518
|
+
for main_block in candidates
|
|
519
|
+
)
|
|
520
|
+
closest_index_candidates = [
|
|
521
|
+
main_block
|
|
522
|
+
for main_block in candidates
|
|
523
|
+
if effective_visual_index_diff(
|
|
524
|
+
child_block,
|
|
525
|
+
main_block,
|
|
526
|
+
ordered_blocks,
|
|
527
|
+
type_by_index=type_by_index,
|
|
528
|
+
)
|
|
529
|
+
== min_effective_index_diff
|
|
530
|
+
]
|
|
531
|
+
|
|
532
|
+
if len(closest_index_candidates) == 1:
|
|
533
|
+
return closest_index_candidates[0]
|
|
534
|
+
|
|
535
|
+
if not use_bbox:
|
|
536
|
+
child_index = _block_index(child_block)
|
|
537
|
+
previous_candidates = [main_block for main_block in closest_index_candidates if _block_index(main_block) < child_index]
|
|
538
|
+
if previous_candidates:
|
|
539
|
+
return max(previous_candidates, key=_block_index)
|
|
540
|
+
return min(closest_index_candidates, key=_block_index)
|
|
541
|
+
|
|
542
|
+
child_bbox = _block_bbox(child_block)
|
|
543
|
+
edge_distances = [
|
|
544
|
+
(
|
|
545
|
+
main_block,
|
|
546
|
+
bbox_distance(child_bbox, _block_bbox(main_block)),
|
|
547
|
+
)
|
|
548
|
+
for main_block in closest_index_candidates
|
|
549
|
+
]
|
|
550
|
+
edge_values = [edge_distance for _, edge_distance in edge_distances]
|
|
551
|
+
if max(edge_values) - min(edge_values) > 2:
|
|
552
|
+
return min(
|
|
553
|
+
edge_distances,
|
|
554
|
+
key=lambda item: (item[1], _block_index(item[0])),
|
|
555
|
+
)[0]
|
|
556
|
+
|
|
557
|
+
child_kind = child_kind_from_type(block_type(child_block, type_by_index))
|
|
558
|
+
if child_kind == "caption" and all(
|
|
559
|
+
main_type_to_visual_type.get(block_type(main_block, type_by_index)) == BlockType.TABLE
|
|
560
|
+
for main_block in closest_index_candidates
|
|
561
|
+
):
|
|
562
|
+
# 表格 caption 位于两个表之间且距离接近时,优先归属后一个表。
|
|
563
|
+
return max(closest_index_candidates, key=_block_index)
|
|
564
|
+
|
|
565
|
+
if child_kind == "footnote":
|
|
566
|
+
# 视觉脚注位于两个主体之间且距离接近时,优先归属前一个主体。
|
|
567
|
+
return min(closest_index_candidates, key=_block_index)
|
|
568
|
+
|
|
569
|
+
return min(
|
|
570
|
+
closest_index_candidates,
|
|
571
|
+
key=lambda main_block: (
|
|
572
|
+
bbox_center_distance(
|
|
573
|
+
child_bbox,
|
|
574
|
+
_block_bbox(main_block),
|
|
575
|
+
),
|
|
576
|
+
_block_index(main_block),
|
|
577
|
+
),
|
|
578
|
+
)
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
def effective_visual_index_diff(
|
|
582
|
+
child_block: BlockDict,
|
|
583
|
+
main_block: BlockDict,
|
|
584
|
+
ordered_blocks: list[BlockDict],
|
|
585
|
+
type_by_index: dict[int, str] | None = None,
|
|
586
|
+
) -> int:
|
|
587
|
+
"""按有效块序列计算视觉子块与主体距离,吸收的 image 子成员视为零成本。"""
|
|
588
|
+
position_by_index = {_block_index(block): position for position, block in enumerate(ordered_blocks)}
|
|
589
|
+
child_pos = position_by_index[_block_index(child_block)]
|
|
590
|
+
main_pos = position_by_index[_block_index(main_block)]
|
|
591
|
+
start_pos = min(child_pos, main_pos)
|
|
592
|
+
end_pos = max(child_pos, main_pos)
|
|
593
|
+
skipped_child_count = 0
|
|
594
|
+
child_kind = child_kind_from_type(block_type(child_block, type_by_index))
|
|
595
|
+
|
|
596
|
+
for block in ordered_blocks[start_pos + 1 : end_pos]:
|
|
597
|
+
if child_kind_from_type(block_type(block, type_by_index)) == child_kind:
|
|
598
|
+
skipped_child_count += 1
|
|
599
|
+
|
|
600
|
+
return end_pos - start_pos - skipped_child_count
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
def is_visual_neighbor(
|
|
604
|
+
child_block: BlockDict,
|
|
605
|
+
main_block: BlockDict,
|
|
606
|
+
ordered_blocks: list[BlockDict],
|
|
607
|
+
position_by_index: dict[int, int],
|
|
608
|
+
type_by_index: dict[int, str] | None = None,
|
|
609
|
+
use_bbox: bool = True,
|
|
610
|
+
) -> bool:
|
|
611
|
+
"""判断视觉标题或脚注与主体之间是否仅隔着允许跳过的关联块。"""
|
|
612
|
+
child_kind = child_kind_from_type(block_type(child_block, type_by_index))
|
|
613
|
+
child_index = _block_index(child_block)
|
|
614
|
+
main_index = _block_index(main_block)
|
|
615
|
+
if child_kind == "footnote" and child_index < main_index:
|
|
616
|
+
return False
|
|
617
|
+
|
|
618
|
+
if child_kind == "caption":
|
|
619
|
+
allowed_between_kinds = {"caption"}
|
|
620
|
+
else:
|
|
621
|
+
allowed_between_kinds = {"caption", "footnote"}
|
|
622
|
+
|
|
623
|
+
child_pos = position_by_index[child_index]
|
|
624
|
+
main_pos = position_by_index[main_index]
|
|
625
|
+
start_pos = min(child_pos, main_pos) + 1
|
|
626
|
+
end_pos = max(child_pos, main_pos)
|
|
627
|
+
|
|
628
|
+
for pos in range(start_pos, end_pos):
|
|
629
|
+
between_block = ordered_blocks[pos]
|
|
630
|
+
between_kind = child_kind_from_type(block_type(between_block, type_by_index))
|
|
631
|
+
if between_kind in allowed_between_kinds:
|
|
632
|
+
continue
|
|
633
|
+
if use_bbox and is_block_outside_visual_gap(
|
|
634
|
+
between_block,
|
|
635
|
+
child_block,
|
|
636
|
+
main_block,
|
|
637
|
+
):
|
|
638
|
+
continue
|
|
639
|
+
return False
|
|
640
|
+
|
|
641
|
+
return True
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
def is_block_outside_visual_gap(
|
|
645
|
+
between_block: BlockDict,
|
|
646
|
+
child_block: BlockDict,
|
|
647
|
+
main_block: BlockDict,
|
|
648
|
+
) -> bool:
|
|
649
|
+
"""判断阅读顺序夹在中间的块是否没有落入视觉父子块的垂直间隔。"""
|
|
650
|
+
visual_gap = vertical_gap_between_blocks(child_block, main_block)
|
|
651
|
+
if visual_gap is None:
|
|
652
|
+
return False
|
|
653
|
+
|
|
654
|
+
if is_bbox_overlapping_visual_relation_block(
|
|
655
|
+
_block_bbox(between_block),
|
|
656
|
+
_block_bbox(child_block),
|
|
657
|
+
_block_bbox(main_block),
|
|
658
|
+
):
|
|
659
|
+
return False
|
|
660
|
+
|
|
661
|
+
if not is_bbox_intersecting_vertical_gap(_block_bbox(between_block), visual_gap):
|
|
662
|
+
return True
|
|
663
|
+
|
|
664
|
+
return False
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
def vertical_gap_between_blocks(
|
|
668
|
+
first_block: BlockDict,
|
|
669
|
+
second_block: BlockDict,
|
|
670
|
+
) -> tuple[float, float] | None:
|
|
671
|
+
"""计算两个块上下分离时的垂直间隔;发生纵向重叠时保持严格阻断。"""
|
|
672
|
+
first_bbox = _block_bbox(first_block)
|
|
673
|
+
second_bbox = _block_bbox(second_block)
|
|
674
|
+
if first_bbox[3] <= second_bbox[1]:
|
|
675
|
+
return first_bbox[3], second_bbox[1]
|
|
676
|
+
if second_bbox[3] <= first_bbox[1]:
|
|
677
|
+
return second_bbox[3], first_bbox[1]
|
|
678
|
+
return None
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def is_bbox_intersecting_vertical_gap(bbox: BBox, vertical_gap: tuple[float, float]) -> bool:
|
|
682
|
+
"""判断 bbox 是否与视觉父子块之间的垂直间隔相交。"""
|
|
683
|
+
bbox = _bbox_for_calculation(bbox)
|
|
684
|
+
gap_top, gap_bottom = vertical_gap
|
|
685
|
+
return bbox[1] < gap_bottom and bbox[3] > gap_top
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
def is_bbox_overlapping_visual_relation_block(bbox: BBox, child_bbox: BBox, main_bbox: BBox) -> bool:
|
|
689
|
+
"""判断 bbox 是否覆盖到父子块本身;覆盖时不能当作普通 index 噪声跳过。"""
|
|
690
|
+
return are_bboxes_overlapping(bbox, child_bbox) or are_bboxes_overlapping(bbox, main_bbox)
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
def are_bboxes_overlapping(first_bbox: BBox, second_bbox: BBox) -> bool:
|
|
694
|
+
"""判断两个 bbox 是否存在二维相交。"""
|
|
695
|
+
first_bbox = _bbox_for_calculation(first_bbox)
|
|
696
|
+
second_bbox = _bbox_for_calculation(second_bbox)
|
|
697
|
+
return not (
|
|
698
|
+
first_bbox[2] <= second_bbox[0]
|
|
699
|
+
or first_bbox[0] >= second_bbox[2]
|
|
700
|
+
or first_bbox[3] <= second_bbox[1]
|
|
701
|
+
or first_bbox[1] >= second_bbox[3]
|
|
702
|
+
)
|
|
703
|
+
|
|
704
|
+
|
|
705
|
+
def block_type(block: BlockDict, type_by_index: dict[int, str] | None = None) -> str:
|
|
706
|
+
"""读取块类型;本地 Hybrid 会传入改写前的原始类型映射。"""
|
|
707
|
+
if type_by_index is not None:
|
|
708
|
+
return type_by_index[_block_index(block)]
|
|
709
|
+
return _block_type(block)
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def child_kind_from_type(block_type: str) -> str | None:
|
|
713
|
+
"""读取通用或已分类视觉子块的 caption/footnote 角色。"""
|
|
714
|
+
child_mapping = VISUAL_CHILD_TYPE_MAPPING.get(block_type)
|
|
715
|
+
return child_mapping[1] if child_mapping else None
|