docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1274 @@
|
|
|
1
|
+
"""合并正文、公式上下文和列表引导块的空间组件。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import statistics
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from .....schema import BBox
|
|
10
|
+
from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_y, _bbox_union_many, _rotate_bbox_from_upright
|
|
11
|
+
from .common import (
|
|
12
|
+
_FIGURE_CAPTION_MARKER_RE,
|
|
13
|
+
_INLINE_MATH_RECOVERY_MARKER,
|
|
14
|
+
_LABELLED_METADATA_RE,
|
|
15
|
+
_LIST_ITEM_RE,
|
|
16
|
+
_PARAGRAPH_FORMULA_CONTEXT_MARKER,
|
|
17
|
+
_SHORT_SAME_BASELINE_PREFIX_RE,
|
|
18
|
+
_URL_LINE_RE,
|
|
19
|
+
_block_starts_with_short_wide_rows,
|
|
20
|
+
_compatible_component_lane_width,
|
|
21
|
+
_component_connection_skips_block,
|
|
22
|
+
_component_declared_lane_interval,
|
|
23
|
+
_component_lane_interval,
|
|
24
|
+
_components_share_lane_role,
|
|
25
|
+
_find_short_opener_pairs,
|
|
26
|
+
_has_parallel_text_component,
|
|
27
|
+
_merge_internal_text_block_group,
|
|
28
|
+
_merge_text_line_content,
|
|
29
|
+
_nearest_following_text_component,
|
|
30
|
+
_nearest_tapered_tail_component,
|
|
31
|
+
_text_component_sort_key,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _merge_short_same_baseline_prefix_blocks(
|
|
36
|
+
blocks: list[dict[str, Any]],
|
|
37
|
+
page_size: tuple[float, float],
|
|
38
|
+
) -> list[dict[str, Any]]:
|
|
39
|
+
"""合并括号序号或时刻等短前缀与右侧同基线正文。"""
|
|
40
|
+
|
|
41
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
42
|
+
consumed: set[int] = set()
|
|
43
|
+
for prefix_index, prefix in enumerate(blocks):
|
|
44
|
+
prefix_rows = prefix.get("_local_line_bboxes")
|
|
45
|
+
prefix_content = str(prefix.get("content") or "").strip()
|
|
46
|
+
if (
|
|
47
|
+
prefix_index in consumed
|
|
48
|
+
or prefix.get("type") != "text"
|
|
49
|
+
or not isinstance(prefix_rows, list)
|
|
50
|
+
or len(prefix_rows) != 1
|
|
51
|
+
or _SHORT_SAME_BASELINE_PREFIX_RE.match(prefix_content) is None
|
|
52
|
+
):
|
|
53
|
+
continue
|
|
54
|
+
prefix_bbox = prefix_rows[0]
|
|
55
|
+
prefix_heights = [
|
|
56
|
+
float(height)
|
|
57
|
+
for height in prefix.get("_line_heights", [])
|
|
58
|
+
if isinstance(height, (int, float)) and float(height) > 0
|
|
59
|
+
]
|
|
60
|
+
prefix_height = statistics.median(prefix_heights) if prefix_heights else max(0.1, prefix_bbox[3] - prefix_bbox[1])
|
|
61
|
+
angle = int(prefix.get("angle", 0) or 0) % 360
|
|
62
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
63
|
+
matches: list[tuple[float, int]] = []
|
|
64
|
+
for host_index, host in enumerate(blocks):
|
|
65
|
+
host_rows = host.get("_local_line_bboxes")
|
|
66
|
+
if (
|
|
67
|
+
host_index == prefix_index
|
|
68
|
+
or host_index in consumed
|
|
69
|
+
or host.get("type") != "text"
|
|
70
|
+
or int(host.get("angle", 0) or 0) % 360 != angle
|
|
71
|
+
or not isinstance(host_rows, list)
|
|
72
|
+
or not host_rows
|
|
73
|
+
):
|
|
74
|
+
continue
|
|
75
|
+
host_bbox = host_rows[0]
|
|
76
|
+
host_width = host_bbox[2] - host_bbox[0]
|
|
77
|
+
horizontal_gap = host_bbox[0] - prefix_bbox[2]
|
|
78
|
+
if (
|
|
79
|
+
host_bbox[0] < prefix_bbox[2]
|
|
80
|
+
or horizontal_gap > 1.25 * prefix_height
|
|
81
|
+
or host_width < 0.15 * local_page_width
|
|
82
|
+
or _bbox_axis_overlap_ratio(
|
|
83
|
+
prefix_bbox,
|
|
84
|
+
host_bbox,
|
|
85
|
+
axis="y",
|
|
86
|
+
)
|
|
87
|
+
< 0.5
|
|
88
|
+
):
|
|
89
|
+
continue
|
|
90
|
+
matches.append((horizontal_gap, host_index))
|
|
91
|
+
if not matches:
|
|
92
|
+
continue
|
|
93
|
+
_gap, host_index = min(matches)
|
|
94
|
+
replacement_index = min(prefix_index, host_index)
|
|
95
|
+
replacement = _merge_internal_text_block_group(
|
|
96
|
+
blocks,
|
|
97
|
+
[prefix_index, host_index],
|
|
98
|
+
preserve_visual_spaces=True,
|
|
99
|
+
)
|
|
100
|
+
replacement["type"] = "text"
|
|
101
|
+
replacements[replacement_index] = replacement
|
|
102
|
+
consumed.update({prefix_index, host_index})
|
|
103
|
+
return [
|
|
104
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _blocks_share_boundary_visual_row(
|
|
109
|
+
first: dict[str, Any],
|
|
110
|
+
second: dict[str, Any],
|
|
111
|
+
pair_height: float,
|
|
112
|
+
) -> bool:
|
|
113
|
+
"""检查前块末行与后块首行是否为被错误切开的同一视觉行。"""
|
|
114
|
+
|
|
115
|
+
first_rows = first.get("_local_line_bboxes")
|
|
116
|
+
second_rows = second.get("_local_line_bboxes")
|
|
117
|
+
if (
|
|
118
|
+
not isinstance(first_rows, list)
|
|
119
|
+
or not isinstance(second_rows, list)
|
|
120
|
+
or max(len(first_rows), len(second_rows)) < 3
|
|
121
|
+
or len(first_rows) + len(second_rows) > 6
|
|
122
|
+
or not _components_share_lane_role(first, second, pair_height)
|
|
123
|
+
):
|
|
124
|
+
return False
|
|
125
|
+
if first["bbox"][1] <= second["bbox"][1]:
|
|
126
|
+
upper_rows, lower_rows = first_rows, second_rows
|
|
127
|
+
else:
|
|
128
|
+
upper_rows, lower_rows = second_rows, first_rows
|
|
129
|
+
upper_boundary = max(upper_rows, key=lambda bbox: (_bbox_center_y(bbox), bbox[0]))
|
|
130
|
+
lower_boundary = min(lower_rows, key=lambda bbox: (_bbox_center_y(bbox), bbox[0]))
|
|
131
|
+
vertical_overlap = max(
|
|
132
|
+
0.0,
|
|
133
|
+
min(upper_boundary[3], lower_boundary[3]) - max(upper_boundary[1], lower_boundary[1]),
|
|
134
|
+
)
|
|
135
|
+
shorter_height = max(
|
|
136
|
+
0.1,
|
|
137
|
+
min(
|
|
138
|
+
upper_boundary[3] - upper_boundary[1],
|
|
139
|
+
lower_boundary[3] - lower_boundary[1],
|
|
140
|
+
),
|
|
141
|
+
)
|
|
142
|
+
horizontal_gap = lower_boundary[0] - upper_boundary[2]
|
|
143
|
+
union_bbox = _bbox_union_many([first["bbox"], second["bbox"]])
|
|
144
|
+
return (
|
|
145
|
+
vertical_overlap / shorter_height >= 0.7
|
|
146
|
+
and -0.2 * pair_height <= horizontal_gap <= 0.75 * pair_height
|
|
147
|
+
and union_bbox[3] - union_bbox[1] <= 6.0 * pair_height
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _merge_overlapping_same_line_text_blocks(
|
|
152
|
+
blocks: list[dict[str, Any]],
|
|
153
|
+
page_size: tuple[float, float],
|
|
154
|
+
) -> list[dict[str, Any]]:
|
|
155
|
+
"""合并块体或边界视觉行重叠的宽正文块,修复错误分栏。"""
|
|
156
|
+
|
|
157
|
+
consumed: set[int] = set()
|
|
158
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
159
|
+
for first_index, first in enumerate(blocks):
|
|
160
|
+
first_bbox = first.get("bbox")
|
|
161
|
+
first_rows = first.get("_local_line_bboxes")
|
|
162
|
+
angle = int(first.get("angle", 0) or 0) % 360
|
|
163
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
164
|
+
if (
|
|
165
|
+
first_index in consumed
|
|
166
|
+
or first.get("type") != "text"
|
|
167
|
+
or not isinstance(first_bbox, (list, tuple))
|
|
168
|
+
or not isinstance(first_rows, list)
|
|
169
|
+
or not 1 <= len(first_rows) <= 5
|
|
170
|
+
or first_bbox[2] - first_bbox[0] < 0.3 * local_page_width
|
|
171
|
+
):
|
|
172
|
+
continue
|
|
173
|
+
first_heights = [
|
|
174
|
+
float(height) for height in first.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
|
|
175
|
+
]
|
|
176
|
+
first_height = statistics.median(first_heights) if first_heights else first_bbox[3] - first_bbox[1]
|
|
177
|
+
for second_index in range(first_index + 1, len(blocks)):
|
|
178
|
+
second = blocks[second_index]
|
|
179
|
+
second_bbox = second.get("bbox")
|
|
180
|
+
second_rows = second.get("_local_line_bboxes")
|
|
181
|
+
if (
|
|
182
|
+
second_index in consumed
|
|
183
|
+
or second.get("type") != "text"
|
|
184
|
+
or int(second.get("angle", 0) or 0) % 360 != angle
|
|
185
|
+
or not isinstance(second_bbox, (list, tuple))
|
|
186
|
+
or not isinstance(second_rows, list)
|
|
187
|
+
or not 1 <= len(second_rows) <= 5
|
|
188
|
+
or second_bbox[2] - second_bbox[0] < 0.3 * local_page_width
|
|
189
|
+
):
|
|
190
|
+
continue
|
|
191
|
+
later_block = max(
|
|
192
|
+
(first, second),
|
|
193
|
+
key=_text_component_sort_key,
|
|
194
|
+
)
|
|
195
|
+
if later_block.get("_hard_break_before") is True:
|
|
196
|
+
continue
|
|
197
|
+
second_heights = [
|
|
198
|
+
float(height) for height in second.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
|
|
199
|
+
]
|
|
200
|
+
second_height = statistics.median(second_heights) if second_heights else second_bbox[3] - second_bbox[1]
|
|
201
|
+
pair_height = max(first_height, second_height, 0.1)
|
|
202
|
+
vertical_overlap = max(
|
|
203
|
+
0.0,
|
|
204
|
+
min(first_bbox[3], second_bbox[3]) - max(first_bbox[1], second_bbox[1]),
|
|
205
|
+
)
|
|
206
|
+
minimum_box_height = max(
|
|
207
|
+
0.1,
|
|
208
|
+
min(
|
|
209
|
+
first_bbox[3] - first_bbox[1],
|
|
210
|
+
second_bbox[3] - second_bbox[1],
|
|
211
|
+
),
|
|
212
|
+
)
|
|
213
|
+
first_fonts = first.get("_font_signatures")
|
|
214
|
+
second_fonts = second.get("_font_signatures")
|
|
215
|
+
fonts_conflict = (
|
|
216
|
+
isinstance(first_fonts, set)
|
|
217
|
+
and isinstance(second_fonts, set)
|
|
218
|
+
and first_fonts
|
|
219
|
+
and second_fonts
|
|
220
|
+
and first_fonts.isdisjoint(second_fonts)
|
|
221
|
+
)
|
|
222
|
+
compact_overlap = (
|
|
223
|
+
len(first_rows) <= 2
|
|
224
|
+
and len(second_rows) <= 2
|
|
225
|
+
and min(len(first_rows), len(second_rows)) == 1
|
|
226
|
+
and abs(first_bbox[0] - second_bbox[0]) <= pair_height
|
|
227
|
+
and vertical_overlap / minimum_box_height >= 0.5
|
|
228
|
+
and _bbox_axis_overlap_ratio(first_bbox, second_bbox, axis="x") >= 0.75
|
|
229
|
+
)
|
|
230
|
+
boundary_overlap = _blocks_share_boundary_visual_row(
|
|
231
|
+
first,
|
|
232
|
+
second,
|
|
233
|
+
pair_height,
|
|
234
|
+
)
|
|
235
|
+
if fonts_conflict or not (compact_overlap or boundary_overlap):
|
|
236
|
+
continue
|
|
237
|
+
union_bbox = _bbox_union_many([first_bbox, second_bbox])
|
|
238
|
+
if not boundary_overlap and union_bbox[3] - union_bbox[1] > 4.0 * pair_height:
|
|
239
|
+
continue
|
|
240
|
+
replacement_index = min(first_index, second_index)
|
|
241
|
+
replacements[replacement_index] = _merge_internal_text_block_group(
|
|
242
|
+
blocks,
|
|
243
|
+
[first_index, second_index],
|
|
244
|
+
)
|
|
245
|
+
consumed.update({first_index, second_index})
|
|
246
|
+
break
|
|
247
|
+
return [
|
|
248
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _merge_inline_math_fragment_text_blocks(
|
|
253
|
+
blocks: list[dict[str, Any]],
|
|
254
|
+
page_size: tuple[float, float],
|
|
255
|
+
) -> list[dict[str, Any]]:
|
|
256
|
+
"""把同一宽正文行上下叠放的多个小数学碎片收回一个文本块。"""
|
|
257
|
+
|
|
258
|
+
consumed: set[int] = set()
|
|
259
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
260
|
+
for host_index, host in enumerate(blocks):
|
|
261
|
+
host_bbox = host.get("bbox")
|
|
262
|
+
host_heights = host.get("_line_heights")
|
|
263
|
+
host_rows = host.get("_local_line_bboxes")
|
|
264
|
+
angle = int(host.get("angle", 0) or 0) % 360
|
|
265
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
266
|
+
if (
|
|
267
|
+
host_index in consumed
|
|
268
|
+
or host.get("type") != "text"
|
|
269
|
+
or (host.get("_single_run_row_id") is None and (not isinstance(host_rows, list) or len(host_rows) > 2))
|
|
270
|
+
or not isinstance(host_bbox, (list, tuple))
|
|
271
|
+
or host_bbox[2] - host_bbox[0] < 0.35 * local_page_width
|
|
272
|
+
or not isinstance(host_heights, list)
|
|
273
|
+
or not host_heights
|
|
274
|
+
):
|
|
275
|
+
continue
|
|
276
|
+
host_height = statistics.median(
|
|
277
|
+
float(height) for height in host_heights if isinstance(height, (int, float)) and height > 0
|
|
278
|
+
)
|
|
279
|
+
fragment_indices: list[int] = []
|
|
280
|
+
for candidate_index, candidate in enumerate(blocks):
|
|
281
|
+
candidate_bbox = candidate.get("bbox")
|
|
282
|
+
candidate_rows = candidate.get("_local_line_bboxes")
|
|
283
|
+
if (
|
|
284
|
+
candidate_index == host_index
|
|
285
|
+
or candidate_index in consumed
|
|
286
|
+
or candidate.get("type") != "text"
|
|
287
|
+
or int(candidate.get("angle", 0) or 0) % 360 != angle
|
|
288
|
+
or not isinstance(candidate_bbox, (list, tuple))
|
|
289
|
+
or not isinstance(candidate_rows, list)
|
|
290
|
+
or len(candidate_rows) != 1
|
|
291
|
+
or candidate_bbox[2] - candidate_bbox[0] > 0.25 * (host_bbox[2] - host_bbox[0])
|
|
292
|
+
or _bbox_axis_overlap_ratio(
|
|
293
|
+
host_bbox,
|
|
294
|
+
candidate_bbox,
|
|
295
|
+
axis="x",
|
|
296
|
+
)
|
|
297
|
+
<= 0.0
|
|
298
|
+
):
|
|
299
|
+
continue
|
|
300
|
+
union_bbox = _bbox_union_many([host_bbox, candidate_bbox])
|
|
301
|
+
vertical_gap = max(
|
|
302
|
+
0.0,
|
|
303
|
+
max(host_bbox[1], candidate_bbox[1]) - min(host_bbox[3], candidate_bbox[3]),
|
|
304
|
+
)
|
|
305
|
+
if vertical_gap <= 0.75 * host_height and union_bbox[3] - union_bbox[1] <= 3.5 * host_height:
|
|
306
|
+
fragment_indices.append(candidate_index)
|
|
307
|
+
if len(fragment_indices) < 2:
|
|
308
|
+
continue
|
|
309
|
+
group_indices = [host_index, *fragment_indices]
|
|
310
|
+
replacement_index = min(group_indices)
|
|
311
|
+
replacements[replacement_index] = _merge_inline_math_recovery_group(
|
|
312
|
+
blocks,
|
|
313
|
+
group_indices,
|
|
314
|
+
)
|
|
315
|
+
consumed.update(group_indices)
|
|
316
|
+
output = [
|
|
317
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
318
|
+
]
|
|
319
|
+
output = _merge_hostless_inline_math_fragment_blocks(output, page_size)
|
|
320
|
+
output = _merge_residual_narrow_math_text_blocks(output, page_size)
|
|
321
|
+
return _merge_inline_math_paragraph_continuations(output, page_size)
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _component_local_union_bbox(
|
|
325
|
+
block: dict[str, Any],
|
|
326
|
+
) -> BBox | None:
|
|
327
|
+
"""合并正文组件持有的正向行框,非法或缺失元数据时返回空。"""
|
|
328
|
+
|
|
329
|
+
rows = block.get("_local_line_bboxes")
|
|
330
|
+
if not isinstance(rows, list):
|
|
331
|
+
return None
|
|
332
|
+
bboxes: list[BBox] = []
|
|
333
|
+
for row in rows:
|
|
334
|
+
if not isinstance(row, (list, tuple)) or len(row) != 4:
|
|
335
|
+
continue
|
|
336
|
+
try:
|
|
337
|
+
bbox = tuple(float(value) for value in row)
|
|
338
|
+
except (TypeError, ValueError):
|
|
339
|
+
continue
|
|
340
|
+
if bbox[2] > bbox[0] and bbox[3] > bbox[1]:
|
|
341
|
+
bboxes.append(bbox) # type: ignore[arg-type]
|
|
342
|
+
return _bbox_union_many(bboxes) if bboxes else None
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _merge_paragraph_formula_context_blocks(
|
|
346
|
+
blocks: list[dict[str, Any]],
|
|
347
|
+
page_size: tuple[float, float],
|
|
348
|
+
) -> list[dict[str, Any]]:
|
|
349
|
+
"""把误似行间公式的复杂行内分式与同栏前后正文恢复成一个块。"""
|
|
350
|
+
|
|
351
|
+
terminal_re = re.compile(
|
|
352
|
+
r"[.!?。!?][\]\)})】》”’'\"]*$",
|
|
353
|
+
)
|
|
354
|
+
consumed: set[int] = set()
|
|
355
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
356
|
+
seed_indices = [
|
|
357
|
+
index
|
|
358
|
+
for index, block in enumerate(blocks)
|
|
359
|
+
if block.get("type") == "text" and block.get(_PARAGRAPH_FORMULA_CONTEXT_MARKER) is True
|
|
360
|
+
]
|
|
361
|
+
for seed_index in seed_indices:
|
|
362
|
+
if seed_index in consumed:
|
|
363
|
+
continue
|
|
364
|
+
group = {seed_index}
|
|
365
|
+
changed = True
|
|
366
|
+
while changed:
|
|
367
|
+
changed = False
|
|
368
|
+
for candidate_index, candidate in enumerate(blocks):
|
|
369
|
+
if candidate_index in group or candidate_index in consumed or candidate.get("type") != "text":
|
|
370
|
+
continue
|
|
371
|
+
for member_index in group:
|
|
372
|
+
member = blocks[member_index]
|
|
373
|
+
if int(candidate.get("angle", 0) or 0) % 360 != int(member.get("angle", 0) or 0) % 360:
|
|
374
|
+
continue
|
|
375
|
+
pair_heights = [
|
|
376
|
+
float(height)
|
|
377
|
+
for block in (candidate, member)
|
|
378
|
+
for height in block.get("_line_heights", [])
|
|
379
|
+
if isinstance(height, (int, float)) and float(height) > 0
|
|
380
|
+
]
|
|
381
|
+
if not pair_heights:
|
|
382
|
+
continue
|
|
383
|
+
pair_height = statistics.median(pair_heights)
|
|
384
|
+
if not _components_share_lane_role(
|
|
385
|
+
candidate,
|
|
386
|
+
member,
|
|
387
|
+
pair_height,
|
|
388
|
+
):
|
|
389
|
+
continue
|
|
390
|
+
candidate_bbox = _component_local_union_bbox(candidate)
|
|
391
|
+
member_bbox = _component_local_union_bbox(member)
|
|
392
|
+
if candidate_bbox is None or member_bbox is None:
|
|
393
|
+
continue
|
|
394
|
+
vertical_gap = max(
|
|
395
|
+
candidate_bbox[1] - member_bbox[3],
|
|
396
|
+
member_bbox[1] - candidate_bbox[3],
|
|
397
|
+
0.0,
|
|
398
|
+
)
|
|
399
|
+
if vertical_gap > 1.5 * pair_height:
|
|
400
|
+
continue
|
|
401
|
+
candidate_below = candidate_bbox[1] >= member_bbox[3]
|
|
402
|
+
member_below = member_bbox[1] >= candidate_bbox[3]
|
|
403
|
+
if candidate_below and (
|
|
404
|
+
candidate.get("_hard_break_before") is True
|
|
405
|
+
or terminal_re.search(str(member.get("content") or "").rstrip()) is not None
|
|
406
|
+
):
|
|
407
|
+
continue
|
|
408
|
+
if member_below and (
|
|
409
|
+
member.get("_hard_break_before") is True
|
|
410
|
+
or terminal_re.search(str(candidate.get("content") or "").rstrip()) is not None
|
|
411
|
+
):
|
|
412
|
+
continue
|
|
413
|
+
group.add(candidate_index)
|
|
414
|
+
changed = True
|
|
415
|
+
break
|
|
416
|
+
if changed:
|
|
417
|
+
break
|
|
418
|
+
if len(group) < 2:
|
|
419
|
+
continue
|
|
420
|
+
ordered_group = sorted(group)
|
|
421
|
+
replacement_index = min(ordered_group)
|
|
422
|
+
merged = _merge_internal_text_block_group(
|
|
423
|
+
blocks,
|
|
424
|
+
ordered_group,
|
|
425
|
+
)
|
|
426
|
+
local_rows = [
|
|
427
|
+
bbox for bbox in merged.get("_local_line_bboxes", []) if isinstance(bbox, (list, tuple)) and len(bbox) == 4
|
|
428
|
+
]
|
|
429
|
+
maximum_width = max(
|
|
430
|
+
(bbox[2] - bbox[0] for bbox in local_rows),
|
|
431
|
+
default=0.0,
|
|
432
|
+
)
|
|
433
|
+
body_rows = [bbox for bbox in local_rows if bbox[2] - bbox[0] >= 0.75 * maximum_width]
|
|
434
|
+
if len(body_rows) >= 2:
|
|
435
|
+
# 复杂分式可能比正文左缘多探出少量 glyph;公开框按重复满行边界稳定收口。
|
|
436
|
+
local_merged_bbox = _bbox_union_many(local_rows)
|
|
437
|
+
local_output_bbox = (
|
|
438
|
+
min(bbox[0] for bbox in body_rows),
|
|
439
|
+
local_merged_bbox[1],
|
|
440
|
+
max(bbox[2] for bbox in body_rows),
|
|
441
|
+
local_merged_bbox[3],
|
|
442
|
+
)
|
|
443
|
+
merged["bbox"] = _rotate_bbox_from_upright(
|
|
444
|
+
local_output_bbox,
|
|
445
|
+
page_size,
|
|
446
|
+
int(merged.get("angle", 0) or 0) % 360,
|
|
447
|
+
)
|
|
448
|
+
replacements[replacement_index] = merged
|
|
449
|
+
consumed.update(ordered_group)
|
|
450
|
+
return [
|
|
451
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
452
|
+
]
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def _merge_residual_narrow_math_text_blocks(
|
|
456
|
+
blocks: list[dict[str, Any]],
|
|
457
|
+
page_size: tuple[float, float],
|
|
458
|
+
) -> list[dict[str, Any]]:
|
|
459
|
+
"""把仍嵌在宽正文行范围内的单个窄数学碎片吸收到唯一宿主块。"""
|
|
460
|
+
|
|
461
|
+
consumed: set[int] = set()
|
|
462
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
463
|
+
for candidate_index, candidate in enumerate(blocks):
|
|
464
|
+
candidate_bbox = candidate.get("bbox")
|
|
465
|
+
candidate_rows = candidate.get("_local_line_bboxes")
|
|
466
|
+
angle = int(candidate.get("angle", 0) or 0) % 360
|
|
467
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
468
|
+
if (
|
|
469
|
+
candidate.get("type") != "text"
|
|
470
|
+
or not isinstance(candidate_bbox, (list, tuple))
|
|
471
|
+
or not isinstance(candidate_rows, list)
|
|
472
|
+
or len(candidate_rows) != 1
|
|
473
|
+
or candidate_bbox[2] - candidate_bbox[0] > 0.05 * local_page_width
|
|
474
|
+
):
|
|
475
|
+
continue
|
|
476
|
+
candidate_heights = [
|
|
477
|
+
float(height) for height in candidate.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
|
|
478
|
+
]
|
|
479
|
+
candidate_height = statistics.median(candidate_heights) if candidate_heights else candidate_bbox[3] - candidate_bbox[1]
|
|
480
|
+
hosts: list[tuple[float, float, int]] = []
|
|
481
|
+
for host_index, host in enumerate(blocks):
|
|
482
|
+
host_bbox = host.get("bbox")
|
|
483
|
+
host_heights = [
|
|
484
|
+
float(height) for height in host.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
|
|
485
|
+
]
|
|
486
|
+
if (
|
|
487
|
+
host_index == candidate_index
|
|
488
|
+
or host_index in consumed
|
|
489
|
+
or host.get("type") != "text"
|
|
490
|
+
or int(host.get("angle", 0) or 0) % 360 != angle
|
|
491
|
+
or not isinstance(host_bbox, (list, tuple))
|
|
492
|
+
or host_bbox[2] - host_bbox[0] < 0.3 * local_page_width
|
|
493
|
+
or candidate_bbox[0] < host_bbox[0]
|
|
494
|
+
or candidate_bbox[2] > host_bbox[2]
|
|
495
|
+
):
|
|
496
|
+
continue
|
|
497
|
+
host_height = statistics.median(host_heights) if host_heights else host_bbox[3] - host_bbox[1]
|
|
498
|
+
vertical_overlap = max(
|
|
499
|
+
0.0,
|
|
500
|
+
min(candidate_bbox[3], host_bbox[3]) - max(candidate_bbox[1], host_bbox[1]),
|
|
501
|
+
)
|
|
502
|
+
vertical_gap = max(
|
|
503
|
+
0.0,
|
|
504
|
+
max(candidate_bbox[1], host_bbox[1]) - min(candidate_bbox[3], host_bbox[3]),
|
|
505
|
+
)
|
|
506
|
+
pair_height = max(candidate_height, host_height, 0.1)
|
|
507
|
+
union_bbox = _bbox_union_many([candidate_bbox, host_bbox])
|
|
508
|
+
if (
|
|
509
|
+
vertical_overlap < 0.35 * min(candidate_height, host_height) and vertical_gap > 0.5 * pair_height
|
|
510
|
+
) or union_bbox[3] - union_bbox[1] > 3.0 * pair_height:
|
|
511
|
+
continue
|
|
512
|
+
hosts.append(
|
|
513
|
+
(
|
|
514
|
+
-vertical_overlap,
|
|
515
|
+
abs(_bbox_center_y(candidate_bbox) - _bbox_center_y(host_bbox)),
|
|
516
|
+
host_index,
|
|
517
|
+
)
|
|
518
|
+
)
|
|
519
|
+
if len(hosts) != 1:
|
|
520
|
+
continue
|
|
521
|
+
host_index = hosts[0][2]
|
|
522
|
+
replacement_index = min(candidate_index, host_index)
|
|
523
|
+
replacements[replacement_index] = _merge_inline_math_recovery_group(
|
|
524
|
+
blocks,
|
|
525
|
+
[candidate_index, host_index],
|
|
526
|
+
)
|
|
527
|
+
consumed.update({candidate_index, host_index})
|
|
528
|
+
return [
|
|
529
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
530
|
+
]
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def _merge_hostless_inline_math_fragment_blocks(
|
|
534
|
+
blocks: list[dict[str, Any]],
|
|
535
|
+
page_size: tuple[float, float],
|
|
536
|
+
) -> list[dict[str, Any]]:
|
|
537
|
+
"""把没有单一宽宿主但在一栏内二维密集排列的数学碎片合成文本块。"""
|
|
538
|
+
|
|
539
|
+
grouped_indices: list[list[int]] = []
|
|
540
|
+
for angle in sorted({int(block.get("angle", 0) or 0) % 360 for block in blocks if block.get("type") == "text"}):
|
|
541
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
542
|
+
candidates = {
|
|
543
|
+
index
|
|
544
|
+
for index, block in enumerate(blocks)
|
|
545
|
+
if block.get("type") == "text"
|
|
546
|
+
and int(block.get("angle", 0) or 0) % 360 == angle
|
|
547
|
+
and isinstance(block.get("bbox"), (list, tuple))
|
|
548
|
+
and isinstance(block.get("_local_line_bboxes"), list)
|
|
549
|
+
and len(block["_local_line_bboxes"]) == 1
|
|
550
|
+
and block["bbox"][2] - block["bbox"][0] <= 0.25 * local_page_width
|
|
551
|
+
}
|
|
552
|
+
while candidates:
|
|
553
|
+
component = {candidates.pop()}
|
|
554
|
+
changed = True
|
|
555
|
+
while changed:
|
|
556
|
+
changed = False
|
|
557
|
+
for candidate_index in list(candidates):
|
|
558
|
+
candidate_bbox = blocks[candidate_index]["bbox"]
|
|
559
|
+
candidate_heights = blocks[candidate_index].get(
|
|
560
|
+
"_line_heights",
|
|
561
|
+
[],
|
|
562
|
+
)
|
|
563
|
+
candidate_height = (
|
|
564
|
+
statistics.median(candidate_heights) if candidate_heights else candidate_bbox[3] - candidate_bbox[1]
|
|
565
|
+
)
|
|
566
|
+
if any(
|
|
567
|
+
(
|
|
568
|
+
max(
|
|
569
|
+
0.0,
|
|
570
|
+
max(candidate_bbox[1], blocks[index]["bbox"][1])
|
|
571
|
+
- min(candidate_bbox[3], blocks[index]["bbox"][3]),
|
|
572
|
+
)
|
|
573
|
+
<= 0.75
|
|
574
|
+
* max(
|
|
575
|
+
candidate_height,
|
|
576
|
+
statistics.median(blocks[index].get("_line_heights", []))
|
|
577
|
+
if blocks[index].get("_line_heights")
|
|
578
|
+
else blocks[index]["bbox"][3] - blocks[index]["bbox"][1],
|
|
579
|
+
)
|
|
580
|
+
and max(
|
|
581
|
+
0.0,
|
|
582
|
+
max(candidate_bbox[0], blocks[index]["bbox"][0])
|
|
583
|
+
- min(candidate_bbox[2], blocks[index]["bbox"][2]),
|
|
584
|
+
)
|
|
585
|
+
<= 1.5
|
|
586
|
+
* max(
|
|
587
|
+
candidate_height,
|
|
588
|
+
statistics.median(blocks[index].get("_line_heights", []))
|
|
589
|
+
if blocks[index].get("_line_heights")
|
|
590
|
+
else blocks[index]["bbox"][3] - blocks[index]["bbox"][1],
|
|
591
|
+
)
|
|
592
|
+
)
|
|
593
|
+
for index in component
|
|
594
|
+
):
|
|
595
|
+
component.add(candidate_index)
|
|
596
|
+
candidates.remove(candidate_index)
|
|
597
|
+
changed = True
|
|
598
|
+
if len(component) < 4:
|
|
599
|
+
continue
|
|
600
|
+
component_heights = [
|
|
601
|
+
float(height)
|
|
602
|
+
for index in component
|
|
603
|
+
for height in blocks[index].get("_line_heights", [])
|
|
604
|
+
if isinstance(height, (int, float)) and height > 0
|
|
605
|
+
]
|
|
606
|
+
if not component_heights:
|
|
607
|
+
continue
|
|
608
|
+
median_height = statistics.median(component_heights)
|
|
609
|
+
union_bbox = _bbox_union_many([blocks[index]["bbox"] for index in component])
|
|
610
|
+
font_signatures = set().union(
|
|
611
|
+
*[
|
|
612
|
+
signatures
|
|
613
|
+
for index in component
|
|
614
|
+
if isinstance(
|
|
615
|
+
(signatures := blocks[index].get("_font_signatures")),
|
|
616
|
+
set,
|
|
617
|
+
)
|
|
618
|
+
]
|
|
619
|
+
)
|
|
620
|
+
if (
|
|
621
|
+
len(font_signatures) < 2
|
|
622
|
+
or not 0.25 * local_page_width <= union_bbox[2] - union_bbox[0] <= 0.5 * local_page_width
|
|
623
|
+
or union_bbox[3] - union_bbox[1] > 3.5 * median_height
|
|
624
|
+
or sum(blocks[index]["bbox"][2] - blocks[index]["bbox"][0] >= 3.5 * median_height for index in component) < 2
|
|
625
|
+
):
|
|
626
|
+
continue
|
|
627
|
+
grouped_indices.append(sorted(component))
|
|
628
|
+
|
|
629
|
+
consumed: set[int] = set()
|
|
630
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
631
|
+
for group in grouped_indices:
|
|
632
|
+
if any(index in consumed for index in group):
|
|
633
|
+
continue
|
|
634
|
+
replacement_index = min(group)
|
|
635
|
+
replacements[replacement_index] = _merge_inline_math_recovery_group(
|
|
636
|
+
blocks,
|
|
637
|
+
group,
|
|
638
|
+
)
|
|
639
|
+
consumed.update(group)
|
|
640
|
+
return [
|
|
641
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
642
|
+
]
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
def _merge_inline_math_recovery_group(
|
|
646
|
+
blocks: list[dict[str, Any]],
|
|
647
|
+
indices: list[int],
|
|
648
|
+
) -> dict[str, Any]:
|
|
649
|
+
"""合并数学碎片并保留仅供后续段落闭合使用的内部标记。"""
|
|
650
|
+
member_bboxes = [blocks[index]["bbox"] for index in indices]
|
|
651
|
+
widths = [bbox[2] - bbox[0] for bbox in member_bboxes]
|
|
652
|
+
maximum_width = max(widths, default=0.0)
|
|
653
|
+
detected_regions = [
|
|
654
|
+
bbox for bbox, width in zip(member_bboxes, widths) if maximum_width <= 0 or width <= 0.25 * maximum_width
|
|
655
|
+
]
|
|
656
|
+
if not detected_regions:
|
|
657
|
+
detected_regions = list(member_bboxes)
|
|
658
|
+
merged = _merge_internal_text_block_group(blocks, indices)
|
|
659
|
+
merged[_INLINE_MATH_RECOVERY_MARKER] = True
|
|
660
|
+
merged["_inline_math_regions"] = [
|
|
661
|
+
*merged.get("_inline_math_regions", []),
|
|
662
|
+
*detected_regions,
|
|
663
|
+
]
|
|
664
|
+
return merged
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
def _merge_inline_math_paragraph_continuations(
|
|
668
|
+
blocks: list[dict[str, Any]],
|
|
669
|
+
page_size: tuple[float, float],
|
|
670
|
+
) -> list[dict[str, Any]]:
|
|
671
|
+
"""在数学碎片恢复后,合并同栏连续且足够宽的正文段落块。"""
|
|
672
|
+
|
|
673
|
+
if sum(block.get(_INLINE_MATH_RECOVERY_MARKER) is True for block in blocks) < 2:
|
|
674
|
+
return blocks
|
|
675
|
+
|
|
676
|
+
lane_groups: list[list[int]] = []
|
|
677
|
+
for index, block in sorted(
|
|
678
|
+
enumerate(blocks),
|
|
679
|
+
key=lambda item: (
|
|
680
|
+
int(item[1].get("angle", 0) or 0) % 360,
|
|
681
|
+
_text_component_sort_key(item[1])
|
|
682
|
+
if isinstance(item[1].get("_local_line_bboxes"), list) and item[1]["_local_line_bboxes"]
|
|
683
|
+
else (float("inf"), float("inf")),
|
|
684
|
+
),
|
|
685
|
+
):
|
|
686
|
+
local_rows = block.get("_local_line_bboxes")
|
|
687
|
+
if (
|
|
688
|
+
block.get("type") != "text"
|
|
689
|
+
or not isinstance(block.get("_lane_is_span"), bool)
|
|
690
|
+
or not isinstance(local_rows, list)
|
|
691
|
+
or not local_rows
|
|
692
|
+
):
|
|
693
|
+
continue
|
|
694
|
+
block_heights = [
|
|
695
|
+
float(height) for height in block.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
|
|
696
|
+
]
|
|
697
|
+
if not block_heights:
|
|
698
|
+
continue
|
|
699
|
+
for lane_group in lane_groups:
|
|
700
|
+
representative = blocks[lane_group[0]]
|
|
701
|
+
if int(representative.get("angle", 0) or 0) % 360 != int(block.get("angle", 0) or 0) % 360:
|
|
702
|
+
continue
|
|
703
|
+
representative_heights = [
|
|
704
|
+
float(height)
|
|
705
|
+
for height in representative.get("_line_heights", [])
|
|
706
|
+
if isinstance(height, (int, float)) and height > 0
|
|
707
|
+
]
|
|
708
|
+
pair_height = statistics.median([*representative_heights, *block_heights])
|
|
709
|
+
if _components_share_lane_role(representative, block, pair_height):
|
|
710
|
+
lane_group.append(index)
|
|
711
|
+
break
|
|
712
|
+
else:
|
|
713
|
+
lane_groups.append([index])
|
|
714
|
+
|
|
715
|
+
candidate_chains: list[list[int]] = []
|
|
716
|
+
for lane_group in lane_groups:
|
|
717
|
+
ordered_indices = sorted(
|
|
718
|
+
lane_group,
|
|
719
|
+
key=lambda index: _text_component_sort_key(blocks[index]),
|
|
720
|
+
)
|
|
721
|
+
chain = [ordered_indices[0]]
|
|
722
|
+
for current_index in ordered_indices[1:]:
|
|
723
|
+
previous_index = chain[-1]
|
|
724
|
+
previous = blocks[previous_index]
|
|
725
|
+
current = blocks[current_index]
|
|
726
|
+
previous_rows = previous["_local_line_bboxes"]
|
|
727
|
+
current_rows = current["_local_line_bboxes"]
|
|
728
|
+
previous_local_bbox = _bbox_union_many(previous_rows)
|
|
729
|
+
current_local_bbox = _bbox_union_many(current_rows)
|
|
730
|
+
pair_heights = [
|
|
731
|
+
float(height)
|
|
732
|
+
for block in (previous, current)
|
|
733
|
+
for height in block.get("_line_heights", [])
|
|
734
|
+
if isinstance(height, (int, float)) and height > 0
|
|
735
|
+
]
|
|
736
|
+
pair_height = statistics.median(pair_heights)
|
|
737
|
+
angle = int(previous.get("angle", 0) or 0) % 360
|
|
738
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
739
|
+
lane_width = _compatible_component_lane_width(
|
|
740
|
+
previous,
|
|
741
|
+
current,
|
|
742
|
+
local_page_width,
|
|
743
|
+
pair_height,
|
|
744
|
+
)
|
|
745
|
+
previous_fonts = previous.get("_font_signatures")
|
|
746
|
+
current_fonts = current.get("_font_signatures")
|
|
747
|
+
fonts_conflict = (
|
|
748
|
+
isinstance(previous_fonts, set)
|
|
749
|
+
and isinstance(current_fonts, set)
|
|
750
|
+
and previous_fonts
|
|
751
|
+
and current_fonts
|
|
752
|
+
and previous_fonts.isdisjoint(current_fonts)
|
|
753
|
+
)
|
|
754
|
+
vertical_gap = current_local_bbox[1] - previous_local_bbox[3]
|
|
755
|
+
connects = (
|
|
756
|
+
_components_share_lane_role(previous, current, pair_height)
|
|
757
|
+
and previous_local_bbox[2] - previous_local_bbox[0] >= 0.8 * lane_width
|
|
758
|
+
and current_local_bbox[2] - current_local_bbox[0] >= 0.8 * lane_width
|
|
759
|
+
and abs(previous_local_bbox[0] - current_local_bbox[0]) <= 0.75 * pair_height
|
|
760
|
+
and -0.75 * pair_height <= vertical_gap <= 0.75 * pair_height
|
|
761
|
+
and not fonts_conflict
|
|
762
|
+
and not _component_connection_skips_block(
|
|
763
|
+
blocks,
|
|
764
|
+
previous_index,
|
|
765
|
+
current_index,
|
|
766
|
+
pair_height,
|
|
767
|
+
)
|
|
768
|
+
)
|
|
769
|
+
if connects:
|
|
770
|
+
chain.append(current_index)
|
|
771
|
+
else:
|
|
772
|
+
candidate_chains.append(chain)
|
|
773
|
+
chain = [current_index]
|
|
774
|
+
candidate_chains.append(chain)
|
|
775
|
+
|
|
776
|
+
consumed: set[int] = set()
|
|
777
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
778
|
+
for chain in candidate_chains:
|
|
779
|
+
if len(chain) < 3 or sum(blocks[index].get(_INLINE_MATH_RECOVERY_MARKER) is True for index in chain) < 2:
|
|
780
|
+
continue
|
|
781
|
+
chain_heights = [
|
|
782
|
+
float(height)
|
|
783
|
+
for index in chain
|
|
784
|
+
for height in blocks[index].get("_line_heights", [])
|
|
785
|
+
if isinstance(height, (int, float)) and height > 0
|
|
786
|
+
]
|
|
787
|
+
median_height = statistics.median(chain_heights)
|
|
788
|
+
local_union = _bbox_union_many([bbox for index in chain for bbox in blocks[index]["_local_line_bboxes"]])
|
|
789
|
+
if local_union[3] - local_union[1] > 24.0 * median_height:
|
|
790
|
+
continue
|
|
791
|
+
replacement_index = min(chain)
|
|
792
|
+
merged = _merge_internal_text_block_group(blocks, chain)
|
|
793
|
+
merged[_INLINE_MATH_RECOVERY_MARKER] = True
|
|
794
|
+
replacements[replacement_index] = merged
|
|
795
|
+
consumed.update(chain)
|
|
796
|
+
return [
|
|
797
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
798
|
+
]
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
def _merge_spatial_text_components(
|
|
802
|
+
blocks: list[dict[str, Any]],
|
|
803
|
+
page_size: tuple[float, float],
|
|
804
|
+
) -> list[dict[str, Any]]:
|
|
805
|
+
"""按短首行、紧邻续行和双栏递减尾行二次连接被栏带拆开的正文块。"""
|
|
806
|
+
|
|
807
|
+
parents = list(range(len(blocks)))
|
|
808
|
+
|
|
809
|
+
def find(index: int) -> int:
|
|
810
|
+
"""查找正文组件所属合并组的根节点。"""
|
|
811
|
+
|
|
812
|
+
while parents[index] != index:
|
|
813
|
+
parents[index] = parents[parents[index]]
|
|
814
|
+
index = parents[index]
|
|
815
|
+
return index
|
|
816
|
+
|
|
817
|
+
def union(first_index: int, second_index: int) -> None:
|
|
818
|
+
"""合并两个已经通过空间连续性校验的正文组件。"""
|
|
819
|
+
|
|
820
|
+
first_root = find(first_index)
|
|
821
|
+
second_root = find(second_index)
|
|
822
|
+
if first_root != second_root:
|
|
823
|
+
parents[second_root] = first_root
|
|
824
|
+
|
|
825
|
+
for angle in sorted(
|
|
826
|
+
{
|
|
827
|
+
int(block.get("angle", 0) or 0) % 360
|
|
828
|
+
for block in blocks
|
|
829
|
+
if block.get("type") == "text" and block.get("_local_line_bboxes")
|
|
830
|
+
}
|
|
831
|
+
):
|
|
832
|
+
candidate_indices = [
|
|
833
|
+
index
|
|
834
|
+
for index, block in enumerate(blocks)
|
|
835
|
+
if block.get("type") == "text"
|
|
836
|
+
and int(block.get("angle", 0) or 0) % 360 == angle
|
|
837
|
+
and isinstance(block.get("_local_line_bboxes"), list)
|
|
838
|
+
and block["_local_line_bboxes"]
|
|
839
|
+
]
|
|
840
|
+
if len(candidate_indices) < 2:
|
|
841
|
+
continue
|
|
842
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
843
|
+
all_heights = [
|
|
844
|
+
float(height)
|
|
845
|
+
for index in candidate_indices
|
|
846
|
+
for height in blocks[index].get("_line_heights", [])
|
|
847
|
+
if isinstance(height, (int, float)) and height > 0
|
|
848
|
+
]
|
|
849
|
+
median_height = statistics.median(all_heights) if all_heights else 1.0
|
|
850
|
+
section_starts = {
|
|
851
|
+
index
|
|
852
|
+
for index in candidate_indices
|
|
853
|
+
if _block_starts_with_short_wide_rows(
|
|
854
|
+
blocks[index],
|
|
855
|
+
local_page_width,
|
|
856
|
+
)
|
|
857
|
+
}
|
|
858
|
+
|
|
859
|
+
opener_pairs = _find_short_opener_pairs(
|
|
860
|
+
blocks,
|
|
861
|
+
candidate_indices,
|
|
862
|
+
local_page_width,
|
|
863
|
+
median_height,
|
|
864
|
+
)
|
|
865
|
+
for opener_index, body_index in opener_pairs:
|
|
866
|
+
section_starts.add(opener_index)
|
|
867
|
+
union(opener_index, body_index)
|
|
868
|
+
|
|
869
|
+
for current_index in sorted(
|
|
870
|
+
candidate_indices,
|
|
871
|
+
key=lambda index: _text_component_sort_key(blocks[index]),
|
|
872
|
+
):
|
|
873
|
+
section_roots = {find(index) for index in section_starts}
|
|
874
|
+
if find(current_index) not in section_roots:
|
|
875
|
+
continue
|
|
876
|
+
next_index = _nearest_following_text_component(
|
|
877
|
+
blocks,
|
|
878
|
+
current_index,
|
|
879
|
+
candidate_indices,
|
|
880
|
+
maximum_gap=0.75 * median_height,
|
|
881
|
+
section_starts=section_starts,
|
|
882
|
+
)
|
|
883
|
+
if next_index is not None:
|
|
884
|
+
union(current_index, next_index)
|
|
885
|
+
|
|
886
|
+
for current_index in candidate_indices:
|
|
887
|
+
if not _has_parallel_text_component(
|
|
888
|
+
blocks,
|
|
889
|
+
current_index,
|
|
890
|
+
candidate_indices,
|
|
891
|
+
):
|
|
892
|
+
continue
|
|
893
|
+
next_index = _nearest_tapered_tail_component(
|
|
894
|
+
blocks,
|
|
895
|
+
current_index,
|
|
896
|
+
candidate_indices,
|
|
897
|
+
median_height,
|
|
898
|
+
section_starts,
|
|
899
|
+
)
|
|
900
|
+
if next_index is not None:
|
|
901
|
+
union(current_index, next_index)
|
|
902
|
+
|
|
903
|
+
grouped_indices: dict[int, list[int]] = {}
|
|
904
|
+
for index in range(len(blocks)):
|
|
905
|
+
grouped_indices.setdefault(find(index), []).append(index)
|
|
906
|
+
|
|
907
|
+
output: list[dict[str, Any]] = []
|
|
908
|
+
for indices in grouped_indices.values():
|
|
909
|
+
if len(indices) == 1:
|
|
910
|
+
output.append(blocks[indices[0]])
|
|
911
|
+
continue
|
|
912
|
+
ordered_indices = sorted(
|
|
913
|
+
indices,
|
|
914
|
+
key=lambda index: _text_component_sort_key(blocks[index]),
|
|
915
|
+
)
|
|
916
|
+
merged = dict(blocks[ordered_indices[0]])
|
|
917
|
+
merged["bbox"] = _bbox_union_many([blocks[index]["bbox"] for index in ordered_indices])
|
|
918
|
+
merged["content"] = _merge_text_line_content([str(blocks[index].get("content", "")) for index in ordered_indices])
|
|
919
|
+
merged["_visual_row_ids"] = set().union(
|
|
920
|
+
*[
|
|
921
|
+
block_ids
|
|
922
|
+
for index in ordered_indices
|
|
923
|
+
if isinstance(
|
|
924
|
+
(block_ids := blocks[index].get("_visual_row_ids")),
|
|
925
|
+
set,
|
|
926
|
+
)
|
|
927
|
+
]
|
|
928
|
+
)
|
|
929
|
+
merged["_single_run_row_id"] = None
|
|
930
|
+
merged["_local_line_bboxes"] = [
|
|
931
|
+
bbox for index in ordered_indices for bbox in blocks[index].get("_local_line_bboxes", [])
|
|
932
|
+
]
|
|
933
|
+
merged["_local_output_line_bboxes"] = [
|
|
934
|
+
bbox for index in ordered_indices for bbox in blocks[index].get("_local_output_line_bboxes", [])
|
|
935
|
+
]
|
|
936
|
+
merged["_output_bbox_repaired"] = any(blocks[index].get("_output_bbox_repaired") is True for index in ordered_indices)
|
|
937
|
+
merged["_line_heights"] = [height for index in ordered_indices for height in blocks[index].get("_line_heights", [])]
|
|
938
|
+
merged["_font_signatures"] = set().union(
|
|
939
|
+
*[
|
|
940
|
+
signatures
|
|
941
|
+
for index in ordered_indices
|
|
942
|
+
if isinstance(
|
|
943
|
+
(signatures := blocks[index].get("_font_signatures")),
|
|
944
|
+
set,
|
|
945
|
+
)
|
|
946
|
+
]
|
|
947
|
+
)
|
|
948
|
+
merged["_inline_math_regions"] = [
|
|
949
|
+
region for index in ordered_indices for region in blocks[index].get("_inline_math_regions", [])
|
|
950
|
+
]
|
|
951
|
+
output.append(merged)
|
|
952
|
+
return output
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
def _merge_list_intro_text_components(
|
|
956
|
+
blocks: list[dict[str, Any]],
|
|
957
|
+
) -> list[dict[str, Any]]:
|
|
958
|
+
"""在编号列表硬边界前合并被误拆的连续引导段和冒号短尾。"""
|
|
959
|
+
|
|
960
|
+
consumed: set[int] = set()
|
|
961
|
+
replacements: dict[int, dict[str, Any]] = {}
|
|
962
|
+
text_indices = [
|
|
963
|
+
index
|
|
964
|
+
for index, block in enumerate(blocks)
|
|
965
|
+
if block.get("type") == "text" and isinstance(block.get("_local_line_bboxes"), list) and block["_local_line_bboxes"]
|
|
966
|
+
]
|
|
967
|
+
for boundary_index in text_indices:
|
|
968
|
+
boundary = blocks[boundary_index]
|
|
969
|
+
if (
|
|
970
|
+
boundary.get("_hard_break_before") is not True
|
|
971
|
+
or _LIST_ITEM_RE.match(
|
|
972
|
+
str(boundary.get("content") or ""),
|
|
973
|
+
)
|
|
974
|
+
is None
|
|
975
|
+
):
|
|
976
|
+
continue
|
|
977
|
+
preceding = [
|
|
978
|
+
index
|
|
979
|
+
for index in text_indices
|
|
980
|
+
if index not in consumed and _text_component_sort_key(blocks[index]) < _text_component_sort_key(boundary)
|
|
981
|
+
]
|
|
982
|
+
if not preceding:
|
|
983
|
+
continue
|
|
984
|
+
immediate_index = max(
|
|
985
|
+
preceding,
|
|
986
|
+
key=lambda index: _text_component_sort_key(blocks[index]),
|
|
987
|
+
)
|
|
988
|
+
immediate = blocks[immediate_index]
|
|
989
|
+
immediate_rows = immediate["_local_line_bboxes"]
|
|
990
|
+
immediate_heights = [
|
|
991
|
+
float(height) for height in immediate.get("_line_heights", []) if isinstance(height, (int, float)) and height > 0
|
|
992
|
+
]
|
|
993
|
+
pair_height = statistics.median(
|
|
994
|
+
immediate_heights
|
|
995
|
+
or [
|
|
996
|
+
immediate_rows[-1][3] - immediate_rows[-1][1],
|
|
997
|
+
],
|
|
998
|
+
)
|
|
999
|
+
interval = _component_lane_interval(immediate)
|
|
1000
|
+
if (
|
|
1001
|
+
interval is None
|
|
1002
|
+
or immediate_rows[-1][2] - immediate_rows[-1][0] > 0.35 * (interval[1] - interval[0])
|
|
1003
|
+
or not str(immediate.get("content") or "").rstrip().endswith((":", ":"))
|
|
1004
|
+
or not _components_share_lane_role(
|
|
1005
|
+
immediate,
|
|
1006
|
+
boundary,
|
|
1007
|
+
pair_height,
|
|
1008
|
+
)
|
|
1009
|
+
):
|
|
1010
|
+
continue
|
|
1011
|
+
|
|
1012
|
+
group = [immediate_index]
|
|
1013
|
+
cursor_index = immediate_index
|
|
1014
|
+
while len(group) < 3:
|
|
1015
|
+
earlier = [
|
|
1016
|
+
index
|
|
1017
|
+
for index in preceding
|
|
1018
|
+
if index not in group
|
|
1019
|
+
and _text_component_sort_key(blocks[index]) < _text_component_sort_key(blocks[cursor_index])
|
|
1020
|
+
and _components_share_lane_role(
|
|
1021
|
+
blocks[index],
|
|
1022
|
+
blocks[cursor_index],
|
|
1023
|
+
pair_height,
|
|
1024
|
+
)
|
|
1025
|
+
]
|
|
1026
|
+
if not earlier:
|
|
1027
|
+
break
|
|
1028
|
+
previous_index = max(
|
|
1029
|
+
earlier,
|
|
1030
|
+
key=lambda index: _text_component_sort_key(blocks[index]),
|
|
1031
|
+
)
|
|
1032
|
+
previous = blocks[previous_index]
|
|
1033
|
+
current = blocks[cursor_index]
|
|
1034
|
+
previous_rows = previous["_local_line_bboxes"]
|
|
1035
|
+
current_rows = current["_local_line_bboxes"]
|
|
1036
|
+
heights = [
|
|
1037
|
+
float(height)
|
|
1038
|
+
for block in (previous, current)
|
|
1039
|
+
for height in block.get("_line_heights", [])
|
|
1040
|
+
if isinstance(height, (int, float)) and height > 0
|
|
1041
|
+
]
|
|
1042
|
+
local_height = statistics.median(heights or [pair_height])
|
|
1043
|
+
vertical_gap = current_rows[0][1] - previous_rows[-1][3]
|
|
1044
|
+
if (
|
|
1045
|
+
previous.get("_hard_break_before") is True
|
|
1046
|
+
or not _components_share_lane_role(
|
|
1047
|
+
previous,
|
|
1048
|
+
current,
|
|
1049
|
+
local_height,
|
|
1050
|
+
)
|
|
1051
|
+
or not -local_height <= vertical_gap <= local_height
|
|
1052
|
+
):
|
|
1053
|
+
break
|
|
1054
|
+
group.append(previous_index)
|
|
1055
|
+
cursor_index = previous_index
|
|
1056
|
+
if len(group) < 2:
|
|
1057
|
+
continue
|
|
1058
|
+
ordered_group = sorted(
|
|
1059
|
+
group,
|
|
1060
|
+
key=lambda index: _text_component_sort_key(blocks[index]),
|
|
1061
|
+
)
|
|
1062
|
+
replacement_index = min(ordered_group)
|
|
1063
|
+
replacements[replacement_index] = _merge_internal_text_block_group(
|
|
1064
|
+
blocks,
|
|
1065
|
+
ordered_group,
|
|
1066
|
+
)
|
|
1067
|
+
consumed.update(ordered_group)
|
|
1068
|
+
|
|
1069
|
+
return [
|
|
1070
|
+
replacements.get(index, block) for index, block in enumerate(blocks) if index not in consumed or index in replacements
|
|
1071
|
+
]
|
|
1072
|
+
|
|
1073
|
+
|
|
1074
|
+
def _merge_unterminated_text_components(
|
|
1075
|
+
blocks: list[dict[str, Any]],
|
|
1076
|
+
) -> list[dict[str, Any]]:
|
|
1077
|
+
"""合并普通栏未终止正文,以及满足严格结构约束的满宽 span 正文。"""
|
|
1078
|
+
|
|
1079
|
+
output = list(blocks)
|
|
1080
|
+
terminal_re = re.compile(
|
|
1081
|
+
r"[.!?。!?::;;][\]\)})】》”’'\"]*$",
|
|
1082
|
+
)
|
|
1083
|
+
while True:
|
|
1084
|
+
text_indices = sorted(
|
|
1085
|
+
(
|
|
1086
|
+
index
|
|
1087
|
+
for index, block in enumerate(output)
|
|
1088
|
+
if block.get("type") == "text"
|
|
1089
|
+
and isinstance(
|
|
1090
|
+
block.get("_local_line_bboxes"),
|
|
1091
|
+
list,
|
|
1092
|
+
)
|
|
1093
|
+
and block["_local_line_bboxes"]
|
|
1094
|
+
),
|
|
1095
|
+
key=lambda index: _text_component_sort_key(output[index]),
|
|
1096
|
+
)
|
|
1097
|
+
merged_pair: tuple[int, int] | None = None
|
|
1098
|
+
for first_index, second_index in zip(
|
|
1099
|
+
text_indices,
|
|
1100
|
+
text_indices[1:],
|
|
1101
|
+
):
|
|
1102
|
+
first = output[first_index]
|
|
1103
|
+
second = output[second_index]
|
|
1104
|
+
second_rows = second["_local_line_bboxes"]
|
|
1105
|
+
first_rows = first["_local_line_bboxes"]
|
|
1106
|
+
heights = [
|
|
1107
|
+
float(height)
|
|
1108
|
+
for block in (first, second)
|
|
1109
|
+
for height in block.get("_line_heights", [])
|
|
1110
|
+
if isinstance(height, (int, float)) and height > 0
|
|
1111
|
+
]
|
|
1112
|
+
pair_height = statistics.median(
|
|
1113
|
+
heights
|
|
1114
|
+
or [
|
|
1115
|
+
first_rows[-1][3] - first_rows[-1][1],
|
|
1116
|
+
second_rows[0][3] - second_rows[0][1],
|
|
1117
|
+
],
|
|
1118
|
+
)
|
|
1119
|
+
second_interval = _component_lane_interval(second)
|
|
1120
|
+
first_declared_interval = _component_declared_lane_interval(
|
|
1121
|
+
first,
|
|
1122
|
+
)
|
|
1123
|
+
second_declared_interval = _component_declared_lane_interval(
|
|
1124
|
+
second,
|
|
1125
|
+
)
|
|
1126
|
+
row_pair_height = statistics.median(
|
|
1127
|
+
[
|
|
1128
|
+
max(
|
|
1129
|
+
0.1,
|
|
1130
|
+
first_rows[-1][3] - first_rows[-1][1],
|
|
1131
|
+
),
|
|
1132
|
+
max(
|
|
1133
|
+
0.1,
|
|
1134
|
+
second_rows[0][3] - second_rows[0][1],
|
|
1135
|
+
),
|
|
1136
|
+
],
|
|
1137
|
+
)
|
|
1138
|
+
span_connection_height = max(
|
|
1139
|
+
pair_height,
|
|
1140
|
+
row_pair_height,
|
|
1141
|
+
)
|
|
1142
|
+
span_tolerance = 0.75 * span_connection_height
|
|
1143
|
+
span_pair = (
|
|
1144
|
+
first.get("_lane_is_span") is True
|
|
1145
|
+
and second.get("_lane_is_span") is True
|
|
1146
|
+
and int(first.get("angle", 0) or 0) % 360 == int(second.get("angle", 0) or 0) % 360
|
|
1147
|
+
and first_declared_interval is not None
|
|
1148
|
+
and second_declared_interval is not None
|
|
1149
|
+
and abs(
|
|
1150
|
+
first_declared_interval[0] - second_declared_interval[0],
|
|
1151
|
+
)
|
|
1152
|
+
<= span_tolerance
|
|
1153
|
+
and abs(
|
|
1154
|
+
first_declared_interval[1] - second_declared_interval[1],
|
|
1155
|
+
)
|
|
1156
|
+
<= span_tolerance
|
|
1157
|
+
)
|
|
1158
|
+
first_content = str(first.get("content") or "")
|
|
1159
|
+
second_content = str(second.get("content") or "")
|
|
1160
|
+
single_numbered_tail = (
|
|
1161
|
+
len(second_rows) == 1
|
|
1162
|
+
and second.get("_hard_break_before") is not True
|
|
1163
|
+
and _LIST_ITEM_RE.match(second_content) is not None
|
|
1164
|
+
and first_content.rstrip().endswith((":", ":"))
|
|
1165
|
+
)
|
|
1166
|
+
url_continuation = _URL_LINE_RE.match(second_content) is not None
|
|
1167
|
+
aligned_short_tail = (
|
|
1168
|
+
len(first_rows) >= 2
|
|
1169
|
+
and len(second_rows) == 1
|
|
1170
|
+
and abs(second_rows[0][0] - first_rows[-1][0]) <= 0.75 * pair_height
|
|
1171
|
+
)
|
|
1172
|
+
narrow_continuation = single_numbered_tail or url_continuation or aligned_short_tail
|
|
1173
|
+
starts_wide_label = (
|
|
1174
|
+
not url_continuation
|
|
1175
|
+
and _LABELLED_METADATA_RE.match(
|
|
1176
|
+
second_content,
|
|
1177
|
+
)
|
|
1178
|
+
is not None
|
|
1179
|
+
and (reference_interval := (second_declared_interval if span_pair else second_interval)) is not None
|
|
1180
|
+
and second_rows[0][2] - second_rows[0][0] >= 0.5 * (reference_interval[1] - reference_interval[0])
|
|
1181
|
+
)
|
|
1182
|
+
if (
|
|
1183
|
+
second.get("_protected_hard_break_before") is True
|
|
1184
|
+
or (second.get("_hard_break_before") is True and (span_pair or not narrow_continuation))
|
|
1185
|
+
or second.get("_leading_emphasis_start") is True
|
|
1186
|
+
or (starts_wide_label and (span_pair or not aligned_short_tail))
|
|
1187
|
+
or first.get("_hanging_indent_group") is not None
|
|
1188
|
+
or second.get("_hanging_indent_group") is not None
|
|
1189
|
+
or _FIGURE_CAPTION_MARKER_RE.match(
|
|
1190
|
+
first_content,
|
|
1191
|
+
)
|
|
1192
|
+
is not None
|
|
1193
|
+
or (
|
|
1194
|
+
not span_pair
|
|
1195
|
+
and not single_numbered_tail
|
|
1196
|
+
and not url_continuation
|
|
1197
|
+
and terminal_re.search(
|
|
1198
|
+
first_content.rstrip(),
|
|
1199
|
+
)
|
|
1200
|
+
is not None
|
|
1201
|
+
)
|
|
1202
|
+
):
|
|
1203
|
+
continue
|
|
1204
|
+
interval = _component_lane_interval(first)
|
|
1205
|
+
if span_pair:
|
|
1206
|
+
interval = first_declared_interval
|
|
1207
|
+
elif interval is None or not _components_share_lane_role(
|
|
1208
|
+
first,
|
|
1209
|
+
second,
|
|
1210
|
+
pair_height,
|
|
1211
|
+
):
|
|
1212
|
+
continue
|
|
1213
|
+
if interval is None:
|
|
1214
|
+
continue
|
|
1215
|
+
lane_width = interval[1] - interval[0]
|
|
1216
|
+
vertical_gap = second_rows[0][1] - first_rows[-1][3]
|
|
1217
|
+
minimum_first_fill = 0.8 if span_pair else 0.5 if single_numbered_tail else 0.65
|
|
1218
|
+
minimum_second_fill = 0.8 if span_pair else 0.65
|
|
1219
|
+
connection_height = span_connection_height if span_pair else pair_height
|
|
1220
|
+
first_reference_width = (
|
|
1221
|
+
max(row[2] - row[0] for row in first_rows) if narrow_continuation else first_rows[-1][2] - first_rows[-1][0]
|
|
1222
|
+
)
|
|
1223
|
+
first_fonts = first.get("_font_signatures")
|
|
1224
|
+
second_fonts = second.get("_font_signatures")
|
|
1225
|
+
fonts_conflict = (
|
|
1226
|
+
isinstance(first_fonts, set)
|
|
1227
|
+
and isinstance(second_fonts, set)
|
|
1228
|
+
and first_fonts
|
|
1229
|
+
and second_fonts
|
|
1230
|
+
and first_fonts.isdisjoint(second_fonts)
|
|
1231
|
+
)
|
|
1232
|
+
if (
|
|
1233
|
+
first_reference_width < minimum_first_fill * lane_width
|
|
1234
|
+
or (
|
|
1235
|
+
(span_pair or not narrow_continuation)
|
|
1236
|
+
and second_rows[0][2] - second_rows[0][0] < minimum_second_fill * lane_width
|
|
1237
|
+
)
|
|
1238
|
+
or not -connection_height <= vertical_gap <= 1.5 * connection_height
|
|
1239
|
+
or fonts_conflict
|
|
1240
|
+
or _component_connection_skips_block(
|
|
1241
|
+
output,
|
|
1242
|
+
first_index,
|
|
1243
|
+
second_index,
|
|
1244
|
+
connection_height,
|
|
1245
|
+
)
|
|
1246
|
+
):
|
|
1247
|
+
continue
|
|
1248
|
+
merged_pair = (first_index, second_index)
|
|
1249
|
+
break
|
|
1250
|
+
if merged_pair is None:
|
|
1251
|
+
return output
|
|
1252
|
+
first_index, second_index = merged_pair
|
|
1253
|
+
output[first_index] = _merge_internal_text_block_group(
|
|
1254
|
+
output,
|
|
1255
|
+
[first_index, second_index],
|
|
1256
|
+
)
|
|
1257
|
+
output.pop(second_index)
|
|
1258
|
+
|
|
1259
|
+
|
|
1260
|
+
__all__ = [
|
|
1261
|
+
"_merge_short_same_baseline_prefix_blocks",
|
|
1262
|
+
"_blocks_share_boundary_visual_row",
|
|
1263
|
+
"_merge_overlapping_same_line_text_blocks",
|
|
1264
|
+
"_merge_inline_math_fragment_text_blocks",
|
|
1265
|
+
"_component_local_union_bbox",
|
|
1266
|
+
"_merge_paragraph_formula_context_blocks",
|
|
1267
|
+
"_merge_residual_narrow_math_text_blocks",
|
|
1268
|
+
"_merge_hostless_inline_math_fragment_blocks",
|
|
1269
|
+
"_merge_inline_math_recovery_group",
|
|
1270
|
+
"_merge_inline_math_paragraph_continuations",
|
|
1271
|
+
"_merge_spatial_text_components",
|
|
1272
|
+
"_merge_list_intro_text_components",
|
|
1273
|
+
"_merge_unterminated_text_components",
|
|
1274
|
+
]
|