docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,692 @@
|
|
|
1
|
+
"""依据栏带、缩进和排版重置寻找正文行分组边界。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import statistics
|
|
7
|
+
from typing import Sequence
|
|
8
|
+
|
|
9
|
+
from .....schema import BBox
|
|
10
|
+
from .....foundation.text import is_hyphen_at_line_end
|
|
11
|
+
from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_x, _rotate_bbox_to_upright
|
|
12
|
+
from ..line_layout import (
|
|
13
|
+
_connection_crosses_table,
|
|
14
|
+
_effective_body_text_row_gap,
|
|
15
|
+
_effective_text_row_gap,
|
|
16
|
+
_horizontal_rule_separates_rows,
|
|
17
|
+
_line_effective_height,
|
|
18
|
+
_line_tight_output_bbox,
|
|
19
|
+
_title_fonts_compatible,
|
|
20
|
+
)
|
|
21
|
+
from ..models import _LineItem, _LocalAxisLine, _TextLane
|
|
22
|
+
from .common import (
|
|
23
|
+
_ABSTRACT_METADATA_RE,
|
|
24
|
+
_BULLET_ITEM_RE,
|
|
25
|
+
_EMAIL_METADATA_RE,
|
|
26
|
+
_FRONT_MATTER_FIELD_RE,
|
|
27
|
+
_LABELLED_METADATA_RE,
|
|
28
|
+
_LIST_ITEM_RE,
|
|
29
|
+
_REFERENCE_ENTRY_RE,
|
|
30
|
+
_URL_LINE_RE,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _local_tight_output_line_bboxes(
|
|
35
|
+
lines: Sequence[_LineItem],
|
|
36
|
+
page_size: tuple[float, float],
|
|
37
|
+
angle: int,
|
|
38
|
+
) -> tuple[list[BBox], bool]:
|
|
39
|
+
"""返回与原行顺序一致的 tight+1pt 局部框及是否存在可靠候选。"""
|
|
40
|
+
|
|
41
|
+
output = []
|
|
42
|
+
changed = False
|
|
43
|
+
for line in lines:
|
|
44
|
+
candidate = _line_tight_output_bbox(line, page_size)
|
|
45
|
+
output.append(
|
|
46
|
+
_rotate_bbox_to_upright(
|
|
47
|
+
candidate or line.bbox,
|
|
48
|
+
page_size,
|
|
49
|
+
angle,
|
|
50
|
+
)
|
|
51
|
+
)
|
|
52
|
+
changed = changed or candidate is not None
|
|
53
|
+
return output, changed
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _starts_structural_reference_entry(
|
|
57
|
+
previous: tuple[_LineItem, BBox],
|
|
58
|
+
current: tuple[_LineItem, BBox],
|
|
59
|
+
) -> bool:
|
|
60
|
+
"""仅在编号行相对续行明显左突时确认新的参考文献条目。"""
|
|
61
|
+
|
|
62
|
+
if _REFERENCE_ENTRY_RE.match(current[0].text.strip()) is None:
|
|
63
|
+
return False
|
|
64
|
+
previous_height = _line_effective_height(*previous)
|
|
65
|
+
current_height = _line_effective_height(*current)
|
|
66
|
+
pair_height = max(previous_height, current_height)
|
|
67
|
+
return (
|
|
68
|
+
current[1][0] <= previous[1][0] - max(5.0, 0.6 * min(previous_height, current_height))
|
|
69
|
+
and -0.75 * pair_height <= _effective_text_row_gap(previous, current) <= 1.5 * pair_height
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _build_hanging_indent_group_map(
|
|
74
|
+
lane: _TextLane,
|
|
75
|
+
table_bboxes: list[BBox],
|
|
76
|
+
axis_lines: list[_LocalAxisLine],
|
|
77
|
+
) -> dict[int, int]:
|
|
78
|
+
"""仅按重复的左突首行和稳定续行缩进识别悬挂缩进条目。"""
|
|
79
|
+
|
|
80
|
+
if len(lane.lines) < 4:
|
|
81
|
+
return {}
|
|
82
|
+
rows = sorted(
|
|
83
|
+
(item for item in lane.lines if item[0].semantic_type is None),
|
|
84
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
85
|
+
)
|
|
86
|
+
if len(rows) < 4:
|
|
87
|
+
return {}
|
|
88
|
+
median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in rows)
|
|
89
|
+
start_tolerance = max(5.0, 0.65 * median_height)
|
|
90
|
+
minimum_indent = max(7.0, 0.8 * median_height)
|
|
91
|
+
continuation_tolerance = max(4.0, 0.55 * median_height)
|
|
92
|
+
|
|
93
|
+
def rows_are_adjacent(
|
|
94
|
+
previous: tuple[_LineItem, BBox],
|
|
95
|
+
current: tuple[_LineItem, BBox],
|
|
96
|
+
) -> bool:
|
|
97
|
+
"""检查相邻行的净空和几何障碍是否允许组成同一缩进序列。"""
|
|
98
|
+
|
|
99
|
+
effective_gap = _effective_text_row_gap(previous, current)
|
|
100
|
+
top_pitch = current[1][1] - previous[1][1]
|
|
101
|
+
robust_pitch_fallback = 0.5 * median_height <= top_pitch <= 1.8 * median_height
|
|
102
|
+
if not -0.6 * median_height <= effective_gap <= 1.3 * median_height and not robust_pitch_fallback:
|
|
103
|
+
return False
|
|
104
|
+
if _connection_crosses_table(
|
|
105
|
+
previous[0].bbox,
|
|
106
|
+
current[0].bbox,
|
|
107
|
+
table_bboxes,
|
|
108
|
+
):
|
|
109
|
+
return False
|
|
110
|
+
return not _horizontal_rule_separates_rows(
|
|
111
|
+
previous[1],
|
|
112
|
+
current[1],
|
|
113
|
+
lane,
|
|
114
|
+
axis_lines,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
def consume_entry(
|
|
118
|
+
start_index: int,
|
|
119
|
+
start_left: float,
|
|
120
|
+
expected_continuation_left: float | None,
|
|
121
|
+
*,
|
|
122
|
+
require_next_start: bool,
|
|
123
|
+
) -> tuple[int, float] | None:
|
|
124
|
+
"""消费一个左突首行及其续行,并返回下一条首行位置。"""
|
|
125
|
+
|
|
126
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
127
|
+
full_width_midparagraph_entry = (
|
|
128
|
+
rows[start_index][1][2] - rows[start_index][1][0] >= 0.8 * lane_width
|
|
129
|
+
and start_index + 1 < len(rows)
|
|
130
|
+
and rows[start_index + 1][1][2] - rows[start_index + 1][1][0] <= 0.75 * lane_width
|
|
131
|
+
)
|
|
132
|
+
if (
|
|
133
|
+
start_index > 0
|
|
134
|
+
and rows_are_adjacent(
|
|
135
|
+
rows[start_index - 1],
|
|
136
|
+
rows[start_index],
|
|
137
|
+
)
|
|
138
|
+
and abs(rows[start_index - 1][1][0] - start_left) <= start_tolerance
|
|
139
|
+
and not full_width_midparagraph_entry
|
|
140
|
+
):
|
|
141
|
+
# 同左缘正文仍在连续时不能从段落中部启动悬挂条目序列。
|
|
142
|
+
return None
|
|
143
|
+
if (
|
|
144
|
+
start_index > 0
|
|
145
|
+
and is_hyphen_at_line_end(rows[start_index - 1][0].text)
|
|
146
|
+
and rows_are_adjacent(rows[start_index - 1], rows[start_index])
|
|
147
|
+
):
|
|
148
|
+
# 排版断词后的下一物理行属于前文,不能被缩进几何误当成新条目首行。
|
|
149
|
+
return None
|
|
150
|
+
continuation_index = start_index + 1
|
|
151
|
+
if continuation_index >= len(rows):
|
|
152
|
+
return None
|
|
153
|
+
first_continuation = rows[continuation_index]
|
|
154
|
+
if not rows_are_adjacent(rows[start_index], first_continuation):
|
|
155
|
+
return None
|
|
156
|
+
continuation_left = first_continuation[1][0]
|
|
157
|
+
if continuation_left < start_left + minimum_indent:
|
|
158
|
+
return None
|
|
159
|
+
if (
|
|
160
|
+
expected_continuation_left is not None
|
|
161
|
+
and abs(continuation_left - expected_continuation_left) > continuation_tolerance
|
|
162
|
+
):
|
|
163
|
+
return None
|
|
164
|
+
|
|
165
|
+
continuation_index += 1
|
|
166
|
+
while continuation_index < len(rows):
|
|
167
|
+
previous = rows[continuation_index - 1]
|
|
168
|
+
current = rows[continuation_index]
|
|
169
|
+
current_left = current[1][0]
|
|
170
|
+
if not rows_are_adjacent(previous, current):
|
|
171
|
+
break
|
|
172
|
+
if current_left < start_left + minimum_indent:
|
|
173
|
+
break
|
|
174
|
+
if abs(current_left - continuation_left) > continuation_tolerance:
|
|
175
|
+
break
|
|
176
|
+
continuation_index += 1
|
|
177
|
+
|
|
178
|
+
if not require_next_start:
|
|
179
|
+
return continuation_index, continuation_left
|
|
180
|
+
if continuation_index >= len(rows):
|
|
181
|
+
return None
|
|
182
|
+
if not rows_are_adjacent(rows[continuation_index - 1], rows[continuation_index]):
|
|
183
|
+
return None
|
|
184
|
+
if abs(rows[continuation_index][1][0] - start_left) > start_tolerance:
|
|
185
|
+
return None
|
|
186
|
+
return continuation_index, continuation_left
|
|
187
|
+
|
|
188
|
+
group_map: dict[int, int] = {}
|
|
189
|
+
group_index = 0
|
|
190
|
+
row_index = 0
|
|
191
|
+
while row_index < len(rows) - 3:
|
|
192
|
+
start_left = rows[row_index][1][0]
|
|
193
|
+
first_entry = consume_entry(
|
|
194
|
+
row_index,
|
|
195
|
+
start_left,
|
|
196
|
+
None,
|
|
197
|
+
require_next_start=True,
|
|
198
|
+
)
|
|
199
|
+
if first_entry is None:
|
|
200
|
+
row_index += 1
|
|
201
|
+
continue
|
|
202
|
+
|
|
203
|
+
_next_start_index, continuation_left = first_entry
|
|
204
|
+
start_indices = [row_index]
|
|
205
|
+
current_start_index = row_index
|
|
206
|
+
end_index: int | None = None
|
|
207
|
+
while True:
|
|
208
|
+
next_entry = consume_entry(
|
|
209
|
+
current_start_index,
|
|
210
|
+
start_left,
|
|
211
|
+
continuation_left,
|
|
212
|
+
require_next_start=True,
|
|
213
|
+
)
|
|
214
|
+
if next_entry is None:
|
|
215
|
+
final_entry = consume_entry(
|
|
216
|
+
current_start_index,
|
|
217
|
+
start_left,
|
|
218
|
+
continuation_left,
|
|
219
|
+
require_next_start=False,
|
|
220
|
+
)
|
|
221
|
+
if final_entry is not None:
|
|
222
|
+
end_index = final_entry[0]
|
|
223
|
+
break
|
|
224
|
+
next_start_index, _continuation_left = next_entry
|
|
225
|
+
prospective_entry = consume_entry(
|
|
226
|
+
next_start_index,
|
|
227
|
+
start_left,
|
|
228
|
+
continuation_left,
|
|
229
|
+
require_next_start=False,
|
|
230
|
+
)
|
|
231
|
+
if prospective_entry is None:
|
|
232
|
+
# 当前条目已经完整确认;后面的普通左对齐段落只作为终止边界,
|
|
233
|
+
# 不能让它反向使此前所有悬挂缩进条目失效。
|
|
234
|
+
end_index = next_start_index
|
|
235
|
+
break
|
|
236
|
+
start_indices.append(next_start_index)
|
|
237
|
+
current_start_index = next_start_index
|
|
238
|
+
if len(start_indices) < 2 or end_index is None:
|
|
239
|
+
row_index += 1
|
|
240
|
+
continue
|
|
241
|
+
|
|
242
|
+
entry_ranges = [
|
|
243
|
+
(start, end)
|
|
244
|
+
for start, end in zip(
|
|
245
|
+
start_indices,
|
|
246
|
+
[*start_indices[1:], end_index],
|
|
247
|
+
strict=True,
|
|
248
|
+
)
|
|
249
|
+
]
|
|
250
|
+
for start, end in entry_ranges:
|
|
251
|
+
for line, _bbox in rows[start:end]:
|
|
252
|
+
group_map[line.source_index] = group_index
|
|
253
|
+
group_index += 1
|
|
254
|
+
row_index = end_index
|
|
255
|
+
|
|
256
|
+
return group_map
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _infer_local_text_lane_map(lane: _TextLane) -> dict[int, _TextLane]:
|
|
260
|
+
"""从连续同左缘正文推导局部栏宽,修正跨栏上文污染的全宽栏带。"""
|
|
261
|
+
|
|
262
|
+
if lane.is_span or len(lane.lines) < 3:
|
|
263
|
+
return {}
|
|
264
|
+
rows = sorted(
|
|
265
|
+
lane.lines,
|
|
266
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
267
|
+
)
|
|
268
|
+
median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in rows)
|
|
269
|
+
left_tolerance = max(3.0, 0.75 * median_height)
|
|
270
|
+
height_ratio_limit = 1.25
|
|
271
|
+
runs: list[list[tuple[_LineItem, BBox]]] = []
|
|
272
|
+
current_run: list[tuple[_LineItem, BBox]] = []
|
|
273
|
+
|
|
274
|
+
def submit_run() -> None:
|
|
275
|
+
"""提交当前连续正文行,语义行和明显左缘变化都会结束该局部区段。"""
|
|
276
|
+
|
|
277
|
+
nonlocal current_run
|
|
278
|
+
if current_run:
|
|
279
|
+
runs.append(current_run)
|
|
280
|
+
current_run = []
|
|
281
|
+
|
|
282
|
+
for item in rows:
|
|
283
|
+
line, bbox = item
|
|
284
|
+
if line.semantic_type is not None:
|
|
285
|
+
submit_run()
|
|
286
|
+
continue
|
|
287
|
+
if not current_run:
|
|
288
|
+
current_run = [item]
|
|
289
|
+
continue
|
|
290
|
+
run_left = statistics.median(member[1][0] for member in current_run)
|
|
291
|
+
run_heights = [_line_effective_height(member, member_bbox) for member, member_bbox in current_run]
|
|
292
|
+
current_height = _line_effective_height(line, bbox)
|
|
293
|
+
if (
|
|
294
|
+
abs(bbox[0] - run_left) <= left_tolerance
|
|
295
|
+
and max([*run_heights, current_height]) / max(0.1, min([*run_heights, current_height])) <= height_ratio_limit
|
|
296
|
+
):
|
|
297
|
+
current_run.append(item)
|
|
298
|
+
else:
|
|
299
|
+
submit_run()
|
|
300
|
+
current_run = [item]
|
|
301
|
+
submit_run()
|
|
302
|
+
|
|
303
|
+
global_width = max(0.1, lane.right - lane.left)
|
|
304
|
+
local_by_source: dict[int, _TextLane] = {}
|
|
305
|
+
for run in runs:
|
|
306
|
+
if len(run) < 3:
|
|
307
|
+
continue
|
|
308
|
+
local_left = statistics.median(bbox[0] for _line, bbox in run)
|
|
309
|
+
local_right = max(bbox[2] for _line, bbox in run)
|
|
310
|
+
local_width = max(0.1, local_right - local_left)
|
|
311
|
+
wide_support = sum(bbox[2] - bbox[0] >= 0.7 * local_width for _line, bbox in run)
|
|
312
|
+
if global_width < 1.4 * local_width or wide_support < 3:
|
|
313
|
+
continue
|
|
314
|
+
local_lane = _TextLane(
|
|
315
|
+
left=local_left,
|
|
316
|
+
right=local_right,
|
|
317
|
+
lines=run,
|
|
318
|
+
is_span=False,
|
|
319
|
+
)
|
|
320
|
+
for line, _bbox in run:
|
|
321
|
+
local_by_source[line.source_index] = local_lane
|
|
322
|
+
return local_by_source
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _structured_text_break_sources(
|
|
326
|
+
lane: _TextLane,
|
|
327
|
+
regular_gap: float,
|
|
328
|
+
gap_mad: float,
|
|
329
|
+
) -> set[int]:
|
|
330
|
+
"""用重复强调首行和前行右侧留白确认结构化正文的新段起点。"""
|
|
331
|
+
|
|
332
|
+
rows = sorted(
|
|
333
|
+
lane.lines,
|
|
334
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
335
|
+
)
|
|
336
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
337
|
+
break_sources: set[int] = set()
|
|
338
|
+
regions: list[list[tuple[_LineItem, BBox]]] = []
|
|
339
|
+
for row in rows:
|
|
340
|
+
if row[0].semantic_type is not None:
|
|
341
|
+
if regions and regions[-1]:
|
|
342
|
+
regions.append([])
|
|
343
|
+
continue
|
|
344
|
+
if not regions:
|
|
345
|
+
regions.append([])
|
|
346
|
+
regions[-1].append(row)
|
|
347
|
+
|
|
348
|
+
for region in regions:
|
|
349
|
+
candidates: list[int] = []
|
|
350
|
+
for index, (line, bbox) in enumerate(region):
|
|
351
|
+
height = _line_effective_height(line, bbox)
|
|
352
|
+
line_width = bbox[2] - bbox[0]
|
|
353
|
+
if (
|
|
354
|
+
line.leading_emphasis_width is not None
|
|
355
|
+
and line.leading_emphasis_width <= 0.2 * lane_width
|
|
356
|
+
and line_width >= 0.95 * lane_width
|
|
357
|
+
and abs(bbox[0] - lane.left) <= 0.75 * height
|
|
358
|
+
):
|
|
359
|
+
candidates.append(index)
|
|
360
|
+
if len(candidates) < 3:
|
|
361
|
+
continue
|
|
362
|
+
for index in candidates:
|
|
363
|
+
if index == 0:
|
|
364
|
+
continue
|
|
365
|
+
previous = region[index - 1]
|
|
366
|
+
current = region[index]
|
|
367
|
+
pair_height = max(
|
|
368
|
+
_line_effective_height(*previous),
|
|
369
|
+
_line_effective_height(*current),
|
|
370
|
+
)
|
|
371
|
+
previous_fill = (previous[1][2] - lane.left) / lane_width
|
|
372
|
+
vertical_gap = _effective_text_row_gap(previous, current)
|
|
373
|
+
if previous_fill <= 0.8 and -0.25 * pair_height <= vertical_gap <= regular_gap + max(
|
|
374
|
+
0.75 * pair_height, 3.0 * gap_mad
|
|
375
|
+
):
|
|
376
|
+
break_sources.add(current[0].source_index)
|
|
377
|
+
return break_sources
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _isolated_indented_paragraph_break_sources(
|
|
381
|
+
lane: _TextLane,
|
|
382
|
+
regular_gap: float,
|
|
383
|
+
gap_mad: float,
|
|
384
|
+
) -> set[int]:
|
|
385
|
+
"""识别短终止尾行之后的缩进首行,并要求下一行回到稳定栏左缘。"""
|
|
386
|
+
|
|
387
|
+
rows = sorted(
|
|
388
|
+
(item for item in lane.lines if item[0].semantic_type is None),
|
|
389
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
390
|
+
)
|
|
391
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
392
|
+
output: set[int] = set()
|
|
393
|
+
terminal_re = re.compile(r"[.!?。!?::;;][\]\)})】》”’'\"]*$")
|
|
394
|
+
for previous, current, following in zip(
|
|
395
|
+
rows,
|
|
396
|
+
rows[1:],
|
|
397
|
+
rows[2:],
|
|
398
|
+
):
|
|
399
|
+
previous_height = _line_effective_height(*previous)
|
|
400
|
+
current_height = _line_effective_height(*current)
|
|
401
|
+
following_height = _line_effective_height(*following)
|
|
402
|
+
pair_height = max(
|
|
403
|
+
previous_height,
|
|
404
|
+
current_height,
|
|
405
|
+
following_height,
|
|
406
|
+
)
|
|
407
|
+
current_indent = current[1][0] - lane.left
|
|
408
|
+
if (
|
|
409
|
+
previous[1][2] - previous[1][0] > 0.3 * lane_width
|
|
410
|
+
or terminal_re.search(previous[0].text.rstrip()) is None
|
|
411
|
+
or not max(5.0, 0.65 * pair_height) <= current_indent <= 3.0 * pair_height
|
|
412
|
+
or current[1][2] - current[1][0] < 0.75 * lane_width
|
|
413
|
+
or abs(following[1][0] - lane.left) > 0.75 * pair_height
|
|
414
|
+
or following[1][2] - following[1][0] < 0.65 * lane_width
|
|
415
|
+
or not _title_fonts_compatible(current[0], following[0])
|
|
416
|
+
):
|
|
417
|
+
continue
|
|
418
|
+
first_gap = _effective_body_text_row_gap(previous, current)
|
|
419
|
+
second_gap = _effective_body_text_row_gap(current, following)
|
|
420
|
+
gap_limit = regular_gap + max(
|
|
421
|
+
0.75 * pair_height,
|
|
422
|
+
3.0 * gap_mad,
|
|
423
|
+
)
|
|
424
|
+
if -0.25 * pair_height <= first_gap <= gap_limit and -0.25 * pair_height <= second_gap <= gap_limit:
|
|
425
|
+
output.add(current[0].source_index)
|
|
426
|
+
return output
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _centered_visual_reset_break_sources(
|
|
430
|
+
lane: _TextLane,
|
|
431
|
+
visual_bboxes: Sequence[BBox],
|
|
432
|
+
local_page_height: float,
|
|
433
|
+
) -> set[int]:
|
|
434
|
+
"""识别视觉主体下方短居中行到更宽居中行的独立注释重启。"""
|
|
435
|
+
|
|
436
|
+
if not visual_bboxes:
|
|
437
|
+
return set()
|
|
438
|
+
rows = sorted(
|
|
439
|
+
(item for item in lane.lines if item[0].semantic_type is None),
|
|
440
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
441
|
+
)
|
|
442
|
+
output: set[int] = set()
|
|
443
|
+
for previous, current in zip(rows, rows[1:]):
|
|
444
|
+
previous_bbox = previous[1]
|
|
445
|
+
current_bbox = current[1]
|
|
446
|
+
previous_width = previous_bbox[2] - previous_bbox[0]
|
|
447
|
+
current_width = current_bbox[2] - current_bbox[0]
|
|
448
|
+
pair_height = max(
|
|
449
|
+
_line_effective_height(*previous),
|
|
450
|
+
_line_effective_height(*current),
|
|
451
|
+
)
|
|
452
|
+
if (
|
|
453
|
+
previous_width > 0.7 * current_width
|
|
454
|
+
or current_bbox[0] > previous_bbox[0] - 0.25 * pair_height
|
|
455
|
+
or current_bbox[2] < previous_bbox[2] + 0.25 * pair_height
|
|
456
|
+
or abs(_bbox_center_x(previous_bbox) - _bbox_center_x(current_bbox)) > 0.1 * current_width
|
|
457
|
+
):
|
|
458
|
+
continue
|
|
459
|
+
vertical_gap = _effective_text_row_gap(previous, current)
|
|
460
|
+
if not -0.25 * pair_height <= vertical_gap <= 0.75 * pair_height:
|
|
461
|
+
continue
|
|
462
|
+
if any(
|
|
463
|
+
-0.25 * pair_height <= previous_bbox[1] - visual_bbox[3] <= max(2.0 * pair_height, 0.03 * local_page_height)
|
|
464
|
+
and _bbox_axis_overlap_ratio(current_bbox, visual_bbox, axis="x") >= 0.8
|
|
465
|
+
and abs(_bbox_center_x(current_bbox) - _bbox_center_x(visual_bbox))
|
|
466
|
+
<= 0.12 * max(current_width, visual_bbox[2] - visual_bbox[0])
|
|
467
|
+
for visual_bbox in visual_bboxes
|
|
468
|
+
):
|
|
469
|
+
output.add(current[0].source_index)
|
|
470
|
+
return output
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _leading_typography_reset_break_sources(
|
|
474
|
+
lane: _TextLane,
|
|
475
|
+
regular_gap: float,
|
|
476
|
+
gap_mad: float,
|
|
477
|
+
) -> set[int]:
|
|
478
|
+
"""识别短尾之后以独立行首字体 run 开启的宽行结构段。"""
|
|
479
|
+
|
|
480
|
+
rows = sorted(
|
|
481
|
+
(item for item in lane.lines if item[0].semantic_type is None),
|
|
482
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
483
|
+
)
|
|
484
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
485
|
+
output: set[int] = set()
|
|
486
|
+
for previous, current in zip(rows, rows[1:]):
|
|
487
|
+
previous_width = previous[1][2] - previous[1][0]
|
|
488
|
+
current_width = current[1][2] - current[1][0]
|
|
489
|
+
pair_height = max(
|
|
490
|
+
_line_effective_height(*previous),
|
|
491
|
+
_line_effective_height(*current),
|
|
492
|
+
)
|
|
493
|
+
if (
|
|
494
|
+
current[0].leading_typography_width is None
|
|
495
|
+
or current[0].leading_typography_width > 0.2 * lane_width
|
|
496
|
+
or previous_width > 0.45 * lane_width
|
|
497
|
+
or current_width < 0.75 * lane_width
|
|
498
|
+
or abs(previous[1][0] - lane.left) > 0.75 * pair_height
|
|
499
|
+
or abs(current[1][0] - lane.left) > 0.75 * pair_height
|
|
500
|
+
or current[0].formula_candidate_only
|
|
501
|
+
or current[0].compact_formula_cluster
|
|
502
|
+
or current[0].inline_math_regions
|
|
503
|
+
):
|
|
504
|
+
continue
|
|
505
|
+
vertical_gap = _effective_body_text_row_gap(previous, current)
|
|
506
|
+
if (
|
|
507
|
+
-0.25 * pair_height
|
|
508
|
+
<= vertical_gap
|
|
509
|
+
<= regular_gap
|
|
510
|
+
+ max(
|
|
511
|
+
0.75 * pair_height,
|
|
512
|
+
3.0 * gap_mad,
|
|
513
|
+
)
|
|
514
|
+
):
|
|
515
|
+
output.add(current[0].source_index)
|
|
516
|
+
return output
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _formula_style_text_row_break_sources(
|
|
520
|
+
lane: _TextLane,
|
|
521
|
+
) -> set[int]:
|
|
522
|
+
"""按相邻显示行几何拆分被公式检测回退为正文的独立文本行。"""
|
|
523
|
+
|
|
524
|
+
rows = sorted(
|
|
525
|
+
(item for item in lane.lines if item[0].semantic_type is None),
|
|
526
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
527
|
+
)
|
|
528
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
529
|
+
matching_edges: set[int] = set()
|
|
530
|
+
for index, (previous, current) in enumerate(zip(rows, rows[1:])):
|
|
531
|
+
previous_line, previous_bbox = previous
|
|
532
|
+
current_line, current_bbox = current
|
|
533
|
+
if not (previous_line.paragraph_formula_context and current_line.paragraph_formula_context):
|
|
534
|
+
continue
|
|
535
|
+
previous_height = _line_effective_height(*previous)
|
|
536
|
+
current_height = _line_effective_height(*current)
|
|
537
|
+
minimum_height = min(previous_height, current_height)
|
|
538
|
+
maximum_height = max(previous_height, current_height)
|
|
539
|
+
if minimum_height < 0.75 * maximum_height:
|
|
540
|
+
continue
|
|
541
|
+
previous_width = previous_bbox[2] - previous_bbox[0]
|
|
542
|
+
current_width = current_bbox[2] - current_bbox[0]
|
|
543
|
+
if min(previous_width, current_width) < 0.45 * lane_width or max(previous_width, current_width) > 0.95 * lane_width:
|
|
544
|
+
continue
|
|
545
|
+
lane_center = 0.5 * (lane.left + lane.right)
|
|
546
|
+
if (
|
|
547
|
+
abs(_bbox_center_x(previous_bbox) - lane_center) > 0.15 * lane_width
|
|
548
|
+
or abs(_bbox_center_x(current_bbox) - lane_center) > 0.15 * lane_width
|
|
549
|
+
):
|
|
550
|
+
continue
|
|
551
|
+
vertical_overlap = max(
|
|
552
|
+
0.0,
|
|
553
|
+
min(previous_bbox[3], current_bbox[3]) - max(previous_bbox[1], current_bbox[1]),
|
|
554
|
+
)
|
|
555
|
+
top_pitch = current_bbox[1] - previous_bbox[1]
|
|
556
|
+
pair_height = statistics.median((previous_height, current_height))
|
|
557
|
+
if vertical_overlap <= 0.2 * minimum_height and 0.9 * pair_height <= top_pitch <= 2.0 * pair_height:
|
|
558
|
+
matching_edges.add(index)
|
|
559
|
+
|
|
560
|
+
output: set[int] = set()
|
|
561
|
+
for index in matching_edges:
|
|
562
|
+
output.add(rows[index][0].source_index)
|
|
563
|
+
output.add(rows[index + 1][0].source_index)
|
|
564
|
+
if index + 2 < len(rows):
|
|
565
|
+
# 同时保护显示行组后的正文起点,避免上下文恢复阶段重新跨界合并。
|
|
566
|
+
output.add(rows[index + 2][0].source_index)
|
|
567
|
+
return output
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _front_matter_keyword_break_sources(
|
|
571
|
+
lane: _TextLane,
|
|
572
|
+
local_page_height: float,
|
|
573
|
+
page_index: int | None,
|
|
574
|
+
) -> set[int]:
|
|
575
|
+
"""把首页关键词和文献元数据行固定为独立文本块起点。"""
|
|
576
|
+
|
|
577
|
+
if page_index != 0:
|
|
578
|
+
return set()
|
|
579
|
+
return {
|
|
580
|
+
line.source_index
|
|
581
|
+
for line, bbox in lane.lines
|
|
582
|
+
if line.semantic_type is None
|
|
583
|
+
and bbox[1] <= 0.65 * local_page_height
|
|
584
|
+
and _FRONT_MATTER_FIELD_RE.match(line.text) is not None
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def _component_starts_with_emphasized_row(
|
|
589
|
+
lines: list[_LineItem],
|
|
590
|
+
) -> bool:
|
|
591
|
+
"""识别行内强调或首行字重显著高于后续正文的组件起点。"""
|
|
592
|
+
|
|
593
|
+
if not lines:
|
|
594
|
+
return False
|
|
595
|
+
if lines[0].leading_emphasis_width is not None:
|
|
596
|
+
return True
|
|
597
|
+
first_weight = lines[0].dominant_font_weight
|
|
598
|
+
following_weights = [line.dominant_font_weight for line in lines[1:] if line.dominant_font_weight is not None]
|
|
599
|
+
if first_weight is None or not following_weights:
|
|
600
|
+
return False
|
|
601
|
+
body_weight = statistics.median(following_weights)
|
|
602
|
+
return first_weight - body_weight >= 100.0 and first_weight >= 1.15 * max(1.0, body_weight)
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
def _explicit_text_break_sources(
|
|
606
|
+
lane: _TextLane,
|
|
607
|
+
) -> set[int]:
|
|
608
|
+
"""用通用列表标记和 E-mail 元数据确认正文中的显式硬分段。"""
|
|
609
|
+
|
|
610
|
+
rows = sorted(
|
|
611
|
+
(item for item in lane.lines if item[0].semantic_type is None),
|
|
612
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
613
|
+
)
|
|
614
|
+
output = {line.source_index for line, _bbox in rows if _ABSTRACT_METADATA_RE.match(line.text) is not None}
|
|
615
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
616
|
+
output.update(
|
|
617
|
+
line.source_index
|
|
618
|
+
for line, bbox in rows
|
|
619
|
+
if _BULLET_ITEM_RE.match(line.text) is not None and bbox[2] - bbox[0] >= 0.8 * lane_width
|
|
620
|
+
)
|
|
621
|
+
for index, (line, bbox) in enumerate(rows):
|
|
622
|
+
if _EMAIL_METADATA_RE.match(line.text) is None or index == 0:
|
|
623
|
+
continue
|
|
624
|
+
previous_line, previous_bbox = rows[index - 1]
|
|
625
|
+
pair_height = max(
|
|
626
|
+
_line_effective_height(previous_line, previous_bbox),
|
|
627
|
+
_line_effective_height(line, bbox),
|
|
628
|
+
)
|
|
629
|
+
if abs(bbox[0] - previous_bbox[0]) <= 0.75 * pair_height:
|
|
630
|
+
output.add(line.source_index)
|
|
631
|
+
for row_index, (previous, current) in enumerate(
|
|
632
|
+
zip(rows, rows[1:]),
|
|
633
|
+
):
|
|
634
|
+
previous_is_label = _LABELLED_METADATA_RE.match(
|
|
635
|
+
previous[0].text,
|
|
636
|
+
)
|
|
637
|
+
current_is_label = _LABELLED_METADATA_RE.match(
|
|
638
|
+
current[0].text,
|
|
639
|
+
)
|
|
640
|
+
label_pair_height = max(
|
|
641
|
+
_line_effective_height(*previous),
|
|
642
|
+
_line_effective_height(*current),
|
|
643
|
+
)
|
|
644
|
+
if (
|
|
645
|
+
previous_is_label is not None
|
|
646
|
+
and current_is_label is not None
|
|
647
|
+
and _URL_LINE_RE.match(current[0].text) is None
|
|
648
|
+
and len(previous_is_label.group("label")) >= 4
|
|
649
|
+
and len(current_is_label.group("label")) >= 4
|
|
650
|
+
and any("\u3400" <= char <= "\u9fff" for char in previous_is_label.group("label"))
|
|
651
|
+
and any("\u3400" <= char <= "\u9fff" for char in current_is_label.group("label"))
|
|
652
|
+
and previous_is_label.group("label").casefold() != current_is_label.group("label").casefold()
|
|
653
|
+
and current[1][1] - previous[1][1] <= 2.0 * label_pair_height
|
|
654
|
+
and previous[1][2] - previous[1][0] <= 0.75 * lane_width
|
|
655
|
+
and current[1][2] - current[1][0] <= 0.75 * lane_width
|
|
656
|
+
):
|
|
657
|
+
output.add(current[0].source_index)
|
|
658
|
+
pair_height = max(
|
|
659
|
+
_line_effective_height(*previous),
|
|
660
|
+
_line_effective_height(*current),
|
|
661
|
+
)
|
|
662
|
+
next_row = rows[row_index + 2] if row_index + 2 < len(rows) else None
|
|
663
|
+
indented_item_continuation = (
|
|
664
|
+
current[1][0] - lane.left >= max(5.0, 0.65 * pair_height)
|
|
665
|
+
and next_row is not None
|
|
666
|
+
and next_row[1][0] - lane.left <= 0.5 * pair_height
|
|
667
|
+
and 0.5 * pair_height <= next_row[1][1] - current[1][1] <= 2.25 * pair_height
|
|
668
|
+
)
|
|
669
|
+
if (
|
|
670
|
+
_LIST_ITEM_RE.match(current[0].text) is not None
|
|
671
|
+
and previous[0].text.rstrip().endswith((":", ":"))
|
|
672
|
+
and previous[1][2] - previous[1][0] <= 0.8 * lane_width
|
|
673
|
+
and indented_item_continuation
|
|
674
|
+
):
|
|
675
|
+
output.add(current[0].source_index)
|
|
676
|
+
return output
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
__all__ = [
|
|
680
|
+
"_local_tight_output_line_bboxes",
|
|
681
|
+
"_starts_structural_reference_entry",
|
|
682
|
+
"_build_hanging_indent_group_map",
|
|
683
|
+
"_infer_local_text_lane_map",
|
|
684
|
+
"_structured_text_break_sources",
|
|
685
|
+
"_isolated_indented_paragraph_break_sources",
|
|
686
|
+
"_centered_visual_reset_break_sources",
|
|
687
|
+
"_leading_typography_reset_break_sources",
|
|
688
|
+
"_formula_style_text_row_break_sources",
|
|
689
|
+
"_front_matter_keyword_break_sources",
|
|
690
|
+
"_component_starts_with_emphasized_row",
|
|
691
|
+
"_explicit_text_break_sources",
|
|
692
|
+
]
|