docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1081 @@
|
|
|
1
|
+
"""识别编号、排版重置及跨页一致的结构标题。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import statistics
|
|
7
|
+
import unicodedata
|
|
8
|
+
from collections import Counter
|
|
9
|
+
from dataclasses import replace
|
|
10
|
+
|
|
11
|
+
from .....schema import BBox
|
|
12
|
+
from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_x, _bbox_center_y, _bbox_union_many, _rotate_bbox_to_upright
|
|
13
|
+
from ..line_layout import (
|
|
14
|
+
_effective_text_row_gap,
|
|
15
|
+
_estimate_lane_gap,
|
|
16
|
+
_font_signatures_share_family,
|
|
17
|
+
_infer_text_lanes,
|
|
18
|
+
_line_canonical_style_scale,
|
|
19
|
+
_line_effective_height,
|
|
20
|
+
_normalized_font_family,
|
|
21
|
+
_title_fonts_compatible,
|
|
22
|
+
)
|
|
23
|
+
from ..models import _DocumentBodyProfile, _DocumentTitleProfile, _LineItem, _PreparedPage, _TextLane
|
|
24
|
+
from .body_profile import _line_uses_document_regular_font
|
|
25
|
+
from .common import (
|
|
26
|
+
_NUMBERED_SECTION_TITLE_RE,
|
|
27
|
+
_SECTION_NUMBER_ONLY_RE,
|
|
28
|
+
_SECTION_TITLE_TERMINAL_RE,
|
|
29
|
+
_UNNUMBERED_SECTION_HEADING_RE,
|
|
30
|
+
_build_physical_title_gap_map,
|
|
31
|
+
_line_inside_visual_container,
|
|
32
|
+
)
|
|
33
|
+
from .page_titles import _classify_page_titles
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _normalized_section_title_text(text: str) -> str:
|
|
37
|
+
"""规范全半角编号和空白,仅用于结构标题规则判断。"""
|
|
38
|
+
|
|
39
|
+
return re.sub(
|
|
40
|
+
r"\s+",
|
|
41
|
+
" ",
|
|
42
|
+
unicodedata.normalize("NFKC", text),
|
|
43
|
+
).strip()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _is_plausible_section_number(
|
|
47
|
+
number: str,
|
|
48
|
+
label: str = "",
|
|
49
|
+
) -> bool:
|
|
50
|
+
"""排除年代、小数值和整句正文冒充的章节编号。"""
|
|
51
|
+
|
|
52
|
+
parts = [int(part) for part in re.findall(r"\d+", number)]
|
|
53
|
+
if not parts or any(part > 99 for part in parts):
|
|
54
|
+
return False
|
|
55
|
+
if len(parts) > 1 and parts[0] == 0:
|
|
56
|
+
return False
|
|
57
|
+
if len(parts) == 1 and parts[0] > 12:
|
|
58
|
+
return False
|
|
59
|
+
stripped_label = label.lstrip()
|
|
60
|
+
if stripped_label and not stripped_label[0].isalpha():
|
|
61
|
+
return False
|
|
62
|
+
if stripped_label and stripped_label[0].isascii() and stripped_label[0].isalpha() and not stripped_label[0].isupper():
|
|
63
|
+
return False
|
|
64
|
+
if len(re.findall(r"\d+(?:\.\d+)?", label)) >= 2:
|
|
65
|
+
return False
|
|
66
|
+
return not any(char in label for char in ",,;;::。!?!?")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _section_title_has_body_followers(
|
|
70
|
+
title_bbox: BBox,
|
|
71
|
+
geometry: list[tuple[_LineItem, BBox]],
|
|
72
|
+
body_height: float,
|
|
73
|
+
local_page_width: float,
|
|
74
|
+
*,
|
|
75
|
+
minimum_count: int,
|
|
76
|
+
) -> bool:
|
|
77
|
+
"""检查紧随标题的同栏常规正文行,避免把页码和数值标成标题。"""
|
|
78
|
+
|
|
79
|
+
followers = 0
|
|
80
|
+
for line, bbox in sorted(
|
|
81
|
+
geometry,
|
|
82
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
83
|
+
):
|
|
84
|
+
if bbox[1] <= title_bbox[1] + 0.4 * body_height:
|
|
85
|
+
continue
|
|
86
|
+
if bbox[1] - title_bbox[3] > 8.0 * body_height:
|
|
87
|
+
break
|
|
88
|
+
line_height = _line_effective_height(line, bbox)
|
|
89
|
+
horizontally_related = (
|
|
90
|
+
_bbox_axis_overlap_ratio(title_bbox, bbox, axis="x") >= 0.15 or abs(bbox[0] - title_bbox[0]) <= 2.5 * body_height
|
|
91
|
+
)
|
|
92
|
+
if (
|
|
93
|
+
line.semantic_type is None
|
|
94
|
+
and horizontally_related
|
|
95
|
+
and bbox[2] - bbox[0] >= 0.25 * local_page_width
|
|
96
|
+
and 0.7 <= line_height / body_height <= 1.4
|
|
97
|
+
):
|
|
98
|
+
followers += 1
|
|
99
|
+
if followers >= minimum_count:
|
|
100
|
+
return True
|
|
101
|
+
return False
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _classify_explicit_section_titles(
|
|
105
|
+
lines: list[_LineItem],
|
|
106
|
+
page_size: tuple[float, float],
|
|
107
|
+
*,
|
|
108
|
+
container_bboxes: list[BBox],
|
|
109
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
110
|
+
) -> None:
|
|
111
|
+
"""以通用编号行和紧凑结构转折补齐正文同字号章节标题。"""
|
|
112
|
+
|
|
113
|
+
if document_body_profile is None or document_body_profile.body_height <= 0:
|
|
114
|
+
return
|
|
115
|
+
body_height = document_body_profile.body_height
|
|
116
|
+
for angle in sorted(
|
|
117
|
+
{
|
|
118
|
+
line.angle
|
|
119
|
+
for line in lines
|
|
120
|
+
if (line.semantic_type is None or line.explicit_section_title) and not line.title_suppressed
|
|
121
|
+
}
|
|
122
|
+
):
|
|
123
|
+
geometry = sorted(
|
|
124
|
+
[
|
|
125
|
+
(
|
|
126
|
+
line,
|
|
127
|
+
_rotate_bbox_to_upright(
|
|
128
|
+
line.bbox,
|
|
129
|
+
page_size,
|
|
130
|
+
angle,
|
|
131
|
+
),
|
|
132
|
+
)
|
|
133
|
+
for line in lines
|
|
134
|
+
if line.angle == angle and line.semantic_type is None and not line.title_suppressed
|
|
135
|
+
],
|
|
136
|
+
key=lambda item: (
|
|
137
|
+
item[1][1],
|
|
138
|
+
item[1][0],
|
|
139
|
+
item[0].source_index,
|
|
140
|
+
),
|
|
141
|
+
)
|
|
142
|
+
if not geometry:
|
|
143
|
+
continue
|
|
144
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
145
|
+
local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
|
|
146
|
+
local_containers = [
|
|
147
|
+
_rotate_bbox_to_upright(
|
|
148
|
+
bbox,
|
|
149
|
+
page_size,
|
|
150
|
+
angle,
|
|
151
|
+
)
|
|
152
|
+
for bbox in container_bboxes
|
|
153
|
+
]
|
|
154
|
+
numbered_groups: list[list[tuple[_LineItem, BBox]]] = []
|
|
155
|
+
grouped_sources: set[int] = set()
|
|
156
|
+
for line, bbox in geometry:
|
|
157
|
+
normalized = _normalized_section_title_text(line.text)
|
|
158
|
+
numbered_match = _NUMBERED_SECTION_TITLE_RE.match(
|
|
159
|
+
normalized,
|
|
160
|
+
)
|
|
161
|
+
if numbered_match is not None and _is_plausible_section_number(
|
|
162
|
+
numbered_match.group("number"),
|
|
163
|
+
numbered_match.group("label"),
|
|
164
|
+
):
|
|
165
|
+
numbered_groups.append([(line, bbox)])
|
|
166
|
+
grouped_sources.add(line.source_index)
|
|
167
|
+
continue
|
|
168
|
+
if _SECTION_NUMBER_ONLY_RE.match(normalized) is None or not _is_plausible_section_number(normalized):
|
|
169
|
+
continue
|
|
170
|
+
marker_height = _line_effective_height(line, bbox)
|
|
171
|
+
companions = [
|
|
172
|
+
(candidate, candidate_bbox)
|
|
173
|
+
for candidate, candidate_bbox in geometry
|
|
174
|
+
if candidate is not line
|
|
175
|
+
and candidate.source_index not in grouped_sources
|
|
176
|
+
and candidate_bbox[0] >= bbox[2]
|
|
177
|
+
and candidate_bbox[0] - bbox[2] <= 4.0 * max(body_height, marker_height)
|
|
178
|
+
and _bbox_axis_overlap_ratio(
|
|
179
|
+
bbox,
|
|
180
|
+
candidate_bbox,
|
|
181
|
+
axis="y",
|
|
182
|
+
)
|
|
183
|
+
>= 0.5
|
|
184
|
+
and candidate_bbox[2] - candidate_bbox[0] <= 0.55 * local_page_width
|
|
185
|
+
and _SECTION_TITLE_TERMINAL_RE.search(
|
|
186
|
+
_normalized_section_title_text(candidate.text),
|
|
187
|
+
)
|
|
188
|
+
is None
|
|
189
|
+
]
|
|
190
|
+
if not companions:
|
|
191
|
+
continue
|
|
192
|
+
companion = min(
|
|
193
|
+
companions,
|
|
194
|
+
key=lambda item: (
|
|
195
|
+
item[1][0] - bbox[2],
|
|
196
|
+
item[1][1],
|
|
197
|
+
),
|
|
198
|
+
)
|
|
199
|
+
numbered_groups.append([(line, bbox), companion])
|
|
200
|
+
grouped_sources.update({line.source_index, companion[0].source_index})
|
|
201
|
+
|
|
202
|
+
for group in numbered_groups:
|
|
203
|
+
title_bbox = _bbox_union_many(
|
|
204
|
+
[bbox for _line, bbox in group],
|
|
205
|
+
)
|
|
206
|
+
group_line_ids = {id(line) for line, _bbox in group}
|
|
207
|
+
preceding = [
|
|
208
|
+
previous_bbox
|
|
209
|
+
for previous_line, previous_bbox in geometry
|
|
210
|
+
if id(previous_line) not in group_line_ids
|
|
211
|
+
and previous_bbox[3] <= title_bbox[1]
|
|
212
|
+
and (
|
|
213
|
+
_bbox_axis_overlap_ratio(
|
|
214
|
+
previous_bbox,
|
|
215
|
+
title_bbox,
|
|
216
|
+
axis="x",
|
|
217
|
+
)
|
|
218
|
+
>= 0.15
|
|
219
|
+
or abs(previous_bbox[0] - title_bbox[0]) <= 2.5 * body_height
|
|
220
|
+
)
|
|
221
|
+
]
|
|
222
|
+
gap_above = title_bbox[1] - max(previous_bbox[3] for previous_bbox in preceding) if preceding else body_height
|
|
223
|
+
if (
|
|
224
|
+
title_bbox[2] - title_bbox[0] > 0.7 * local_page_width
|
|
225
|
+
or not 0.1 * local_page_height <= _bbox_center_y(title_bbox) <= 0.93 * local_page_height
|
|
226
|
+
or any(
|
|
227
|
+
_bbox_axis_overlap_ratio(
|
|
228
|
+
title_bbox,
|
|
229
|
+
container_bbox,
|
|
230
|
+
axis="x",
|
|
231
|
+
)
|
|
232
|
+
>= 0.8
|
|
233
|
+
and _bbox_axis_overlap_ratio(
|
|
234
|
+
title_bbox,
|
|
235
|
+
container_bbox,
|
|
236
|
+
axis="y",
|
|
237
|
+
)
|
|
238
|
+
>= 0.8
|
|
239
|
+
for container_bbox in local_containers
|
|
240
|
+
)
|
|
241
|
+
or not _section_title_has_body_followers(
|
|
242
|
+
title_bbox,
|
|
243
|
+
geometry,
|
|
244
|
+
body_height,
|
|
245
|
+
local_page_width,
|
|
246
|
+
minimum_count=1,
|
|
247
|
+
)
|
|
248
|
+
or gap_above < 0.4 * body_height
|
|
249
|
+
):
|
|
250
|
+
continue
|
|
251
|
+
for line, _bbox in group:
|
|
252
|
+
line.semantic_type = "paragraph_title"
|
|
253
|
+
line.structural_title = True
|
|
254
|
+
line.explicit_section_title = True
|
|
255
|
+
|
|
256
|
+
for line, bbox in geometry:
|
|
257
|
+
if line.semantic_type is not None:
|
|
258
|
+
continue
|
|
259
|
+
normalized = _normalized_section_title_text(line.text)
|
|
260
|
+
canonical_heading = normalized.strip(
|
|
261
|
+
"[][]【】()()",
|
|
262
|
+
).replace(" ", "")
|
|
263
|
+
if (
|
|
264
|
+
_UNNUMBERED_SECTION_HEADING_RE.fullmatch(
|
|
265
|
+
canonical_heading,
|
|
266
|
+
)
|
|
267
|
+
is None
|
|
268
|
+
or not 2 <= len(normalized) <= 24
|
|
269
|
+
or _SECTION_TITLE_TERMINAL_RE.search(normalized) is not None
|
|
270
|
+
or any(char in normalized for char in "[][]")
|
|
271
|
+
or bbox[2] - bbox[0] > 0.3 * local_page_width
|
|
272
|
+
or _bbox_center_y(bbox) < 0.35 * local_page_height
|
|
273
|
+
or not 0.75 <= _line_effective_height(line, bbox) / body_height <= 1.4
|
|
274
|
+
or any(
|
|
275
|
+
_bbox_axis_overlap_ratio(
|
|
276
|
+
bbox,
|
|
277
|
+
container_bbox,
|
|
278
|
+
axis="x",
|
|
279
|
+
)
|
|
280
|
+
>= 0.8
|
|
281
|
+
and _bbox_axis_overlap_ratio(
|
|
282
|
+
bbox,
|
|
283
|
+
container_bbox,
|
|
284
|
+
axis="y",
|
|
285
|
+
)
|
|
286
|
+
>= 0.8
|
|
287
|
+
for container_bbox in local_containers
|
|
288
|
+
)
|
|
289
|
+
or not _section_title_has_body_followers(
|
|
290
|
+
bbox,
|
|
291
|
+
geometry,
|
|
292
|
+
body_height,
|
|
293
|
+
local_page_width,
|
|
294
|
+
minimum_count=2,
|
|
295
|
+
)
|
|
296
|
+
):
|
|
297
|
+
continue
|
|
298
|
+
preceding = [
|
|
299
|
+
previous_bbox
|
|
300
|
+
for previous_line, previous_bbox in geometry
|
|
301
|
+
if previous_line is not line
|
|
302
|
+
and previous_bbox[3] <= bbox[1]
|
|
303
|
+
and (
|
|
304
|
+
_bbox_axis_overlap_ratio(
|
|
305
|
+
previous_bbox,
|
|
306
|
+
bbox,
|
|
307
|
+
axis="x",
|
|
308
|
+
)
|
|
309
|
+
>= 0.15
|
|
310
|
+
or abs(previous_bbox[0] - bbox[0]) <= 2.5 * body_height
|
|
311
|
+
)
|
|
312
|
+
]
|
|
313
|
+
gap_above = bbox[1] - max(previous_bbox[3] for previous_bbox in preceding) if preceding else body_height
|
|
314
|
+
if gap_above >= 0.5 * body_height:
|
|
315
|
+
line.semantic_type = "paragraph_title"
|
|
316
|
+
line.structural_title = True
|
|
317
|
+
line.explicit_section_title = True
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def _classify_document_structural_titles(
|
|
321
|
+
prepared_pages: list[_PreparedPage],
|
|
322
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
323
|
+
*,
|
|
324
|
+
legacy_body_profile: _DocumentBodyProfile | None,
|
|
325
|
+
document_title_profile: _DocumentTitleProfile | None,
|
|
326
|
+
) -> None:
|
|
327
|
+
"""用跨页稳定栏带和段前后转折补齐正文同字号标题。"""
|
|
328
|
+
|
|
329
|
+
if document_body_profile is None or not document_body_profile.has_style_scale_repairs:
|
|
330
|
+
return
|
|
331
|
+
probe_pages = [
|
|
332
|
+
replace(
|
|
333
|
+
prepared,
|
|
334
|
+
remaining_lines=[
|
|
335
|
+
replace(
|
|
336
|
+
line,
|
|
337
|
+
style_scale_repaired=True,
|
|
338
|
+
structural_title=False,
|
|
339
|
+
)
|
|
340
|
+
for line in prepared.remaining_lines
|
|
341
|
+
],
|
|
342
|
+
)
|
|
343
|
+
for prepared in prepared_pages
|
|
344
|
+
]
|
|
345
|
+
_classify_document_structural_title_candidates(
|
|
346
|
+
probe_pages,
|
|
347
|
+
document_body_profile,
|
|
348
|
+
)
|
|
349
|
+
canonical_candidate_sources = {
|
|
350
|
+
(page_index, line.source_index)
|
|
351
|
+
for page_index, prepared in enumerate(probe_pages)
|
|
352
|
+
for line in prepared.remaining_lines
|
|
353
|
+
if line.structural_title
|
|
354
|
+
}
|
|
355
|
+
legacy_title_sources = _collect_legacy_paragraph_title_sources(
|
|
356
|
+
prepared_pages,
|
|
357
|
+
legacy_body_profile,
|
|
358
|
+
document_title_profile,
|
|
359
|
+
)
|
|
360
|
+
body_height = max(0.1, document_body_profile.body_height)
|
|
361
|
+
canonical_style_candidate_pages: dict[
|
|
362
|
+
tuple[str, int, float, int],
|
|
363
|
+
list[int],
|
|
364
|
+
] = {}
|
|
365
|
+
for page_index, prepared in enumerate(prepared_pages):
|
|
366
|
+
for line in prepared.remaining_lines:
|
|
367
|
+
line_key = (page_index, line.source_index)
|
|
368
|
+
if line_key not in canonical_candidate_sources:
|
|
369
|
+
continue
|
|
370
|
+
local_bbox = _rotate_bbox_to_upright(
|
|
371
|
+
line.source_bbox or line.bbox,
|
|
372
|
+
prepared.page_size,
|
|
373
|
+
line.angle,
|
|
374
|
+
)
|
|
375
|
+
layout_ratio = (local_bbox[3] - local_bbox[1]) / body_height
|
|
376
|
+
if (
|
|
377
|
+
line_key in legacy_title_sources
|
|
378
|
+
or layout_ratio >= 1.8
|
|
379
|
+
or not _line_uses_document_regular_font(
|
|
380
|
+
line,
|
|
381
|
+
document_body_profile,
|
|
382
|
+
)
|
|
383
|
+
):
|
|
384
|
+
line.semantic_type = "paragraph_title"
|
|
385
|
+
line.structural_title = True
|
|
386
|
+
if (
|
|
387
|
+
line_key not in legacy_title_sources
|
|
388
|
+
and layout_ratio >= 1.8
|
|
389
|
+
and (
|
|
390
|
+
style_key := _canonical_title_style_key(
|
|
391
|
+
line,
|
|
392
|
+
)
|
|
393
|
+
)
|
|
394
|
+
is not None
|
|
395
|
+
):
|
|
396
|
+
canonical_style_candidate_pages.setdefault(
|
|
397
|
+
style_key,
|
|
398
|
+
[],
|
|
399
|
+
).append(page_index)
|
|
400
|
+
canonical_style_prototypes = {
|
|
401
|
+
style_key
|
|
402
|
+
for style_key, page_indices in canonical_style_candidate_pages.items()
|
|
403
|
+
if len(set(page_indices)) >= 2 or max(Counter(page_indices).values(), default=0) >= 3
|
|
404
|
+
}
|
|
405
|
+
if canonical_style_prototypes:
|
|
406
|
+
for prepared in prepared_pages:
|
|
407
|
+
prepared.canonical_formula_geometry = True
|
|
408
|
+
for line in prepared.remaining_lines:
|
|
409
|
+
style_key = _canonical_title_style_key(line)
|
|
410
|
+
if style_key in canonical_style_prototypes:
|
|
411
|
+
line.style_scale_repaired = True
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _promote_noninitial_document_title_band(
|
|
415
|
+
lines: list[_LineItem],
|
|
416
|
+
page_size: tuple[float, float],
|
|
417
|
+
*,
|
|
418
|
+
page_index: int,
|
|
419
|
+
container_bboxes: list[BBox],
|
|
420
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
421
|
+
title_candidate_source_indices: set[int],
|
|
422
|
+
) -> None:
|
|
423
|
+
"""把非首页中已确认且显著大于正文的最强段落标题带升为文档标题。"""
|
|
424
|
+
|
|
425
|
+
if page_index == 0 or document_body_profile is None:
|
|
426
|
+
return
|
|
427
|
+
body_height = max(0.1, document_body_profile.body_height)
|
|
428
|
+
page_candidates: list[
|
|
429
|
+
tuple[
|
|
430
|
+
tuple[float, float, int, float],
|
|
431
|
+
list[tuple[_LineItem, BBox]],
|
|
432
|
+
]
|
|
433
|
+
] = []
|
|
434
|
+
for angle in sorted(
|
|
435
|
+
{
|
|
436
|
+
line.angle
|
|
437
|
+
for line in lines
|
|
438
|
+
if line.source_index in title_candidate_source_indices
|
|
439
|
+
and line.semantic_type in {None, "paragraph_title"}
|
|
440
|
+
and not line.title_suppressed
|
|
441
|
+
}
|
|
442
|
+
):
|
|
443
|
+
geometry = sorted(
|
|
444
|
+
[
|
|
445
|
+
(
|
|
446
|
+
line,
|
|
447
|
+
_rotate_bbox_to_upright(
|
|
448
|
+
line.bbox,
|
|
449
|
+
page_size,
|
|
450
|
+
angle,
|
|
451
|
+
),
|
|
452
|
+
)
|
|
453
|
+
for line in lines
|
|
454
|
+
if line.angle == angle
|
|
455
|
+
and line.source_index in title_candidate_source_indices
|
|
456
|
+
and line.semantic_type in {None, "paragraph_title"}
|
|
457
|
+
and not line.title_suppressed
|
|
458
|
+
],
|
|
459
|
+
key=lambda item: (
|
|
460
|
+
item[1][1],
|
|
461
|
+
item[1][0],
|
|
462
|
+
item[0].source_index,
|
|
463
|
+
),
|
|
464
|
+
)
|
|
465
|
+
if not geometry:
|
|
466
|
+
continue
|
|
467
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
468
|
+
local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
|
|
469
|
+
local_containers = [
|
|
470
|
+
_rotate_bbox_to_upright(
|
|
471
|
+
bbox,
|
|
472
|
+
page_size,
|
|
473
|
+
angle,
|
|
474
|
+
)
|
|
475
|
+
for bbox in container_bboxes
|
|
476
|
+
]
|
|
477
|
+
for index, (line, bbox) in enumerate(geometry):
|
|
478
|
+
line_scale = _line_effective_height(line, bbox)
|
|
479
|
+
width_ratio = (bbox[2] - bbox[0]) / max(
|
|
480
|
+
0.1,
|
|
481
|
+
local_page_width,
|
|
482
|
+
)
|
|
483
|
+
centered = abs(_bbox_center_x(bbox) - 0.5 * local_page_width) <= 0.08 * local_page_width
|
|
484
|
+
if (
|
|
485
|
+
line_scale < 1.25 * body_height
|
|
486
|
+
or not 0.45 <= width_ratio <= 0.85
|
|
487
|
+
or not centered
|
|
488
|
+
or not 0.12 * local_page_height <= _bbox_center_y(bbox) <= 0.75 * local_page_height
|
|
489
|
+
or _line_inside_visual_container(
|
|
490
|
+
bbox,
|
|
491
|
+
local_containers,
|
|
492
|
+
)
|
|
493
|
+
):
|
|
494
|
+
continue
|
|
495
|
+
|
|
496
|
+
title_members = [(line, bbox)]
|
|
497
|
+
cursor = index + 1
|
|
498
|
+
while cursor < len(geometry):
|
|
499
|
+
candidate_line, candidate_bbox = geometry[cursor]
|
|
500
|
+
candidate_scale = _line_effective_height(
|
|
501
|
+
candidate_line,
|
|
502
|
+
candidate_bbox,
|
|
503
|
+
)
|
|
504
|
+
vertical_gap = max(
|
|
505
|
+
0.0,
|
|
506
|
+
candidate_bbox[1] - title_members[-1][1][3],
|
|
507
|
+
)
|
|
508
|
+
if (
|
|
509
|
+
candidate_scale < 1.2 * body_height
|
|
510
|
+
or vertical_gap
|
|
511
|
+
> 1.5
|
|
512
|
+
* max(
|
|
513
|
+
line_scale,
|
|
514
|
+
candidate_scale,
|
|
515
|
+
)
|
|
516
|
+
or abs(_bbox_center_x(candidate_bbox) - 0.5 * local_page_width) > 0.1 * local_page_width
|
|
517
|
+
or candidate_bbox[2] - candidate_bbox[0] > 0.85 * local_page_width
|
|
518
|
+
or _line_inside_visual_container(
|
|
519
|
+
candidate_bbox,
|
|
520
|
+
local_containers,
|
|
521
|
+
)
|
|
522
|
+
or not _title_fonts_compatible(
|
|
523
|
+
title_members[-1][0],
|
|
524
|
+
candidate_line,
|
|
525
|
+
)
|
|
526
|
+
):
|
|
527
|
+
break
|
|
528
|
+
title_members.append(
|
|
529
|
+
(candidate_line, candidate_bbox),
|
|
530
|
+
)
|
|
531
|
+
cursor += 1
|
|
532
|
+
|
|
533
|
+
title_scales = [_line_effective_height(title_line, title_bbox) for title_line, title_bbox in title_members]
|
|
534
|
+
title_bbox = _bbox_union_many(
|
|
535
|
+
[member_bbox for _member_line, member_bbox in title_members],
|
|
536
|
+
)
|
|
537
|
+
page_candidates.append(
|
|
538
|
+
(
|
|
539
|
+
(
|
|
540
|
+
statistics.median(title_scales) / body_height,
|
|
541
|
+
sum(member_bbox[2] - member_bbox[0] for _member_line, member_bbox in title_members) / local_page_width,
|
|
542
|
+
len(title_members),
|
|
543
|
+
-_bbox_center_y(title_bbox) / local_page_height,
|
|
544
|
+
),
|
|
545
|
+
title_members,
|
|
546
|
+
)
|
|
547
|
+
)
|
|
548
|
+
|
|
549
|
+
if not page_candidates:
|
|
550
|
+
return
|
|
551
|
+
_score, title_members = max(
|
|
552
|
+
page_candidates,
|
|
553
|
+
key=lambda item: item[0],
|
|
554
|
+
)
|
|
555
|
+
for title_line, _title_bbox in title_members:
|
|
556
|
+
title_line.semantic_type = "doc_title"
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _canonical_title_style_key(
|
|
560
|
+
line: _LineItem,
|
|
561
|
+
) -> tuple[str, int, float, int] | None:
|
|
562
|
+
"""返回 canonical-only 标题向同样式正文传播时使用的稳定键。"""
|
|
563
|
+
|
|
564
|
+
if line.font_signature is None or line.em_height <= 0:
|
|
565
|
+
return None
|
|
566
|
+
font_family = _normalized_font_family(line.font_signature)
|
|
567
|
+
if font_family is None:
|
|
568
|
+
return None
|
|
569
|
+
return (
|
|
570
|
+
font_family,
|
|
571
|
+
line.font_signature[1],
|
|
572
|
+
round(line.em_height * 4.0) / 4.0,
|
|
573
|
+
line.angle,
|
|
574
|
+
)
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _classify_document_structural_title_candidates(
|
|
578
|
+
prepared_pages: list[_PreparedPage],
|
|
579
|
+
document_body_profile: _DocumentBodyProfile,
|
|
580
|
+
) -> None:
|
|
581
|
+
"""在 canonical 行副本上收集所有满足结构转折的标题候选。"""
|
|
582
|
+
|
|
583
|
+
body_height = max(0.1, document_body_profile.body_height)
|
|
584
|
+
strong_candidates: list[tuple[int, _LineItem, tuple[str, int] | None, float]] = []
|
|
585
|
+
start_candidates: list[tuple[int, _LineItem, tuple[str, int] | None, float]] = []
|
|
586
|
+
accepted_anchor_positions: list[tuple[int, int, float, float]] = []
|
|
587
|
+
for page_index, prepared in enumerate(prepared_pages):
|
|
588
|
+
container_bboxes = [
|
|
589
|
+
block["bbox"] for block in prepared.fixed_blocks if not isinstance(block.get("_inline_visual_row_id"), int)
|
|
590
|
+
]
|
|
591
|
+
for angle in sorted({line.angle for line in prepared.remaining_lines if line.semantic_type is None}):
|
|
592
|
+
geometry = sorted(
|
|
593
|
+
[
|
|
594
|
+
(
|
|
595
|
+
line,
|
|
596
|
+
_rotate_bbox_to_upright(
|
|
597
|
+
line.source_bbox or line.bbox,
|
|
598
|
+
prepared.page_size,
|
|
599
|
+
angle,
|
|
600
|
+
),
|
|
601
|
+
)
|
|
602
|
+
for line in prepared.remaining_lines
|
|
603
|
+
if line.angle == angle and line.semantic_type is None
|
|
604
|
+
],
|
|
605
|
+
key=lambda item: (
|
|
606
|
+
item[1][1],
|
|
607
|
+
item[1][0],
|
|
608
|
+
item[0].source_index,
|
|
609
|
+
),
|
|
610
|
+
)
|
|
611
|
+
if len(geometry) < 4:
|
|
612
|
+
continue
|
|
613
|
+
local_page_width = prepared.page_size[1] if angle in {90, 270} else prepared.page_size[0]
|
|
614
|
+
local_page_height = prepared.page_size[0] if angle in {90, 270} else prepared.page_size[1]
|
|
615
|
+
median_height = statistics.median(_line_canonical_style_scale(line, bbox) for line, bbox in geometry)
|
|
616
|
+
lanes = _infer_text_lanes(
|
|
617
|
+
geometry,
|
|
618
|
+
local_page_width,
|
|
619
|
+
median_height,
|
|
620
|
+
)
|
|
621
|
+
physical_gaps = _build_physical_title_gap_map(geometry)
|
|
622
|
+
local_containers = [
|
|
623
|
+
_rotate_bbox_to_upright(
|
|
624
|
+
bbox,
|
|
625
|
+
prepared.page_size,
|
|
626
|
+
angle,
|
|
627
|
+
)
|
|
628
|
+
for bbox in container_bboxes
|
|
629
|
+
]
|
|
630
|
+
for line, bbox in geometry:
|
|
631
|
+
if page_index == 0 and _bbox_center_y(bbox) < 0.64 * local_page_height:
|
|
632
|
+
continue
|
|
633
|
+
related_lanes = [
|
|
634
|
+
lane
|
|
635
|
+
for lane in lanes
|
|
636
|
+
if not lane.is_span
|
|
637
|
+
and len(lane.lines) >= 3
|
|
638
|
+
and lane.left - body_height <= _bbox_center_x(bbox) <= lane.right + body_height
|
|
639
|
+
]
|
|
640
|
+
if not related_lanes:
|
|
641
|
+
continue
|
|
642
|
+
lane = max(
|
|
643
|
+
related_lanes,
|
|
644
|
+
key=lambda item: (
|
|
645
|
+
len(item.lines),
|
|
646
|
+
item.right - item.left,
|
|
647
|
+
),
|
|
648
|
+
)
|
|
649
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
650
|
+
width_ratio = (bbox[2] - bbox[0]) / lane_width
|
|
651
|
+
style_ratio = _line_canonical_style_scale(line, bbox) / body_height
|
|
652
|
+
left_offset = (bbox[0] - lane.left) / body_height
|
|
653
|
+
regular_font = _line_uses_document_regular_font(
|
|
654
|
+
line,
|
|
655
|
+
document_body_profile,
|
|
656
|
+
)
|
|
657
|
+
if (
|
|
658
|
+
not 0.75 <= style_ratio <= 1.35
|
|
659
|
+
or width_ratio > 0.8
|
|
660
|
+
or left_offset > 0.75
|
|
661
|
+
or left_offset < (-0.75 if regular_font else -3.0)
|
|
662
|
+
or _line_inside_visual_container(
|
|
663
|
+
bbox,
|
|
664
|
+
local_containers,
|
|
665
|
+
)
|
|
666
|
+
):
|
|
667
|
+
continue
|
|
668
|
+
followers = [
|
|
669
|
+
(other_line, other_bbox)
|
|
670
|
+
for other_line, other_bbox in geometry
|
|
671
|
+
if other_line is not line
|
|
672
|
+
and bbox[1] < other_bbox[1]
|
|
673
|
+
and other_bbox[1] - bbox[3] <= 3.0 * body_height
|
|
674
|
+
and lane.left - body_height <= other_bbox[0] <= lane.left + 3.5 * body_height
|
|
675
|
+
and other_bbox[2] - other_bbox[0] >= 0.4 * lane_width
|
|
676
|
+
]
|
|
677
|
+
if not followers:
|
|
678
|
+
continue
|
|
679
|
+
gap_above, gap_below = physical_gaps.get(
|
|
680
|
+
line.source_index,
|
|
681
|
+
(None, None),
|
|
682
|
+
)
|
|
683
|
+
layout_ratio = (bbox[3] - bbox[1]) / body_height
|
|
684
|
+
if regular_font and layout_ratio < 1.3 and line.font_coverage < 0.75:
|
|
685
|
+
continue
|
|
686
|
+
standard_transition = (
|
|
687
|
+
gap_above is not None
|
|
688
|
+
and gap_below is not None
|
|
689
|
+
and gap_above >= 0.65 * body_height
|
|
690
|
+
and gap_below >= 0.35 * body_height
|
|
691
|
+
and (not regular_font or gap_below <= 1.5 * body_height)
|
|
692
|
+
)
|
|
693
|
+
low_coverage_wide_transition = (
|
|
694
|
+
gap_above is not None
|
|
695
|
+
and gap_below is not None
|
|
696
|
+
and gap_above >= 0.5 * body_height
|
|
697
|
+
and gap_below >= 0.45 * body_height
|
|
698
|
+
and gap_below <= 0.7 * body_height
|
|
699
|
+
and width_ratio >= 0.75
|
|
700
|
+
and line.font_coverage <= 0.7
|
|
701
|
+
and layout_ratio >= 1.8
|
|
702
|
+
)
|
|
703
|
+
if low_coverage_wide_transition and any(
|
|
704
|
+
anchor_page_index == page_index
|
|
705
|
+
and anchor_angle == angle
|
|
706
|
+
and abs(anchor_left - lane.left) <= body_height
|
|
707
|
+
and 0 < _bbox_center_y(bbox) - anchor_center_y <= 12.0 * body_height
|
|
708
|
+
for (
|
|
709
|
+
anchor_page_index,
|
|
710
|
+
anchor_angle,
|
|
711
|
+
anchor_left,
|
|
712
|
+
anchor_center_y,
|
|
713
|
+
) in accepted_anchor_positions
|
|
714
|
+
):
|
|
715
|
+
low_coverage_wide_transition = False
|
|
716
|
+
compact_regular_transition = (
|
|
717
|
+
gap_above is not None
|
|
718
|
+
and gap_below is not None
|
|
719
|
+
and gap_above >= 0.65 * body_height
|
|
720
|
+
and gap_below >= 0.15 * body_height
|
|
721
|
+
and width_ratio <= 0.45
|
|
722
|
+
and line.font_coverage >= 0.75
|
|
723
|
+
and layout_ratio <= 1.3
|
|
724
|
+
)
|
|
725
|
+
family_key = (
|
|
726
|
+
(
|
|
727
|
+
_normalized_font_family(line.font_signature),
|
|
728
|
+
line.font_signature[1],
|
|
729
|
+
)
|
|
730
|
+
if line.font_signature is not None
|
|
731
|
+
else None
|
|
732
|
+
)
|
|
733
|
+
if (
|
|
734
|
+
standard_transition
|
|
735
|
+
or low_coverage_wide_transition
|
|
736
|
+
or compact_regular_transition
|
|
737
|
+
or (gap_above is None and gap_below is not None and gap_below >= 0.6 * body_height and not regular_font)
|
|
738
|
+
):
|
|
739
|
+
strong_candidates.append(
|
|
740
|
+
(
|
|
741
|
+
page_index,
|
|
742
|
+
line,
|
|
743
|
+
family_key,
|
|
744
|
+
layout_ratio,
|
|
745
|
+
),
|
|
746
|
+
)
|
|
747
|
+
accepted_anchor_positions.append(
|
|
748
|
+
(
|
|
749
|
+
page_index,
|
|
750
|
+
angle,
|
|
751
|
+
lane.left,
|
|
752
|
+
_bbox_center_y(bbox),
|
|
753
|
+
)
|
|
754
|
+
)
|
|
755
|
+
continue
|
|
756
|
+
if (
|
|
757
|
+
gap_above is None
|
|
758
|
+
and bbox[1] <= 0.18 * local_page_height
|
|
759
|
+
and width_ratio <= 0.6
|
|
760
|
+
and (line.font_coverage >= 0.75 or (not regular_font and line.font_coverage >= 0.5))
|
|
761
|
+
):
|
|
762
|
+
start_candidates.append(
|
|
763
|
+
(
|
|
764
|
+
page_index,
|
|
765
|
+
line,
|
|
766
|
+
family_key,
|
|
767
|
+
layout_ratio,
|
|
768
|
+
),
|
|
769
|
+
)
|
|
770
|
+
|
|
771
|
+
for _page_index, line, _family_key, _layout_ratio in strong_candidates:
|
|
772
|
+
line.semantic_type = "paragraph_title"
|
|
773
|
+
line.structural_title = True
|
|
774
|
+
strong_families_by_page = {
|
|
775
|
+
(page_index, family_key) for page_index, _line, family_key, _layout_ratio in strong_candidates if family_key is not None
|
|
776
|
+
}
|
|
777
|
+
for page_index, line, family_key, _layout_ratio in start_candidates:
|
|
778
|
+
if not _line_uses_document_regular_font(
|
|
779
|
+
line,
|
|
780
|
+
document_body_profile,
|
|
781
|
+
) or (family_key is not None and (page_index, family_key) in strong_families_by_page):
|
|
782
|
+
line.semantic_type = "paragraph_title"
|
|
783
|
+
line.structural_title = True
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
def _collect_legacy_paragraph_title_sources(
|
|
787
|
+
prepared_pages: list[_PreparedPage],
|
|
788
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
789
|
+
document_title_profile: _DocumentTitleProfile | None,
|
|
790
|
+
) -> set[tuple[int, int]]:
|
|
791
|
+
"""在行副本上使用 legacy 尺度收集原本成立的段落标题身份。"""
|
|
792
|
+
|
|
793
|
+
if document_body_profile is None:
|
|
794
|
+
return set()
|
|
795
|
+
legacy_profile = replace(
|
|
796
|
+
document_body_profile,
|
|
797
|
+
has_style_scale_repairs=False,
|
|
798
|
+
)
|
|
799
|
+
output: set[tuple[int, int]] = set()
|
|
800
|
+
for page_index, prepared in enumerate(prepared_pages):
|
|
801
|
+
probe_lines = [
|
|
802
|
+
replace(
|
|
803
|
+
line,
|
|
804
|
+
style_scale_repaired=False,
|
|
805
|
+
structural_title=False,
|
|
806
|
+
)
|
|
807
|
+
for line in prepared.remaining_lines
|
|
808
|
+
]
|
|
809
|
+
container_bboxes = [
|
|
810
|
+
block["bbox"] for block in prepared.fixed_blocks if not isinstance(block.get("_inline_visual_row_id"), int)
|
|
811
|
+
]
|
|
812
|
+
caption_container_bboxes = [block["bbox"] for block in prepared.fixed_blocks if block.get("type") in {"image", "code"}]
|
|
813
|
+
_classify_page_titles(
|
|
814
|
+
probe_lines,
|
|
815
|
+
prepared.page_size,
|
|
816
|
+
page_index=page_index,
|
|
817
|
+
container_bboxes=container_bboxes,
|
|
818
|
+
caption_container_bboxes=caption_container_bboxes,
|
|
819
|
+
document_body_profile=legacy_profile,
|
|
820
|
+
document_title_profile=document_title_profile,
|
|
821
|
+
)
|
|
822
|
+
output.update((page_index, line.source_index) for line in probe_lines if line.semantic_type == "paragraph_title")
|
|
823
|
+
return output
|
|
824
|
+
|
|
825
|
+
|
|
826
|
+
def _classify_inline_typography_reset_titles(
|
|
827
|
+
lines: list[_LineItem],
|
|
828
|
+
page_size: tuple[float, float],
|
|
829
|
+
*,
|
|
830
|
+
container_bboxes: list[BBox],
|
|
831
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
832
|
+
) -> None:
|
|
833
|
+
"""用短段尾、字体切换和缩进正文识别行内结构标题。"""
|
|
834
|
+
|
|
835
|
+
if document_body_profile is None:
|
|
836
|
+
return
|
|
837
|
+
for angle in sorted({line.angle for line in lines if line.semantic_type is None and not line.title_suppressed}):
|
|
838
|
+
line_geometry = [
|
|
839
|
+
(line, _rotate_bbox_to_upright(line.bbox, page_size, angle))
|
|
840
|
+
for line in lines
|
|
841
|
+
if line.angle == angle and line.semantic_type is None and not line.title_suppressed
|
|
842
|
+
]
|
|
843
|
+
if len(line_geometry) < 3:
|
|
844
|
+
continue
|
|
845
|
+
median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in line_geometry)
|
|
846
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
847
|
+
lanes = _infer_text_lanes(
|
|
848
|
+
line_geometry,
|
|
849
|
+
local_page_width,
|
|
850
|
+
median_height,
|
|
851
|
+
)
|
|
852
|
+
local_container_bboxes = [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in container_bboxes]
|
|
853
|
+
for lane in lanes:
|
|
854
|
+
rows = sorted(
|
|
855
|
+
lane.lines,
|
|
856
|
+
key=lambda item: (
|
|
857
|
+
item[1][1],
|
|
858
|
+
item[1][0],
|
|
859
|
+
item[0].source_index,
|
|
860
|
+
),
|
|
861
|
+
)
|
|
862
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
863
|
+
for previous, current, following in zip(
|
|
864
|
+
rows,
|
|
865
|
+
rows[1:],
|
|
866
|
+
rows[2:],
|
|
867
|
+
):
|
|
868
|
+
previous_line, previous_bbox = previous
|
|
869
|
+
current_line, current_bbox = current
|
|
870
|
+
following_line, following_bbox = following
|
|
871
|
+
if any(
|
|
872
|
+
line.semantic_type is not None
|
|
873
|
+
for line in (
|
|
874
|
+
previous_line,
|
|
875
|
+
current_line,
|
|
876
|
+
following_line,
|
|
877
|
+
)
|
|
878
|
+
):
|
|
879
|
+
continue
|
|
880
|
+
if (
|
|
881
|
+
previous_line.font_signature is None
|
|
882
|
+
or current_line.font_signature is None
|
|
883
|
+
or following_line.font_signature is None
|
|
884
|
+
or previous_line.font_coverage < 0.65
|
|
885
|
+
or current_line.font_coverage < 0.65
|
|
886
|
+
or following_line.font_coverage < 0.65
|
|
887
|
+
):
|
|
888
|
+
continue
|
|
889
|
+
if not _font_signatures_share_family(
|
|
890
|
+
previous_line.font_signature,
|
|
891
|
+
following_line.font_signature,
|
|
892
|
+
) or _font_signatures_share_family(
|
|
893
|
+
current_line.font_signature,
|
|
894
|
+
previous_line.font_signature,
|
|
895
|
+
):
|
|
896
|
+
continue
|
|
897
|
+
previous_height = _line_effective_height(*previous)
|
|
898
|
+
current_height = _line_effective_height(*current)
|
|
899
|
+
following_height = _line_effective_height(*following)
|
|
900
|
+
neighbor_height = statistics.median(
|
|
901
|
+
(previous_height, following_height),
|
|
902
|
+
)
|
|
903
|
+
pair_height = max(
|
|
904
|
+
previous_height,
|
|
905
|
+
current_height,
|
|
906
|
+
following_height,
|
|
907
|
+
)
|
|
908
|
+
previous_width = previous_bbox[2] - previous_bbox[0]
|
|
909
|
+
current_width = current_bbox[2] - current_bbox[0]
|
|
910
|
+
following_width = following_bbox[2] - following_bbox[0]
|
|
911
|
+
following_indent = following_bbox[0] - current_bbox[0]
|
|
912
|
+
if not (
|
|
913
|
+
previous_width <= 0.45 * lane_width
|
|
914
|
+
and 0.2 * lane_width <= current_width <= 0.65 * lane_width
|
|
915
|
+
and following_width >= 0.75 * lane_width
|
|
916
|
+
and abs(current_bbox[0] - lane.left) <= 0.75 * pair_height
|
|
917
|
+
and 0.75 * pair_height <= following_indent <= 3.0 * pair_height
|
|
918
|
+
and 0.85 <= current_height / max(0.1, neighbor_height) <= 1.15
|
|
919
|
+
and -0.25 * pair_height <= _effective_text_row_gap(previous, current) <= 0.5 * pair_height
|
|
920
|
+
and -0.25 * pair_height <= _effective_text_row_gap(current, following) <= 0.5 * pair_height
|
|
921
|
+
and not _line_inside_visual_container(
|
|
922
|
+
current_bbox,
|
|
923
|
+
local_container_bboxes,
|
|
924
|
+
)
|
|
925
|
+
):
|
|
926
|
+
continue
|
|
927
|
+
current_line.semantic_type = "paragraph_title"
|
|
928
|
+
current_line.structural_title = True
|
|
929
|
+
|
|
930
|
+
|
|
931
|
+
def _classify_body_height_section_titles(
|
|
932
|
+
lines: list[_LineItem],
|
|
933
|
+
page_size: tuple[float, float],
|
|
934
|
+
*,
|
|
935
|
+
container_bboxes: list[BBox],
|
|
936
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
937
|
+
page_index: int = 1,
|
|
938
|
+
) -> None:
|
|
939
|
+
"""用重复的短行加正文组结构识别与正文同字号的独立章节标题。"""
|
|
940
|
+
|
|
941
|
+
if document_body_profile is None:
|
|
942
|
+
return
|
|
943
|
+
body_height = document_body_profile.body_height
|
|
944
|
+
if body_height <= 0:
|
|
945
|
+
return
|
|
946
|
+
|
|
947
|
+
for angle in sorted({line.angle for line in lines if line.semantic_type is None and not line.title_suppressed}):
|
|
948
|
+
line_geometry = sorted(
|
|
949
|
+
[
|
|
950
|
+
(line, _rotate_bbox_to_upright(line.bbox, page_size, angle))
|
|
951
|
+
for line in lines
|
|
952
|
+
if line.angle == angle and line.semantic_type is None and not line.title_suppressed
|
|
953
|
+
],
|
|
954
|
+
key=lambda item: (item[1][1], item[1][0], item[0].source_index),
|
|
955
|
+
)
|
|
956
|
+
if len(line_geometry) < 8:
|
|
957
|
+
continue
|
|
958
|
+
median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in line_geometry)
|
|
959
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
960
|
+
local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
|
|
961
|
+
lanes = _infer_text_lanes(
|
|
962
|
+
line_geometry,
|
|
963
|
+
local_page_width,
|
|
964
|
+
median_height,
|
|
965
|
+
)
|
|
966
|
+
lane_by_source: dict[int, _TextLane] = {}
|
|
967
|
+
regular_gaps: list[float] = []
|
|
968
|
+
for lane in lanes:
|
|
969
|
+
lane.lines.sort(key=lambda item: (item[1][1], item[1][0], item[0].source_index))
|
|
970
|
+
regular_gap, _gap_mad = _estimate_lane_gap(lane)
|
|
971
|
+
regular_gaps.append(regular_gap)
|
|
972
|
+
for line, _bbox in lane.lines:
|
|
973
|
+
lane_by_source[line.source_index] = lane
|
|
974
|
+
if not lane_by_source:
|
|
975
|
+
continue
|
|
976
|
+
|
|
977
|
+
page_regular_gap = statistics.median(regular_gaps) if regular_gaps else 0.2 * body_height
|
|
978
|
+
physical_gaps = _build_physical_title_gap_map(line_geometry)
|
|
979
|
+
local_container_bboxes = [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in container_bboxes]
|
|
980
|
+
candidates: list[tuple[_LineItem, BBox, _TextLane]] = []
|
|
981
|
+
for line, bbox in line_geometry:
|
|
982
|
+
lane = lane_by_source.get(line.source_index)
|
|
983
|
+
if lane is None:
|
|
984
|
+
continue
|
|
985
|
+
line_height = _line_effective_height(line, bbox)
|
|
986
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
987
|
+
if not 0.9 <= line_height / body_height <= 1.1:
|
|
988
|
+
continue
|
|
989
|
+
if bbox[2] - bbox[0] > 0.22 * lane_width:
|
|
990
|
+
continue
|
|
991
|
+
if abs(bbox[0] - lane.left) > 0.75 * body_height:
|
|
992
|
+
continue
|
|
993
|
+
if _line_inside_visual_container(bbox, local_container_bboxes):
|
|
994
|
+
continue
|
|
995
|
+
|
|
996
|
+
followers = _body_height_section_followers(
|
|
997
|
+
line,
|
|
998
|
+
bbox,
|
|
999
|
+
line_geometry,
|
|
1000
|
+
lane_by_source,
|
|
1001
|
+
body_height,
|
|
1002
|
+
)
|
|
1003
|
+
if len(followers) < 3:
|
|
1004
|
+
continue
|
|
1005
|
+
gap_above = physical_gaps.get(line.source_index, (None, None))[0]
|
|
1006
|
+
if gap_above is not None and gap_above <= 0.25 * body_height:
|
|
1007
|
+
continue
|
|
1008
|
+
starts_body = gap_above is None and bbox[1] <= 0.2 * local_page_height
|
|
1009
|
+
has_extra_gap = gap_above is not None and gap_above - page_regular_gap >= 0.75 * body_height
|
|
1010
|
+
has_full_width_follower = any(
|
|
1011
|
+
follower_bbox[2] - follower_bbox[0]
|
|
1012
|
+
>= 0.75
|
|
1013
|
+
* max(
|
|
1014
|
+
0.1,
|
|
1015
|
+
lane_by_source[follower_line.source_index].right - lane_by_source[follower_line.source_index].left,
|
|
1016
|
+
)
|
|
1017
|
+
for follower_line, follower_bbox in followers
|
|
1018
|
+
if follower_line.source_index in lane_by_source
|
|
1019
|
+
)
|
|
1020
|
+
if starts_body or has_extra_gap or has_full_width_follower:
|
|
1021
|
+
candidates.append((line, bbox, lane))
|
|
1022
|
+
|
|
1023
|
+
for line, bbox, _lane in candidates:
|
|
1024
|
+
compatible_count = sum(
|
|
1025
|
+
1
|
|
1026
|
+
for peer_line, peer_bbox, _peer_lane in candidates
|
|
1027
|
+
if abs(peer_bbox[0] - bbox[0]) <= body_height
|
|
1028
|
+
and 0.9 <= _line_effective_height(peer_line, peer_bbox) / _line_effective_height(line, bbox) <= 1.1
|
|
1029
|
+
and _title_fonts_compatible(line, peer_line)
|
|
1030
|
+
)
|
|
1031
|
+
if compatible_count >= 2:
|
|
1032
|
+
# 只标记结构锚点本身,避免普通正文被标题邻行扩展再次吞入。
|
|
1033
|
+
line.semantic_type = "paragraph_title"
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
def _body_height_section_followers(
|
|
1037
|
+
candidate_line: _LineItem,
|
|
1038
|
+
candidate_bbox: BBox,
|
|
1039
|
+
line_geometry: list[tuple[_LineItem, BBox]],
|
|
1040
|
+
lane_by_source: dict[int, _TextLane],
|
|
1041
|
+
body_height: float,
|
|
1042
|
+
) -> list[tuple[_LineItem, BBox]]:
|
|
1043
|
+
"""返回短标题后方同锚点、同正文尺度且行距稳定的前三行。"""
|
|
1044
|
+
|
|
1045
|
+
followers: list[tuple[_LineItem, BBox]] = []
|
|
1046
|
+
previous_top = candidate_bbox[1]
|
|
1047
|
+
for line, bbox in line_geometry:
|
|
1048
|
+
if line is candidate_line or bbox[1] <= candidate_bbox[1] + 0.4 * body_height:
|
|
1049
|
+
continue
|
|
1050
|
+
if not (candidate_bbox[0] - 0.75 * body_height <= bbox[0] <= candidate_bbox[0] + 1.5 * body_height):
|
|
1051
|
+
continue
|
|
1052
|
+
top_pitch = bbox[1] - previous_top
|
|
1053
|
+
if top_pitch < 0.5 * body_height:
|
|
1054
|
+
continue
|
|
1055
|
+
if top_pitch > 1.8 * body_height:
|
|
1056
|
+
break
|
|
1057
|
+
if not 0.9 <= _line_effective_height(line, bbox) / body_height <= 1.1:
|
|
1058
|
+
break
|
|
1059
|
+
if line.source_index not in lane_by_source:
|
|
1060
|
+
break
|
|
1061
|
+
followers.append((line, bbox))
|
|
1062
|
+
previous_top = bbox[1]
|
|
1063
|
+
if len(followers) == 3:
|
|
1064
|
+
break
|
|
1065
|
+
return followers
|
|
1066
|
+
|
|
1067
|
+
|
|
1068
|
+
__all__ = [
|
|
1069
|
+
"_normalized_section_title_text",
|
|
1070
|
+
"_is_plausible_section_number",
|
|
1071
|
+
"_section_title_has_body_followers",
|
|
1072
|
+
"_classify_explicit_section_titles",
|
|
1073
|
+
"_classify_document_structural_titles",
|
|
1074
|
+
"_promote_noninitial_document_title_band",
|
|
1075
|
+
"_canonical_title_style_key",
|
|
1076
|
+
"_classify_document_structural_title_candidates",
|
|
1077
|
+
"_collect_legacy_paragraph_title_sources",
|
|
1078
|
+
"_classify_inline_typography_reset_titles",
|
|
1079
|
+
"_classify_body_height_section_titles",
|
|
1080
|
+
"_body_height_section_followers",
|
|
1081
|
+
]
|