docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""保留既有入口的 Flash PDF 门面,内部实现按职责显式组织。"""
|
|
2
|
+
|
|
3
|
+
from .text_assembly.annotations import (
|
|
4
|
+
_merge_image_caption_text_blocks as _merge_image_caption_text_blocks,
|
|
5
|
+
_caption_image_group_bboxes as _caption_image_group_bboxes,
|
|
6
|
+
_caption_seed_matches_image as _caption_seed_matches_image,
|
|
7
|
+
_caption_body_has_structural_gap as _caption_body_has_structural_gap,
|
|
8
|
+
_caption_tail_matches_seed as _caption_tail_matches_seed,
|
|
9
|
+
_merge_multiline_title_blocks as _merge_multiline_title_blocks,
|
|
10
|
+
_merge_fragmented_header_blocks as _merge_fragmented_header_blocks,
|
|
11
|
+
_merge_front_matter_column_blocks as _merge_front_matter_column_blocks,
|
|
12
|
+
_merge_repeated_compact_title_continuations as _merge_repeated_compact_title_continuations,
|
|
13
|
+
)
|
|
14
|
+
from .text_assembly.assembly import _build_text_blocks as _build_text_blocks
|
|
15
|
+
from .text_assembly.common import (
|
|
16
|
+
_REFERENCE_ENTRY_RE as _REFERENCE_ENTRY_RE,
|
|
17
|
+
_FIGURE_CAPTION_MARKER_RE as _FIGURE_CAPTION_MARKER_RE,
|
|
18
|
+
_INLINE_MATH_RECOVERY_MARKER as _INLINE_MATH_RECOVERY_MARKER,
|
|
19
|
+
_PARAGRAPH_FORMULA_CONTEXT_MARKER as _PARAGRAPH_FORMULA_CONTEXT_MARKER,
|
|
20
|
+
_FRONT_MATTER_FIELD_RE as _FRONT_MATTER_FIELD_RE,
|
|
21
|
+
_LIST_ITEM_RE as _LIST_ITEM_RE,
|
|
22
|
+
_BULLET_ITEM_RE as _BULLET_ITEM_RE,
|
|
23
|
+
_EMAIL_METADATA_RE as _EMAIL_METADATA_RE,
|
|
24
|
+
_ABSTRACT_METADATA_RE as _ABSTRACT_METADATA_RE,
|
|
25
|
+
_LABELLED_METADATA_RE as _LABELLED_METADATA_RE,
|
|
26
|
+
_URL_LINE_RE as _URL_LINE_RE,
|
|
27
|
+
_SHORT_SAME_BASELINE_PREFIX_RE as _SHORT_SAME_BASELINE_PREFIX_RE,
|
|
28
|
+
_merge_internal_text_block_group as _merge_internal_text_block_group,
|
|
29
|
+
_component_declared_lane_interval as _component_declared_lane_interval,
|
|
30
|
+
_component_lane_interval as _component_lane_interval,
|
|
31
|
+
_component_reference_width as _component_reference_width,
|
|
32
|
+
_compatible_component_lane_width as _compatible_component_lane_width,
|
|
33
|
+
_components_share_lane_role as _components_share_lane_role,
|
|
34
|
+
_block_starts_with_short_wide_rows as _block_starts_with_short_wide_rows,
|
|
35
|
+
_find_short_opener_pairs as _find_short_opener_pairs,
|
|
36
|
+
_nearest_following_text_component as _nearest_following_text_component,
|
|
37
|
+
_has_parallel_text_component as _has_parallel_text_component,
|
|
38
|
+
_nearest_tapered_tail_component as _nearest_tapered_tail_component,
|
|
39
|
+
_component_connection_skips_block as _component_connection_skips_block,
|
|
40
|
+
_text_component_sort_key as _text_component_sort_key,
|
|
41
|
+
_merge_text_line_content as _merge_text_line_content,
|
|
42
|
+
)
|
|
43
|
+
from .text_assembly.footnotes import (
|
|
44
|
+
_build_grouped_page_footnote_blocks as _build_grouped_page_footnote_blocks,
|
|
45
|
+
_split_page_footnote_entries as _split_page_footnote_entries,
|
|
46
|
+
_find_page_footnote_marker_rows as _find_page_footnote_marker_rows,
|
|
47
|
+
_find_geometric_page_footnote_marker_rows as _find_geometric_page_footnote_marker_rows,
|
|
48
|
+
_split_marked_page_footnote_entries as _split_marked_page_footnote_entries,
|
|
49
|
+
_split_unmarked_page_footnote_entries as _split_unmarked_page_footnote_entries,
|
|
50
|
+
_tight_page_footnote_bboxes as _tight_page_footnote_bboxes,
|
|
51
|
+
)
|
|
52
|
+
from .text_assembly.merging import (
|
|
53
|
+
_merge_short_same_baseline_prefix_blocks as _merge_short_same_baseline_prefix_blocks,
|
|
54
|
+
_blocks_share_boundary_visual_row as _blocks_share_boundary_visual_row,
|
|
55
|
+
_merge_overlapping_same_line_text_blocks as _merge_overlapping_same_line_text_blocks,
|
|
56
|
+
_merge_inline_math_fragment_text_blocks as _merge_inline_math_fragment_text_blocks,
|
|
57
|
+
_component_local_union_bbox as _component_local_union_bbox,
|
|
58
|
+
_merge_paragraph_formula_context_blocks as _merge_paragraph_formula_context_blocks,
|
|
59
|
+
_merge_residual_narrow_math_text_blocks as _merge_residual_narrow_math_text_blocks,
|
|
60
|
+
_merge_hostless_inline_math_fragment_blocks as _merge_hostless_inline_math_fragment_blocks,
|
|
61
|
+
_merge_inline_math_recovery_group as _merge_inline_math_recovery_group,
|
|
62
|
+
_merge_inline_math_paragraph_continuations as _merge_inline_math_paragraph_continuations,
|
|
63
|
+
_merge_spatial_text_components as _merge_spatial_text_components,
|
|
64
|
+
_merge_list_intro_text_components as _merge_list_intro_text_components,
|
|
65
|
+
_merge_unterminated_text_components as _merge_unterminated_text_components,
|
|
66
|
+
)
|
|
67
|
+
from .text_assembly.rows import (
|
|
68
|
+
_local_tight_output_line_bboxes as _local_tight_output_line_bboxes,
|
|
69
|
+
_starts_structural_reference_entry as _starts_structural_reference_entry,
|
|
70
|
+
_build_hanging_indent_group_map as _build_hanging_indent_group_map,
|
|
71
|
+
_infer_local_text_lane_map as _infer_local_text_lane_map,
|
|
72
|
+
_structured_text_break_sources as _structured_text_break_sources,
|
|
73
|
+
_isolated_indented_paragraph_break_sources as _isolated_indented_paragraph_break_sources,
|
|
74
|
+
_centered_visual_reset_break_sources as _centered_visual_reset_break_sources,
|
|
75
|
+
_leading_typography_reset_break_sources as _leading_typography_reset_break_sources,
|
|
76
|
+
_formula_style_text_row_break_sources as _formula_style_text_row_break_sources,
|
|
77
|
+
_front_matter_keyword_break_sources as _front_matter_keyword_break_sources,
|
|
78
|
+
_component_starts_with_emphasized_row as _component_starts_with_emphasized_row,
|
|
79
|
+
_explicit_text_break_sources as _explicit_text_break_sources,
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
__all__ = []
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""保留既有入口的 Flash PDF 门面,内部实现按职责显式组织。"""
|
|
2
|
+
|
|
3
|
+
from .inline.detection import (
|
|
4
|
+
detect_pdf_text_link_lines as detect_pdf_text_link_lines,
|
|
5
|
+
detect_pdf_text_style_lines as detect_pdf_text_style_lines,
|
|
6
|
+
)
|
|
7
|
+
from .inline.matching import (
|
|
8
|
+
_assign_lines_to_blocks as _assign_lines_to_blocks,
|
|
9
|
+
_realign_repaired_text_evidence as _realign_repaired_text_evidence,
|
|
10
|
+
_partition_resplit_text_evidence as _partition_resplit_text_evidence,
|
|
11
|
+
)
|
|
12
|
+
from .inline.materialize import (
|
|
13
|
+
apply_pdf_text_links as apply_pdf_text_links,
|
|
14
|
+
apply_pdf_text_scripts as apply_pdf_text_scripts,
|
|
15
|
+
apply_pdf_text_styles as apply_pdf_text_styles,
|
|
16
|
+
materialize_pdf_inline_spans as materialize_pdf_inline_spans,
|
|
17
|
+
)
|
|
18
|
+
from .inline.scripts import (
|
|
19
|
+
detect_pdf_text_script_lines as detect_pdf_text_script_lines,
|
|
20
|
+
_fraction_member_indices as _fraction_member_indices,
|
|
21
|
+
_refine_math_script_tokens as _refine_math_script_tokens,
|
|
22
|
+
_script_line_char_roles as _script_line_char_roles,
|
|
23
|
+
)
|
|
24
|
+
from .inline.types import (
|
|
25
|
+
PDF_NATIVE_SCRIPT_MARKUP_KEY as PDF_NATIVE_SCRIPT_MARKUP_KEY,
|
|
26
|
+
PDF_FONT_FORCE_BOLD_FLAG as PDF_FONT_FORCE_BOLD_FLAG,
|
|
27
|
+
PDF_FONT_ITALIC_FLAG as PDF_FONT_ITALIC_FLAG,
|
|
28
|
+
PDFTextLinkLine as PDFTextLinkLine,
|
|
29
|
+
PDFTextLinkRange as PDFTextLinkRange,
|
|
30
|
+
PDFTextScriptLine as PDFTextScriptLine,
|
|
31
|
+
PDFTextScriptRange as PDFTextScriptRange,
|
|
32
|
+
PDFTextStyle as PDFTextStyle,
|
|
33
|
+
PDFTextStyleLine as PDFTextStyleLine,
|
|
34
|
+
PDFTextStyleRange as PDFTextStyleRange,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
__all__ = [
|
|
38
|
+
"PDF_NATIVE_SCRIPT_MARKUP_KEY",
|
|
39
|
+
"PDF_FONT_FORCE_BOLD_FLAG",
|
|
40
|
+
"PDF_FONT_ITALIC_FLAG",
|
|
41
|
+
"PDFTextLinkLine",
|
|
42
|
+
"PDFTextLinkRange",
|
|
43
|
+
"PDFTextScriptLine",
|
|
44
|
+
"PDFTextScriptRange",
|
|
45
|
+
"PDFTextStyle",
|
|
46
|
+
"PDFTextStyleLine",
|
|
47
|
+
"PDFTextStyleRange",
|
|
48
|
+
"apply_pdf_text_links",
|
|
49
|
+
"apply_pdf_text_scripts",
|
|
50
|
+
"apply_pdf_text_styles",
|
|
51
|
+
"detect_pdf_text_link_lines",
|
|
52
|
+
"detect_pdf_text_script_lines",
|
|
53
|
+
"detect_pdf_text_style_lines",
|
|
54
|
+
"materialize_pdf_inline_spans",
|
|
55
|
+
]
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""统计全文和栏内正文排版基线。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import statistics
|
|
6
|
+
|
|
7
|
+
from ..geometry import _rotate_bbox_to_upright
|
|
8
|
+
from ..inline.types import PDF_FONT_FORCE_BOLD_FLAG, PDF_FONT_ITALIC_FLAG
|
|
9
|
+
from ..line_layout import _estimate_lane_gap, _font_signatures_share_family, _line_canonical_style_scale, _line_effective_height
|
|
10
|
+
from ..models import _DocumentBodyProfile, _LaneBodyProfile, _LineItem, _PreparedPage, _TextLane
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _infer_document_body_profile(
|
|
14
|
+
prepared_pages: list[_PreparedPage],
|
|
15
|
+
*,
|
|
16
|
+
use_canonical_scale: bool = False,
|
|
17
|
+
) -> _DocumentBodyProfile | None:
|
|
18
|
+
"""按跨页覆盖和累计行宽推断全文正文行高及常规字体集合。"""
|
|
19
|
+
|
|
20
|
+
samples: list[
|
|
21
|
+
tuple[
|
|
22
|
+
float,
|
|
23
|
+
int,
|
|
24
|
+
float,
|
|
25
|
+
tuple[str, int] | None,
|
|
26
|
+
float | None,
|
|
27
|
+
]
|
|
28
|
+
] = []
|
|
29
|
+
for page_index, prepared in enumerate(prepared_pages):
|
|
30
|
+
for line in prepared.remaining_lines:
|
|
31
|
+
if line.semantic_type is not None:
|
|
32
|
+
continue
|
|
33
|
+
local_bbox = _rotate_bbox_to_upright(
|
|
34
|
+
line.bbox,
|
|
35
|
+
prepared.page_size,
|
|
36
|
+
line.angle,
|
|
37
|
+
)
|
|
38
|
+
local_page_width = prepared.page_size[1] if line.angle in {90, 270} else prepared.page_size[0]
|
|
39
|
+
height = (
|
|
40
|
+
_line_canonical_style_scale(line, local_bbox)
|
|
41
|
+
if use_canonical_scale
|
|
42
|
+
else _line_effective_height(line, local_bbox)
|
|
43
|
+
)
|
|
44
|
+
normalized_width = max(0.0, local_bbox[2] - local_bbox[0]) / max(
|
|
45
|
+
0.1,
|
|
46
|
+
local_page_width,
|
|
47
|
+
)
|
|
48
|
+
if height <= 0 or normalized_width <= 0:
|
|
49
|
+
continue
|
|
50
|
+
samples.append(
|
|
51
|
+
(
|
|
52
|
+
height,
|
|
53
|
+
page_index,
|
|
54
|
+
normalized_width,
|
|
55
|
+
line.font_signature if line.font_coverage >= 0.75 else None,
|
|
56
|
+
line.dominant_font_weight,
|
|
57
|
+
)
|
|
58
|
+
)
|
|
59
|
+
if not samples:
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
height_clusters: list[list[tuple[float, int, float]]] = []
|
|
63
|
+
for height, page_index, normalized_width, _font, _weight in sorted(
|
|
64
|
+
samples,
|
|
65
|
+
key=lambda item: item[0],
|
|
66
|
+
):
|
|
67
|
+
target = next(
|
|
68
|
+
(
|
|
69
|
+
cluster
|
|
70
|
+
for cluster in height_clusters
|
|
71
|
+
if abs(height - statistics.median(item[0] for item in cluster))
|
|
72
|
+
<= 0.1 * statistics.median(item[0] for item in cluster)
|
|
73
|
+
),
|
|
74
|
+
None,
|
|
75
|
+
)
|
|
76
|
+
if target is None:
|
|
77
|
+
height_clusters.append([(height, page_index, normalized_width)])
|
|
78
|
+
else:
|
|
79
|
+
target.append((height, page_index, normalized_width))
|
|
80
|
+
cross_page_clusters = [cluster for cluster in height_clusters if len({item[1] for item in cluster}) >= 2]
|
|
81
|
+
eligible_clusters = cross_page_clusters or height_clusters
|
|
82
|
+
body_cluster = max(
|
|
83
|
+
eligible_clusters,
|
|
84
|
+
key=lambda cluster: (
|
|
85
|
+
sum(item[2] for item in cluster),
|
|
86
|
+
len({item[1] for item in cluster}),
|
|
87
|
+
len(cluster),
|
|
88
|
+
),
|
|
89
|
+
)
|
|
90
|
+
body_height = statistics.median(item[0] for item in body_cluster)
|
|
91
|
+
|
|
92
|
+
body_weights = [
|
|
93
|
+
weight
|
|
94
|
+
for height, _page_index, _width, _font, weight in samples
|
|
95
|
+
if weight is not None and 0.9 <= height / body_height <= 1.1
|
|
96
|
+
]
|
|
97
|
+
body_weight = statistics.median(body_weights) if body_weights else None
|
|
98
|
+
|
|
99
|
+
font_pages: dict[tuple[str, int], set[int]] = {}
|
|
100
|
+
font_widths: dict[tuple[str, int], float] = {}
|
|
101
|
+
font_weights: dict[tuple[str, int], list[float]] = {}
|
|
102
|
+
for height, page_index, width, font, weight in samples:
|
|
103
|
+
if font is None or not 0.9 <= height / body_height <= 1.1:
|
|
104
|
+
continue
|
|
105
|
+
# 常规字体支持必须来自正文高度带;跨页重复的大标题不能反向污染正文画像。
|
|
106
|
+
font_pages.setdefault(font, set()).add(page_index)
|
|
107
|
+
font_widths[font] = font_widths.get(font, 0.0) + width
|
|
108
|
+
if weight is not None:
|
|
109
|
+
font_weights.setdefault(font, []).append(weight)
|
|
110
|
+
|
|
111
|
+
regular_fonts = frozenset(
|
|
112
|
+
font
|
|
113
|
+
for font, pages in font_pages.items()
|
|
114
|
+
if len(pages) >= 3
|
|
115
|
+
and font_widths.get(font, 0.0) >= 2.0
|
|
116
|
+
and font_widths.get(font, 0.0) >= 0.75 * len(pages)
|
|
117
|
+
and _document_font_is_regular(
|
|
118
|
+
font,
|
|
119
|
+
font_weights.get(font, []),
|
|
120
|
+
body_weight,
|
|
121
|
+
)
|
|
122
|
+
)
|
|
123
|
+
return _DocumentBodyProfile(
|
|
124
|
+
body_height=max(0.1, body_height),
|
|
125
|
+
body_weight=body_weight,
|
|
126
|
+
regular_fonts=regular_fonts,
|
|
127
|
+
has_style_scale_repairs=any(
|
|
128
|
+
line.style_scale_repaired for prepared in prepared_pages for line in prepared.remaining_lines
|
|
129
|
+
),
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _document_font_is_regular(
|
|
134
|
+
font: tuple[str, int],
|
|
135
|
+
weights: list[float],
|
|
136
|
+
body_weight: float | None,
|
|
137
|
+
) -> bool:
|
|
138
|
+
"""用字体样式位和全文正文基准过滤斜体、粗体等强调字体。"""
|
|
139
|
+
|
|
140
|
+
if font[1] & (PDF_FONT_ITALIC_FLAG | PDF_FONT_FORCE_BOLD_FLAG):
|
|
141
|
+
return False
|
|
142
|
+
if not weights or body_weight is None:
|
|
143
|
+
return True
|
|
144
|
+
median_weight = statistics.median(weights)
|
|
145
|
+
return median_weight < max(body_weight + 100.0, 1.15 * body_weight)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _infer_lane_body_profile(lane: _TextLane) -> _LaneBodyProfile:
|
|
149
|
+
"""从栏带的长行主体估计正文行高、主字体、字重、常规行距和样式占比。"""
|
|
150
|
+
|
|
151
|
+
available = [item for item in lane.lines if item[0].semantic_type is None]
|
|
152
|
+
if not available:
|
|
153
|
+
return _LaneBodyProfile(1.0, None, None, 0.35, {})
|
|
154
|
+
lane_width = max(0.1, lane.right - lane.left)
|
|
155
|
+
long_lines = [item for item in available if item[1][2] - item[1][0] >= 0.45 * lane_width]
|
|
156
|
+
body_rows = long_lines or available
|
|
157
|
+
body_line_ids = {id(line) for line, _bbox in body_rows}
|
|
158
|
+
body_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in body_rows)
|
|
159
|
+
|
|
160
|
+
font_support: dict[tuple[str, int], float] = {}
|
|
161
|
+
style_support: dict[tuple[str, int], float] = {}
|
|
162
|
+
total_style_width = 0.0
|
|
163
|
+
for line, bbox in available:
|
|
164
|
+
if line.font_signature is None or line.font_coverage < 0.75:
|
|
165
|
+
continue
|
|
166
|
+
line_width = max(0.1, bbox[2] - bbox[0])
|
|
167
|
+
style_support[line.font_signature] = style_support.get(line.font_signature, 0.0) + line_width
|
|
168
|
+
total_style_width += line_width
|
|
169
|
+
if id(line) in body_line_ids and 0.75 <= _line_effective_height(line, bbox) / body_height <= 1.35:
|
|
170
|
+
font_support[line.font_signature] = font_support.get(line.font_signature, 0.0) + line_width
|
|
171
|
+
body_font = max(font_support, key=font_support.get) if font_support else None
|
|
172
|
+
if total_style_width > 0:
|
|
173
|
+
style_support = {signature: width / total_style_width for signature, width in style_support.items()}
|
|
174
|
+
|
|
175
|
+
body_weights = [
|
|
176
|
+
line.dominant_font_weight
|
|
177
|
+
for line, bbox in body_rows
|
|
178
|
+
if line.dominant_font_weight is not None
|
|
179
|
+
and (body_font is None or line.font_signature == body_font)
|
|
180
|
+
and 0.75 <= _line_effective_height(line, bbox) / body_height <= 1.35
|
|
181
|
+
]
|
|
182
|
+
regular_gap, _gap_mad = _estimate_lane_gap(lane)
|
|
183
|
+
return _LaneBodyProfile(
|
|
184
|
+
body_height=max(0.1, body_height),
|
|
185
|
+
body_font=body_font,
|
|
186
|
+
body_weight=statistics.median(body_weights) if body_weights else None,
|
|
187
|
+
regular_gap=regular_gap,
|
|
188
|
+
style_support=style_support,
|
|
189
|
+
body_row_count=len(body_rows),
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _line_uses_document_regular_font(
|
|
194
|
+
line: _LineItem,
|
|
195
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
196
|
+
) -> bool:
|
|
197
|
+
"""判断当前行是否使用跨页反复出现且未加粗的常规字体。"""
|
|
198
|
+
|
|
199
|
+
return (
|
|
200
|
+
document_body_profile is not None
|
|
201
|
+
and line.font_signature is not None
|
|
202
|
+
and line.font_coverage >= 0.5
|
|
203
|
+
and any(
|
|
204
|
+
_font_signatures_share_family(line.font_signature, regular_font)
|
|
205
|
+
for regular_font in document_body_profile.regular_fonts
|
|
206
|
+
)
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
__all__ = [
|
|
211
|
+
"_infer_document_body_profile",
|
|
212
|
+
"_document_font_is_regular",
|
|
213
|
+
"_infer_lane_body_profile",
|
|
214
|
+
"_line_uses_document_regular_font",
|
|
215
|
+
]
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""提供标题分类共享的几何与文本规则。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from .....schema import BBox
|
|
8
|
+
from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_x, _bbox_center_y
|
|
9
|
+
from ..line_layout import _effective_text_row_gap
|
|
10
|
+
from ..models import _LineItem
|
|
11
|
+
|
|
12
|
+
_NUMBERED_SECTION_TITLE_RE = re.compile(
|
|
13
|
+
r"^(?P<number>\d+(?:\s*\.\s*\d+)*)\s+(?P<label>\S.*)$",
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
_SECTION_NUMBER_ONLY_RE = re.compile(
|
|
18
|
+
r"^\d+(?:\s*\.\s*\d+)*\.?$",
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
_SECTION_TITLE_TERMINAL_RE = re.compile(
|
|
23
|
+
r"[.!?。!?::;;,,]$",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
_UNNUMBERED_SECTION_HEADING_RE = re.compile(
|
|
28
|
+
r"^(?:introduction|references?|bibliography|acknowledg(?:e)?ments?|引言|绪论|参考文献|参考资料)$",
|
|
29
|
+
re.IGNORECASE,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _build_physical_title_gap_map(
|
|
34
|
+
line_geometry: list[tuple[_LineItem, BBox]],
|
|
35
|
+
) -> dict[int, tuple[float | None, float | None]]:
|
|
36
|
+
"""为每行记录同方向且水平投影相交的最近上、下物理行净空。"""
|
|
37
|
+
|
|
38
|
+
output: dict[int, tuple[float | None, float | None]] = {}
|
|
39
|
+
for line, bbox in line_geometry:
|
|
40
|
+
line_center = _bbox_center_y(bbox)
|
|
41
|
+
above_gaps: list[float] = []
|
|
42
|
+
below_gaps: list[float] = []
|
|
43
|
+
for other_line, other_bbox in line_geometry:
|
|
44
|
+
if other_line is line:
|
|
45
|
+
continue
|
|
46
|
+
if _bbox_axis_overlap_ratio(bbox, other_bbox, axis="x") < 0.1:
|
|
47
|
+
# 双栏同高度正文并非当前行的物理上下文,不能抹掉真实标题留白。
|
|
48
|
+
continue
|
|
49
|
+
if line.visual_row_id is not None and other_line.visual_row_id == line.visual_row_id:
|
|
50
|
+
continue
|
|
51
|
+
other_center = _bbox_center_y(other_bbox)
|
|
52
|
+
if other_center < line_center:
|
|
53
|
+
above_gaps.append(
|
|
54
|
+
max(
|
|
55
|
+
0.0,
|
|
56
|
+
_effective_text_row_gap(
|
|
57
|
+
(other_line, other_bbox),
|
|
58
|
+
(line, bbox),
|
|
59
|
+
),
|
|
60
|
+
)
|
|
61
|
+
)
|
|
62
|
+
elif other_center > line_center:
|
|
63
|
+
below_gaps.append(
|
|
64
|
+
max(
|
|
65
|
+
0.0,
|
|
66
|
+
_effective_text_row_gap(
|
|
67
|
+
(line, bbox),
|
|
68
|
+
(other_line, other_bbox),
|
|
69
|
+
),
|
|
70
|
+
)
|
|
71
|
+
)
|
|
72
|
+
output[line.source_index] = (
|
|
73
|
+
min(above_gaps) if above_gaps else None,
|
|
74
|
+
min(below_gaps) if below_gaps else None,
|
|
75
|
+
)
|
|
76
|
+
return output
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _line_inside_visual_container(
|
|
80
|
+
line_bbox: BBox,
|
|
81
|
+
container_bboxes: list[BBox],
|
|
82
|
+
) -> bool:
|
|
83
|
+
"""检查文本行中心是否落入视觉容器,容器内标签不得借标题原型晋升。"""
|
|
84
|
+
|
|
85
|
+
center_x = _bbox_center_x(line_bbox)
|
|
86
|
+
center_y = _bbox_center_y(line_bbox)
|
|
87
|
+
return any(bbox[0] <= center_x <= bbox[2] and bbox[1] <= center_y <= bbox[3] for bbox in container_bboxes)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _line_near_visual_container(
|
|
91
|
+
line_bbox: BBox,
|
|
92
|
+
container_bboxes: list[BBox],
|
|
93
|
+
body_height: float,
|
|
94
|
+
) -> bool:
|
|
95
|
+
"""检查短行是否紧邻图、表等容器,以纯几何方式抑制 caption 误判。"""
|
|
96
|
+
|
|
97
|
+
for container_bbox in container_bboxes:
|
|
98
|
+
if _bbox_axis_overlap_ratio(line_bbox, container_bbox, axis="x") < 0.35:
|
|
99
|
+
continue
|
|
100
|
+
container_width = max(0.1, container_bbox[2] - container_bbox[0])
|
|
101
|
+
if line_bbox[2] - line_bbox[0] > 0.8 * container_width:
|
|
102
|
+
continue
|
|
103
|
+
vertical_gap = max(line_bbox[1] - container_bbox[3], container_bbox[1] - line_bbox[3], 0.0)
|
|
104
|
+
if vertical_gap <= 2.0 * body_height:
|
|
105
|
+
return True
|
|
106
|
+
return False
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
__all__ = [
|
|
110
|
+
"_NUMBERED_SECTION_TITLE_RE",
|
|
111
|
+
"_SECTION_NUMBER_ONLY_RE",
|
|
112
|
+
"_SECTION_TITLE_TERMINAL_RE",
|
|
113
|
+
"_UNNUMBERED_SECTION_HEADING_RE",
|
|
114
|
+
"_build_physical_title_gap_map",
|
|
115
|
+
"_line_inside_visual_container",
|
|
116
|
+
"_line_near_visual_container",
|
|
117
|
+
]
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""使用既有页面分类探测构建全文标题原型。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import statistics
|
|
6
|
+
from dataclasses import replace
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
from ..geometry import _rotate_bbox_to_upright
|
|
10
|
+
from ..line_layout import _infer_text_lanes, _line_effective_height, _normalized_font_family
|
|
11
|
+
from ..models import _DocumentBodyProfile, _DocumentTitleProfile, _PreparedPage, _TitleStylePrototype
|
|
12
|
+
from .common import _line_inside_visual_container
|
|
13
|
+
from .page_titles import _classify_page_titles
|
|
14
|
+
from .prototype import _title_profile_alignment, _title_profile_seed_matches_cluster
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _infer_document_title_profile(
|
|
18
|
+
prepared_pages: list[_PreparedPage],
|
|
19
|
+
document_body_profile: _DocumentBodyProfile | None,
|
|
20
|
+
) -> _DocumentTitleProfile | None:
|
|
21
|
+
"""在副本上复用现有标题判定,并把跨页稳定样式聚成标题原型。"""
|
|
22
|
+
|
|
23
|
+
if document_body_profile is None:
|
|
24
|
+
return None
|
|
25
|
+
seeds: list[
|
|
26
|
+
tuple[
|
|
27
|
+
str,
|
|
28
|
+
int,
|
|
29
|
+
float,
|
|
30
|
+
float | None,
|
|
31
|
+
Literal["left", "center"],
|
|
32
|
+
float,
|
|
33
|
+
int,
|
|
34
|
+
]
|
|
35
|
+
] = []
|
|
36
|
+
for page_index, prepared in enumerate(prepared_pages):
|
|
37
|
+
probe_lines = [replace(line) for line in prepared.remaining_lines]
|
|
38
|
+
container_bboxes = [
|
|
39
|
+
block["bbox"] for block in prepared.fixed_blocks if not isinstance(block.get("_inline_visual_row_id"), int)
|
|
40
|
+
]
|
|
41
|
+
_classify_page_titles(
|
|
42
|
+
probe_lines,
|
|
43
|
+
prepared.page_size,
|
|
44
|
+
page_index=page_index,
|
|
45
|
+
container_bboxes=container_bboxes,
|
|
46
|
+
document_body_profile=document_body_profile,
|
|
47
|
+
)
|
|
48
|
+
seen_rows: set[tuple[int, int]] = set()
|
|
49
|
+
for angle in sorted({line.angle for line in probe_lines}):
|
|
50
|
+
geometry = [
|
|
51
|
+
(
|
|
52
|
+
line,
|
|
53
|
+
_rotate_bbox_to_upright(
|
|
54
|
+
line.bbox,
|
|
55
|
+
prepared.page_size,
|
|
56
|
+
angle,
|
|
57
|
+
),
|
|
58
|
+
)
|
|
59
|
+
for line in probe_lines
|
|
60
|
+
if line.angle == angle
|
|
61
|
+
]
|
|
62
|
+
if not geometry:
|
|
63
|
+
continue
|
|
64
|
+
median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in geometry)
|
|
65
|
+
local_page_width = prepared.page_size[1] if angle in {90, 270} else prepared.page_size[0]
|
|
66
|
+
lanes = _infer_text_lanes(
|
|
67
|
+
geometry,
|
|
68
|
+
local_page_width,
|
|
69
|
+
median_height,
|
|
70
|
+
)
|
|
71
|
+
local_container_bboxes = [
|
|
72
|
+
_rotate_bbox_to_upright(
|
|
73
|
+
bbox,
|
|
74
|
+
prepared.page_size,
|
|
75
|
+
angle,
|
|
76
|
+
)
|
|
77
|
+
for bbox in container_bboxes
|
|
78
|
+
]
|
|
79
|
+
for lane in lanes:
|
|
80
|
+
for line, bbox in lane.lines:
|
|
81
|
+
line_height_ratio = _line_effective_height(line, bbox) / document_body_profile.body_height
|
|
82
|
+
large_unresolved_seed = (
|
|
83
|
+
page_index > 0
|
|
84
|
+
and line.semantic_type is None
|
|
85
|
+
and line_height_ratio >= 1.3
|
|
86
|
+
and bbox[2] - bbox[0] <= 0.8 * local_page_width
|
|
87
|
+
and not _line_inside_visual_container(
|
|
88
|
+
bbox,
|
|
89
|
+
local_container_bboxes,
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
if (
|
|
93
|
+
(line.semantic_type != "paragraph_title" and not large_unresolved_seed)
|
|
94
|
+
or line.font_signature is None
|
|
95
|
+
or line.font_coverage < 0.75
|
|
96
|
+
):
|
|
97
|
+
continue
|
|
98
|
+
row_identity = line.visual_row_id if line.visual_row_id is not None else line.source_index
|
|
99
|
+
row_key = (angle, row_identity)
|
|
100
|
+
if row_key in seen_rows:
|
|
101
|
+
continue
|
|
102
|
+
alignment = _title_profile_alignment(
|
|
103
|
+
bbox,
|
|
104
|
+
lane,
|
|
105
|
+
document_body_profile.body_height,
|
|
106
|
+
)
|
|
107
|
+
if alignment is None:
|
|
108
|
+
continue
|
|
109
|
+
font_family = _normalized_font_family(line.font_signature)
|
|
110
|
+
if font_family is None:
|
|
111
|
+
continue
|
|
112
|
+
mode, anchor_offset = alignment
|
|
113
|
+
seeds.append(
|
|
114
|
+
(
|
|
115
|
+
font_family,
|
|
116
|
+
line.font_signature[1],
|
|
117
|
+
line_height_ratio,
|
|
118
|
+
line.dominant_font_weight,
|
|
119
|
+
mode,
|
|
120
|
+
anchor_offset,
|
|
121
|
+
page_index,
|
|
122
|
+
)
|
|
123
|
+
)
|
|
124
|
+
seen_rows.add(row_key)
|
|
125
|
+
|
|
126
|
+
clusters: list[list[tuple[str, int, float, float | None, Literal["left", "center"], float, int]]] = []
|
|
127
|
+
for seed in sorted(seeds, key=lambda item: (item[0], item[1], item[4], item[2])):
|
|
128
|
+
target = next(
|
|
129
|
+
(cluster for cluster in clusters if _title_profile_seed_matches_cluster(seed, cluster)),
|
|
130
|
+
None,
|
|
131
|
+
)
|
|
132
|
+
if target is None:
|
|
133
|
+
clusters.append([seed])
|
|
134
|
+
else:
|
|
135
|
+
target.append(seed)
|
|
136
|
+
|
|
137
|
+
prototypes: list[_TitleStylePrototype] = []
|
|
138
|
+
for cluster in clusters:
|
|
139
|
+
support_pages = len({seed[6] for seed in cluster})
|
|
140
|
+
if len(cluster) < 3 and support_pages < 2:
|
|
141
|
+
continue
|
|
142
|
+
weights = [seed[3] for seed in cluster if seed[3] is not None]
|
|
143
|
+
prototypes.append(
|
|
144
|
+
_TitleStylePrototype(
|
|
145
|
+
font_family=cluster[0][0],
|
|
146
|
+
font_flags=cluster[0][1],
|
|
147
|
+
height_ratio=statistics.median(seed[2] for seed in cluster),
|
|
148
|
+
weight=statistics.median(weights) if weights else None,
|
|
149
|
+
alignment=cluster[0][4],
|
|
150
|
+
anchor_offset=statistics.median(seed[5] for seed in cluster),
|
|
151
|
+
support_count=len(cluster),
|
|
152
|
+
support_pages=support_pages,
|
|
153
|
+
)
|
|
154
|
+
)
|
|
155
|
+
if not prototypes:
|
|
156
|
+
return None
|
|
157
|
+
prototypes.sort(
|
|
158
|
+
key=lambda item: (item.support_pages, item.support_count),
|
|
159
|
+
reverse=True,
|
|
160
|
+
)
|
|
161
|
+
return _DocumentTitleProfile(tuple(prototypes))
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
__all__ = ["_infer_document_title_profile"]
|