docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
"""识别并构造 Flash 原生 PDF 的目录正文块。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
import statistics
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
from ....schema import BBox, BlockType
|
|
12
|
+
|
|
13
|
+
from .models import _LineItem
|
|
14
|
+
from .geometry import _bbox_center_x, _bbox_center_y, _bbox_overlap_in_smaller, _bbox_union_many, _rotate_bbox_to_upright
|
|
15
|
+
from .line_layout import _line_effective_height, _lines_tight_output_bbox
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
_INDEX_PAGE_NUMBER_RE = re.compile(
|
|
19
|
+
r"(?<![A-Za-z0-90-9])(?:[0-90-9]+|[ivxlcdmIVXLCDM]+)"
|
|
20
|
+
r"[\s.。.、,,;;::))\]】]*$"
|
|
21
|
+
)
|
|
22
|
+
_INDEX_MIN_ROWS = 5
|
|
23
|
+
_INDEX_MIN_PAGE_NUMBER_RATIO = 0.7
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(slots=True)
|
|
27
|
+
class _IndexRow:
|
|
28
|
+
"""保存目录候选视觉行的成员、局部几何和合并文本。"""
|
|
29
|
+
|
|
30
|
+
members: list[_LineItem]
|
|
31
|
+
local_member_bboxes: list[BBox]
|
|
32
|
+
local_bbox: BBox
|
|
33
|
+
content: str
|
|
34
|
+
ends_in_page_number: bool
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _extract_index_blocks(
|
|
38
|
+
lines: list[_LineItem],
|
|
39
|
+
page_size: tuple[float, float],
|
|
40
|
+
container_bboxes: list[BBox],
|
|
41
|
+
*,
|
|
42
|
+
require_heading: bool = False,
|
|
43
|
+
) -> tuple[list[dict[str, object]], list[_LineItem]]:
|
|
44
|
+
"""用页码行尾和稳定版式识别目录;预判阶段可强制要求几何目录标题。"""
|
|
45
|
+
|
|
46
|
+
claimed_line_ids: set[int] = set()
|
|
47
|
+
blocks: list[dict[str, object]] = []
|
|
48
|
+
for angle in sorted({line.angle for line in lines if line.semantic_type is None}):
|
|
49
|
+
rows = _build_index_rows(lines, page_size, angle)
|
|
50
|
+
if len(rows) < _INDEX_MIN_ROWS:
|
|
51
|
+
continue
|
|
52
|
+
local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
|
|
53
|
+
local_containers = [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in container_bboxes]
|
|
54
|
+
eligible_rows = [
|
|
55
|
+
row
|
|
56
|
+
for row in rows
|
|
57
|
+
if not any(_bbox_overlap_in_smaller(row.local_bbox, container_bbox) >= 0.35 for container_bbox in local_containers)
|
|
58
|
+
]
|
|
59
|
+
for band in _split_index_bands(eligible_rows):
|
|
60
|
+
candidate = _trim_index_band_edges(band)
|
|
61
|
+
if not _index_band_has_stable_layout(candidate, local_page_width):
|
|
62
|
+
continue
|
|
63
|
+
heading_row = _find_index_heading_row(
|
|
64
|
+
eligible_rows,
|
|
65
|
+
candidate,
|
|
66
|
+
local_page_width,
|
|
67
|
+
)
|
|
68
|
+
if require_heading and heading_row is None:
|
|
69
|
+
continue
|
|
70
|
+
if heading_row is not None:
|
|
71
|
+
for line in heading_row.members:
|
|
72
|
+
line.semantic_type = "paragraph_title"
|
|
73
|
+
for row in candidate:
|
|
74
|
+
for line in row.members:
|
|
75
|
+
line.semantic_type = BlockType.INDEX
|
|
76
|
+
claimed_line_ids.add(id(line))
|
|
77
|
+
candidate_lines = [line for row in candidate for line in row.members]
|
|
78
|
+
block: dict[str, object] = {
|
|
79
|
+
"type": BlockType.INDEX,
|
|
80
|
+
"bbox": _bbox_union_many([line.bbox for line in candidate_lines]),
|
|
81
|
+
"angle": angle,
|
|
82
|
+
"content": "\n".join(row.content for row in candidate),
|
|
83
|
+
}
|
|
84
|
+
tight_output_bbox = _lines_tight_output_bbox(
|
|
85
|
+
candidate_lines,
|
|
86
|
+
page_size,
|
|
87
|
+
)
|
|
88
|
+
if tight_output_bbox is not None:
|
|
89
|
+
block["_tight_output_bbox"] = tight_output_bbox
|
|
90
|
+
blocks.append(block)
|
|
91
|
+
|
|
92
|
+
remaining_lines = [line for line in lines if id(line) not in claimed_line_ids]
|
|
93
|
+
return blocks, remaining_lines
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _build_index_rows(
|
|
97
|
+
lines: list[_LineItem],
|
|
98
|
+
page_size: tuple[float, float],
|
|
99
|
+
angle: int,
|
|
100
|
+
) -> list[_IndexRow]:
|
|
101
|
+
"""按 visual_row_id 复原当前方向的完整视觉行,保留左右分裂成员。"""
|
|
102
|
+
|
|
103
|
+
row_groups: dict[tuple[str, int], list[_LineItem]] = {}
|
|
104
|
+
for line in lines:
|
|
105
|
+
if line.angle != angle or line.semantic_type is not None:
|
|
106
|
+
continue
|
|
107
|
+
key = ("visual", line.visual_row_id) if line.visual_row_id is not None else ("source", line.source_index)
|
|
108
|
+
row_groups.setdefault(key, []).append(line)
|
|
109
|
+
|
|
110
|
+
rows: list[_IndexRow] = []
|
|
111
|
+
for members in row_groups.values():
|
|
112
|
+
ordered = sorted(
|
|
113
|
+
members,
|
|
114
|
+
key=lambda line: (
|
|
115
|
+
_rotate_bbox_to_upright(line.bbox, page_size, angle)[0],
|
|
116
|
+
line.run_index,
|
|
117
|
+
line.source_index,
|
|
118
|
+
),
|
|
119
|
+
)
|
|
120
|
+
local_member_bboxes = [_rotate_bbox_to_upright(line.bbox, page_size, angle) for line in ordered]
|
|
121
|
+
local_bbox = _bbox_union_many(local_member_bboxes)
|
|
122
|
+
content = " ".join(part for line in ordered if (part := line.text.strip()))
|
|
123
|
+
if not content:
|
|
124
|
+
continue
|
|
125
|
+
rows.append(
|
|
126
|
+
_IndexRow(
|
|
127
|
+
members=ordered,
|
|
128
|
+
local_member_bboxes=local_member_bboxes,
|
|
129
|
+
local_bbox=local_bbox,
|
|
130
|
+
content=content,
|
|
131
|
+
ends_in_page_number=_index_row_ends_in_page_number(content),
|
|
132
|
+
)
|
|
133
|
+
)
|
|
134
|
+
rows.sort(
|
|
135
|
+
key=lambda row: (
|
|
136
|
+
row.local_bbox[1],
|
|
137
|
+
row.local_bbox[0],
|
|
138
|
+
min(line.source_index for line in row.members),
|
|
139
|
+
)
|
|
140
|
+
)
|
|
141
|
+
return rows
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _index_row_ends_in_page_number(content: str) -> bool:
|
|
145
|
+
"""识别行尾的半角、全角阿拉伯页码或罗马页码。"""
|
|
146
|
+
|
|
147
|
+
return _INDEX_PAGE_NUMBER_RE.search(content.rstrip()) is not None
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _split_index_bands(rows: list[_IndexRow]) -> list[list[_IndexRow]]:
|
|
151
|
+
"""按显著纵向断层拆分候选带,同时容纳目录章节之间的加大行距。"""
|
|
152
|
+
|
|
153
|
+
if not rows:
|
|
154
|
+
return []
|
|
155
|
+
median_height = statistics.median(max(_line_effective_height(line, row.local_bbox) for line in row.members) for row in rows)
|
|
156
|
+
maximum_pitch = 3.25 * max(0.1, median_height)
|
|
157
|
+
bands: list[list[_IndexRow]] = [[rows[0]]]
|
|
158
|
+
for row in rows[1:]:
|
|
159
|
+
previous = bands[-1][-1]
|
|
160
|
+
if _bbox_center_y(row.local_bbox) - _bbox_center_y(previous.local_bbox) > maximum_pitch:
|
|
161
|
+
bands.append([row])
|
|
162
|
+
else:
|
|
163
|
+
bands[-1].append(row)
|
|
164
|
+
return bands
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _trim_index_band_edges(rows: list[_IndexRow]) -> list[_IndexRow]:
|
|
168
|
+
"""移除候选带两端不带页码的标题或邻接正文,内部少量续行继续保留。"""
|
|
169
|
+
|
|
170
|
+
start = 0
|
|
171
|
+
end = len(rows)
|
|
172
|
+
while start < end and not rows[start].ends_in_page_number:
|
|
173
|
+
start += 1
|
|
174
|
+
while end > start and not rows[end - 1].ends_in_page_number:
|
|
175
|
+
end -= 1
|
|
176
|
+
return rows[start:end]
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _index_band_has_stable_layout(
|
|
180
|
+
rows: list[_IndexRow],
|
|
181
|
+
local_page_width: float,
|
|
182
|
+
) -> bool:
|
|
183
|
+
"""联合页码比例、右边界、行宽、缩进和行距确认目录候选。"""
|
|
184
|
+
|
|
185
|
+
if len(rows) < _INDEX_MIN_ROWS or local_page_width <= 0:
|
|
186
|
+
return False
|
|
187
|
+
page_number_rows = [row for row in rows if row.ends_in_page_number]
|
|
188
|
+
required_page_number_rows = max(
|
|
189
|
+
4,
|
|
190
|
+
math.ceil(_INDEX_MIN_PAGE_NUMBER_RATIO * len(rows)),
|
|
191
|
+
)
|
|
192
|
+
if len(page_number_rows) < required_page_number_rows:
|
|
193
|
+
return False
|
|
194
|
+
|
|
195
|
+
row_heights = [max(0.1, row.local_bbox[3] - row.local_bbox[1]) for row in rows]
|
|
196
|
+
median_height = statistics.median(row_heights)
|
|
197
|
+
median_right = statistics.median(row.local_bbox[2] for row in page_number_rows)
|
|
198
|
+
right_tolerance = max(1.5 * median_height, 0.02 * local_page_width)
|
|
199
|
+
aligned_right_ratio = sum(abs(row.local_bbox[2] - median_right) <= right_tolerance for row in page_number_rows) / len(
|
|
200
|
+
page_number_rows
|
|
201
|
+
)
|
|
202
|
+
if median_right < 0.75 * local_page_width or aligned_right_ratio < 0.7:
|
|
203
|
+
return False
|
|
204
|
+
|
|
205
|
+
wide_row_ratio = sum(row.local_bbox[2] - row.local_bbox[0] >= 0.55 * local_page_width for row in rows) / len(rows)
|
|
206
|
+
right_sidecar_count = sum(_index_row_has_right_sidecar(row, local_page_width, median_height) for row in rows)
|
|
207
|
+
if wide_row_ratio < 0.7 and right_sidecar_count < 2:
|
|
208
|
+
return False
|
|
209
|
+
|
|
210
|
+
left_body_ratio = sum(row.local_bbox[0] <= 0.3 * local_page_width for row in rows) / len(rows)
|
|
211
|
+
if left_body_ratio < 0.7:
|
|
212
|
+
return False
|
|
213
|
+
|
|
214
|
+
pitches = [
|
|
215
|
+
_bbox_center_y(current.local_bbox) - _bbox_center_y(previous.local_bbox) for previous, current in zip(rows, rows[1:])
|
|
216
|
+
]
|
|
217
|
+
if not pitches or min(pitches) <= 0:
|
|
218
|
+
return False
|
|
219
|
+
median_pitch = statistics.median(pitches)
|
|
220
|
+
regular_pitch_ratio = sum(0.55 * median_pitch <= pitch <= 1.75 * median_pitch for pitch in pitches) / len(pitches)
|
|
221
|
+
return regular_pitch_ratio >= 0.7
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _find_index_heading_row(
|
|
225
|
+
all_rows: list[_IndexRow],
|
|
226
|
+
candidate: list[_IndexRow],
|
|
227
|
+
local_page_width: float,
|
|
228
|
+
) -> _IndexRow | None:
|
|
229
|
+
"""用候选带上方的居中、短行和垂直邻接关系保留目录标题。"""
|
|
230
|
+
|
|
231
|
+
if not candidate:
|
|
232
|
+
return None
|
|
233
|
+
first_position = next(
|
|
234
|
+
(index for index, row in enumerate(all_rows) if row is candidate[0]),
|
|
235
|
+
None,
|
|
236
|
+
)
|
|
237
|
+
if first_position is None or first_position == 0:
|
|
238
|
+
return None
|
|
239
|
+
heading = all_rows[first_position - 1]
|
|
240
|
+
if heading.ends_in_page_number:
|
|
241
|
+
return None
|
|
242
|
+
candidate_heights = [max(0.1, row.local_bbox[3] - row.local_bbox[1]) for row in candidate]
|
|
243
|
+
median_height = statistics.median(candidate_heights)
|
|
244
|
+
heading_width = heading.local_bbox[2] - heading.local_bbox[0]
|
|
245
|
+
centered = abs(_bbox_center_x(heading.local_bbox) - 0.5 * local_page_width) <= 0.15 * local_page_width
|
|
246
|
+
vertical_pitch = _bbox_center_y(candidate[0].local_bbox) - _bbox_center_y(heading.local_bbox)
|
|
247
|
+
if (
|
|
248
|
+
heading_width > 0.4 * local_page_width
|
|
249
|
+
or not centered
|
|
250
|
+
or not 1.25 * median_height <= vertical_pitch <= 5.0 * median_height
|
|
251
|
+
):
|
|
252
|
+
return None
|
|
253
|
+
return heading
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _index_row_has_right_sidecar(
|
|
257
|
+
row: _IndexRow,
|
|
258
|
+
local_page_width: float,
|
|
259
|
+
median_height: float,
|
|
260
|
+
) -> bool:
|
|
261
|
+
"""检查视觉行末尾是否存在靠近页面右侧的窄页码片段。"""
|
|
262
|
+
|
|
263
|
+
if len(row.members) < 2:
|
|
264
|
+
return False
|
|
265
|
+
last_bbox = row.local_member_bboxes[-1]
|
|
266
|
+
return last_bbox[0] >= 0.75 * local_page_width and last_bbox[2] - last_bbox[0] <= max(
|
|
267
|
+
0.12 * local_page_width, 4.0 * median_height
|
|
268
|
+
)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""提供行内证据共享的字符和几何规范化原语。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from typing import Any, Iterable, cast
|
|
7
|
+
|
|
8
|
+
from .....schema import BBox
|
|
9
|
+
from .types import (
|
|
10
|
+
_LIGATURE_REPLACEMENTS,
|
|
11
|
+
_PDF_CONTROL_CHAR_RE,
|
|
12
|
+
_PDF_SEPARATOR_SPACE_CHARS,
|
|
13
|
+
_PDF_ZERO_WIDTH_CHARS,
|
|
14
|
+
PDF_TEXT_STYLE_ORDER,
|
|
15
|
+
PDFTextStyle,
|
|
16
|
+
PDFTextStyleLine,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _style_line_reading_order_key(
|
|
21
|
+
line: PDFTextStyleLine,
|
|
22
|
+
) -> tuple[int, float, float]:
|
|
23
|
+
"""优先使用原生 source_index 排序,重复索引时再以 bbox 保持稳定。"""
|
|
24
|
+
|
|
25
|
+
return line.source_index, line.bbox[1], line.bbox[0]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _coerce_bbox(value: Any) -> BBox | None:
|
|
29
|
+
"""把 list、tuple 或 pdftext bbox 对象收敛为合法有限 bbox。"""
|
|
30
|
+
|
|
31
|
+
raw_bbox = getattr(value, "bbox", value)
|
|
32
|
+
try:
|
|
33
|
+
if raw_bbox is None or len(raw_bbox) != 4:
|
|
34
|
+
return None
|
|
35
|
+
bbox = tuple(float(item) for item in raw_bbox)
|
|
36
|
+
except (TypeError, ValueError):
|
|
37
|
+
return None
|
|
38
|
+
if not all(math.isfinite(item) for item in bbox):
|
|
39
|
+
return None
|
|
40
|
+
if bbox[2] <= bbox[0] or bbox[3] <= bbox[1]:
|
|
41
|
+
return None
|
|
42
|
+
return bbox # type: ignore[return-value]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _ordered_line_chars(line: Any) -> list[dict[str, Any]]:
|
|
46
|
+
"""按 char_idx 修复异常乱序字符,同时保留缺少索引时的来源顺序。"""
|
|
47
|
+
|
|
48
|
+
chars = [char for char in getattr(line, "chars", []) if isinstance(char, dict)]
|
|
49
|
+
indexed_chars = [char.get("char_idx") for char in chars]
|
|
50
|
+
if (
|
|
51
|
+
chars
|
|
52
|
+
and all(isinstance(index, int) for index in indexed_chars)
|
|
53
|
+
and any(first > second for first, second in zip(indexed_chars, indexed_chars[1:]))
|
|
54
|
+
):
|
|
55
|
+
return sorted(chars, key=lambda char: int(char["char_idx"]))
|
|
56
|
+
return chars
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _normalize_match_fragment(value: Any) -> str:
|
|
60
|
+
"""把单个字符片段规范为忽略排版空白的确定性匹配文本。"""
|
|
61
|
+
|
|
62
|
+
output: list[str] = []
|
|
63
|
+
for char in str(value or ""):
|
|
64
|
+
if char in _PDF_ZERO_WIDTH_CHARS or char == "\u00ad":
|
|
65
|
+
continue
|
|
66
|
+
if char == "\x02":
|
|
67
|
+
output.append("-")
|
|
68
|
+
continue
|
|
69
|
+
if char.isspace() or char in _PDF_SEPARATOR_SPACE_CHARS:
|
|
70
|
+
continue
|
|
71
|
+
if _PDF_CONTROL_CHAR_RE.fullmatch(char):
|
|
72
|
+
continue
|
|
73
|
+
output.append(_LIGATURE_REPLACEMENTS.get(char, char))
|
|
74
|
+
return "".join(output)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _canonical_styles(styles: Iterable[str]) -> tuple[PDFTextStyle, ...]:
|
|
78
|
+
"""按公开富文本协议顺序过滤、去重并规范样式集合。"""
|
|
79
|
+
|
|
80
|
+
style_set = set(styles)
|
|
81
|
+
return cast(
|
|
82
|
+
tuple[PDFTextStyle, ...],
|
|
83
|
+
tuple(style for style in PDF_TEXT_STYLE_ORDER if style in style_set),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _bbox_intersection_area(first: BBox, second: BBox) -> float:
|
|
88
|
+
"""返回两个合法 bbox 的相交面积。"""
|
|
89
|
+
|
|
90
|
+
width = max(0.0, min(first[2], second[2]) - max(first[0], second[0]))
|
|
91
|
+
height = max(0.0, min(first[3], second[3]) - max(first[1], second[1]))
|
|
92
|
+
return width * height
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _bbox_overlap_ratio(first: BBox, second: BBox) -> float:
|
|
96
|
+
"""返回 first 面积中落入 second 的比例。"""
|
|
97
|
+
|
|
98
|
+
intersection_width = max(0.0, min(first[2], second[2]) - max(first[0], second[0]))
|
|
99
|
+
intersection_height = max(0.0, min(first[3], second[3]) - max(first[1], second[1]))
|
|
100
|
+
first_area = max(0.01, (first[2] - first[0]) * (first[3] - first[1]))
|
|
101
|
+
return intersection_width * intersection_height / first_area
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
__all__ = [
|
|
105
|
+
"_style_line_reading_order_key",
|
|
106
|
+
"_coerce_bbox",
|
|
107
|
+
"_ordered_line_chars",
|
|
108
|
+
"_normalize_match_fragment",
|
|
109
|
+
"_canonical_styles",
|
|
110
|
+
"_bbox_intersection_area",
|
|
111
|
+
"_bbox_overlap_ratio",
|
|
112
|
+
]
|