docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,636 @@
|
|
|
1
|
+
"""提供 PDF 字符 loose/tight/origin 驱动的通用上下标几何分类。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
import statistics
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Any, Literal
|
|
9
|
+
|
|
10
|
+
from ....document.pdf.text.contracts import Char
|
|
11
|
+
|
|
12
|
+
from ....schema import BBox
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
SCRIPT_BODY_COMPARABLE_HEIGHT_RATIO = 0.9
|
|
16
|
+
SCRIPT_BASELINE_ABSOLUTE_TOLERANCE = 0.35
|
|
17
|
+
SCRIPT_BASELINE_LOOSE_HEIGHT_RATIO = 0.04
|
|
18
|
+
SCRIPT_ORIGIN_MIN_SHIFT_ABSOLUTE = 0.5
|
|
19
|
+
SCRIPT_ORIGIN_MIN_SHIFT_RATIO = 0.12
|
|
20
|
+
SCRIPT_TIGHT_HEIGHT_RATIO = 0.9
|
|
21
|
+
SCRIPT_STRONG_SHIFT_RATIO = 0.3
|
|
22
|
+
SCRIPT_STRONG_MAX_HEIGHT_RATIO = 1.1
|
|
23
|
+
SCRIPT_CONSENSUS_TIGHT_HEIGHT_RATIO = 0.88
|
|
24
|
+
SCRIPT_CONSENSUS_ORIGIN_SHIFT_RATIO = 0.22
|
|
25
|
+
SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO = 0.3
|
|
26
|
+
SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO = 0.15
|
|
27
|
+
SCRIPT_LOOSE_HEIGHT_ANOMALY_RATIO = 1.35
|
|
28
|
+
SCRIPT_COMPONENT_FORWARD_GAP_RATIO = 1.5
|
|
29
|
+
SCRIPT_COMPONENT_X_BACKTRACK_RATIO = 0.5
|
|
30
|
+
SCRIPT_COMPONENT_SEED_GAP_RATIO = 0.5
|
|
31
|
+
SCRIPT_COMPONENT_SEED_MAX_POSITION_DISTANCE = 4
|
|
32
|
+
SCRIPT_COMPONENT_MIN_OFFSET_RATIO = 0.08
|
|
33
|
+
SCRIPT_COMPONENT_MAX_HEIGHT_RATIO = 1.1
|
|
34
|
+
CONTROL_LINE_BREAK_CHARS = {"\r", "\n"}
|
|
35
|
+
|
|
36
|
+
ScriptRole = Literal["body", "sup", "sub"]
|
|
37
|
+
ScriptMarkRole = Literal["sup", "sub"]
|
|
38
|
+
_ScriptFontKey = tuple[str, int | None, int | None]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True, slots=True)
|
|
42
|
+
class ScriptCharFeature:
|
|
43
|
+
"""保存单个字符参与纯几何上下标判定所需的只读特征。"""
|
|
44
|
+
|
|
45
|
+
index: int
|
|
46
|
+
text: str
|
|
47
|
+
loose_bbox: BBox
|
|
48
|
+
tight_bbox: BBox | None
|
|
49
|
+
origin: tuple[float, float] | None
|
|
50
|
+
is_valid: bool
|
|
51
|
+
is_body_anchor: bool
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def loose_height(self) -> float:
|
|
55
|
+
"""返回 loose bbox 高度。"""
|
|
56
|
+
return self.loose_bbox[3] - self.loose_bbox[1]
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def loose_center_y(self) -> float:
|
|
60
|
+
"""返回 loose bbox 中心 y。"""
|
|
61
|
+
return (self.loose_bbox[1] + self.loose_bbox[3]) / 2
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def tight_height(self) -> float:
|
|
65
|
+
"""返回 tight bbox 高度,无有效框时返回零。"""
|
|
66
|
+
if self.tight_bbox is None:
|
|
67
|
+
return 0.0
|
|
68
|
+
return self.tight_bbox[3] - self.tight_bbox[1]
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def tight_center_y(self) -> float | None:
|
|
72
|
+
"""返回 tight bbox 中心 y。"""
|
|
73
|
+
if self.tight_bbox is None:
|
|
74
|
+
return None
|
|
75
|
+
return (self.tight_bbox[1] + self.tight_bbox[3]) / 2
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@dataclass(frozen=True, slots=True)
|
|
79
|
+
class ScriptBodyBand:
|
|
80
|
+
"""表示当前视觉组件的正文 origin 基线与双 bbox 参考高度。"""
|
|
81
|
+
|
|
82
|
+
baseline: float
|
|
83
|
+
tight_height: float
|
|
84
|
+
loose_height: float
|
|
85
|
+
member_indices: frozenset[int]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass(frozen=True, slots=True)
|
|
89
|
+
class ScriptBaselineCluster:
|
|
90
|
+
"""表示共享近似字符 origin 的正文或角标基线簇。"""
|
|
91
|
+
|
|
92
|
+
baseline: float
|
|
93
|
+
member_indices: tuple[int, ...]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _coerce_finite_bbox(value: Any) -> BBox | None:
|
|
97
|
+
"""把可迭代四元组收敛为合法有限 bbox。"""
|
|
98
|
+
try:
|
|
99
|
+
raw_bbox = getattr(value, "bbox", value)
|
|
100
|
+
if raw_bbox is None or len(raw_bbox) != 4:
|
|
101
|
+
return None
|
|
102
|
+
bbox = tuple(float(item) for item in raw_bbox)
|
|
103
|
+
except (TypeError, ValueError):
|
|
104
|
+
return None
|
|
105
|
+
if not all(math.isfinite(item) for item in bbox) or bbox[2] <= bbox[0] or bbox[3] <= bbox[1]:
|
|
106
|
+
return None
|
|
107
|
+
return bbox # type: ignore[return-value]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _char_geometry_key(char: Char) -> int | None:
|
|
111
|
+
"""返回可用于 side-map 查询的合法 PDFium char_idx。"""
|
|
112
|
+
char_idx = char.get("char_idx")
|
|
113
|
+
if isinstance(char_idx, bool) or not isinstance(char_idx, int):
|
|
114
|
+
return None
|
|
115
|
+
return char_idx
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _script_role(offset: float) -> ScriptMarkRole:
|
|
119
|
+
"""把相对正文 origin 基线的纵向偏移转换为上下标角色。"""
|
|
120
|
+
return "sub" if offset > 0 else "sup"
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def build_script_features(
|
|
124
|
+
chars: list[Char],
|
|
125
|
+
tight_bboxes: dict[int, BBox],
|
|
126
|
+
origins: dict[int, tuple[float, float]],
|
|
127
|
+
protected_body_indices: set[int],
|
|
128
|
+
) -> list[ScriptCharFeature]:
|
|
129
|
+
"""一次性构造 loose/tight/origin 三类字符几何特征。"""
|
|
130
|
+
features = []
|
|
131
|
+
for index, char in enumerate(chars):
|
|
132
|
+
text = str(char.get("char", ""))
|
|
133
|
+
char_idx = _char_geometry_key(char)
|
|
134
|
+
loose_bbox = _coerce_finite_bbox(char.get("bbox")) or (0.0, 0.0, 0.0, 0.0)
|
|
135
|
+
tight_bbox = _coerce_finite_bbox(tight_bboxes.get(char_idx)) if char_idx is not None else None
|
|
136
|
+
raw_origin = origins.get(char_idx) if char_idx is not None else None
|
|
137
|
+
origin = None
|
|
138
|
+
if raw_origin is not None:
|
|
139
|
+
try:
|
|
140
|
+
candidate_origin = (float(raw_origin[0]), float(raw_origin[1]))
|
|
141
|
+
except (IndexError, TypeError, ValueError):
|
|
142
|
+
candidate_origin = None
|
|
143
|
+
if candidate_origin is not None and all(math.isfinite(value) for value in candidate_origin):
|
|
144
|
+
origin = candidate_origin
|
|
145
|
+
is_valid = text not in CONTROL_LINE_BREAK_CHARS and not text.isspace() and tight_bbox is not None and origin is not None
|
|
146
|
+
features.append(
|
|
147
|
+
ScriptCharFeature(
|
|
148
|
+
index=index,
|
|
149
|
+
text=text,
|
|
150
|
+
loose_bbox=loose_bbox,
|
|
151
|
+
tight_bbox=tight_bbox,
|
|
152
|
+
origin=origin,
|
|
153
|
+
is_valid=is_valid,
|
|
154
|
+
is_body_anchor=is_valid and text.isalnum() and index not in protected_body_indices,
|
|
155
|
+
)
|
|
156
|
+
)
|
|
157
|
+
return features
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def split_script_visual_components(features: list[ScriptCharFeature]) -> list[list[int]]:
|
|
161
|
+
"""按换行、x 回退和大间隙切分独立视觉组件。"""
|
|
162
|
+
valid_heights = [feature.loose_height for feature in features if feature.loose_height > 0]
|
|
163
|
+
scale = statistics.median(valid_heights) if valid_heights else 1.0
|
|
164
|
+
components: list[list[int]] = []
|
|
165
|
+
current: list[int] = []
|
|
166
|
+
previous_visible: ScriptCharFeature | None = None
|
|
167
|
+
for feature in features:
|
|
168
|
+
if feature.text in CONTROL_LINE_BREAK_CHARS or feature.text.isspace():
|
|
169
|
+
if current:
|
|
170
|
+
components.append(current)
|
|
171
|
+
current = []
|
|
172
|
+
previous_visible = None
|
|
173
|
+
continue
|
|
174
|
+
if feature.is_valid and previous_visible is not None:
|
|
175
|
+
x_backtrack = previous_visible.loose_bbox[0] - feature.loose_bbox[0]
|
|
176
|
+
forward_gap = feature.loose_bbox[0] - previous_visible.loose_bbox[2]
|
|
177
|
+
if (
|
|
178
|
+
x_backtrack > scale * SCRIPT_COMPONENT_X_BACKTRACK_RATIO
|
|
179
|
+
or forward_gap > scale * SCRIPT_COMPONENT_FORWARD_GAP_RATIO
|
|
180
|
+
):
|
|
181
|
+
if current:
|
|
182
|
+
components.append(current)
|
|
183
|
+
current = []
|
|
184
|
+
previous_visible = None
|
|
185
|
+
current.append(feature.index)
|
|
186
|
+
if feature.is_valid:
|
|
187
|
+
previous_visible = feature
|
|
188
|
+
if current:
|
|
189
|
+
components.append(current)
|
|
190
|
+
return components
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _quantile(values: list[float], fraction: float) -> float:
|
|
194
|
+
"""返回适合小样本基线簇的稳定分位数。"""
|
|
195
|
+
if not values:
|
|
196
|
+
return 0.0
|
|
197
|
+
ordered = sorted(values)
|
|
198
|
+
index = min(len(ordered) - 1, max(0, round((len(ordered) - 1) * fraction)))
|
|
199
|
+
return ordered[index]
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _cluster_baselines(
|
|
203
|
+
features: list[ScriptCharFeature],
|
|
204
|
+
component_indices: list[int],
|
|
205
|
+
) -> tuple[list[ScriptBaselineCluster], float]:
|
|
206
|
+
"""按字符 origin y 聚类当前组件的字母数字基线。"""
|
|
207
|
+
anchors = [features[index] for index in component_indices if features[index].is_body_anchor]
|
|
208
|
+
if not anchors:
|
|
209
|
+
return [], 0.0
|
|
210
|
+
median_loose_height = statistics.median(feature.loose_height for feature in anchors)
|
|
211
|
+
tolerance = max(SCRIPT_BASELINE_ABSOLUTE_TOLERANCE, median_loose_height * SCRIPT_BASELINE_LOOSE_HEIGHT_RATIO)
|
|
212
|
+
groups: list[list[int]] = []
|
|
213
|
+
for feature in sorted(anchors, key=lambda item: item.origin[1] if item.origin is not None else 0.0):
|
|
214
|
+
baseline = feature.origin[1] if feature.origin is not None else 0.0
|
|
215
|
+
if not groups:
|
|
216
|
+
groups.append([feature.index])
|
|
217
|
+
continue
|
|
218
|
+
previous_baseline = statistics.median(
|
|
219
|
+
features[index].origin[1] for index in groups[-1] if features[index].origin is not None
|
|
220
|
+
)
|
|
221
|
+
if abs(baseline - previous_baseline) <= tolerance:
|
|
222
|
+
groups[-1].append(feature.index)
|
|
223
|
+
else:
|
|
224
|
+
groups.append([feature.index])
|
|
225
|
+
return (
|
|
226
|
+
[
|
|
227
|
+
ScriptBaselineCluster(
|
|
228
|
+
baseline=statistics.median(features[index].origin[1] for index in group if features[index].origin is not None),
|
|
229
|
+
member_indices=tuple(group),
|
|
230
|
+
)
|
|
231
|
+
for group in groups
|
|
232
|
+
],
|
|
233
|
+
tolerance,
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _cluster_tight_height(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
|
|
238
|
+
"""返回基线簇的高分位 tight 高度。"""
|
|
239
|
+
return _quantile(
|
|
240
|
+
[features[index].tight_height for index in cluster.member_indices if features[index].tight_height > 0], 0.9
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _cluster_loose_height(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
|
|
245
|
+
"""返回基线簇的高分位 loose 高度。"""
|
|
246
|
+
return _quantile(
|
|
247
|
+
[features[index].loose_height for index in cluster.member_indices if features[index].loose_height > 0], 0.9
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _cluster_tight_center(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
|
|
252
|
+
"""返回基线簇的 tight bbox 中心中位数。"""
|
|
253
|
+
return statistics.median(
|
|
254
|
+
features[index].tight_center_y for index in cluster.member_indices if features[index].tight_center_y is not None
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _cluster_loose_center(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
|
|
259
|
+
"""返回基线簇的 loose bbox 中心中位数。"""
|
|
260
|
+
return statistics.median(features[index].loose_center_y for index in cluster.member_indices)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _cluster_has_consistent_displacement(
|
|
264
|
+
features: list[ScriptCharFeature],
|
|
265
|
+
cluster: ScriptBaselineCluster,
|
|
266
|
+
body_cluster: ScriptBaselineCluster,
|
|
267
|
+
body_band: ScriptBodyBand,
|
|
268
|
+
) -> bool:
|
|
269
|
+
"""要求 origin 与双 bbox 至少两项同向,并排除普通混合字体的弱中心偏移。"""
|
|
270
|
+
origin_shift = cluster.baseline - body_band.baseline
|
|
271
|
+
tight_shift = _cluster_tight_center(features, cluster) - _cluster_tight_center(features, body_cluster)
|
|
272
|
+
loose_shift = _cluster_loose_center(features, cluster) - _cluster_loose_center(features, body_cluster)
|
|
273
|
+
shifts = (origin_shift, tight_shift, loose_shift)
|
|
274
|
+
expected_positive = origin_shift > 0
|
|
275
|
+
if sum(value > 0.05 if expected_positive else value < -0.05 for value in shifts) < 2:
|
|
276
|
+
return False
|
|
277
|
+
origin_ratio = abs(origin_shift) / max(body_band.tight_height, 1e-6)
|
|
278
|
+
tight_ratio = abs(tight_shift) / max(body_band.tight_height, 1e-6)
|
|
279
|
+
loose_ratio = abs(loose_shift) / max(body_band.loose_height, 1e-6)
|
|
280
|
+
return origin_ratio >= SCRIPT_CONSENSUS_ORIGIN_SHIFT_RATIO or (
|
|
281
|
+
tight_ratio >= SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO and loose_ratio >= SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _choose_body_band(
|
|
286
|
+
features: list[ScriptCharFeature],
|
|
287
|
+
clusters: list[ScriptBaselineCluster],
|
|
288
|
+
) -> tuple[ScriptBodyBand, ScriptBaselineCluster] | None:
|
|
289
|
+
"""在接近最高字形的基线簇中按字符数选择正文基线。"""
|
|
290
|
+
if not clusters:
|
|
291
|
+
return None
|
|
292
|
+
maximum_height = max(_cluster_tight_height(features, cluster) for cluster in clusters)
|
|
293
|
+
comparable = [
|
|
294
|
+
cluster
|
|
295
|
+
for cluster in clusters
|
|
296
|
+
if _cluster_tight_height(features, cluster) >= maximum_height * SCRIPT_BODY_COMPARABLE_HEIGHT_RATIO
|
|
297
|
+
]
|
|
298
|
+
body_cluster = max(
|
|
299
|
+
comparable,
|
|
300
|
+
key=lambda cluster: (
|
|
301
|
+
len(cluster.member_indices),
|
|
302
|
+
_cluster_tight_height(features, cluster),
|
|
303
|
+
_cluster_loose_height(features, cluster),
|
|
304
|
+
),
|
|
305
|
+
)
|
|
306
|
+
return (
|
|
307
|
+
ScriptBodyBand(
|
|
308
|
+
baseline=body_cluster.baseline,
|
|
309
|
+
tight_height=_cluster_tight_height(features, body_cluster),
|
|
310
|
+
loose_height=_cluster_loose_height(features, body_cluster),
|
|
311
|
+
member_indices=frozenset(body_cluster.member_indices),
|
|
312
|
+
),
|
|
313
|
+
body_cluster,
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def _nearest_cluster(
|
|
318
|
+
feature: ScriptCharFeature,
|
|
319
|
+
clusters: list[ScriptBaselineCluster],
|
|
320
|
+
tolerance: float,
|
|
321
|
+
) -> ScriptBaselineCluster | None:
|
|
322
|
+
"""把非字母数字字符附着到最近 origin 基线簇。"""
|
|
323
|
+
if feature.origin is None or not clusters:
|
|
324
|
+
return None
|
|
325
|
+
nearest = min(clusters, key=lambda cluster: abs(feature.origin[1] - cluster.baseline))
|
|
326
|
+
return nearest if abs(feature.origin[1] - nearest.baseline) <= tolerance else None
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _horizontal_gap(first: BBox, second: BBox) -> float:
|
|
330
|
+
"""返回两个字符 tight bbox 的水平间隙。"""
|
|
331
|
+
return max(0.0, first[0] - second[2], second[0] - first[2])
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _drop_unseeded_punctuation(
|
|
335
|
+
features: list[ScriptCharFeature],
|
|
336
|
+
component_indices: list[int],
|
|
337
|
+
roles: list[ScriptRole],
|
|
338
|
+
body_height: float,
|
|
339
|
+
) -> None:
|
|
340
|
+
"""移除未邻近同类字母数字种子的标点。"""
|
|
341
|
+
seeded = [
|
|
342
|
+
features[index]
|
|
343
|
+
for index in component_indices
|
|
344
|
+
if roles[index] != "body" and features[index].text.isalnum() and features[index].tight_bbox is not None
|
|
345
|
+
]
|
|
346
|
+
for index in component_indices:
|
|
347
|
+
feature = features[index]
|
|
348
|
+
role = roles[index]
|
|
349
|
+
if role == "body" or feature.text.isalnum() or feature.tight_bbox is None:
|
|
350
|
+
continue
|
|
351
|
+
nearby = any(
|
|
352
|
+
roles[seed.index] == role
|
|
353
|
+
and _horizontal_gap(feature.tight_bbox, seed.tight_bbox) <= max(2.0, body_height * SCRIPT_COMPONENT_SEED_GAP_RATIO)
|
|
354
|
+
and abs(feature.index - seed.index) <= SCRIPT_COMPONENT_SEED_MAX_POSITION_DISTANCE
|
|
355
|
+
for seed in seeded
|
|
356
|
+
)
|
|
357
|
+
if not nearby:
|
|
358
|
+
roles[index] = "body"
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _apply_consensus_candidates(
|
|
362
|
+
features: list[ScriptCharFeature],
|
|
363
|
+
component_indices: list[int],
|
|
364
|
+
body_band: ScriptBodyBand,
|
|
365
|
+
roles: list[ScriptRole],
|
|
366
|
+
) -> None:
|
|
367
|
+
"""用 origin/tight/loose 三证据一致性补充孤立边界字符。"""
|
|
368
|
+
body_members = [
|
|
369
|
+
features[index]
|
|
370
|
+
for index in body_band.member_indices
|
|
371
|
+
if features[index].tight_bbox is not None
|
|
372
|
+
and features[index].tight_center_y is not None
|
|
373
|
+
and features[index].origin is not None
|
|
374
|
+
]
|
|
375
|
+
for index in component_indices:
|
|
376
|
+
feature = features[index]
|
|
377
|
+
if (
|
|
378
|
+
roles[index] != "body"
|
|
379
|
+
or index in body_band.member_indices
|
|
380
|
+
or not feature.is_valid
|
|
381
|
+
or feature.tight_bbox is None
|
|
382
|
+
or feature.tight_center_y is None
|
|
383
|
+
or feature.origin is None
|
|
384
|
+
or not body_members
|
|
385
|
+
):
|
|
386
|
+
continue
|
|
387
|
+
reference = min(
|
|
388
|
+
body_members,
|
|
389
|
+
key=lambda member: (_horizontal_gap(feature.tight_bbox, member.tight_bbox), abs(feature.index - member.index)),
|
|
390
|
+
)
|
|
391
|
+
if _horizontal_gap(feature.tight_bbox, reference.tight_bbox) > max(2.5, reference.tight_height * 1.2):
|
|
392
|
+
continue
|
|
393
|
+
if feature.tight_height / max(reference.tight_height, 1e-6) > SCRIPT_CONSENSUS_TIGHT_HEIGHT_RATIO:
|
|
394
|
+
continue
|
|
395
|
+
shifts = (
|
|
396
|
+
feature.origin[1] - reference.origin[1],
|
|
397
|
+
feature.tight_center_y - reference.tight_center_y,
|
|
398
|
+
feature.loose_center_y - reference.loose_center_y,
|
|
399
|
+
)
|
|
400
|
+
positive_votes = sum(value > 0.05 for value in shifts)
|
|
401
|
+
negative_votes = sum(value < -0.05 for value in shifts)
|
|
402
|
+
if max(positive_votes, negative_votes) < 2:
|
|
403
|
+
continue
|
|
404
|
+
role = "sub" if positive_votes > negative_votes else "sup"
|
|
405
|
+
if _script_role(shifts[0]) != role:
|
|
406
|
+
continue
|
|
407
|
+
origin_ratio = abs(shifts[0]) / max(reference.tight_height, 1e-6)
|
|
408
|
+
tight_ratio = abs(shifts[1]) / max(reference.tight_height, 1e-6)
|
|
409
|
+
loose_ratio = abs(shifts[2]) / max(reference.loose_height, 1e-6)
|
|
410
|
+
if not feature.text.isalnum():
|
|
411
|
+
expected_positive = role == "sub"
|
|
412
|
+
if not all(value > 0.05 if expected_positive else value < -0.05 for value in shifts) or not (
|
|
413
|
+
origin_ratio >= SCRIPT_STRONG_SHIFT_RATIO
|
|
414
|
+
and tight_ratio >= SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO
|
|
415
|
+
and loose_ratio >= SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO
|
|
416
|
+
):
|
|
417
|
+
continue
|
|
418
|
+
if origin_ratio >= SCRIPT_CONSENSUS_ORIGIN_SHIFT_RATIO or (
|
|
419
|
+
tight_ratio >= SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO
|
|
420
|
+
and loose_ratio >= SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO
|
|
421
|
+
):
|
|
422
|
+
roles[index] = role
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _expand_component_neighbors(
|
|
426
|
+
features: list[ScriptCharFeature],
|
|
427
|
+
component_indices: list[int],
|
|
428
|
+
body_band: ScriptBodyBand,
|
|
429
|
+
protected_body_indices: set[int],
|
|
430
|
+
roles: list[ScriptRole],
|
|
431
|
+
) -> None:
|
|
432
|
+
"""把同侧连续字符并入已有角标 run。"""
|
|
433
|
+
component_set = set(component_indices)
|
|
434
|
+
changed = True
|
|
435
|
+
while changed:
|
|
436
|
+
changed = False
|
|
437
|
+
for index in component_indices:
|
|
438
|
+
feature = features[index]
|
|
439
|
+
if (
|
|
440
|
+
roles[index] != "body"
|
|
441
|
+
or index in protected_body_indices
|
|
442
|
+
or not feature.is_valid
|
|
443
|
+
or feature.origin is None
|
|
444
|
+
or feature.text.isspace()
|
|
445
|
+
or feature.tight_height > body_band.tight_height * SCRIPT_COMPONENT_MAX_HEIGHT_RATIO
|
|
446
|
+
):
|
|
447
|
+
continue
|
|
448
|
+
neighbor_roles = {
|
|
449
|
+
roles[neighbor]
|
|
450
|
+
for neighbor in (index - 1, index + 1)
|
|
451
|
+
if neighbor in component_set and roles[neighbor] != "body"
|
|
452
|
+
}
|
|
453
|
+
if len(neighbor_roles) != 1:
|
|
454
|
+
continue
|
|
455
|
+
role = next(iter(neighbor_roles))
|
|
456
|
+
shift = feature.origin[1] - body_band.baseline
|
|
457
|
+
if abs(shift) < body_band.tight_height * SCRIPT_COMPONENT_MIN_OFFSET_RATIO or _script_role(shift) != role:
|
|
458
|
+
continue
|
|
459
|
+
roles[index] = role
|
|
460
|
+
changed = True
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _script_font_key(char: Char) -> _ScriptFontKey | None:
|
|
464
|
+
"""读取原始字体身份,不合并子集名称,也不把原始字号当作有效字形高度。"""
|
|
465
|
+
font = char.get("font")
|
|
466
|
+
if not isinstance(font, dict):
|
|
467
|
+
return None
|
|
468
|
+
name = font.get("name")
|
|
469
|
+
if not isinstance(name, str) or not name:
|
|
470
|
+
return None
|
|
471
|
+
flags, weight = font.get("flags"), font.get("weight")
|
|
472
|
+
return name, flags if type(flags) is int else None, weight if type(weight) is int else None
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def _local_font_run(
|
|
476
|
+
features: list[ScriptCharFeature],
|
|
477
|
+
font_keys: list[_ScriptFontKey | None],
|
|
478
|
+
component: set[int],
|
|
479
|
+
cluster: ScriptBaselineCluster,
|
|
480
|
+
) -> set[int]:
|
|
481
|
+
"""限定同一视觉组件内连续的同字体 ASCII/全角英数字串。"""
|
|
482
|
+
keys = {font_keys[index] for index in cluster.member_indices}
|
|
483
|
+
if len(keys) != 1 or None in keys:
|
|
484
|
+
return set()
|
|
485
|
+
key = next(iter(keys))
|
|
486
|
+
|
|
487
|
+
def belongs(index: int) -> bool:
|
|
488
|
+
"""让空白、括号、运算符、CJK、缺失几何和字体边界阻断局部参考。"""
|
|
489
|
+
if index not in component or not features[index].is_valid or font_keys[index] != key:
|
|
490
|
+
return False
|
|
491
|
+
text = features[index].text
|
|
492
|
+
return len(text) == 1 and (
|
|
493
|
+
(text.isascii() and text.isalnum()) or "0" <= text <= "9" or "A" <= text <= "Z" or "a" <= text <= "z"
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
left = min(cluster.member_indices)
|
|
497
|
+
right = left + 1
|
|
498
|
+
while belongs(left - 1):
|
|
499
|
+
left -= 1
|
|
500
|
+
while belongs(right):
|
|
501
|
+
right += 1
|
|
502
|
+
if not all(left <= index < right and belongs(index) for index in cluster.member_indices):
|
|
503
|
+
return set()
|
|
504
|
+
return set(range(left, right))
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _recheck_weak_clusters_with_local_font_body(
|
|
508
|
+
features: list[ScriptCharFeature],
|
|
509
|
+
font_keys: list[_ScriptFontKey | None],
|
|
510
|
+
component_indices: list[int],
|
|
511
|
+
body_band: ScriptBodyBand,
|
|
512
|
+
cluster_roles: dict[ScriptBaselineCluster, ScriptRole],
|
|
513
|
+
) -> None:
|
|
514
|
+
"""用同字体局部正文撤销弱误判;不制造角标,不让撤销结果成为新参考。"""
|
|
515
|
+
initial_roles = dict(cluster_roles)
|
|
516
|
+
component = set(component_indices)
|
|
517
|
+
for cluster, role in initial_roles.items():
|
|
518
|
+
if role == "body" or abs(cluster.baseline - body_band.baseline) >= body_band.tight_height * SCRIPT_STRONG_SHIFT_RATIO:
|
|
519
|
+
continue
|
|
520
|
+
run = _local_font_run(features, font_keys, component, cluster)
|
|
521
|
+
if not run:
|
|
522
|
+
continue
|
|
523
|
+
references: list[tuple[float, int, ScriptBaselineCluster]] = []
|
|
524
|
+
for reference, reference_role in initial_roles.items():
|
|
525
|
+
if reference_role != "body":
|
|
526
|
+
continue
|
|
527
|
+
indices = tuple(index for index in reference.member_indices if index in run)
|
|
528
|
+
if len(indices) < 2:
|
|
529
|
+
continue
|
|
530
|
+
gap = min(
|
|
531
|
+
_horizontal_gap(features[first].tight_bbox, features[second].tight_bbox)
|
|
532
|
+
for first in cluster.member_indices
|
|
533
|
+
for second in indices
|
|
534
|
+
if features[first].tight_bbox is not None and features[second].tight_bbox is not None
|
|
535
|
+
)
|
|
536
|
+
if gap > max(2.5, body_band.tight_height * 1.2):
|
|
537
|
+
continue
|
|
538
|
+
baseline = statistics.median(features[index].origin[1] for index in indices if features[index].origin is not None)
|
|
539
|
+
references.append((gap, -len(indices), ScriptBaselineCluster(baseline, indices)))
|
|
540
|
+
if not references:
|
|
541
|
+
continue
|
|
542
|
+
reference = min(references, key=lambda item: item[:2])[2]
|
|
543
|
+
height = _cluster_tight_height(features, reference)
|
|
544
|
+
if height <= 0:
|
|
545
|
+
continue
|
|
546
|
+
shift = cluster.baseline - reference.baseline
|
|
547
|
+
strong_shift = abs(shift) >= height * SCRIPT_STRONG_SHIFT_RATIO
|
|
548
|
+
tight_ratio = _cluster_tight_height(features, cluster) / height
|
|
549
|
+
if abs(shift) < max(SCRIPT_ORIGIN_MIN_SHIFT_ABSOLUTE, height * SCRIPT_ORIGIN_MIN_SHIFT_RATIO) or (
|
|
550
|
+
tight_ratio > SCRIPT_TIGHT_HEIGHT_RATIO and not (strong_shift and tight_ratio <= SCRIPT_STRONG_MAX_HEIGHT_RATIO)
|
|
551
|
+
):
|
|
552
|
+
cluster_roles[cluster] = "body"
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _assign_component(
|
|
556
|
+
features: list[ScriptCharFeature],
|
|
557
|
+
component_indices: list[int],
|
|
558
|
+
protected_body_indices: set[int],
|
|
559
|
+
roles: list[ScriptRole],
|
|
560
|
+
font_keys: list[_ScriptFontKey | None],
|
|
561
|
+
) -> None:
|
|
562
|
+
"""在单个视觉组件内按 origin 基线簇和双 bbox 一致性分配角色。"""
|
|
563
|
+
clusters, tolerance = _cluster_baselines(features, component_indices)
|
|
564
|
+
body_result = _choose_body_band(features, clusters)
|
|
565
|
+
if body_result is None:
|
|
566
|
+
return
|
|
567
|
+
body_band, body_cluster = body_result
|
|
568
|
+
if body_band.tight_height <= 0 or body_band.loose_height <= 0:
|
|
569
|
+
return
|
|
570
|
+
cluster_roles: dict[ScriptBaselineCluster, ScriptRole] = {body_cluster: "body"}
|
|
571
|
+
for cluster in clusters:
|
|
572
|
+
if cluster is body_cluster:
|
|
573
|
+
continue
|
|
574
|
+
shift = cluster.baseline - body_band.baseline
|
|
575
|
+
tight_ratio = _cluster_tight_height(features, cluster) / body_band.tight_height
|
|
576
|
+
loose_ratio = _cluster_loose_height(features, cluster) / body_band.loose_height
|
|
577
|
+
minimum_shift = max(SCRIPT_ORIGIN_MIN_SHIFT_ABSOLUTE, body_band.tight_height * SCRIPT_ORIGIN_MIN_SHIFT_RATIO)
|
|
578
|
+
strong_shift = abs(shift) >= body_band.tight_height * SCRIPT_STRONG_SHIFT_RATIO
|
|
579
|
+
if (
|
|
580
|
+
abs(shift) < minimum_shift
|
|
581
|
+
or (
|
|
582
|
+
tight_ratio > SCRIPT_TIGHT_HEIGHT_RATIO and not (strong_shift and tight_ratio <= SCRIPT_STRONG_MAX_HEIGHT_RATIO)
|
|
583
|
+
)
|
|
584
|
+
or (
|
|
585
|
+
loose_ratio > SCRIPT_LOOSE_HEIGHT_ANOMALY_RATIO
|
|
586
|
+
and not _cluster_has_consistent_displacement(features, cluster, body_cluster, body_band)
|
|
587
|
+
)
|
|
588
|
+
):
|
|
589
|
+
cluster_roles[cluster] = "body"
|
|
590
|
+
else:
|
|
591
|
+
cluster_roles[cluster] = _script_role(shift)
|
|
592
|
+
_recheck_weak_clusters_with_local_font_body(features, font_keys, component_indices, body_band, cluster_roles)
|
|
593
|
+
for index in component_indices:
|
|
594
|
+
feature = features[index]
|
|
595
|
+
if (
|
|
596
|
+
index in protected_body_indices
|
|
597
|
+
or feature.origin is None
|
|
598
|
+
or feature.text.isspace()
|
|
599
|
+
or feature.text in CONTROL_LINE_BREAK_CHARS
|
|
600
|
+
):
|
|
601
|
+
continue
|
|
602
|
+
cluster = _nearest_cluster(feature, clusters, tolerance)
|
|
603
|
+
if cluster is not None:
|
|
604
|
+
roles[index] = cluster_roles.get(cluster, "body")
|
|
605
|
+
_drop_unseeded_punctuation(features, component_indices, roles, body_band.tight_height)
|
|
606
|
+
_apply_consensus_candidates(features, component_indices, body_band, roles)
|
|
607
|
+
_expand_component_neighbors(features, component_indices, body_band, protected_body_indices, roles)
|
|
608
|
+
for index in protected_body_indices.intersection(component_indices):
|
|
609
|
+
roles[index] = "body"
|
|
610
|
+
|
|
611
|
+
|
|
612
|
+
def classify_char_script_roles(
|
|
613
|
+
chars: list[Char],
|
|
614
|
+
*,
|
|
615
|
+
tight_bboxes: dict[int, BBox],
|
|
616
|
+
origins: dict[int, tuple[float, float]],
|
|
617
|
+
protected_body_indices: set[int] | None = None,
|
|
618
|
+
) -> list[ScriptRole]:
|
|
619
|
+
"""按视觉组件、origin 基线簇和双 bbox 一致性识别上下标。"""
|
|
620
|
+
protected = protected_body_indices or set()
|
|
621
|
+
features = build_script_features(chars, tight_bboxes, origins, protected)
|
|
622
|
+
font_keys = [_script_font_key(char) for char in chars]
|
|
623
|
+
roles: list[ScriptRole] = ["body"] * len(features)
|
|
624
|
+
for component_indices in split_script_visual_components(features):
|
|
625
|
+
_assign_component(features, component_indices, protected, roles, font_keys)
|
|
626
|
+
return roles
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
__all__ = [
|
|
630
|
+
"CONTROL_LINE_BREAK_CHARS",
|
|
631
|
+
"ScriptCharFeature",
|
|
632
|
+
"ScriptRole",
|
|
633
|
+
"build_script_features",
|
|
634
|
+
"classify_char_script_roles",
|
|
635
|
+
"split_script_visual_components",
|
|
636
|
+
]
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""原生 PDF 的公共复用能力;模型区域融合可独立使用这些纯数据操作。"""
|
|
2
|
+
|
|
3
|
+
from .native_text import _build_native_line_items as build_native_line_items
|
|
4
|
+
from .line_merging import (
|
|
5
|
+
_merge_overlapping_inline_text_clusters as merge_overlapping_inline_text_clusters,
|
|
6
|
+
_merge_same_baseline_text_lines as merge_same_baseline_text_lines,
|
|
7
|
+
)
|
|
8
|
+
from .script_geometry import ScriptRole, classify_char_script_roles
|
|
9
|
+
from .spatial_text import project_ocr_table_text as project_table_text
|
|
10
|
+
from .table_text_styles import render_native_table_html_with_scripts
|
|
11
|
+
from .table_recovery import (
|
|
12
|
+
NativeTableInput,
|
|
13
|
+
coerce_native_table_rectangles,
|
|
14
|
+
coerce_native_table_rules,
|
|
15
|
+
recover_native_pdf_table,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"build_native_line_items",
|
|
20
|
+
"merge_overlapping_inline_text_clusters",
|
|
21
|
+
"merge_same_baseline_text_lines",
|
|
22
|
+
"ScriptRole",
|
|
23
|
+
"classify_char_script_roles",
|
|
24
|
+
"project_table_text",
|
|
25
|
+
"render_native_table_html_with_scripts",
|
|
26
|
+
"NativeTableInput",
|
|
27
|
+
"coerce_native_table_rectangles",
|
|
28
|
+
"coerce_native_table_rules",
|
|
29
|
+
"recover_native_pdf_table",
|
|
30
|
+
]
|