docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""PDF 字符到文本片段及行的共享纯数据接口。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .contracts import Bbox, Char, Line, Span
|
|
6
|
+
from .groups import assign_scripts, get_lines
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def get_spans(chars: list[Char], superscript_height_threshold: float = 0.8, line_distance_threshold: float = 0.1) -> list[Span]:
|
|
10
|
+
"""直接从自有字符记录构建片段,避免容器与数组往返转换。"""
|
|
11
|
+
spans: list[Span] = []
|
|
12
|
+
for char in chars:
|
|
13
|
+
current = spans[-1] if spans else None
|
|
14
|
+
box = char["bbox"]
|
|
15
|
+
new_span = current is None
|
|
16
|
+
if current is not None:
|
|
17
|
+
previous = current["chars"][-1]
|
|
18
|
+
height = current["bbox"].height
|
|
19
|
+
new_span = (
|
|
20
|
+
char["font"] != current["font"]
|
|
21
|
+
or char["rotation"] != current["rotation"]
|
|
22
|
+
or previous["char"] in {"\x02", "\n"}
|
|
23
|
+
or (
|
|
24
|
+
box.y_start < current["bbox"].y_start - height * line_distance_threshold
|
|
25
|
+
and box.y_end < height * superscript_height_threshold + current["bbox"].y_start
|
|
26
|
+
and box.x_start > current["bbox"].x_end
|
|
27
|
+
)
|
|
28
|
+
)
|
|
29
|
+
if new_span:
|
|
30
|
+
spans.append(
|
|
31
|
+
{
|
|
32
|
+
"bbox": box.copy(),
|
|
33
|
+
"text": char["char"],
|
|
34
|
+
"font": char["font"],
|
|
35
|
+
"chars": [char],
|
|
36
|
+
"char_start_idx": char["char_idx"],
|
|
37
|
+
"char_end_idx": char["char_idx"],
|
|
38
|
+
"rotation": char["rotation"],
|
|
39
|
+
"url": "",
|
|
40
|
+
"superscript": False,
|
|
41
|
+
"subscript": False,
|
|
42
|
+
}
|
|
43
|
+
)
|
|
44
|
+
else:
|
|
45
|
+
current["bbox"].merge_inplace(box)
|
|
46
|
+
current["text"] += char["char"]
|
|
47
|
+
current["chars"].append(char)
|
|
48
|
+
current["char_end_idx"] = char["char_idx"]
|
|
49
|
+
return spans
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def get_lines_from_chars(
|
|
53
|
+
chars: list[Char], superscript_height_threshold: float = 0.7, line_distance_threshold: float = 0.1
|
|
54
|
+
) -> list[Line]:
|
|
55
|
+
"""由已物化字符生成基础文本行,不访问 PDFium 或源文档。"""
|
|
56
|
+
spans = get_spans(chars, superscript_height_threshold, line_distance_threshold)
|
|
57
|
+
lines = get_lines(spans)
|
|
58
|
+
assign_scripts(lines, superscript_height_threshold, line_distance_threshold)
|
|
59
|
+
return lines
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
__all__ = ["Bbox", "Char", "Line", "Span", "get_lines_from_chars"]
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
# Portions derived from pdftext 0.7.1, Copyright Vik Paruchuri, Apache-2.0.
|
|
2
|
+
# Changed in DocVortex: owned character types and source-index mappings replace upstream containers.
|
|
3
|
+
"""DocVortex 自有 PDF 字符与几何数据,不携带 PDFium 句柄。"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
from typing import Any, TypedDict
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Bbox:
|
|
10
|
+
__slots__ = ("bbox", "ensure_nonzero_area")
|
|
11
|
+
|
|
12
|
+
def __init__(self, bbox: list[float], ensure_nonzero_area: bool = False) -> None:
|
|
13
|
+
"""建立独立矩形对象并按需保证面积。"""
|
|
14
|
+
if ensure_nonzero_area:
|
|
15
|
+
bbox = list(bbox)
|
|
16
|
+
bbox[2] = max(bbox[0], bbox[2] + 1)
|
|
17
|
+
bbox[3] = max(bbox[1], bbox[3] + 1)
|
|
18
|
+
self.bbox = bbox
|
|
19
|
+
self.ensure_nonzero_area = ensure_nonzero_area
|
|
20
|
+
|
|
21
|
+
def __getitem__(self, item: int | slice) -> float | list[float]:
|
|
22
|
+
"""读取矩形坐标。"""
|
|
23
|
+
return self.bbox[item]
|
|
24
|
+
|
|
25
|
+
def __repr__(self) -> str:
|
|
26
|
+
"""返回矩形的可读表示。"""
|
|
27
|
+
return f"Bbox({self.bbox})"
|
|
28
|
+
|
|
29
|
+
def __reduce__(self) -> tuple:
|
|
30
|
+
# ensure_nonzero_area is already applied at construction; don't re-apply on unpickle
|
|
31
|
+
"""保存矩形数值以支持跨进程序列化。"""
|
|
32
|
+
return (Bbox, (self.bbox,))
|
|
33
|
+
|
|
34
|
+
def copy(self) -> Bbox:
|
|
35
|
+
"""复制矩形避免共享累加状态。"""
|
|
36
|
+
return Bbox(list(self.bbox))
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def height(self) -> float:
|
|
40
|
+
"""计算矩形的 height 几何属性。"""
|
|
41
|
+
return self.bbox[3] - self.bbox[1]
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def width(self) -> float:
|
|
45
|
+
"""计算矩形的 width 几何属性。"""
|
|
46
|
+
return self.bbox[2] - self.bbox[0]
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def area(self) -> float:
|
|
50
|
+
"""计算矩形的 area 几何属性。"""
|
|
51
|
+
return self.width * self.height
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def center(self) -> list[float]:
|
|
55
|
+
"""计算矩形的 center 几何属性。"""
|
|
56
|
+
return [(self.bbox[0] + self.bbox[2]) / 2, (self.bbox[1] + self.bbox[3]) / 2]
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def size(self) -> list[float]:
|
|
60
|
+
"""计算矩形的 size 几何属性。"""
|
|
61
|
+
return [self.width, self.height]
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def x_start(self) -> float:
|
|
65
|
+
"""计算矩形的 x_start 几何属性。"""
|
|
66
|
+
return self.bbox[0]
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def y_start(self) -> float:
|
|
70
|
+
"""计算矩形的 y_start 几何属性。"""
|
|
71
|
+
return self.bbox[1]
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def x_end(self) -> float:
|
|
75
|
+
"""计算矩形的 x_end 几何属性。"""
|
|
76
|
+
return self.bbox[2]
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def y_end(self) -> float:
|
|
80
|
+
"""计算矩形的 y_end 几何属性。"""
|
|
81
|
+
return self.bbox[3]
|
|
82
|
+
|
|
83
|
+
def merge(self, other: Bbox) -> Bbox:
|
|
84
|
+
"""返回覆盖两个矩形的新对象。"""
|
|
85
|
+
self_bbox = self.bbox
|
|
86
|
+
other_bbox = other.bbox
|
|
87
|
+
return Bbox(
|
|
88
|
+
[
|
|
89
|
+
min(self_bbox[0], other_bbox[0]),
|
|
90
|
+
min(self_bbox[1], other_bbox[1]),
|
|
91
|
+
max(self_bbox[2], other_bbox[2]),
|
|
92
|
+
max(self_bbox[3], other_bbox[3]),
|
|
93
|
+
]
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
def merge_inplace(self, other: Bbox) -> Bbox:
|
|
97
|
+
# Mutates this bbox; only safe on accumulator bboxes that aren't shared
|
|
98
|
+
"""仅修改当前累加矩形。"""
|
|
99
|
+
self_bbox = self.bbox
|
|
100
|
+
other_bbox = other.bbox
|
|
101
|
+
if other_bbox[0] < self_bbox[0]:
|
|
102
|
+
self_bbox[0] = other_bbox[0]
|
|
103
|
+
if other_bbox[1] < self_bbox[1]:
|
|
104
|
+
self_bbox[1] = other_bbox[1]
|
|
105
|
+
if other_bbox[2] > self_bbox[2]:
|
|
106
|
+
self_bbox[2] = other_bbox[2]
|
|
107
|
+
if other_bbox[3] > self_bbox[3]:
|
|
108
|
+
self_bbox[3] = other_bbox[3]
|
|
109
|
+
return self
|
|
110
|
+
|
|
111
|
+
def overlap_x(self, other: Bbox) -> float:
|
|
112
|
+
"""计算矩形的 overlap_x 几何属性。"""
|
|
113
|
+
return max(0, min(self.bbox[2], other.bbox[2]) - max(self.bbox[0], other.bbox[0]))
|
|
114
|
+
|
|
115
|
+
def overlap_y(self, other: Bbox) -> float:
|
|
116
|
+
"""计算矩形的 overlap_y 几何属性。"""
|
|
117
|
+
return max(0, min(self.bbox[3], other.bbox[3]) - max(self.bbox[1], other.bbox[1]))
|
|
118
|
+
|
|
119
|
+
def intersection_area(self, other: Bbox) -> float:
|
|
120
|
+
"""计算矩形的 intersection_area 几何属性。"""
|
|
121
|
+
return self.overlap_x(other) * self.overlap_y(other)
|
|
122
|
+
|
|
123
|
+
def intersection_pct(self, other: Bbox) -> float:
|
|
124
|
+
"""计算矩形的 intersection_pct 几何属性。"""
|
|
125
|
+
if self.area <= 0:
|
|
126
|
+
return 0
|
|
127
|
+
|
|
128
|
+
intersection = self.intersection_area(other)
|
|
129
|
+
return intersection / self.area
|
|
130
|
+
|
|
131
|
+
def rotate(self, page_width: float, page_height: float, rotation: int) -> Bbox:
|
|
132
|
+
"""将矩形转换到旋转后的页面坐标。"""
|
|
133
|
+
if rotation not in [0, 90, 180, 270]:
|
|
134
|
+
raise ValueError("Rotation must be one of [0, 90, 180, 270] degrees.")
|
|
135
|
+
|
|
136
|
+
x_min, y_min, x_max, y_max = self.bbox
|
|
137
|
+
|
|
138
|
+
if rotation == 0:
|
|
139
|
+
return Bbox(list(self.bbox))
|
|
140
|
+
elif rotation == 90:
|
|
141
|
+
new_x_min = page_height - y_max
|
|
142
|
+
new_y_min = x_min
|
|
143
|
+
new_x_max = page_height - y_min
|
|
144
|
+
new_y_max = x_max
|
|
145
|
+
elif rotation == 180:
|
|
146
|
+
new_x_min = page_width - x_max
|
|
147
|
+
new_y_min = page_height - y_max
|
|
148
|
+
new_x_max = page_width - x_min
|
|
149
|
+
new_y_max = page_height - y_min
|
|
150
|
+
elif rotation == 270:
|
|
151
|
+
new_x_min = y_min
|
|
152
|
+
new_y_min = page_width - x_max
|
|
153
|
+
new_x_max = y_max
|
|
154
|
+
new_y_max = page_width - x_min
|
|
155
|
+
|
|
156
|
+
# Ensure that x_min < x_max and y_min < y_max; must stay a list so
|
|
157
|
+
# merge_inplace can mutate it
|
|
158
|
+
rotated_bbox = [
|
|
159
|
+
min(new_x_min, new_x_max),
|
|
160
|
+
min(new_y_min, new_y_max),
|
|
161
|
+
max(new_x_min, new_x_max),
|
|
162
|
+
max(new_y_min, new_y_max),
|
|
163
|
+
]
|
|
164
|
+
|
|
165
|
+
return Bbox(rotated_bbox)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
class _CharValue(TypedDict):
|
|
169
|
+
bbox: Bbox
|
|
170
|
+
char: str
|
|
171
|
+
rotation: float
|
|
172
|
+
font: dict[str, Any]
|
|
173
|
+
char_idx: int
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class Char(_CharValue, total=False):
|
|
177
|
+
"""字符及原始索引映射,几何缺失保留为显式空值。"""
|
|
178
|
+
|
|
179
|
+
source_indices: tuple[int, ...]
|
|
180
|
+
raw_code: int
|
|
181
|
+
loose_bbox: tuple[float, float, float, float] | None
|
|
182
|
+
tight_bbox: tuple[float, float, float, float] | None
|
|
183
|
+
origin: tuple[float, float] | None
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class Span(TypedDict):
|
|
187
|
+
"""基础字体片段,包含已解码文本及原始字符引用。"""
|
|
188
|
+
|
|
189
|
+
bbox: Bbox
|
|
190
|
+
text: str
|
|
191
|
+
font: dict[str, Any]
|
|
192
|
+
chars: list[Char]
|
|
193
|
+
char_start_idx: int
|
|
194
|
+
char_end_idx: int
|
|
195
|
+
rotation: float
|
|
196
|
+
url: str
|
|
197
|
+
superscript: bool
|
|
198
|
+
subscript: bool
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
class Line(TypedDict):
|
|
202
|
+
"""基础文本行,几何与片段顺序均保持可追溯。"""
|
|
203
|
+
|
|
204
|
+
spans: list[Span]
|
|
205
|
+
bbox: Bbox
|
|
206
|
+
rotation: float
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
Spans = list[Span]
|
|
210
|
+
Lines = list[Line]
|
|
211
|
+
__all__ = ["Bbox", "Char", "Span", "Line", "Spans", "Lines"]
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
# Portions derived from pdftext 0.7.1, Copyright Vik Paruchuri, Apache-2.0.
|
|
2
|
+
# Changed in DocVortex: direct PDFium extraction collects character and extended geometry together.
|
|
3
|
+
"""在一次 PDFium 字符遍历内收集原始码值、字体及可选几何。"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
from ctypes import byref, c_double, c_int, create_string_buffer
|
|
8
|
+
import math
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
import pypdfium2 as pdfium
|
|
12
|
+
import pypdfium2.raw as raw
|
|
13
|
+
|
|
14
|
+
from .contracts import Bbox, Char
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def transform_point(
|
|
18
|
+
point: tuple[float, float], page_bbox: tuple[float, float, float, float], rotation: int
|
|
19
|
+
) -> tuple[float, float]:
|
|
20
|
+
"""保留浮点页面框,将 PDF 原始坐标转换到视觉页面坐标。"""
|
|
21
|
+
left, bottom, right, top = page_bbox
|
|
22
|
+
width, height = abs(right - left), abs(top - bottom)
|
|
23
|
+
x, y = point[0] - min(left, right), max(bottom, top) - point[1]
|
|
24
|
+
rotation %= 360
|
|
25
|
+
if rotation == 90:
|
|
26
|
+
return height - y, x
|
|
27
|
+
if rotation == 180:
|
|
28
|
+
return width - x, height - y
|
|
29
|
+
if rotation == 270:
|
|
30
|
+
return y, width - x
|
|
31
|
+
return x, y
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def visual_bbox(
|
|
35
|
+
box: tuple[float, float, float, float], page_bbox: tuple[float, float, float, float], rotation: int
|
|
36
|
+
) -> tuple[float, float, float, float] | None:
|
|
37
|
+
"""转换原始矩形,几何缺失或零面积时返回空值而非伪造坐标。"""
|
|
38
|
+
left, bottom, right, top = box
|
|
39
|
+
points = [
|
|
40
|
+
transform_point(point, page_bbox, rotation) for point in ((left, bottom), (left, top), (right, bottom), (right, top))
|
|
41
|
+
]
|
|
42
|
+
result = (min(p[0] for p in points), min(p[1] for p in points), max(p[0] for p in points), max(p[1] for p in points))
|
|
43
|
+
return result if all(math.isfinite(v) for v in result) and result[2] > result[0] and result[3] > result[1] else None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _font_name(handle: Any, index: int, buffer: Any, flags: c_int) -> tuple[str, int]:
|
|
47
|
+
"""复用字体缓冲区,超长名称按 PDFium 返回长度重新读取。"""
|
|
48
|
+
try:
|
|
49
|
+
length = raw.FPDFText_GetFontInfo(handle, index, buffer, len(buffer), byref(flags))
|
|
50
|
+
if length > len(buffer):
|
|
51
|
+
buffer = create_string_buffer(length)
|
|
52
|
+
raw.FPDFText_GetFontInfo(handle, index, buffer, length, byref(flags))
|
|
53
|
+
return (buffer.value.decode("utf-8", errors="replace"), flags.value) if length > 0 else ("", 0)
|
|
54
|
+
except pdfium.PdfiumError:
|
|
55
|
+
return "", 0
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def get_chars(
|
|
59
|
+
textpage: pdfium.PdfTextPage, page_bbox: list[float], page_rotation: int, *, include_geometry: bool = False
|
|
60
|
+
) -> list[Char]:
|
|
61
|
+
"""读取原始字符记录;原始码值始终保留,随后统一解码和去重。"""
|
|
62
|
+
handle = textpage.raw
|
|
63
|
+
left, bottom, right, top = page_bbox
|
|
64
|
+
width, height = math.ceil(abs(right - left)), math.ceil(abs(top - bottom))
|
|
65
|
+
rect = raw.FS_RECTF()
|
|
66
|
+
tight_left, tight_right, tight_bottom, tight_top = c_double(), c_double(), c_double(), c_double()
|
|
67
|
+
origin_x, origin_y = c_double(), c_double()
|
|
68
|
+
font_buffer, font_flags = create_string_buffer(256), c_int()
|
|
69
|
+
fonts: dict[tuple[Any, ...], dict[str, Any]] = {}
|
|
70
|
+
chars: list[Char] = []
|
|
71
|
+
for index in range(textpage.count_chars()):
|
|
72
|
+
code = int(raw.FPDFText_GetUnicode(handle, index))
|
|
73
|
+
rotation = float(raw.FPDFText_GetCharAngle(handle, index))
|
|
74
|
+
loose: tuple[float, float, float, float] | None = None
|
|
75
|
+
tight: tuple[float, float, float, float] | None = None
|
|
76
|
+
if rotation == 0 or include_geometry:
|
|
77
|
+
try:
|
|
78
|
+
if raw.FPDFText_GetLooseCharBox(handle, index, rect):
|
|
79
|
+
loose = (float(rect.left), float(rect.bottom), float(rect.right), float(rect.top))
|
|
80
|
+
except Exception:
|
|
81
|
+
if rotation == 0:
|
|
82
|
+
raise
|
|
83
|
+
if rotation != 0 or include_geometry:
|
|
84
|
+
try:
|
|
85
|
+
if raw.FPDFText_GetCharBox(handle, index, tight_left, tight_right, tight_bottom, tight_top):
|
|
86
|
+
tight = (tight_left.value, tight_bottom.value, tight_right.value, tight_top.value)
|
|
87
|
+
except Exception:
|
|
88
|
+
if rotation != 0:
|
|
89
|
+
raise
|
|
90
|
+
selected = loose if rotation == 0 else tight
|
|
91
|
+
if selected is None:
|
|
92
|
+
raise pdfium.PdfiumError("Failed to get charbox.")
|
|
93
|
+
x0, y0, x1, y1 = selected
|
|
94
|
+
# 布局框保留基线的整数页面高度,原始扩展几何另用浮点页面框。
|
|
95
|
+
ys = (height - (y0 - bottom), height - (y1 - bottom))
|
|
96
|
+
box = Bbox([min(x0, x1) - left, min(ys), max(x0, x1) - left, max(ys)])
|
|
97
|
+
if page_rotation:
|
|
98
|
+
box = box.rotate(width, height, page_rotation)
|
|
99
|
+
name, flags = _font_name(handle, index, font_buffer, font_flags)
|
|
100
|
+
size, weight = raw.FPDFText_GetFontSize(handle, index), raw.FPDFText_GetFontWeight(handle, index)
|
|
101
|
+
font = fonts.setdefault((name, flags, size, weight), {"name": name, "flags": flags, "size": size, "weight": weight})
|
|
102
|
+
char: Char = {
|
|
103
|
+
"bbox": box,
|
|
104
|
+
"char": chr(code) if not 0xD800 <= code <= 0xDFFF else "\ufffd",
|
|
105
|
+
"rotation": rotation,
|
|
106
|
+
"font": font,
|
|
107
|
+
"char_idx": index,
|
|
108
|
+
"source_indices": (index,),
|
|
109
|
+
"raw_code": code,
|
|
110
|
+
}
|
|
111
|
+
if include_geometry:
|
|
112
|
+
char["loose_bbox"] = visual_bbox(loose, tuple(page_bbox), page_rotation) if loose else None
|
|
113
|
+
char["tight_bbox"] = visual_bbox(tight, tuple(page_bbox), page_rotation) if tight else None
|
|
114
|
+
char["origin"] = None
|
|
115
|
+
try:
|
|
116
|
+
if raw.FPDFText_GetCharOrigin(handle, index, origin_x, origin_y):
|
|
117
|
+
origin = transform_point((origin_x.value, origin_y.value), tuple(page_bbox), page_rotation)
|
|
118
|
+
if all(math.isfinite(v) for v in origin):
|
|
119
|
+
char["origin"] = origin
|
|
120
|
+
except Exception:
|
|
121
|
+
pass
|
|
122
|
+
chars.append(char)
|
|
123
|
+
return chars
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def deduplicate_chars(chars: list[Char]) -> list[Char]:
|
|
127
|
+
"""按词文本、字体、方向及取整位置去重,保留最早原始索引。"""
|
|
128
|
+
if not chars:
|
|
129
|
+
return []
|
|
130
|
+
groups: list[list[Char]] = [[chars[0]]]
|
|
131
|
+
for char in chars[1:]:
|
|
132
|
+
previous = groups[-1][-1]
|
|
133
|
+
if (
|
|
134
|
+
previous["char"] in {"\x02", "\n", " "}
|
|
135
|
+
or char["font"] != previous["font"]
|
|
136
|
+
or char["rotation"] != previous["rotation"]
|
|
137
|
+
):
|
|
138
|
+
groups.append([])
|
|
139
|
+
groups[-1].append(char)
|
|
140
|
+
seen: dict[tuple[Any, ...], list[Char]] = {}
|
|
141
|
+
result: list[Char] = []
|
|
142
|
+
for group in groups:
|
|
143
|
+
box = group[0]["bbox"].copy()
|
|
144
|
+
for char in group[1:]:
|
|
145
|
+
box.merge_inplace(char["bbox"])
|
|
146
|
+
font = group[0]["font"]
|
|
147
|
+
key = (
|
|
148
|
+
tuple(round(float(v), 0) for v in box.bbox),
|
|
149
|
+
"".join(c["char"] for c in group),
|
|
150
|
+
group[0]["rotation"],
|
|
151
|
+
tuple(font.get(k) for k in ("name", "flags", "size", "weight")),
|
|
152
|
+
)
|
|
153
|
+
if key not in seen:
|
|
154
|
+
seen[key] = group
|
|
155
|
+
result.extend(group)
|
|
156
|
+
else:
|
|
157
|
+
for retained, duplicate in zip(seen[key], group):
|
|
158
|
+
retained["source_indices"] = (
|
|
159
|
+
*retained.get("source_indices", (retained["char_idx"],)),
|
|
160
|
+
*duplicate.get("source_indices", (duplicate["char_idx"],)),
|
|
161
|
+
)
|
|
162
|
+
return result
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
__all__ = ["transform_point", "visual_bbox"]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""几何载荷与独立矩形类型之间的纯值转换。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
from .contracts import Bbox
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def char_bbox_values(bbox: object) -> tuple[float, float, float, float] | None:
|
|
8
|
+
"""读取固定四元坐标,拒绝结构不完整的几何对象。"""
|
|
9
|
+
if isinstance(bbox, Bbox):
|
|
10
|
+
bbox = bbox.bbox
|
|
11
|
+
if isinstance(bbox, (tuple, list)) and len(bbox) == 4:
|
|
12
|
+
return tuple(float(value) for value in bbox)
|
|
13
|
+
return None
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
__all__ = ["char_bbox_values"]
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
# Portions derived from pdftext 0.7.1, Copyright Vik Paruchuri, Apache-2.0.
|
|
2
|
+
# Changed in DocVortex: grouping uses owned dictionaries without upstream container adapters.
|
|
3
|
+
"""基础文本行与上下标分组;保留已验证的几何判断。"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import math
|
|
7
|
+
import unicodedata
|
|
8
|
+
from .contracts import Line, Lines, Spans
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def is_math_symbol(char: str) -> bool:
|
|
12
|
+
"""判断单字符数学符号。"""
|
|
13
|
+
if len(char) != 1:
|
|
14
|
+
return False
|
|
15
|
+
|
|
16
|
+
category = unicodedata.category(char)
|
|
17
|
+
return category == "Sm"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _top2(values: list[float]) -> tuple[float, int, float]:
|
|
21
|
+
# Returns (max1, max1_idx, max2) so that max-excluding-index can be answered in O(1)
|
|
22
|
+
"""在线性时间内找到两个最大值。"""
|
|
23
|
+
max1 = max2 = float("-inf")
|
|
24
|
+
max1_idx = -1
|
|
25
|
+
for idx, v in enumerate(values):
|
|
26
|
+
if v > max1:
|
|
27
|
+
max2 = max1
|
|
28
|
+
max1 = v
|
|
29
|
+
max1_idx = idx
|
|
30
|
+
elif v > max2:
|
|
31
|
+
max2 = v
|
|
32
|
+
return max1, max1_idx, max2
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _bottom2(values: list[float]) -> tuple[float, int, float]:
|
|
36
|
+
"""在线性时间内找到两个最小值。"""
|
|
37
|
+
min1 = min2 = float("inf")
|
|
38
|
+
min1_idx = -1
|
|
39
|
+
for idx, v in enumerate(values):
|
|
40
|
+
if v < min1:
|
|
41
|
+
min2 = min1
|
|
42
|
+
min1 = v
|
|
43
|
+
min1_idx = idx
|
|
44
|
+
elif v < min2:
|
|
45
|
+
min2 = v
|
|
46
|
+
return min1, min1_idx, min2
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def assign_scripts(lines: Lines, height_threshold: float = 0.8, line_distance_threshold: float = 0.1) -> None:
|
|
50
|
+
"""根据邻接片段几何设置基础上下标提示。"""
|
|
51
|
+
for line in lines:
|
|
52
|
+
spans = line["spans"]
|
|
53
|
+
if len(spans) < 2:
|
|
54
|
+
continue
|
|
55
|
+
|
|
56
|
+
line_bbox = line["bbox"].bbox
|
|
57
|
+
line_height = line_bbox[3] - line_bbox[1]
|
|
58
|
+
# Skip vertical lines
|
|
59
|
+
if line_height > line_bbox[2] - line_bbox[0]:
|
|
60
|
+
continue
|
|
61
|
+
|
|
62
|
+
# Precompute per-span geometry once; the loop below would otherwise
|
|
63
|
+
# recompute these via Bbox properties O(n^2) times per line
|
|
64
|
+
heights = []
|
|
65
|
+
y_starts = []
|
|
66
|
+
y_ends = []
|
|
67
|
+
v_above = []
|
|
68
|
+
v_below = []
|
|
69
|
+
for s in spans:
|
|
70
|
+
bbox = s["bbox"].bbox
|
|
71
|
+
height = bbox[3] - bbox[1]
|
|
72
|
+
heights.append(height)
|
|
73
|
+
y_starts.append(bbox[1])
|
|
74
|
+
y_ends.append(bbox[3])
|
|
75
|
+
v_above.append(bbox[1] - height * line_distance_threshold)
|
|
76
|
+
v_below.append(bbox[3] + height * line_distance_threshold)
|
|
77
|
+
|
|
78
|
+
above_max1, above_max1_idx, above_max2 = _top2(v_above)
|
|
79
|
+
below_min1, below_min1_idx, below_min2 = _bottom2(v_below)
|
|
80
|
+
max_line_height = max(1, line_height)
|
|
81
|
+
last_idx = len(spans) - 1
|
|
82
|
+
|
|
83
|
+
for i, span in enumerate(spans):
|
|
84
|
+
is_first = i == 0 or not spans[i - 1]["text"].strip()
|
|
85
|
+
is_last = i == last_idx or not spans[i + 1]["text"].strip()
|
|
86
|
+
span_height = heights[i]
|
|
87
|
+
span_top = y_starts[i]
|
|
88
|
+
span_bottom = y_ends[i]
|
|
89
|
+
|
|
90
|
+
line_fullheight = span_height / max_line_height <= height_threshold
|
|
91
|
+
next_fullheight = is_last or span_height / max(1, heights[i + 1]) <= height_threshold
|
|
92
|
+
prev_fullheight = is_first or span_height / max(1, heights[i - 1]) <= height_threshold
|
|
93
|
+
|
|
94
|
+
# any(span_top < v_above[j] for j != i) == span_top < max(v_above excluding i)
|
|
95
|
+
above = span_top < (above_max2 if i == above_max1_idx else above_max1)
|
|
96
|
+
prev_above = is_first or span_top < y_starts[i - 1]
|
|
97
|
+
next_above = is_last or span_top < y_starts[i + 1]
|
|
98
|
+
|
|
99
|
+
below = span_bottom > (below_min2 if i == below_min1_idx else below_min1)
|
|
100
|
+
prev_below = is_first or span_bottom > y_ends[i - 1]
|
|
101
|
+
next_below = is_last or span_bottom > y_ends[i + 1]
|
|
102
|
+
|
|
103
|
+
span_text = span["text"].strip()
|
|
104
|
+
span_text_okay = all(
|
|
105
|
+
[
|
|
106
|
+
(len(span_text) == 1 or span_text.isdigit()), # Ensure that the span text is a single char or a number
|
|
107
|
+
span_text.isalnum()
|
|
108
|
+
or is_math_symbol(span_text), # Ensure that the span text is an alphanumeric or a math symbol
|
|
109
|
+
]
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
if all([(prev_fullheight or next_fullheight), (prev_above or next_above), above, line_fullheight, span_text_okay]):
|
|
113
|
+
span["superscript"] = True
|
|
114
|
+
elif all(
|
|
115
|
+
[(prev_fullheight or next_fullheight), (prev_below or next_below), below, line_fullheight, span_text_okay]
|
|
116
|
+
):
|
|
117
|
+
span["subscript"] = True
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def get_lines(spans: Spans) -> Lines:
|
|
121
|
+
"""按换行、角度和位置将片段聚合为行。"""
|
|
122
|
+
lines: Lines = []
|
|
123
|
+
line: Line = None
|
|
124
|
+
|
|
125
|
+
def line_break() -> None:
|
|
126
|
+
"""以当前片段开始一个新的文本行。"""
|
|
127
|
+
lines.append({"spans": [span], "bbox": span["bbox"].copy(), "rotation": span["rotation"]})
|
|
128
|
+
|
|
129
|
+
for span in spans:
|
|
130
|
+
if lines:
|
|
131
|
+
line = lines[-1]
|
|
132
|
+
|
|
133
|
+
if not line:
|
|
134
|
+
line_break()
|
|
135
|
+
continue
|
|
136
|
+
|
|
137
|
+
# we break if the previous span ends with a linebreak
|
|
138
|
+
last_text = line["spans"][-1]["text"]
|
|
139
|
+
if any(last_text.endswith(suffix) for suffix in ["\n", "\x02"]):
|
|
140
|
+
line_break()
|
|
141
|
+
continue
|
|
142
|
+
|
|
143
|
+
# rotations are radians from FPDFText_GetCharAngle; compare circularly.
|
|
144
|
+
# Only break on roughly perpendicular text: pdfium reports a 180-degree
|
|
145
|
+
# flip for ordinary text rendered with negative-scale matrices, which
|
|
146
|
+
# still belongs to the same visual line
|
|
147
|
+
if span["rotation"] != line["rotation"]:
|
|
148
|
+
rotation_diff = abs(span["rotation"] - line["rotation"]) % (2 * math.pi)
|
|
149
|
+
rotation_diff = min(rotation_diff, 2 * math.pi - rotation_diff)
|
|
150
|
+
if math.radians(45) <= rotation_diff <= math.radians(135):
|
|
151
|
+
line_break()
|
|
152
|
+
continue
|
|
153
|
+
|
|
154
|
+
# sometimes pdfium doesn't inject a linebreak, so we check the span positions
|
|
155
|
+
if span["bbox"].y_start > line["bbox"].y_end:
|
|
156
|
+
line_break()
|
|
157
|
+
continue
|
|
158
|
+
|
|
159
|
+
line["spans"].append(span)
|
|
160
|
+
line["bbox"].merge_inplace(span["bbox"])
|
|
161
|
+
|
|
162
|
+
return lines
|