docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,535 @@
|
|
|
1
|
+
"""识别带填充背景的等宽代码区域并投影其空间文本。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
import statistics
|
|
7
|
+
import unicodedata
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from .spatial_text import project_pdf_spatial_text
|
|
11
|
+
from ....schema import BBox
|
|
12
|
+
from ....document.pdf.document import PDFPathInfo
|
|
13
|
+
|
|
14
|
+
from .geometry import (
|
|
15
|
+
_bbox_area,
|
|
16
|
+
_bbox_center_x,
|
|
17
|
+
_bbox_center_y,
|
|
18
|
+
_bbox_overlap_in_first,
|
|
19
|
+
_bbox_overlap_in_smaller,
|
|
20
|
+
_bbox_union_many,
|
|
21
|
+
)
|
|
22
|
+
from .models import _CodeCandidate, _LineItem, _PageSource
|
|
23
|
+
from .native_text import _sanitize_pdf_control_text
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
_MONOSPACE_FONT_HINTS = (
|
|
27
|
+
"mono",
|
|
28
|
+
"courier",
|
|
29
|
+
"consolas",
|
|
30
|
+
"menlo",
|
|
31
|
+
"typewriter",
|
|
32
|
+
"fixed",
|
|
33
|
+
"code",
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _build_code_blocks(
|
|
38
|
+
source: _PageSource,
|
|
39
|
+
excluded_bboxes: list[BBox],
|
|
40
|
+
claimed_line_indices: set[int],
|
|
41
|
+
) -> tuple[list[dict[str, Any]], set[int]]:
|
|
42
|
+
"""检测代码背景、唯一认领其文本,并输出页内 code block。"""
|
|
43
|
+
|
|
44
|
+
candidates = _detect_code_candidates(
|
|
45
|
+
source,
|
|
46
|
+
excluded_bboxes,
|
|
47
|
+
claimed_line_indices,
|
|
48
|
+
)
|
|
49
|
+
return _materialize_code_candidates(source, candidates)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _build_rule_delimited_code_blocks(
|
|
53
|
+
source: _PageSource,
|
|
54
|
+
excluded_bboxes: list[BBox],
|
|
55
|
+
claimed_line_indices: set[int] | None = None,
|
|
56
|
+
) -> tuple[list[dict[str, Any]], set[int]]:
|
|
57
|
+
"""在表格认领前物化外框或上下横线限定的代码清单。"""
|
|
58
|
+
|
|
59
|
+
candidates = _detect_rule_delimited_code_candidates(
|
|
60
|
+
source,
|
|
61
|
+
excluded_bboxes,
|
|
62
|
+
claimed_line_indices or set(),
|
|
63
|
+
)
|
|
64
|
+
return _materialize_code_candidates(source, candidates)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _materialize_code_candidates(
|
|
68
|
+
source: _PageSource,
|
|
69
|
+
candidates: list[_CodeCandidate],
|
|
70
|
+
) -> tuple[list[dict[str, Any]], set[int]]:
|
|
71
|
+
"""统一投影代码候选,确保来源行只被一个 code block 认领。"""
|
|
72
|
+
|
|
73
|
+
if not candidates:
|
|
74
|
+
return [], set()
|
|
75
|
+
lines_by_index = {line.source_index: line for line in source.lines}
|
|
76
|
+
blocks: list[dict[str, Any]] = []
|
|
77
|
+
claimed: set[int] = set()
|
|
78
|
+
for candidate in candidates:
|
|
79
|
+
members = [
|
|
80
|
+
lines_by_index[source_index] for source_index in sorted(candidate.line_indices) if source_index in lines_by_index
|
|
81
|
+
]
|
|
82
|
+
if not members:
|
|
83
|
+
continue
|
|
84
|
+
content = project_pdf_spatial_text(
|
|
85
|
+
_code_member_chars(members),
|
|
86
|
+
candidate.bbox,
|
|
87
|
+
candidate.angle,
|
|
88
|
+
preserve_blank_rows=True,
|
|
89
|
+
)
|
|
90
|
+
if not content:
|
|
91
|
+
content = _fallback_code_content(members)
|
|
92
|
+
if not content:
|
|
93
|
+
continue
|
|
94
|
+
blocks.append(
|
|
95
|
+
{
|
|
96
|
+
"type": "code",
|
|
97
|
+
"bbox": candidate.bbox,
|
|
98
|
+
"angle": candidate.angle,
|
|
99
|
+
"content": content,
|
|
100
|
+
}
|
|
101
|
+
)
|
|
102
|
+
claimed.update(candidate.line_indices)
|
|
103
|
+
blocks.sort(key=lambda block: (block["bbox"][1], block["bbox"][0]))
|
|
104
|
+
return blocks, claimed
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _vertical_rule_candidate_height_coverage(
|
|
108
|
+
rule_bbox: BBox,
|
|
109
|
+
candidate_bbox: BBox,
|
|
110
|
+
) -> float:
|
|
111
|
+
"""计算竖轨在候选高度方向上的实际覆盖比例,长轨超出候选时仍按交集计量。"""
|
|
112
|
+
|
|
113
|
+
candidate_height = max(0.0, candidate_bbox[3] - candidate_bbox[1])
|
|
114
|
+
if candidate_height <= 0:
|
|
115
|
+
return 0.0
|
|
116
|
+
overlap = max(
|
|
117
|
+
0.0,
|
|
118
|
+
min(rule_bbox[3], candidate_bbox[3]) - max(rule_bbox[1], candidate_bbox[1]),
|
|
119
|
+
)
|
|
120
|
+
return overlap / candidate_height
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _detect_rule_delimited_code_candidates(
|
|
124
|
+
source: _PageSource,
|
|
125
|
+
excluded_bboxes: list[BBox],
|
|
126
|
+
claimed_line_indices: set[int],
|
|
127
|
+
) -> list[_CodeCandidate]:
|
|
128
|
+
"""按规则边界、稳定行距和缩进层次识别非等宽代码清单。"""
|
|
129
|
+
|
|
130
|
+
page_width, page_height = source.page_size
|
|
131
|
+
if page_width <= 0 or page_height <= 0:
|
|
132
|
+
return []
|
|
133
|
+
available = [line for line in source.lines if line.source_index not in claimed_line_indices and line.angle == 0]
|
|
134
|
+
if len(available) < 5:
|
|
135
|
+
return []
|
|
136
|
+
median_height = statistics.median(max(0.1, line.effective_height or line.bbox[3] - line.bbox[1]) for line in available)
|
|
137
|
+
horizontal_rules = sorted(
|
|
138
|
+
[
|
|
139
|
+
line
|
|
140
|
+
for line in source.drawing_lines
|
|
141
|
+
if line.orientation == "horizontal" and line.bbox[2] - line.bbox[0] >= 0.22 * page_width
|
|
142
|
+
],
|
|
143
|
+
key=lambda line: (line.bbox[1], line.bbox[0]),
|
|
144
|
+
)
|
|
145
|
+
vertical_rules = [line for line in source.drawing_lines if line.orientation == "vertical"]
|
|
146
|
+
raw_candidates: list[_CodeCandidate] = []
|
|
147
|
+
endpoint_tolerance = max(2.0, 0.75 * median_height)
|
|
148
|
+
for top_index, top_rule in enumerate(horizontal_rules[:-1]):
|
|
149
|
+
for bottom_rule in horizontal_rules[top_index + 1 :]:
|
|
150
|
+
if (
|
|
151
|
+
abs(top_rule.bbox[0] - bottom_rule.bbox[0]) > endpoint_tolerance
|
|
152
|
+
or abs(top_rule.bbox[2] - bottom_rule.bbox[2]) > endpoint_tolerance
|
|
153
|
+
):
|
|
154
|
+
continue
|
|
155
|
+
candidate_bbox = _bbox_union_many([top_rule.bbox, bottom_rule.bbox])
|
|
156
|
+
candidate_height = candidate_bbox[3] - candidate_bbox[1]
|
|
157
|
+
if not 6.0 * median_height <= candidate_height <= 0.5 * page_height:
|
|
158
|
+
continue
|
|
159
|
+
if any(_bbox_overlap_in_smaller(candidate_bbox, excluded_bbox) >= 0.5 for excluded_bbox in excluded_bboxes):
|
|
160
|
+
continue
|
|
161
|
+
interior_rules = [
|
|
162
|
+
rule
|
|
163
|
+
for rule in horizontal_rules
|
|
164
|
+
if top_rule.bbox[3] + 0.5 * median_height < rule.bbox[1] < bottom_rule.bbox[1] - 0.5 * median_height
|
|
165
|
+
and _bbox_overlap_in_first(rule.bbox, candidate_bbox) >= 0.8
|
|
166
|
+
and rule.bbox[2] - rule.bbox[0] >= 0.6 * (candidate_bbox[2] - candidate_bbox[0])
|
|
167
|
+
]
|
|
168
|
+
if interior_rules:
|
|
169
|
+
continue
|
|
170
|
+
internal_vertical_rules = [
|
|
171
|
+
rule
|
|
172
|
+
for rule in vertical_rules
|
|
173
|
+
if candidate_bbox[0] + median_height < _bbox_center_x(rule.bbox) < candidate_bbox[2] - median_height
|
|
174
|
+
# 表格竖轨可能贯穿候选上下边界,必须相对候选高度计算覆盖率。
|
|
175
|
+
and _vertical_rule_candidate_height_coverage(
|
|
176
|
+
rule.bbox,
|
|
177
|
+
candidate_bbox,
|
|
178
|
+
)
|
|
179
|
+
>= 0.6
|
|
180
|
+
]
|
|
181
|
+
if internal_vertical_rules:
|
|
182
|
+
continue
|
|
183
|
+
members = [
|
|
184
|
+
line
|
|
185
|
+
for line in available
|
|
186
|
+
if candidate_bbox[0] - 0.5 * median_height
|
|
187
|
+
<= _bbox_center_x(line.bbox)
|
|
188
|
+
<= candidate_bbox[2] + 0.5 * median_height
|
|
189
|
+
and top_rule.bbox[3] <= _bbox_center_y(line.bbox) <= bottom_rule.bbox[1]
|
|
190
|
+
]
|
|
191
|
+
if not _rule_delimited_code_members_are_structured(
|
|
192
|
+
members,
|
|
193
|
+
candidate_bbox,
|
|
194
|
+
median_height,
|
|
195
|
+
):
|
|
196
|
+
continue
|
|
197
|
+
raw_candidates.append(
|
|
198
|
+
_CodeCandidate(
|
|
199
|
+
bbox=candidate_bbox,
|
|
200
|
+
angle=0,
|
|
201
|
+
line_indices={line.source_index for line in members},
|
|
202
|
+
)
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
accepted: list[_CodeCandidate] = []
|
|
206
|
+
for candidate in sorted(raw_candidates, key=lambda item: _bbox_area(item.bbox)):
|
|
207
|
+
if any(_bbox_overlap_in_smaller(candidate.bbox, existing.bbox) >= 0.85 for existing in accepted):
|
|
208
|
+
continue
|
|
209
|
+
accepted.append(candidate)
|
|
210
|
+
return sorted(accepted, key=lambda item: (item.bbox[1], item.bbox[0]))
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _rule_delimited_code_members_are_structured(
|
|
214
|
+
members: list[_LineItem],
|
|
215
|
+
candidate_bbox: BBox,
|
|
216
|
+
median_height: float,
|
|
217
|
+
) -> bool:
|
|
218
|
+
"""验证候选具有稳定基线、代码缩进或窄行号槽,并排除规则多列表格。"""
|
|
219
|
+
|
|
220
|
+
if len(members) < 5:
|
|
221
|
+
return False
|
|
222
|
+
rows: dict[int, list[_LineItem]] = {}
|
|
223
|
+
fallback_row = 1_000_000
|
|
224
|
+
for line in members:
|
|
225
|
+
row_key = line.visual_row_id
|
|
226
|
+
if row_key is None:
|
|
227
|
+
row_key = fallback_row
|
|
228
|
+
fallback_row += 1
|
|
229
|
+
rows.setdefault(row_key, []).append(line)
|
|
230
|
+
if len(rows) < 5:
|
|
231
|
+
return False
|
|
232
|
+
ordered_rows = sorted(
|
|
233
|
+
rows.values(),
|
|
234
|
+
key=lambda row: min(_bbox_center_y(line.bbox) for line in row),
|
|
235
|
+
)
|
|
236
|
+
row_centers = [statistics.median(_bbox_center_y(line.bbox) for line in row) for row in ordered_rows]
|
|
237
|
+
row_gaps = [current - previous for previous, current in zip(row_centers, row_centers[1:]) if current > previous]
|
|
238
|
+
if not row_gaps:
|
|
239
|
+
return False
|
|
240
|
+
base_pitch = statistics.median(row_gaps)
|
|
241
|
+
if base_pitch <= 0 or sum(0.65 * base_pitch <= gap <= 1.8 * base_pitch for gap in row_gaps) / len(row_gaps) < 0.75:
|
|
242
|
+
return False
|
|
243
|
+
if sum(len(row) >= 3 for row in ordered_rows) / len(ordered_rows) >= 0.25:
|
|
244
|
+
return False
|
|
245
|
+
|
|
246
|
+
left_positions = sorted(line.bbox[0] for line in members)
|
|
247
|
+
indent_clusters: list[list[float]] = []
|
|
248
|
+
for position in left_positions:
|
|
249
|
+
if not indent_clusters or position - statistics.median(indent_clusters[-1]) > 0.75 * median_height:
|
|
250
|
+
indent_clusters.append([position])
|
|
251
|
+
else:
|
|
252
|
+
indent_clusters[-1].append(position)
|
|
253
|
+
narrow_gutter_rows = 0
|
|
254
|
+
for row in ordered_rows:
|
|
255
|
+
ordered = sorted(row, key=lambda line: line.bbox[0])
|
|
256
|
+
if (
|
|
257
|
+
len(ordered) >= 2
|
|
258
|
+
and ordered[0].bbox[2] - ordered[0].bbox[0] <= 2.0 * median_height
|
|
259
|
+
and ordered[1].bbox[0] - ordered[0].bbox[2] >= 0.5 * median_height
|
|
260
|
+
):
|
|
261
|
+
narrow_gutter_rows += 1
|
|
262
|
+
has_line_number_gutter = narrow_gutter_rows / len(ordered_rows) >= 0.35
|
|
263
|
+
has_indent_hierarchy = len(indent_clusters) >= 3 and sum(len(cluster) >= 2 for cluster in indent_clusters) >= 2
|
|
264
|
+
if not has_line_number_gutter and not has_indent_hierarchy:
|
|
265
|
+
return False
|
|
266
|
+
# 稳定行号槽属于强代码证据,允许右侧长语句或注释自然触及清单边界。
|
|
267
|
+
if has_line_number_gutter:
|
|
268
|
+
return True
|
|
269
|
+
|
|
270
|
+
occupied_width = max(line.bbox[2] for line in members) - min(line.bbox[0] for line in members)
|
|
271
|
+
return occupied_width <= 0.95 * max(
|
|
272
|
+
0.1,
|
|
273
|
+
candidate_bbox[2] - candidate_bbox[0],
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _code_member_chars(lines: list[_LineItem]) -> list[dict[str, Any]]:
|
|
278
|
+
"""按 char_idx 去重代码成员字符,避免区域内斜向水印混入空间投影。"""
|
|
279
|
+
|
|
280
|
+
output: list[dict[str, Any]] = []
|
|
281
|
+
seen: set[tuple[str, int]] = set()
|
|
282
|
+
for line in lines:
|
|
283
|
+
for fallback_index, char in enumerate(line.chars):
|
|
284
|
+
if not isinstance(char, dict):
|
|
285
|
+
continue
|
|
286
|
+
try:
|
|
287
|
+
identity = ("source", int(char.get("char_idx")))
|
|
288
|
+
except (TypeError, ValueError):
|
|
289
|
+
identity = ("fallback", fallback_index + line.source_index * 1_000_000)
|
|
290
|
+
if identity in seen:
|
|
291
|
+
continue
|
|
292
|
+
seen.add(identity)
|
|
293
|
+
output.append(char)
|
|
294
|
+
return output
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _detect_code_candidates(
|
|
298
|
+
source: _PageSource,
|
|
299
|
+
excluded_bboxes: list[BBox],
|
|
300
|
+
claimed_line_indices: set[int],
|
|
301
|
+
) -> list[_CodeCandidate]:
|
|
302
|
+
"""以非白填充矩形、等宽字体和规则空间栅格筛选代码候选。"""
|
|
303
|
+
|
|
304
|
+
page_width, page_height = source.page_size
|
|
305
|
+
page_area = max(0.0, page_width) * max(0.0, page_height)
|
|
306
|
+
if page_area <= 0 or not source.path_infos:
|
|
307
|
+
return []
|
|
308
|
+
raw_candidates: list[_CodeCandidate] = []
|
|
309
|
+
for path_info in source.path_infos:
|
|
310
|
+
bbox = path_info.bbox
|
|
311
|
+
width = bbox[2] - bbox[0]
|
|
312
|
+
height = bbox[3] - bbox[1]
|
|
313
|
+
if (
|
|
314
|
+
path_info.form_depth != 0
|
|
315
|
+
or not path_info.fill_visible
|
|
316
|
+
or path_info.segment_count < 4
|
|
317
|
+
or not _path_has_visible_nonwhite_fill(path_info)
|
|
318
|
+
or width < 0.5 * page_width
|
|
319
|
+
or height < 0.008 * page_height
|
|
320
|
+
or _bbox_area(bbox) >= 0.8 * page_area
|
|
321
|
+
or any(_bbox_overlap_in_smaller(bbox, excluded_bbox) >= 0.5 for excluded_bbox in excluded_bboxes)
|
|
322
|
+
):
|
|
323
|
+
continue
|
|
324
|
+
members = [
|
|
325
|
+
line
|
|
326
|
+
for line in source.lines
|
|
327
|
+
if line.source_index not in claimed_line_indices and _bbox_overlap_in_first(line.bbox, bbox) >= 0.8
|
|
328
|
+
]
|
|
329
|
+
if not members:
|
|
330
|
+
continue
|
|
331
|
+
dominant_angle = _dominant_code_angle(members)
|
|
332
|
+
angle_members = [line for line in members if line.angle == dominant_angle]
|
|
333
|
+
total_support = sum(_estimated_line_character_count(line) for line in members)
|
|
334
|
+
angle_support = sum(_estimated_line_character_count(line) for line in angle_members)
|
|
335
|
+
if total_support <= 0 or angle_support / total_support < 0.8:
|
|
336
|
+
continue
|
|
337
|
+
monospace_ratio, cell_widths = _monospace_character_support(angle_members)
|
|
338
|
+
if monospace_ratio < 0.8 or not _monospace_advances_are_stable(cell_widths):
|
|
339
|
+
continue
|
|
340
|
+
median_cell_width = statistics.median(width_value for values in cell_widths.values() for width_value in values)
|
|
341
|
+
if not _code_rows_have_spatial_structure(
|
|
342
|
+
angle_members,
|
|
343
|
+
bbox,
|
|
344
|
+
median_cell_width,
|
|
345
|
+
):
|
|
346
|
+
continue
|
|
347
|
+
raw_candidates.append(
|
|
348
|
+
_CodeCandidate(
|
|
349
|
+
bbox=bbox,
|
|
350
|
+
angle=dominant_angle,
|
|
351
|
+
line_indices={line.source_index for line in angle_members},
|
|
352
|
+
)
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
accepted: list[_CodeCandidate] = []
|
|
356
|
+
for candidate in sorted(raw_candidates, key=lambda item: _bbox_area(item.bbox)):
|
|
357
|
+
if any(_bbox_overlap_in_smaller(candidate.bbox, existing.bbox) >= 0.9 for existing in accepted):
|
|
358
|
+
continue
|
|
359
|
+
accepted.append(candidate)
|
|
360
|
+
return sorted(accepted, key=lambda item: (item.bbox[1], item.bbox[0]))
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _path_has_visible_nonwhite_fill(path_info: PDFPathInfo) -> bool:
|
|
364
|
+
"""检查填充色是否可见且与白色背景存在最小颜色差。"""
|
|
365
|
+
|
|
366
|
+
if path_info.fill_rgba is None:
|
|
367
|
+
return False
|
|
368
|
+
red, green, blue, alpha = path_info.fill_rgba
|
|
369
|
+
return alpha > 0 and max(255 - red, 255 - green, 255 - blue) >= 5
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _dominant_code_angle(lines: list[_LineItem]) -> int:
|
|
373
|
+
"""按估算字符数选择代码区域的主文本方向。"""
|
|
374
|
+
|
|
375
|
+
support: dict[int, float] = {}
|
|
376
|
+
for line in lines:
|
|
377
|
+
support[line.angle] = support.get(line.angle, 0.0) + _estimated_line_character_count(line)
|
|
378
|
+
return max(sorted(support), key=lambda angle: support[angle])
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _estimated_line_character_count(line: _LineItem) -> float:
|
|
382
|
+
"""优先按字符对象计数,缺失时用行宽和缓存字宽估算字符支持。"""
|
|
383
|
+
|
|
384
|
+
valid_chars = [char for char in line.chars if isinstance(char, dict) and str(char.get("char") or "").strip()]
|
|
385
|
+
if valid_chars:
|
|
386
|
+
return float(len(valid_chars))
|
|
387
|
+
line_width = max(0.1, line.bbox[2] - line.bbox[0])
|
|
388
|
+
glyph_width = max(0.1, line.median_glyph_width or line_width)
|
|
389
|
+
return max(1.0, line_width / glyph_width)
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def _font_name_looks_monospaced(name: str | None) -> bool:
|
|
393
|
+
"""按字体元数据中的通用等宽提示判断字体族,不匹配文档内容。"""
|
|
394
|
+
|
|
395
|
+
normalized = (name or "").replace("-", "").replace("_", "").casefold()
|
|
396
|
+
return any(hint in normalized for hint in _MONOSPACE_FONT_HINTS)
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def _monospace_character_support(
|
|
400
|
+
lines: list[_LineItem],
|
|
401
|
+
) -> tuple[float, dict[str, list[float]]]:
|
|
402
|
+
"""统计等宽字体字符占比,并按东西文宽度组收集字符 advance。"""
|
|
403
|
+
|
|
404
|
+
supported = 0.0
|
|
405
|
+
total = 0.0
|
|
406
|
+
widths: dict[str, list[float]] = {"narrow": [], "wide": []}
|
|
407
|
+
for line in lines:
|
|
408
|
+
fallback_monospace = _font_name_looks_monospaced(line.font_signature[0] if line.font_signature is not None else None)
|
|
409
|
+
for char in line.chars:
|
|
410
|
+
if not isinstance(char, dict):
|
|
411
|
+
continue
|
|
412
|
+
value = str(char.get("char") or "")
|
|
413
|
+
if not value.strip():
|
|
414
|
+
continue
|
|
415
|
+
total += 1.0
|
|
416
|
+
font = char.get("font")
|
|
417
|
+
font_name = font.get("name") if isinstance(font, dict) else None
|
|
418
|
+
if _font_name_looks_monospaced(font_name) or fallback_monospace:
|
|
419
|
+
supported += 1.0
|
|
420
|
+
try:
|
|
421
|
+
x0, _y0, x1, _y1 = [float(item) for item in char.get("bbox", [])]
|
|
422
|
+
except (TypeError, ValueError):
|
|
423
|
+
continue
|
|
424
|
+
width = x1 - x0
|
|
425
|
+
if not math.isfinite(width) or width <= 0.1:
|
|
426
|
+
continue
|
|
427
|
+
width_group = "wide" if unicodedata.east_asian_width(value[0]) in {"W", "F"} else "narrow"
|
|
428
|
+
widths[width_group].append(width)
|
|
429
|
+
|
|
430
|
+
if total <= 0:
|
|
431
|
+
fallback_support = sum(_estimated_line_character_count(line) for line in lines)
|
|
432
|
+
if fallback_support <= 0:
|
|
433
|
+
return 0.0, widths
|
|
434
|
+
supported_support = sum(
|
|
435
|
+
_estimated_line_character_count(line)
|
|
436
|
+
for line in lines
|
|
437
|
+
if _font_name_looks_monospaced(line.font_signature[0] if line.font_signature is not None else None)
|
|
438
|
+
)
|
|
439
|
+
fallback_widths = [
|
|
440
|
+
line.median_glyph_width for line in lines if line.median_glyph_width is not None and line.median_glyph_width > 0
|
|
441
|
+
]
|
|
442
|
+
widths["narrow"].extend(fallback_widths)
|
|
443
|
+
return supported_support / fallback_support, widths
|
|
444
|
+
return supported / total, widths
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _monospace_advances_are_stable(
|
|
448
|
+
widths: dict[str, list[float]],
|
|
449
|
+
) -> bool:
|
|
450
|
+
"""验证各字符宽度组的中位绝对偏差足够小,并校验中西文宽度关系。"""
|
|
451
|
+
|
|
452
|
+
populated = [values for values in widths.values() if values]
|
|
453
|
+
if not populated or sum(len(values) for values in populated) < 3:
|
|
454
|
+
return False
|
|
455
|
+
medians: dict[str, float] = {}
|
|
456
|
+
for group_name, values in widths.items():
|
|
457
|
+
if not values:
|
|
458
|
+
continue
|
|
459
|
+
median_width = statistics.median(values)
|
|
460
|
+
mad = statistics.median(abs(value - median_width) for value in values)
|
|
461
|
+
if mad / max(0.1, median_width) > 0.2:
|
|
462
|
+
return False
|
|
463
|
+
medians[group_name] = median_width
|
|
464
|
+
if "narrow" in medians and "wide" in medians:
|
|
465
|
+
ratio = medians["wide"] / medians["narrow"]
|
|
466
|
+
if not 1.25 <= ratio <= 2.25:
|
|
467
|
+
return False
|
|
468
|
+
return True
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def _code_rows_have_spatial_structure(
|
|
472
|
+
lines: list[_LineItem],
|
|
473
|
+
candidate_bbox: BBox,
|
|
474
|
+
median_cell_width: float,
|
|
475
|
+
) -> bool:
|
|
476
|
+
"""验证代码行具有规则基线,并且左缘落在一致的等宽字符槽。"""
|
|
477
|
+
|
|
478
|
+
rows: list[list[_LineItem]] = []
|
|
479
|
+
for line in sorted(lines, key=lambda item: (_bbox_center_y(item.bbox), item.bbox[0])):
|
|
480
|
+
target = next(
|
|
481
|
+
(
|
|
482
|
+
row
|
|
483
|
+
for row in rows
|
|
484
|
+
if any(
|
|
485
|
+
min(member.bbox[3], line.bbox[3]) - max(member.bbox[1], line.bbox[1])
|
|
486
|
+
>= 0.5
|
|
487
|
+
* min(
|
|
488
|
+
member.bbox[3] - member.bbox[1],
|
|
489
|
+
line.bbox[3] - line.bbox[1],
|
|
490
|
+
)
|
|
491
|
+
for member in row
|
|
492
|
+
)
|
|
493
|
+
),
|
|
494
|
+
None,
|
|
495
|
+
)
|
|
496
|
+
if target is None:
|
|
497
|
+
rows.append([line])
|
|
498
|
+
else:
|
|
499
|
+
target.append(line)
|
|
500
|
+
if not rows:
|
|
501
|
+
return False
|
|
502
|
+
|
|
503
|
+
residuals = []
|
|
504
|
+
for line in lines:
|
|
505
|
+
slot = (line.bbox[0] - candidate_bbox[0]) / max(0.1, median_cell_width)
|
|
506
|
+
residuals.append(abs(slot - round(slot)))
|
|
507
|
+
if sum(residual <= 0.35 for residual in residuals) / len(residuals) < 0.8:
|
|
508
|
+
return False
|
|
509
|
+
if len(rows) == 1:
|
|
510
|
+
return True
|
|
511
|
+
|
|
512
|
+
row_tops = sorted(min(line.bbox[1] for line in row) for row in rows)
|
|
513
|
+
deltas = [current - previous for previous, current in zip(row_tops, row_tops[1:]) if current > previous]
|
|
514
|
+
if not deltas:
|
|
515
|
+
return False
|
|
516
|
+
lower_count = max(1, math.ceil(0.6 * len(deltas)))
|
|
517
|
+
base_pitch = statistics.median(sorted(deltas)[:lower_count])
|
|
518
|
+
if base_pitch <= 0:
|
|
519
|
+
return False
|
|
520
|
+
return all(abs(delta / base_pitch - round(delta / base_pitch)) <= 0.35 for delta in deltas)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _fallback_code_content(lines: list[_LineItem]) -> str:
|
|
524
|
+
"""空间投影失败时按视觉行和水平位置保留代码文本的最小结构。"""
|
|
525
|
+
|
|
526
|
+
ordered = sorted(
|
|
527
|
+
lines,
|
|
528
|
+
key=lambda line: (
|
|
529
|
+
round(_bbox_center_y(line.bbox), 1),
|
|
530
|
+
_bbox_center_x(line.bbox),
|
|
531
|
+
line.source_index,
|
|
532
|
+
),
|
|
533
|
+
)
|
|
534
|
+
content = "\n".join(line.text for line in ordered if line.text)
|
|
535
|
+
return _sanitize_pdf_control_text(content, preserve_newlines=True).strip()
|