docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1138 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from ctypes import byref, c_int, create_string_buffer
|
|
6
|
+
from io import BytesIO
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import pypdfium2 as pdfium
|
|
10
|
+
import pypdfium2.raw as pdfium_c
|
|
11
|
+
from loguru import logger
|
|
12
|
+
from pypdf import PdfReader
|
|
13
|
+
from pypdf.generic import ContentStream
|
|
14
|
+
|
|
15
|
+
from .pdfium import PdfiumFontError, close_pdfium_child, pdfium_guard
|
|
16
|
+
|
|
17
|
+
MAX_SAMPLE_PAGES = 10
|
|
18
|
+
CHARS_THRESHOLD = 50
|
|
19
|
+
HIGH_IMAGE_COVERAGE_THRESHOLD = 0.8
|
|
20
|
+
TEXT_QUALITY_MIN_CHARS = 300
|
|
21
|
+
TEXT_QUALITY_BAD_THRESHOLD = 0.03
|
|
22
|
+
UNICODE_MAP_ERROR_RATIO_THRESHOLD = 0.04
|
|
23
|
+
CID_FONT_USAGE_RATIO_THRESHOLD = 0.01
|
|
24
|
+
CID_FONT_USAGE_COUNT_THRESHOLD = 30
|
|
25
|
+
LATIN_CJK_FONT_USAGE_RATIO_THRESHOLD = 0.01
|
|
26
|
+
LATIN_CJK_FONT_USAGE_COUNT_THRESHOLD = 30
|
|
27
|
+
LATIN_CJK_FONT_CJK_RATIO_THRESHOLD = 0.8
|
|
28
|
+
LATIN_CHARSET_MIN_LATIN_GLYPHS = 10
|
|
29
|
+
LATIN_CHARSET_MIN_LATIN_RATIO = 0.5
|
|
30
|
+
MAX_PAGE_ASPECT_RATIO = 10.0
|
|
31
|
+
SUSPICIOUS_CJK_72XX_START = 0x7280
|
|
32
|
+
SUSPICIOUS_CJK_72XX_END = 0x72DF
|
|
33
|
+
SUSPICIOUS_CJK_72XX_COUNT_THRESHOLD = 30
|
|
34
|
+
SUSPICIOUS_CJK_72XX_CJK_RATIO_THRESHOLD = 0.026
|
|
35
|
+
SUSPICIOUS_CJK_72XX_WHITELIST = set("犀犁犄犊犒犟犬犯状犷犹狂狄狈狐狗狙狞")
|
|
36
|
+
ASCII_PUNCT_CHARS = set("!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~")
|
|
37
|
+
ASCII_PUNCT_RUN_MIN_LENGTH = 4
|
|
38
|
+
SUSPICIOUS_ASCII_PUNCT_MIN_TEXT_CHARS = 100
|
|
39
|
+
SUSPICIOUS_ASCII_PUNCT_RATIO_THRESHOLD = 0.25
|
|
40
|
+
SUSPICIOUS_ASCII_PUNCT_RUN_RATIO_THRESHOLD = 0.10
|
|
41
|
+
SUSPICIOUS_CROSS_SCRIPT_MIN_TEXT_CHARS = 300
|
|
42
|
+
SUSPICIOUS_CROSS_SCRIPT_MIN_CJK_CHARS = 100
|
|
43
|
+
SUSPICIOUS_CROSS_SCRIPT_MIN_OTHER_SCRIPT_CHARS = 120
|
|
44
|
+
SUSPICIOUS_CROSS_SCRIPT_OTHER_SCRIPT_RATIO = 0.18
|
|
45
|
+
SUSPICIOUS_CROSS_SCRIPT_MIN_DENSE_SCRIPTS = 3
|
|
46
|
+
SUSPICIOUS_CROSS_SCRIPT_DENSE_SCRIPT_CHARS = 5
|
|
47
|
+
SUSPICIOUS_CROSS_SCRIPT_RANGES = (
|
|
48
|
+
(0x0370, 0x03FF, "Greek"),
|
|
49
|
+
(0x0400, 0x052F, "Cyrillic"),
|
|
50
|
+
(0x0600, 0x06FF, "Arabic"),
|
|
51
|
+
(0x0700, 0x074F, "Syriac"),
|
|
52
|
+
(0x0750, 0x077F, "Arabic Supplement"),
|
|
53
|
+
(0x0780, 0x07BF, "Thaana"),
|
|
54
|
+
(0x07C0, 0x07FF, "NKo"),
|
|
55
|
+
(0x0800, 0x083F, "Samaritan"),
|
|
56
|
+
(0x0840, 0x085F, "Mandaic"),
|
|
57
|
+
(0x0860, 0x086F, "Syriac Supplement"),
|
|
58
|
+
(0x0870, 0x089F, "Arabic Extended-B"),
|
|
59
|
+
(0x0900, 0x097F, "Devanagari"),
|
|
60
|
+
(0x0C80, 0x0CFF, "Kannada"),
|
|
61
|
+
(0x0E00, 0x0E7F, "Thai"),
|
|
62
|
+
(0x1000, 0x109F, "Myanmar"),
|
|
63
|
+
(0x1100, 0x11FF, "Hangul Jamo"),
|
|
64
|
+
(0x1200, 0x137F, "Ethiopic"),
|
|
65
|
+
(0x13A0, 0x13FF, "Cherokee"),
|
|
66
|
+
(0x1400, 0x167F, "Canadian Syllabics"),
|
|
67
|
+
(0x1800, 0x18AF, "Mongolian"),
|
|
68
|
+
(0x1A20, 0x1AAF, "Tai Tham"),
|
|
69
|
+
(0x2C00, 0x2C5F, "Glagolitic"),
|
|
70
|
+
(0xA000, 0xA48F, "Yi"),
|
|
71
|
+
)
|
|
72
|
+
CJK_TEXT_RANGES = (
|
|
73
|
+
(0x3400, 0x4DBF),
|
|
74
|
+
(0x4E00, 0x9FFF),
|
|
75
|
+
(0xF900, 0xFAFF),
|
|
76
|
+
(0x20000, 0x2EBEF),
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
_ALLOWED_CONTROL_CODES = {9, 10, 13}
|
|
80
|
+
_PRIVATE_USE_AREA_START = 0xE000
|
|
81
|
+
_PRIVATE_USE_AREA_END = 0xF8FF
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _is_disallowed_control_unicode(unicode_code: int) -> bool:
|
|
85
|
+
return (0 <= unicode_code < 32 or 127 <= unicode_code <= 159) and unicode_code not in _ALLOWED_CONTROL_CODES
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def classify(pdf_doc: pdfium.PdfDocument, pdf_bytes: bytes) -> str:
|
|
89
|
+
"""
|
|
90
|
+
Fast PDF classification path.
|
|
91
|
+
|
|
92
|
+
The path uses pdfium + pypdf to detect text PDFs and garbled PDFs.
|
|
93
|
+
|
|
94
|
+
Returns:
|
|
95
|
+
"txt" if the PDF can be parsed as text, otherwise "ocr".
|
|
96
|
+
"""
|
|
97
|
+
|
|
98
|
+
try:
|
|
99
|
+
with pdfium_guard():
|
|
100
|
+
page_count = len(pdf_doc)
|
|
101
|
+
if page_count == 0:
|
|
102
|
+
return "ocr"
|
|
103
|
+
|
|
104
|
+
page_indices = get_sample_page_indices(page_count, MAX_SAMPLE_PAGES)
|
|
105
|
+
if not page_indices:
|
|
106
|
+
return "ocr"
|
|
107
|
+
|
|
108
|
+
extreme_page_index, extreme_ratio = get_extreme_aspect_ratio_page_pdfium(
|
|
109
|
+
pdf_doc,
|
|
110
|
+
page_indices,
|
|
111
|
+
)
|
|
112
|
+
if extreme_page_index is not None:
|
|
113
|
+
logger.debug(
|
|
114
|
+
"Classify PDF as OCR due to extreme sampled-page aspect ratio: "
|
|
115
|
+
f"page={extreme_page_index + 1}, ratio={extreme_ratio:.2f}"
|
|
116
|
+
)
|
|
117
|
+
return "ocr"
|
|
118
|
+
|
|
119
|
+
text_samples = _collect_pdfium_text_samples(pdf_doc, page_indices)
|
|
120
|
+
avg_cleaned_chars_per_page = _get_avg_cleaned_chars_per_page_from_samples(text_samples)
|
|
121
|
+
if avg_cleaned_chars_per_page < CHARS_THRESHOLD:
|
|
122
|
+
return "ocr"
|
|
123
|
+
|
|
124
|
+
unicode_map_error_signal = _get_unicode_map_error_signal_from_samples(text_samples)
|
|
125
|
+
if unicode_map_error_signal["unicode_map_error_ratio"] >= UNICODE_MAP_ERROR_RATIO_THRESHOLD:
|
|
126
|
+
logger.debug(
|
|
127
|
+
"Classify PDF as OCR due to PDFium Unicode map errors: "
|
|
128
|
+
f"errors={unicode_map_error_signal['unicode_map_error_count']}, "
|
|
129
|
+
f"total={unicode_map_error_signal['total_chars']}, "
|
|
130
|
+
f"ratio={unicode_map_error_signal['unicode_map_error_ratio']:.4f}"
|
|
131
|
+
)
|
|
132
|
+
return "ocr"
|
|
133
|
+
|
|
134
|
+
font_resource_signals = _get_font_resource_signals_pypdf(
|
|
135
|
+
pdf_bytes,
|
|
136
|
+
page_indices,
|
|
137
|
+
)
|
|
138
|
+
cid_font_usage_signal = _get_cid_font_usage_signal_from_samples(
|
|
139
|
+
text_samples,
|
|
140
|
+
font_resource_signals["cid_without_to_unicode_usage"],
|
|
141
|
+
)
|
|
142
|
+
if cid_font_usage_signal["triggered"]:
|
|
143
|
+
logger.debug(
|
|
144
|
+
"Classify PDF as OCR due to high CID font usage without ToUnicode: "
|
|
145
|
+
f"page={cid_font_usage_signal['page_index'] + 1}, "
|
|
146
|
+
f"fonts={cid_font_usage_signal['font_names']}, "
|
|
147
|
+
f"chars={cid_font_usage_signal['cid_font_char_count']}, "
|
|
148
|
+
f"total={cid_font_usage_signal['total_chars']}, "
|
|
149
|
+
f"ratio={cid_font_usage_signal['cid_font_usage_ratio']:.4f}"
|
|
150
|
+
)
|
|
151
|
+
return "ocr"
|
|
152
|
+
|
|
153
|
+
latin_cjk_font_usage_signal = _get_latin_font_cjk_usage_signal_from_samples(
|
|
154
|
+
text_samples,
|
|
155
|
+
font_resource_signals["latin_charset_with_to_unicode"],
|
|
156
|
+
count_threshold=LATIN_CJK_FONT_USAGE_COUNT_THRESHOLD,
|
|
157
|
+
usage_ratio_threshold=LATIN_CJK_FONT_USAGE_RATIO_THRESHOLD,
|
|
158
|
+
cjk_ratio_threshold=LATIN_CJK_FONT_CJK_RATIO_THRESHOLD,
|
|
159
|
+
)
|
|
160
|
+
if latin_cjk_font_usage_signal["triggered"]:
|
|
161
|
+
logger.debug(
|
|
162
|
+
"Classify PDF as OCR due to Latin CharSet font decoding as CJK: "
|
|
163
|
+
f"page={latin_cjk_font_usage_signal['page_index'] + 1}, "
|
|
164
|
+
f"fonts={latin_cjk_font_usage_signal['font_names']}, "
|
|
165
|
+
f"chars={latin_cjk_font_usage_signal['font_char_count']}, "
|
|
166
|
+
f"cjk={latin_cjk_font_usage_signal['cjk_char_count']}, "
|
|
167
|
+
f"total={latin_cjk_font_usage_signal['total_chars']}, "
|
|
168
|
+
f"usage_ratio={latin_cjk_font_usage_signal['font_usage_ratio']:.4f}, "
|
|
169
|
+
f"cjk_ratio={latin_cjk_font_usage_signal['font_cjk_ratio']:.4f}"
|
|
170
|
+
)
|
|
171
|
+
return "ocr"
|
|
172
|
+
|
|
173
|
+
text_quality_signal = _get_text_quality_signal_from_samples(text_samples)
|
|
174
|
+
total_chars = text_quality_signal["total_chars"]
|
|
175
|
+
abnormal_ratio = text_quality_signal["abnormal_ratio"]
|
|
176
|
+
|
|
177
|
+
if total_chars >= TEXT_QUALITY_MIN_CHARS and abnormal_ratio >= TEXT_QUALITY_BAD_THRESHOLD:
|
|
178
|
+
return "ocr"
|
|
179
|
+
|
|
180
|
+
u72xx_signal = _get_u72xx_text_signal_from_samples(text_samples)
|
|
181
|
+
if (
|
|
182
|
+
u72xx_signal["u72xx_count"] >= SUSPICIOUS_CJK_72XX_COUNT_THRESHOLD
|
|
183
|
+
and u72xx_signal["u72xx_cjk_ratio"] >= SUSPICIOUS_CJK_72XX_CJK_RATIO_THRESHOLD
|
|
184
|
+
):
|
|
185
|
+
logger.debug(
|
|
186
|
+
"Classify PDF as OCR due to suspicious U+7280-U+72DF text: "
|
|
187
|
+
f"count={u72xx_signal['u72xx_count']}, "
|
|
188
|
+
f"cjk_ratio={u72xx_signal['u72xx_cjk_ratio']:.4f}"
|
|
189
|
+
)
|
|
190
|
+
return "ocr"
|
|
191
|
+
|
|
192
|
+
cross_script_signal = _get_cross_script_text_signal_from_samples(text_samples)
|
|
193
|
+
if cross_script_signal["triggered"]:
|
|
194
|
+
logger.debug(
|
|
195
|
+
"Classify PDF as OCR due to suspicious cross-script text: "
|
|
196
|
+
f"chars={cross_script_signal['total_chars']}, "
|
|
197
|
+
f"cjk={cross_script_signal['cjk_chars']}, "
|
|
198
|
+
f"suspicious={cross_script_signal['suspicious_chars']}, "
|
|
199
|
+
f"ratio={cross_script_signal['suspicious_ratio']:.4f}, "
|
|
200
|
+
f"scripts={cross_script_signal['top_scripts']}"
|
|
201
|
+
)
|
|
202
|
+
return "ocr"
|
|
203
|
+
|
|
204
|
+
ascii_punct_signal = _get_sampled_ascii_punct_signal_from_samples(text_samples)
|
|
205
|
+
if ascii_punct_signal["triggered"]:
|
|
206
|
+
logger.debug(
|
|
207
|
+
"Classify PDF as OCR due to suspicious sampled-page ASCII punctuation "
|
|
208
|
+
f"text: page={ascii_punct_signal['page_index'] + 1}, "
|
|
209
|
+
f"text_chars={ascii_punct_signal['cleaned_text_chars']}, "
|
|
210
|
+
f"ascii_punct_ratio="
|
|
211
|
+
f"{ascii_punct_signal['ascii_punct_ratio']:.4f}, "
|
|
212
|
+
f"punct_run_ratio={ascii_punct_signal['punct_run_ratio']:.4f}"
|
|
213
|
+
)
|
|
214
|
+
return "ocr"
|
|
215
|
+
|
|
216
|
+
if get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices) >= HIGH_IMAGE_COVERAGE_THRESHOLD:
|
|
217
|
+
return "ocr"
|
|
218
|
+
|
|
219
|
+
except PdfiumFontError:
|
|
220
|
+
raise
|
|
221
|
+
except Exception as e:
|
|
222
|
+
logger.error(f"Failed to classify PDF: {e}")
|
|
223
|
+
return "ocr"
|
|
224
|
+
|
|
225
|
+
return "txt"
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def get_sample_page_indices(page_count: int, max_pages: int = MAX_SAMPLE_PAGES) -> list[int]:
|
|
229
|
+
if page_count <= 0 or max_pages <= 0:
|
|
230
|
+
return []
|
|
231
|
+
|
|
232
|
+
sample_count = min(page_count, max_pages)
|
|
233
|
+
if sample_count == page_count:
|
|
234
|
+
return list(range(page_count))
|
|
235
|
+
if sample_count == 1:
|
|
236
|
+
return [0]
|
|
237
|
+
|
|
238
|
+
indices = []
|
|
239
|
+
seen = set()
|
|
240
|
+
for i in range(sample_count):
|
|
241
|
+
page_index = round(i * (page_count - 1) / (sample_count - 1))
|
|
242
|
+
page_index = max(0, min(page_count - 1, page_index))
|
|
243
|
+
if page_index not in seen:
|
|
244
|
+
indices.append(page_index)
|
|
245
|
+
seen.add(page_index)
|
|
246
|
+
|
|
247
|
+
if len(indices) < sample_count:
|
|
248
|
+
for page_index in range(page_count):
|
|
249
|
+
if page_index in seen:
|
|
250
|
+
continue
|
|
251
|
+
indices.append(page_index)
|
|
252
|
+
seen.add(page_index)
|
|
253
|
+
if len(indices) == sample_count:
|
|
254
|
+
break
|
|
255
|
+
|
|
256
|
+
return sorted(indices)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def get_extreme_aspect_ratio_page_pdfium(
|
|
260
|
+
pdf_doc: pdfium.PdfDocument,
|
|
261
|
+
page_indices: list[int],
|
|
262
|
+
max_page_aspect_ratio: float = MAX_PAGE_ASPECT_RATIO,
|
|
263
|
+
) -> tuple[Any, Any]:
|
|
264
|
+
with pdfium_guard():
|
|
265
|
+
for page_index in page_indices:
|
|
266
|
+
page = None
|
|
267
|
+
try:
|
|
268
|
+
page = pdf_doc[page_index]
|
|
269
|
+
page_width, page_height = page.get_size()
|
|
270
|
+
if page_width <= 0 or page_height <= 0:
|
|
271
|
+
continue
|
|
272
|
+
|
|
273
|
+
aspect_ratio = max(page_width / page_height, page_height / page_width)
|
|
274
|
+
if aspect_ratio > max_page_aspect_ratio:
|
|
275
|
+
return page_index, aspect_ratio
|
|
276
|
+
finally:
|
|
277
|
+
close_pdfium_child(page)
|
|
278
|
+
|
|
279
|
+
return None, None
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _collect_pdfium_text_sample_from_page(page_index: int, page: Any) -> dict[str, Any]:
|
|
283
|
+
"""从单页 PDFium 对象提取纯 Python 文本统计,并在调用方释放子对象。"""
|
|
284
|
+
text_page = None
|
|
285
|
+
try:
|
|
286
|
+
text_page = page.get_textpage()
|
|
287
|
+
text = text_page.get_text_bounded()
|
|
288
|
+
char_count = text_page.count_chars()
|
|
289
|
+
null_char_count = 0
|
|
290
|
+
replacement_char_count = 0
|
|
291
|
+
control_char_count = 0
|
|
292
|
+
private_use_char_count = 0
|
|
293
|
+
unicode_map_error_count = 0
|
|
294
|
+
font_name_counts = {}
|
|
295
|
+
non_generated_char_count = 0
|
|
296
|
+
font_non_generated_char_counts = {}
|
|
297
|
+
font_non_generated_cjk_char_counts = {}
|
|
298
|
+
|
|
299
|
+
for char_index in range(char_count):
|
|
300
|
+
unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
|
|
301
|
+
is_generated = pdfium_c.FPDFText_IsGenerated(text_page, char_index) == 1
|
|
302
|
+
if not is_generated:
|
|
303
|
+
non_generated_char_count += 1
|
|
304
|
+
|
|
305
|
+
if unicode_code == 0:
|
|
306
|
+
null_char_count += 1
|
|
307
|
+
elif unicode_code == 0xFFFD:
|
|
308
|
+
replacement_char_count += 1
|
|
309
|
+
elif _is_disallowed_control_unicode(unicode_code):
|
|
310
|
+
control_char_count += 1
|
|
311
|
+
elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END:
|
|
312
|
+
private_use_char_count += 1
|
|
313
|
+
|
|
314
|
+
if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index):
|
|
315
|
+
unicode_map_error_count += 1
|
|
316
|
+
|
|
317
|
+
font_name = _normalize_pdf_font_name(_get_pdfium_char_font_name(text_page, char_index))
|
|
318
|
+
if font_name:
|
|
319
|
+
font_name_counts[font_name] = font_name_counts.get(font_name, 0) + 1
|
|
320
|
+
if not is_generated:
|
|
321
|
+
font_non_generated_char_counts[font_name] = font_non_generated_char_counts.get(font_name, 0) + 1
|
|
322
|
+
if _is_cjk_unicode_code(unicode_code):
|
|
323
|
+
font_non_generated_cjk_char_counts[font_name] = font_non_generated_cjk_char_counts.get(font_name, 0) + 1
|
|
324
|
+
|
|
325
|
+
return {
|
|
326
|
+
"page_index": page_index,
|
|
327
|
+
"text": text,
|
|
328
|
+
"cleaned_text": re.sub(r"\s+", "", text),
|
|
329
|
+
"char_count": char_count,
|
|
330
|
+
"null_char_count": null_char_count,
|
|
331
|
+
"replacement_char_count": replacement_char_count,
|
|
332
|
+
"control_char_count": control_char_count,
|
|
333
|
+
"private_use_char_count": private_use_char_count,
|
|
334
|
+
"unicode_map_error_count": unicode_map_error_count,
|
|
335
|
+
"font_name_counts": font_name_counts,
|
|
336
|
+
"non_generated_char_count": non_generated_char_count,
|
|
337
|
+
"font_non_generated_char_counts": font_non_generated_char_counts,
|
|
338
|
+
"font_non_generated_cjk_char_counts": font_non_generated_cjk_char_counts,
|
|
339
|
+
}
|
|
340
|
+
finally:
|
|
341
|
+
close_pdfium_child(text_page)
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _collect_pdfium_text_samples(pdf_doc: pdfium.PdfDocument, page_indices: list[int]) -> list[dict[str, Any]]:
|
|
345
|
+
"""一次性收集抽样页文本统计,返回纯 Python 数据,避免缓存 PDFium 子对象。"""
|
|
346
|
+
text_samples = []
|
|
347
|
+
|
|
348
|
+
with pdfium_guard():
|
|
349
|
+
for page_index in page_indices:
|
|
350
|
+
page = None
|
|
351
|
+
try:
|
|
352
|
+
page = pdf_doc[page_index]
|
|
353
|
+
text_samples.append(_collect_pdfium_text_sample_from_page(page_index, page))
|
|
354
|
+
finally:
|
|
355
|
+
close_pdfium_child(page)
|
|
356
|
+
|
|
357
|
+
return text_samples
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def _get_avg_cleaned_chars_per_page_from_samples(text_samples: list[dict[str, Any]]) -> float:
|
|
361
|
+
"""基于已缓存的抽样页文本计算平均有效字符数。"""
|
|
362
|
+
cleaned_total_chars = 0
|
|
363
|
+
|
|
364
|
+
for text_sample in text_samples:
|
|
365
|
+
cleaned_total_chars += len(text_sample["cleaned_text"])
|
|
366
|
+
|
|
367
|
+
if not text_samples:
|
|
368
|
+
return 0.0
|
|
369
|
+
return cleaned_total_chars / len(text_samples)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _get_text_quality_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
|
|
373
|
+
"""基于已缓存的抽样页字符计数统计异常字符质量信号。"""
|
|
374
|
+
total_chars = 0
|
|
375
|
+
null_char_count = 0
|
|
376
|
+
replacement_char_count = 0
|
|
377
|
+
control_char_count = 0
|
|
378
|
+
private_use_char_count = 0
|
|
379
|
+
|
|
380
|
+
for text_sample in text_samples:
|
|
381
|
+
total_chars += text_sample["char_count"]
|
|
382
|
+
null_char_count += text_sample["null_char_count"]
|
|
383
|
+
replacement_char_count += text_sample["replacement_char_count"]
|
|
384
|
+
control_char_count += text_sample["control_char_count"]
|
|
385
|
+
private_use_char_count += text_sample["private_use_char_count"]
|
|
386
|
+
|
|
387
|
+
abnormal_chars = null_char_count + replacement_char_count + control_char_count + private_use_char_count
|
|
388
|
+
|
|
389
|
+
abnormal_ratio = 0.0
|
|
390
|
+
if total_chars > 0:
|
|
391
|
+
abnormal_ratio = abnormal_chars / total_chars
|
|
392
|
+
|
|
393
|
+
return {
|
|
394
|
+
"total_chars": total_chars,
|
|
395
|
+
"abnormal_ratio": abnormal_ratio,
|
|
396
|
+
"null_char_count": null_char_count,
|
|
397
|
+
"replacement_char_count": replacement_char_count,
|
|
398
|
+
"control_char_count": control_char_count,
|
|
399
|
+
"private_use_char_count": private_use_char_count,
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _get_unicode_map_error_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
|
|
404
|
+
"""统计 PDFium 字符级 Unicode 映射失败比例,用于识别无法可靠抽取的乱码文本。"""
|
|
405
|
+
total_chars = 0
|
|
406
|
+
unicode_map_error_count = 0
|
|
407
|
+
|
|
408
|
+
for text_sample in text_samples:
|
|
409
|
+
total_chars += text_sample["char_count"]
|
|
410
|
+
unicode_map_error_count += text_sample["unicode_map_error_count"]
|
|
411
|
+
|
|
412
|
+
unicode_map_error_ratio = 0.0
|
|
413
|
+
if total_chars > 0:
|
|
414
|
+
unicode_map_error_ratio = unicode_map_error_count / total_chars
|
|
415
|
+
|
|
416
|
+
return {
|
|
417
|
+
"total_chars": total_chars,
|
|
418
|
+
"unicode_map_error_count": unicode_map_error_count,
|
|
419
|
+
"unicode_map_error_ratio": unicode_map_error_ratio,
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _is_cjk_unicode_code(unicode_code: int) -> bool:
|
|
424
|
+
"""判断 Unicode 码点是否属于分类器认可的 CJK 文本范围。"""
|
|
425
|
+
return any(start <= unicode_code <= end for start, end in CJK_TEXT_RANGES)
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _get_cjk_glyph_name_code(glyph_name: str) -> int | None:
|
|
429
|
+
"""解析 uniXXXX/uXXXXX 形式的 CJK glyph name,其他名称返回 None。"""
|
|
430
|
+
match = re.fullmatch(
|
|
431
|
+
r"(?:uni([0-9A-Fa-f]{4,6})|u([0-9A-Fa-f]{4,6}))",
|
|
432
|
+
glyph_name,
|
|
433
|
+
)
|
|
434
|
+
if match is None:
|
|
435
|
+
return None
|
|
436
|
+
unicode_code = int(match.group(1) or match.group(2), 16)
|
|
437
|
+
if not _is_cjk_unicode_code(unicode_code):
|
|
438
|
+
return None
|
|
439
|
+
return unicode_code
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _get_empty_latin_charset_with_to_unicode_signal() -> dict[str, Any]:
|
|
443
|
+
"""构造未触发的 Type1 Latin CharSet 字体候选信号。"""
|
|
444
|
+
return {
|
|
445
|
+
"triggered": False,
|
|
446
|
+
"charset_glyph_count": 0,
|
|
447
|
+
"latin_glyph_count": 0,
|
|
448
|
+
"latin_glyph_ratio": 0.0,
|
|
449
|
+
"cjk_charset_glyph_count": 0,
|
|
450
|
+
"cjk_charset_glyph_ratio": 0.0,
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _get_latin_charset_with_to_unicode_signal(font: Any) -> dict[str, Any]:
|
|
455
|
+
"""识别带 ToUnicode 且 CharSet 明显为 Latin 的 Type1 字体候选。"""
|
|
456
|
+
signal = _get_empty_latin_charset_with_to_unicode_signal()
|
|
457
|
+
if str(font.get("/Subtype")) != "/Type1":
|
|
458
|
+
return signal
|
|
459
|
+
|
|
460
|
+
descriptor = _resolve_pdf_object(font.get("/FontDescriptor"))
|
|
461
|
+
to_unicode = _resolve_pdf_object(font.get("/ToUnicode"))
|
|
462
|
+
if descriptor is None or to_unicode is None:
|
|
463
|
+
return signal
|
|
464
|
+
|
|
465
|
+
charset = descriptor.get("/CharSet")
|
|
466
|
+
if charset is None:
|
|
467
|
+
return signal
|
|
468
|
+
|
|
469
|
+
glyph_names = set(re.findall(r"/([^/\s]+)", str(charset)))
|
|
470
|
+
charset_glyph_count = len(glyph_names)
|
|
471
|
+
latin_glyph_count = sum(1 for glyph_name in glyph_names if re.fullmatch(r"[A-Za-z]", glyph_name))
|
|
472
|
+
cjk_charset_glyph_count = sum(1 for glyph_name in glyph_names if _get_cjk_glyph_name_code(glyph_name) is not None)
|
|
473
|
+
latin_glyph_ratio = latin_glyph_count / charset_glyph_count if charset_glyph_count else 0.0
|
|
474
|
+
cjk_charset_glyph_ratio = cjk_charset_glyph_count / charset_glyph_count if charset_glyph_count else 0.0
|
|
475
|
+
|
|
476
|
+
signal.update(
|
|
477
|
+
{
|
|
478
|
+
"charset_glyph_count": charset_glyph_count,
|
|
479
|
+
"latin_glyph_count": latin_glyph_count,
|
|
480
|
+
"latin_glyph_ratio": latin_glyph_ratio,
|
|
481
|
+
"cjk_charset_glyph_count": cjk_charset_glyph_count,
|
|
482
|
+
"cjk_charset_glyph_ratio": cjk_charset_glyph_ratio,
|
|
483
|
+
}
|
|
484
|
+
)
|
|
485
|
+
signal["triggered"] = (
|
|
486
|
+
latin_glyph_count >= LATIN_CHARSET_MIN_LATIN_GLYPHS
|
|
487
|
+
and latin_glyph_ratio >= LATIN_CHARSET_MIN_LATIN_RATIO
|
|
488
|
+
and cjk_charset_glyph_count == 0
|
|
489
|
+
)
|
|
490
|
+
return signal
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def _normalize_pdf_font_name(font_name: Any) -> str:
|
|
494
|
+
"""规范化 PDF 字体名,统一 pypdf 的 NameObject 和 PDFium 返回值格式。"""
|
|
495
|
+
if font_name is None:
|
|
496
|
+
return ""
|
|
497
|
+
normalized_name = str(font_name).strip().lstrip("/")
|
|
498
|
+
return re.sub(r"^[A-Z]{6}\+", "", normalized_name, count=1)
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def _get_pdfium_char_font_name(text_page: Any, char_index: int) -> str:
|
|
502
|
+
"""读取 PDFium 字符级字体名,用于统计可疑 CID 字体的实际使用比例。"""
|
|
503
|
+
flags = c_int()
|
|
504
|
+
buffer_length = pdfium_c.FPDFText_GetFontInfo(
|
|
505
|
+
text_page,
|
|
506
|
+
char_index,
|
|
507
|
+
None,
|
|
508
|
+
0,
|
|
509
|
+
byref(flags),
|
|
510
|
+
)
|
|
511
|
+
if buffer_length <= 0:
|
|
512
|
+
return ""
|
|
513
|
+
|
|
514
|
+
font_name_buffer = create_string_buffer(buffer_length)
|
|
515
|
+
actual_length = pdfium_c.FPDFText_GetFontInfo(
|
|
516
|
+
text_page,
|
|
517
|
+
char_index,
|
|
518
|
+
font_name_buffer,
|
|
519
|
+
buffer_length,
|
|
520
|
+
byref(flags),
|
|
521
|
+
)
|
|
522
|
+
if actual_length <= 0:
|
|
523
|
+
return ""
|
|
524
|
+
|
|
525
|
+
return font_name_buffer.value.decode("utf-8", errors="ignore")
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _get_cid_font_usage_signal_from_samples(
|
|
529
|
+
text_samples: list[dict[str, Any]], cid_font_usage: dict[int, dict[str, Any]]
|
|
530
|
+
) -> dict[str, Any]:
|
|
531
|
+
"""结合内容流精确计数与 PDFium 总字符数计算可疑 CID 字体使用比例。"""
|
|
532
|
+
best_signal = {
|
|
533
|
+
"triggered": False,
|
|
534
|
+
"page_index": None,
|
|
535
|
+
"font_names": [],
|
|
536
|
+
"cid_font_char_count": 0,
|
|
537
|
+
"total_chars": 0,
|
|
538
|
+
"cid_font_usage_ratio": 0.0,
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
for text_sample in text_samples:
|
|
542
|
+
page_index = text_sample.get("page_index")
|
|
543
|
+
total_chars = text_sample["char_count"]
|
|
544
|
+
if total_chars <= 0:
|
|
545
|
+
continue
|
|
546
|
+
|
|
547
|
+
page_usage = cid_font_usage.get(page_index) or {}
|
|
548
|
+
cid_font_char_count = int(page_usage.get("cid_font_char_count", 0))
|
|
549
|
+
matched_font_names = sorted(page_usage.get("font_names") or [])
|
|
550
|
+
|
|
551
|
+
cid_font_usage_ratio = cid_font_char_count / total_chars
|
|
552
|
+
signal = {
|
|
553
|
+
"triggered": False,
|
|
554
|
+
"page_index": page_index,
|
|
555
|
+
"font_names": matched_font_names,
|
|
556
|
+
"cid_font_char_count": cid_font_char_count,
|
|
557
|
+
"total_chars": total_chars,
|
|
558
|
+
"cid_font_usage_ratio": cid_font_usage_ratio,
|
|
559
|
+
}
|
|
560
|
+
if cid_font_char_count >= CID_FONT_USAGE_COUNT_THRESHOLD and cid_font_usage_ratio >= CID_FONT_USAGE_RATIO_THRESHOLD:
|
|
561
|
+
signal["triggered"] = True
|
|
562
|
+
return signal
|
|
563
|
+
|
|
564
|
+
if (
|
|
565
|
+
signal["cid_font_usage_ratio"],
|
|
566
|
+
signal["cid_font_char_count"],
|
|
567
|
+
) > (
|
|
568
|
+
best_signal["cid_font_usage_ratio"],
|
|
569
|
+
best_signal["cid_font_char_count"],
|
|
570
|
+
):
|
|
571
|
+
best_signal = signal
|
|
572
|
+
|
|
573
|
+
return best_signal
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _get_latin_font_cjk_usage_signal_from_samples(
|
|
577
|
+
text_samples: list[dict[str, Any]],
|
|
578
|
+
font_signal: dict[str, Any],
|
|
579
|
+
count_threshold: int,
|
|
580
|
+
usage_ratio_threshold: float,
|
|
581
|
+
cjk_ratio_threshold: float,
|
|
582
|
+
) -> dict[str, Any]:
|
|
583
|
+
"""按单个 Latin 候选字体统计实际使用量及 PDFium 解码后的 CJK 比例。"""
|
|
584
|
+
best_signal = {
|
|
585
|
+
"triggered": False,
|
|
586
|
+
"page_index": None,
|
|
587
|
+
"font_names": [],
|
|
588
|
+
"font_char_count": 0,
|
|
589
|
+
"cjk_char_count": 0,
|
|
590
|
+
"total_chars": 0,
|
|
591
|
+
"font_usage_ratio": 0.0,
|
|
592
|
+
"font_cjk_ratio": 0.0,
|
|
593
|
+
}
|
|
594
|
+
if not font_signal or not font_signal.get("triggered"):
|
|
595
|
+
return best_signal
|
|
596
|
+
|
|
597
|
+
page_fonts = font_signal.get("page_fonts") or {}
|
|
598
|
+
for text_sample in text_samples:
|
|
599
|
+
page_index = text_sample.get("page_index")
|
|
600
|
+
total_chars = text_sample.get("non_generated_char_count", 0)
|
|
601
|
+
if total_chars <= 0:
|
|
602
|
+
continue
|
|
603
|
+
|
|
604
|
+
font_name_counts = text_sample.get("font_non_generated_char_counts") or {}
|
|
605
|
+
font_cjk_char_counts = text_sample.get("font_non_generated_cjk_char_counts") or {}
|
|
606
|
+
candidate_font_names = {_normalize_pdf_font_name(font_name) for font_name in page_fonts.get(page_index, set())}
|
|
607
|
+
candidate_font_names.discard("")
|
|
608
|
+
|
|
609
|
+
for font_name in sorted(candidate_font_names):
|
|
610
|
+
font_char_count = font_name_counts.get(font_name, 0)
|
|
611
|
+
cjk_char_count = font_cjk_char_counts.get(font_name, 0)
|
|
612
|
+
font_usage_ratio = font_char_count / total_chars
|
|
613
|
+
font_cjk_ratio = cjk_char_count / font_char_count if font_char_count else 0.0
|
|
614
|
+
signal = {
|
|
615
|
+
"triggered": False,
|
|
616
|
+
"page_index": page_index,
|
|
617
|
+
"font_names": [font_name] if font_char_count else [],
|
|
618
|
+
"font_char_count": font_char_count,
|
|
619
|
+
"cjk_char_count": cjk_char_count,
|
|
620
|
+
"total_chars": total_chars,
|
|
621
|
+
"font_usage_ratio": font_usage_ratio,
|
|
622
|
+
"font_cjk_ratio": font_cjk_ratio,
|
|
623
|
+
}
|
|
624
|
+
if (
|
|
625
|
+
font_char_count >= count_threshold
|
|
626
|
+
and font_usage_ratio >= usage_ratio_threshold
|
|
627
|
+
and font_cjk_ratio >= cjk_ratio_threshold
|
|
628
|
+
):
|
|
629
|
+
signal["triggered"] = True
|
|
630
|
+
return signal
|
|
631
|
+
|
|
632
|
+
if (
|
|
633
|
+
signal["font_cjk_ratio"],
|
|
634
|
+
signal["font_usage_ratio"],
|
|
635
|
+
signal["font_char_count"],
|
|
636
|
+
) > (
|
|
637
|
+
best_signal["font_cjk_ratio"],
|
|
638
|
+
best_signal["font_usage_ratio"],
|
|
639
|
+
best_signal["font_char_count"],
|
|
640
|
+
):
|
|
641
|
+
best_signal = signal
|
|
642
|
+
|
|
643
|
+
return best_signal
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
def _get_u72xx_text_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
|
|
647
|
+
"""基于已缓存的抽样页文本统计扣除常用字后的 U+7280-U+72DF 字符占比。"""
|
|
648
|
+
cjk_chars = 0
|
|
649
|
+
u72xx_count = 0
|
|
650
|
+
|
|
651
|
+
for text_sample in text_samples:
|
|
652
|
+
for char in text_sample["cleaned_text"]:
|
|
653
|
+
unicode_code = ord(char)
|
|
654
|
+
if 0x4E00 <= unicode_code <= 0x9FFF:
|
|
655
|
+
cjk_chars += 1
|
|
656
|
+
if (
|
|
657
|
+
SUSPICIOUS_CJK_72XX_START <= unicode_code <= SUSPICIOUS_CJK_72XX_END
|
|
658
|
+
and char not in SUSPICIOUS_CJK_72XX_WHITELIST
|
|
659
|
+
):
|
|
660
|
+
u72xx_count += 1
|
|
661
|
+
|
|
662
|
+
u72xx_cjk_ratio = 0.0
|
|
663
|
+
if cjk_chars > 0:
|
|
664
|
+
u72xx_cjk_ratio = u72xx_count / cjk_chars
|
|
665
|
+
|
|
666
|
+
return {
|
|
667
|
+
"cjk_chars": cjk_chars,
|
|
668
|
+
"u72xx_count": u72xx_count,
|
|
669
|
+
"u72xx_cjk_ratio": u72xx_cjk_ratio,
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
def _get_sample_cleaned_text(text_sample: Any) -> str:
|
|
674
|
+
"""兼容 dict 和测试替身对象,读取抽样页的 cleaned_text 字段。"""
|
|
675
|
+
if isinstance(text_sample, dict):
|
|
676
|
+
return str(text_sample.get("cleaned_text", ""))
|
|
677
|
+
return str(getattr(text_sample, "cleaned_text", ""))
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
def _is_cjk_text_char(char: str) -> bool:
|
|
681
|
+
"""判断字符是否属于中文文档中可接受的 CJK 文字范围。"""
|
|
682
|
+
return _is_cjk_unicode_code(ord(char))
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
def _get_cross_script_name(char: str) -> str | None:
|
|
686
|
+
"""识别中文文档乱码中常见的跨脚本字符块名称。"""
|
|
687
|
+
unicode_code = ord(char)
|
|
688
|
+
for start, end, script_name in SUSPICIOUS_CROSS_SCRIPT_RANGES:
|
|
689
|
+
if start <= unicode_code <= end:
|
|
690
|
+
return script_name
|
|
691
|
+
return None
|
|
692
|
+
|
|
693
|
+
|
|
694
|
+
def _get_cross_script_text_signal_from_samples(text_samples: list[Any]) -> dict[str, Any]:
|
|
695
|
+
"""统计中文文档文本层中大比例跨脚本混入信号,用于识别合法 Unicode 错码。"""
|
|
696
|
+
total_chars = 0
|
|
697
|
+
cjk_chars = 0
|
|
698
|
+
suspicious_chars = 0
|
|
699
|
+
script_counts: dict[str, int] = {}
|
|
700
|
+
|
|
701
|
+
for text_sample in text_samples:
|
|
702
|
+
for char in _get_sample_cleaned_text(text_sample):
|
|
703
|
+
total_chars += 1
|
|
704
|
+
if _is_cjk_text_char(char):
|
|
705
|
+
cjk_chars += 1
|
|
706
|
+
|
|
707
|
+
script_name = _get_cross_script_name(char)
|
|
708
|
+
if script_name is None:
|
|
709
|
+
continue
|
|
710
|
+
|
|
711
|
+
suspicious_chars += 1
|
|
712
|
+
script_counts[script_name] = script_counts.get(script_name, 0) + 1
|
|
713
|
+
|
|
714
|
+
suspicious_ratio = 0.0
|
|
715
|
+
if total_chars > 0:
|
|
716
|
+
suspicious_ratio = suspicious_chars / total_chars
|
|
717
|
+
dense_script_count = sum(1 for count in script_counts.values() if count >= SUSPICIOUS_CROSS_SCRIPT_DENSE_SCRIPT_CHARS)
|
|
718
|
+
top_scripts = sorted(
|
|
719
|
+
script_counts.items(),
|
|
720
|
+
key=lambda item: (-item[1], item[0]),
|
|
721
|
+
)[:5]
|
|
722
|
+
triggered = (
|
|
723
|
+
total_chars >= SUSPICIOUS_CROSS_SCRIPT_MIN_TEXT_CHARS
|
|
724
|
+
and cjk_chars >= SUSPICIOUS_CROSS_SCRIPT_MIN_CJK_CHARS
|
|
725
|
+
and suspicious_chars >= SUSPICIOUS_CROSS_SCRIPT_MIN_OTHER_SCRIPT_CHARS
|
|
726
|
+
and suspicious_ratio >= SUSPICIOUS_CROSS_SCRIPT_OTHER_SCRIPT_RATIO
|
|
727
|
+
and dense_script_count >= SUSPICIOUS_CROSS_SCRIPT_MIN_DENSE_SCRIPTS
|
|
728
|
+
)
|
|
729
|
+
|
|
730
|
+
return {
|
|
731
|
+
"triggered": triggered,
|
|
732
|
+
"total_chars": total_chars,
|
|
733
|
+
"cjk_chars": cjk_chars,
|
|
734
|
+
"suspicious_chars": suspicious_chars,
|
|
735
|
+
"suspicious_ratio": suspicious_ratio,
|
|
736
|
+
"script_counts": script_counts,
|
|
737
|
+
"top_scripts": top_scripts,
|
|
738
|
+
"dense_script_count": dense_script_count,
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def _count_ascii_punct_run_chars(text: str) -> int:
|
|
743
|
+
"""统计连续 ASCII 标点字符数,仅累计长度达到阈值的 run。"""
|
|
744
|
+
run_chars = 0
|
|
745
|
+
current_run = 0
|
|
746
|
+
current_run_types: set[str] = set()
|
|
747
|
+
|
|
748
|
+
for char in text:
|
|
749
|
+
if char in ASCII_PUNCT_CHARS:
|
|
750
|
+
current_run += 1
|
|
751
|
+
current_run_types.add(char)
|
|
752
|
+
continue
|
|
753
|
+
|
|
754
|
+
if current_run >= ASCII_PUNCT_RUN_MIN_LENGTH and len(current_run_types) >= 2:
|
|
755
|
+
run_chars += current_run
|
|
756
|
+
current_run = 0
|
|
757
|
+
current_run_types.clear()
|
|
758
|
+
|
|
759
|
+
if current_run >= ASCII_PUNCT_RUN_MIN_LENGTH and len(current_run_types) >= 2:
|
|
760
|
+
run_chars += current_run
|
|
761
|
+
|
|
762
|
+
return run_chars
|
|
763
|
+
|
|
764
|
+
|
|
765
|
+
def _get_sampled_ascii_punct_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
|
|
766
|
+
"""检查所有抽样页的 ASCII 标点密集度,用于识别无 ToUnicode 的乱码文本。"""
|
|
767
|
+
best_signal = {
|
|
768
|
+
"triggered": False,
|
|
769
|
+
"page_index": None,
|
|
770
|
+
"cleaned_text_chars": 0,
|
|
771
|
+
"ascii_punct_count": 0,
|
|
772
|
+
"ascii_punct_ratio": 0.0,
|
|
773
|
+
"ascii_punct_run_chars": 0,
|
|
774
|
+
"punct_run_ratio": 0.0,
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
for text_sample in text_samples:
|
|
778
|
+
page_index = text_sample.get("page_index")
|
|
779
|
+
cleaned_text = text_sample["cleaned_text"]
|
|
780
|
+
cleaned_text_chars = len(cleaned_text)
|
|
781
|
+
ascii_punct_count = sum(1 for char in cleaned_text if char in ASCII_PUNCT_CHARS)
|
|
782
|
+
ascii_punct_run_chars = _count_ascii_punct_run_chars(cleaned_text)
|
|
783
|
+
|
|
784
|
+
ascii_punct_ratio = 0.0
|
|
785
|
+
punct_run_ratio = 0.0
|
|
786
|
+
if cleaned_text_chars > 0:
|
|
787
|
+
ascii_punct_ratio = ascii_punct_count / cleaned_text_chars
|
|
788
|
+
punct_run_ratio = ascii_punct_run_chars / cleaned_text_chars
|
|
789
|
+
|
|
790
|
+
signal = {
|
|
791
|
+
"triggered": False,
|
|
792
|
+
"page_index": page_index,
|
|
793
|
+
"cleaned_text_chars": cleaned_text_chars,
|
|
794
|
+
"ascii_punct_count": ascii_punct_count,
|
|
795
|
+
"ascii_punct_ratio": ascii_punct_ratio,
|
|
796
|
+
"ascii_punct_run_chars": ascii_punct_run_chars,
|
|
797
|
+
"punct_run_ratio": punct_run_ratio,
|
|
798
|
+
}
|
|
799
|
+
if (
|
|
800
|
+
cleaned_text_chars >= SUSPICIOUS_ASCII_PUNCT_MIN_TEXT_CHARS
|
|
801
|
+
and ascii_punct_ratio >= SUSPICIOUS_ASCII_PUNCT_RATIO_THRESHOLD
|
|
802
|
+
and punct_run_ratio >= SUSPICIOUS_ASCII_PUNCT_RUN_RATIO_THRESHOLD
|
|
803
|
+
):
|
|
804
|
+
signal["triggered"] = True
|
|
805
|
+
return signal
|
|
806
|
+
|
|
807
|
+
# 未触发时保留最可疑的抽样页指标,方便日志扩展和后续排查阈值边界。
|
|
808
|
+
if (
|
|
809
|
+
signal["punct_run_ratio"],
|
|
810
|
+
signal["ascii_punct_ratio"],
|
|
811
|
+
signal["cleaned_text_chars"],
|
|
812
|
+
) > (
|
|
813
|
+
best_signal["punct_run_ratio"],
|
|
814
|
+
best_signal["ascii_punct_ratio"],
|
|
815
|
+
best_signal["cleaned_text_chars"],
|
|
816
|
+
):
|
|
817
|
+
best_signal = signal
|
|
818
|
+
|
|
819
|
+
return best_signal
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def _get_pdf_object_cache_key(obj_ref: Any, obj: Any) -> tuple[Any, ...]:
|
|
823
|
+
"""为 pypdf 间接或直接对象生成可复用的身份键。"""
|
|
824
|
+
idnum = getattr(obj_ref, "idnum", None)
|
|
825
|
+
generation = getattr(obj_ref, "generation", None)
|
|
826
|
+
if idnum is not None:
|
|
827
|
+
return "indirect", idnum, generation
|
|
828
|
+
return "direct", id(obj)
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
def _get_font_resource_analysis(
|
|
832
|
+
font_ref: Any,
|
|
833
|
+
font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]],
|
|
834
|
+
) -> tuple[Any, dict[str, bool]]:
|
|
835
|
+
"""按字体资源对象身份缓存 CID 与 Type1 字体语义分析结果。"""
|
|
836
|
+
font = _resolve_pdf_object(font_ref)
|
|
837
|
+
if not font:
|
|
838
|
+
raise ValueError("Unable to resolve PDF font resource")
|
|
839
|
+
|
|
840
|
+
cache_key = _get_pdf_object_cache_key(font_ref, font)
|
|
841
|
+
analysis = font_analysis_cache.get(cache_key)
|
|
842
|
+
if analysis is None:
|
|
843
|
+
subtype = str(font.get("/Subtype"))
|
|
844
|
+
encoding = str(font.get("/Encoding"))
|
|
845
|
+
cid_without_to_unicode = (
|
|
846
|
+
subtype == "/Type0"
|
|
847
|
+
and encoding in ("/Identity-H", "/Identity-V")
|
|
848
|
+
and "/DescendantFonts" in font
|
|
849
|
+
and "/ToUnicode" not in font
|
|
850
|
+
)
|
|
851
|
+
latin_charset_signal = _get_latin_charset_with_to_unicode_signal(font)
|
|
852
|
+
analysis = {
|
|
853
|
+
"cid_without_to_unicode": cid_without_to_unicode,
|
|
854
|
+
"latin_charset_with_to_unicode": latin_charset_signal["triggered"],
|
|
855
|
+
}
|
|
856
|
+
font_analysis_cache[cache_key] = analysis
|
|
857
|
+
return font, analysis
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
def _get_pdf_string_raw_bytes(value: Any) -> bytes:
|
|
861
|
+
"""读取 pypdf 字符串对象的原始字节,禁止用已解码文本替代。"""
|
|
862
|
+
raw_bytes = getattr(value, "original_bytes", None)
|
|
863
|
+
if isinstance(raw_bytes, bytes):
|
|
864
|
+
return raw_bytes
|
|
865
|
+
if isinstance(value, bytes):
|
|
866
|
+
return value
|
|
867
|
+
raise ValueError("PDF text string does not expose original bytes")
|
|
868
|
+
|
|
869
|
+
|
|
870
|
+
def _count_identity_cid_string(value: Any) -> int:
|
|
871
|
+
"""按 Identity-H/V 的双字节编码统计文本字符串中的 CID 数量。"""
|
|
872
|
+
raw_bytes = _get_pdf_string_raw_bytes(value)
|
|
873
|
+
if len(raw_bytes) % 2:
|
|
874
|
+
raise ValueError("Identity CID text string has an odd byte length")
|
|
875
|
+
return len(raw_bytes) // 2
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
def _resource_graph_has_cid_without_to_unicode(
|
|
879
|
+
resources: Any,
|
|
880
|
+
font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]],
|
|
881
|
+
active_form_keys: frozenset[tuple[Any, ...]] = frozenset(),
|
|
882
|
+
) -> bool:
|
|
883
|
+
"""递归检查页面及 Form 资源图中是否存在缺少 ToUnicode 的 Identity CID 字体。"""
|
|
884
|
+
resources = _resolve_pdf_object(resources)
|
|
885
|
+
if not resources:
|
|
886
|
+
return False
|
|
887
|
+
|
|
888
|
+
fonts = _resolve_pdf_object(resources.get("/Font")) or {}
|
|
889
|
+
for font_ref in fonts.values():
|
|
890
|
+
_font, analysis = _get_font_resource_analysis(
|
|
891
|
+
font_ref,
|
|
892
|
+
font_analysis_cache,
|
|
893
|
+
)
|
|
894
|
+
if analysis["cid_without_to_unicode"]:
|
|
895
|
+
return True
|
|
896
|
+
|
|
897
|
+
xobjects = _resolve_pdf_object(resources.get("/XObject")) or {}
|
|
898
|
+
for xobject_ref in xobjects.values():
|
|
899
|
+
xobject = _resolve_pdf_object(xobject_ref)
|
|
900
|
+
if not xobject or str(xobject.get("/Subtype")) != "/Form":
|
|
901
|
+
continue
|
|
902
|
+
|
|
903
|
+
form_key = _get_pdf_object_cache_key(xobject_ref, xobject)
|
|
904
|
+
if form_key in active_form_keys:
|
|
905
|
+
continue
|
|
906
|
+
form_resources = xobject.get("/Resources")
|
|
907
|
+
if form_resources is None:
|
|
908
|
+
continue
|
|
909
|
+
if _resource_graph_has_cid_without_to_unicode(
|
|
910
|
+
form_resources,
|
|
911
|
+
font_analysis_cache,
|
|
912
|
+
active_form_keys | {form_key},
|
|
913
|
+
):
|
|
914
|
+
return True
|
|
915
|
+
return False
|
|
916
|
+
|
|
917
|
+
|
|
918
|
+
def _count_cid_font_usage_in_content(
|
|
919
|
+
reader: PdfReader,
|
|
920
|
+
content: Any,
|
|
921
|
+
resources: Any,
|
|
922
|
+
font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]],
|
|
923
|
+
*,
|
|
924
|
+
inherited_font: tuple[str, bool] | None = None,
|
|
925
|
+
active_form_keys: frozenset[tuple[Any, ...]] = frozenset(),
|
|
926
|
+
) -> Counter[str]:
|
|
927
|
+
"""按实际 Tf 资源递归统计内容流中缺少 ToUnicode 的 Identity CID 字形。"""
|
|
928
|
+
counts: Counter[str] = Counter()
|
|
929
|
+
if content is None:
|
|
930
|
+
return counts
|
|
931
|
+
|
|
932
|
+
resources = _resolve_pdf_object(resources)
|
|
933
|
+
if resources is None:
|
|
934
|
+
raise ValueError("PDF content stream has no resolvable resources")
|
|
935
|
+
|
|
936
|
+
fonts = _resolve_pdf_object(resources.get("/Font")) or {}
|
|
937
|
+
xobjects = _resolve_pdf_object(resources.get("/XObject")) or {}
|
|
938
|
+
current_font = inherited_font
|
|
939
|
+
font_stack: list[tuple[str, bool] | None] = []
|
|
940
|
+
|
|
941
|
+
for operands, operator in ContentStream(content, reader).operations:
|
|
942
|
+
if operator == b"q":
|
|
943
|
+
font_stack.append(current_font)
|
|
944
|
+
continue
|
|
945
|
+
if operator == b"Q":
|
|
946
|
+
current_font = font_stack.pop() if font_stack else inherited_font
|
|
947
|
+
continue
|
|
948
|
+
if operator == b"Tf":
|
|
949
|
+
if not operands:
|
|
950
|
+
raise ValueError("PDF Tf operator has no font resource name")
|
|
951
|
+
font_key = operands[0]
|
|
952
|
+
font_ref = fonts.get(font_key)
|
|
953
|
+
if font_ref is None:
|
|
954
|
+
raise ValueError(f"Unable to resolve PDF font resource {font_key}")
|
|
955
|
+
font, analysis = _get_font_resource_analysis(
|
|
956
|
+
font_ref,
|
|
957
|
+
font_analysis_cache,
|
|
958
|
+
)
|
|
959
|
+
font_name = _normalize_pdf_font_name(font.get("/BaseFont") or font_key)
|
|
960
|
+
current_font = (
|
|
961
|
+
font_name,
|
|
962
|
+
analysis["cid_without_to_unicode"],
|
|
963
|
+
)
|
|
964
|
+
continue
|
|
965
|
+
|
|
966
|
+
if operator in (b"Tj", b"'", b'"'):
|
|
967
|
+
if current_font is None:
|
|
968
|
+
raise ValueError("PDF text is shown before selecting a font")
|
|
969
|
+
if current_font[1]:
|
|
970
|
+
counts[current_font[0]] += _count_identity_cid_string(operands[-1])
|
|
971
|
+
continue
|
|
972
|
+
if operator == b"TJ":
|
|
973
|
+
if current_font is None:
|
|
974
|
+
raise ValueError("PDF text is shown before selecting a font")
|
|
975
|
+
if current_font[1]:
|
|
976
|
+
for value in operands[0]:
|
|
977
|
+
if isinstance(value, (int, float)):
|
|
978
|
+
continue
|
|
979
|
+
counts[current_font[0]] += _count_identity_cid_string(value)
|
|
980
|
+
continue
|
|
981
|
+
if operator != b"Do":
|
|
982
|
+
continue
|
|
983
|
+
|
|
984
|
+
if not operands:
|
|
985
|
+
raise ValueError("PDF Do operator has no XObject resource name")
|
|
986
|
+
xobject_key = operands[0]
|
|
987
|
+
xobject_ref = xobjects.get(xobject_key)
|
|
988
|
+
if xobject_ref is None:
|
|
989
|
+
raise ValueError(f"Unable to resolve PDF XObject resource {xobject_key}")
|
|
990
|
+
xobject = _resolve_pdf_object(xobject_ref)
|
|
991
|
+
if not xobject:
|
|
992
|
+
raise ValueError(f"Unable to resolve PDF XObject {xobject_key}")
|
|
993
|
+
if str(xobject.get("/Subtype")) != "/Form":
|
|
994
|
+
continue
|
|
995
|
+
|
|
996
|
+
form_key = _get_pdf_object_cache_key(xobject_ref, xobject)
|
|
997
|
+
if form_key in active_form_keys:
|
|
998
|
+
raise ValueError(f"Cyclic PDF Form XObject reference {xobject_key}")
|
|
999
|
+
form_resources = xobject.get("/Resources")
|
|
1000
|
+
child_resources = resources if form_resources is None else form_resources
|
|
1001
|
+
counts.update(
|
|
1002
|
+
_count_cid_font_usage_in_content(
|
|
1003
|
+
reader,
|
|
1004
|
+
xobject,
|
|
1005
|
+
child_resources,
|
|
1006
|
+
font_analysis_cache,
|
|
1007
|
+
inherited_font=current_font,
|
|
1008
|
+
active_form_keys=active_form_keys | {form_key},
|
|
1009
|
+
)
|
|
1010
|
+
)
|
|
1011
|
+
return counts
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def _get_font_resource_signals_pypdf(
|
|
1015
|
+
pdf_bytes: bytes,
|
|
1016
|
+
page_indices: list[int],
|
|
1017
|
+
) -> dict[str, Any]:
|
|
1018
|
+
"""一次扫描抽样页字体资源,收集 CID 缺映射和 Type1 Latin 候选字体。"""
|
|
1019
|
+
reader = PdfReader(BytesIO(pdf_bytes))
|
|
1020
|
+
cid_page_fonts: dict[int, set[str]] = {}
|
|
1021
|
+
cid_page_usage: dict[int, dict[str, Any]] = {}
|
|
1022
|
+
latin_charset_page_fonts: dict[int, set[str]] = {}
|
|
1023
|
+
font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]] = {}
|
|
1024
|
+
|
|
1025
|
+
for page_index in page_indices:
|
|
1026
|
+
page = reader.pages[page_index]
|
|
1027
|
+
resources = _resolve_pdf_object(page.get("/Resources"))
|
|
1028
|
+
if not resources:
|
|
1029
|
+
continue
|
|
1030
|
+
|
|
1031
|
+
fonts = _resolve_pdf_object(resources.get("/Font")) or {}
|
|
1032
|
+
|
|
1033
|
+
# Type1 Latin 信号仍按字体名使用 PDFium 统计;CID 用量在后续按资源对象精确计算。
|
|
1034
|
+
page_latin_font_resources: dict[str, dict[tuple[Any, ...], bool]] = {}
|
|
1035
|
+
for font_key, font_ref in fonts.items():
|
|
1036
|
+
font, analysis = _get_font_resource_analysis(
|
|
1037
|
+
font_ref,
|
|
1038
|
+
font_analysis_cache,
|
|
1039
|
+
)
|
|
1040
|
+
font_name = _normalize_pdf_font_name(font.get("/BaseFont") or font_key)
|
|
1041
|
+
if not font_name:
|
|
1042
|
+
continue
|
|
1043
|
+
|
|
1044
|
+
cache_key = _get_pdf_object_cache_key(font_ref, font)
|
|
1045
|
+
|
|
1046
|
+
if analysis["cid_without_to_unicode"]:
|
|
1047
|
+
cid_page_fonts.setdefault(page_index, set()).add(font_name)
|
|
1048
|
+
|
|
1049
|
+
page_latin_font_resources.setdefault(font_name, {})[cache_key] = analysis["latin_charset_with_to_unicode"]
|
|
1050
|
+
|
|
1051
|
+
for font_name, resource_states in page_latin_font_resources.items():
|
|
1052
|
+
if len(resource_states) == 1 and set(resource_states.values()) == {True}:
|
|
1053
|
+
latin_charset_page_fonts.setdefault(page_index, set()).add(font_name)
|
|
1054
|
+
|
|
1055
|
+
if _resource_graph_has_cid_without_to_unicode(
|
|
1056
|
+
resources,
|
|
1057
|
+
font_analysis_cache,
|
|
1058
|
+
):
|
|
1059
|
+
usage_counts = _count_cid_font_usage_in_content(
|
|
1060
|
+
reader,
|
|
1061
|
+
page.get_contents(),
|
|
1062
|
+
resources,
|
|
1063
|
+
font_analysis_cache,
|
|
1064
|
+
)
|
|
1065
|
+
cid_page_usage[page_index] = {
|
|
1066
|
+
"font_names": sorted(font_name for font_name, char_count in usage_counts.items() if char_count > 0),
|
|
1067
|
+
"cid_font_char_count": sum(usage_counts.values()),
|
|
1068
|
+
}
|
|
1069
|
+
|
|
1070
|
+
return {
|
|
1071
|
+
"cid_without_to_unicode": {
|
|
1072
|
+
"triggered": bool(cid_page_fonts),
|
|
1073
|
+
"page_fonts": cid_page_fonts,
|
|
1074
|
+
},
|
|
1075
|
+
"cid_without_to_unicode_usage": cid_page_usage,
|
|
1076
|
+
"latin_charset_with_to_unicode": {
|
|
1077
|
+
"triggered": bool(latin_charset_page_fonts),
|
|
1078
|
+
"page_fonts": latin_charset_page_fonts,
|
|
1079
|
+
},
|
|
1080
|
+
}
|
|
1081
|
+
|
|
1082
|
+
|
|
1083
|
+
def _resolve_pdf_object(obj: Any) -> Any:
|
|
1084
|
+
if hasattr(obj, "get_object"):
|
|
1085
|
+
return obj.get_object()
|
|
1086
|
+
return obj
|
|
1087
|
+
|
|
1088
|
+
|
|
1089
|
+
def _get_pdfium_page_object_bounds(page_object: Any) -> tuple[float, float, float, float]:
|
|
1090
|
+
"""兼容 pypdfium2 4.x/5.x,统一获取页面对象的边界坐标。"""
|
|
1091
|
+
get_bounds = getattr(page_object, "get_bounds", None)
|
|
1092
|
+
if callable(get_bounds):
|
|
1093
|
+
return get_bounds()
|
|
1094
|
+
|
|
1095
|
+
get_pos = getattr(page_object, "get_pos", None)
|
|
1096
|
+
if callable(get_pos):
|
|
1097
|
+
return get_pos()
|
|
1098
|
+
|
|
1099
|
+
raise AttributeError("PDFium page object has neither get_bounds() nor get_pos()")
|
|
1100
|
+
|
|
1101
|
+
|
|
1102
|
+
def get_high_image_coverage_ratio_pdfium(pdf_doc: pdfium.PdfDocument, page_indices: list[int]) -> float:
|
|
1103
|
+
high_image_coverage_pages = 0
|
|
1104
|
+
|
|
1105
|
+
with pdfium_guard():
|
|
1106
|
+
for page_index in page_indices:
|
|
1107
|
+
page = None
|
|
1108
|
+
try:
|
|
1109
|
+
page = pdf_doc[page_index]
|
|
1110
|
+
page_bbox: tuple[float, float, float, float] = page.get_bbox()
|
|
1111
|
+
page_area = abs((page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1]))
|
|
1112
|
+
image_area = 0.0
|
|
1113
|
+
|
|
1114
|
+
for page_object in page.get_objects(filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3):
|
|
1115
|
+
try:
|
|
1116
|
+
left, bottom, right, top = _get_pdfium_page_object_bounds(page_object)
|
|
1117
|
+
image_area += max(0.0, right - left) * max(0.0, top - bottom)
|
|
1118
|
+
finally:
|
|
1119
|
+
close_pdfium_child(page_object)
|
|
1120
|
+
|
|
1121
|
+
coverage_ratio = min(image_area / page_area, 1.0) if page_area > 0 else 0.0
|
|
1122
|
+
if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD:
|
|
1123
|
+
high_image_coverage_pages += 1
|
|
1124
|
+
finally:
|
|
1125
|
+
close_pdfium_child(page)
|
|
1126
|
+
|
|
1127
|
+
if not page_indices:
|
|
1128
|
+
return 0.0
|
|
1129
|
+
return high_image_coverage_pages / len(page_indices)
|
|
1130
|
+
|
|
1131
|
+
|
|
1132
|
+
if __name__ == "__main__":
|
|
1133
|
+
from .document import PDFDocument
|
|
1134
|
+
|
|
1135
|
+
with open("/Users/myhloli/pdf/luanma2x10.pdf", "rb") as f:
|
|
1136
|
+
p_bytes = f.read()
|
|
1137
|
+
pdf_doc = PDFDocument(p_bytes)
|
|
1138
|
+
logger.info(f"PDF classify result: {pdf_doc.classify()}")
|