docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""PDF 字符去重及原始文字几何提取,保持原生提取算法与资源语义。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
import logging
|
|
5
|
+
import math
|
|
6
|
+
from typing import Any, Iterator, cast
|
|
7
|
+
import pypdfium2 as pdfium
|
|
8
|
+
import pypdfium2.raw as pdfium_c
|
|
9
|
+
from .text.extract import get_chars, deduplicate_chars
|
|
10
|
+
from .text.contracts import Char
|
|
11
|
+
from .text.geometry import char_bbox_values as _char_bbox_values
|
|
12
|
+
|
|
13
|
+
from .native_contracts import (
|
|
14
|
+
NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE,
|
|
15
|
+
OFFSET_DUPLICATE_CHAR_BBOX_TOLERANCE,
|
|
16
|
+
OFFSET_DUPLICATE_MIN_BBOX_OVERLAP_RATIO,
|
|
17
|
+
OFFSET_DUPLICATE_TRANSLATION_TOLERANCE,
|
|
18
|
+
PDFPageImage,
|
|
19
|
+
PDFPageTextGeometry,
|
|
20
|
+
)
|
|
21
|
+
from .native_lifecycle import _try_close
|
|
22
|
+
|
|
23
|
+
logger = logging.getLogger("docvortex.document.pdf.document")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _get_visible_char_signature(
|
|
27
|
+
char: Char,
|
|
28
|
+
) -> tuple[str, tuple[Any, Any, Any, Any], float]:
|
|
29
|
+
"""生成可见字符去重签名,不把 bbox 放入签名以便单独做近重合判断。"""
|
|
30
|
+
font = char.get("font") or {}
|
|
31
|
+
font_key = (
|
|
32
|
+
font.get("name"),
|
|
33
|
+
font.get("flags"),
|
|
34
|
+
font.get("size"),
|
|
35
|
+
font.get("weight"),
|
|
36
|
+
)
|
|
37
|
+
rotation_key = round(float(char.get("rotation") or 0.0), 3)
|
|
38
|
+
return str(char.get("char", "")), font_key, rotation_key
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _is_near_identical_bbox(
|
|
42
|
+
bbox_a: tuple[float, float, float, float],
|
|
43
|
+
bbox_b: tuple[float, float, float, float],
|
|
44
|
+
) -> bool:
|
|
45
|
+
"""判断两个字符 bbox 是否属于同一视觉位置的一点内抖动。"""
|
|
46
|
+
return all(abs(coord_a - coord_b) <= NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE for coord_a, coord_b in zip(bbox_a, bbox_b))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _calculate_bbox_overlap_in_smaller_area(
|
|
50
|
+
bbox_a: tuple[float, float, float, float],
|
|
51
|
+
bbox_b: tuple[float, float, float, float],
|
|
52
|
+
) -> float:
|
|
53
|
+
"""计算两个字符框交集占较小字符框面积的比例。"""
|
|
54
|
+
intersection_width = max(
|
|
55
|
+
0.0,
|
|
56
|
+
min(bbox_a[2], bbox_b[2]) - max(bbox_a[0], bbox_b[0]),
|
|
57
|
+
)
|
|
58
|
+
intersection_height = max(
|
|
59
|
+
0.0,
|
|
60
|
+
min(bbox_a[3], bbox_b[3]) - max(bbox_a[1], bbox_b[1]),
|
|
61
|
+
)
|
|
62
|
+
bbox_a_area = max(0.0, bbox_a[2] - bbox_a[0]) * max(
|
|
63
|
+
0.0,
|
|
64
|
+
bbox_a[3] - bbox_a[1],
|
|
65
|
+
)
|
|
66
|
+
bbox_b_area = max(0.0, bbox_b[2] - bbox_b[0]) * max(
|
|
67
|
+
0.0,
|
|
68
|
+
bbox_b[3] - bbox_b[1],
|
|
69
|
+
)
|
|
70
|
+
smaller_area = min(bbox_a_area, bbox_b_area)
|
|
71
|
+
if smaller_area == 0:
|
|
72
|
+
return 0.0
|
|
73
|
+
return intersection_width * intersection_height / smaller_area
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _is_adjacent_offset_duplicate_char(
|
|
77
|
+
previous_char: Char,
|
|
78
|
+
current_char: Char,
|
|
79
|
+
) -> bool:
|
|
80
|
+
"""识别相邻字符中由对角平移阴影产生的第二个重复字符。"""
|
|
81
|
+
if _get_visible_char_signature(previous_char) != _get_visible_char_signature(current_char):
|
|
82
|
+
return False
|
|
83
|
+
|
|
84
|
+
previous_bbox = _char_bbox_values(previous_char.get("bbox"))
|
|
85
|
+
current_bbox = _char_bbox_values(current_char.get("bbox"))
|
|
86
|
+
if previous_bbox is None or current_bbox is None:
|
|
87
|
+
return False
|
|
88
|
+
|
|
89
|
+
x_start_offset = current_bbox[0] - previous_bbox[0]
|
|
90
|
+
y_start_offset = current_bbox[1] - previous_bbox[1]
|
|
91
|
+
x_end_offset = current_bbox[2] - previous_bbox[2]
|
|
92
|
+
y_end_offset = current_bbox[3] - previous_bbox[3]
|
|
93
|
+
|
|
94
|
+
# 阴影层应是同一字符框的刚性平移,避免把大小不同的相邻同字误判为重复。
|
|
95
|
+
if (
|
|
96
|
+
abs(x_start_offset - x_end_offset) > OFFSET_DUPLICATE_TRANSLATION_TOLERANCE
|
|
97
|
+
or abs(y_start_offset - y_end_offset) > OFFSET_DUPLICATE_TRANSLATION_TOLERANCE
|
|
98
|
+
):
|
|
99
|
+
return False
|
|
100
|
+
|
|
101
|
+
if not (
|
|
102
|
+
NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE < abs(x_start_offset) <= OFFSET_DUPLICATE_CHAR_BBOX_TOLERANCE
|
|
103
|
+
and NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE < abs(y_start_offset) <= OFFSET_DUPLICATE_CHAR_BBOX_TOLERANCE
|
|
104
|
+
):
|
|
105
|
+
return False
|
|
106
|
+
|
|
107
|
+
return _calculate_bbox_overlap_in_smaller_area(previous_bbox, current_bbox) >= OFFSET_DUPLICATE_MIN_BBOX_OVERLAP_RATIO
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _get_near_identical_bbox_bucket_key(
|
|
111
|
+
bbox_coords: tuple[float, float, float, float],
|
|
112
|
+
) -> tuple[int, int]:
|
|
113
|
+
"""按字符 bbox 左上角生成空间桶 key,缩小近重合判断的候选范围。"""
|
|
114
|
+
return (
|
|
115
|
+
math.floor(bbox_coords[0] / NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE),
|
|
116
|
+
math.floor(bbox_coords[1] / NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE),
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _iter_neighbor_bbox_bucket_keys(
|
|
121
|
+
bucket_key: tuple[int, int],
|
|
122
|
+
) -> Iterator[tuple[int, int]]:
|
|
123
|
+
"""遍历当前桶及周围 8 个邻近桶,覆盖 bbox 容差范围内的候选字符。"""
|
|
124
|
+
bucket_x, bucket_y = bucket_key
|
|
125
|
+
for offset_x in (-1, 0, 1):
|
|
126
|
+
for offset_y in (-1, 0, 1):
|
|
127
|
+
yield bucket_x + offset_x, bucket_y + offset_y
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _deduplicate_near_identical_chars(chars: list[Char]) -> list[Char]:
|
|
131
|
+
"""清理 PDFium 文本层边界处同字符、同位置及对角阴影重复字符。"""
|
|
132
|
+
seen_visible_char_bboxes: dict[
|
|
133
|
+
tuple[str, tuple[Any, Any, Any, Any], float],
|
|
134
|
+
dict[tuple[int, int], list[tuple[float, float, float, float]]],
|
|
135
|
+
] = {}
|
|
136
|
+
deduplicated_chars: list[Char] = []
|
|
137
|
+
|
|
138
|
+
for char in chars:
|
|
139
|
+
text = str(char.get("char", ""))
|
|
140
|
+
if not text or text.isspace():
|
|
141
|
+
deduplicated_chars.append(char)
|
|
142
|
+
continue
|
|
143
|
+
|
|
144
|
+
visible_char_key = _get_visible_char_signature(char)
|
|
145
|
+
bbox_coords = _char_bbox_values(char.get("bbox"))
|
|
146
|
+
if bbox_coords is None:
|
|
147
|
+
deduplicated_chars.append(char)
|
|
148
|
+
continue
|
|
149
|
+
|
|
150
|
+
if deduplicated_chars and _is_adjacent_offset_duplicate_char(
|
|
151
|
+
deduplicated_chars[-1],
|
|
152
|
+
char,
|
|
153
|
+
):
|
|
154
|
+
continue
|
|
155
|
+
|
|
156
|
+
bbox_bucket_key = _get_near_identical_bbox_bucket_key(bbox_coords)
|
|
157
|
+
visible_char_bbox_buckets = seen_visible_char_bboxes.setdefault(
|
|
158
|
+
visible_char_key,
|
|
159
|
+
{},
|
|
160
|
+
)
|
|
161
|
+
if any(
|
|
162
|
+
_is_near_identical_bbox(bbox_coords, seen_bbox)
|
|
163
|
+
for neighbor_bucket_key in _iter_neighbor_bbox_bucket_keys(bbox_bucket_key)
|
|
164
|
+
for seen_bbox in visible_char_bbox_buckets.get(neighbor_bucket_key, [])
|
|
165
|
+
):
|
|
166
|
+
continue
|
|
167
|
+
|
|
168
|
+
visible_char_bbox_buckets.setdefault(bbox_bucket_key, []).append(bbox_coords)
|
|
169
|
+
deduplicated_chars.append(char)
|
|
170
|
+
|
|
171
|
+
return deduplicated_chars
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _restore_pdfium_surrogate_pairs(
|
|
175
|
+
chars: list[Char],
|
|
176
|
+
textpage: pdfium.PdfTextPage,
|
|
177
|
+
*,
|
|
178
|
+
raw_codes: dict[int, int] | None = None,
|
|
179
|
+
) -> list[Char]:
|
|
180
|
+
"""利用 PDFium 原始 UTF-16 code unit 恢复 pdftext 丢失的补充平面字符。"""
|
|
181
|
+
if not any(
|
|
182
|
+
len(text := str(char.get("char", ""))) == 1 and (text == "\ufffd" or 0xD800 <= ord(text) <= 0xDFFF) for char in chars
|
|
183
|
+
):
|
|
184
|
+
return chars
|
|
185
|
+
|
|
186
|
+
try:
|
|
187
|
+
textpage_raw = textpage.raw
|
|
188
|
+
char_count = int(textpage.count_chars())
|
|
189
|
+
except Exception:
|
|
190
|
+
textpage_raw = None
|
|
191
|
+
char_count = 0
|
|
192
|
+
|
|
193
|
+
restored_chars: list[Char] = []
|
|
194
|
+
consumed_char_indices: set[int] = set()
|
|
195
|
+
|
|
196
|
+
def get_unicode(handle: object, index: int) -> int:
|
|
197
|
+
"""优先复用首次读取的原始码值,仅为独立辅助调用读取原生接口。"""
|
|
198
|
+
return raw_codes[index] if raw_codes is not None else int(pdfium_c.FPDFText_GetUnicode(handle, index))
|
|
199
|
+
|
|
200
|
+
for char in chars:
|
|
201
|
+
text = str(char.get("char", ""))
|
|
202
|
+
raw_char_idx = char.get("char_idx")
|
|
203
|
+
try:
|
|
204
|
+
char_idx = int(raw_char_idx) if raw_char_idx is not None else -1
|
|
205
|
+
except (TypeError, ValueError):
|
|
206
|
+
char_idx = -1
|
|
207
|
+
|
|
208
|
+
if char_idx in consumed_char_indices:
|
|
209
|
+
continue
|
|
210
|
+
|
|
211
|
+
raw_code = None
|
|
212
|
+
if (
|
|
213
|
+
textpage_raw is not None
|
|
214
|
+
and 0 <= char_idx < char_count
|
|
215
|
+
and len(text) == 1
|
|
216
|
+
and (text == "\ufffd" or 0xD800 <= ord(text) <= 0xDFFF)
|
|
217
|
+
):
|
|
218
|
+
raw_code = int(get_unicode(textpage_raw, char_idx))
|
|
219
|
+
|
|
220
|
+
high_surrogate = None
|
|
221
|
+
low_surrogate = None
|
|
222
|
+
if raw_code is not None and 0xD800 <= raw_code <= 0xDBFF and char_idx + 1 < char_count:
|
|
223
|
+
next_code = int(get_unicode(textpage_raw, char_idx + 1))
|
|
224
|
+
if 0xDC00 <= next_code <= 0xDFFF:
|
|
225
|
+
high_surrogate = raw_code
|
|
226
|
+
low_surrogate = next_code
|
|
227
|
+
consumed_char_indices.add(char_idx + 1)
|
|
228
|
+
elif raw_code is not None and 0xDC00 <= raw_code <= 0xDFFF and char_idx > 0:
|
|
229
|
+
previous_code = int(get_unicode(textpage_raw, char_idx - 1))
|
|
230
|
+
if 0xD800 <= previous_code <= 0xDBFF:
|
|
231
|
+
high_surrogate = previous_code
|
|
232
|
+
low_surrogate = raw_code
|
|
233
|
+
|
|
234
|
+
if high_surrogate is not None and low_surrogate is not None:
|
|
235
|
+
restored_char = cast(Char, dict(char))
|
|
236
|
+
restored_char["char"] = chr(0x10000 + ((high_surrogate - 0xD800) << 10) + (low_surrogate - 0xDC00))
|
|
237
|
+
restored_char["source_indices"] = tuple(
|
|
238
|
+
sorted(
|
|
239
|
+
set(
|
|
240
|
+
(
|
|
241
|
+
*char.get("source_indices", (char_idx,)),
|
|
242
|
+
char_idx + 1 if raw_code is not None and raw_code <= 0xDBFF else char_idx - 1,
|
|
243
|
+
)
|
|
244
|
+
)
|
|
245
|
+
)
|
|
246
|
+
)
|
|
247
|
+
restored_chars.append(restored_char)
|
|
248
|
+
continue
|
|
249
|
+
|
|
250
|
+
if len(text) == 1 and 0xD800 <= ord(text) <= 0xDFFF:
|
|
251
|
+
restored_char = cast(Char, dict(char))
|
|
252
|
+
restored_char["char"] = "\ufffd"
|
|
253
|
+
restored_chars.append(restored_char)
|
|
254
|
+
continue
|
|
255
|
+
|
|
256
|
+
restored_chars.append(char)
|
|
257
|
+
|
|
258
|
+
return restored_chars
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _page_to_image(page: pdfium.PdfPage, scale: float, max_edge: int) -> PDFPageImage:
|
|
262
|
+
"""按原缩放与长边上限复制页面像素,并返回独立持有的图片。"""
|
|
263
|
+
long_edge_length = max(*page.get_size())
|
|
264
|
+
if (long_edge_length * scale) > max_edge:
|
|
265
|
+
scale = max_edge / long_edge_length
|
|
266
|
+
|
|
267
|
+
bitmap = None
|
|
268
|
+
try:
|
|
269
|
+
bitmap = page.render(scale=scale) # type: ignore
|
|
270
|
+
bitmap = cast(pdfium.PdfBitmap, bitmap)
|
|
271
|
+
pil_image = bitmap.to_pil()
|
|
272
|
+
finally:
|
|
273
|
+
_try_close(bitmap)
|
|
274
|
+
|
|
275
|
+
return PDFPageImage(pil_image=pil_image, scale=scale)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _extract_page_text_geometry(
|
|
279
|
+
page: pdfium.PdfPage,
|
|
280
|
+
*,
|
|
281
|
+
include_extended_geometry: bool,
|
|
282
|
+
) -> PDFPageTextGeometry:
|
|
283
|
+
"""在调用方持有的页面和锁内读取字符,使批量提取与独立接口共用实现。"""
|
|
284
|
+
textpage = None
|
|
285
|
+
try:
|
|
286
|
+
textpage = page.get_textpage()
|
|
287
|
+
raw_page_bbox: list[float] = list(page.get_bbox())
|
|
288
|
+
page_rotation: int = 0
|
|
289
|
+
try:
|
|
290
|
+
page_rotation = page.get_rotation()
|
|
291
|
+
except Exception:
|
|
292
|
+
pass
|
|
293
|
+
chars = get_chars(textpage, raw_page_bbox, page_rotation, include_geometry=include_extended_geometry)
|
|
294
|
+
raw_codes = {char["char_idx"]: char["raw_code"] for char in chars}
|
|
295
|
+
chars = deduplicate_chars(chars)
|
|
296
|
+
chars = _restore_pdfium_surrogate_pairs(chars, textpage, raw_codes=raw_codes)
|
|
297
|
+
chars = _deduplicate_near_identical_chars(chars)
|
|
298
|
+
if include_extended_geometry:
|
|
299
|
+
loose_bboxes = {
|
|
300
|
+
char["char_idx"]: char["loose_bbox"]
|
|
301
|
+
for char in chars
|
|
302
|
+
if char.get("loose_bbox") is not None and abs(char["rotation"]) > 1e-9
|
|
303
|
+
}
|
|
304
|
+
tight_bboxes = {char["char_idx"]: char["tight_bbox"] for char in chars if char.get("tight_bbox") is not None}
|
|
305
|
+
origins = {char["char_idx"]: char["origin"] for char in chars if char.get("origin") is not None}
|
|
306
|
+
else:
|
|
307
|
+
loose_bboxes, tight_bboxes, origins = {}, {}, {}
|
|
308
|
+
finally:
|
|
309
|
+
_try_close(textpage)
|
|
310
|
+
return PDFPageTextGeometry(
|
|
311
|
+
chars=chars,
|
|
312
|
+
tight_bboxes=tight_bboxes,
|
|
313
|
+
origins=origins,
|
|
314
|
+
loose_bboxes=loose_bboxes,
|
|
315
|
+
)
|
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import threading
|
|
3
|
+
from contextlib import contextmanager
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from io import BytesIO
|
|
6
|
+
from typing import Any, Iterator, Sequence, TypeVar
|
|
7
|
+
|
|
8
|
+
from loguru import logger
|
|
9
|
+
|
|
10
|
+
from .font_runtime import PdfiumFontError, PdfiumRuntimeInfo, _FontProvider
|
|
11
|
+
|
|
12
|
+
_pdfium_lock = threading.RLock()
|
|
13
|
+
_font_provider: _FontProvider | None = None
|
|
14
|
+
|
|
15
|
+
T = TypeVar("T")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _check_runtime_process() -> None:
|
|
19
|
+
"""拒绝复用 fork 继承的字体接口;渲染 worker 应通过 spawn/forkserver 独立初始化。"""
|
|
20
|
+
if _font_provider is not None and _font_provider.pid != os.getpid():
|
|
21
|
+
raise PdfiumFontError("Inherited PDFium font runtime: use spawn/forkserver instead of forking after PDF use")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def initialize_pdfium_runtime() -> PdfiumRuntimeInfo:
|
|
25
|
+
"""幂等安装本进程的固定 CJK 字体提供器;必须先于首次 PDF 字体使用。"""
|
|
26
|
+
global _font_provider
|
|
27
|
+
_check_runtime_process()
|
|
28
|
+
with _pdfium_lock:
|
|
29
|
+
if _font_provider is None:
|
|
30
|
+
import pypdfium2 as pdfium
|
|
31
|
+
import pypdfium2.raw as raw
|
|
32
|
+
|
|
33
|
+
provider = _FontProvider(raw, str(pdfium.PDFIUM_INFO))
|
|
34
|
+
# 注册前保留强引用,保证即使安装阶段回调失败,C 指针也不会指向被回收的对象。
|
|
35
|
+
_font_provider = provider
|
|
36
|
+
provider.install()
|
|
37
|
+
_font_provider.raise_if_failed()
|
|
38
|
+
return _font_provider.info
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class PdfiumRewriteResult:
|
|
43
|
+
"""记录 PDFium 安全重写结果,供调用方按实际保留页修正原始页号。"""
|
|
44
|
+
|
|
45
|
+
pdf_bytes: bytes
|
|
46
|
+
retained_page_indices: list[int] | None = None
|
|
47
|
+
broken_page_indices: list[int] = field(default_factory=list)
|
|
48
|
+
used_original: bool = False
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@contextmanager
|
|
52
|
+
def pdfium_guard() -> Iterator[None]:
|
|
53
|
+
"""串行化访问并确保字体运行时就绪,在原生栈退出后传播字体故障。"""
|
|
54
|
+
_check_runtime_process()
|
|
55
|
+
with _pdfium_lock:
|
|
56
|
+
initialize_pdfium_runtime()
|
|
57
|
+
assert _font_provider is not None
|
|
58
|
+
try:
|
|
59
|
+
yield
|
|
60
|
+
finally:
|
|
61
|
+
_font_provider.raise_if_failed()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def close_pdfium_document(pdf_doc: Any) -> None:
|
|
65
|
+
"""清理时只持锁,不安装字体、不重建已销毁的 PDFium 运行时。"""
|
|
66
|
+
if pdf_doc is None:
|
|
67
|
+
return
|
|
68
|
+
with _pdfium_lock:
|
|
69
|
+
pdf_doc.close()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def close_pdfium_child(pdfium_obj: Any) -> None:
|
|
73
|
+
"""显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。"""
|
|
74
|
+
if pdfium_obj is None:
|
|
75
|
+
return
|
|
76
|
+
close = getattr(pdfium_obj, "close", None)
|
|
77
|
+
if callable(close):
|
|
78
|
+
with _pdfium_lock:
|
|
79
|
+
close()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def close_pdfium_objects_safely(*pdfium_objs: object, owner: str = "pdfium cleanup") -> None:
|
|
83
|
+
"""清理多个 PDFium 对象时逐个尝试关闭,避免前一个关闭失败阻断后续对象释放。"""
|
|
84
|
+
for pdfium_obj in pdfium_objs:
|
|
85
|
+
if pdfium_obj is None:
|
|
86
|
+
continue
|
|
87
|
+
try:
|
|
88
|
+
close_pdfium_child(pdfium_obj)
|
|
89
|
+
except Exception as exc:
|
|
90
|
+
logger.warning(f"Failed to close PDFium object during {owner}: {exc}")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def get_loadable_pdfium_page_indices(
|
|
94
|
+
src_pdf_bytes: bytes,
|
|
95
|
+
start_page_id: int = 0,
|
|
96
|
+
end_page_id: int | None = None,
|
|
97
|
+
) -> tuple[list[int], list[int]]:
|
|
98
|
+
"""逐页探测 PDFium 可加载页面,返回可保留页和损坏页的 0-based 索引。"""
|
|
99
|
+
import pypdfium2 as pdfium
|
|
100
|
+
|
|
101
|
+
loadable_page_indices = []
|
|
102
|
+
broken_page_indices = []
|
|
103
|
+
pdf_doc = None
|
|
104
|
+
|
|
105
|
+
try:
|
|
106
|
+
with pdfium_guard():
|
|
107
|
+
pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
|
|
108
|
+
total_page_count = len(pdf_doc)
|
|
109
|
+
if total_page_count == 0:
|
|
110
|
+
return [], []
|
|
111
|
+
|
|
112
|
+
normalized_start_page_id = max(0, start_page_id)
|
|
113
|
+
normalized_end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else total_page_count - 1
|
|
114
|
+
if normalized_end_page_id > total_page_count - 1:
|
|
115
|
+
normalized_end_page_id = total_page_count - 1
|
|
116
|
+
if normalized_start_page_id > normalized_end_page_id:
|
|
117
|
+
return [], []
|
|
118
|
+
|
|
119
|
+
for page_index in range(
|
|
120
|
+
normalized_start_page_id,
|
|
121
|
+
normalized_end_page_id + 1,
|
|
122
|
+
):
|
|
123
|
+
page = None
|
|
124
|
+
try:
|
|
125
|
+
page = pdf_doc[page_index]
|
|
126
|
+
page.get_size()
|
|
127
|
+
loadable_page_indices.append(page_index)
|
|
128
|
+
except Exception:
|
|
129
|
+
broken_page_indices.append(page_index)
|
|
130
|
+
finally:
|
|
131
|
+
close_pdfium_child(page)
|
|
132
|
+
finally:
|
|
133
|
+
close_pdfium_document(pdf_doc)
|
|
134
|
+
|
|
135
|
+
return loadable_page_indices, broken_page_indices
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _normalize_rewrite_page_indices(
|
|
139
|
+
total_page_count: int,
|
|
140
|
+
start_page_id: int = 0,
|
|
141
|
+
end_page_id: int | None = None,
|
|
142
|
+
page_indices: Sequence[int] | None = None,
|
|
143
|
+
) -> list[int]:
|
|
144
|
+
"""按 rewrite_pdf_bytes_with_pdfium 的规则归一化实际导出的 0-based 页号。"""
|
|
145
|
+
if total_page_count == 0:
|
|
146
|
+
return []
|
|
147
|
+
|
|
148
|
+
if page_indices is not None:
|
|
149
|
+
return sorted({int(page_index) for page_index in page_indices if 0 <= int(page_index) < total_page_count})
|
|
150
|
+
|
|
151
|
+
normalized_start_page_id = max(0, start_page_id)
|
|
152
|
+
normalized_end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else total_page_count - 1
|
|
153
|
+
if normalized_end_page_id > total_page_count - 1:
|
|
154
|
+
normalized_end_page_id = total_page_count - 1
|
|
155
|
+
if normalized_start_page_id > normalized_end_page_id:
|
|
156
|
+
return []
|
|
157
|
+
return list(range(normalized_start_page_id, normalized_end_page_id + 1))
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _get_rewrite_page_indices_from_pdf(
|
|
161
|
+
src_pdf_bytes: bytes,
|
|
162
|
+
start_page_id: int = 0,
|
|
163
|
+
end_page_id: int | None = None,
|
|
164
|
+
page_indices: Sequence[int] | None = None,
|
|
165
|
+
) -> list[int]:
|
|
166
|
+
"""读取源 PDF 页数并计算本次重写会保留的原始页号。"""
|
|
167
|
+
import pypdfium2 as pdfium
|
|
168
|
+
|
|
169
|
+
pdf_doc = None
|
|
170
|
+
try:
|
|
171
|
+
with pdfium_guard():
|
|
172
|
+
pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
|
|
173
|
+
return _normalize_rewrite_page_indices(
|
|
174
|
+
len(pdf_doc),
|
|
175
|
+
start_page_id=start_page_id,
|
|
176
|
+
end_page_id=end_page_id,
|
|
177
|
+
page_indices=page_indices,
|
|
178
|
+
)
|
|
179
|
+
finally:
|
|
180
|
+
close_pdfium_document(pdf_doc)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def rewrite_pdf_bytes_with_pdfium(
|
|
184
|
+
src_pdf_bytes: bytes,
|
|
185
|
+
start_page_id: int = 0,
|
|
186
|
+
end_page_id: int | None = None,
|
|
187
|
+
page_indices: Sequence[int] | None = None,
|
|
188
|
+
) -> bytes:
|
|
189
|
+
import pypdfium2 as pdfium
|
|
190
|
+
|
|
191
|
+
pdf_doc = None
|
|
192
|
+
output_doc = None
|
|
193
|
+
try:
|
|
194
|
+
with pdfium_guard():
|
|
195
|
+
pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
|
|
196
|
+
total_page_count = len(pdf_doc)
|
|
197
|
+
if total_page_count == 0:
|
|
198
|
+
return b""
|
|
199
|
+
|
|
200
|
+
normalized_page_indices = _normalize_rewrite_page_indices(
|
|
201
|
+
total_page_count,
|
|
202
|
+
start_page_id=start_page_id,
|
|
203
|
+
end_page_id=end_page_id,
|
|
204
|
+
page_indices=page_indices,
|
|
205
|
+
)
|
|
206
|
+
if not normalized_page_indices:
|
|
207
|
+
return b""
|
|
208
|
+
|
|
209
|
+
output_doc = pdfium.PdfDocument.new()
|
|
210
|
+
output_doc.import_pages(pdf_doc, normalized_page_indices)
|
|
211
|
+
|
|
212
|
+
output_buffer = BytesIO()
|
|
213
|
+
output_doc.save(output_buffer)
|
|
214
|
+
return output_buffer.getvalue()
|
|
215
|
+
finally:
|
|
216
|
+
close_pdfium_objects_safely(
|
|
217
|
+
output_doc,
|
|
218
|
+
pdf_doc,
|
|
219
|
+
owner="rewrite_pdf_bytes_with_pdfium",
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def safe_rewrite_pdf_bytes_with_pdfium(
|
|
224
|
+
src_pdf_bytes: bytes,
|
|
225
|
+
start_page_id: int = 0,
|
|
226
|
+
end_page_id: int | None = None,
|
|
227
|
+
page_indices: Sequence[int] | None = None,
|
|
228
|
+
) -> bytes:
|
|
229
|
+
"""安全重写 PDF 字节;常规重写失败时跳过损坏页并保留可加载页面。"""
|
|
230
|
+
return safe_rewrite_pdf_bytes_with_pdfium_result(
|
|
231
|
+
src_pdf_bytes,
|
|
232
|
+
start_page_id=start_page_id,
|
|
233
|
+
end_page_id=end_page_id,
|
|
234
|
+
page_indices=page_indices,
|
|
235
|
+
).pdf_bytes
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def safe_rewrite_pdf_bytes_with_pdfium_result(
|
|
239
|
+
src_pdf_bytes: bytes,
|
|
240
|
+
start_page_id: int = 0,
|
|
241
|
+
end_page_id: int | None = None,
|
|
242
|
+
page_indices: Sequence[int] | None = None,
|
|
243
|
+
) -> PdfiumRewriteResult:
|
|
244
|
+
"""安全重写 PDF 字节,并返回重写后 PDF 对应的原始页号映射。"""
|
|
245
|
+
try:
|
|
246
|
+
rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium(
|
|
247
|
+
src_pdf_bytes,
|
|
248
|
+
start_page_id=start_page_id,
|
|
249
|
+
end_page_id=end_page_id,
|
|
250
|
+
page_indices=page_indices,
|
|
251
|
+
)
|
|
252
|
+
if rebuilt_pdf_bytes:
|
|
253
|
+
retained_page_indices = _get_rewrite_page_indices_from_pdf(
|
|
254
|
+
src_pdf_bytes,
|
|
255
|
+
start_page_id=start_page_id,
|
|
256
|
+
end_page_id=end_page_id,
|
|
257
|
+
page_indices=page_indices,
|
|
258
|
+
)
|
|
259
|
+
return PdfiumRewriteResult(
|
|
260
|
+
pdf_bytes=rebuilt_pdf_bytes,
|
|
261
|
+
retained_page_indices=retained_page_indices,
|
|
262
|
+
)
|
|
263
|
+
logger.warning("PDFium rewrite returned empty bytes, trying to skip broken pages.")
|
|
264
|
+
except PdfiumFontError:
|
|
265
|
+
raise
|
|
266
|
+
except Exception as fallback_error:
|
|
267
|
+
logger.warning(f"Error in converting PDF bytes with pdfium: {fallback_error}, trying to skip broken pages.")
|
|
268
|
+
|
|
269
|
+
try:
|
|
270
|
+
if page_indices is not None:
|
|
271
|
+
requested_page_indices = sorted({int(page_index) for page_index in page_indices if int(page_index) >= 0})
|
|
272
|
+
if not requested_page_indices:
|
|
273
|
+
logger.warning("PDFium safe rewrite received no valid requested pages, using original PDF bytes.")
|
|
274
|
+
return PdfiumRewriteResult(pdf_bytes=src_pdf_bytes, retained_page_indices=None, used_original=True)
|
|
275
|
+
probe_start_page_id = requested_page_indices[0]
|
|
276
|
+
probe_end_page_id = requested_page_indices[-1]
|
|
277
|
+
else:
|
|
278
|
+
requested_page_indices = None
|
|
279
|
+
probe_start_page_id = start_page_id
|
|
280
|
+
probe_end_page_id = end_page_id
|
|
281
|
+
|
|
282
|
+
loadable_page_indices, broken_page_indices = get_loadable_pdfium_page_indices(
|
|
283
|
+
src_pdf_bytes,
|
|
284
|
+
start_page_id=probe_start_page_id,
|
|
285
|
+
end_page_id=probe_end_page_id,
|
|
286
|
+
)
|
|
287
|
+
if requested_page_indices is not None:
|
|
288
|
+
requested_page_index_set = set(requested_page_indices)
|
|
289
|
+
loadable_page_indices = [
|
|
290
|
+
page_index for page_index in loadable_page_indices if page_index in requested_page_index_set
|
|
291
|
+
]
|
|
292
|
+
broken_page_indices = [page_index for page_index in broken_page_indices if page_index in requested_page_index_set]
|
|
293
|
+
|
|
294
|
+
if broken_page_indices:
|
|
295
|
+
skipped_pages = [page_index + 1 for page_index in broken_page_indices]
|
|
296
|
+
logger.warning(f"Skipped broken PDF pages during PDFium rewrite: {skipped_pages}")
|
|
297
|
+
if not loadable_page_indices:
|
|
298
|
+
logger.warning("PDFium skip-broken-page rewrite found no loadable pages, using original PDF bytes.")
|
|
299
|
+
return PdfiumRewriteResult(
|
|
300
|
+
pdf_bytes=src_pdf_bytes,
|
|
301
|
+
retained_page_indices=None,
|
|
302
|
+
broken_page_indices=broken_page_indices,
|
|
303
|
+
used_original=True,
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium(
|
|
307
|
+
src_pdf_bytes,
|
|
308
|
+
start_page_id=probe_start_page_id,
|
|
309
|
+
end_page_id=probe_end_page_id,
|
|
310
|
+
page_indices=loadable_page_indices,
|
|
311
|
+
)
|
|
312
|
+
if rebuilt_pdf_bytes:
|
|
313
|
+
return PdfiumRewriteResult(
|
|
314
|
+
pdf_bytes=rebuilt_pdf_bytes,
|
|
315
|
+
retained_page_indices=loadable_page_indices,
|
|
316
|
+
broken_page_indices=broken_page_indices,
|
|
317
|
+
)
|
|
318
|
+
logger.warning("PDFium skip-broken-page rewrite returned empty bytes, using original PDF bytes.")
|
|
319
|
+
except PdfiumFontError:
|
|
320
|
+
raise
|
|
321
|
+
except Exception as fallback_error:
|
|
322
|
+
logger.warning(
|
|
323
|
+
f"Error in converting PDF bytes with skip-broken-page fallback: {fallback_error}, using original PDF bytes."
|
|
324
|
+
)
|
|
325
|
+
return PdfiumRewriteResult(pdf_bytes=src_pdf_bytes, retained_page_indices=None, used_original=True)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from loguru import logger
|
|
2
|
+
from PIL import Image
|
|
3
|
+
from pypdfium2 import PdfBitmap, PdfPage
|
|
4
|
+
|
|
5
|
+
from .pdfium import pdfium_guard
|
|
6
|
+
|
|
7
|
+
DEFAULT_PDF_IMAGE_DPI = 200
|
|
8
|
+
DEFAULT_MAX_RENDER_EDGE = 3500
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def estimate_page_image_bytes(page_size: tuple[float, float], dpi: int = DEFAULT_PDF_IMAGE_DPI) -> int:
|
|
12
|
+
"""按现有渲染缩放估算四通道页图字节,用于限制批量驻留内存。"""
|
|
13
|
+
import math
|
|
14
|
+
|
|
15
|
+
width, height = page_size
|
|
16
|
+
scale = min(dpi / 72, DEFAULT_MAX_RENDER_EDGE / max(width, height))
|
|
17
|
+
return math.ceil(width * scale) * math.ceil(height * scale) * 4
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def page_to_image(
|
|
21
|
+
page: PdfPage,
|
|
22
|
+
dpi: int = DEFAULT_PDF_IMAGE_DPI,
|
|
23
|
+
max_width_or_height: int = DEFAULT_MAX_RENDER_EDGE,
|
|
24
|
+
) -> tuple[Image.Image, float]:
|
|
25
|
+
"""按既有 DPI 与长边上限渲染页面,并独立持有返回图片的像素。"""
|
|
26
|
+
with pdfium_guard():
|
|
27
|
+
scale = dpi / 72
|
|
28
|
+
|
|
29
|
+
long_side_length = max(*page.get_size())
|
|
30
|
+
if (long_side_length * scale) > max_width_or_height:
|
|
31
|
+
scale = max_width_or_height / long_side_length
|
|
32
|
+
|
|
33
|
+
bitmap: PdfBitmap | None = None
|
|
34
|
+
try:
|
|
35
|
+
bitmap = page.render(scale=scale) # type: ignore
|
|
36
|
+
image = bitmap.to_pil().copy()
|
|
37
|
+
finally:
|
|
38
|
+
if bitmap is not None:
|
|
39
|
+
try:
|
|
40
|
+
bitmap.close()
|
|
41
|
+
except Exception as e:
|
|
42
|
+
logger.error(f"Failed to close bitmap: {e}")
|
|
43
|
+
return image, scale
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
__all__ = ["page_to_image"]
|