docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""PDF 页面坐标、方向和裁图使用的无状态几何原语。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import base64
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import cv2
|
|
9
|
+
import numpy as np
|
|
10
|
+
from loguru import logger
|
|
11
|
+
|
|
12
|
+
from ...schema import BBox
|
|
13
|
+
from ...foundation.geometry import normalize_to_int_bbox
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _normalize_page_size(page_image: Any) -> tuple[int, int]:
|
|
17
|
+
"""从PIL或numpy图像中读取页面宽高,供归一化bbox还原为像素bbox。"""
|
|
18
|
+
if hasattr(page_image, "size"):
|
|
19
|
+
return page_image.size
|
|
20
|
+
|
|
21
|
+
height, width = page_image.shape[:2]
|
|
22
|
+
return width, height
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _bbox_to_pixel_bbox(bbox: BBox | None, page_size: tuple[int, int]) -> BBox | None:
|
|
26
|
+
"""将归一化或像素bbox统一成像素bbox,异常bbox返回None。"""
|
|
27
|
+
if bbox is None or len(bbox) != 4:
|
|
28
|
+
return None
|
|
29
|
+
|
|
30
|
+
try:
|
|
31
|
+
x0, y0, x1, y1 = [float(v) for v in bbox]
|
|
32
|
+
except (TypeError, ValueError):
|
|
33
|
+
return None
|
|
34
|
+
|
|
35
|
+
width, height = page_size
|
|
36
|
+
if all(0.0 <= value <= 1.0 for value in [x0, y0, x1, y1]):
|
|
37
|
+
x0, y0, x1, y1 = x0 * width, y0 * height, x1 * width, y1 * height
|
|
38
|
+
|
|
39
|
+
left, right = sorted([x0, x1])
|
|
40
|
+
top, bottom = sorted([y0, y1])
|
|
41
|
+
if right <= left or bottom <= top:
|
|
42
|
+
return None
|
|
43
|
+
return (left, top, right, bottom)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _normalize_layout_bbox_to_unit(bbox: BBox | None, page_size: tuple[int, int]) -> list[float] | None:
|
|
47
|
+
"""将 layout 像素 bbox 归一化为 VLM ContentBlock 需要的 0-1 坐标。"""
|
|
48
|
+
pixel_bbox = _bbox_to_pixel_bbox(bbox, page_size)
|
|
49
|
+
if pixel_bbox is None:
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
page_width, page_height = page_size
|
|
53
|
+
if page_width <= 0 or page_height <= 0:
|
|
54
|
+
return None
|
|
55
|
+
|
|
56
|
+
x0, y0, x1, y1 = pixel_bbox
|
|
57
|
+
unit_bbox = [
|
|
58
|
+
round(max(0.0, min(1.0, float(x0) / page_width)), 3),
|
|
59
|
+
round(max(0.0, min(1.0, float(y0) / page_height)), 3),
|
|
60
|
+
round(max(0.0, min(1.0, float(x1) / page_width)), 3),
|
|
61
|
+
round(max(0.0, min(1.0, float(y1) / page_height)), 3),
|
|
62
|
+
]
|
|
63
|
+
if unit_bbox[2] <= unit_bbox[0] or unit_bbox[3] <= unit_bbox[1]:
|
|
64
|
+
return None
|
|
65
|
+
return unit_bbox
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _medium_bbox_to_quad(bbox: list[float] | tuple[float, ...]) -> np.ndarray:
|
|
69
|
+
"""将普通 bbox 转为表格模型 OCR token 使用的四点框。"""
|
|
70
|
+
x0, y0, x1, y1 = [float(v) for v in bbox]
|
|
71
|
+
return np.asarray([[x0, y0], [x1, y0], [x1, y1], [x0, y1]], dtype=np.float32)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _normalize_medium_content(value: Any) -> str:
|
|
75
|
+
"""将 medium 本地模型输出的文本字段规范成 Hybrid block 可消费的字符串。"""
|
|
76
|
+
if isinstance(value, list):
|
|
77
|
+
return "\n".join(str(item) for item in value if str(item).strip())
|
|
78
|
+
if isinstance(value, str):
|
|
79
|
+
return value.strip()
|
|
80
|
+
return ""
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _table_bbox_center(bbox: BBox) -> tuple[float, float]:
|
|
84
|
+
"""计算 bbox 中心点,用于判断图片或公式应归属哪个表格。"""
|
|
85
|
+
return (float(bbox[0]) + float(bbox[2])) / 2.0, (float(bbox[1]) + float(bbox[3])) / 2.0
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _normalize_visual_block_angle(angle: Any) -> int:
|
|
89
|
+
"""规范视觉块角度为 0/90/180/270,无法识别的角度按 0 处理。"""
|
|
90
|
+
try:
|
|
91
|
+
normalized_angle = int(float(angle or 0)) % 360
|
|
92
|
+
except (TypeError, ValueError):
|
|
93
|
+
logger.warning(f"Unsupported visual block angle: {angle}, using 0")
|
|
94
|
+
return 0
|
|
95
|
+
if normalized_angle not in {0, 90, 180, 270}:
|
|
96
|
+
logger.warning(f"Unsupported visual block angle: {angle}, using 0")
|
|
97
|
+
return 0
|
|
98
|
+
return normalized_angle
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _rotate_visual_block_image_to_upright(image: np.ndarray, angle: int) -> np.ndarray:
|
|
102
|
+
"""按 layout 视觉块角度把裁图旋转至正向,角度语义与方向分类模型保持一致。"""
|
|
103
|
+
if angle == 270:
|
|
104
|
+
return cv2.rotate(image, cv2.ROTATE_90_CLOCKWISE)
|
|
105
|
+
if angle == 90:
|
|
106
|
+
return cv2.rotate(image, cv2.ROTATE_90_COUNTERCLOCKWISE)
|
|
107
|
+
if angle == 180:
|
|
108
|
+
return cv2.rotate(image, cv2.ROTATE_180)
|
|
109
|
+
return image
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _rotate_medium_table_bbox(
|
|
113
|
+
bbox: BBox,
|
|
114
|
+
image_width: float,
|
|
115
|
+
image_height: float,
|
|
116
|
+
angle: int,
|
|
117
|
+
) -> BBox:
|
|
118
|
+
"""把原表格裁图中的 bbox 同步转换到旋转后裁图坐标系。"""
|
|
119
|
+
x0, y0, x1, y1 = [float(value) for value in bbox]
|
|
120
|
+
if angle == 270:
|
|
121
|
+
# 顺时针旋转 90 度后,新 x 轴来自原 y 轴的反方向。
|
|
122
|
+
return (image_height - y1, x0, image_height - y0, x1)
|
|
123
|
+
if angle == 90:
|
|
124
|
+
# 逆时针旋转 90 度后,新 y 轴来自原 x 轴的反方向。
|
|
125
|
+
return (y0, image_width - x1, y1, image_width - x0)
|
|
126
|
+
if angle == 180:
|
|
127
|
+
return (image_width - x1, image_height - y1, image_width - x0, image_height - y0)
|
|
128
|
+
return (x0, y0, x1, y1)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _get_medium_table_virtual_image_bbox(
|
|
132
|
+
bbox: BBox,
|
|
133
|
+
image_size: tuple[int, int],
|
|
134
|
+
box_size: float = 10.0,
|
|
135
|
+
) -> BBox:
|
|
136
|
+
"""在图片中心生成小 OCR token 框,避免图片大框干扰单元格匹配。"""
|
|
137
|
+
image_width, image_height = image_size
|
|
138
|
+
center_x, center_y = _table_bbox_center(bbox)
|
|
139
|
+
half_size = box_size / 2.0
|
|
140
|
+
return (
|
|
141
|
+
max(0.0, center_x - half_size),
|
|
142
|
+
max(0.0, center_y - half_size),
|
|
143
|
+
min(float(image_width), center_x + half_size),
|
|
144
|
+
min(float(image_height), center_y + half_size),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _encode_page_crop_as_jpeg_data_uri(
|
|
149
|
+
np_image: np.ndarray,
|
|
150
|
+
page_bbox: BBox,
|
|
151
|
+
angle: int,
|
|
152
|
+
) -> str:
|
|
153
|
+
"""从页面原图按像素框裁剪,按视觉块方向回正后编码为 JPEG data URI。"""
|
|
154
|
+
image_h, image_w = np_image.shape[:2]
|
|
155
|
+
image_bbox = normalize_to_int_bbox(page_bbox, image_size=(image_h, image_w))
|
|
156
|
+
if image_bbox is None:
|
|
157
|
+
return ""
|
|
158
|
+
x0, y0, x1, y1 = image_bbox
|
|
159
|
+
crop_rgb = np_image[y0:y1, x0:x1].copy()
|
|
160
|
+
if crop_rgb.size == 0:
|
|
161
|
+
return ""
|
|
162
|
+
|
|
163
|
+
crop_rgb = _rotate_visual_block_image_to_upright(crop_rgb, angle)
|
|
164
|
+
crop_bgr = cv2.cvtColor(crop_rgb, cv2.COLOR_RGB2BGR)
|
|
165
|
+
success, encoded = cv2.imencode(".jpg", crop_bgr)
|
|
166
|
+
if not success:
|
|
167
|
+
return ""
|
|
168
|
+
return f"data:image/jpeg;base64,{base64.b64encode(encoded.tobytes()).decode('ascii')}"
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _sidecar_bbox_to_page_bbox(
|
|
172
|
+
bbox: BBox | None,
|
|
173
|
+
page_size: tuple[float, float],
|
|
174
|
+
render_scale: float,
|
|
175
|
+
) -> BBox | None:
|
|
176
|
+
"""将公式或 OCR sidecar bbox 转为 PDF point 坐标,供原生字符匹配和组行。"""
|
|
177
|
+
if bbox is None or len(bbox) != 4 or render_scale <= 0:
|
|
178
|
+
return None
|
|
179
|
+
try:
|
|
180
|
+
x0, y0, x1, y1 = [float(value) for value in bbox]
|
|
181
|
+
except (TypeError, ValueError):
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
page_width, page_height = page_size
|
|
185
|
+
if page_width <= 0 or page_height <= 0:
|
|
186
|
+
return None
|
|
187
|
+
if all(0.0 <= value <= 1.0 for value in [x0, y0, x1, y1]):
|
|
188
|
+
x0, y0, x1, y1 = x0 * page_width, y0 * page_height, x1 * page_width, y1 * page_height
|
|
189
|
+
else:
|
|
190
|
+
x0, y0, x1, y1 = (
|
|
191
|
+
x0 / render_scale,
|
|
192
|
+
y0 / render_scale,
|
|
193
|
+
x1 / render_scale,
|
|
194
|
+
y1 / render_scale,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
left, right = sorted([max(0.0, min(page_width, x0)), max(0.0, min(page_width, x1))])
|
|
198
|
+
top, bottom = sorted([max(0.0, min(page_height, y0)), max(0.0, min(page_height, y1))])
|
|
199
|
+
if right <= left or bottom <= top:
|
|
200
|
+
return None
|
|
201
|
+
return (left, top, right, bottom)
|
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
"""视觉块容器补全、方向归一化与页面裁图。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from typing import TYPE_CHECKING, Any
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
from loguru import logger
|
|
10
|
+
|
|
11
|
+
from ...schema import BBox, BlockType
|
|
12
|
+
from ...foundation.geometry import calculate_overlap_area_2_minbox_area_ratio, calculate_overlap_area_in_bbox1_area_ratio
|
|
13
|
+
|
|
14
|
+
from .constants import (
|
|
15
|
+
IMAGE_BLOCK_CONTAINMENT_THRESHOLD,
|
|
16
|
+
IMAGE_BLOCK_LAYOUT_COVERAGE_THRESHOLD,
|
|
17
|
+
IMAGE_BLOCK_LAYOUT_MIN_VISUAL_COUNT,
|
|
18
|
+
LOCAL_LAYOUT_IMAGE_BLOCK_AREA_TYPES,
|
|
19
|
+
LOCAL_LAYOUT_IMAGE_BLOCK_BODY_TYPES,
|
|
20
|
+
MODEL_JSON_VISUAL_BLOCK_TYPES,
|
|
21
|
+
)
|
|
22
|
+
from .visual_geometry import (
|
|
23
|
+
_bbox_to_pixel_bbox,
|
|
24
|
+
_encode_page_crop_as_jpeg_data_uri,
|
|
25
|
+
_normalize_page_size,
|
|
26
|
+
_normalize_visual_block_angle,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from .document import PDFDocument
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _normalize_model_bbox_for_containment(raw_bbox: Any) -> BBox | None:
|
|
34
|
+
"""校验模型 block 的四点框,返回可用于面积包含判断的浮点坐标。"""
|
|
35
|
+
try:
|
|
36
|
+
if raw_bbox is None or len(raw_bbox) != 4:
|
|
37
|
+
return None
|
|
38
|
+
bbox = tuple(float(value) for value in raw_bbox)
|
|
39
|
+
except (TypeError, ValueError):
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
if not all(math.isfinite(value) for value in bbox):
|
|
43
|
+
return None
|
|
44
|
+
if bbox[2] <= bbox[0] or bbox[3] <= bbox[1]:
|
|
45
|
+
return None
|
|
46
|
+
return bbox
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _collapse_image_blocks(
|
|
50
|
+
page_model_list: list[dict[str, Any]],
|
|
51
|
+
containment_threshold: float = IMAGE_BLOCK_CONTAINMENT_THRESHOLD,
|
|
52
|
+
) -> None:
|
|
53
|
+
"""将 image_block 折叠为单个图片,并删除面积上被其包裹的非容器子块。"""
|
|
54
|
+
image_blocks = [block for block in page_model_list if block.get("type") == "image_block"]
|
|
55
|
+
if not image_blocks:
|
|
56
|
+
return
|
|
57
|
+
|
|
58
|
+
image_block_ids = {id(block) for block in image_blocks}
|
|
59
|
+
image_block_bboxes = [
|
|
60
|
+
bbox for block in image_blocks if (bbox := _normalize_model_bbox_for_containment(block.get("bbox"))) is not None
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
retained_blocks: list[dict[str, Any]] = []
|
|
64
|
+
for block in page_model_list:
|
|
65
|
+
if id(block) in image_block_ids:
|
|
66
|
+
block["type"] = BlockType.IMAGE
|
|
67
|
+
retained_blocks.append(block)
|
|
68
|
+
continue
|
|
69
|
+
|
|
70
|
+
block_bbox = _normalize_model_bbox_for_containment(block.get("bbox"))
|
|
71
|
+
is_contained = block_bbox is not None and any(
|
|
72
|
+
calculate_overlap_area_in_bbox1_area_ratio(block_bbox, image_block_bbox) >= containment_threshold
|
|
73
|
+
for image_block_bbox in image_block_bboxes
|
|
74
|
+
)
|
|
75
|
+
if not is_contained:
|
|
76
|
+
retained_blocks.append(block)
|
|
77
|
+
|
|
78
|
+
page_model_list[:] = retained_blocks
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _supplement_missing_image_block_containers(
|
|
82
|
+
model_list: list[list[dict[str, Any]]],
|
|
83
|
+
layout_blocks_list: list[list[dict[str, Any]]],
|
|
84
|
+
containment_threshold: float = IMAGE_BLOCK_CONTAINMENT_THRESHOLD,
|
|
85
|
+
coverage_threshold: float = IMAGE_BLOCK_LAYOUT_COVERAGE_THRESHOLD,
|
|
86
|
+
min_visual_count: int = IMAGE_BLOCK_LAYOUT_MIN_VISUAL_COUNT,
|
|
87
|
+
) -> None:
|
|
88
|
+
"""用本地 layout 整图框为 xhigh 结果补充缺失的 image_block 容器。"""
|
|
89
|
+
if len(model_list) != len(layout_blocks_list):
|
|
90
|
+
raise ValueError(
|
|
91
|
+
"Hybrid image-block fallback page count mismatch: "
|
|
92
|
+
f"model_list={len(model_list)}, layout_blocks={len(layout_blocks_list)}"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
for page_model_list, page_layout_blocks in zip(model_list, layout_blocks_list):
|
|
96
|
+
existing_image_block_bboxes = [
|
|
97
|
+
bbox
|
|
98
|
+
for block in page_model_list
|
|
99
|
+
if block.get("type") == "image_block"
|
|
100
|
+
if (bbox := _normalize_model_bbox_for_containment(block.get("bbox"))) is not None
|
|
101
|
+
]
|
|
102
|
+
|
|
103
|
+
existing_claimed_block_ids: set[int] = set()
|
|
104
|
+
if existing_image_block_bboxes:
|
|
105
|
+
for block in page_model_list:
|
|
106
|
+
if block.get("type") == "image_block":
|
|
107
|
+
continue
|
|
108
|
+
block_bbox = _normalize_model_bbox_for_containment(block.get("bbox"))
|
|
109
|
+
if block_bbox is not None and any(
|
|
110
|
+
calculate_overlap_area_in_bbox1_area_ratio(block_bbox, image_block_bbox) >= containment_threshold
|
|
111
|
+
for image_block_bbox in existing_image_block_bboxes
|
|
112
|
+
):
|
|
113
|
+
existing_claimed_block_ids.add(id(block))
|
|
114
|
+
|
|
115
|
+
candidates: list[tuple[int, float, int, int, dict[str, Any], set[int]]] = []
|
|
116
|
+
for layout_order, layout_block in enumerate(page_layout_blocks):
|
|
117
|
+
if layout_block.get("type") != BlockType.IMAGE or layout_block.get("sub_type") == "seal":
|
|
118
|
+
continue
|
|
119
|
+
|
|
120
|
+
layout_bbox = _normalize_model_bbox_for_containment(layout_block.get("bbox"))
|
|
121
|
+
if layout_bbox is None:
|
|
122
|
+
continue
|
|
123
|
+
if any(
|
|
124
|
+
calculate_overlap_area_2_minbox_area_ratio(layout_bbox, image_block_bbox) >= containment_threshold
|
|
125
|
+
for image_block_bbox in existing_image_block_bboxes
|
|
126
|
+
):
|
|
127
|
+
continue
|
|
128
|
+
|
|
129
|
+
contained_blocks: list[tuple[int, dict[str, Any], BBox]] = []
|
|
130
|
+
for block_index, block in enumerate(page_model_list):
|
|
131
|
+
if block.get("type") == "image_block" or id(block) in existing_claimed_block_ids:
|
|
132
|
+
continue
|
|
133
|
+
block_bbox = _normalize_model_bbox_for_containment(block.get("bbox"))
|
|
134
|
+
if block_bbox is None:
|
|
135
|
+
continue
|
|
136
|
+
if calculate_overlap_area_in_bbox1_area_ratio(block_bbox, layout_bbox) >= containment_threshold:
|
|
137
|
+
contained_blocks.append((block_index, block, block_bbox))
|
|
138
|
+
|
|
139
|
+
contained_visuals = [
|
|
140
|
+
(block_index, block)
|
|
141
|
+
for block_index, block, _ in contained_blocks
|
|
142
|
+
if block.get("type") in LOCAL_LAYOUT_IMAGE_BLOCK_BODY_TYPES
|
|
143
|
+
]
|
|
144
|
+
if len(contained_visuals) < min_visual_count:
|
|
145
|
+
continue
|
|
146
|
+
|
|
147
|
+
layout_area = (layout_bbox[2] - layout_bbox[0]) * (layout_bbox[3] - layout_bbox[1])
|
|
148
|
+
contained_area = sum(
|
|
149
|
+
(block_bbox[2] - block_bbox[0]) * (block_bbox[3] - block_bbox[1])
|
|
150
|
+
for _, block, block_bbox in contained_blocks
|
|
151
|
+
if block.get("type") in LOCAL_LAYOUT_IMAGE_BLOCK_AREA_TYPES
|
|
152
|
+
)
|
|
153
|
+
coverage_ratio = contained_area / layout_area
|
|
154
|
+
if coverage_ratio < coverage_threshold and not math.isclose(
|
|
155
|
+
coverage_ratio,
|
|
156
|
+
coverage_threshold,
|
|
157
|
+
rel_tol=0.0,
|
|
158
|
+
abs_tol=1e-12,
|
|
159
|
+
):
|
|
160
|
+
continue
|
|
161
|
+
|
|
162
|
+
contained_block_ids = {id(block) for _, block, _ in contained_blocks}
|
|
163
|
+
first_block_index = min(block_index for block_index, _, _ in contained_blocks)
|
|
164
|
+
candidates.append(
|
|
165
|
+
(
|
|
166
|
+
-len(contained_visuals),
|
|
167
|
+
layout_area,
|
|
168
|
+
layout_order,
|
|
169
|
+
first_block_index,
|
|
170
|
+
layout_block,
|
|
171
|
+
contained_block_ids,
|
|
172
|
+
)
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
claimed_block_ids: set[int] = set()
|
|
176
|
+
selected_containers: list[tuple[int, dict[str, Any]]] = []
|
|
177
|
+
for _, _, _, first_block_index, layout_block, block_ids in sorted(candidates):
|
|
178
|
+
if claimed_block_ids.intersection(block_ids):
|
|
179
|
+
continue
|
|
180
|
+
claimed_block_ids.update(block_ids)
|
|
181
|
+
selected_containers.append(
|
|
182
|
+
(
|
|
183
|
+
first_block_index,
|
|
184
|
+
{
|
|
185
|
+
"type": "image_block",
|
|
186
|
+
"bbox": list(layout_block["bbox"]),
|
|
187
|
+
"angle": layout_block.get("angle", 0),
|
|
188
|
+
"content": None,
|
|
189
|
+
},
|
|
190
|
+
)
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
for insert_index, image_block in sorted(selected_containers, reverse=True):
|
|
194
|
+
page_model_list.insert(insert_index, image_block)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _attach_visual_block_images(
|
|
198
|
+
model_list: list[list[dict[str, Any]]],
|
|
199
|
+
images_list: list[dict[str, Any]],
|
|
200
|
+
page_start_index: int = 0,
|
|
201
|
+
) -> None:
|
|
202
|
+
"""在窗口页图释放前,为最终 model_list 视觉块写入回正后的页面裁图。"""
|
|
203
|
+
if len(model_list) != len(images_list):
|
|
204
|
+
raise ValueError(f"Hybrid visual crop page count mismatch: model_list={len(model_list)}, images={len(images_list)}")
|
|
205
|
+
|
|
206
|
+
for page_offset, (page_model_list, image_dict) in enumerate(zip(model_list, images_list)):
|
|
207
|
+
_attach_prepared_visual_block_images(
|
|
208
|
+
[_prepare_page_visual_blocks(page_model_list)], [image_dict], page_start_index + page_offset
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _prepare_page_visual_blocks(page_model_list: list[dict[str, Any]]) -> list[tuple[int, dict[str, Any]]]:
|
|
213
|
+
"""先折叠视觉容器,再记录原始块索引,供按需页图任务复用。"""
|
|
214
|
+
_collapse_image_blocks(page_model_list)
|
|
215
|
+
return [(index, block) for index, block in enumerate(page_model_list) if block.get("type") in MODEL_JSON_VISUAL_BLOCK_TYPES]
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _visual_page_ranges(
|
|
219
|
+
prepared_pages: list[list[tuple[int, dict[str, Any]]]],
|
|
220
|
+
image_bytes_by_page: dict[int, int] | None = None,
|
|
221
|
+
*,
|
|
222
|
+
window_size: int = 64,
|
|
223
|
+
) -> list[tuple[int, int]]:
|
|
224
|
+
"""在指定窗口和 32MiB 像素预算内合并需裁图页,不改变单页清晰度。"""
|
|
225
|
+
ranges: list[tuple[int, int]] = []
|
|
226
|
+
batch_bytes = 0
|
|
227
|
+
for page_index, blocks in enumerate(prepared_pages):
|
|
228
|
+
if not blocks:
|
|
229
|
+
continue
|
|
230
|
+
page_bytes = (image_bytes_by_page or {}).get(page_index, 0)
|
|
231
|
+
if (
|
|
232
|
+
ranges
|
|
233
|
+
and page_index == ranges[-1][1] + 1
|
|
234
|
+
and page_index // window_size == ranges[-1][0] // window_size
|
|
235
|
+
and batch_bytes + page_bytes <= 32 * 1024 * 1024
|
|
236
|
+
):
|
|
237
|
+
ranges[-1] = (ranges[-1][0], page_index)
|
|
238
|
+
batch_bytes += page_bytes
|
|
239
|
+
else:
|
|
240
|
+
ranges.append((page_index, page_index))
|
|
241
|
+
batch_bytes = page_bytes
|
|
242
|
+
return ranges
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def attach_visual_block_images_from_pdf(
|
|
246
|
+
document: PDFDocument,
|
|
247
|
+
model_list: list[list[dict[str, Any]]],
|
|
248
|
+
*,
|
|
249
|
+
window_size: int = 64,
|
|
250
|
+
timeout: int | None = None,
|
|
251
|
+
threads: int | None = None,
|
|
252
|
+
) -> None:
|
|
253
|
+
"""按当前 PDF 全部物理页的视觉块需求原地补图;页图由本函数释放,文档仍归调用方。"""
|
|
254
|
+
from .images import load_images_from_pdf_bytes_range
|
|
255
|
+
from .raster import estimate_page_image_bytes
|
|
256
|
+
|
|
257
|
+
if isinstance(window_size, bool) or not isinstance(window_size, int) or window_size <= 0:
|
|
258
|
+
raise ValueError("window_size must be a positive integer")
|
|
259
|
+
if len(model_list) != document.page_count:
|
|
260
|
+
raise ValueError(f"PDF visual crop page count mismatch: model_list={len(model_list)}, document={document.page_count}")
|
|
261
|
+
|
|
262
|
+
prepared_visuals = [_prepare_page_visual_blocks(page) for page in model_list]
|
|
263
|
+
image_bytes = {
|
|
264
|
+
index: estimate_page_image_bytes(document.page_size(index)) for index, blocks in enumerate(prepared_visuals) if blocks
|
|
265
|
+
}
|
|
266
|
+
for start, end in _visual_page_ranges(prepared_visuals, image_bytes, window_size=window_size):
|
|
267
|
+
images = load_images_from_pdf_bytes_range(
|
|
268
|
+
document.bytes,
|
|
269
|
+
start_page_id=start,
|
|
270
|
+
end_page_id=end,
|
|
271
|
+
image_type="pil_img",
|
|
272
|
+
timeout=timeout,
|
|
273
|
+
threads=threads,
|
|
274
|
+
)
|
|
275
|
+
try:
|
|
276
|
+
_attach_prepared_visual_block_images(prepared_visuals[start : end + 1], images, page_start_index=start)
|
|
277
|
+
finally:
|
|
278
|
+
for item in images:
|
|
279
|
+
if item.get("img_pil") is not None:
|
|
280
|
+
item["img_pil"].close()
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _attach_prepared_visual_block_images(
|
|
284
|
+
prepared_pages: list[list[tuple[int, dict[str, Any]]]],
|
|
285
|
+
images_list: list[dict[str, Any]],
|
|
286
|
+
page_start_index: int = 0,
|
|
287
|
+
) -> None:
|
|
288
|
+
"""按所选 PDF 的物理页索引裁图,输入块已经整理且无需再次折叠。"""
|
|
289
|
+
if len(prepared_pages) != len(images_list):
|
|
290
|
+
raise ValueError(f"Hybrid visual crop page count mismatch: model_list={len(prepared_pages)}, images={len(images_list)}")
|
|
291
|
+
for page_offset, (visual_blocks, image_dict) in enumerate(zip(prepared_pages, images_list)):
|
|
292
|
+
if not visual_blocks:
|
|
293
|
+
continue
|
|
294
|
+
|
|
295
|
+
page_index = page_start_index + page_offset
|
|
296
|
+
page_pil_image = image_dict.get("img_pil")
|
|
297
|
+
if page_pil_image is None:
|
|
298
|
+
logger.warning(f"Skipping model visual block crops without page image: page={page_index}")
|
|
299
|
+
continue
|
|
300
|
+
|
|
301
|
+
converted_page_image = None
|
|
302
|
+
try:
|
|
303
|
+
if getattr(page_pil_image, "mode", None) == "RGB":
|
|
304
|
+
page_rgb_image = page_pil_image
|
|
305
|
+
else:
|
|
306
|
+
converted_page_image = page_pil_image.convert("RGB")
|
|
307
|
+
page_rgb_image = converted_page_image
|
|
308
|
+
|
|
309
|
+
page_size = _normalize_page_size(page_rgb_image)
|
|
310
|
+
np_image = np.asarray(page_rgb_image)
|
|
311
|
+
for block_idx, block in visual_blocks:
|
|
312
|
+
try:
|
|
313
|
+
pixel_bbox = _bbox_to_pixel_bbox(block.get("bbox"), page_size)
|
|
314
|
+
if pixel_bbox is None:
|
|
315
|
+
raise ValueError("invalid bbox")
|
|
316
|
+
angle = _normalize_visual_block_angle(block.get("angle", 0))
|
|
317
|
+
image_base64 = _encode_page_crop_as_jpeg_data_uri(
|
|
318
|
+
np_image,
|
|
319
|
+
pixel_bbox,
|
|
320
|
+
angle,
|
|
321
|
+
)
|
|
322
|
+
if not image_base64:
|
|
323
|
+
raise ValueError("empty crop or JPEG encoding failure")
|
|
324
|
+
block["image_base64"] = image_base64
|
|
325
|
+
except Exception as exc:
|
|
326
|
+
logger.warning(
|
|
327
|
+
"Skipping invalid model visual block crop: "
|
|
328
|
+
f"page={page_index}, block={block_idx}, type={block.get('type')}, "
|
|
329
|
+
f"bbox={block.get('bbox')}, error={exc}"
|
|
330
|
+
)
|
|
331
|
+
finally:
|
|
332
|
+
if converted_page_image is not None:
|
|
333
|
+
converted_page_image.close()
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
attach_visual_block_images = _attach_visual_block_images
|
|
337
|
+
supplement_missing_image_block_containers = _supplement_missing_image_block_containers
|
|
338
|
+
|
|
339
|
+
__all__ = [
|
|
340
|
+
"attach_visual_block_images",
|
|
341
|
+
"attach_visual_block_images_from_pdf",
|
|
342
|
+
"supplement_missing_image_block_containers",
|
|
343
|
+
]
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""文件输入准备与有效页范围,不包含档位或 OCR 路由。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import TYPE_CHECKING, cast
|
|
8
|
+
|
|
9
|
+
from docvortex.document.contracts import HtmlSourceContext
|
|
10
|
+
from ..errors import InvalidRequestError
|
|
11
|
+
from ..schema import FILE_SUFFIXES, FileSuffix
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from .pdf.document import PDFDocument
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(slots=True)
|
|
18
|
+
class PreparedSource:
|
|
19
|
+
"""保存输入字节、有效页面映射及底层文档的资源所有权。"""
|
|
20
|
+
|
|
21
|
+
data: bytes
|
|
22
|
+
file_suffix: FileSuffix
|
|
23
|
+
source_context: HtmlSourceContext | None = None
|
|
24
|
+
page_index_map: list[int] | None = None
|
|
25
|
+
broken_page_indices: tuple[int, ...] = ()
|
|
26
|
+
document: PDFDocument | None = None
|
|
27
|
+
owns_document: bool = False
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def prepare_source(
|
|
31
|
+
source: str | Path | bytes | PDFDocument,
|
|
32
|
+
*,
|
|
33
|
+
file_suffix: FileSuffix | None = None,
|
|
34
|
+
page_range: str = "",
|
|
35
|
+
source_context: HtmlSourceContext | None = None,
|
|
36
|
+
) -> PreparedSource:
|
|
37
|
+
"""准备原生解析输入,调用者持有的 PDFDocument 不由引擎关闭。"""
|
|
38
|
+
from .page_range import normalize_page_range_input, parse_page_range
|
|
39
|
+
|
|
40
|
+
document = None
|
|
41
|
+
path = Path(source) if isinstance(source, (str, Path)) else None
|
|
42
|
+
if path is not None:
|
|
43
|
+
data = path.read_bytes()
|
|
44
|
+
elif isinstance(source, bytes):
|
|
45
|
+
data = source
|
|
46
|
+
else:
|
|
47
|
+
from .pdf.document import PDFDocument
|
|
48
|
+
|
|
49
|
+
if not isinstance(source, PDFDocument):
|
|
50
|
+
raise TypeError("source must be a path, bytes, or PDFDocument")
|
|
51
|
+
document, data, file_suffix = source, source.bytes, "pdf"
|
|
52
|
+
if file_suffix:
|
|
53
|
+
suffix = file_suffix
|
|
54
|
+
else:
|
|
55
|
+
from .detection import guess_suffix_by_bytes
|
|
56
|
+
|
|
57
|
+
suffix = guess_suffix_by_bytes(data, str(path) if path else None)
|
|
58
|
+
if suffix not in FILE_SUFFIXES:
|
|
59
|
+
raise InvalidRequestError("file_type_unsupported", f"Unsupported native input format: {suffix}", "file_suffix")
|
|
60
|
+
page_range = normalize_page_range_input(page_range)
|
|
61
|
+
if suffix != "pdf" and page_range:
|
|
62
|
+
raise InvalidRequestError("page_range_invalid", "Page ranges are supported only for PDF input", "page_range")
|
|
63
|
+
if suffix == "html" and source_context is None and path is not None:
|
|
64
|
+
absolute = path.resolve()
|
|
65
|
+
source_context = HtmlSourceContext(source_uri=absolute.as_uri(), local_resource_root=absolute.parent)
|
|
66
|
+
prepared = PreparedSource(data, cast(FileSuffix, suffix), source_context=source_context, document=document)
|
|
67
|
+
if suffix == "pdf":
|
|
68
|
+
from .pdf.document import PDFDocument
|
|
69
|
+
from .pdf.pdfium import safe_rewrite_pdf_bytes_with_pdfium_result
|
|
70
|
+
|
|
71
|
+
if document is None:
|
|
72
|
+
document = PDFDocument(data)
|
|
73
|
+
prepared.document, prepared.owns_document = document, True
|
|
74
|
+
try:
|
|
75
|
+
indices = parse_page_range(page_range, document.page_count)
|
|
76
|
+
if indices != list(range(document.page_count)):
|
|
77
|
+
rewrite = safe_rewrite_pdf_bytes_with_pdfium_result(data, page_indices=indices)
|
|
78
|
+
prepared.data = rewrite.pdf_bytes or data
|
|
79
|
+
prepared.page_index_map = None if rewrite.used_original else rewrite.retained_page_indices
|
|
80
|
+
prepared.broken_page_indices = tuple(rewrite.broken_page_indices)
|
|
81
|
+
if prepared.owns_document:
|
|
82
|
+
document.close()
|
|
83
|
+
prepared.document = PDFDocument(prepared.data)
|
|
84
|
+
prepared.owns_document = True
|
|
85
|
+
except Exception:
|
|
86
|
+
if prepared.owns_document and prepared.document is not None:
|
|
87
|
+
prepared.document.close()
|
|
88
|
+
raise
|
|
89
|
+
return prepared
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
__all__ = ["PreparedSource", "prepare_source", "HtmlSourceContext"]
|
docvortex/errors.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""独立文档引擎的可定位错误,不依赖宿主 HTTP 或任务状态。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class DocumentError(ValueError):
|
|
7
|
+
"""携带稳定错误码的文档输入或处理错误。"""
|
|
8
|
+
|
|
9
|
+
def __init__(self, code: str, message: str, param: str | None = None) -> None:
|
|
10
|
+
"""保存供命令行及上层应用映射的错误信息。"""
|
|
11
|
+
super().__init__(message)
|
|
12
|
+
self.code = code
|
|
13
|
+
self.message = message
|
|
14
|
+
self.param = param
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class InvalidRequestError(DocumentError):
|
|
18
|
+
"""表示页范围或格式选项不符合文档输入契约。"""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
__all__ = ["DocumentError", "InvalidRequestError"]
|