docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""跨页表格的表头、宽度和边界行结构判定。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from bs4 import Tag
|
|
8
|
+
|
|
9
|
+
from ...schema import BlockType
|
|
10
|
+
|
|
11
|
+
from .blocks import _bbox_for_calculation, _is_continuation_caption, _is_post_table_non_continuation_caption, _table_children
|
|
12
|
+
from .html import _colspan, _rowspan, calculate_row_rendered_segments
|
|
13
|
+
from .models import MAX_HEADER_ROWS, TableMergeState
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def detect_table_headers(
|
|
17
|
+
state1: TableMergeState, state2: TableMergeState, max_header_rows: int = MAX_HEADER_ROWS
|
|
18
|
+
) -> tuple[int, bool, list[list[str]]]:
|
|
19
|
+
"""检测并比较两个表格的表头,仅扫描前几行."""
|
|
20
|
+
front_rows1 = state1.front_header_info[:max_header_rows]
|
|
21
|
+
front_rows2 = state2.front_header_info[:max_header_rows]
|
|
22
|
+
|
|
23
|
+
min_rows = min(len(front_rows1), len(front_rows2), max_header_rows)
|
|
24
|
+
header_rows = 0
|
|
25
|
+
headers_match = True
|
|
26
|
+
header_texts = []
|
|
27
|
+
|
|
28
|
+
for row_idx in range(min_rows):
|
|
29
|
+
row1 = front_rows1[row_idx]
|
|
30
|
+
row2 = front_rows2[row_idx]
|
|
31
|
+
structure_match = (
|
|
32
|
+
row1.cell_count == row2.cell_count
|
|
33
|
+
and row1.effective_cols == row2.effective_cols
|
|
34
|
+
and row1.colspans == row2.colspans
|
|
35
|
+
and row1.rowspans == row2.rowspans
|
|
36
|
+
and row1.normalized_texts == row2.normalized_texts
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
if structure_match:
|
|
40
|
+
header_rows += 1
|
|
41
|
+
header_texts.append(list(row1.display_texts))
|
|
42
|
+
else:
|
|
43
|
+
headers_match = header_rows > 0
|
|
44
|
+
break
|
|
45
|
+
|
|
46
|
+
if header_rows == 0:
|
|
47
|
+
header_rows, headers_match, header_texts = _detect_table_headers_visual(state1, state2, max_header_rows=max_header_rows)
|
|
48
|
+
|
|
49
|
+
return header_rows, headers_match, header_texts
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _detect_table_headers_visual(
|
|
53
|
+
state1: TableMergeState,
|
|
54
|
+
state2: TableMergeState,
|
|
55
|
+
max_header_rows: int = MAX_HEADER_ROWS,
|
|
56
|
+
) -> tuple[int, bool, list[list[str]]]:
|
|
57
|
+
"""基于视觉一致性检测表头(只比较文本内容,忽略colspan/rowspan差异)."""
|
|
58
|
+
front_rows1 = state1.front_header_info[:max_header_rows]
|
|
59
|
+
front_rows2 = state2.front_header_info[:max_header_rows]
|
|
60
|
+
|
|
61
|
+
min_rows = min(len(front_rows1), len(front_rows2), max_header_rows)
|
|
62
|
+
header_rows = 0
|
|
63
|
+
headers_match = True
|
|
64
|
+
header_texts = []
|
|
65
|
+
|
|
66
|
+
for row_idx in range(min_rows):
|
|
67
|
+
row1 = front_rows1[row_idx]
|
|
68
|
+
row2 = front_rows2[row_idx]
|
|
69
|
+
# OCR 识别表头时可能丢失 colspan/rowspan,这里用渲染段数约束视觉一致性。
|
|
70
|
+
rendered_segments1 = calculate_row_rendered_segments(state1.rows, row_idx)
|
|
71
|
+
rendered_segments2 = calculate_row_rendered_segments(state2.rows, row_idx)
|
|
72
|
+
if row1.normalized_texts == row2.normalized_texts and rendered_segments1 == rendered_segments2:
|
|
73
|
+
header_rows += 1
|
|
74
|
+
header_texts.append(list(row1.display_texts))
|
|
75
|
+
else:
|
|
76
|
+
headers_match = header_rows > 0
|
|
77
|
+
break
|
|
78
|
+
|
|
79
|
+
if header_rows == 0:
|
|
80
|
+
headers_match = False
|
|
81
|
+
|
|
82
|
+
return header_rows, headers_match, header_texts
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _expand_header_count_by_rowspan(rows: list[Tag], header_count: int) -> int:
|
|
86
|
+
"""按表头 rowspan 覆盖范围扩展跳过行数。
|
|
87
|
+
|
|
88
|
+
跨页续表的第一行表头可能包含 rowspan。如果只跳过已匹配的首行,
|
|
89
|
+
被该 rowspan 覆盖的后续表头行会失去占位来源,合并后形成半截表头。
|
|
90
|
+
因此跳过重复表头时,需要覆盖所有由已跳过表头行跨行占据的行。
|
|
91
|
+
"""
|
|
92
|
+
if header_count <= 0 or not rows:
|
|
93
|
+
return header_count
|
|
94
|
+
|
|
95
|
+
expanded_header_count = min(header_count, len(rows))
|
|
96
|
+
row_idx = 0
|
|
97
|
+
while row_idx < expanded_header_count:
|
|
98
|
+
row = rows[row_idx]
|
|
99
|
+
for cell in row.find_all(["td", "th"]):
|
|
100
|
+
rowspan = _rowspan(cell)
|
|
101
|
+
if rowspan > 1:
|
|
102
|
+
expanded_header_count = max(expanded_header_count, row_idx + rowspan)
|
|
103
|
+
expanded_header_count = min(expanded_header_count, len(rows))
|
|
104
|
+
row_idx += 1
|
|
105
|
+
|
|
106
|
+
return expanded_header_count
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def can_merge_by_structure(
|
|
110
|
+
current_state: TableMergeState,
|
|
111
|
+
previous_state: TableMergeState,
|
|
112
|
+
current_bbox: Any = None,
|
|
113
|
+
previous_bbox: Any = None,
|
|
114
|
+
) -> bool:
|
|
115
|
+
"""仅基于表格结构判断是否可合并(不检查 caption/footnote)。
|
|
116
|
+
|
|
117
|
+
供外部工具调用,忽略 caption 和 footnote 检查。
|
|
118
|
+
"""
|
|
119
|
+
if (
|
|
120
|
+
current_bbox is not None
|
|
121
|
+
and previous_bbox is not None
|
|
122
|
+
and not _table_widths_are_compatible(
|
|
123
|
+
current_bbox,
|
|
124
|
+
previous_bbox,
|
|
125
|
+
)
|
|
126
|
+
):
|
|
127
|
+
return False
|
|
128
|
+
|
|
129
|
+
if (
|
|
130
|
+
previous_state.total_cols <= 0
|
|
131
|
+
or current_state.total_cols <= 0
|
|
132
|
+
or previous_state.last_data_row_metrics is None
|
|
133
|
+
or current_state.last_data_row_metrics is None
|
|
134
|
+
):
|
|
135
|
+
return False
|
|
136
|
+
|
|
137
|
+
if previous_state.total_cols == current_state.total_cols:
|
|
138
|
+
return True
|
|
139
|
+
|
|
140
|
+
return check_rows_match(previous_state, current_state)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _table_widths_are_compatible(current_bbox: Any, previous_bbox: Any) -> bool:
|
|
144
|
+
"""使用千分位 bbox 判断两张表的宽度相对差是否小于百分之十。"""
|
|
145
|
+
current_calc_bbox = _bbox_for_calculation(current_bbox)
|
|
146
|
+
previous_calc_bbox = _bbox_for_calculation(previous_bbox)
|
|
147
|
+
if current_calc_bbox is None or previous_calc_bbox is None:
|
|
148
|
+
return False
|
|
149
|
+
|
|
150
|
+
current_width = current_calc_bbox[2] - current_calc_bbox[0]
|
|
151
|
+
previous_width = previous_calc_bbox[2] - previous_calc_bbox[0]
|
|
152
|
+
min_width = min(current_width, previous_width)
|
|
153
|
+
return min_width > 0 and abs(current_width - previous_width) / min_width < 0.1
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def can_merge_tables(current_state: TableMergeState, previous_state: TableMergeState) -> bool:
|
|
157
|
+
"""根据 dict 表格的辅助文本、宽度和 HTML 结构判断是否可合并。"""
|
|
158
|
+
current_table_block = current_state.owner_block
|
|
159
|
+
previous_table_block = previous_state.owner_block
|
|
160
|
+
|
|
161
|
+
if not isinstance(previous_table_block, dict) or not isinstance(current_table_block, dict):
|
|
162
|
+
return False
|
|
163
|
+
|
|
164
|
+
previous_children = _table_children(previous_table_block)
|
|
165
|
+
current_children = _table_children(current_table_block)
|
|
166
|
+
footnote_count = sum(1 for block in previous_children if block.get("type") == BlockType.TABLE_FOOTNOTE)
|
|
167
|
+
caption_blocks = [block for block in current_children if block.get("type") == BlockType.TABLE_CAPTION]
|
|
168
|
+
merge_caption_blocks = [
|
|
169
|
+
block for block in caption_blocks if not _is_post_table_non_continuation_caption(current_table_block, block)
|
|
170
|
+
]
|
|
171
|
+
if merge_caption_blocks:
|
|
172
|
+
has_continuation_marker = any(_is_continuation_caption(block) for block in merge_caption_blocks)
|
|
173
|
+
|
|
174
|
+
if not has_continuation_marker:
|
|
175
|
+
return False
|
|
176
|
+
|
|
177
|
+
if footnote_count > 1:
|
|
178
|
+
return False
|
|
179
|
+
elif footnote_count > 0:
|
|
180
|
+
return False
|
|
181
|
+
|
|
182
|
+
if not _table_widths_are_compatible(current_table_block.get("bbox"), previous_table_block.get("bbox")):
|
|
183
|
+
return False
|
|
184
|
+
|
|
185
|
+
return can_merge_by_structure(current_state, previous_state)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def check_rows_match(previous_state: TableMergeState, current_state: TableMergeState) -> bool:
|
|
189
|
+
"""检查表格边界行是否匹配."""
|
|
190
|
+
last_row_metrics = previous_state.last_data_row_metrics
|
|
191
|
+
if last_row_metrics is None:
|
|
192
|
+
return False
|
|
193
|
+
|
|
194
|
+
header_count, _, _ = detect_table_headers(previous_state, current_state)
|
|
195
|
+
header_count = _expand_header_count_by_rowspan(current_state.rows, header_count)
|
|
196
|
+
first_data_row_metrics = current_state.front_first_data_row_metrics.get(header_count)
|
|
197
|
+
if first_data_row_metrics is None:
|
|
198
|
+
return False
|
|
199
|
+
|
|
200
|
+
previous_rendered_segments = calculate_row_rendered_segments(previous_state.rows, last_row_metrics.row_idx)
|
|
201
|
+
current_rendered_segments = calculate_row_rendered_segments(current_state.rows, first_data_row_metrics.row_idx)
|
|
202
|
+
|
|
203
|
+
return (
|
|
204
|
+
last_row_metrics.effective_cols == first_data_row_metrics.effective_cols
|
|
205
|
+
or last_row_metrics.actual_cols == first_data_row_metrics.actual_cols
|
|
206
|
+
or previous_rendered_segments == current_rendered_segments
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def check_row_columns_match(row1: Tag, row2: Tag) -> bool:
|
|
211
|
+
"""判断两行显式单元格数量与 colspan 结构是否一致。"""
|
|
212
|
+
cells1 = row1.find_all(["td", "th"])
|
|
213
|
+
cells2 = row2.find_all(["td", "th"])
|
|
214
|
+
if len(cells1) != len(cells2):
|
|
215
|
+
return False
|
|
216
|
+
for cell1, cell2 in zip(cells1, cells2):
|
|
217
|
+
colspan1 = _colspan(cell1)
|
|
218
|
+
colspan2 = _colspan(cell2)
|
|
219
|
+
if colspan1 != colspan2:
|
|
220
|
+
return False
|
|
221
|
+
return True
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""HTML Flash 解析使用的来源上下文契约。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from docvortex.foundation.type_identity import preserve_type_module
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True, slots=True)
|
|
11
|
+
class HtmlSourceContext:
|
|
12
|
+
"""保存相对链接解析及 HTML 解码所需的来源上下文。"""
|
|
13
|
+
|
|
14
|
+
source_uri: str | None = None
|
|
15
|
+
local_resource_root: Path | None = None
|
|
16
|
+
transport_encoding: str | None = None
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
__all__ = ["HtmlSourceContext"]
|
|
20
|
+
|
|
21
|
+
# 保持既有公开类型的 pickle 路径,所有旧、新入口指向同一个类。
|
|
22
|
+
preserve_type_module(HtmlSourceContext, "docvortex.analyzers.native.html.contracts")
|
|
@@ -0,0 +1,389 @@
|
|
|
1
|
+
"""根据文件内容和容器结构识别 DocVortex 支持的输入后缀。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from io import BytesIO
|
|
6
|
+
from functools import lru_cache
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from xml.etree import ElementTree
|
|
9
|
+
from zipfile import BadZipFile, ZipFile
|
|
10
|
+
|
|
11
|
+
from loguru import logger
|
|
12
|
+
from typing import TYPE_CHECKING
|
|
13
|
+
|
|
14
|
+
if TYPE_CHECKING:
|
|
15
|
+
from magika import Magika
|
|
16
|
+
|
|
17
|
+
from .filetypes import CSV_EXTENSIONS, HTML_EXTENSIONS, IMAGE_EXTENSIONS, rtf_header_offset
|
|
18
|
+
|
|
19
|
+
PDF_SIG_BYTES = b"%PDF"
|
|
20
|
+
OLE2_SIG_BYTES = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"
|
|
21
|
+
OOXML_ROOT_RELS = "_rels/.rels"
|
|
22
|
+
OOXML_CONTENT_TYPES = "[Content_Types].xml"
|
|
23
|
+
OOXML_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
|
|
24
|
+
OOXML_CONTENT_TYPES_NS = "http://schemas.openxmlformats.org/package/2006/content-types"
|
|
25
|
+
OOXML_OFFICE_DOCUMENT_REL = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument"
|
|
26
|
+
OOXML_MAIN_CONTENT_TYPES = {
|
|
27
|
+
("application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"): "docx",
|
|
28
|
+
("application/vnd.openxmlformats-officedocument.presentationml.presentation.main+xml"): "pptx",
|
|
29
|
+
("application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"): "xlsx",
|
|
30
|
+
}
|
|
31
|
+
ODF_MIMETYPE_SUFFIXES = {
|
|
32
|
+
"application/vnd.oasis.opendocument.text": "odt",
|
|
33
|
+
"application/vnd.oasis.opendocument.spreadsheet": "ods",
|
|
34
|
+
"application/vnd.oasis.opendocument.presentation": "odp",
|
|
35
|
+
}
|
|
36
|
+
ODF_MANIFEST_PATH = "META-INF/manifest.xml"
|
|
37
|
+
ODF_MANIFEST_NS = "urn:oasis:names:tc:opendocument:xmlns:manifest:1.0"
|
|
38
|
+
# OLE2 compound file 内部 stream 名 → 旧 Office 格式后缀
|
|
39
|
+
# doc: WordDocument stream;xls: Workbook 或 Book stream;ppt: PowerPoint Document stream
|
|
40
|
+
OLE2_STREAM_SUFFIX_MAP: dict[str, str] = {
|
|
41
|
+
"WordDocument": "doc",
|
|
42
|
+
"Workbook": "xls",
|
|
43
|
+
"Book": "xls",
|
|
44
|
+
"PowerPoint Document": "ppt",
|
|
45
|
+
}
|
|
46
|
+
_STRONG_CONTENT_SUFFIXES = frozenset(
|
|
47
|
+
{
|
|
48
|
+
"pdf",
|
|
49
|
+
"doc",
|
|
50
|
+
"docx",
|
|
51
|
+
"ppt",
|
|
52
|
+
"pptx",
|
|
53
|
+
"xls",
|
|
54
|
+
"xlsx",
|
|
55
|
+
"rtf",
|
|
56
|
+
"epub",
|
|
57
|
+
"ofd",
|
|
58
|
+
"odt",
|
|
59
|
+
"ods",
|
|
60
|
+
"odp",
|
|
61
|
+
*IMAGE_EXTENSIONS,
|
|
62
|
+
}
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@lru_cache(maxsize=1)
|
|
67
|
+
def _magika() -> Magika:
|
|
68
|
+
"""惰性创建文件类型识别器,避免导入 parser 时加载模型。"""
|
|
69
|
+
from magika import Magika
|
|
70
|
+
|
|
71
|
+
return Magika()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _strip_package_part_name(part_name: str | None) -> str:
|
|
75
|
+
"""规范化 OPC part 路径,方便匹配 Content_Types 中的 PartName。"""
|
|
76
|
+
if not part_name:
|
|
77
|
+
return ""
|
|
78
|
+
return part_name.replace("\\", "/").lstrip("/")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _ooxml_relationship_targets(root: ElementTree.Element) -> list[str]:
|
|
82
|
+
"""从根关系文件中提取 Office 主文档关系目标。"""
|
|
83
|
+
targets = []
|
|
84
|
+
for relationship in root:
|
|
85
|
+
if relationship.tag not in {
|
|
86
|
+
f"{{{OOXML_PACKAGE_REL_NS}}}Relationship",
|
|
87
|
+
"Relationship",
|
|
88
|
+
}:
|
|
89
|
+
continue
|
|
90
|
+
if relationship.get("TargetMode") == "External":
|
|
91
|
+
continue
|
|
92
|
+
if relationship.get("Type") != OOXML_OFFICE_DOCUMENT_REL:
|
|
93
|
+
continue
|
|
94
|
+
target = _strip_package_part_name(relationship.get("Target"))
|
|
95
|
+
if target:
|
|
96
|
+
targets.append(target)
|
|
97
|
+
return targets
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _ooxml_content_type_overrides(root: ElementTree.Element) -> dict[str, str]:
|
|
101
|
+
"""读取 Content_Types 中每个显式 part 的 ContentType 映射。"""
|
|
102
|
+
overrides = {}
|
|
103
|
+
for override in root:
|
|
104
|
+
if override.tag not in {
|
|
105
|
+
f"{{{OOXML_CONTENT_TYPES_NS}}}Override",
|
|
106
|
+
"Override",
|
|
107
|
+
}:
|
|
108
|
+
continue
|
|
109
|
+
part_name = _strip_package_part_name(override.get("PartName"))
|
|
110
|
+
content_type = override.get("ContentType")
|
|
111
|
+
if part_name and content_type:
|
|
112
|
+
overrides[part_name] = content_type
|
|
113
|
+
return overrides
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _guess_ooxml_suffix_from_zip(package: ZipFile) -> str | None:
|
|
117
|
+
"""根据 OOXML 包内标准主文档关系和主内容类型判断 Office 子类型。"""
|
|
118
|
+
rels_root = ElementTree.fromstring(package.read(OOXML_ROOT_RELS))
|
|
119
|
+
content_types_root = ElementTree.fromstring(package.read(OOXML_CONTENT_TYPES))
|
|
120
|
+
|
|
121
|
+
overrides = _ooxml_content_type_overrides(content_types_root)
|
|
122
|
+
for target in _ooxml_relationship_targets(rels_root):
|
|
123
|
+
suffix = OOXML_MAIN_CONTENT_TYPES.get(overrides.get(target, ""))
|
|
124
|
+
if suffix:
|
|
125
|
+
return suffix
|
|
126
|
+
return None
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _guess_ooxml_suffix_by_bytes(file_bytes: bytes) -> str | None:
|
|
130
|
+
"""优先用 OOXML 包结构识别 docx/pptx/xlsx,避免 Magika 被内嵌对象误导。"""
|
|
131
|
+
try:
|
|
132
|
+
with ZipFile(BytesIO(file_bytes)) as package:
|
|
133
|
+
return _guess_ooxml_suffix_from_zip(package)
|
|
134
|
+
except (
|
|
135
|
+
BadZipFile,
|
|
136
|
+
KeyError,
|
|
137
|
+
ElementTree.ParseError,
|
|
138
|
+
RuntimeError,
|
|
139
|
+
OSError,
|
|
140
|
+
ValueError,
|
|
141
|
+
):
|
|
142
|
+
return None
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _guess_ooxml_suffix_by_path(file_path: Path) -> str | None:
|
|
146
|
+
"""从文件路径读取 OOXML 包结构;失败时交给 Magika 原有逻辑兜底。"""
|
|
147
|
+
try:
|
|
148
|
+
with ZipFile(file_path) as package:
|
|
149
|
+
return _guess_ooxml_suffix_from_zip(package)
|
|
150
|
+
except (
|
|
151
|
+
BadZipFile,
|
|
152
|
+
KeyError,
|
|
153
|
+
ElementTree.ParseError,
|
|
154
|
+
RuntimeError,
|
|
155
|
+
OSError,
|
|
156
|
+
ValueError,
|
|
157
|
+
):
|
|
158
|
+
return None
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _guess_odf_suffix_from_zip(package: ZipFile) -> str | None:
|
|
162
|
+
"""按 ODF mimetype、manifest 根条目依次识别 odt/ods/odp。"""
|
|
163
|
+
try:
|
|
164
|
+
mimetype_info = package.getinfo("mimetype")
|
|
165
|
+
if mimetype_info.file_size <= 256:
|
|
166
|
+
mimetype = package.read(mimetype_info).decode("ascii", errors="strict").strip()
|
|
167
|
+
if suffix := ODF_MIMETYPE_SUFFIXES.get(mimetype):
|
|
168
|
+
return suffix
|
|
169
|
+
except (KeyError, UnicodeDecodeError, RuntimeError, OSError, ValueError):
|
|
170
|
+
pass
|
|
171
|
+
try:
|
|
172
|
+
manifest_info = package.getinfo(ODF_MANIFEST_PATH)
|
|
173
|
+
if manifest_info.file_size > 1024 * 1024:
|
|
174
|
+
return None
|
|
175
|
+
root = ElementTree.fromstring(package.read(manifest_info))
|
|
176
|
+
except (KeyError, ElementTree.ParseError, RuntimeError, OSError, ValueError):
|
|
177
|
+
return None
|
|
178
|
+
for entry in root.iter(f"{{{ODF_MANIFEST_NS}}}file-entry"):
|
|
179
|
+
if entry.get(f"{{{ODF_MANIFEST_NS}}}full-path") != "/":
|
|
180
|
+
continue
|
|
181
|
+
media_type = entry.get(f"{{{ODF_MANIFEST_NS}}}media-type", "").strip()
|
|
182
|
+
return ODF_MIMETYPE_SUFFIXES.get(media_type)
|
|
183
|
+
return None
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _guess_odf_suffix_by_bytes(file_bytes: bytes) -> str | None:
|
|
187
|
+
"""从内存 ZIP 包识别 ODF,失败时不影响后续 OLE/Magika/CSV 路由。"""
|
|
188
|
+
try:
|
|
189
|
+
with ZipFile(BytesIO(file_bytes)) as package:
|
|
190
|
+
return _guess_odf_suffix_from_zip(package)
|
|
191
|
+
except (BadZipFile, RuntimeError, OSError, ValueError):
|
|
192
|
+
return None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _guess_odf_suffix_by_path(file_path: Path) -> str | None:
|
|
196
|
+
"""从路径 ZIP 包识别 ODF,保持现有 OOXML 检测优先级。"""
|
|
197
|
+
try:
|
|
198
|
+
with ZipFile(file_path) as package:
|
|
199
|
+
return _guess_odf_suffix_from_zip(package)
|
|
200
|
+
except (BadZipFile, RuntimeError, OSError, ValueError):
|
|
201
|
+
return None
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _guess_epub_suffix_by_bytes(file_bytes: bytes) -> str | None:
|
|
205
|
+
"""从内存 ZIP 包验证 EPUB 强内容身份。"""
|
|
206
|
+
from ..analyzers.native.epub import detect_epub
|
|
207
|
+
|
|
208
|
+
return "epub" if detect_epub(file_bytes) else None
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _guess_epub_suffix_by_path(file_path: Path) -> str | None:
|
|
212
|
+
"""从路径 ZIP 包验证 EPUB 强内容身份。"""
|
|
213
|
+
from ..analyzers.native.epub import detect_epub_path
|
|
214
|
+
|
|
215
|
+
return "epub" if detect_epub_path(file_path) else None
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _guess_ofd_suffix_by_bytes(file_bytes: bytes) -> str | None:
|
|
219
|
+
"""从内存 ZIP 包验证 OFD 强内容身份。"""
|
|
220
|
+
from ..analyzers.native.ofd import detect_ofd
|
|
221
|
+
|
|
222
|
+
return "ofd" if detect_ofd(file_bytes) else None
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _guess_ofd_suffix_by_path(file_path: Path) -> str | None:
|
|
226
|
+
"""从路径 ZIP 包验证 OFD 强内容身份。"""
|
|
227
|
+
from ..analyzers.native.ofd import detect_ofd_path
|
|
228
|
+
|
|
229
|
+
return "ofd" if detect_ofd_path(file_path) else None
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _guess_ole2_suffix_by_bytes(file_bytes: bytes) -> str | None:
|
|
233
|
+
"""用 OLE2 magic + olefile 内部 stream 区分 doc/xls/ppt。
|
|
234
|
+
|
|
235
|
+
olefile 是纯 Python 库且已是核心依赖(mineru.model.flash.office.legacy 使用)。
|
|
236
|
+
在 OOXML 识别失败后、Magika 兜底前插入此层,避免 Magika 对 OLE2 返回 unknown。
|
|
237
|
+
"""
|
|
238
|
+
if len(file_bytes) < 8 or file_bytes[:8] != OLE2_SIG_BYTES:
|
|
239
|
+
return None
|
|
240
|
+
try:
|
|
241
|
+
import olefile # type: ignore[import-untyped]
|
|
242
|
+
|
|
243
|
+
with olefile.OleFileIO(BytesIO(file_bytes)) as ole:
|
|
244
|
+
for stream_name in ole.listdir(streams=True):
|
|
245
|
+
name = "/".join(stream_name)
|
|
246
|
+
suffix = OLE2_STREAM_SUFFIX_MAP.get(name)
|
|
247
|
+
if suffix:
|
|
248
|
+
return suffix
|
|
249
|
+
except Exception:
|
|
250
|
+
return None
|
|
251
|
+
return None
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _guess_ole2_suffix_by_path(file_path: Path) -> str | None:
|
|
255
|
+
"""从文件路径读取 OLE2 容器并识别旧 Office 格式。"""
|
|
256
|
+
try:
|
|
257
|
+
with open(file_path, "rb") as f:
|
|
258
|
+
return _guess_ole2_suffix_by_bytes(f.read())
|
|
259
|
+
except OSError:
|
|
260
|
+
return None
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _has_pdf_signature_by_path(file_path: Path) -> bool:
|
|
264
|
+
"""读取文件头判断路径指向的内容是否具有 PDF 强签名。"""
|
|
265
|
+
try:
|
|
266
|
+
with open(file_path, "rb") as file:
|
|
267
|
+
return file.read(len(PDF_SIG_BYTES)) == PDF_SIG_BYTES
|
|
268
|
+
except OSError:
|
|
269
|
+
return False
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _has_rtf_signature_by_path(file_path: Path) -> bool:
|
|
273
|
+
"""读取有限文件头并按共享规则识别 RTF 根组。"""
|
|
274
|
+
try:
|
|
275
|
+
with open(file_path, "rb") as file:
|
|
276
|
+
return rtf_header_offset(file.read(128)) is not None
|
|
277
|
+
except OSError:
|
|
278
|
+
return False
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _resolve_signatureless_csv_suffix(detected_suffix: str, file_path: str | Path | None) -> str:
|
|
282
|
+
"""以 .csv/.tsv 扩展名兜底无签名分隔文本,并保留强内容类型的优先级。"""
|
|
283
|
+
extension = Path(file_path).suffix.lower().lstrip(".") if file_path else ""
|
|
284
|
+
if extension in CSV_EXTENSIONS:
|
|
285
|
+
if detected_suffix in _STRONG_CONTENT_SUFFIXES:
|
|
286
|
+
return detected_suffix
|
|
287
|
+
return "csv"
|
|
288
|
+
if detected_suffix == "csv":
|
|
289
|
+
if extension in ODF_MIMETYPE_SUFFIXES.values():
|
|
290
|
+
return "txt"
|
|
291
|
+
return extension or "txt"
|
|
292
|
+
return detected_suffix
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _resolve_signatureless_html_suffix(detected_suffix: str, file_path: str | Path | None) -> str:
|
|
296
|
+
"""用 HTML_EXTENSIONS 兜底短文本,并把 Magika 的 HTML 结果统一规范为 html。"""
|
|
297
|
+
extension = Path(file_path).suffix.lower().lstrip(".") if file_path else ""
|
|
298
|
+
if extension in HTML_EXTENSIONS and detected_suffix not in _STRONG_CONTENT_SUFFIXES:
|
|
299
|
+
return "html"
|
|
300
|
+
return "html" if detected_suffix == "html" else detected_suffix
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _reject_unverified_package_suffix(detected_suffix: str) -> str:
|
|
304
|
+
"""拒绝未通过包身份验证、仅由启发式工具猜出的 ODF/EPUB/OFD 类型。"""
|
|
305
|
+
package_suffixes = {*ODF_MIMETYPE_SUFFIXES.values(), "epub", "ofd"}
|
|
306
|
+
return "unknown" if detected_suffix in package_suffixes else detected_suffix
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def guess_suffix_by_bytes(file_bytes: bytes, file_path: str | None = None) -> str:
|
|
310
|
+
if file_bytes[: len(PDF_SIG_BYTES)] == PDF_SIG_BYTES:
|
|
311
|
+
return "pdf"
|
|
312
|
+
if rtf_header_offset(file_bytes[:128]) is not None:
|
|
313
|
+
return "rtf"
|
|
314
|
+
|
|
315
|
+
ofd_suffix = _guess_ofd_suffix_by_bytes(file_bytes)
|
|
316
|
+
if ofd_suffix:
|
|
317
|
+
return ofd_suffix
|
|
318
|
+
|
|
319
|
+
epub_suffix = _guess_epub_suffix_by_bytes(file_bytes)
|
|
320
|
+
if epub_suffix:
|
|
321
|
+
return epub_suffix
|
|
322
|
+
|
|
323
|
+
ooxml_suffix = _guess_ooxml_suffix_by_bytes(file_bytes)
|
|
324
|
+
if ooxml_suffix:
|
|
325
|
+
return ooxml_suffix
|
|
326
|
+
|
|
327
|
+
odf_suffix = _guess_odf_suffix_by_bytes(file_bytes)
|
|
328
|
+
if odf_suffix:
|
|
329
|
+
return odf_suffix
|
|
330
|
+
|
|
331
|
+
ole2_suffix = _guess_ole2_suffix_by_bytes(file_bytes)
|
|
332
|
+
if ole2_suffix:
|
|
333
|
+
return ole2_suffix
|
|
334
|
+
|
|
335
|
+
suffix = _magika().identify_bytes(file_bytes).prediction.output.label
|
|
336
|
+
if (
|
|
337
|
+
file_path
|
|
338
|
+
and suffix in ["ai", "html"]
|
|
339
|
+
and Path(file_path).suffix.lower() in [".pdf"]
|
|
340
|
+
and file_bytes[:4] == PDF_SIG_BYTES
|
|
341
|
+
):
|
|
342
|
+
suffix = "pdf"
|
|
343
|
+
suffix = _resolve_signatureless_csv_suffix(_reject_unverified_package_suffix(suffix), file_path)
|
|
344
|
+
return _resolve_signatureless_html_suffix(suffix, file_path)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def guess_suffix_by_path(file_path: str | Path) -> str:
|
|
348
|
+
if not isinstance(file_path, Path):
|
|
349
|
+
file_path = Path(file_path)
|
|
350
|
+
|
|
351
|
+
if _has_rtf_signature_by_path(file_path):
|
|
352
|
+
return "rtf"
|
|
353
|
+
|
|
354
|
+
ofd_suffix = _guess_ofd_suffix_by_path(file_path)
|
|
355
|
+
if ofd_suffix:
|
|
356
|
+
return ofd_suffix
|
|
357
|
+
|
|
358
|
+
epub_suffix = _guess_epub_suffix_by_path(file_path)
|
|
359
|
+
if epub_suffix:
|
|
360
|
+
return epub_suffix
|
|
361
|
+
|
|
362
|
+
ooxml_suffix = _guess_ooxml_suffix_by_path(file_path)
|
|
363
|
+
if ooxml_suffix:
|
|
364
|
+
return ooxml_suffix
|
|
365
|
+
|
|
366
|
+
odf_suffix = _guess_odf_suffix_by_path(file_path)
|
|
367
|
+
if odf_suffix:
|
|
368
|
+
return odf_suffix
|
|
369
|
+
|
|
370
|
+
ole2_suffix = _guess_ole2_suffix_by_path(file_path)
|
|
371
|
+
if ole2_suffix:
|
|
372
|
+
return ole2_suffix
|
|
373
|
+
|
|
374
|
+
if _has_pdf_signature_by_path(file_path):
|
|
375
|
+
return "pdf"
|
|
376
|
+
|
|
377
|
+
suffix = _magika().identify_path(file_path).prediction.output.label
|
|
378
|
+
if suffix in ["ai", "html"] and file_path.suffix.lower() in [".pdf"]:
|
|
379
|
+
try:
|
|
380
|
+
with open(file_path, "rb") as f:
|
|
381
|
+
if f.read(4) == PDF_SIG_BYTES:
|
|
382
|
+
suffix = "pdf"
|
|
383
|
+
except Exception as e:
|
|
384
|
+
logger.warning(f"Failed to read file {file_path} for PDF signature check: {e}")
|
|
385
|
+
suffix = _resolve_signatureless_csv_suffix(_reject_unverified_package_suffix(suffix), file_path)
|
|
386
|
+
return _resolve_signatureless_html_suffix(suffix, file_path)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
__all__ = ["guess_suffix_by_bytes", "guess_suffix_by_path"]
|