docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
docvortex/schema.py
ADDED
|
@@ -0,0 +1,1141 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import math
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Annotated, Any, ClassVar, Literal, TypeAlias, TypeVar, Union, cast, get_args
|
|
8
|
+
|
|
9
|
+
from pydantic import (
|
|
10
|
+
BaseModel,
|
|
11
|
+
ConfigDict,
|
|
12
|
+
Field,
|
|
13
|
+
TypeAdapter,
|
|
14
|
+
field_validator,
|
|
15
|
+
model_validator,
|
|
16
|
+
model_serializer,
|
|
17
|
+
JsonValue,
|
|
18
|
+
SerializationInfo,
|
|
19
|
+
SerializerFunctionWrapHandler,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
from .foundation.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
|
|
23
|
+
|
|
24
|
+
# 这些字符串不能作为公开 Block.type discriminator,只用于 raw 阶段或 Block 内部枚举值。
|
|
25
|
+
RawBlockType: TypeAlias = Literal[
|
|
26
|
+
"algorithm",
|
|
27
|
+
"caption",
|
|
28
|
+
"footnote",
|
|
29
|
+
"formula_number",
|
|
30
|
+
"phonetic",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
RAW_ALGORITHM: RawBlockType = "algorithm"
|
|
34
|
+
RAW_CAPTION: RawBlockType = "caption"
|
|
35
|
+
RAW_FOOTNOTE: RawBlockType = "footnote"
|
|
36
|
+
RAW_FORMULA_NUMBER: RawBlockType = "formula_number"
|
|
37
|
+
RAW_PHONETIC: RawBlockType = "phonetic"
|
|
38
|
+
|
|
39
|
+
RAW_ONLY_BLOCK_TYPES = frozenset(
|
|
40
|
+
{
|
|
41
|
+
RAW_ALGORITHM,
|
|
42
|
+
RAW_CAPTION,
|
|
43
|
+
RAW_FOOTNOTE,
|
|
44
|
+
RAW_FORMULA_NUMBER,
|
|
45
|
+
RAW_PHONETIC,
|
|
46
|
+
}
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
FileSuffix: TypeAlias = Literal[
|
|
50
|
+
"pdf",
|
|
51
|
+
"doc",
|
|
52
|
+
"docx",
|
|
53
|
+
"ppt",
|
|
54
|
+
"pptx",
|
|
55
|
+
"xls",
|
|
56
|
+
"xlsx",
|
|
57
|
+
"rtf",
|
|
58
|
+
"csv",
|
|
59
|
+
"epub",
|
|
60
|
+
"html",
|
|
61
|
+
"ofd",
|
|
62
|
+
"odt",
|
|
63
|
+
"ods",
|
|
64
|
+
"odp",
|
|
65
|
+
]
|
|
66
|
+
FILE_SUFFIXES: frozenset[FileSuffix] = frozenset(cast(tuple[FileSuffix, ...], get_args(FileSuffix)))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class BlockType(str, Enum):
|
|
70
|
+
IMAGE = "image"
|
|
71
|
+
IMAGE_BODY = "image_body"
|
|
72
|
+
IMAGE_CAPTION = "image_caption"
|
|
73
|
+
IMAGE_FOOTNOTE = "image_footnote"
|
|
74
|
+
|
|
75
|
+
TABLE = "table"
|
|
76
|
+
TABLE_BODY = "table_body"
|
|
77
|
+
TABLE_CAPTION = "table_caption"
|
|
78
|
+
TABLE_FOOTNOTE = "table_footnote"
|
|
79
|
+
|
|
80
|
+
CHART = "chart"
|
|
81
|
+
CHART_BODY = "chart_body"
|
|
82
|
+
CHART_CAPTION = "chart_caption"
|
|
83
|
+
CHART_FOOTNOTE = "chart_footnote"
|
|
84
|
+
|
|
85
|
+
# Added in vlm 2.5
|
|
86
|
+
CODE = "code"
|
|
87
|
+
CODE_BODY = "code_body"
|
|
88
|
+
ALGORITHM_BODY = "algorithm_body"
|
|
89
|
+
CODE_CAPTION = "code_caption"
|
|
90
|
+
CODE_FOOTNOTE = "code_footnote"
|
|
91
|
+
|
|
92
|
+
TEXT = "text"
|
|
93
|
+
EQUATION = "equation" # 行间公式(独立公式)
|
|
94
|
+
LIST = "list"
|
|
95
|
+
INDEX = "index"
|
|
96
|
+
|
|
97
|
+
# Added in vlm 2.5
|
|
98
|
+
REF_TEXT = "ref_text"
|
|
99
|
+
HEADER = "header"
|
|
100
|
+
FOOTER = "footer"
|
|
101
|
+
PAGE_NUMBER = "page_number"
|
|
102
|
+
ASIDE_TEXT = "aside_text"
|
|
103
|
+
PAGE_FOOTNOTE = "page_footnote"
|
|
104
|
+
|
|
105
|
+
# Added in pp_doclayout_v2
|
|
106
|
+
DOC_TITLE = "doc_title"
|
|
107
|
+
PARAGRAPH_TITLE = "paragraph_title"
|
|
108
|
+
|
|
109
|
+
def __str__(self) -> str:
|
|
110
|
+
return self.value
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
BlockTypes = Literal[
|
|
114
|
+
BlockType.IMAGE,
|
|
115
|
+
BlockType.TABLE,
|
|
116
|
+
BlockType.CHART,
|
|
117
|
+
BlockType.IMAGE_BODY,
|
|
118
|
+
BlockType.TABLE_BODY,
|
|
119
|
+
BlockType.CHART_BODY,
|
|
120
|
+
BlockType.IMAGE_CAPTION,
|
|
121
|
+
BlockType.TABLE_CAPTION,
|
|
122
|
+
BlockType.CHART_CAPTION,
|
|
123
|
+
BlockType.IMAGE_FOOTNOTE,
|
|
124
|
+
BlockType.TABLE_FOOTNOTE,
|
|
125
|
+
BlockType.CHART_FOOTNOTE,
|
|
126
|
+
BlockType.TEXT,
|
|
127
|
+
BlockType.EQUATION,
|
|
128
|
+
BlockType.LIST,
|
|
129
|
+
BlockType.INDEX,
|
|
130
|
+
BlockType.CODE,
|
|
131
|
+
BlockType.CODE_BODY,
|
|
132
|
+
BlockType.ALGORITHM_BODY,
|
|
133
|
+
BlockType.CODE_CAPTION,
|
|
134
|
+
BlockType.CODE_FOOTNOTE,
|
|
135
|
+
BlockType.REF_TEXT,
|
|
136
|
+
BlockType.HEADER,
|
|
137
|
+
BlockType.FOOTER,
|
|
138
|
+
BlockType.PAGE_NUMBER,
|
|
139
|
+
BlockType.ASIDE_TEXT,
|
|
140
|
+
BlockType.PAGE_FOOTNOTE,
|
|
141
|
+
BlockType.DOC_TITLE,
|
|
142
|
+
BlockType.PARAGRAPH_TITLE,
|
|
143
|
+
]
|
|
144
|
+
|
|
145
|
+
PageBlockTypes = Literal[
|
|
146
|
+
BlockType.IMAGE,
|
|
147
|
+
BlockType.TABLE,
|
|
148
|
+
BlockType.CHART,
|
|
149
|
+
BlockType.TEXT,
|
|
150
|
+
BlockType.EQUATION,
|
|
151
|
+
BlockType.LIST,
|
|
152
|
+
BlockType.INDEX,
|
|
153
|
+
BlockType.CODE,
|
|
154
|
+
BlockType.REF_TEXT,
|
|
155
|
+
BlockType.HEADER,
|
|
156
|
+
BlockType.FOOTER,
|
|
157
|
+
BlockType.PAGE_NUMBER,
|
|
158
|
+
BlockType.ASIDE_TEXT,
|
|
159
|
+
BlockType.PAGE_FOOTNOTE,
|
|
160
|
+
BlockType.DOC_TITLE,
|
|
161
|
+
BlockType.PARAGRAPH_TITLE,
|
|
162
|
+
]
|
|
163
|
+
|
|
164
|
+
BLOCK_TYPES = {
|
|
165
|
+
BlockType.IMAGE,
|
|
166
|
+
BlockType.TABLE,
|
|
167
|
+
BlockType.CHART,
|
|
168
|
+
BlockType.IMAGE_BODY,
|
|
169
|
+
BlockType.TABLE_BODY,
|
|
170
|
+
BlockType.CHART_BODY,
|
|
171
|
+
BlockType.IMAGE_CAPTION,
|
|
172
|
+
BlockType.TABLE_CAPTION,
|
|
173
|
+
BlockType.CHART_CAPTION,
|
|
174
|
+
BlockType.IMAGE_FOOTNOTE,
|
|
175
|
+
BlockType.TABLE_FOOTNOTE,
|
|
176
|
+
BlockType.CHART_FOOTNOTE,
|
|
177
|
+
BlockType.TEXT,
|
|
178
|
+
BlockType.EQUATION,
|
|
179
|
+
BlockType.LIST,
|
|
180
|
+
BlockType.INDEX,
|
|
181
|
+
BlockType.CODE,
|
|
182
|
+
BlockType.CODE_BODY,
|
|
183
|
+
BlockType.ALGORITHM_BODY,
|
|
184
|
+
BlockType.CODE_CAPTION,
|
|
185
|
+
BlockType.CODE_FOOTNOTE,
|
|
186
|
+
BlockType.REF_TEXT,
|
|
187
|
+
BlockType.HEADER,
|
|
188
|
+
BlockType.FOOTER,
|
|
189
|
+
BlockType.PAGE_NUMBER,
|
|
190
|
+
BlockType.ASIDE_TEXT,
|
|
191
|
+
BlockType.PAGE_FOOTNOTE,
|
|
192
|
+
BlockType.DOC_TITLE,
|
|
193
|
+
BlockType.PARAGRAPH_TITLE,
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
PAGE_BLOCK_TYPES = {
|
|
197
|
+
BlockType.IMAGE,
|
|
198
|
+
BlockType.TABLE,
|
|
199
|
+
BlockType.CHART,
|
|
200
|
+
BlockType.TEXT,
|
|
201
|
+
BlockType.EQUATION,
|
|
202
|
+
BlockType.LIST,
|
|
203
|
+
BlockType.INDEX,
|
|
204
|
+
BlockType.CODE,
|
|
205
|
+
BlockType.REF_TEXT,
|
|
206
|
+
BlockType.HEADER,
|
|
207
|
+
BlockType.FOOTER,
|
|
208
|
+
BlockType.PAGE_NUMBER,
|
|
209
|
+
BlockType.ASIDE_TEXT,
|
|
210
|
+
BlockType.PAGE_FOOTNOTE,
|
|
211
|
+
BlockType.DOC_TITLE,
|
|
212
|
+
BlockType.PARAGRAPH_TITLE,
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
# 页面装饰与辅助文本不参与正文、列表和视觉对象之间的语义边界判断。
|
|
216
|
+
PAGE_AUXILIARY_BLOCK_TYPES = {
|
|
217
|
+
BlockType.HEADER,
|
|
218
|
+
BlockType.FOOTER,
|
|
219
|
+
BlockType.PAGE_NUMBER,
|
|
220
|
+
BlockType.ASIDE_TEXT,
|
|
221
|
+
}
|
|
222
|
+
# 页面脚注需要参与输出,但不会阻断正文、列表、续表或视觉对象之间的关系判断。
|
|
223
|
+
MERGE_TRANSPARENT_BLOCK_TYPES = {
|
|
224
|
+
*PAGE_AUXILIARY_BLOCK_TYPES,
|
|
225
|
+
BlockType.PAGE_FOOTNOTE,
|
|
226
|
+
}
|
|
227
|
+
VISUAL_RELATION_IGNORED_TYPES = MERGE_TRANSPARENT_BLOCK_TYPES
|
|
228
|
+
VISUAL_MAIN_TYPES = {
|
|
229
|
+
BlockType.IMAGE_BODY: BlockType.IMAGE,
|
|
230
|
+
BlockType.TABLE_BODY: BlockType.TABLE,
|
|
231
|
+
BlockType.CHART_BODY: BlockType.CHART,
|
|
232
|
+
BlockType.CODE_BODY: BlockType.CODE,
|
|
233
|
+
}
|
|
234
|
+
VISUAL_TYPE_MAPPING = {
|
|
235
|
+
BlockType.IMAGE: {
|
|
236
|
+
"body": BlockType.IMAGE_BODY,
|
|
237
|
+
"caption": BlockType.IMAGE_CAPTION,
|
|
238
|
+
"footnote": BlockType.IMAGE_FOOTNOTE,
|
|
239
|
+
},
|
|
240
|
+
BlockType.TABLE: {
|
|
241
|
+
"body": BlockType.TABLE_BODY,
|
|
242
|
+
"caption": BlockType.TABLE_CAPTION,
|
|
243
|
+
"footnote": BlockType.TABLE_FOOTNOTE,
|
|
244
|
+
},
|
|
245
|
+
BlockType.CHART: {
|
|
246
|
+
"body": BlockType.CHART_BODY,
|
|
247
|
+
"caption": BlockType.CHART_CAPTION,
|
|
248
|
+
"footnote": BlockType.CHART_FOOTNOTE,
|
|
249
|
+
},
|
|
250
|
+
BlockType.CODE: {
|
|
251
|
+
"body": BlockType.CODE_BODY,
|
|
252
|
+
"caption": BlockType.CODE_CAPTION,
|
|
253
|
+
"footnote": BlockType.CODE_FOOTNOTE,
|
|
254
|
+
},
|
|
255
|
+
}
|
|
256
|
+
# ── model types ─────────────────────────────────────────────────────
|
|
257
|
+
|
|
258
|
+
BBox: TypeAlias = tuple[float, float, float, float]
|
|
259
|
+
IntBBox: TypeAlias = tuple[int, int, int, int]
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _remove_block_fields(value: Any, excluded_fields: set[str]) -> Any:
|
|
263
|
+
"""递归删除序列化结果中指定的 block 字段,覆盖任意深度的容器。"""
|
|
264
|
+
if isinstance(value, list):
|
|
265
|
+
return [_remove_block_fields(item, excluded_fields) for item in value]
|
|
266
|
+
if not isinstance(value, dict):
|
|
267
|
+
return value
|
|
268
|
+
|
|
269
|
+
result = {
|
|
270
|
+
key: _remove_block_fields(item, excluded_fields)
|
|
271
|
+
for key, item in value.items()
|
|
272
|
+
if not ("type" in value and key in excluded_fields)
|
|
273
|
+
}
|
|
274
|
+
return result
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
class _StrictMiddleModel(BaseModel):
|
|
278
|
+
"""Model/Middle JSON 严格模型基类,提供无副作用的统一序列化入口。"""
|
|
279
|
+
|
|
280
|
+
model_config = ConfigDict(extra="forbid", strict=True, validate_assignment=True)
|
|
281
|
+
|
|
282
|
+
def to_dict(
|
|
283
|
+
self,
|
|
284
|
+
*,
|
|
285
|
+
skip_defaults: bool = True,
|
|
286
|
+
exclude_none: bool = False,
|
|
287
|
+
exclude_block_fields: set[str] | None = None,
|
|
288
|
+
) -> dict[str, Any]:
|
|
289
|
+
"""序列化对象,并按字段名递归排除任意层级的 block 字段。"""
|
|
290
|
+
payload = self.model_dump(
|
|
291
|
+
mode="json",
|
|
292
|
+
exclude_defaults=skip_defaults,
|
|
293
|
+
exclude_none=exclude_none,
|
|
294
|
+
)
|
|
295
|
+
if exclude_block_fields:
|
|
296
|
+
payload = _remove_block_fields(payload, set(exclude_block_fields))
|
|
297
|
+
return payload
|
|
298
|
+
|
|
299
|
+
def to_json(
|
|
300
|
+
self,
|
|
301
|
+
*,
|
|
302
|
+
skip_defaults: bool = True,
|
|
303
|
+
exclude_none: bool = False,
|
|
304
|
+
exclude_block_fields: set[str] | None = None,
|
|
305
|
+
indent: int | None = 4,
|
|
306
|
+
) -> str:
|
|
307
|
+
"""将对象编码为 UTF-8 友好的 JSON 字符串,不执行图片文件写入。"""
|
|
308
|
+
return json.dumps(
|
|
309
|
+
self.to_dict(
|
|
310
|
+
skip_defaults=skip_defaults,
|
|
311
|
+
exclude_none=exclude_none,
|
|
312
|
+
exclude_block_fields=exclude_block_fields,
|
|
313
|
+
),
|
|
314
|
+
ensure_ascii=False,
|
|
315
|
+
indent=indent,
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
InlineStyle: TypeAlias = Literal[
|
|
320
|
+
"bold",
|
|
321
|
+
"italic",
|
|
322
|
+
"underline",
|
|
323
|
+
"emphasis",
|
|
324
|
+
"strikethrough",
|
|
325
|
+
"superscript",
|
|
326
|
+
"subscript",
|
|
327
|
+
]
|
|
328
|
+
|
|
329
|
+
INLINE_STYLE_ORDER: tuple[InlineStyle, ...] = (
|
|
330
|
+
"bold",
|
|
331
|
+
"italic",
|
|
332
|
+
"underline",
|
|
333
|
+
"emphasis",
|
|
334
|
+
"strikethrough",
|
|
335
|
+
"superscript",
|
|
336
|
+
"subscript",
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
class TextSpan(_StrictMiddleModel):
|
|
341
|
+
"""保存普通行内文字及其可见字体样式。"""
|
|
342
|
+
|
|
343
|
+
type: Literal["text"]
|
|
344
|
+
content: str = Field(min_length=1)
|
|
345
|
+
styles: list[InlineStyle] = Field(default_factory=list)
|
|
346
|
+
|
|
347
|
+
@field_validator("styles")
|
|
348
|
+
@classmethod
|
|
349
|
+
def _normalize_styles(cls, value: list[InlineStyle]) -> list[InlineStyle]:
|
|
350
|
+
"""按公开固定顺序去重样式,并禁止同时声明上下标。"""
|
|
351
|
+
unique = set(value)
|
|
352
|
+
if "superscript" in unique and "subscript" in unique:
|
|
353
|
+
raise ValueError("text span cannot be both superscript and subscript")
|
|
354
|
+
return [style for style in INLINE_STYLE_ORDER if style in unique]
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
class EquationInlineSpan(_StrictMiddleModel):
|
|
358
|
+
"""保存不含外层定界符的行内 LaTeX。"""
|
|
359
|
+
|
|
360
|
+
type: Literal["equation_inline"]
|
|
361
|
+
content: str = Field(min_length=1)
|
|
362
|
+
|
|
363
|
+
@field_validator("content")
|
|
364
|
+
@classmethod
|
|
365
|
+
def _validate_content(cls, value: str) -> str:
|
|
366
|
+
"""拒绝只包含空白的行内公式,同时保留公式原始空白。"""
|
|
367
|
+
if not value.strip():
|
|
368
|
+
raise ValueError("inline equation content must not be blank")
|
|
369
|
+
return value
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
class CodeInlineSpan(_StrictMiddleModel):
|
|
373
|
+
"""保存需要按字面量显示的行内代码。"""
|
|
374
|
+
|
|
375
|
+
type: Literal["code_inline"]
|
|
376
|
+
content: str = Field(min_length=1)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
NonLinkInlineSpan: TypeAlias = Annotated[
|
|
380
|
+
Union[TextSpan, EquationInlineSpan, CodeInlineSpan],
|
|
381
|
+
Field(discriminator="type"),
|
|
382
|
+
]
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
class HyperlinkSpan(_StrictMiddleModel):
|
|
386
|
+
"""保存安全超链接目标及其非链接行内子节点。"""
|
|
387
|
+
|
|
388
|
+
type: Literal["hyperlink"]
|
|
389
|
+
url: str = Field(min_length=1)
|
|
390
|
+
content: list[NonLinkInlineSpan] = Field(min_length=1)
|
|
391
|
+
|
|
392
|
+
@field_validator("url")
|
|
393
|
+
@classmethod
|
|
394
|
+
def _validate_url(cls, value: str) -> str:
|
|
395
|
+
"""复用统一策略拒绝危险协议、本地路径、畸形 URL 和控制字符。"""
|
|
396
|
+
normalized = sanitize_hyperlink_target(
|
|
397
|
+
value,
|
|
398
|
+
allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
|
|
399
|
+
allow_relative=True,
|
|
400
|
+
allow_fragment=True,
|
|
401
|
+
)
|
|
402
|
+
if normalized is None:
|
|
403
|
+
raise ValueError("hyperlink span url is unsafe or malformed")
|
|
404
|
+
return normalized
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
InlineSpan: TypeAlias = Annotated[
|
|
408
|
+
Union[TextSpan, EquationInlineSpan, CodeInlineSpan, HyperlinkSpan],
|
|
409
|
+
Field(discriminator="type"),
|
|
410
|
+
]
|
|
411
|
+
|
|
412
|
+
INLINE_SPAN_ADAPTER = TypeAdapter(InlineSpan)
|
|
413
|
+
INLINE_SPAN_LIST_ADAPTER = TypeAdapter(list[InlineSpan])
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _normalize_typed_inline_spans(spans: list[InlineSpan]) -> list[InlineSpan]:
|
|
417
|
+
"""递归合并相邻同样式文字及相邻同目标链接。"""
|
|
418
|
+
normalized: list[InlineSpan] = []
|
|
419
|
+
for span in spans:
|
|
420
|
+
current: InlineSpan
|
|
421
|
+
if isinstance(span, HyperlinkSpan):
|
|
422
|
+
children = _normalize_typed_inline_spans(list(span.content))
|
|
423
|
+
non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
|
|
424
|
+
if not non_link_children:
|
|
425
|
+
continue
|
|
426
|
+
current = span.model_copy(update={"content": non_link_children}, deep=True)
|
|
427
|
+
else:
|
|
428
|
+
current = span.model_copy(deep=True)
|
|
429
|
+
if (
|
|
430
|
+
normalized
|
|
431
|
+
and isinstance(normalized[-1], TextSpan)
|
|
432
|
+
and isinstance(current, TextSpan)
|
|
433
|
+
and normalized[-1].styles == current.styles
|
|
434
|
+
):
|
|
435
|
+
previous = normalized[-1]
|
|
436
|
+
normalized[-1] = previous.model_copy(update={"content": f"{previous.content}{current.content}"})
|
|
437
|
+
continue
|
|
438
|
+
if (
|
|
439
|
+
normalized
|
|
440
|
+
and isinstance(normalized[-1], HyperlinkSpan)
|
|
441
|
+
and isinstance(current, HyperlinkSpan)
|
|
442
|
+
and normalized[-1].url == current.url
|
|
443
|
+
):
|
|
444
|
+
previous_link = normalized[-1]
|
|
445
|
+
merged_children = _normalize_typed_inline_spans([*previous_link.content, *current.content])
|
|
446
|
+
normalized[-1] = previous_link.model_copy(update={"content": merged_children})
|
|
447
|
+
continue
|
|
448
|
+
normalized.append(current)
|
|
449
|
+
return normalized
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def parse_inline_span(value: Any) -> InlineSpan:
|
|
453
|
+
"""把字典或现有模型严格解析为一个公开行内 Span。"""
|
|
454
|
+
return INLINE_SPAN_ADAPTER.validate_python(value)
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def parse_inline_spans(value: Any) -> list[InlineSpan]:
|
|
458
|
+
"""严格解析并规范化完整行内 Span 列表。"""
|
|
459
|
+
return _normalize_typed_inline_spans(INLINE_SPAN_LIST_ADAPTER.validate_python(value))
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
class BlockBase(_StrictMiddleModel):
|
|
463
|
+
"""所有公开 Middle JSON block 的最小公共字段。"""
|
|
464
|
+
|
|
465
|
+
type: BlockTypes
|
|
466
|
+
index: int | None = Field(default=None, ge=0)
|
|
467
|
+
bbox: BBox | None = None
|
|
468
|
+
|
|
469
|
+
@field_validator("bbox", mode="before")
|
|
470
|
+
@classmethod
|
|
471
|
+
def _validate_bbox(cls, value: Any) -> BBox | None:
|
|
472
|
+
"""接受 JSON 数组形式的 bbox,并严格校验归一化坐标。"""
|
|
473
|
+
if value is None:
|
|
474
|
+
return None
|
|
475
|
+
if not isinstance(value, (list, tuple)) or len(value) != 4:
|
|
476
|
+
raise ValueError("bbox must contain exactly four numbers")
|
|
477
|
+
if any(isinstance(item, bool) or not isinstance(item, (int, float)) for item in value):
|
|
478
|
+
raise ValueError("bbox values must be numbers")
|
|
479
|
+
bbox = tuple(float(item) for item in value)
|
|
480
|
+
if not all(math.isfinite(item) and 0.0 <= item <= 1.0 for item in bbox):
|
|
481
|
+
raise ValueError("bbox values must be finite normalized coordinates")
|
|
482
|
+
if bbox[2] <= bbox[0] or bbox[3] <= bbox[1]:
|
|
483
|
+
raise ValueError("bbox must satisfy x1 > x0 and y1 > y0")
|
|
484
|
+
return bbox # type: ignore[return-value]
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
class StringContentBlock(BlockBase):
|
|
488
|
+
"""所有字符串内容 block 的共享结构。"""
|
|
489
|
+
|
|
490
|
+
content: str
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
class InlineContentBlock(BlockBase):
|
|
494
|
+
"""所有结构化行内内容 block 的共享结构。"""
|
|
495
|
+
|
|
496
|
+
content: list[InlineSpan]
|
|
497
|
+
|
|
498
|
+
@field_validator("content")
|
|
499
|
+
@classmethod
|
|
500
|
+
def _normalize_content(cls, value: list[InlineSpan]) -> list[InlineSpan]:
|
|
501
|
+
"""在严格对象边界合并相邻同语义 Span。"""
|
|
502
|
+
return _normalize_typed_inline_spans(value)
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
class ContinuableTextBlockBase(InlineContentBlock):
|
|
506
|
+
"""正文与参考文献共享的跨块续接结构。"""
|
|
507
|
+
|
|
508
|
+
continues_prev: bool | None = None
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
class TextBlock(ContinuableTextBlockBase):
|
|
512
|
+
type: Literal[BlockType.TEXT] # type: ignore[reportIncompatibleVariableOverride]
|
|
513
|
+
anchor: str | None = None
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
class RefTextBlock(ContinuableTextBlockBase):
|
|
517
|
+
type: Literal[BlockType.REF_TEXT] # type: ignore[reportIncompatibleVariableOverride]
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
class TitleBlockBase(InlineContentBlock):
|
|
521
|
+
"""文档标题与段落标题的全局层级公共结构。"""
|
|
522
|
+
|
|
523
|
+
anchor: str | None = None
|
|
524
|
+
level: int
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
class DocTitleBlock(TitleBlockBase):
|
|
528
|
+
type: Literal[BlockType.DOC_TITLE] # type: ignore[reportIncompatibleVariableOverride]
|
|
529
|
+
level: int = Field(ge=1, le=1)
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
class ParagraphTitleBlock(TitleBlockBase):
|
|
533
|
+
type: Literal[BlockType.PARAGRAPH_TITLE] # type: ignore[reportIncompatibleVariableOverride]
|
|
534
|
+
level: int = Field(ge=2, le=6)
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
class PageAuxTextBlock(InlineContentBlock):
|
|
538
|
+
"""页眉、页脚、页码和边栏的共享文本结构。"""
|
|
539
|
+
|
|
540
|
+
type: Literal[ # type: ignore[reportIncompatibleVariableOverride]
|
|
541
|
+
BlockType.HEADER,
|
|
542
|
+
BlockType.FOOTER,
|
|
543
|
+
BlockType.PAGE_NUMBER,
|
|
544
|
+
BlockType.ASIDE_TEXT,
|
|
545
|
+
]
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
class PageFootnoteBlock(InlineContentBlock):
|
|
549
|
+
"""保存需要参与默认输出并可被文档内链接引用的页面脚注。"""
|
|
550
|
+
|
|
551
|
+
type: Literal[BlockType.PAGE_FOOTNOTE] # type: ignore[reportIncompatibleVariableOverride]
|
|
552
|
+
anchor: str | None = None
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
class ImagePayloadBlock(BlockBase):
|
|
556
|
+
"""统一携带 sidecar、data URI 或远程 URL 的图片 block 基类。"""
|
|
557
|
+
|
|
558
|
+
image_base64: str | None = Field(default=None, repr=False)
|
|
559
|
+
image_path: str | None = None
|
|
560
|
+
image_url: str | None = None
|
|
561
|
+
|
|
562
|
+
@field_validator("image_path")
|
|
563
|
+
@classmethod
|
|
564
|
+
def _validate_image_path(cls, value: str | None) -> str | None:
|
|
565
|
+
"""校验已记录的图片路径只能是安全的 POSIX 相对路径。"""
|
|
566
|
+
if value is None:
|
|
567
|
+
return None
|
|
568
|
+
from .foundation.image_payload import validate_image_sidecar_path
|
|
569
|
+
|
|
570
|
+
return validate_image_sidecar_path(value)
|
|
571
|
+
|
|
572
|
+
@field_validator("image_url")
|
|
573
|
+
@classmethod
|
|
574
|
+
def _validate_image_url(cls, value: str | None) -> str | None:
|
|
575
|
+
"""校验远程图片 URL,禁止活动协议、相对地址与内嵌凭据。"""
|
|
576
|
+
if value is None:
|
|
577
|
+
return None
|
|
578
|
+
from .foundation.image_payload import validate_remote_image_url
|
|
579
|
+
|
|
580
|
+
return validate_remote_image_url(value)
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
class ImagePayloadContentBlock(ImagePayloadBlock):
|
|
584
|
+
"""统一携带字符串内容和图片载荷的 block 结构。"""
|
|
585
|
+
|
|
586
|
+
content: str
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
class EquationBlock(ImagePayloadContentBlock):
|
|
590
|
+
type: Literal[BlockType.EQUATION] # type: ignore[reportIncompatibleVariableOverride]
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
class ImageBodyBlock(ImagePayloadContentBlock):
|
|
594
|
+
type: Literal[BlockType.IMAGE_BODY] # type: ignore[reportIncompatibleVariableOverride]
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
class TableBodyBlock(ImagePayloadContentBlock):
|
|
598
|
+
type: Literal[BlockType.TABLE_BODY] # type: ignore[reportIncompatibleVariableOverride]
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
class ChartBodyBlock(ImagePayloadContentBlock):
|
|
602
|
+
type: Literal[BlockType.CHART_BODY] # type: ignore[reportIncompatibleVariableOverride]
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
class CodeBodyBlock(StringContentBlock):
|
|
606
|
+
type: Literal[BlockType.CODE_BODY] # type: ignore[reportIncompatibleVariableOverride]
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
class AlgorithmBodyBlock(InlineContentBlock):
|
|
610
|
+
"""保存预格式算法文字与行内公式 Span。"""
|
|
611
|
+
|
|
612
|
+
type: Literal[BlockType.ALGORITHM_BODY] # type: ignore[reportIncompatibleVariableOverride]
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
class ImageAnnotationBlock(InlineContentBlock):
|
|
616
|
+
"""图片标题与图片脚注的共享结构。"""
|
|
617
|
+
|
|
618
|
+
type: Literal[BlockType.IMAGE_CAPTION, BlockType.IMAGE_FOOTNOTE] # type: ignore[reportIncompatibleVariableOverride]
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
class TableAnnotationBlock(InlineContentBlock):
|
|
622
|
+
"""表格标题与表格脚注的共享结构。"""
|
|
623
|
+
|
|
624
|
+
type: Literal[BlockType.TABLE_CAPTION, BlockType.TABLE_FOOTNOTE] # type: ignore[reportIncompatibleVariableOverride]
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
class ChartAnnotationBlock(InlineContentBlock):
|
|
628
|
+
"""图表标题与图表脚注的共享结构。"""
|
|
629
|
+
|
|
630
|
+
type: Literal[BlockType.CHART_CAPTION, BlockType.CHART_FOOTNOTE] # type: ignore[reportIncompatibleVariableOverride]
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
class CodeAnnotationBlock(InlineContentBlock):
|
|
634
|
+
"""代码标题与代码脚注的共享结构。"""
|
|
635
|
+
|
|
636
|
+
type: Literal[BlockType.CODE_CAPTION, BlockType.CODE_FOOTNOTE] # type: ignore[reportIncompatibleVariableOverride]
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
ListChildBlock: TypeAlias = Annotated[
|
|
640
|
+
Union[TextBlock, RefTextBlock, "ListBlock"],
|
|
641
|
+
Field(discriminator="type"),
|
|
642
|
+
]
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
class ListBlock(BlockBase):
|
|
646
|
+
type: Literal[BlockType.LIST] # type: ignore[reportIncompatibleVariableOverride]
|
|
647
|
+
content: list[ListChildBlock]
|
|
648
|
+
sub_type: Literal[BlockType.TEXT, BlockType.REF_TEXT] | None = None
|
|
649
|
+
continues_prev: bool | None = None
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
IndexChildBlock: TypeAlias = Annotated[
|
|
653
|
+
Union[TextBlock, DocTitleBlock, ParagraphTitleBlock, "IndexBlock"],
|
|
654
|
+
Field(discriminator="type"),
|
|
655
|
+
]
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
class IndexBlock(BlockBase):
|
|
659
|
+
type: Literal[BlockType.INDEX] # type: ignore[reportIncompatibleVariableOverride]
|
|
660
|
+
content: list[IndexChildBlock]
|
|
661
|
+
|
|
662
|
+
|
|
663
|
+
class _VisualBlockBase(BlockBase):
|
|
664
|
+
"""视觉父块的共享结构约束。"""
|
|
665
|
+
|
|
666
|
+
_body_types: ClassVar[tuple[str, ...]]
|
|
667
|
+
|
|
668
|
+
@model_validator(mode="after")
|
|
669
|
+
def _validate_visual_children(self) -> _VisualBlockBase:
|
|
670
|
+
"""校验视觉父块只有一个 body,且父子定位字段保持一致。"""
|
|
671
|
+
children = getattr(self, "content", [])
|
|
672
|
+
bodies = [child for child in children if child.type in self._body_types]
|
|
673
|
+
if len(bodies) != 1:
|
|
674
|
+
expected = "/".join(str(item) for item in self._body_types)
|
|
675
|
+
raise ValueError(f"{self.type} must contain exactly one {expected}")
|
|
676
|
+
body = bodies[0]
|
|
677
|
+
if self.index is not None and body.index != self.index:
|
|
678
|
+
raise ValueError(f"{self.type} body index must equal parent index")
|
|
679
|
+
if self.bbox is not None and body.bbox is not None and body.bbox != self.bbox:
|
|
680
|
+
raise ValueError(f"{self.type} body bbox must equal parent bbox")
|
|
681
|
+
return self
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
ImageChildBlock: TypeAlias = Annotated[
|
|
685
|
+
Union[ImageBodyBlock, ImageAnnotationBlock],
|
|
686
|
+
Field(discriminator="type"),
|
|
687
|
+
]
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
class ImageBlock(_VisualBlockBase):
|
|
691
|
+
type: Literal[BlockType.IMAGE] # type: ignore[reportIncompatibleVariableOverride]
|
|
692
|
+
content: list[ImageChildBlock]
|
|
693
|
+
sub_type: str | None = None
|
|
694
|
+
_body_types: ClassVar[tuple[str, ...]] = (BlockType.IMAGE_BODY,)
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
TableChildBlock: TypeAlias = Annotated[
|
|
698
|
+
Union[TableBodyBlock, TableAnnotationBlock],
|
|
699
|
+
Field(discriminator="type"),
|
|
700
|
+
]
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
class TableBlock(_VisualBlockBase):
|
|
704
|
+
type: Literal[BlockType.TABLE] # type: ignore[reportIncompatibleVariableOverride]
|
|
705
|
+
content: list[TableChildBlock]
|
|
706
|
+
continues_prev: bool | None = None
|
|
707
|
+
cell_merge: list[Literal[0, 1]] | None = None
|
|
708
|
+
_body_types: ClassVar[tuple[str, ...]] = (BlockType.TABLE_BODY,)
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
ChartChildBlock: TypeAlias = Annotated[
|
|
712
|
+
Union[ChartBodyBlock, ChartAnnotationBlock],
|
|
713
|
+
Field(discriminator="type"),
|
|
714
|
+
]
|
|
715
|
+
|
|
716
|
+
|
|
717
|
+
class ChartBlock(_VisualBlockBase):
|
|
718
|
+
type: Literal[BlockType.CHART] # type: ignore[reportIncompatibleVariableOverride]
|
|
719
|
+
content: list[ChartChildBlock]
|
|
720
|
+
sub_type: str | None = None
|
|
721
|
+
_body_types: ClassVar[tuple[str, ...]] = (BlockType.CHART_BODY,)
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
CodeChildBlock: TypeAlias = Annotated[
|
|
725
|
+
Union[CodeBodyBlock, AlgorithmBodyBlock, CodeAnnotationBlock],
|
|
726
|
+
Field(discriminator="type"),
|
|
727
|
+
]
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
class CodeBlock(_VisualBlockBase):
|
|
731
|
+
type: Literal[BlockType.CODE] # type: ignore[reportIncompatibleVariableOverride]
|
|
732
|
+
content: list[CodeChildBlock]
|
|
733
|
+
sub_type: Literal[BlockType.CODE, RAW_ALGORITHM]
|
|
734
|
+
guess_lang: str | None = None
|
|
735
|
+
_body_types: ClassVar[tuple[str, ...]] = (BlockType.CODE_BODY, BlockType.ALGORITHM_BODY)
|
|
736
|
+
|
|
737
|
+
@model_validator(mode="after")
|
|
738
|
+
def _validate_language(self) -> CodeBlock:
|
|
739
|
+
"""代码块要求语言,算法块则禁止携带代码语言猜测结果。"""
|
|
740
|
+
body = next(child for child in self.content if child.type in self._body_types)
|
|
741
|
+
if self.sub_type == BlockType.CODE:
|
|
742
|
+
if body.type != BlockType.CODE_BODY:
|
|
743
|
+
raise ValueError("code block must contain code_body")
|
|
744
|
+
if not isinstance(self.guess_lang, str) or not self.guess_lang.strip():
|
|
745
|
+
raise ValueError("code block must contain a non-empty guess_lang")
|
|
746
|
+
else:
|
|
747
|
+
if body.type != BlockType.ALGORITHM_BODY:
|
|
748
|
+
raise ValueError("algorithm block must contain algorithm_body")
|
|
749
|
+
if self.guess_lang is not None:
|
|
750
|
+
raise ValueError("algorithm block must not contain guess_lang")
|
|
751
|
+
return self
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
ListBlock.model_rebuild()
|
|
755
|
+
IndexBlock.model_rebuild()
|
|
756
|
+
|
|
757
|
+
|
|
758
|
+
PageBlock: TypeAlias = Annotated[
|
|
759
|
+
Union[
|
|
760
|
+
TextBlock,
|
|
761
|
+
RefTextBlock,
|
|
762
|
+
DocTitleBlock,
|
|
763
|
+
ParagraphTitleBlock,
|
|
764
|
+
PageAuxTextBlock,
|
|
765
|
+
PageFootnoteBlock,
|
|
766
|
+
EquationBlock,
|
|
767
|
+
ListBlock,
|
|
768
|
+
IndexBlock,
|
|
769
|
+
ImageBlock,
|
|
770
|
+
TableBlock,
|
|
771
|
+
ChartBlock,
|
|
772
|
+
CodeBlock,
|
|
773
|
+
],
|
|
774
|
+
Field(discriminator="type"),
|
|
775
|
+
]
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
Block: TypeAlias = Annotated[
|
|
779
|
+
Union[
|
|
780
|
+
TextBlock,
|
|
781
|
+
RefTextBlock,
|
|
782
|
+
DocTitleBlock,
|
|
783
|
+
ParagraphTitleBlock,
|
|
784
|
+
PageAuxTextBlock,
|
|
785
|
+
PageFootnoteBlock,
|
|
786
|
+
EquationBlock,
|
|
787
|
+
ListBlock,
|
|
788
|
+
IndexBlock,
|
|
789
|
+
ImageBodyBlock,
|
|
790
|
+
ImageAnnotationBlock,
|
|
791
|
+
ImageBlock,
|
|
792
|
+
TableBodyBlock,
|
|
793
|
+
TableAnnotationBlock,
|
|
794
|
+
TableBlock,
|
|
795
|
+
ChartBodyBlock,
|
|
796
|
+
ChartAnnotationBlock,
|
|
797
|
+
ChartBlock,
|
|
798
|
+
CodeBodyBlock,
|
|
799
|
+
AlgorithmBodyBlock,
|
|
800
|
+
CodeAnnotationBlock,
|
|
801
|
+
CodeBlock,
|
|
802
|
+
],
|
|
803
|
+
Field(discriminator="type"),
|
|
804
|
+
]
|
|
805
|
+
|
|
806
|
+
BLOCK_ADAPTER = TypeAdapter(Block)
|
|
807
|
+
|
|
808
|
+
|
|
809
|
+
def parse_block(value: Any) -> Block:
|
|
810
|
+
"""将字典或已有模型严格解析成对应的具体 Block 类型。"""
|
|
811
|
+
return BLOCK_ADAPTER.validate_python(value)
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _iter_child_blocks(block: BlockBase) -> list[BlockBase]:
|
|
815
|
+
"""返回容器 block 的直接子块,叶子 block 返回空列表。"""
|
|
816
|
+
content = getattr(block, "content", None)
|
|
817
|
+
if not isinstance(content, list):
|
|
818
|
+
return []
|
|
819
|
+
return [child for child in content if isinstance(child, BlockBase)]
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
_RAW_INLINE_CONTENT_TYPES = {
|
|
823
|
+
BlockType.TEXT,
|
|
824
|
+
BlockType.REF_TEXT,
|
|
825
|
+
BlockType.DOC_TITLE,
|
|
826
|
+
BlockType.PARAGRAPH_TITLE,
|
|
827
|
+
BlockType.HEADER,
|
|
828
|
+
BlockType.FOOTER,
|
|
829
|
+
BlockType.PAGE_NUMBER,
|
|
830
|
+
BlockType.ASIDE_TEXT,
|
|
831
|
+
BlockType.PAGE_FOOTNOTE,
|
|
832
|
+
BlockType.IMAGE_CAPTION,
|
|
833
|
+
BlockType.IMAGE_FOOTNOTE,
|
|
834
|
+
BlockType.TABLE_CAPTION,
|
|
835
|
+
BlockType.TABLE_FOOTNOTE,
|
|
836
|
+
BlockType.CHART_CAPTION,
|
|
837
|
+
BlockType.CHART_FOOTNOTE,
|
|
838
|
+
BlockType.CODE_CAPTION,
|
|
839
|
+
BlockType.CODE_FOOTNOTE,
|
|
840
|
+
RAW_ALGORITHM,
|
|
841
|
+
RAW_CAPTION,
|
|
842
|
+
RAW_FOOTNOTE,
|
|
843
|
+
RAW_PHONETIC,
|
|
844
|
+
}
|
|
845
|
+
|
|
846
|
+
|
|
847
|
+
def _looks_like_raw_inline_span_list(content: list[Any]) -> bool:
|
|
848
|
+
"""区分 PDF 扁平 LIST/INDEX 的 Span 载荷与已经成树的文本子块。"""
|
|
849
|
+
if not content:
|
|
850
|
+
return False
|
|
851
|
+
for item in content:
|
|
852
|
+
if not isinstance(item, dict):
|
|
853
|
+
return False
|
|
854
|
+
span_type = item.get("type")
|
|
855
|
+
span_content = item.get("content")
|
|
856
|
+
if span_type in {"text", "equation_inline", "code_inline"} and isinstance(span_content, str):
|
|
857
|
+
continue
|
|
858
|
+
if span_type == "hyperlink" and isinstance(span_content, list) and isinstance(item.get("url"), str):
|
|
859
|
+
continue
|
|
860
|
+
return False
|
|
861
|
+
return True
|
|
862
|
+
|
|
863
|
+
|
|
864
|
+
def _validate_raw_block_inline_content(block: dict[str, Any], *, location: str) -> None:
|
|
865
|
+
"""递归校验 raw block 的自然语言 content 已切换为 Span 列表。"""
|
|
866
|
+
block_type = block.get("type")
|
|
867
|
+
content = block.get("content")
|
|
868
|
+
if block_type in _RAW_INLINE_CONTENT_TYPES:
|
|
869
|
+
if not isinstance(content, list):
|
|
870
|
+
raise ValueError(f"ModelJson inline content must be a span list: {location}, type={block_type}")
|
|
871
|
+
try:
|
|
872
|
+
parse_inline_spans(content)
|
|
873
|
+
except ValueError as exc:
|
|
874
|
+
raise ValueError(f"Invalid ModelJson inline spans: {location}, type={block_type}: {exc}") from exc
|
|
875
|
+
return
|
|
876
|
+
if block_type not in {BlockType.LIST, BlockType.INDEX} or not isinstance(content, list):
|
|
877
|
+
return
|
|
878
|
+
if _looks_like_raw_inline_span_list(content):
|
|
879
|
+
try:
|
|
880
|
+
parse_inline_spans(content)
|
|
881
|
+
except ValueError as exc:
|
|
882
|
+
raise ValueError(f"Invalid ModelJson inline spans: {location}, type={block_type}: {exc}") from exc
|
|
883
|
+
return
|
|
884
|
+
for child_index, child in enumerate(content):
|
|
885
|
+
if isinstance(child, dict):
|
|
886
|
+
_validate_raw_block_inline_content(child, location=f"{location}.content[{child_index}]")
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
class Producer(_StrictMiddleModel):
|
|
890
|
+
"""记录语言无关的文档生产者,避免绑定宿主产品元数据。"""
|
|
891
|
+
|
|
892
|
+
name: str = Field(min_length=1)
|
|
893
|
+
version: str = Field(min_length=1)
|
|
894
|
+
|
|
895
|
+
|
|
896
|
+
class DocumentMetadata(_StrictMiddleModel):
|
|
897
|
+
"""承载文档格式和真实生产者,读取时不补造来源信息。"""
|
|
898
|
+
|
|
899
|
+
file_suffix: FileSuffix
|
|
900
|
+
producer: Producer
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
def _require_document_wire_identity(schema: dict[str, Any]) -> None:
|
|
904
|
+
"""JSON 文档必须显式携带协议身份;构造器的常量默认值仅便利 Python 调用。"""
|
|
905
|
+
identity = "schema" if "schema" in schema["properties"] else "schema_id"
|
|
906
|
+
schema["required"] = list(dict.fromkeys([identity, "schema_version", *schema.get("required", [])]))
|
|
907
|
+
|
|
908
|
+
|
|
909
|
+
_DocumentT = TypeVar("_DocumentT", bound="DocumentModel")
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
class DocumentModel(_StrictMiddleModel):
|
|
913
|
+
"""公共文档封装,只持有生产者及可序列化扩展信息。"""
|
|
914
|
+
|
|
915
|
+
model_config = ConfigDict(serialize_by_alias=True, json_schema_extra=_require_document_wire_identity)
|
|
916
|
+
|
|
917
|
+
metadata: DocumentMetadata
|
|
918
|
+
extensions: dict[str, JsonValue] = Field(default_factory=dict)
|
|
919
|
+
schema_version: Literal["2.0"] = "2.0"
|
|
920
|
+
schema_id: str = Field(alias="schema")
|
|
921
|
+
|
|
922
|
+
def to_dict(
|
|
923
|
+
self,
|
|
924
|
+
*,
|
|
925
|
+
skip_defaults: bool = True,
|
|
926
|
+
exclude_none: bool = False,
|
|
927
|
+
exclude_block_fields: set[str] | None = None,
|
|
928
|
+
) -> dict[str, Any]:
|
|
929
|
+
"""只对页面树省略块字段,保护具有同名键的来源和应用扩展。"""
|
|
930
|
+
payload = super().to_dict(skip_defaults=skip_defaults, exclude_none=exclude_none)
|
|
931
|
+
if exclude_block_fields and "pages" in payload:
|
|
932
|
+
payload["pages"] = _remove_block_fields(payload["pages"], exclude_block_fields)
|
|
933
|
+
return payload
|
|
934
|
+
|
|
935
|
+
@model_serializer(mode="wrap")
|
|
936
|
+
def _serialize_document(self, handler: SerializerFunctionWrapHandler, info: SerializationInfo):
|
|
937
|
+
"""保留已声明的协议默认字段,同时尊重调用方显式的字段筛选。"""
|
|
938
|
+
# 不声明通用 dict 返回类型,避免 Pydantic 将序列化 Schema 降为任意对象。
|
|
939
|
+
payload = handler(self)
|
|
940
|
+
use_alias = info.by_alias is not False
|
|
941
|
+
for field_name in ("schema_id", "schema_version", "extensions"):
|
|
942
|
+
if info.exclude and field_name in info.exclude:
|
|
943
|
+
continue
|
|
944
|
+
if info.include is not None and field_name not in info.include:
|
|
945
|
+
continue
|
|
946
|
+
key = "schema" if field_name == "schema_id" and use_alias else field_name
|
|
947
|
+
if key not in payload:
|
|
948
|
+
payload[key] = deepcopy(getattr(self, field_name))
|
|
949
|
+
return payload
|
|
950
|
+
|
|
951
|
+
@classmethod
|
|
952
|
+
def from_dict(cls: type[_DocumentT], value: dict[str, Any]) -> _DocumentT:
|
|
953
|
+
"""联合校验协议身份及版本后读取文档,不猜测或迁移历史格式。"""
|
|
954
|
+
expected_schema = cls.model_fields["schema_id"].default
|
|
955
|
+
expected_version = cls.model_fields["schema_version"].default
|
|
956
|
+
if (
|
|
957
|
+
not isinstance(value, dict)
|
|
958
|
+
or value.get("schema") != expected_schema
|
|
959
|
+
or value.get("schema_version") != expected_version
|
|
960
|
+
):
|
|
961
|
+
raise ValueError(f"Expected {expected_schema} schema version {expected_version}; reparse the source document")
|
|
962
|
+
return cls.model_validate(value)
|
|
963
|
+
|
|
964
|
+
@classmethod
|
|
965
|
+
def from_json(cls: type[_DocumentT], value: str | bytes) -> _DocumentT:
|
|
966
|
+
"""从 JSON 文本恢复共享文档,并复用唯一的协议校验入口。"""
|
|
967
|
+
return cls.from_dict(json.loads(value))
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
class ModelJson(DocumentModel):
|
|
971
|
+
"""Analyze 返回的完整严格 Model JSON 对象。"""
|
|
972
|
+
|
|
973
|
+
schema_id: Literal["docvortex.model"] = Field(default="docvortex.model", alias="schema")
|
|
974
|
+
pages: list[list[dict[str, Any]]]
|
|
975
|
+
page_index_map: list[int]
|
|
976
|
+
|
|
977
|
+
@model_validator(mode="after")
|
|
978
|
+
def _validate_page_index_map(self) -> ModelJson:
|
|
979
|
+
"""校验显式抽页映射及每个 raw 文本块的 Span 契约。"""
|
|
980
|
+
if self.page_index_map:
|
|
981
|
+
if len(self.page_index_map) != len(self.pages):
|
|
982
|
+
raise ValueError(f"page_index_map length mismatch: pages={len(self.pages)}, mapping={len(self.page_index_map)}")
|
|
983
|
+
if any(page_idx < 0 for page_idx in self.page_index_map):
|
|
984
|
+
raise ValueError("page_index_map values must be non-negative integers")
|
|
985
|
+
if len(self.page_index_map) != len(set(self.page_index_map)):
|
|
986
|
+
raise ValueError("page_index_map values must be unique")
|
|
987
|
+
if any(current <= previous for previous, current in zip(self.page_index_map, self.page_index_map[1:])):
|
|
988
|
+
raise ValueError("page_index_map values must preserve strictly increasing order")
|
|
989
|
+
for page_index, page in enumerate(self.pages):
|
|
990
|
+
for block_index, block in enumerate(page):
|
|
991
|
+
if isinstance(block, dict):
|
|
992
|
+
_validate_raw_block_inline_content(block, location=f"pages[{page_index}][{block_index}]")
|
|
993
|
+
return self
|
|
994
|
+
|
|
995
|
+
@property
|
|
996
|
+
def is_full_document(self) -> bool:
|
|
997
|
+
"""返回当前 Model JSON 是否表示整本文档解析。"""
|
|
998
|
+
return not self.page_index_map
|
|
999
|
+
|
|
1000
|
+
@property
|
|
1001
|
+
def resolved_page_indices(self) -> list[int]:
|
|
1002
|
+
"""返回显式抽页映射或整本文档的顺序页号副本。"""
|
|
1003
|
+
if self.is_full_document:
|
|
1004
|
+
return list(range(len(self.pages)))
|
|
1005
|
+
return list(self.page_index_map)
|
|
1006
|
+
|
|
1007
|
+
|
|
1008
|
+
class PageInfo(_StrictMiddleModel):
|
|
1009
|
+
"""一页的严格 Middle JSON 内容。"""
|
|
1010
|
+
|
|
1011
|
+
page_idx: int = Field(ge=0)
|
|
1012
|
+
blocks: list[PageBlock] = Field(default_factory=list)
|
|
1013
|
+
|
|
1014
|
+
@model_validator(mode="after")
|
|
1015
|
+
def _validate_page_tree(self) -> PageInfo:
|
|
1016
|
+
"""校验顶层 index 顺序,并禁止嵌套块携带跨块延续标记。"""
|
|
1017
|
+
indices: list[int] = []
|
|
1018
|
+
for block in self.blocks:
|
|
1019
|
+
if block.index is None:
|
|
1020
|
+
raise ValueError("top-level block index is required")
|
|
1021
|
+
indices.append(block.index)
|
|
1022
|
+
if len(indices) != len(set(indices)):
|
|
1023
|
+
raise ValueError("top-level block indices must be unique")
|
|
1024
|
+
if any(current <= previous for previous, current in zip(indices, indices[1:])):
|
|
1025
|
+
raise ValueError("top-level block indices must be strictly increasing")
|
|
1026
|
+
|
|
1027
|
+
pending = [child for block in self.blocks for child in _iter_child_blocks(block)]
|
|
1028
|
+
while pending:
|
|
1029
|
+
child = pending.pop()
|
|
1030
|
+
if "continues_prev" in child.model_fields_set:
|
|
1031
|
+
raise ValueError("nested blocks must not contain continues_prev")
|
|
1032
|
+
pending.extend(_iter_child_blocks(child))
|
|
1033
|
+
return self
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
class MiddleJson(DocumentModel):
|
|
1037
|
+
"""Analyze 返回的完整严格 Middle JSON 对象。"""
|
|
1038
|
+
|
|
1039
|
+
schema_id: Literal["docvortex.middle"] = Field(default="docvortex.middle", alias="schema")
|
|
1040
|
+
pages: list[PageInfo]
|
|
1041
|
+
is_full_document: bool
|
|
1042
|
+
|
|
1043
|
+
@model_validator(mode="after")
|
|
1044
|
+
def _validate_document(self) -> MiddleJson:
|
|
1045
|
+
"""校验页号唯一有序,并要求固定版式文档顶层 block 均具有 bbox。"""
|
|
1046
|
+
page_indices = [page.page_idx for page in self.pages]
|
|
1047
|
+
if len(page_indices) != len(set(page_indices)):
|
|
1048
|
+
raise ValueError("page_idx values must be unique")
|
|
1049
|
+
if any(current <= previous for previous, current in zip(page_indices, page_indices[1:])):
|
|
1050
|
+
raise ValueError("page_idx values must be strictly increasing")
|
|
1051
|
+
if self.metadata.file_suffix in {"pdf", "ofd"}:
|
|
1052
|
+
for page in self.pages:
|
|
1053
|
+
for block in page.blocks:
|
|
1054
|
+
if block.bbox is None:
|
|
1055
|
+
raise ValueError(
|
|
1056
|
+
f"Fixed-layout top-level block requires bbox: "
|
|
1057
|
+
f"file_suffix={self.metadata.file_suffix}, page_idx={page.page_idx}, index={block.index}"
|
|
1058
|
+
)
|
|
1059
|
+
return self
|
|
1060
|
+
|
|
1061
|
+
|
|
1062
|
+
__all__ = [
|
|
1063
|
+
"DocumentMetadata",
|
|
1064
|
+
"RawBlockType",
|
|
1065
|
+
"RAW_ALGORITHM",
|
|
1066
|
+
"RAW_CAPTION",
|
|
1067
|
+
"RAW_FOOTNOTE",
|
|
1068
|
+
"RAW_FORMULA_NUMBER",
|
|
1069
|
+
"RAW_PHONETIC",
|
|
1070
|
+
"RAW_ONLY_BLOCK_TYPES",
|
|
1071
|
+
"FileSuffix",
|
|
1072
|
+
"FILE_SUFFIXES",
|
|
1073
|
+
"BlockType",
|
|
1074
|
+
"BlockTypes",
|
|
1075
|
+
"PageBlockTypes",
|
|
1076
|
+
"BLOCK_TYPES",
|
|
1077
|
+
"PAGE_BLOCK_TYPES",
|
|
1078
|
+
"PAGE_AUXILIARY_BLOCK_TYPES",
|
|
1079
|
+
"MERGE_TRANSPARENT_BLOCK_TYPES",
|
|
1080
|
+
"VISUAL_RELATION_IGNORED_TYPES",
|
|
1081
|
+
"VISUAL_MAIN_TYPES",
|
|
1082
|
+
"VISUAL_TYPE_MAPPING",
|
|
1083
|
+
"BBox",
|
|
1084
|
+
"IntBBox",
|
|
1085
|
+
"InlineStyle",
|
|
1086
|
+
"INLINE_STYLE_ORDER",
|
|
1087
|
+
"TextSpan",
|
|
1088
|
+
"EquationInlineSpan",
|
|
1089
|
+
"CodeInlineSpan",
|
|
1090
|
+
"NonLinkInlineSpan",
|
|
1091
|
+
"HyperlinkSpan",
|
|
1092
|
+
"InlineSpan",
|
|
1093
|
+
"INLINE_SPAN_ADAPTER",
|
|
1094
|
+
"INLINE_SPAN_LIST_ADAPTER",
|
|
1095
|
+
"parse_inline_span",
|
|
1096
|
+
"parse_inline_spans",
|
|
1097
|
+
"BlockBase",
|
|
1098
|
+
"StringContentBlock",
|
|
1099
|
+
"InlineContentBlock",
|
|
1100
|
+
"ContinuableTextBlockBase",
|
|
1101
|
+
"TextBlock",
|
|
1102
|
+
"RefTextBlock",
|
|
1103
|
+
"TitleBlockBase",
|
|
1104
|
+
"DocTitleBlock",
|
|
1105
|
+
"ParagraphTitleBlock",
|
|
1106
|
+
"PageAuxTextBlock",
|
|
1107
|
+
"PageFootnoteBlock",
|
|
1108
|
+
"ImagePayloadBlock",
|
|
1109
|
+
"ImagePayloadContentBlock",
|
|
1110
|
+
"EquationBlock",
|
|
1111
|
+
"ImageBodyBlock",
|
|
1112
|
+
"TableBodyBlock",
|
|
1113
|
+
"ChartBodyBlock",
|
|
1114
|
+
"CodeBodyBlock",
|
|
1115
|
+
"AlgorithmBodyBlock",
|
|
1116
|
+
"ImageAnnotationBlock",
|
|
1117
|
+
"TableAnnotationBlock",
|
|
1118
|
+
"ChartAnnotationBlock",
|
|
1119
|
+
"CodeAnnotationBlock",
|
|
1120
|
+
"ListChildBlock",
|
|
1121
|
+
"ListBlock",
|
|
1122
|
+
"IndexChildBlock",
|
|
1123
|
+
"IndexBlock",
|
|
1124
|
+
"ImageChildBlock",
|
|
1125
|
+
"ImageBlock",
|
|
1126
|
+
"TableChildBlock",
|
|
1127
|
+
"TableBlock",
|
|
1128
|
+
"ChartChildBlock",
|
|
1129
|
+
"ChartBlock",
|
|
1130
|
+
"CodeChildBlock",
|
|
1131
|
+
"CodeBlock",
|
|
1132
|
+
"PageBlock",
|
|
1133
|
+
"Block",
|
|
1134
|
+
"BLOCK_ADAPTER",
|
|
1135
|
+
"parse_block",
|
|
1136
|
+
"Producer",
|
|
1137
|
+
"DocumentModel",
|
|
1138
|
+
"ModelJson",
|
|
1139
|
+
"PageInfo",
|
|
1140
|
+
"MiddleJson",
|
|
1141
|
+
]
|