docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""raw block 的文本、代码和公式内容清理规则。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from ..content.inline import map_text_span_content, normalize_inline_spans
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def code_content_clean(content: str | None) -> str:
|
|
12
|
+
"""去除代码块外层 Markdown 围栏并保留代码正文。"""
|
|
13
|
+
if not content:
|
|
14
|
+
return ""
|
|
15
|
+
lines = content.splitlines()
|
|
16
|
+
start_idx = 1 if lines and lines[0].startswith("```") else 0
|
|
17
|
+
end_idx = len(lines)
|
|
18
|
+
if lines and end_idx > start_idx and lines[end_idx - 1].strip() == "```":
|
|
19
|
+
end_idx -= 1
|
|
20
|
+
if start_idx < end_idx:
|
|
21
|
+
return "\n".join(lines[start_idx:end_idx]).strip()
|
|
22
|
+
return ""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def clean_content(content: str | None) -> str | None:
|
|
26
|
+
"""将成对的行间公式分隔符改为兼容文本清理的方括号。"""
|
|
27
|
+
if content and content.count("\\[") == content.count("\\]") and content.count("\\[") > 0:
|
|
28
|
+
|
|
29
|
+
def replace_pattern(match: re.Match[str]) -> str:
|
|
30
|
+
"""替换单个成对公式片段。"""
|
|
31
|
+
return f"[{match.group(1)}]"
|
|
32
|
+
|
|
33
|
+
content = re.sub(r"\\\[(.*?)\\\]", replace_pattern, content)
|
|
34
|
+
return content
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def clean_inline_content(content: Any) -> list[dict[str, Any]]:
|
|
38
|
+
"""严格规范化 raw Span 列表,并返回可继续后处理的 JSON 字典。"""
|
|
39
|
+
if content is None:
|
|
40
|
+
return []
|
|
41
|
+
if not isinstance(content, list):
|
|
42
|
+
raise TypeError("inline content must be a list of spans")
|
|
43
|
+
spans = normalize_inline_spans(content)
|
|
44
|
+
return [span.model_dump(mode="json") for span in spans]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def collapse_inline_newlines(content: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
48
|
+
"""把标题 Span 中的换行及后续空白收敛为单个空格。"""
|
|
49
|
+
spans = map_text_span_content(normalize_inline_spans(content), lambda value: re.sub(r"\n\s*", " ", value))
|
|
50
|
+
return [span.model_dump(mode="json") for span in spans]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
__all__ = ["clean_content", "clean_inline_content", "code_content_clean", "collapse_inline_newlines"]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""将规范化分析结果转换为独立的语义文档。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
from copy import deepcopy
|
|
5
|
+
from ..schema import MiddleJson, ModelJson
|
|
6
|
+
from .pages import model_json_to_pages
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def model_json_to_middle_json(model_json: ModelJson) -> MiddleJson:
|
|
10
|
+
"""只执行确定性后处理,智能增强由调用方另行执行。"""
|
|
11
|
+
return MiddleJson(
|
|
12
|
+
pages=model_json_to_pages(model_json),
|
|
13
|
+
is_full_document=model_json.is_full_document,
|
|
14
|
+
metadata=model_json.metadata.model_copy(deep=True),
|
|
15
|
+
extensions=deepcopy(model_json.extensions),
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
__all__ = ["model_json_to_middle_json"]
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""PDF 与 Office 的列表、目录和标题编号后处理。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from ..content.spans import inline_span_plain_text, text_spans
|
|
9
|
+
from ..schema import BlockType, parse_inline_spans
|
|
10
|
+
from ..foundation.geometry import calculate_overlap_area_in_bbox1_area_ratio
|
|
11
|
+
|
|
12
|
+
from ..content.inline import inline_plain_text, slice_inline_spans
|
|
13
|
+
from .visual import _bbox_for_calculation
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def fix_office_paragraph_titles(model_list: list[list[dict[str, Any]]]) -> None:
|
|
17
|
+
"""按文档级标题序列内化 Office 自动编号,并清除私有编号元数据。"""
|
|
18
|
+
counters: dict[int, int] = {}
|
|
19
|
+
for page_model_list in model_list:
|
|
20
|
+
for block in page_model_list:
|
|
21
|
+
if block.get("type") != BlockType.PARAGRAPH_TITLE:
|
|
22
|
+
continue
|
|
23
|
+
raw_level = block.get("level")
|
|
24
|
+
normalized_level = raw_level if type(raw_level) is int else 2
|
|
25
|
+
level = min(max(normalized_level, 2), 6)
|
|
26
|
+
block["level"] = level
|
|
27
|
+
numbering_depth = level - 1
|
|
28
|
+
is_numbered_style = block.pop("is_numbered_style", None)
|
|
29
|
+
block.pop("section_number", None)
|
|
30
|
+
content = block.get("content")
|
|
31
|
+
if not isinstance(content, list):
|
|
32
|
+
continue
|
|
33
|
+
if is_numbered_style is True:
|
|
34
|
+
for ancestor_level in range(1, numbering_depth):
|
|
35
|
+
counters.setdefault(ancestor_level, 1)
|
|
36
|
+
counters[numbering_depth] = counters.get(numbering_depth, 0) + 1
|
|
37
|
+
_clear_deeper_title_counters(counters, numbering_depth)
|
|
38
|
+
section_number = ".".join(str(counters[ancestor_level]) for ancestor_level in range(1, numbering_depth + 1))
|
|
39
|
+
block["content"] = [*text_spans(f"{section_number} "), *content]
|
|
40
|
+
continue
|
|
41
|
+
if is_numbered_style is False:
|
|
42
|
+
number_match = re.match(r"^\s*(\d+(?:\.\d+)*)\b", _visible_text(content))
|
|
43
|
+
if number_match is None:
|
|
44
|
+
continue
|
|
45
|
+
number_parts = [int(part) for part in number_match.group(1).split(".")]
|
|
46
|
+
if len(number_parts) != numbering_depth:
|
|
47
|
+
continue
|
|
48
|
+
counters.update((part_level, number) for part_level, number in enumerate(number_parts, start=1))
|
|
49
|
+
_clear_deeper_title_counters(counters, numbering_depth)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def fix_office_index_title_blocks(model_list: list[list[dict[str, Any]]]) -> None:
|
|
53
|
+
"""按正文目标 anchor 将 Office 目录文本叶子转换为对应正文或标题类型。"""
|
|
54
|
+
target_by_anchor: dict[str, tuple[str, int | None]] = {}
|
|
55
|
+
for page_model_list in model_list:
|
|
56
|
+
for block in page_model_list:
|
|
57
|
+
block_type = block.get("type")
|
|
58
|
+
if block_type not in {BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE}:
|
|
59
|
+
continue
|
|
60
|
+
anchor = block.get("anchor")
|
|
61
|
+
level = block.get("level")
|
|
62
|
+
if not isinstance(anchor, str) or not anchor.strip():
|
|
63
|
+
continue
|
|
64
|
+
if block_type in {BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE} and type(level) is not int:
|
|
65
|
+
continue
|
|
66
|
+
target_by_anchor.setdefault(anchor.strip(), (block_type, level if type(level) is int else None))
|
|
67
|
+
|
|
68
|
+
for page_model_list in model_list:
|
|
69
|
+
for block in page_model_list:
|
|
70
|
+
if block.get("type") == BlockType.INDEX:
|
|
71
|
+
_rewrite_office_index_title_leaves(block, target_by_anchor)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _rewrite_office_index_title_leaves(
|
|
75
|
+
index_block: dict[str, Any],
|
|
76
|
+
target_by_anchor: dict[str, tuple[str, int | None]],
|
|
77
|
+
) -> None:
|
|
78
|
+
"""递归改写目录叶子;未匹配 anchor 时降级为不带 anchor 的普通文本。"""
|
|
79
|
+
content = index_block.get("content")
|
|
80
|
+
if not isinstance(content, list):
|
|
81
|
+
return
|
|
82
|
+
for child in content:
|
|
83
|
+
if not isinstance(child, dict):
|
|
84
|
+
continue
|
|
85
|
+
if child.get("type") == BlockType.INDEX:
|
|
86
|
+
_rewrite_office_index_title_leaves(child, target_by_anchor)
|
|
87
|
+
continue
|
|
88
|
+
if child.get("type") != BlockType.TEXT:
|
|
89
|
+
continue
|
|
90
|
+
anchor = child.get("anchor")
|
|
91
|
+
normalized_anchor = anchor.strip() if isinstance(anchor, str) else ""
|
|
92
|
+
target = target_by_anchor.get(normalized_anchor)
|
|
93
|
+
if target is None:
|
|
94
|
+
child.pop("anchor", None)
|
|
95
|
+
continue
|
|
96
|
+
target_type, target_level = target
|
|
97
|
+
child["type"] = target_type
|
|
98
|
+
if target_level is None:
|
|
99
|
+
child.pop("level", None)
|
|
100
|
+
else:
|
|
101
|
+
child["level"] = target_level
|
|
102
|
+
child["anchor"] = normalized_anchor
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def fix_pdf_list_blocks(
|
|
106
|
+
list_blocks: list[dict[str, Any]],
|
|
107
|
+
text_blocks: list[dict[str, Any]],
|
|
108
|
+
ref_text_blocks: list[dict[str, Any]],
|
|
109
|
+
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], list[dict[str, Any]]]:
|
|
110
|
+
"""按 bbox 把 PDF text/ref_text 归入 list,并推断列表子类型。"""
|
|
111
|
+
for list_block in list_blocks:
|
|
112
|
+
list_block["content"] = []
|
|
113
|
+
need_remove_blocks = []
|
|
114
|
+
for block in text_blocks + ref_text_blocks:
|
|
115
|
+
for list_block in list_blocks:
|
|
116
|
+
if (
|
|
117
|
+
calculate_overlap_area_in_bbox1_area_ratio(
|
|
118
|
+
_bbox_for_calculation(block["bbox"]),
|
|
119
|
+
_bbox_for_calculation(list_block["bbox"]),
|
|
120
|
+
)
|
|
121
|
+
>= 0.8
|
|
122
|
+
):
|
|
123
|
+
list_block["content"].append(block)
|
|
124
|
+
need_remove_blocks.append(block)
|
|
125
|
+
break
|
|
126
|
+
for block in need_remove_blocks:
|
|
127
|
+
if block in text_blocks:
|
|
128
|
+
text_blocks.remove(block)
|
|
129
|
+
elif block in ref_text_blocks:
|
|
130
|
+
ref_text_blocks.remove(block)
|
|
131
|
+
list_blocks = [block for block in list_blocks if block["content"]]
|
|
132
|
+
for list_block in list_blocks:
|
|
133
|
+
type_count: dict[str, int] = {}
|
|
134
|
+
for sub_block in list_block["content"]:
|
|
135
|
+
sub_block_type = sub_block["type"]
|
|
136
|
+
type_count[sub_block_type] = type_count.get(sub_block_type, 0) + 1
|
|
137
|
+
list_block["sub_type"] = max(type_count, key=type_count.get) if type_count else "text" # type: ignore[arg-type]
|
|
138
|
+
return list_blocks, text_blocks, ref_text_blocks
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def fix_pdf_index_blocks(index_blocks: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
142
|
+
"""将 PDF 目录块的多行内容拆分为多个文本子块。"""
|
|
143
|
+
for index_block in index_blocks:
|
|
144
|
+
raw_content = index_block.get("content")
|
|
145
|
+
if not isinstance(raw_content, list):
|
|
146
|
+
index_block["content"] = []
|
|
147
|
+
continue
|
|
148
|
+
spans = parse_inline_spans(raw_content)
|
|
149
|
+
visible = inline_plain_text(spans)
|
|
150
|
+
children: list[dict[str, Any]] = []
|
|
151
|
+
start = 0
|
|
152
|
+
for match in re.finditer("\n", visible):
|
|
153
|
+
content = slice_inline_spans(spans, start, match.start())
|
|
154
|
+
if content:
|
|
155
|
+
children.append({"type": BlockType.TEXT, "content": [span.model_dump(mode="json") for span in content]})
|
|
156
|
+
start = match.end()
|
|
157
|
+
content = slice_inline_spans(spans, start)
|
|
158
|
+
if content:
|
|
159
|
+
children.append({"type": BlockType.TEXT, "content": [span.model_dump(mode="json") for span in content]})
|
|
160
|
+
index_block["content"] = children
|
|
161
|
+
return index_blocks
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def fix_office_index_blocks(index_blocks: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
165
|
+
"""递归移除 Office 目录层级私有字段,保留已规范化的标题叶子。"""
|
|
166
|
+
pending_blocks = list(index_blocks)
|
|
167
|
+
while pending_blocks:
|
|
168
|
+
block = pending_blocks.pop()
|
|
169
|
+
block.pop("ilevel", None)
|
|
170
|
+
content = block.get("content")
|
|
171
|
+
if isinstance(content, list):
|
|
172
|
+
pending_blocks.extend(child for child in content if isinstance(child, dict))
|
|
173
|
+
return index_blocks
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _clear_deeper_title_counters(counters: dict[int, int], level: int) -> None:
|
|
177
|
+
"""删除当前标题层级之后的旧计数。"""
|
|
178
|
+
for counter_level in [value for value in counters if value > level]:
|
|
179
|
+
del counters[counter_level]
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _visible_text(content: list[dict[str, Any]]) -> str:
|
|
183
|
+
"""提取结构化 Span 的可见文本,供显式标题编号识别。"""
|
|
184
|
+
return inline_span_plain_text(content)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def fix_office_list_blocks(list_blocks: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
188
|
+
"""将每层 Office 列表的局部序号写入文本内容,并移除原始元数据。"""
|
|
189
|
+
|
|
190
|
+
def get_ordered_list_start(list_block: dict[str, Any]) -> int:
|
|
191
|
+
"""读取有序列表起始编号,保留合法的零值。"""
|
|
192
|
+
start = list_block.get("start")
|
|
193
|
+
if start is None:
|
|
194
|
+
return 1
|
|
195
|
+
try:
|
|
196
|
+
start = int(start)
|
|
197
|
+
except (TypeError, ValueError):
|
|
198
|
+
return 1
|
|
199
|
+
return start if start >= 0 else 1
|
|
200
|
+
|
|
201
|
+
def fix_list_block(list_block: dict[str, Any]) -> None:
|
|
202
|
+
"""递归处理列表树;每个有序列表只维护当前层的独立编号。"""
|
|
203
|
+
is_ordered = list_block.get("attribute") == "ordered"
|
|
204
|
+
ordered_number = get_ordered_list_start(list_block)
|
|
205
|
+
content = list_block.get("content")
|
|
206
|
+
if isinstance(content, list):
|
|
207
|
+
for child_block in content:
|
|
208
|
+
if not isinstance(child_block, dict):
|
|
209
|
+
continue
|
|
210
|
+
child_type = child_block.get("type")
|
|
211
|
+
if child_type == BlockType.TEXT:
|
|
212
|
+
child_content = child_block.get("content")
|
|
213
|
+
if not isinstance(child_content, list):
|
|
214
|
+
child_block.pop("list_label", None)
|
|
215
|
+
continue
|
|
216
|
+
exact_label = child_block.pop("list_label", None)
|
|
217
|
+
if isinstance(exact_label, str) and exact_label.strip():
|
|
218
|
+
prefix = f"{exact_label.strip()} "
|
|
219
|
+
if is_ordered:
|
|
220
|
+
ordered_number += 1
|
|
221
|
+
elif is_ordered:
|
|
222
|
+
prefix = f"{ordered_number}. "
|
|
223
|
+
ordered_number += 1
|
|
224
|
+
else:
|
|
225
|
+
prefix = "- "
|
|
226
|
+
child_block["content"] = [*text_spans(prefix), *child_content]
|
|
227
|
+
elif child_type == BlockType.LIST:
|
|
228
|
+
fix_list_block(child_block)
|
|
229
|
+
list_block.pop("attribute", None)
|
|
230
|
+
list_block.pop("ilevel", None)
|
|
231
|
+
list_block.pop("start", None)
|
|
232
|
+
|
|
233
|
+
for list_block in list_blocks:
|
|
234
|
+
if isinstance(list_block, dict):
|
|
235
|
+
fix_list_block(list_block)
|
|
236
|
+
return list_blocks
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""单页 raw block 的内容清理、列表整理和视觉分组流水线。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from ..schema import RAW_ALGORITHM, RAW_CAPTION, RAW_FOOTNOTE, BlockType, VISUAL_MAIN_TYPES
|
|
8
|
+
from ..foundation.language import guess_code_language
|
|
9
|
+
|
|
10
|
+
from ..foundation.text import clean_isolated_formula
|
|
11
|
+
|
|
12
|
+
from .content import clean_inline_content, code_content_clean, collapse_inline_newlines
|
|
13
|
+
from .lists import fix_office_index_blocks, fix_office_list_blocks, fix_pdf_index_blocks, fix_pdf_list_blocks
|
|
14
|
+
from .visual import (
|
|
15
|
+
fallback_inline_caption_fragments,
|
|
16
|
+
fallback_leading_table_continuation_captions,
|
|
17
|
+
fallback_no_bbox_caption_fragments,
|
|
18
|
+
regroup_visual_blocks,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
BlockDict = dict[str, Any]
|
|
22
|
+
|
|
23
|
+
_REPLACED_BLOCK_TYPES = {
|
|
24
|
+
BlockType.TEXT,
|
|
25
|
+
BlockType.REF_TEXT,
|
|
26
|
+
BlockType.LIST,
|
|
27
|
+
BlockType.INDEX,
|
|
28
|
+
RAW_CAPTION,
|
|
29
|
+
RAW_FOOTNOTE,
|
|
30
|
+
BlockType.IMAGE_BODY,
|
|
31
|
+
BlockType.TABLE_BODY,
|
|
32
|
+
BlockType.CHART_BODY,
|
|
33
|
+
BlockType.CODE_BODY,
|
|
34
|
+
BlockType.IMAGE_CAPTION,
|
|
35
|
+
BlockType.IMAGE_FOOTNOTE,
|
|
36
|
+
BlockType.TABLE_CAPTION,
|
|
37
|
+
BlockType.TABLE_FOOTNOTE,
|
|
38
|
+
BlockType.CHART_CAPTION,
|
|
39
|
+
BlockType.CHART_FOOTNOTE,
|
|
40
|
+
BlockType.CODE_CAPTION,
|
|
41
|
+
BlockType.CODE_FOOTNOTE,
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
_INLINE_CONTENT_BLOCK_TYPES = {
|
|
46
|
+
BlockType.TEXT,
|
|
47
|
+
BlockType.REF_TEXT,
|
|
48
|
+
BlockType.DOC_TITLE,
|
|
49
|
+
BlockType.PARAGRAPH_TITLE,
|
|
50
|
+
BlockType.HEADER,
|
|
51
|
+
BlockType.FOOTER,
|
|
52
|
+
BlockType.PAGE_NUMBER,
|
|
53
|
+
BlockType.ASIDE_TEXT,
|
|
54
|
+
BlockType.PAGE_FOOTNOTE,
|
|
55
|
+
RAW_CAPTION,
|
|
56
|
+
RAW_FOOTNOTE,
|
|
57
|
+
BlockType.IMAGE_CAPTION,
|
|
58
|
+
BlockType.IMAGE_FOOTNOTE,
|
|
59
|
+
BlockType.TABLE_CAPTION,
|
|
60
|
+
BlockType.TABLE_FOOTNOTE,
|
|
61
|
+
BlockType.CHART_CAPTION,
|
|
62
|
+
BlockType.CHART_FOOTNOTE,
|
|
63
|
+
BlockType.CODE_CAPTION,
|
|
64
|
+
BlockType.CODE_FOOTNOTE,
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _normalize_raw_blocks(page_model_list: list[BlockDict]) -> list[BlockDict]:
|
|
69
|
+
"""原地规范化 raw block 类型、内容、索引和代码子类型。"""
|
|
70
|
+
blocks: list[BlockDict] = []
|
|
71
|
+
for index, block in enumerate(page_model_list):
|
|
72
|
+
code_block_sub_type = None
|
|
73
|
+
block_type = block.get("type", "")
|
|
74
|
+
block_content = block.get("content", "")
|
|
75
|
+
if block_type == BlockType.IMAGE:
|
|
76
|
+
block_type = BlockType.IMAGE_BODY
|
|
77
|
+
elif block_type == BlockType.TABLE:
|
|
78
|
+
block_type = BlockType.TABLE_BODY
|
|
79
|
+
elif block_type == BlockType.CHART:
|
|
80
|
+
block_type = BlockType.CHART_BODY
|
|
81
|
+
elif block_type == BlockType.CODE:
|
|
82
|
+
code_block_sub_type = block_type
|
|
83
|
+
block_content = code_content_clean(block_content)
|
|
84
|
+
block_type = BlockType.CODE_BODY
|
|
85
|
+
elif block_type == RAW_ALGORITHM:
|
|
86
|
+
code_block_sub_type = block_type
|
|
87
|
+
block_content = clean_inline_content(block_content)
|
|
88
|
+
block_type = BlockType.CODE_BODY
|
|
89
|
+
elif block_type == BlockType.EQUATION:
|
|
90
|
+
block_content = clean_isolated_formula(block_content)
|
|
91
|
+
|
|
92
|
+
if block_type in [BlockType.IMAGE_BODY, BlockType.CHART_BODY] and block_content is None:
|
|
93
|
+
block_content = ""
|
|
94
|
+
if block_type == BlockType.DOC_TITLE:
|
|
95
|
+
block["level"] = 1
|
|
96
|
+
elif block_type == BlockType.PARAGRAPH_TITLE:
|
|
97
|
+
raw_level = block.get("level")
|
|
98
|
+
normalized_level = raw_level if type(raw_level) is int else 2
|
|
99
|
+
block["level"] = min(max(normalized_level, 2), 6)
|
|
100
|
+
|
|
101
|
+
if block_type in _INLINE_CONTENT_BLOCK_TYPES:
|
|
102
|
+
block_content = clean_inline_content(block_content)
|
|
103
|
+
if block_type in [BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE]:
|
|
104
|
+
block_content = collapse_inline_newlines(block_content)
|
|
105
|
+
|
|
106
|
+
block["type"] = block_type
|
|
107
|
+
block["content"] = block_content
|
|
108
|
+
block["index"] = index
|
|
109
|
+
if code_block_sub_type:
|
|
110
|
+
block["sub_type"] = code_block_sub_type
|
|
111
|
+
blocks.append(block)
|
|
112
|
+
return blocks
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _apply_caption_fallbacks(blocks: list[BlockDict], *, use_bbox: bool) -> None:
|
|
116
|
+
"""按 PDF 或 Office 结构应用视觉标题兜底规则。"""
|
|
117
|
+
if use_bbox:
|
|
118
|
+
fallback_inline_caption_fragments(blocks, VISUAL_MAIN_TYPES)
|
|
119
|
+
fallback_leading_table_continuation_captions(blocks, VISUAL_MAIN_TYPES)
|
|
120
|
+
else:
|
|
121
|
+
fallback_no_bbox_caption_fragments(blocks, VISUAL_MAIN_TYPES)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _partition_textual_blocks(
|
|
125
|
+
blocks: list[BlockDict],
|
|
126
|
+
) -> tuple[list[BlockDict], list[BlockDict], list[BlockDict], list[BlockDict]]:
|
|
127
|
+
"""按 text、ref_text、list、index 四类分区 raw block。"""
|
|
128
|
+
text_blocks: list[BlockDict] = []
|
|
129
|
+
ref_text_blocks: list[BlockDict] = []
|
|
130
|
+
list_blocks: list[BlockDict] = []
|
|
131
|
+
index_blocks: list[BlockDict] = []
|
|
132
|
+
for block in blocks:
|
|
133
|
+
block_type = block["type"]
|
|
134
|
+
if block_type == BlockType.TEXT:
|
|
135
|
+
text_blocks.append(block)
|
|
136
|
+
elif block_type == BlockType.REF_TEXT:
|
|
137
|
+
ref_text_blocks.append(block)
|
|
138
|
+
elif block_type == BlockType.LIST:
|
|
139
|
+
list_blocks.append(block)
|
|
140
|
+
elif block_type == BlockType.INDEX:
|
|
141
|
+
index_blocks.append(block)
|
|
142
|
+
return text_blocks, ref_text_blocks, list_blocks, index_blocks
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _prepare_lists_and_indices(
|
|
146
|
+
blocks: list[BlockDict],
|
|
147
|
+
*,
|
|
148
|
+
use_bbox: bool,
|
|
149
|
+
) -> tuple[list[BlockDict], list[BlockDict], list[BlockDict], list[BlockDict]]:
|
|
150
|
+
"""根据文档类型整理列表、目录以及被列表吸收的文本块。"""
|
|
151
|
+
text_blocks, ref_text_blocks, list_blocks, index_blocks = _partition_textual_blocks(blocks)
|
|
152
|
+
if use_bbox:
|
|
153
|
+
list_blocks, text_blocks, ref_text_blocks = fix_pdf_list_blocks(
|
|
154
|
+
list_blocks,
|
|
155
|
+
text_blocks,
|
|
156
|
+
ref_text_blocks,
|
|
157
|
+
)
|
|
158
|
+
index_blocks = fix_pdf_index_blocks(index_blocks)
|
|
159
|
+
else:
|
|
160
|
+
list_blocks = fix_office_list_blocks(list_blocks)
|
|
161
|
+
index_blocks = fix_office_index_blocks(index_blocks)
|
|
162
|
+
return text_blocks, ref_text_blocks, list_blocks, index_blocks
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _finalize_code_blocks(code_blocks: list[BlockDict]) -> None:
|
|
166
|
+
"""确定代码语言,并把算法唯一主体切换为 algorithm_body。"""
|
|
167
|
+
for code_block in code_blocks:
|
|
168
|
+
if code_block["sub_type"] == RAW_ALGORITHM:
|
|
169
|
+
for sub_block in code_block["content"]:
|
|
170
|
+
if sub_block.get("type") == BlockType.CODE_BODY:
|
|
171
|
+
sub_block["type"] = BlockType.ALGORITHM_BODY
|
|
172
|
+
break
|
|
173
|
+
continue
|
|
174
|
+
guess_lang = code_block.get("guess_lang")
|
|
175
|
+
if isinstance(guess_lang, str) and guess_lang.strip():
|
|
176
|
+
code_block["guess_lang"] = guess_lang.strip()
|
|
177
|
+
continue
|
|
178
|
+
for sub_block in code_block["content"]:
|
|
179
|
+
if sub_block.get("type") == BlockType.CODE_BODY:
|
|
180
|
+
code_block["guess_lang"] = guess_code_language(sub_block.get("content", ""))
|
|
181
|
+
break
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def process_page_blocks(
|
|
185
|
+
page_model_list: list[BlockDict],
|
|
186
|
+
*,
|
|
187
|
+
use_bbox: bool | None = None,
|
|
188
|
+
) -> list[BlockDict]:
|
|
189
|
+
"""按固定阶段将单页 raw model-list 转换为可对象化的顶层 blocks。"""
|
|
190
|
+
resolved_use_bbox = any(block.get("bbox") for block in page_model_list) if use_bbox is None else use_bbox
|
|
191
|
+
blocks = _normalize_raw_blocks(page_model_list)
|
|
192
|
+
_apply_caption_fallbacks(blocks, use_bbox=resolved_use_bbox)
|
|
193
|
+
text_blocks, ref_text_blocks, list_blocks, index_blocks = _prepare_lists_and_indices(
|
|
194
|
+
blocks,
|
|
195
|
+
use_bbox=resolved_use_bbox,
|
|
196
|
+
)
|
|
197
|
+
visual_groups, unmatched_child_blocks = regroup_visual_blocks(
|
|
198
|
+
blocks,
|
|
199
|
+
use_bbox=resolved_use_bbox,
|
|
200
|
+
)
|
|
201
|
+
image_blocks = visual_groups[BlockType.IMAGE]
|
|
202
|
+
table_blocks = visual_groups[BlockType.TABLE]
|
|
203
|
+
chart_blocks = visual_groups[BlockType.CHART]
|
|
204
|
+
code_blocks = visual_groups[BlockType.CODE]
|
|
205
|
+
_finalize_code_blocks(code_blocks)
|
|
206
|
+
for block in unmatched_child_blocks:
|
|
207
|
+
block["type"] = BlockType.TEXT
|
|
208
|
+
text_blocks.append(block)
|
|
209
|
+
|
|
210
|
+
result = [block for block in blocks if block["type"] not in _REPLACED_BLOCK_TYPES]
|
|
211
|
+
result.extend(
|
|
212
|
+
list_blocks + text_blocks + ref_text_blocks + index_blocks + image_blocks + table_blocks + chart_blocks + code_blocks
|
|
213
|
+
)
|
|
214
|
+
return result
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""严格 ModelJson 到 PageInfo 列表的唯一转换边界。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from pydantic import TypeAdapter
|
|
9
|
+
|
|
10
|
+
from .lists import fix_office_index_title_blocks, fix_office_paragraph_titles
|
|
11
|
+
from .page_blocks import process_page_blocks
|
|
12
|
+
from .paragraphs import merge_para_text_blocks
|
|
13
|
+
from ..content.table import merge_table
|
|
14
|
+
from ..schema import ModelJson, PageInfo
|
|
15
|
+
|
|
16
|
+
PAGE_INFO_LIST_ADAPTER = TypeAdapter(list[PageInfo])
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _document_uses_bbox(model_list: list[list[dict[str, Any]]]) -> bool:
|
|
20
|
+
"""按整份文档是否出现 bbox 判定 PDF/Office,避免空白首页误判。"""
|
|
21
|
+
return any(
|
|
22
|
+
block.get("bbox") is not None for page_model_list in model_list for block in page_model_list if isinstance(block, dict)
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _remove_private_block_metadata(block: dict[str, Any]) -> None:
|
|
27
|
+
"""递归清除对象化边界之前仅供 Analyze 计算使用的临时字段。"""
|
|
28
|
+
for field_name in ("lines", "_lines", "angle", "score", "label"):
|
|
29
|
+
block.pop(field_name, None)
|
|
30
|
+
content = block.get("content")
|
|
31
|
+
if isinstance(content, list):
|
|
32
|
+
for child in content:
|
|
33
|
+
if isinstance(child, dict):
|
|
34
|
+
_remove_private_block_metadata(child)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _blocks_to_raw_page_info(
|
|
38
|
+
page_model_list: list[dict[str, Any]],
|
|
39
|
+
*,
|
|
40
|
+
page_idx: int,
|
|
41
|
+
use_bbox: bool,
|
|
42
|
+
) -> dict[str, Any]:
|
|
43
|
+
"""运行单页后处理流水线并保留 raw dict,供跨页处理继续消费。"""
|
|
44
|
+
page_blocks = process_page_blocks(page_model_list, use_bbox=use_bbox)
|
|
45
|
+
page_blocks.sort(key=lambda block: block["index"])
|
|
46
|
+
return {"page_idx": page_idx, "blocks": page_blocks}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def blocks_to_page_info(
|
|
50
|
+
page_model_list: list[dict[str, Any]],
|
|
51
|
+
*,
|
|
52
|
+
page_idx: int = 0,
|
|
53
|
+
use_bbox: bool | None = None,
|
|
54
|
+
) -> PageInfo:
|
|
55
|
+
"""无副作用地把单页 raw blocks 转换为严格 PageInfo 对象。"""
|
|
56
|
+
copied_blocks = deepcopy(page_model_list)
|
|
57
|
+
resolved_use_bbox = _document_uses_bbox([copied_blocks]) if use_bbox is None else use_bbox
|
|
58
|
+
if not resolved_use_bbox:
|
|
59
|
+
fix_office_paragraph_titles([copied_blocks])
|
|
60
|
+
fix_office_index_title_blocks([copied_blocks])
|
|
61
|
+
raw_page = _blocks_to_raw_page_info(
|
|
62
|
+
copied_blocks,
|
|
63
|
+
page_idx=page_idx,
|
|
64
|
+
use_bbox=resolved_use_bbox,
|
|
65
|
+
)
|
|
66
|
+
for block in raw_page["blocks"]:
|
|
67
|
+
_remove_private_block_metadata(block)
|
|
68
|
+
return PageInfo.model_validate(raw_page)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def model_json_to_pages(model_json: ModelJson) -> list[PageInfo]:
|
|
72
|
+
"""从严格 ModelJson 无副作用地构造可递归序列化的 PageInfo。"""
|
|
73
|
+
page_indices = model_json.resolved_page_indices
|
|
74
|
+
copied_model_list = deepcopy(model_json.pages)
|
|
75
|
+
use_bbox = _document_uses_bbox(copied_model_list)
|
|
76
|
+
if not use_bbox:
|
|
77
|
+
fix_office_paragraph_titles(copied_model_list)
|
|
78
|
+
fix_office_index_title_blocks(copied_model_list)
|
|
79
|
+
|
|
80
|
+
raw_pages = [
|
|
81
|
+
_blocks_to_raw_page_info(
|
|
82
|
+
page_model_list,
|
|
83
|
+
page_idx=page_idx,
|
|
84
|
+
use_bbox=use_bbox,
|
|
85
|
+
)
|
|
86
|
+
for page_model_list, page_idx in zip(copied_model_list, page_indices, strict=True)
|
|
87
|
+
]
|
|
88
|
+
if use_bbox:
|
|
89
|
+
merge_para_text_blocks(raw_pages)
|
|
90
|
+
merge_table(raw_pages)
|
|
91
|
+
|
|
92
|
+
for page in raw_pages:
|
|
93
|
+
for block in page["blocks"]:
|
|
94
|
+
_remove_private_block_metadata(block)
|
|
95
|
+
return PAGE_INFO_LIST_ADAPTER.validate_python(raw_pages)
|