docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""Flash 各格式构造 Middle JSON 2.0 行内 Span 的轻量工具。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from ..schema import (
|
|
9
|
+
INLINE_STYLE_ORDER,
|
|
10
|
+
CodeInlineSpan,
|
|
11
|
+
EquationInlineSpan,
|
|
12
|
+
HyperlinkSpan,
|
|
13
|
+
InlineSpan,
|
|
14
|
+
TextSpan,
|
|
15
|
+
parse_inline_spans,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def append_text_span(output: list[dict[str, Any]], content: str, styles: Iterable[str] = ()) -> None:
|
|
20
|
+
"""追加非空 TextSpan,并合并相邻同样式文字。"""
|
|
21
|
+
if not content:
|
|
22
|
+
return
|
|
23
|
+
style_set = set(styles)
|
|
24
|
+
normalized_styles = [style for style in INLINE_STYLE_ORDER if style in style_set]
|
|
25
|
+
if output and output[-1].get("type") == "text" and output[-1].get("styles", []) == normalized_styles:
|
|
26
|
+
output[-1]["content"] = f"{output[-1].get('content', '')}{content}"
|
|
27
|
+
return
|
|
28
|
+
span: dict[str, Any] = {"type": "text", "content": content}
|
|
29
|
+
if normalized_styles:
|
|
30
|
+
span["styles"] = normalized_styles
|
|
31
|
+
output.append(span)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def text_spans(content: str, styles: Iterable[str] = ()) -> list[dict[str, Any]]:
|
|
35
|
+
"""把一段文字构造成零个或一个规范 TextSpan。"""
|
|
36
|
+
output: list[dict[str, Any]] = []
|
|
37
|
+
append_text_span(output, content, styles)
|
|
38
|
+
return output
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def append_equation_span(output: list[dict[str, Any]], latex: str) -> None:
|
|
42
|
+
"""追加非空行内公式 Span。"""
|
|
43
|
+
normalized = latex.strip()
|
|
44
|
+
if normalized:
|
|
45
|
+
output.append({"type": "equation_inline", "content": normalized})
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def append_code_span(output: list[dict[str, Any]], content: str) -> None:
|
|
49
|
+
"""追加非空行内代码 Span。"""
|
|
50
|
+
if content:
|
|
51
|
+
output.append({"type": "code_inline", "content": content})
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def append_hyperlink_span(output: list[dict[str, Any]], children: list[dict[str, Any]], url: str | None) -> None:
|
|
55
|
+
"""追加安全超链接;目标缺失时把标签子 Span 直接降级为普通内容。"""
|
|
56
|
+
if not children:
|
|
57
|
+
return
|
|
58
|
+
normalized_url = (url or "").strip()
|
|
59
|
+
if not normalized_url:
|
|
60
|
+
extend_inline_spans(output, children)
|
|
61
|
+
return
|
|
62
|
+
non_link_children = [child for child in children if child.get("type") != "hyperlink"]
|
|
63
|
+
if not non_link_children:
|
|
64
|
+
return
|
|
65
|
+
output.append({"type": "hyperlink", "url": normalized_url, "content": non_link_children})
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def extend_inline_spans(output: list[dict[str, Any]], spans: Iterable[dict[str, Any]]) -> None:
|
|
69
|
+
"""追加 Span 序列,并在边界合并同样式文字。"""
|
|
70
|
+
for span in spans:
|
|
71
|
+
if span.get("type") == "text" and isinstance(span.get("content"), str):
|
|
72
|
+
append_text_span(output, str(span["content"]), span.get("styles", []))
|
|
73
|
+
else:
|
|
74
|
+
output.append(span)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def normalize_span_dicts(spans: Iterable[dict[str, Any] | InlineSpan]) -> list[dict[str, Any]]:
|
|
78
|
+
"""严格校验 Span 并返回确定性的 JSON 字典列表。"""
|
|
79
|
+
parsed = parse_inline_spans(list(spans))
|
|
80
|
+
output: list[dict[str, Any]] = []
|
|
81
|
+
for span in parsed:
|
|
82
|
+
payload = span.model_dump(mode="json")
|
|
83
|
+
if payload.get("type") == "text":
|
|
84
|
+
append_text_span(output, str(payload["content"]), payload.get("styles", []))
|
|
85
|
+
else:
|
|
86
|
+
output.append(payload)
|
|
87
|
+
return output
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def inline_span_plain_text(spans: Iterable[dict[str, Any]]) -> str:
|
|
91
|
+
"""从 raw Span 字典中提取可见文本。"""
|
|
92
|
+
parts: list[str] = []
|
|
93
|
+
for span in spans:
|
|
94
|
+
span_type = span.get("type")
|
|
95
|
+
if span_type in {"text", "equation_inline", "code_inline"}:
|
|
96
|
+
content = span.get("content")
|
|
97
|
+
if isinstance(content, str):
|
|
98
|
+
parts.append(content)
|
|
99
|
+
elif span_type == "hyperlink":
|
|
100
|
+
children = span.get("content")
|
|
101
|
+
if isinstance(children, list):
|
|
102
|
+
parts.append(inline_span_plain_text(child for child in children if isinstance(child, dict)))
|
|
103
|
+
return "".join(parts)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def slice_span_dicts(
|
|
107
|
+
spans: Iterable[dict[str, Any] | InlineSpan], start: int = 0, end: int | None = None
|
|
108
|
+
) -> list[dict[str, Any]]:
|
|
109
|
+
"""按可见字符偏移裁剪 raw Span,并保留样式和链接。"""
|
|
110
|
+
parsed = parse_inline_spans(list(spans))
|
|
111
|
+
visible_length = len(_typed_plain_text(parsed))
|
|
112
|
+
resolved_start = min(max(start, 0), visible_length)
|
|
113
|
+
resolved_end = visible_length if end is None else min(max(end, resolved_start), visible_length)
|
|
114
|
+
output: list[InlineSpan] = []
|
|
115
|
+
cursor = 0
|
|
116
|
+
for span in parsed:
|
|
117
|
+
span_length = len(_typed_plain_text([span]))
|
|
118
|
+
span_end = cursor + span_length
|
|
119
|
+
overlap_start = max(resolved_start, cursor)
|
|
120
|
+
overlap_end = min(resolved_end, span_end)
|
|
121
|
+
if overlap_start < overlap_end:
|
|
122
|
+
sliced = _slice_typed_span(span, overlap_start - cursor, overlap_end - cursor)
|
|
123
|
+
if sliced is not None:
|
|
124
|
+
output.append(sliced)
|
|
125
|
+
cursor = span_end
|
|
126
|
+
if cursor >= resolved_end:
|
|
127
|
+
break
|
|
128
|
+
return normalize_span_dicts(output)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def strip_span_dicts(spans: Iterable[dict[str, Any] | InlineSpan]) -> list[dict[str, Any]]:
|
|
132
|
+
"""裁剪 Span 首尾空白,并保持内部样式与链接结构。"""
|
|
133
|
+
parsed = parse_inline_spans(list(spans))
|
|
134
|
+
visible = _typed_plain_text(parsed)
|
|
135
|
+
start = len(visible) - len(visible.lstrip())
|
|
136
|
+
end = len(visible.rstrip())
|
|
137
|
+
return slice_span_dicts(parsed, start, end)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _typed_plain_text(spans: Iterable[InlineSpan]) -> str:
|
|
141
|
+
"""提取已验证 Span 的可见文本。"""
|
|
142
|
+
parts: list[str] = []
|
|
143
|
+
for span in spans:
|
|
144
|
+
if isinstance(span, (TextSpan, EquationInlineSpan, CodeInlineSpan)):
|
|
145
|
+
parts.append(span.content)
|
|
146
|
+
elif isinstance(span, HyperlinkSpan):
|
|
147
|
+
parts.append(_typed_plain_text(span.content))
|
|
148
|
+
return "".join(parts)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _slice_typed_span(span: InlineSpan, start: int, end: int) -> InlineSpan | None:
|
|
152
|
+
"""裁剪一个已验证 Span 的局部区间。"""
|
|
153
|
+
if start >= end:
|
|
154
|
+
return None
|
|
155
|
+
if isinstance(span, TextSpan):
|
|
156
|
+
content = span.content[start:end]
|
|
157
|
+
return span.model_copy(update={"content": content}) if content else None
|
|
158
|
+
if isinstance(span, EquationInlineSpan):
|
|
159
|
+
content = span.content[start:end]
|
|
160
|
+
return span.model_copy(update={"content": content}) if content.strip() else None
|
|
161
|
+
if isinstance(span, CodeInlineSpan):
|
|
162
|
+
content = span.content[start:end]
|
|
163
|
+
return span.model_copy(update={"content": content}) if content else None
|
|
164
|
+
if isinstance(span, HyperlinkSpan):
|
|
165
|
+
children = slice_span_dicts(span.content, start, end)
|
|
166
|
+
parsed_children = parse_inline_spans(children)
|
|
167
|
+
non_links = [child for child in parsed_children if not isinstance(child, HyperlinkSpan)]
|
|
168
|
+
return span.model_copy(update={"content": non_links}) if non_links else None
|
|
169
|
+
return None
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
__all__ = [
|
|
173
|
+
"append_code_span",
|
|
174
|
+
"append_equation_span",
|
|
175
|
+
"append_hyperlink_span",
|
|
176
|
+
"append_text_span",
|
|
177
|
+
"extend_inline_spans",
|
|
178
|
+
"inline_span_plain_text",
|
|
179
|
+
"normalize_span_dicts",
|
|
180
|
+
"slice_span_dicts",
|
|
181
|
+
"strip_span_dicts",
|
|
182
|
+
"text_spans",
|
|
183
|
+
]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""跨页表格结构检测和内容合并的稳定入口。"""
|
|
2
|
+
|
|
3
|
+
from .content import merge_table_content
|
|
4
|
+
from .document import merge_table
|
|
5
|
+
from .html import build_row_rendered_cell_segments, build_table_state_from_html, calculate_row_rendered_segments
|
|
6
|
+
from .structure import can_merge_by_structure, detect_table_headers
|
|
7
|
+
from .structure import _expand_header_count_by_rowspan as expand_header_count_by_rowspan
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"merge_table",
|
|
11
|
+
"merge_table_content",
|
|
12
|
+
"build_table_state_from_html",
|
|
13
|
+
"build_row_rendered_cell_segments",
|
|
14
|
+
"can_merge_by_structure",
|
|
15
|
+
"calculate_row_rendered_segments",
|
|
16
|
+
"detect_table_headers",
|
|
17
|
+
"expand_header_count_by_rowspan",
|
|
18
|
+
]
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""DocVortex table block 的主体、辅助文本和 bbox 访问规则。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from bs4 import BeautifulSoup
|
|
9
|
+
|
|
10
|
+
from ...schema import BlockType
|
|
11
|
+
|
|
12
|
+
from .html import _build_front_cache, _scan_rows
|
|
13
|
+
from .models import MAX_HEADER_ROWS, BlockDict, CalculationBBox, TableMergeState
|
|
14
|
+
from .rules import is_table_continuation_text
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _bbox_for_calculation(bbox: Any) -> CalculationBBox | None:
|
|
18
|
+
"""复制归一化 bbox 并放大为千分位整数,原始字段保持不变。"""
|
|
19
|
+
if not isinstance(bbox, (list, tuple)) or len(bbox) != 4:
|
|
20
|
+
return None
|
|
21
|
+
if any(isinstance(value, bool) or not isinstance(value, (int, float)) for value in bbox):
|
|
22
|
+
return None
|
|
23
|
+
|
|
24
|
+
values = tuple(float(value) for value in bbox)
|
|
25
|
+
if not all(math.isfinite(value) and 0 <= value <= 1 for value in values):
|
|
26
|
+
return None
|
|
27
|
+
|
|
28
|
+
x0, y0, x1, y1 = (int(round(value * 1000)) for value in values)
|
|
29
|
+
if x1 <= x0 or y1 <= y0:
|
|
30
|
+
return None
|
|
31
|
+
return x0, y0, x1, y1
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _table_children(table_block: BlockDict) -> list[BlockDict]:
|
|
35
|
+
"""读取 table 根块下的合法 dict 子块。"""
|
|
36
|
+
content = table_block.get("content")
|
|
37
|
+
if not isinstance(content, list):
|
|
38
|
+
return []
|
|
39
|
+
return [block for block in content if isinstance(block, dict)]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _find_table_body_block(table_block: BlockDict) -> BlockDict | None:
|
|
43
|
+
"""查找 dict table block 中的主体子块。"""
|
|
44
|
+
for block in _table_children(table_block):
|
|
45
|
+
if block.get("type") == BlockType.TABLE_BODY:
|
|
46
|
+
return block
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _build_post_body_child_index(table_block: BlockDict, offset: int) -> int | None:
|
|
51
|
+
"""为复制到前表的 footnote 生成表体后的安全 index。"""
|
|
52
|
+
body_block = _find_table_body_block(table_block)
|
|
53
|
+
if body_block is None:
|
|
54
|
+
return None
|
|
55
|
+
body_index = body_block.get("index")
|
|
56
|
+
if not isinstance(body_index, int):
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
child_indices = [block.get("index") for block in _table_children(table_block) if isinstance(block.get("index"), int)]
|
|
60
|
+
return max([body_index, *child_indices]) + offset
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _block_text(block: BlockDict) -> str:
|
|
64
|
+
"""递归读取 dict block 的文本内容,供续表标记判断使用。"""
|
|
65
|
+
content = block.get("content")
|
|
66
|
+
if isinstance(content, str):
|
|
67
|
+
return content
|
|
68
|
+
if not isinstance(content, list):
|
|
69
|
+
return ""
|
|
70
|
+
return "".join(_block_text(child) for child in content if isinstance(child, dict))
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _is_continuation_caption(caption_block: BlockDict) -> bool:
|
|
74
|
+
"""判断 dict caption 文本是否带有续表标记。"""
|
|
75
|
+
return is_table_continuation_text(_block_text(caption_block))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _is_post_table_non_continuation_caption(table_block: BlockDict, caption_block: BlockDict) -> bool:
|
|
79
|
+
"""判断 caption 是否是误挂到表格下方的新段落标题。
|
|
80
|
+
|
|
81
|
+
这类 caption 位于 table body 下方,且不含续表标记;它不应作为
|
|
82
|
+
当前表的新标题阻断跨页关系判断。
|
|
83
|
+
"""
|
|
84
|
+
if _is_continuation_caption(caption_block):
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
body_block = _find_table_body_block(table_block)
|
|
88
|
+
if body_block is None:
|
|
89
|
+
return False
|
|
90
|
+
|
|
91
|
+
body_bbox = _bbox_for_calculation(body_block.get("bbox"))
|
|
92
|
+
caption_bbox = _bbox_for_calculation(caption_block.get("bbox"))
|
|
93
|
+
if body_bbox is None or caption_bbox is None:
|
|
94
|
+
return False
|
|
95
|
+
|
|
96
|
+
return caption_bbox[1] >= body_bbox[3]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _build_table_state(table_block: BlockDict, max_header_rows: int = MAX_HEADER_ROWS) -> TableMergeState | None:
|
|
100
|
+
"""从 dict table block 构建结构缓存,非法主体安全返回空。"""
|
|
101
|
+
body_block = _find_table_body_block(table_block)
|
|
102
|
+
if body_block is None:
|
|
103
|
+
return None
|
|
104
|
+
|
|
105
|
+
html = body_block.get("content")
|
|
106
|
+
if not isinstance(html, str) or not html:
|
|
107
|
+
return None
|
|
108
|
+
|
|
109
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
110
|
+
tbody = soup.find("tbody") or soup.find("table")
|
|
111
|
+
rows = soup.find_all("tr")
|
|
112
|
+
if tbody is None or not rows:
|
|
113
|
+
return None
|
|
114
|
+
|
|
115
|
+
scan = _scan_rows(rows)
|
|
116
|
+
if scan.total_cols <= 0 or scan.last_nonempty_row_metrics is None:
|
|
117
|
+
return None
|
|
118
|
+
front_header_info, front_first_data_row_metrics = _build_front_cache(rows, max_header_rows=max_header_rows)
|
|
119
|
+
|
|
120
|
+
return TableMergeState(
|
|
121
|
+
owner_block=table_block,
|
|
122
|
+
body_block=body_block,
|
|
123
|
+
soup=soup,
|
|
124
|
+
tbody=tbody,
|
|
125
|
+
rows=rows,
|
|
126
|
+
total_cols=scan.total_cols,
|
|
127
|
+
front_header_info=front_header_info,
|
|
128
|
+
front_first_data_row_metrics=front_first_data_row_metrics,
|
|
129
|
+
last_data_row_metrics=scan.last_nonempty_row_metrics,
|
|
130
|
+
row_effective_cols=scan.row_effective_cols,
|
|
131
|
+
tail_occupied=scan.tail_occupied,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _get_or_create_table_state(
|
|
136
|
+
table_block: BlockDict,
|
|
137
|
+
state_cache: dict[int, TableMergeState],
|
|
138
|
+
max_header_rows: int = MAX_HEADER_ROWS,
|
|
139
|
+
) -> TableMergeState | None:
|
|
140
|
+
"""按 table dict 对象身份复用 HTML 结构扫描结果。"""
|
|
141
|
+
cache_key = id(table_block)
|
|
142
|
+
state = state_cache.get(cache_key)
|
|
143
|
+
if state is not None:
|
|
144
|
+
return state
|
|
145
|
+
|
|
146
|
+
try:
|
|
147
|
+
state = _build_table_state(table_block, max_header_rows=max_header_rows)
|
|
148
|
+
except (AssertionError, TypeError, ValueError):
|
|
149
|
+
return None
|
|
150
|
+
if state is not None:
|
|
151
|
+
state_cache[cache_key] = state
|
|
152
|
+
return state
|