docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1237 @@
|
|
|
1
|
+
"""把静态 XHTML/HTML DOM 投影为 DocVortex raw blocks。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import html
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Protocol, TypeAlias
|
|
9
|
+
|
|
10
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
11
|
+
|
|
12
|
+
from docvortex.schema import RAW_ALGORITHM, BlockType, VISUAL_TYPE_MAPPING
|
|
13
|
+
from docvortex.content.spans import (
|
|
14
|
+
append_code_span,
|
|
15
|
+
append_equation_span,
|
|
16
|
+
append_hyperlink_span,
|
|
17
|
+
append_text_span,
|
|
18
|
+
extend_inline_spans,
|
|
19
|
+
inline_span_plain_text,
|
|
20
|
+
strip_span_dicts,
|
|
21
|
+
text_spans,
|
|
22
|
+
)
|
|
23
|
+
from docvortex.foundation.xml_names import local_name
|
|
24
|
+
from docvortex.content.markup.formula import FormulaExtraction, extract_formula
|
|
25
|
+
from docvortex.content.markup.styles import MarkupStylesheet, TextStyle
|
|
26
|
+
from docvortex.foundation.type_identity import preserve_type_module
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
BLOCK_TAGS = frozenset(
|
|
30
|
+
{
|
|
31
|
+
"address",
|
|
32
|
+
"article",
|
|
33
|
+
"aside",
|
|
34
|
+
"blockquote",
|
|
35
|
+
"body",
|
|
36
|
+
"dd",
|
|
37
|
+
"details",
|
|
38
|
+
"div",
|
|
39
|
+
"dl",
|
|
40
|
+
"dt",
|
|
41
|
+
"figcaption",
|
|
42
|
+
"figure",
|
|
43
|
+
"footer",
|
|
44
|
+
"h1",
|
|
45
|
+
"h2",
|
|
46
|
+
"h3",
|
|
47
|
+
"h4",
|
|
48
|
+
"h5",
|
|
49
|
+
"h6",
|
|
50
|
+
"header",
|
|
51
|
+
"hr",
|
|
52
|
+
"main",
|
|
53
|
+
"math",
|
|
54
|
+
"nav",
|
|
55
|
+
"ol",
|
|
56
|
+
"p",
|
|
57
|
+
"pre",
|
|
58
|
+
"section",
|
|
59
|
+
"summary",
|
|
60
|
+
"svg",
|
|
61
|
+
"table",
|
|
62
|
+
"ul",
|
|
63
|
+
}
|
|
64
|
+
)
|
|
65
|
+
SKIPPED_TAGS = frozenset(
|
|
66
|
+
{
|
|
67
|
+
"audio",
|
|
68
|
+
"button",
|
|
69
|
+
"canvas",
|
|
70
|
+
"embed",
|
|
71
|
+
"form",
|
|
72
|
+
"head",
|
|
73
|
+
"iframe",
|
|
74
|
+
"input",
|
|
75
|
+
"noscript",
|
|
76
|
+
"object",
|
|
77
|
+
"script",
|
|
78
|
+
"select",
|
|
79
|
+
"style",
|
|
80
|
+
"template",
|
|
81
|
+
"textarea",
|
|
82
|
+
"video",
|
|
83
|
+
}
|
|
84
|
+
)
|
|
85
|
+
_WHITESPACE_RE = re.compile(r"[\t\r\n\f ]+")
|
|
86
|
+
_XLINK_HREF = "{http://www.w3.org/1999/xlink}href"
|
|
87
|
+
_MAX_TABLE_SPAN = 1_000
|
|
88
|
+
_CAPTION_TOKENS = frozenset(
|
|
89
|
+
{
|
|
90
|
+
"caption",
|
|
91
|
+
"figure-caption",
|
|
92
|
+
"image-caption",
|
|
93
|
+
"table-caption",
|
|
94
|
+
"chart-caption",
|
|
95
|
+
"code-caption",
|
|
96
|
+
"docvortex-caption",
|
|
97
|
+
}
|
|
98
|
+
)
|
|
99
|
+
_FOOTNOTE_TOKENS = frozenset(
|
|
100
|
+
{
|
|
101
|
+
"footnote",
|
|
102
|
+
"figure-footnote",
|
|
103
|
+
"image-footnote",
|
|
104
|
+
"table-footnote",
|
|
105
|
+
"chart-footnote",
|
|
106
|
+
"code-footnote",
|
|
107
|
+
"docvortex-footnote",
|
|
108
|
+
}
|
|
109
|
+
)
|
|
110
|
+
_VISUAL_ELEMENT_TAGS = frozenset({"img", "image", "pre", "svg", "table"})
|
|
111
|
+
_LIST_PAGE_BLOCK_TAGS = frozenset({"figure", "image", "img", "math", "pre", "svg", "table"})
|
|
112
|
+
_InlineSpanDict: TypeAlias = dict[str, object]
|
|
113
|
+
_InlineProjectionSegment: TypeAlias = list[_InlineSpanDict] | dict[str, object]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def clean_text_node(value: str | None) -> str:
|
|
117
|
+
"""折叠普通标记文档文本节点中的排版空白。"""
|
|
118
|
+
return _WHITESPACE_RE.sub(" ", value) if value else ""
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def visible_text(element: etree._Element) -> str:
|
|
122
|
+
"""提取元素折叠空白后的可见纯文本。"""
|
|
123
|
+
return _WHITESPACE_RE.sub(" ", html.unescape("".join(element.itertext()))).strip()
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _semantic_tokens(element: etree._Element) -> frozenset[str]:
|
|
127
|
+
"""按 class/id 的完整空白 token 返回小写集合,不执行任意 substring 匹配。"""
|
|
128
|
+
value = f"{element.get('class') or ''} {element.get('id') or ''}".casefold()
|
|
129
|
+
return frozenset(value.split())
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _raw_visual_type(value: object) -> BlockType | None:
|
|
133
|
+
"""把 raw visual 主体或 algorithm 规范为统一父块类型。"""
|
|
134
|
+
if value in {BlockType.IMAGE, BlockType.TABLE, BlockType.CHART}:
|
|
135
|
+
return BlockType(value)
|
|
136
|
+
if value in {BlockType.CODE, RAW_ALGORITHM}:
|
|
137
|
+
return BlockType.CODE
|
|
138
|
+
return None
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _append_inline_segment(
|
|
142
|
+
segments: list[_InlineProjectionSegment],
|
|
143
|
+
segment: _InlineProjectionSegment,
|
|
144
|
+
) -> None:
|
|
145
|
+
"""追加行内投影片段,并合并相邻 Span 组以保持稳定 block 粒度。"""
|
|
146
|
+
if isinstance(segment, list) and segments and isinstance(segments[-1], list):
|
|
147
|
+
extend_inline_spans(segments[-1], segment)
|
|
148
|
+
elif not isinstance(segment, list) or segment:
|
|
149
|
+
segments.append(segment)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _append_list_block_content(parts: list[_InlineSpanDict], rendered: list[_InlineSpanDict]) -> None:
|
|
153
|
+
"""用换行包围列表项内的块级正文,避免相邻段落静默粘连。"""
|
|
154
|
+
if not rendered:
|
|
155
|
+
return
|
|
156
|
+
last_visible = inline_span_plain_text(parts)
|
|
157
|
+
if last_visible and not last_visible.endswith("\n"):
|
|
158
|
+
append_text_span(parts, "\n")
|
|
159
|
+
extend_inline_spans(parts, rendered)
|
|
160
|
+
append_text_span(parts, "\n")
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def entity_text(element: etree._Element) -> str:
|
|
164
|
+
"""把 lxml 保留的安全命名实体恢复为可见文本。"""
|
|
165
|
+
name = getattr(element, "name", "")
|
|
166
|
+
return html.unescape(f"&{name};") if name else ""
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def bounded_table_span(value: str) -> str | None:
|
|
170
|
+
"""规范化有界表格跨度,避免异常整数放大渲染网格。"""
|
|
171
|
+
if not value.isdigit():
|
|
172
|
+
return None
|
|
173
|
+
normalized = value.lstrip("0")
|
|
174
|
+
if not normalized or len(normalized) > len(str(_MAX_TABLE_SPAN)):
|
|
175
|
+
return None
|
|
176
|
+
span = int(normalized)
|
|
177
|
+
return str(span) if span <= _MAX_TABLE_SPAN else None
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def visible_raw_text_with_style(
|
|
181
|
+
element: etree._Element,
|
|
182
|
+
stylesheet: MarkupStylesheet,
|
|
183
|
+
style: TextStyle,
|
|
184
|
+
visibility_hidden: bool,
|
|
185
|
+
) -> str:
|
|
186
|
+
"""递归提取遵守整树隐藏和继承 visibility 的原始文本。"""
|
|
187
|
+
parts: list[str] = [] if visibility_hidden else [element.text or ""]
|
|
188
|
+
for child in element:
|
|
189
|
+
if isinstance(child.tag, str):
|
|
190
|
+
resolved = stylesheet.resolve(child, style, visibility_hidden)
|
|
191
|
+
if not resolved.subtree_hidden:
|
|
192
|
+
parts.append(
|
|
193
|
+
visible_raw_text_with_style(
|
|
194
|
+
child,
|
|
195
|
+
stylesheet,
|
|
196
|
+
resolved.text,
|
|
197
|
+
resolved.visibility_hidden,
|
|
198
|
+
)
|
|
199
|
+
)
|
|
200
|
+
elif not visibility_hidden:
|
|
201
|
+
parts.append(entity_text(child))
|
|
202
|
+
if not visibility_hidden:
|
|
203
|
+
parts.append(child.tail or "")
|
|
204
|
+
return "".join(parts)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
@dataclass(frozen=True, slots=True)
|
|
208
|
+
class ResolvedMarkupImage:
|
|
209
|
+
"""保存标记文档图片解析后的互斥载荷和说明文本。"""
|
|
210
|
+
|
|
211
|
+
image_base64: str | None = None
|
|
212
|
+
image_url: str | None = None
|
|
213
|
+
alt: str = ""
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
class MarkupContext(Protocol):
|
|
217
|
+
"""定义 projector 向具体容器请求链接、图片和 anchor 的边界。"""
|
|
218
|
+
|
|
219
|
+
def resolve_link(self, href: str) -> str | None:
|
|
220
|
+
"""解析一个安全链接目标。"""
|
|
221
|
+
|
|
222
|
+
def resolve_image(self, source: str, *, alt: str = "") -> ResolvedMarkupImage | None:
|
|
223
|
+
"""解析图片为 data URI、远程 URL 或可见降级文本。"""
|
|
224
|
+
|
|
225
|
+
def heading_anchor(self, heading: etree._Element) -> str | None:
|
|
226
|
+
"""返回标题对应的规范 anchor。"""
|
|
227
|
+
|
|
228
|
+
def heading_label(self, anchor: str) -> str | None:
|
|
229
|
+
"""返回规范 anchor 对应的标题标签。"""
|
|
230
|
+
|
|
231
|
+
def note_anchor(self, note: etree._Element) -> str | None:
|
|
232
|
+
"""返回脚注节点对应的规范 anchor。"""
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
class MarkupProjector:
|
|
236
|
+
"""按 DOM 顺序把一个静态内容根节点投影为统一 raw blocks。"""
|
|
237
|
+
|
|
238
|
+
def __init__(
|
|
239
|
+
self,
|
|
240
|
+
root: etree._Element,
|
|
241
|
+
context: MarkupContext,
|
|
242
|
+
stylesheet: MarkupStylesheet,
|
|
243
|
+
*,
|
|
244
|
+
single_document_title: bool = False,
|
|
245
|
+
document_title_emitted: bool = False,
|
|
246
|
+
) -> None:
|
|
247
|
+
"""绑定 DOM、格式适配器、有限 CSS 和标题策略。"""
|
|
248
|
+
self.root = root
|
|
249
|
+
self.context = context
|
|
250
|
+
self.stylesheet = stylesheet
|
|
251
|
+
self.single_document_title = single_document_title
|
|
252
|
+
self.document_title_emitted = document_title_emitted
|
|
253
|
+
|
|
254
|
+
def convert(self) -> list[dict[str, object]]:
|
|
255
|
+
"""转换内容根节点的子树并返回按 DOM 顺序排列的 raw blocks。"""
|
|
256
|
+
resolved = self.stylesheet.resolve(self.root, TextStyle())
|
|
257
|
+
if resolved.subtree_hidden:
|
|
258
|
+
return []
|
|
259
|
+
name = local_name(self.root)
|
|
260
|
+
if name == "figure" or (name in {"aside", "div", "section"} and self._has_contextual_visual_annotation(self.root)):
|
|
261
|
+
return self._parse_figure(self.root, resolved.text, resolved.visibility_hidden)
|
|
262
|
+
return self._parse_container_contents(self.root, resolved.text, resolved.visibility_hidden)
|
|
263
|
+
|
|
264
|
+
def convert_svg(self) -> list[dict[str, object]]:
|
|
265
|
+
"""把 standalone SVG 根节点尽力转换为文本和静态图片。"""
|
|
266
|
+
resolved = self.stylesheet.resolve(self.root, TextStyle())
|
|
267
|
+
return [] if resolved.subtree_hidden else self._parse_svg(self.root, resolved.text, resolved.visibility_hidden)
|
|
268
|
+
|
|
269
|
+
def project_block(self, element: etree._Element) -> list[dict[str, object]]:
|
|
270
|
+
"""把一个已知块元素按默认继承样式投影,供版本化 HTML 解码复用。"""
|
|
271
|
+
return self._parse_block(element, TextStyle())
|
|
272
|
+
|
|
273
|
+
def project_inline_content(self, element: etree._Element) -> list[_InlineSpanDict]:
|
|
274
|
+
"""把一个已知行内容器恢复为结构化 Span。"""
|
|
275
|
+
resolved = self.stylesheet.resolve(element, TextStyle())
|
|
276
|
+
if resolved.subtree_hidden:
|
|
277
|
+
return []
|
|
278
|
+
content, extras = self._render_inline_children(element, resolved.text, resolved.visibility_hidden)
|
|
279
|
+
if extras:
|
|
280
|
+
raise ValueError("inline projection produced unexpected block content")
|
|
281
|
+
return strip_span_dicts(content)
|
|
282
|
+
|
|
283
|
+
def _parse_container_contents(
|
|
284
|
+
self,
|
|
285
|
+
element: etree._Element,
|
|
286
|
+
style: TextStyle,
|
|
287
|
+
visibility_hidden: bool = False,
|
|
288
|
+
) -> list[dict[str, object]]:
|
|
289
|
+
"""把连续行内内容和块级子元素按源顺序拆成 raw blocks。"""
|
|
290
|
+
blocks: list[dict[str, object]] = []
|
|
291
|
+
inline_parts: list[_InlineSpanDict] = []
|
|
292
|
+
if not visibility_hidden:
|
|
293
|
+
extend_inline_spans(inline_parts, self._render_text(element.text, style))
|
|
294
|
+
|
|
295
|
+
def flush_inline() -> None:
|
|
296
|
+
"""把当前连续行内片段写为普通正文 block。"""
|
|
297
|
+
content = strip_span_dicts(inline_parts)
|
|
298
|
+
inline_parts.clear()
|
|
299
|
+
if content:
|
|
300
|
+
blocks.append({"type": BlockType.TEXT, "content": content})
|
|
301
|
+
|
|
302
|
+
for child in element:
|
|
303
|
+
if not isinstance(child.tag, str):
|
|
304
|
+
if not visibility_hidden:
|
|
305
|
+
extend_inline_spans(inline_parts, self._render_text(entity_text(child), style))
|
|
306
|
+
extend_inline_spans(inline_parts, self._render_text(child.tail, style))
|
|
307
|
+
continue
|
|
308
|
+
name = local_name(child)
|
|
309
|
+
if name in BLOCK_TAGS:
|
|
310
|
+
flush_inline()
|
|
311
|
+
blocks.extend(self._parse_block(child, style, visibility_hidden))
|
|
312
|
+
else:
|
|
313
|
+
for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
|
|
314
|
+
if isinstance(segment, list):
|
|
315
|
+
extend_inline_spans(inline_parts, segment)
|
|
316
|
+
else:
|
|
317
|
+
flush_inline()
|
|
318
|
+
blocks.append(segment)
|
|
319
|
+
if not visibility_hidden:
|
|
320
|
+
extend_inline_spans(inline_parts, self._render_text(child.tail, style))
|
|
321
|
+
flush_inline()
|
|
322
|
+
return blocks
|
|
323
|
+
|
|
324
|
+
def _parse_block(
|
|
325
|
+
self,
|
|
326
|
+
element: etree._Element,
|
|
327
|
+
inherited: TextStyle,
|
|
328
|
+
inherited_visibility_hidden: bool = False,
|
|
329
|
+
) -> list[dict[str, object]]:
|
|
330
|
+
"""把一个块级元素分派到对应 raw block 转换逻辑。"""
|
|
331
|
+
resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
|
|
332
|
+
if resolved.subtree_hidden:
|
|
333
|
+
return []
|
|
334
|
+
name = local_name(element)
|
|
335
|
+
if name in SKIPPED_TAGS or name == "hr":
|
|
336
|
+
return []
|
|
337
|
+
if self.context.note_anchor(element) is not None:
|
|
338
|
+
return self._parse_note_element(element, resolved.text, resolved.visibility_hidden)
|
|
339
|
+
if name in {"h1", "h2", "h3", "h4", "h5", "h6", "p"}:
|
|
340
|
+
return self._parse_textual_block(element, name, resolved.text, resolved.visibility_hidden)
|
|
341
|
+
if name in {"ul", "ol"}:
|
|
342
|
+
list_block, extras = self._parse_list(element, resolved.text, resolved.visibility_hidden)
|
|
343
|
+
return ([list_block] if list_block is not None else []) + extras
|
|
344
|
+
if name == "table":
|
|
345
|
+
return self._parse_table(element, resolved.text, resolved.visibility_hidden)
|
|
346
|
+
if name == "pre":
|
|
347
|
+
content = self._visible_raw_text(element, resolved.text, resolved.visibility_hidden)
|
|
348
|
+
language = self._code_language_hint(element)
|
|
349
|
+
block: dict[str, object] = {"type": BlockType.CODE, "content": content}
|
|
350
|
+
if language:
|
|
351
|
+
block["guess_lang"] = language
|
|
352
|
+
return [block] if content.strip() else []
|
|
353
|
+
if name == "math":
|
|
354
|
+
if resolved.visibility_hidden:
|
|
355
|
+
return []
|
|
356
|
+
formula = self._formula_extraction(element)
|
|
357
|
+
if formula is not None:
|
|
358
|
+
return [{"type": BlockType.EQUATION, "content": formula.latex}]
|
|
359
|
+
fallback = self._visible_plain_text(element, resolved.text, resolved.visibility_hidden)
|
|
360
|
+
return [{"type": BlockType.TEXT, "content": text_spans(fallback)}] if fallback else []
|
|
361
|
+
if name == "figure":
|
|
362
|
+
return self._parse_figure(element, resolved.text, resolved.visibility_hidden)
|
|
363
|
+
if name in {"aside", "div", "section"} and self._has_contextual_visual_annotation(element):
|
|
364
|
+
return self._parse_figure(element, resolved.text, resolved.visibility_hidden)
|
|
365
|
+
if name == "svg":
|
|
366
|
+
return self._parse_svg(element, resolved.text, resolved.visibility_hidden)
|
|
367
|
+
return self._parse_container_contents(element, resolved.text, resolved.visibility_hidden)
|
|
368
|
+
|
|
369
|
+
def _parse_textual_block(
|
|
370
|
+
self,
|
|
371
|
+
element: etree._Element,
|
|
372
|
+
name: str,
|
|
373
|
+
style: TextStyle,
|
|
374
|
+
visibility_hidden: bool,
|
|
375
|
+
) -> list[dict[str, object]]:
|
|
376
|
+
"""转换标题或段落,并旁路其中的视觉 blocks。"""
|
|
377
|
+
blocks: list[dict[str, object]] = []
|
|
378
|
+
text_emitted = False
|
|
379
|
+
for segment in self._render_inline_children_ordered(element, style, visibility_hidden):
|
|
380
|
+
if not isinstance(segment, list):
|
|
381
|
+
blocks.append(segment)
|
|
382
|
+
continue
|
|
383
|
+
content = strip_span_dicts(segment)
|
|
384
|
+
if not content:
|
|
385
|
+
continue
|
|
386
|
+
if text_emitted:
|
|
387
|
+
blocks.append({"type": BlockType.TEXT, "content": content})
|
|
388
|
+
continue
|
|
389
|
+
if name == "h1" and (not self.single_document_title or not self.document_title_emitted):
|
|
390
|
+
block: dict[str, object] = {"type": BlockType.DOC_TITLE, "level": 1, "content": content}
|
|
391
|
+
self.document_title_emitted = True
|
|
392
|
+
elif name.startswith("h"):
|
|
393
|
+
level = min(max(int(name[1:]), 2), 6)
|
|
394
|
+
block = {
|
|
395
|
+
"type": BlockType.PARAGRAPH_TITLE,
|
|
396
|
+
"level": level,
|
|
397
|
+
"is_numbered_style": False,
|
|
398
|
+
"content": content,
|
|
399
|
+
}
|
|
400
|
+
else:
|
|
401
|
+
block = {"type": BlockType.TEXT, "content": content}
|
|
402
|
+
if name.startswith("h") and (anchor := self.context.heading_anchor(element)):
|
|
403
|
+
block["anchor"] = anchor
|
|
404
|
+
blocks.append(block)
|
|
405
|
+
text_emitted = True
|
|
406
|
+
return blocks
|
|
407
|
+
|
|
408
|
+
def _parse_note_element(
|
|
409
|
+
self,
|
|
410
|
+
element: etree._Element,
|
|
411
|
+
style: TextStyle,
|
|
412
|
+
visibility_hidden: bool = False,
|
|
413
|
+
) -> list[dict[str, object]]:
|
|
414
|
+
"""逐块转换单条脚注,并只给首个文本脚注挂载 anchor。"""
|
|
415
|
+
blocks = self._parse_container_contents(element, style, visibility_hidden)
|
|
416
|
+
anchor = self.context.note_anchor(element)
|
|
417
|
+
anchor_attached = False
|
|
418
|
+
for block in blocks:
|
|
419
|
+
content = block.get("content")
|
|
420
|
+
if block.get("type") != BlockType.TEXT or not isinstance(content, list) or not content:
|
|
421
|
+
continue
|
|
422
|
+
block["type"] = BlockType.PAGE_FOOTNOTE
|
|
423
|
+
if anchor is not None and not anchor_attached:
|
|
424
|
+
block["anchor"] = anchor
|
|
425
|
+
anchor_attached = True
|
|
426
|
+
return blocks
|
|
427
|
+
|
|
428
|
+
def _render_inline_children(
|
|
429
|
+
self,
|
|
430
|
+
element: etree._Element,
|
|
431
|
+
style: TextStyle,
|
|
432
|
+
visibility_hidden: bool = False,
|
|
433
|
+
) -> tuple[list[_InlineSpanDict], list[dict[str, object]]]:
|
|
434
|
+
"""渲染元素的连续行内 Span,并旁路其中的视觉 blocks。"""
|
|
435
|
+
segments = self._render_inline_children_ordered(element, style, visibility_hidden)
|
|
436
|
+
content: list[_InlineSpanDict] = []
|
|
437
|
+
for segment in segments:
|
|
438
|
+
if isinstance(segment, list):
|
|
439
|
+
extend_inline_spans(content, segment)
|
|
440
|
+
return (
|
|
441
|
+
content,
|
|
442
|
+
[segment for segment in segments if not isinstance(segment, list)],
|
|
443
|
+
)
|
|
444
|
+
|
|
445
|
+
def _render_inline_children_ordered(
|
|
446
|
+
self,
|
|
447
|
+
element: etree._Element,
|
|
448
|
+
style: TextStyle,
|
|
449
|
+
visibility_hidden: bool = False,
|
|
450
|
+
) -> list[_InlineProjectionSegment]:
|
|
451
|
+
"""按 DOM 顺序返回连续文本与旁路 block,保留 inline visual 前后边界。"""
|
|
452
|
+
segments: list[_InlineProjectionSegment] = []
|
|
453
|
+
if not visibility_hidden:
|
|
454
|
+
_append_inline_segment(segments, self._render_text(element.text, style))
|
|
455
|
+
for child in element:
|
|
456
|
+
if not isinstance(child.tag, str):
|
|
457
|
+
if not visibility_hidden:
|
|
458
|
+
_append_inline_segment(segments, self._render_text(entity_text(child), style))
|
|
459
|
+
_append_inline_segment(segments, self._render_text(child.tail, style))
|
|
460
|
+
continue
|
|
461
|
+
for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
|
|
462
|
+
_append_inline_segment(segments, segment)
|
|
463
|
+
if not visibility_hidden:
|
|
464
|
+
_append_inline_segment(segments, self._render_text(child.tail, style))
|
|
465
|
+
return segments
|
|
466
|
+
|
|
467
|
+
def _render_inline_element(
|
|
468
|
+
self,
|
|
469
|
+
element: etree._Element,
|
|
470
|
+
inherited: TextStyle,
|
|
471
|
+
inherited_visibility_hidden: bool = False,
|
|
472
|
+
) -> tuple[list[_InlineSpanDict], list[dict[str, object]]]:
|
|
473
|
+
"""把一个行内元素转换为结构化 Span 和可选视觉块。"""
|
|
474
|
+
segments = self._render_inline_element_ordered(element, inherited, inherited_visibility_hidden)
|
|
475
|
+
content: list[_InlineSpanDict] = []
|
|
476
|
+
for segment in segments:
|
|
477
|
+
if isinstance(segment, list):
|
|
478
|
+
extend_inline_spans(content, segment)
|
|
479
|
+
return (
|
|
480
|
+
content,
|
|
481
|
+
[segment for segment in segments if not isinstance(segment, list)],
|
|
482
|
+
)
|
|
483
|
+
|
|
484
|
+
def _render_inline_element_ordered(
|
|
485
|
+
self,
|
|
486
|
+
element: etree._Element,
|
|
487
|
+
inherited: TextStyle,
|
|
488
|
+
inherited_visibility_hidden: bool = False,
|
|
489
|
+
) -> list[_InlineProjectionSegment]:
|
|
490
|
+
"""递归投影单个行内元素,并在嵌套 visual 位置保留顺序分段。"""
|
|
491
|
+
resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
|
|
492
|
+
if resolved.subtree_hidden:
|
|
493
|
+
return []
|
|
494
|
+
name = local_name(element)
|
|
495
|
+
if name in SKIPPED_TAGS:
|
|
496
|
+
return []
|
|
497
|
+
if name == "br":
|
|
498
|
+
return [] if resolved.visibility_hidden else [text_spans("\n")]
|
|
499
|
+
if name in {"img", "image"}:
|
|
500
|
+
return [] if resolved.visibility_hidden else self._image_blocks(element)
|
|
501
|
+
if name == "math":
|
|
502
|
+
if resolved.visibility_hidden:
|
|
503
|
+
return []
|
|
504
|
+
formula = self._formula_extraction(element)
|
|
505
|
+
if formula is not None:
|
|
506
|
+
if formula.display == "block":
|
|
507
|
+
return [{"type": BlockType.EQUATION, "content": formula.latex}]
|
|
508
|
+
spans: list[_InlineSpanDict] = []
|
|
509
|
+
append_equation_span(spans, formula.latex)
|
|
510
|
+
return [spans]
|
|
511
|
+
fallback = self._visible_plain_text(element, resolved.text, resolved.visibility_hidden)
|
|
512
|
+
return [text_spans(fallback)] if fallback else []
|
|
513
|
+
if name == "code":
|
|
514
|
+
if resolved.visibility_hidden:
|
|
515
|
+
return []
|
|
516
|
+
code = self._visible_raw_text(element, resolved.text, resolved.visibility_hidden)
|
|
517
|
+
spans = []
|
|
518
|
+
append_code_span(spans, code)
|
|
519
|
+
return [spans] if spans else []
|
|
520
|
+
if name in BLOCK_TAGS:
|
|
521
|
+
return self._parse_block(element, inherited, inherited_visibility_hidden)
|
|
522
|
+
segments = self._render_inline_children_ordered(element, resolved.text, resolved.visibility_hidden)
|
|
523
|
+
if name == "a":
|
|
524
|
+
href = element.get("href") or element.get(_XLINK_HREF) or ""
|
|
525
|
+
target = self.context.resolve_link(href)
|
|
526
|
+
if target:
|
|
527
|
+
linked: list[_InlineProjectionSegment] = []
|
|
528
|
+
for segment in segments:
|
|
529
|
+
if not isinstance(segment, list) or not segment:
|
|
530
|
+
linked.append(segment)
|
|
531
|
+
continue
|
|
532
|
+
wrapped: list[_InlineSpanDict] = []
|
|
533
|
+
append_hyperlink_span(wrapped, segment, target)
|
|
534
|
+
linked.append(wrapped)
|
|
535
|
+
return linked
|
|
536
|
+
return segments
|
|
537
|
+
|
|
538
|
+
@staticmethod
|
|
539
|
+
def _render_text(value: str | None, style: TextStyle) -> list[_InlineSpanDict]:
|
|
540
|
+
"""折叠文本节点并直接投影为带样式 TextSpan。"""
|
|
541
|
+
text = clean_text_node(value)
|
|
542
|
+
if not text:
|
|
543
|
+
return []
|
|
544
|
+
return text_spans(text, style.names())
|
|
545
|
+
|
|
546
|
+
def _visible_raw_text(
|
|
547
|
+
self,
|
|
548
|
+
element: etree._Element,
|
|
549
|
+
style: TextStyle,
|
|
550
|
+
visibility_hidden: bool = False,
|
|
551
|
+
) -> str:
|
|
552
|
+
"""递归提取可见原始文本,并允许后代显式恢复 visibility。"""
|
|
553
|
+
return visible_raw_text_with_style(element, self.stylesheet, style, visibility_hidden)
|
|
554
|
+
|
|
555
|
+
def _visible_plain_text(
|
|
556
|
+
self,
|
|
557
|
+
element: etree._Element,
|
|
558
|
+
style: TextStyle,
|
|
559
|
+
visibility_hidden: bool = False,
|
|
560
|
+
) -> str:
|
|
561
|
+
"""返回折叠空白并还原实体后的可见纯文本。"""
|
|
562
|
+
value = self._visible_raw_text(element, style, visibility_hidden)
|
|
563
|
+
return _WHITESPACE_RE.sub(" ", html.unescape(value)).strip()
|
|
564
|
+
|
|
565
|
+
def _image_blocks(
|
|
566
|
+
self,
|
|
567
|
+
element: etree._Element,
|
|
568
|
+
*,
|
|
569
|
+
caption: str | None = None,
|
|
570
|
+
emit_alt_caption: bool = True,
|
|
571
|
+
) -> list[dict[str, object]]:
|
|
572
|
+
"""把可解析图片转换为 image block,并用 caption/alt 补说明。"""
|
|
573
|
+
source = element.get("src") or element.get("href") or element.get(_XLINK_HREF) or ""
|
|
574
|
+
requested_alt = (caption or element.get("alt") or element.get("title") or "").strip()
|
|
575
|
+
resolved = self.context.resolve_image(source, alt=requested_alt)
|
|
576
|
+
alt = (resolved.alt if resolved is not None else requested_alt).strip()
|
|
577
|
+
if resolved is None or not (resolved.image_base64 or resolved.image_url):
|
|
578
|
+
return [{"type": BlockType.TEXT, "content": text_spans(alt)}] if alt else []
|
|
579
|
+
block: dict[str, object] = {"type": BlockType.IMAGE, "content": ""}
|
|
580
|
+
if resolved.image_base64:
|
|
581
|
+
block["image_base64"] = resolved.image_base64
|
|
582
|
+
if resolved.image_url:
|
|
583
|
+
block["image_url"] = resolved.image_url
|
|
584
|
+
blocks: list[dict[str, object]] = [block]
|
|
585
|
+
annotation = (caption or (alt if emit_alt_caption else "")).strip()
|
|
586
|
+
if annotation:
|
|
587
|
+
blocks.append({"type": BlockType.IMAGE_CAPTION, "content": text_spans(annotation)})
|
|
588
|
+
return blocks
|
|
589
|
+
|
|
590
|
+
def _parse_figure(
|
|
591
|
+
self,
|
|
592
|
+
element: etree._Element,
|
|
593
|
+
style: TextStyle,
|
|
594
|
+
visibility_hidden: bool = False,
|
|
595
|
+
) -> list[dict[str, object]]:
|
|
596
|
+
"""按标准标签或完整 token 解析 visual 主体、caption 与 footnote。"""
|
|
597
|
+
annotations = [
|
|
598
|
+
(child, kind)
|
|
599
|
+
for child in element
|
|
600
|
+
if isinstance(child.tag, str) and (kind := self._visual_annotation_kind(child)) is not None
|
|
601
|
+
]
|
|
602
|
+
annotation_elements = {child for child, _ in annotations}
|
|
603
|
+
docvortex_figure = "docvortex-figure" in (element.get("class") or "").casefold().split()
|
|
604
|
+
blocks, visual_blocks_by_child = self._parse_figure_contents(
|
|
605
|
+
element,
|
|
606
|
+
style,
|
|
607
|
+
visibility_hidden,
|
|
608
|
+
annotation_elements=annotation_elements,
|
|
609
|
+
emit_alt_caption=not docvortex_figure and not annotations,
|
|
610
|
+
)
|
|
611
|
+
annotation_targets = self._figure_annotation_targets(
|
|
612
|
+
element,
|
|
613
|
+
annotation_elements,
|
|
614
|
+
visual_blocks_by_child,
|
|
615
|
+
)
|
|
616
|
+
|
|
617
|
+
annotations_by_visual: dict[int, list[dict[str, object]]] = {}
|
|
618
|
+
unbound_annotations: list[dict[str, object]] = []
|
|
619
|
+
for annotation, kind in annotations:
|
|
620
|
+
resolved = self.stylesheet.resolve(annotation, style, visibility_hidden)
|
|
621
|
+
if resolved.subtree_hidden:
|
|
622
|
+
continue
|
|
623
|
+
target = annotation_targets.get(annotation)
|
|
624
|
+
visual_type = _raw_visual_type(target.get("type")) if target is not None else None
|
|
625
|
+
annotation_type = VISUAL_TYPE_MAPPING[visual_type][kind] if visual_type is not None else BlockType.TEXT
|
|
626
|
+
annotation_blocks: list[dict[str, object]] = []
|
|
627
|
+
for segment in self._render_inline_children_ordered(annotation, resolved.text, resolved.visibility_hidden):
|
|
628
|
+
if isinstance(segment, list):
|
|
629
|
+
if content := strip_span_dicts(segment):
|
|
630
|
+
annotation_blocks.append({"type": annotation_type, "content": content})
|
|
631
|
+
continue
|
|
632
|
+
if visual_type is not None and segment.get("type") == BlockType.TEXT:
|
|
633
|
+
segment = {**segment, "type": annotation_type}
|
|
634
|
+
annotation_blocks.append(segment)
|
|
635
|
+
if target is not None and visual_type is not None:
|
|
636
|
+
annotations_by_visual.setdefault(id(target), []).extend(annotation_blocks)
|
|
637
|
+
else:
|
|
638
|
+
unbound_annotations.extend(annotation_blocks)
|
|
639
|
+
|
|
640
|
+
output: list[dict[str, object]] = []
|
|
641
|
+
for block in blocks:
|
|
642
|
+
output.append(block)
|
|
643
|
+
output.extend(annotations_by_visual.get(id(block), ()))
|
|
644
|
+
output.extend(unbound_annotations)
|
|
645
|
+
return output
|
|
646
|
+
|
|
647
|
+
def _parse_figure_contents(
|
|
648
|
+
self,
|
|
649
|
+
element: etree._Element,
|
|
650
|
+
style: TextStyle,
|
|
651
|
+
visibility_hidden: bool,
|
|
652
|
+
*,
|
|
653
|
+
annotation_elements: set[etree._Element],
|
|
654
|
+
emit_alt_caption: bool,
|
|
655
|
+
) -> tuple[list[dict[str, object]], dict[etree._Element, list[dict[str, object]]]]:
|
|
656
|
+
"""按 DOM 顺序缓冲 figure 文本,并在 visual extras 前后切分正文 block。"""
|
|
657
|
+
blocks: list[dict[str, object]] = []
|
|
658
|
+
visual_blocks_by_child: dict[etree._Element, list[dict[str, object]]] = {}
|
|
659
|
+
inline_parts: list[_InlineSpanDict] = []
|
|
660
|
+
if not visibility_hidden:
|
|
661
|
+
extend_inline_spans(inline_parts, self._render_text(element.text, style))
|
|
662
|
+
|
|
663
|
+
def flush_inline() -> None:
|
|
664
|
+
"""把 figure 当前连续文本写为普通正文 block。"""
|
|
665
|
+
content = strip_span_dicts(inline_parts)
|
|
666
|
+
inline_parts.clear()
|
|
667
|
+
if content:
|
|
668
|
+
blocks.append({"type": BlockType.TEXT, "content": content})
|
|
669
|
+
|
|
670
|
+
for child in element:
|
|
671
|
+
if not isinstance(child.tag, str):
|
|
672
|
+
if not visibility_hidden:
|
|
673
|
+
extend_inline_spans(inline_parts, self._render_text(entity_text(child), style))
|
|
674
|
+
extend_inline_spans(inline_parts, self._render_text(child.tail, style))
|
|
675
|
+
continue
|
|
676
|
+
if child in annotation_elements:
|
|
677
|
+
if not visibility_hidden:
|
|
678
|
+
extend_inline_spans(inline_parts, self._render_text(child.tail, style))
|
|
679
|
+
continue
|
|
680
|
+
|
|
681
|
+
first_child_block = len(blocks)
|
|
682
|
+
name = local_name(child)
|
|
683
|
+
if name in {"img", "image"}:
|
|
684
|
+
flush_inline()
|
|
685
|
+
child_style = self.stylesheet.resolve(child, style, visibility_hidden)
|
|
686
|
+
if not child_style.subtree_hidden and not child_style.visibility_hidden:
|
|
687
|
+
blocks.extend(self._image_blocks(child, emit_alt_caption=emit_alt_caption))
|
|
688
|
+
elif name in BLOCK_TAGS:
|
|
689
|
+
flush_inline()
|
|
690
|
+
blocks.extend(self._parse_block(child, style, visibility_hidden))
|
|
691
|
+
else:
|
|
692
|
+
for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
|
|
693
|
+
if isinstance(segment, list):
|
|
694
|
+
extend_inline_spans(inline_parts, segment)
|
|
695
|
+
else:
|
|
696
|
+
flush_inline()
|
|
697
|
+
blocks.append(segment)
|
|
698
|
+
if not visibility_hidden:
|
|
699
|
+
extend_inline_spans(inline_parts, self._render_text(child.tail, style))
|
|
700
|
+
child_visuals = [block for block in blocks[first_child_block:] if _raw_visual_type(block.get("type")) is not None]
|
|
701
|
+
if child_visuals:
|
|
702
|
+
visual_blocks_by_child[child] = child_visuals
|
|
703
|
+
flush_inline()
|
|
704
|
+
return blocks, visual_blocks_by_child
|
|
705
|
+
|
|
706
|
+
@staticmethod
|
|
707
|
+
def _figure_annotation_targets(
|
|
708
|
+
figure: etree._Element,
|
|
709
|
+
annotations: set[etree._Element],
|
|
710
|
+
visual_blocks_by_child: dict[etree._Element, list[dict[str, object]]],
|
|
711
|
+
) -> dict[etree._Element, dict[str, object] | None]:
|
|
712
|
+
"""用双向线性扫描绑定全部 annotation,优先最近前序 visual。"""
|
|
713
|
+
children = [child for child in figure if isinstance(child.tag, str)]
|
|
714
|
+
targets: dict[etree._Element, dict[str, object] | None] = {}
|
|
715
|
+
previous_visual: dict[str, object] | None = None
|
|
716
|
+
for child in children:
|
|
717
|
+
if visuals := visual_blocks_by_child.get(child):
|
|
718
|
+
previous_visual = visuals[-1]
|
|
719
|
+
if child in annotations:
|
|
720
|
+
targets[child] = previous_visual
|
|
721
|
+
|
|
722
|
+
next_visual: dict[str, object] | None = None
|
|
723
|
+
for child in reversed(children):
|
|
724
|
+
if visuals := visual_blocks_by_child.get(child):
|
|
725
|
+
next_visual = visuals[0]
|
|
726
|
+
if child in annotations and targets[child] is None:
|
|
727
|
+
targets[child] = next_visual
|
|
728
|
+
return targets
|
|
729
|
+
|
|
730
|
+
def _has_contextual_visual_annotation(self, element: etree._Element) -> bool:
|
|
731
|
+
"""仅在直属完整 token annotation 与 visual 后代并存时启用非标准容器解析。"""
|
|
732
|
+
children = [child for child in element if isinstance(child.tag, str)]
|
|
733
|
+
if not any(self._visual_annotation_kind(child) is not None for child in children):
|
|
734
|
+
return False
|
|
735
|
+
return any(
|
|
736
|
+
local_name(candidate) in _VISUAL_ELEMENT_TAGS
|
|
737
|
+
for child in children
|
|
738
|
+
if self._visual_annotation_kind(child) is None
|
|
739
|
+
for candidate in [child, *child.iterdescendants()]
|
|
740
|
+
if isinstance(candidate.tag, str)
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
@staticmethod
|
|
744
|
+
def _visual_annotation_kind(element: etree._Element) -> str | None:
|
|
745
|
+
"""用标准标签、role 或完整 class/id token 返回 caption/footnote 角色。"""
|
|
746
|
+
if local_name(element) == "figcaption":
|
|
747
|
+
return "caption"
|
|
748
|
+
tokens = _semantic_tokens(element)
|
|
749
|
+
roles = frozenset((element.get("role") or "").casefold().split())
|
|
750
|
+
if tokens & _CAPTION_TOKENS or roles & {"caption", "doc-subtitle"}:
|
|
751
|
+
return "caption"
|
|
752
|
+
if tokens & _FOOTNOTE_TOKENS or roles & {"doc-footnote", "note"}:
|
|
753
|
+
return "footnote"
|
|
754
|
+
return None
|
|
755
|
+
|
|
756
|
+
def _parse_svg(
|
|
757
|
+
self,
|
|
758
|
+
element: etree._Element,
|
|
759
|
+
style: TextStyle,
|
|
760
|
+
visibility_hidden: bool = False,
|
|
761
|
+
) -> list[dict[str, object]]:
|
|
762
|
+
"""从 SVG 尽力提取 title/desc/text 和静态 image。"""
|
|
763
|
+
blocks: list[dict[str, object]] = []
|
|
764
|
+
texts: list[str] = []
|
|
765
|
+
|
|
766
|
+
def visit(parent: etree._Element, inherited: TextStyle, inherited_visibility_hidden: bool) -> None:
|
|
767
|
+
"""按 SVG 树顺序访问候选节点,并允许可见后代恢复输出。"""
|
|
768
|
+
for child in parent:
|
|
769
|
+
if not isinstance(child.tag, str):
|
|
770
|
+
continue
|
|
771
|
+
resolved = self.stylesheet.resolve(child, inherited, inherited_visibility_hidden)
|
|
772
|
+
if resolved.subtree_hidden:
|
|
773
|
+
continue
|
|
774
|
+
name = local_name(child)
|
|
775
|
+
if name in {"title", "desc", "text"}:
|
|
776
|
+
value = self._visible_plain_text(child, resolved.text, resolved.visibility_hidden)
|
|
777
|
+
if value and value not in texts:
|
|
778
|
+
texts.append(value)
|
|
779
|
+
elif name == "image":
|
|
780
|
+
if not resolved.visibility_hidden:
|
|
781
|
+
blocks.extend(self._image_blocks(child))
|
|
782
|
+
else:
|
|
783
|
+
visit(child, resolved.text, resolved.visibility_hidden)
|
|
784
|
+
|
|
785
|
+
visit(element, style, visibility_hidden)
|
|
786
|
+
if texts:
|
|
787
|
+
blocks.insert(0, {"type": BlockType.TEXT, "content": text_spans("\n".join(texts))})
|
|
788
|
+
return blocks
|
|
789
|
+
|
|
790
|
+
def _parse_table(
|
|
791
|
+
self,
|
|
792
|
+
table: etree._Element,
|
|
793
|
+
style: TextStyle,
|
|
794
|
+
visibility_hidden: bool = False,
|
|
795
|
+
) -> list[dict[str, object]]:
|
|
796
|
+
"""重建白名单化 HTML 表格,并把 caption 投影为表格说明。"""
|
|
797
|
+
markup = self._serialize_table_node(table, style, visibility_hidden)
|
|
798
|
+
if not markup:
|
|
799
|
+
return []
|
|
800
|
+
blocks: list[dict[str, object]] = [{"type": BlockType.TABLE, "content": markup}]
|
|
801
|
+
caption_element = next(
|
|
802
|
+
(child for child in table if isinstance(child.tag, str) and local_name(child) == "caption"),
|
|
803
|
+
None,
|
|
804
|
+
)
|
|
805
|
+
if caption_element is not None:
|
|
806
|
+
caption_style = self.stylesheet.resolve(caption_element, style, visibility_hidden)
|
|
807
|
+
if not caption_style.subtree_hidden:
|
|
808
|
+
caption = self._visible_plain_text(caption_element, caption_style.text, caption_style.visibility_hidden)
|
|
809
|
+
if caption:
|
|
810
|
+
blocks.append({"type": BlockType.TABLE_CAPTION, "content": text_spans(caption)})
|
|
811
|
+
return blocks
|
|
812
|
+
|
|
813
|
+
def _serialize_table_node(
|
|
814
|
+
self,
|
|
815
|
+
element: etree._Element,
|
|
816
|
+
inherited: TextStyle,
|
|
817
|
+
inherited_visibility_hidden: bool = False,
|
|
818
|
+
*,
|
|
819
|
+
row_link_target: str | None = None,
|
|
820
|
+
) -> str:
|
|
821
|
+
"""递归序列化安全表格结构、行内样式、链接、公式和图片。"""
|
|
822
|
+
resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
|
|
823
|
+
if resolved.subtree_hidden:
|
|
824
|
+
return ""
|
|
825
|
+
name = local_name(element)
|
|
826
|
+
if name == "caption" or name in SKIPPED_TAGS:
|
|
827
|
+
return ""
|
|
828
|
+
allowed = {
|
|
829
|
+
"a",
|
|
830
|
+
"b",
|
|
831
|
+
"br",
|
|
832
|
+
"code",
|
|
833
|
+
"col",
|
|
834
|
+
"colgroup",
|
|
835
|
+
"em",
|
|
836
|
+
"i",
|
|
837
|
+
"img",
|
|
838
|
+
"math",
|
|
839
|
+
"p",
|
|
840
|
+
"s",
|
|
841
|
+
"span",
|
|
842
|
+
"strong",
|
|
843
|
+
"sub",
|
|
844
|
+
"sup",
|
|
845
|
+
"table",
|
|
846
|
+
"tbody",
|
|
847
|
+
"td",
|
|
848
|
+
"tfoot",
|
|
849
|
+
"th",
|
|
850
|
+
"thead",
|
|
851
|
+
"tr",
|
|
852
|
+
"u",
|
|
853
|
+
}
|
|
854
|
+
if name not in allowed:
|
|
855
|
+
return self._serialize_table_children(
|
|
856
|
+
element,
|
|
857
|
+
resolved.text,
|
|
858
|
+
resolved.visibility_hidden,
|
|
859
|
+
row_link_target=row_link_target,
|
|
860
|
+
)
|
|
861
|
+
if name == "br":
|
|
862
|
+
return "" if resolved.visibility_hidden else "<br>"
|
|
863
|
+
if name == "math":
|
|
864
|
+
if resolved.visibility_hidden:
|
|
865
|
+
return ""
|
|
866
|
+
formula = self._formula_extraction(element)
|
|
867
|
+
if formula is not None:
|
|
868
|
+
return f"<eq>{html.escape(formula.latex, quote=False)}</eq>"
|
|
869
|
+
fallback = self._visible_plain_text(element, resolved.text, resolved.visibility_hidden)
|
|
870
|
+
return html.escape(fallback, quote=False)
|
|
871
|
+
if name == "img":
|
|
872
|
+
if resolved.visibility_hidden:
|
|
873
|
+
return ""
|
|
874
|
+
source = element.get("src") or ""
|
|
875
|
+
alt_text = (element.get("alt") or "").strip()
|
|
876
|
+
image = self.context.resolve_image(source, alt=alt_text)
|
|
877
|
+
if image is None:
|
|
878
|
+
return html.escape(alt_text, quote=False)
|
|
879
|
+
image_source = image.image_base64 or image.image_url
|
|
880
|
+
alt = html.escape(image.alt or alt_text, quote=True)
|
|
881
|
+
return f'<img src="{html.escape(image_source, quote=True)}" alt="{alt}">' if image_source else alt
|
|
882
|
+
if name == "tr":
|
|
883
|
+
row_link_target = self._toc_table_row_target(element)
|
|
884
|
+
attributes: list[str] = []
|
|
885
|
+
if name in {"td", "th"}:
|
|
886
|
+
for attribute in ("colspan", "rowspan", "scope"):
|
|
887
|
+
value = (element.get(attribute) or "").strip()
|
|
888
|
+
if attribute == "scope" and value in {"col", "colgroup", "row", "rowgroup"}:
|
|
889
|
+
attributes.append(f'{attribute}="{value}"')
|
|
890
|
+
elif span := bounded_table_span(value):
|
|
891
|
+
attributes.append(f'{attribute}="{span}"')
|
|
892
|
+
elif name in {"col", "colgroup"}:
|
|
893
|
+
if span := bounded_table_span((element.get("span") or "").strip()):
|
|
894
|
+
attributes.append(f'span="{span}"')
|
|
895
|
+
if name == "a":
|
|
896
|
+
target = self.context.resolve_link(element.get("href") or "")
|
|
897
|
+
if target:
|
|
898
|
+
attributes.append(f'href="{html.escape(target, quote=True)}"')
|
|
899
|
+
inner = self._serialize_table_children(
|
|
900
|
+
element,
|
|
901
|
+
resolved.text,
|
|
902
|
+
resolved.visibility_hidden,
|
|
903
|
+
row_link_target=row_link_target,
|
|
904
|
+
)
|
|
905
|
+
if resolved.visibility_hidden and not inner:
|
|
906
|
+
return ""
|
|
907
|
+
if name in {"td", "th"} and row_link_target and self._table_cell_can_inherit_toc_link(element):
|
|
908
|
+
inner = f'<a href="{html.escape(row_link_target, quote=True)}">{inner}</a>'
|
|
909
|
+
attrs = f" {' '.join(attributes)}" if attributes else ""
|
|
910
|
+
return f"<{name}{attrs}>{inner}</{name}>"
|
|
911
|
+
|
|
912
|
+
def _serialize_table_children(
|
|
913
|
+
self,
|
|
914
|
+
element: etree._Element,
|
|
915
|
+
style: TextStyle,
|
|
916
|
+
visibility_hidden: bool = False,
|
|
917
|
+
*,
|
|
918
|
+
row_link_target: str | None = None,
|
|
919
|
+
) -> str:
|
|
920
|
+
"""序列化表格节点的文本、子元素和 tail。"""
|
|
921
|
+
parts = [] if visibility_hidden else [self._render_table_text(element.text, style)]
|
|
922
|
+
for child in element:
|
|
923
|
+
if isinstance(child.tag, str):
|
|
924
|
+
parts.append(self._serialize_table_node(child, style, visibility_hidden, row_link_target=row_link_target))
|
|
925
|
+
elif not visibility_hidden:
|
|
926
|
+
parts.append(self._render_table_text(entity_text(child), style))
|
|
927
|
+
if not visibility_hidden:
|
|
928
|
+
parts.append(self._render_table_text(child.tail, style))
|
|
929
|
+
return "".join(parts)
|
|
930
|
+
|
|
931
|
+
def _toc_table_row_target(self, row: etree._Element) -> str | None:
|
|
932
|
+
"""为严格匹配单一目标标题的目录表格行返回内部链接。"""
|
|
933
|
+
links = [
|
|
934
|
+
element
|
|
935
|
+
for element in row.iter()
|
|
936
|
+
if isinstance(element.tag, str) and local_name(element) == "a" and (element.get("href") or "").strip()
|
|
937
|
+
]
|
|
938
|
+
if not links:
|
|
939
|
+
return None
|
|
940
|
+
resolved_targets: list[str] = []
|
|
941
|
+
for link in links:
|
|
942
|
+
target = self.context.resolve_link(link.get("href") or "")
|
|
943
|
+
if target is None or not target.startswith("#"):
|
|
944
|
+
return None
|
|
945
|
+
resolved_targets.append(target)
|
|
946
|
+
if len(set(resolved_targets)) != 1:
|
|
947
|
+
return None
|
|
948
|
+
target = resolved_targets[0]
|
|
949
|
+
title = self.context.heading_label(target[1:])
|
|
950
|
+
if title is None:
|
|
951
|
+
return None
|
|
952
|
+
cells = [child for child in row if isinstance(child.tag, str) and local_name(child) in {"td", "th"}]
|
|
953
|
+
row_label = " ".join(value for cell in cells if (value := visible_text(cell)))
|
|
954
|
+
normalized_row = _WHITESPACE_RE.sub(" ", html.unescape(row_label)).strip().casefold()
|
|
955
|
+
normalized_title = _WHITESPACE_RE.sub(" ", html.unescape(title)).strip().casefold()
|
|
956
|
+
return target if normalized_row and normalized_row == normalized_title else None
|
|
957
|
+
|
|
958
|
+
@staticmethod
|
|
959
|
+
def _table_cell_can_inherit_toc_link(cell: etree._Element) -> bool:
|
|
960
|
+
"""只允许纯文本与行内样式单元格继承目录行的唯一内部链接。"""
|
|
961
|
+
if not visible_text(cell):
|
|
962
|
+
return False
|
|
963
|
+
allowed_inline = {"b", "br", "code", "em", "i", "s", "span", "strong", "sub", "sup", "u"}
|
|
964
|
+
return all(isinstance(child.tag, str) and local_name(child) in allowed_inline for child in cell.iterdescendants())
|
|
965
|
+
|
|
966
|
+
@staticmethod
|
|
967
|
+
def _render_table_text(value: str | None, style: TextStyle) -> str:
|
|
968
|
+
"""把表格文字转义后包装为 renderer 支持的安全 HTML 样式标签。"""
|
|
969
|
+
rendered = html.escape(clean_text_node(value), quote=False)
|
|
970
|
+
if not rendered:
|
|
971
|
+
return ""
|
|
972
|
+
for enabled, tag in (
|
|
973
|
+
(style.bold, "strong"),
|
|
974
|
+
(style.italic, "em"),
|
|
975
|
+
(style.underline, "u"),
|
|
976
|
+
(style.strikethrough, "s"),
|
|
977
|
+
(style.superscript, "sup"),
|
|
978
|
+
(style.subscript, "sub"),
|
|
979
|
+
):
|
|
980
|
+
if enabled:
|
|
981
|
+
rendered = f"<{tag}>{rendered}</{tag}>"
|
|
982
|
+
return rendered
|
|
983
|
+
|
|
984
|
+
def _parse_list(
|
|
985
|
+
self,
|
|
986
|
+
element: etree._Element,
|
|
987
|
+
style: TextStyle,
|
|
988
|
+
visibility_hidden: bool = False,
|
|
989
|
+
) -> tuple[dict[str, object] | None, list[dict[str, object]]]:
|
|
990
|
+
"""解析有序/无序列表,并投影为连续阿拉伯编号结构。"""
|
|
991
|
+
if self._list_contains_page_blocks(element):
|
|
992
|
+
return self._parse_list_with_page_blocks(element, style, visibility_hidden)
|
|
993
|
+
ordered = local_name(element) == "ol"
|
|
994
|
+
items = [child for child in element if isinstance(child.tag, str) and local_name(child) == "li"]
|
|
995
|
+
if not items:
|
|
996
|
+
return None, []
|
|
997
|
+
children: list[dict[str, object]] = []
|
|
998
|
+
extras: list[dict[str, object]] = []
|
|
999
|
+
for item in items:
|
|
1000
|
+
item_style = self.stylesheet.resolve(item, style, visibility_hidden)
|
|
1001
|
+
if item_style.subtree_hidden:
|
|
1002
|
+
continue
|
|
1003
|
+
if self.context.note_anchor(item) is not None:
|
|
1004
|
+
extras.extend(self._parse_note_element(item, item_style.text, item_style.visibility_hidden))
|
|
1005
|
+
continue
|
|
1006
|
+
content_parts: list[_InlineSpanDict] = []
|
|
1007
|
+
if not item_style.visibility_hidden:
|
|
1008
|
+
extend_inline_spans(content_parts, self._render_text(item.text, item_style.text))
|
|
1009
|
+
nested_lists: list[dict[str, object]] = []
|
|
1010
|
+
for child in item:
|
|
1011
|
+
if not isinstance(child.tag, str):
|
|
1012
|
+
if not item_style.visibility_hidden:
|
|
1013
|
+
extend_inline_spans(content_parts, self._render_text(entity_text(child), item_style.text))
|
|
1014
|
+
extend_inline_spans(content_parts, self._render_text(child.tail, item_style.text))
|
|
1015
|
+
continue
|
|
1016
|
+
name = local_name(child)
|
|
1017
|
+
if name in {"ul", "ol"}:
|
|
1018
|
+
nested_style = self.stylesheet.resolve(child, item_style.text, item_style.visibility_hidden)
|
|
1019
|
+
if not nested_style.subtree_hidden:
|
|
1020
|
+
nested, nested_extras = self._parse_list(child, nested_style.text, nested_style.visibility_hidden)
|
|
1021
|
+
if nested is not None:
|
|
1022
|
+
nested_lists.append(nested)
|
|
1023
|
+
extras.extend(nested_extras)
|
|
1024
|
+
elif name in {"table", "figure", "svg"}:
|
|
1025
|
+
extras.extend(self._parse_block(child, item_style.text, item_style.visibility_hidden))
|
|
1026
|
+
elif name in BLOCK_TAGS:
|
|
1027
|
+
child_style = self.stylesheet.resolve(child, item_style.text, item_style.visibility_hidden)
|
|
1028
|
+
if not child_style.subtree_hidden:
|
|
1029
|
+
if self.context.note_anchor(child) is not None:
|
|
1030
|
+
extras.extend(self._parse_note_element(child, child_style.text, child_style.visibility_hidden))
|
|
1031
|
+
else:
|
|
1032
|
+
rendered, child_extras = self._render_inline_children(
|
|
1033
|
+
child,
|
|
1034
|
+
child_style.text,
|
|
1035
|
+
child_style.visibility_hidden,
|
|
1036
|
+
)
|
|
1037
|
+
_append_list_block_content(content_parts, rendered)
|
|
1038
|
+
extras.extend(child_extras)
|
|
1039
|
+
else:
|
|
1040
|
+
rendered, child_extras = self._render_inline_element(child, item_style.text, item_style.visibility_hidden)
|
|
1041
|
+
extend_inline_spans(content_parts, rendered)
|
|
1042
|
+
extras.extend(child_extras)
|
|
1043
|
+
if not item_style.visibility_hidden:
|
|
1044
|
+
extend_inline_spans(content_parts, self._render_text(child.tail, item_style.text))
|
|
1045
|
+
content = strip_span_dicts(content_parts)
|
|
1046
|
+
if content:
|
|
1047
|
+
children.append({"type": BlockType.TEXT, "content": content})
|
|
1048
|
+
children.extend(nested_lists)
|
|
1049
|
+
if not children:
|
|
1050
|
+
return None, extras
|
|
1051
|
+
block: dict[str, object] = {
|
|
1052
|
+
"type": BlockType.LIST,
|
|
1053
|
+
"attribute": "ordered" if ordered else "unordered",
|
|
1054
|
+
"content": children,
|
|
1055
|
+
}
|
|
1056
|
+
if ordered:
|
|
1057
|
+
block["start"] = self._ordered_list_start(element)
|
|
1058
|
+
return block, extras
|
|
1059
|
+
|
|
1060
|
+
@staticmethod
|
|
1061
|
+
def _list_contains_page_blocks(element: etree._Element) -> bool:
|
|
1062
|
+
"""判断列表是否含可能提升为页面兄弟的 visual、code 或公式子树。"""
|
|
1063
|
+
return any(
|
|
1064
|
+
isinstance(candidate.tag, str) and local_name(candidate) in _LIST_PAGE_BLOCK_TAGS
|
|
1065
|
+
for candidate in element.iterdescendants()
|
|
1066
|
+
)
|
|
1067
|
+
|
|
1068
|
+
def _parse_list_with_page_blocks(
|
|
1069
|
+
self,
|
|
1070
|
+
element: etree._Element,
|
|
1071
|
+
style: TextStyle,
|
|
1072
|
+
visibility_hidden: bool,
|
|
1073
|
+
) -> tuple[dict[str, object] | None, list[dict[str, object]]]:
|
|
1074
|
+
"""把含 visual 的列表切成有序 list/text/page block 片段,保持 DOM 阅读顺序。"""
|
|
1075
|
+
ordered = local_name(element) == "ol"
|
|
1076
|
+
list_start = self._ordered_list_start(element) if ordered else 1
|
|
1077
|
+
items = [child for child in element if isinstance(child.tag, str) and local_name(child) == "li"]
|
|
1078
|
+
pending_children: list[dict[str, object]] = []
|
|
1079
|
+
pending_start = list_start
|
|
1080
|
+
output: list[dict[str, object]] = []
|
|
1081
|
+
visible_item_ordinal = 0
|
|
1082
|
+
has_page_blocks = False
|
|
1083
|
+
|
|
1084
|
+
def flush_pending() -> None:
|
|
1085
|
+
"""把当前连续列表项写为一个顶层 list block。"""
|
|
1086
|
+
nonlocal pending_children
|
|
1087
|
+
if not pending_children:
|
|
1088
|
+
return
|
|
1089
|
+
output.append(self._build_raw_list_block(pending_children, ordered=ordered, start=pending_start))
|
|
1090
|
+
pending_children = []
|
|
1091
|
+
|
|
1092
|
+
for item in items:
|
|
1093
|
+
item_style = self.stylesheet.resolve(item, style, visibility_hidden)
|
|
1094
|
+
if item_style.subtree_hidden:
|
|
1095
|
+
continue
|
|
1096
|
+
if self.context.note_anchor(item) is not None:
|
|
1097
|
+
flush_pending()
|
|
1098
|
+
output.extend(self._parse_note_element(item, item_style.text, item_style.visibility_hidden))
|
|
1099
|
+
has_page_blocks = True
|
|
1100
|
+
continue
|
|
1101
|
+
|
|
1102
|
+
segments = self._normalize_list_item_segments(
|
|
1103
|
+
self._render_inline_children_ordered(item, item_style.text, item_style.visibility_hidden)
|
|
1104
|
+
)
|
|
1105
|
+
page_positions = [
|
|
1106
|
+
index
|
|
1107
|
+
for index, segment in enumerate(segments)
|
|
1108
|
+
if not isinstance(segment, list) and segment.get("type") != BlockType.LIST
|
|
1109
|
+
]
|
|
1110
|
+
if not page_positions:
|
|
1111
|
+
item_children = self._list_item_children(segments)
|
|
1112
|
+
if item_children:
|
|
1113
|
+
if not pending_children:
|
|
1114
|
+
pending_start = list_start + visible_item_ordinal
|
|
1115
|
+
pending_children.extend(item_children)
|
|
1116
|
+
visible_item_ordinal += 1
|
|
1117
|
+
continue
|
|
1118
|
+
|
|
1119
|
+
first_page_position = page_positions[0]
|
|
1120
|
+
prefix_children = self._list_item_children(segments[:first_page_position])
|
|
1121
|
+
if not pending_children:
|
|
1122
|
+
pending_start = list_start + visible_item_ordinal
|
|
1123
|
+
pending_children.extend(prefix_children or [{"type": BlockType.TEXT, "content": []}])
|
|
1124
|
+
flush_pending()
|
|
1125
|
+
|
|
1126
|
+
for segment in segments[first_page_position:]:
|
|
1127
|
+
if isinstance(segment, list):
|
|
1128
|
+
content = strip_span_dicts(segment)
|
|
1129
|
+
if content:
|
|
1130
|
+
output.append({"type": BlockType.TEXT, "content": content})
|
|
1131
|
+
else:
|
|
1132
|
+
output.append(segment)
|
|
1133
|
+
visible_item_ordinal += 1
|
|
1134
|
+
has_page_blocks = True
|
|
1135
|
+
|
|
1136
|
+
if not has_page_blocks:
|
|
1137
|
+
return (
|
|
1138
|
+
self._build_raw_list_block(pending_children, ordered=ordered, start=list_start) if pending_children else None,
|
|
1139
|
+
[],
|
|
1140
|
+
)
|
|
1141
|
+
flush_pending()
|
|
1142
|
+
return None, output
|
|
1143
|
+
|
|
1144
|
+
@staticmethod
|
|
1145
|
+
def _normalize_list_item_segments(
|
|
1146
|
+
segments: list[_InlineProjectionSegment],
|
|
1147
|
+
) -> list[_InlineProjectionSegment]:
|
|
1148
|
+
"""把列表内部普通 text block 还原为文本片段,保留 visual/list 页面边界。"""
|
|
1149
|
+
normalized: list[_InlineProjectionSegment] = []
|
|
1150
|
+
for segment in segments:
|
|
1151
|
+
if isinstance(segment, dict) and segment.get("type") == BlockType.TEXT:
|
|
1152
|
+
content = segment.get("content")
|
|
1153
|
+
_append_inline_segment(normalized, content if isinstance(content, list) else [])
|
|
1154
|
+
else:
|
|
1155
|
+
_append_inline_segment(normalized, segment)
|
|
1156
|
+
return normalized
|
|
1157
|
+
|
|
1158
|
+
@staticmethod
|
|
1159
|
+
def _list_item_children(segments: list[_InlineProjectionSegment]) -> list[dict[str, object]]:
|
|
1160
|
+
"""把无页面 visual 的列表片段收敛为一个文本叶子及其嵌套列表。"""
|
|
1161
|
+
content: list[_InlineSpanDict] = []
|
|
1162
|
+
for segment in segments:
|
|
1163
|
+
if isinstance(segment, list):
|
|
1164
|
+
extend_inline_spans(content, segment)
|
|
1165
|
+
content = strip_span_dicts(content)
|
|
1166
|
+
children = [{"type": BlockType.TEXT, "content": content}] if content else []
|
|
1167
|
+
children.extend(segment for segment in segments if isinstance(segment, dict) and segment.get("type") == BlockType.LIST)
|
|
1168
|
+
return children
|
|
1169
|
+
|
|
1170
|
+
@staticmethod
|
|
1171
|
+
def _build_raw_list_block(
|
|
1172
|
+
children: list[dict[str, object]],
|
|
1173
|
+
*,
|
|
1174
|
+
ordered: bool,
|
|
1175
|
+
start: int,
|
|
1176
|
+
) -> dict[str, object]:
|
|
1177
|
+
"""构造一段可由既有无坐标后处理编号的 raw list block。"""
|
|
1178
|
+
block: dict[str, object] = {
|
|
1179
|
+
"type": BlockType.LIST,
|
|
1180
|
+
"attribute": "ordered" if ordered else "unordered",
|
|
1181
|
+
"content": children,
|
|
1182
|
+
}
|
|
1183
|
+
if ordered:
|
|
1184
|
+
block["start"] = start
|
|
1185
|
+
return block
|
|
1186
|
+
|
|
1187
|
+
@staticmethod
|
|
1188
|
+
def _ordered_list_start(element: etree._Element) -> int:
|
|
1189
|
+
"""读取有序列表唯一通用起始值,非法或负值统一回退为一。"""
|
|
1190
|
+
try:
|
|
1191
|
+
start = int(element.get("start") or 1)
|
|
1192
|
+
except ValueError:
|
|
1193
|
+
return 1
|
|
1194
|
+
return start if start >= 0 else 1
|
|
1195
|
+
|
|
1196
|
+
@staticmethod
|
|
1197
|
+
def _formula_extraction(element: etree._Element) -> FormulaExtraction | None:
|
|
1198
|
+
"""调用共享公式优先级,返回裸 LaTeX 及来源信息。"""
|
|
1199
|
+
return extract_formula(element)
|
|
1200
|
+
|
|
1201
|
+
@staticmethod
|
|
1202
|
+
def _code_language_hint(element: etree._Element) -> str | None:
|
|
1203
|
+
"""从 pre/code 的标准 class 或 data 属性提取安全语言提示。"""
|
|
1204
|
+
candidates = [element, *[child for child in element if isinstance(child.tag, str) and local_name(child) == "code"]]
|
|
1205
|
+
for candidate in candidates:
|
|
1206
|
+
for attribute in ("data-language", "data-lang"):
|
|
1207
|
+
value = (candidate.get(attribute) or "").strip()
|
|
1208
|
+
if re.fullmatch(r"[A-Za-z0-9_.+#-]+", value):
|
|
1209
|
+
return value
|
|
1210
|
+
for token in (candidate.get("class") or "").split():
|
|
1211
|
+
normalized = token.casefold()
|
|
1212
|
+
for prefix in ("language-", "lang-"):
|
|
1213
|
+
if normalized.startswith(prefix):
|
|
1214
|
+
value = token[len(prefix) :]
|
|
1215
|
+
if re.fullmatch(r"[A-Za-z0-9_.+#-]+", value):
|
|
1216
|
+
return value
|
|
1217
|
+
return None
|
|
1218
|
+
|
|
1219
|
+
|
|
1220
|
+
__all__ = [
|
|
1221
|
+
"BLOCK_TAGS",
|
|
1222
|
+
"MarkupContext",
|
|
1223
|
+
"MarkupProjector",
|
|
1224
|
+
"ResolvedMarkupImage",
|
|
1225
|
+
"SKIPPED_TAGS",
|
|
1226
|
+
"bounded_table_span",
|
|
1227
|
+
"clean_text_node",
|
|
1228
|
+
"entity_text",
|
|
1229
|
+
"local_name",
|
|
1230
|
+
"visible_raw_text_with_style",
|
|
1231
|
+
"visible_text",
|
|
1232
|
+
]
|
|
1233
|
+
|
|
1234
|
+
# 保持既有公开类型的 pickle 路径,所有旧、新入口指向同一个类。
|
|
1235
|
+
preserve_type_module(ResolvedMarkupImage, "docvortex.analyzers.native._shared.markup.projector")
|
|
1236
|
+
preserve_type_module(MarkupContext, "docvortex.analyzers.native._shared.markup.projector")
|
|
1237
|
+
preserve_type_module(MarkupProjector, "docvortex.analyzers.native._shared.markup.projector")
|