docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,763 @@
|
|
|
1
|
+
"""把 DocVortex HTML v1 固定 DOM 解析为无资源副作用的 typed plan。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from typing import cast
|
|
7
|
+
from urllib.parse import unquote
|
|
8
|
+
|
|
9
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
10
|
+
|
|
11
|
+
from ...schema import PAGE_BLOCK_TYPES, RAW_ALGORITHM, BlockType, VISUAL_TYPE_MAPPING
|
|
12
|
+
from docvortex.content.markup import extract_formula
|
|
13
|
+
from docvortex.content.markup.projector import BLOCK_TAGS, local_name
|
|
14
|
+
from .contracts import (
|
|
15
|
+
AnnotationWireSpec,
|
|
16
|
+
CodeBodyWireSpec,
|
|
17
|
+
EquationWireSpec,
|
|
18
|
+
FlowchartBodyWireSpec,
|
|
19
|
+
IndexBlockWireSpec,
|
|
20
|
+
IndexLeafWireSpec,
|
|
21
|
+
IndexWireSpec,
|
|
22
|
+
ListBlockWireSpec,
|
|
23
|
+
ListLeafWireSpec,
|
|
24
|
+
ListWireSpec,
|
|
25
|
+
DOCVORTEX_HTML_VERSION,
|
|
26
|
+
DocVortexHtmlWirePlan,
|
|
27
|
+
PageWireSpec,
|
|
28
|
+
RichVisualBodyWireSpec,
|
|
29
|
+
TableBodyWireSpec,
|
|
30
|
+
TextWireSpec,
|
|
31
|
+
VisualBodyWireSpec,
|
|
32
|
+
VisualWireSpec,
|
|
33
|
+
WireFallbackReason,
|
|
34
|
+
WireRenderMode,
|
|
35
|
+
WIRE_BLOCK_CLASS,
|
|
36
|
+
WIRE_DOCUMENT_CLASS,
|
|
37
|
+
WIRE_INDEX_CLASS,
|
|
38
|
+
WIRE_LIST_CONTENT_CLASS,
|
|
39
|
+
WIRE_LIST_MARKER_CLASS,
|
|
40
|
+
WIRE_PAGE_BREAK_CLASS,
|
|
41
|
+
WIRE_PAGE_CLASS,
|
|
42
|
+
WIRE_VISUAL_BODY_CLASS,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
_SIMPLE_TEXT_TYPES = frozenset(
|
|
47
|
+
{
|
|
48
|
+
BlockType.TEXT,
|
|
49
|
+
BlockType.REF_TEXT,
|
|
50
|
+
BlockType.DOC_TITLE,
|
|
51
|
+
BlockType.PARAGRAPH_TITLE,
|
|
52
|
+
BlockType.HEADER,
|
|
53
|
+
BlockType.FOOTER,
|
|
54
|
+
BlockType.PAGE_NUMBER,
|
|
55
|
+
BlockType.ASIDE_TEXT,
|
|
56
|
+
BlockType.PAGE_FOOTNOTE,
|
|
57
|
+
}
|
|
58
|
+
)
|
|
59
|
+
_PAGE_AUXILIARY_TYPES = frozenset({BlockType.HEADER, BlockType.FOOTER, BlockType.PAGE_NUMBER, BlockType.ASIDE_TEXT})
|
|
60
|
+
_LIST_LEAF_TYPES = frozenset({BlockType.TEXT, BlockType.REF_TEXT})
|
|
61
|
+
_INDEX_LEAF_TYPES = frozenset({BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE})
|
|
62
|
+
_OWNED_VISUAL_IMAGE_TOKENS = frozenset(
|
|
63
|
+
{"docvortex-chart-image", "docvortex-flowchart-fallback", "docvortex-image", "docvortex-table-image"}
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class NonCanonicalWire(ValueError):
|
|
68
|
+
"""表示当前 DOM 不是 renderer 能生成的 canonical v1 wire。"""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def parse_docvortex_html_wire(body: etree._Element) -> tuple[DocVortexHtmlWirePlan | None, WireFallbackReason | None]:
|
|
72
|
+
"""发现并解析 canonical v1 wire,非法结构只返回统一回退原因。"""
|
|
73
|
+
roots = [
|
|
74
|
+
element
|
|
75
|
+
for element in body.iter()
|
|
76
|
+
if isinstance(element.tag, str) and element.get("data-docvortex-html-version") is not None
|
|
77
|
+
]
|
|
78
|
+
if not roots:
|
|
79
|
+
return None, None
|
|
80
|
+
if len(roots) != 1:
|
|
81
|
+
return None, "non_canonical_wire"
|
|
82
|
+
root = roots[0]
|
|
83
|
+
if (root.get("data-docvortex-html-version") or "").strip() != DOCVORTEX_HTML_VERSION:
|
|
84
|
+
return None, "unsupported_version"
|
|
85
|
+
try:
|
|
86
|
+
_validate_wire_root_ownership(body, root)
|
|
87
|
+
return _parse_wire_root(root), None
|
|
88
|
+
except NonCanonicalWire:
|
|
89
|
+
return None, "non_canonical_wire"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _parse_wire_root(root: etree._Element) -> DocVortexHtmlWirePlan:
|
|
93
|
+
"""解析根、渲染模式、页面容器和全部顶层 block。"""
|
|
94
|
+
if local_name(root) != "article" or _class_tokens(root) != {
|
|
95
|
+
WIRE_DOCUMENT_CLASS,
|
|
96
|
+
f"{WIRE_DOCUMENT_CLASS}--{(root.get('data-render-mode') or '').strip()}",
|
|
97
|
+
}:
|
|
98
|
+
raise NonCanonicalWire
|
|
99
|
+
_validate_structural_text(root)
|
|
100
|
+
mode_value = (root.get("data-render-mode") or "").strip()
|
|
101
|
+
if mode_value not in {"default", "full"}:
|
|
102
|
+
raise NonCanonicalWire
|
|
103
|
+
mode = cast(WireRenderMode, mode_value)
|
|
104
|
+
wrappers: list[tuple[etree._Element, int | None]] = []
|
|
105
|
+
if mode == "default":
|
|
106
|
+
for child in _element_children(root):
|
|
107
|
+
if local_name(child) != "div" or _class_tokens(child) != {WIRE_BLOCK_CLASS}:
|
|
108
|
+
raise NonCanonicalWire
|
|
109
|
+
wrappers.append((child, None))
|
|
110
|
+
else:
|
|
111
|
+
for child in _element_children(root):
|
|
112
|
+
if local_name(child) == "hr" and _class_tokens(child) == {WIRE_PAGE_BREAK_CLASS}:
|
|
113
|
+
continue
|
|
114
|
+
if local_name(child) != "section" or _class_tokens(child) != {WIRE_PAGE_CLASS}:
|
|
115
|
+
raise NonCanonicalWire
|
|
116
|
+
_validate_structural_text(child)
|
|
117
|
+
page_idx = _non_negative_integer(child, "data-page-idx", required=True)
|
|
118
|
+
for wrapper in _element_children(child):
|
|
119
|
+
if local_name(wrapper) != "div" or _class_tokens(wrapper) != {WIRE_BLOCK_CLASS}:
|
|
120
|
+
raise NonCanonicalWire
|
|
121
|
+
wrappers.append((wrapper, page_idx))
|
|
122
|
+
nested_wrappers = [
|
|
123
|
+
element
|
|
124
|
+
for element in root.iterdescendants()
|
|
125
|
+
if isinstance(element.tag, str) and WIRE_BLOCK_CLASS in _class_tokens(element)
|
|
126
|
+
]
|
|
127
|
+
if len(nested_wrappers) != len(wrappers):
|
|
128
|
+
raise NonCanonicalWire
|
|
129
|
+
target_ids = _collect_anchor_target_ids(wrappers)
|
|
130
|
+
blocks = tuple(_parse_top_block(wrapper, section_page_idx, target_ids) for wrapper, section_page_idx in wrappers)
|
|
131
|
+
return DocVortexHtmlWirePlan(root, mode, blocks)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _validate_wire_root_ownership(body: etree._Element, root: etree._Element) -> None:
|
|
135
|
+
"""要求 canonical wire 根独占从自身到 body 的可见内容路径。"""
|
|
136
|
+
current = root
|
|
137
|
+
while current is not body:
|
|
138
|
+
parent = current.getparent()
|
|
139
|
+
if parent is None or (parent.text or "").strip():
|
|
140
|
+
raise NonCanonicalWire
|
|
141
|
+
for sibling in parent:
|
|
142
|
+
if sibling is current:
|
|
143
|
+
if (sibling.tail or "").strip():
|
|
144
|
+
raise NonCanonicalWire
|
|
145
|
+
continue
|
|
146
|
+
if isinstance(sibling.tag, str) or (sibling.tail or "").strip():
|
|
147
|
+
raise NonCanonicalWire
|
|
148
|
+
current = parent
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _parse_top_block(
|
|
152
|
+
wrapper: etree._Element,
|
|
153
|
+
section_page_idx: int | None,
|
|
154
|
+
target_ids: dict[str, frozenset[str]],
|
|
155
|
+
) -> PageWireSpec:
|
|
156
|
+
"""解析一个顶层 wrapper,并构造与 block 家族匹配的 typed spec。"""
|
|
157
|
+
_validate_structural_text(wrapper)
|
|
158
|
+
block_type = _page_block_type(wrapper)
|
|
159
|
+
page_idx = _non_negative_integer(wrapper, "data-page-idx", required=True)
|
|
160
|
+
if section_page_idx is not None and page_idx != section_page_idx:
|
|
161
|
+
raise NonCanonicalWire
|
|
162
|
+
block_index = _non_negative_integer(wrapper, "data-block-index", required=False)
|
|
163
|
+
_validate_top_metadata(wrapper, block_type)
|
|
164
|
+
roots = _element_children(wrapper)
|
|
165
|
+
if len(roots) != 1:
|
|
166
|
+
raise NonCanonicalWire
|
|
167
|
+
content_root = roots[0]
|
|
168
|
+
if block_type in _SIMPLE_TEXT_TYPES:
|
|
169
|
+
_validate_simple_content(content_root, block_type)
|
|
170
|
+
return TextWireSpec(wrapper, content_root, block_type, page_idx, block_index)
|
|
171
|
+
if block_type == BlockType.EQUATION:
|
|
172
|
+
_validate_equation_content(content_root)
|
|
173
|
+
return EquationWireSpec(wrapper, content_root, page_idx, block_index)
|
|
174
|
+
if block_type == BlockType.LIST:
|
|
175
|
+
root = _parse_list_container(content_root, top_wrapper=wrapper)
|
|
176
|
+
return ListBlockWireSpec(wrapper, page_idx, block_index, root)
|
|
177
|
+
if block_type == BlockType.INDEX:
|
|
178
|
+
root = _parse_index_root(content_root, wrapper, target_ids)
|
|
179
|
+
return IndexBlockWireSpec(wrapper, page_idx, block_index, root)
|
|
180
|
+
if block_type in VISUAL_TYPE_MAPPING:
|
|
181
|
+
return _parse_visual_content(wrapper, content_root, block_type, page_idx, block_index)
|
|
182
|
+
raise NonCanonicalWire
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _validate_top_metadata(wrapper: etree._Element, block_type: BlockType) -> None:
|
|
186
|
+
"""校验 subtype、语言、anchor 和 level 只出现在 renderer 定义的位置。"""
|
|
187
|
+
sub_type = (wrapper.get("data-block-sub-type") or "").strip()
|
|
188
|
+
guess_lang = (wrapper.get("data-guess-lang") or "").strip()
|
|
189
|
+
anchor = (wrapper.get("data-anchor") or "").strip()
|
|
190
|
+
level = _optional_integer(wrapper, "data-level")
|
|
191
|
+
if block_type == BlockType.CODE:
|
|
192
|
+
if sub_type not in {BlockType.CODE, RAW_ALGORITHM}:
|
|
193
|
+
raise NonCanonicalWire
|
|
194
|
+
if sub_type == BlockType.CODE and not guess_lang:
|
|
195
|
+
raise NonCanonicalWire
|
|
196
|
+
if sub_type == RAW_ALGORITHM and guess_lang:
|
|
197
|
+
raise NonCanonicalWire
|
|
198
|
+
elif block_type == BlockType.LIST:
|
|
199
|
+
if sub_type and sub_type not in _LIST_LEAF_TYPES:
|
|
200
|
+
raise NonCanonicalWire
|
|
201
|
+
if guess_lang:
|
|
202
|
+
raise NonCanonicalWire
|
|
203
|
+
elif block_type in {BlockType.IMAGE, BlockType.CHART}:
|
|
204
|
+
if guess_lang:
|
|
205
|
+
raise NonCanonicalWire
|
|
206
|
+
elif sub_type or guess_lang:
|
|
207
|
+
raise NonCanonicalWire
|
|
208
|
+
if block_type == BlockType.DOC_TITLE:
|
|
209
|
+
if level != 1:
|
|
210
|
+
raise NonCanonicalWire
|
|
211
|
+
elif block_type == BlockType.PARAGRAPH_TITLE:
|
|
212
|
+
if level is None or not 2 <= level <= 6:
|
|
213
|
+
raise NonCanonicalWire
|
|
214
|
+
elif level is not None:
|
|
215
|
+
raise NonCanonicalWire
|
|
216
|
+
if anchor and block_type not in {
|
|
217
|
+
BlockType.TEXT,
|
|
218
|
+
BlockType.DOC_TITLE,
|
|
219
|
+
BlockType.PARAGRAPH_TITLE,
|
|
220
|
+
BlockType.PAGE_FOOTNOTE,
|
|
221
|
+
}:
|
|
222
|
+
raise NonCanonicalWire
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _validate_simple_content(content_root: etree._Element, block_type: BlockType) -> None:
|
|
226
|
+
"""校验文本、标题、脚注和页面辅助 block 的固定行内容器。"""
|
|
227
|
+
expected_tags: dict[BlockType, frozenset[str]] = {
|
|
228
|
+
BlockType.TEXT: frozenset({"p"}),
|
|
229
|
+
BlockType.REF_TEXT: frozenset({"p"}),
|
|
230
|
+
BlockType.DOC_TITLE: frozenset({"h1"}),
|
|
231
|
+
BlockType.PARAGRAPH_TITLE: frozenset({"h2", "h3", "h4", "h5", "h6"}),
|
|
232
|
+
BlockType.PAGE_FOOTNOTE: frozenset({"div"}),
|
|
233
|
+
**{value: frozenset({"div"}) for value in _PAGE_AUXILIARY_TYPES},
|
|
234
|
+
}
|
|
235
|
+
if local_name(content_root) not in expected_tags[block_type]:
|
|
236
|
+
raise NonCanonicalWire
|
|
237
|
+
_validate_inline_region(content_root)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _validate_equation_content(content_root: etree._Element) -> None:
|
|
241
|
+
"""校验行间公式为 renderer math carrier 或公式图片。"""
|
|
242
|
+
if local_name(content_root) == "math":
|
|
243
|
+
_validate_formula_carrier(content_root, expected_display="block")
|
|
244
|
+
return
|
|
245
|
+
if local_name(content_root) != "img" or _class_tokens(content_root) != {"docvortex-equation-image"}:
|
|
246
|
+
raise NonCanonicalWire
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _validate_formula_carrier(element: etree._Element, *, expected_display: str) -> None:
|
|
250
|
+
"""校验公式 carrier 的类型、显示模式和非空 LaTeX。"""
|
|
251
|
+
if local_name(element) != "math" or (element.get("data-block-type") or "").strip() != BlockType.EQUATION:
|
|
252
|
+
raise NonCanonicalWire
|
|
253
|
+
if (element.get("data-formula-display") or "").strip() != expected_display:
|
|
254
|
+
raise NonCanonicalWire
|
|
255
|
+
formula = extract_formula(element)
|
|
256
|
+
if formula is None or not formula.latex:
|
|
257
|
+
raise NonCanonicalWire
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _parse_visual_content(
|
|
261
|
+
wrapper: etree._Element,
|
|
262
|
+
content_root: etree._Element,
|
|
263
|
+
parent_type: BlockType,
|
|
264
|
+
page_idx: int,
|
|
265
|
+
block_index: int | None,
|
|
266
|
+
) -> VisualWireSpec:
|
|
267
|
+
"""解析 figure 下唯一 body 与有序 annotation 子节点。"""
|
|
268
|
+
if local_name(content_root) != "figure":
|
|
269
|
+
raise NonCanonicalWire
|
|
270
|
+
_validate_structural_text(content_root)
|
|
271
|
+
mapping = VISUAL_TYPE_MAPPING[parent_type]
|
|
272
|
+
sub_type = (wrapper.get("data-block-sub-type") or "").strip()
|
|
273
|
+
body_type = BlockType.ALGORITHM_BODY if parent_type == BlockType.CODE and sub_type == RAW_ALGORITHM else mapping["body"]
|
|
274
|
+
allowed_types = frozenset({body_type, mapping["caption"], mapping["footnote"]})
|
|
275
|
+
children = _element_children(content_root)
|
|
276
|
+
child_types = [(child.get("data-block-type") or "").strip() for child in children]
|
|
277
|
+
if child_types.count(body_type) != 1 or any(value not in allowed_types for value in child_types):
|
|
278
|
+
raise NonCanonicalWire
|
|
279
|
+
guess_lang = (wrapper.get("data-guess-lang") or "").strip()
|
|
280
|
+
parsed_children: list[VisualBodyWireSpec | AnnotationWireSpec] = []
|
|
281
|
+
for child, child_type in zip(children, child_types, strict=True):
|
|
282
|
+
child_index = _non_negative_integer(child, "data-block-index", required=False)
|
|
283
|
+
if child_type == body_type:
|
|
284
|
+
if local_name(child) != "div" or (block_index is not None and child_index != block_index):
|
|
285
|
+
raise NonCanonicalWire
|
|
286
|
+
parsed_children.append(_parse_visual_body(child, parent_type, sub_type))
|
|
287
|
+
continue
|
|
288
|
+
if local_name(child) != "p":
|
|
289
|
+
raise NonCanonicalWire
|
|
290
|
+
_validate_inline_region(child)
|
|
291
|
+
parsed_children.append(AnnotationWireSpec(child, BlockType(child_type)))
|
|
292
|
+
return VisualWireSpec(
|
|
293
|
+
wrapper,
|
|
294
|
+
content_root,
|
|
295
|
+
parent_type,
|
|
296
|
+
page_idx,
|
|
297
|
+
block_index,
|
|
298
|
+
sub_type,
|
|
299
|
+
guess_lang,
|
|
300
|
+
tuple(parsed_children),
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _parse_visual_body(body: etree._Element, parent_type: BlockType, sub_type: str) -> VisualBodyWireSpec:
|
|
305
|
+
"""按父 visual 类型解析唯一 canonical body 载荷。"""
|
|
306
|
+
expected_class = f"{WIRE_VISUAL_BODY_CLASS}--{'image' if parent_type == BlockType.IMAGE else str(parent_type)}"
|
|
307
|
+
if _class_tokens(body) != {WIRE_VISUAL_BODY_CLASS, expected_class}:
|
|
308
|
+
raise NonCanonicalWire
|
|
309
|
+
if parent_type == BlockType.CODE:
|
|
310
|
+
return _parse_code_body(body, sub_type)
|
|
311
|
+
if parent_type == BlockType.TABLE:
|
|
312
|
+
return _parse_table_body(body)
|
|
313
|
+
if parent_type == BlockType.IMAGE and _looks_like_flowchart_body(body):
|
|
314
|
+
return _parse_flowchart_body(body)
|
|
315
|
+
return _parse_rich_visual_body(body, parent_type, sub_type)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _parse_code_body(body: etree._Element, sub_type: str) -> CodeBodyWireSpec:
|
|
319
|
+
"""解析普通代码或 algorithm 的固定内容载体。"""
|
|
320
|
+
_validate_structural_text(body)
|
|
321
|
+
children = _element_children(body)
|
|
322
|
+
if sub_type == BlockType.CODE:
|
|
323
|
+
if len(children) != 1 or local_name(children[0]) != "pre":
|
|
324
|
+
raise NonCanonicalWire
|
|
325
|
+
pre = children[0]
|
|
326
|
+
_validate_structural_text(pre)
|
|
327
|
+
code_children = _element_children(pre)
|
|
328
|
+
if len(code_children) != 1 or local_name(code_children[0]) != "code" or _element_children(code_children[0]):
|
|
329
|
+
raise NonCanonicalWire
|
|
330
|
+
return CodeBodyWireSpec(body, "code", code_children[0])
|
|
331
|
+
if sub_type != RAW_ALGORITHM:
|
|
332
|
+
raise NonCanonicalWire
|
|
333
|
+
if not children:
|
|
334
|
+
empty = etree.Element("div")
|
|
335
|
+
return CodeBodyWireSpec(body, "algorithm", empty)
|
|
336
|
+
if len(children) != 1 or local_name(children[0]) != "div" or _class_tokens(children[0]) != {"docvortex-algorithm"}:
|
|
337
|
+
raise NonCanonicalWire
|
|
338
|
+
_validate_inline_region(children[0])
|
|
339
|
+
return CodeBodyWireSpec(body, "algorithm", _clone_fragment(children[0]))
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _parse_table_body(body: etree._Element) -> TableBodyWireSpec:
|
|
343
|
+
"""解析结构表格、空间文本、图片或空 table body。"""
|
|
344
|
+
_validate_structural_text(body)
|
|
345
|
+
children = _element_children(body)
|
|
346
|
+
if not children:
|
|
347
|
+
return TableBodyWireSpec(body, "empty", None)
|
|
348
|
+
if len(children) != 1:
|
|
349
|
+
raise NonCanonicalWire
|
|
350
|
+
child = children[0]
|
|
351
|
+
name = local_name(child)
|
|
352
|
+
classes = _class_tokens(child)
|
|
353
|
+
if name == "table":
|
|
354
|
+
return TableBodyWireSpec(body, "html", child)
|
|
355
|
+
if name == "pre" and classes in ({"docvortex-table-text"}, {"docvortex-raw-fallback"}) and not _element_children(child):
|
|
356
|
+
return TableBodyWireSpec(body, "text", child)
|
|
357
|
+
if name == "img" and classes == {"docvortex-table-image"}:
|
|
358
|
+
return TableBodyWireSpec(body, "image", child)
|
|
359
|
+
raise NonCanonicalWire
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _looks_like_flowchart_body(body: etree._Element) -> bool:
|
|
363
|
+
"""判断 body 是否使用 renderer 的 flowchart 固定外壳。"""
|
|
364
|
+
children = _element_children(body)
|
|
365
|
+
return bool(children and local_name(children[0]) == "div" and "docvortex-flowchart" in _class_tokens(children[0]))
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def _parse_flowchart_body(body: etree._Element) -> FlowchartBodyWireSpec:
|
|
369
|
+
"""解析 flowchart canvas、可选 raster 和源码 details。"""
|
|
370
|
+
_validate_structural_text(body)
|
|
371
|
+
children = _element_children(body)
|
|
372
|
+
if len(children) != 2:
|
|
373
|
+
raise NonCanonicalWire
|
|
374
|
+
display, details = children
|
|
375
|
+
display_classes = _class_tokens(display)
|
|
376
|
+
if local_name(display) != "div" or "docvortex-flowchart" not in display_classes:
|
|
377
|
+
raise NonCanonicalWire
|
|
378
|
+
_validate_structural_text(display)
|
|
379
|
+
display_children = _element_children(display)
|
|
380
|
+
if not 1 <= len(display_children) <= 2:
|
|
381
|
+
raise NonCanonicalWire
|
|
382
|
+
canvas = display_children[0]
|
|
383
|
+
if local_name(canvas) != "div" or _class_tokens(canvas) != {"docvortex-flowchart-canvas"}:
|
|
384
|
+
raise NonCanonicalWire
|
|
385
|
+
_validate_structural_text(canvas)
|
|
386
|
+
if _element_children(canvas):
|
|
387
|
+
raise NonCanonicalWire
|
|
388
|
+
fallback_image = None
|
|
389
|
+
if len(display_children) == 2:
|
|
390
|
+
fallback_image = display_children[1]
|
|
391
|
+
if local_name(fallback_image) != "img" or _class_tokens(fallback_image) != {"docvortex-flowchart-fallback"}:
|
|
392
|
+
raise NonCanonicalWire
|
|
393
|
+
if local_name(details) != "details" or _class_tokens(details) != {"docvortex-details", "docvortex-flowchart-details"}:
|
|
394
|
+
raise NonCanonicalWire
|
|
395
|
+
_validate_structural_text(details)
|
|
396
|
+
details_children = _element_children(details)
|
|
397
|
+
if len(details_children) != 2:
|
|
398
|
+
raise NonCanonicalWire
|
|
399
|
+
summary, source = details_children
|
|
400
|
+
if (
|
|
401
|
+
local_name(summary) != "summary"
|
|
402
|
+
or _element_children(summary)
|
|
403
|
+
or " ".join(summary.itertext()).strip() != "flowchart source"
|
|
404
|
+
):
|
|
405
|
+
raise NonCanonicalWire
|
|
406
|
+
if local_name(source) != "pre" or _class_tokens(source) != {"docvortex-flowchart-source"}:
|
|
407
|
+
raise NonCanonicalWire
|
|
408
|
+
_validate_structural_text(source)
|
|
409
|
+
code_children = _element_children(source)
|
|
410
|
+
if len(code_children) != 1 or local_name(code_children[0]) != "code" or _element_children(code_children[0]):
|
|
411
|
+
raise NonCanonicalWire
|
|
412
|
+
return FlowchartBodyWireSpec(body, code_children[0], fallback_image)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _parse_rich_visual_body(body: etree._Element, parent_type: BlockType, sub_type: str) -> RichVisualBodyWireSpec:
|
|
416
|
+
"""区分 renderer-owned 主图与开放但受 sanitizer 约束的富内容 carrier。"""
|
|
417
|
+
allowed_token = "docvortex-image" if parent_type == BlockType.IMAGE else "docvortex-chart-image"
|
|
418
|
+
children = _element_children(body)
|
|
419
|
+
primary_image = (
|
|
420
|
+
children[0] if children and local_name(children[0]) == "img" and allowed_token in _class_tokens(children[0]) else None
|
|
421
|
+
)
|
|
422
|
+
if primary_image is not None and _class_tokens(primary_image) != {allowed_token}:
|
|
423
|
+
raise NonCanonicalWire
|
|
424
|
+
for element in body.iterdescendants():
|
|
425
|
+
if not isinstance(element.tag, str) or element is primary_image:
|
|
426
|
+
continue
|
|
427
|
+
if _class_tokens(element) & _OWNED_VISUAL_IMAGE_TOKENS:
|
|
428
|
+
raise NonCanonicalWire
|
|
429
|
+
if primary_image is None:
|
|
430
|
+
return RichVisualBodyWireSpec(body, parent_type, sub_type, None, _clone_fragment(body))
|
|
431
|
+
if (body.text or "").strip() or (primary_image.tail or "").strip():
|
|
432
|
+
raise NonCanonicalWire
|
|
433
|
+
remaining = children[1:]
|
|
434
|
+
if not remaining:
|
|
435
|
+
return RichVisualBodyWireSpec(body, parent_type, sub_type, primary_image, None)
|
|
436
|
+
if len(remaining) != 1:
|
|
437
|
+
raise NonCanonicalWire
|
|
438
|
+
details = remaining[0]
|
|
439
|
+
if local_name(details) != "details" or _class_tokens(details) != {"docvortex-details"}:
|
|
440
|
+
raise NonCanonicalWire
|
|
441
|
+
if (details.text or "").strip() or (details.tail or "").strip():
|
|
442
|
+
raise NonCanonicalWire
|
|
443
|
+
details_children = _element_children(details)
|
|
444
|
+
if not details_children or local_name(details_children[0]) != "summary" or _element_children(details_children[0]):
|
|
445
|
+
raise NonCanonicalWire
|
|
446
|
+
expected_summary = sub_type or ("image content" if parent_type == BlockType.IMAGE else "chart content")
|
|
447
|
+
if " ".join(details_children[0].itertext()).strip() != expected_summary:
|
|
448
|
+
raise NonCanonicalWire
|
|
449
|
+
fragment = _clone_fragment(details, after_child=details_children[0])
|
|
450
|
+
return RichVisualBodyWireSpec(body, parent_type, sub_type, primary_image, fragment)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _parse_list_container(container: etree._Element, *, top_wrapper: etree._Element | None = None) -> ListWireSpec:
|
|
454
|
+
"""递归解析 renderer 生成的列表 carrier、叶子和嵌套列表。"""
|
|
455
|
+
if local_name(container) not in {"ol", "ul"} or (container.get("data-block-type") or "").strip() != BlockType.LIST:
|
|
456
|
+
raise NonCanonicalWire
|
|
457
|
+
_validate_structural_text(container)
|
|
458
|
+
block_index = _non_negative_integer(container, "data-block-index", required=False)
|
|
459
|
+
sub_type = (container.get("data-block-sub-type") or "").strip()
|
|
460
|
+
if sub_type and sub_type not in _LIST_LEAF_TYPES:
|
|
461
|
+
raise NonCanonicalWire
|
|
462
|
+
if top_wrapper is not None and sub_type != (top_wrapper.get("data-block-sub-type") or "").strip():
|
|
463
|
+
raise NonCanonicalWire
|
|
464
|
+
classes = _class_tokens(container)
|
|
465
|
+
if "docvortex-list" not in classes or len(classes) != 2:
|
|
466
|
+
raise NonCanonicalWire
|
|
467
|
+
children: list[ListLeafWireSpec | ListWireSpec] = []
|
|
468
|
+
for item in _element_children(container):
|
|
469
|
+
if local_name(item) != "li":
|
|
470
|
+
raise NonCanonicalWire
|
|
471
|
+
item_type = (item.get("data-block-type") or "").strip()
|
|
472
|
+
nested_lists = [child for child in _element_children(item) if local_name(child) in {"ol", "ul"}]
|
|
473
|
+
if item_type:
|
|
474
|
+
if item_type not in _LIST_LEAF_TYPES:
|
|
475
|
+
raise NonCanonicalWire
|
|
476
|
+
leaf = _parse_list_leaf(item, BlockType(item_type), nested_lists)
|
|
477
|
+
children.append(leaf)
|
|
478
|
+
elif any(child not in nested_lists for child in _element_children(item)) or not nested_lists:
|
|
479
|
+
raise NonCanonicalWire
|
|
480
|
+
children.extend(_parse_list_container(nested) for nested in nested_lists)
|
|
481
|
+
start = _canonical_list_start(container)
|
|
482
|
+
return ListWireSpec(container, block_index, local_name(container) == "ol", start, sub_type, classes, tuple(children))
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def _parse_list_leaf(
|
|
486
|
+
item: etree._Element,
|
|
487
|
+
block_type: BlockType,
|
|
488
|
+
nested_lists: list[etree._Element],
|
|
489
|
+
) -> ListLeafWireSpec:
|
|
490
|
+
"""解析一个列表叶子的唯一 marker/content carrier。"""
|
|
491
|
+
block_index = _non_negative_integer(item, "data-block-index", required=False)
|
|
492
|
+
candidates = [child for child in _element_children(item) if child not in nested_lists]
|
|
493
|
+
content_carriers = [child for child in candidates if _class_tokens(child) == {WIRE_LIST_CONTENT_CLASS}]
|
|
494
|
+
marker_carriers = [child for child in candidates if _class_tokens(child) == {WIRE_LIST_MARKER_CLASS}]
|
|
495
|
+
if len(content_carriers) > 1 or len(marker_carriers) > 1:
|
|
496
|
+
raise NonCanonicalWire
|
|
497
|
+
allowed = [*content_carriers, *marker_carriers, *nested_lists]
|
|
498
|
+
if any(child not in allowed for child in _element_children(item)):
|
|
499
|
+
raise NonCanonicalWire
|
|
500
|
+
if content_carriers:
|
|
501
|
+
_validate_structural_text(item)
|
|
502
|
+
content_element = content_carriers[0]
|
|
503
|
+
if local_name(content_element) != "span":
|
|
504
|
+
raise NonCanonicalWire
|
|
505
|
+
_validate_inline_region(content_element)
|
|
506
|
+
else:
|
|
507
|
+
if marker_carriers or (item.text or "").strip() or any((child.tail or "").strip() for child in item):
|
|
508
|
+
raise NonCanonicalWire
|
|
509
|
+
content_element = None
|
|
510
|
+
marker = ""
|
|
511
|
+
if marker_carriers:
|
|
512
|
+
marker_element = marker_carriers[0]
|
|
513
|
+
if local_name(marker_element) != "span" or _element_children(marker_element):
|
|
514
|
+
raise NonCanonicalWire
|
|
515
|
+
marker = "".join(marker_element.itertext()).strip()
|
|
516
|
+
return ListLeafWireSpec(block_type, block_index, content_element, marker)
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _parse_index_root(
|
|
520
|
+
content_root: etree._Element,
|
|
521
|
+
wrapper: etree._Element,
|
|
522
|
+
target_ids: dict[str, frozenset[str]],
|
|
523
|
+
) -> IndexWireSpec:
|
|
524
|
+
"""解析目录根与唯一直属 ul。"""
|
|
525
|
+
if local_name(content_root) != "nav" or _class_tokens(content_root) != {WIRE_INDEX_CLASS}:
|
|
526
|
+
raise NonCanonicalWire
|
|
527
|
+
if (content_root.get("data-block-type") or "").strip() != BlockType.INDEX:
|
|
528
|
+
raise NonCanonicalWire
|
|
529
|
+
_validate_structural_text(content_root)
|
|
530
|
+
root_index = _non_negative_integer(content_root, "data-block-index", required=False)
|
|
531
|
+
wrapper_index = _non_negative_integer(wrapper, "data-block-index", required=False)
|
|
532
|
+
if wrapper_index is not None and root_index != wrapper_index:
|
|
533
|
+
raise NonCanonicalWire
|
|
534
|
+
lists = _element_children(content_root)
|
|
535
|
+
if len(lists) != 1 or local_name(lists[0]) != "ul":
|
|
536
|
+
raise NonCanonicalWire
|
|
537
|
+
return _parse_index_list(lists[0], nested=False, target_ids=target_ids, block_index=root_index)
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def _parse_index_list(
|
|
541
|
+
container: etree._Element,
|
|
542
|
+
*,
|
|
543
|
+
nested: bool,
|
|
544
|
+
target_ids: dict[str, frozenset[str]],
|
|
545
|
+
block_index: int | None = None,
|
|
546
|
+
) -> IndexWireSpec:
|
|
547
|
+
"""递归解析目录叶子、linked carrier 和嵌套 IndexBlock。"""
|
|
548
|
+
_validate_structural_text(container)
|
|
549
|
+
if nested:
|
|
550
|
+
if (container.get("data-block-type") or "").strip() != BlockType.INDEX:
|
|
551
|
+
raise NonCanonicalWire
|
|
552
|
+
block_index = _non_negative_integer(container, "data-block-index", required=False)
|
|
553
|
+
children: list[IndexLeafWireSpec | IndexWireSpec] = []
|
|
554
|
+
for item in _element_children(container):
|
|
555
|
+
if local_name(item) != "li":
|
|
556
|
+
raise NonCanonicalWire
|
|
557
|
+
item_type_value = (item.get("data-block-type") or "").strip()
|
|
558
|
+
nested_lists = [child for child in _element_children(item) if local_name(child) == "ul"]
|
|
559
|
+
if item_type_value:
|
|
560
|
+
try:
|
|
561
|
+
item_type = BlockType(item_type_value)
|
|
562
|
+
except ValueError as exc:
|
|
563
|
+
raise NonCanonicalWire from exc
|
|
564
|
+
if item_type not in _INDEX_LEAF_TYPES:
|
|
565
|
+
raise NonCanonicalWire
|
|
566
|
+
children.append(_parse_index_leaf(item, item_type, nested_lists, target_ids))
|
|
567
|
+
elif any(child not in nested_lists for child in _element_children(item)) or not nested_lists:
|
|
568
|
+
raise NonCanonicalWire
|
|
569
|
+
children.extend(_parse_index_list(nested_list, nested=True, target_ids=target_ids) for nested_list in nested_lists)
|
|
570
|
+
return IndexWireSpec(container, block_index, tuple(children))
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def _parse_index_leaf(
|
|
574
|
+
item: etree._Element,
|
|
575
|
+
item_type: BlockType,
|
|
576
|
+
nested_lists: list[etree._Element],
|
|
577
|
+
target_ids: dict[str, frozenset[str]],
|
|
578
|
+
) -> IndexLeafWireSpec:
|
|
579
|
+
"""解析 linked/unlinked 目录叶子并封闭 anchor 外结构。"""
|
|
580
|
+
block_index = _non_negative_integer(item, "data-block-index", required=False)
|
|
581
|
+
anchor = (item.get("data-anchor") or "").strip()
|
|
582
|
+
level = _optional_integer(item, "data-level")
|
|
583
|
+
_validate_index_leaf_metadata(item_type, anchor, level)
|
|
584
|
+
direct_content = [child for child in _element_children(item) if child not in nested_lists]
|
|
585
|
+
linked = [
|
|
586
|
+
child
|
|
587
|
+
for child in direct_content
|
|
588
|
+
if _is_canonical_index_link(child, item_type=item_type, anchor=anchor, target_ids=target_ids)
|
|
589
|
+
]
|
|
590
|
+
if linked:
|
|
591
|
+
if len(linked) != 1 or len(direct_content) != 1:
|
|
592
|
+
raise NonCanonicalWire
|
|
593
|
+
_validate_structural_text(item)
|
|
594
|
+
content_element = linked[0]
|
|
595
|
+
_validate_inline_region(content_element)
|
|
596
|
+
else:
|
|
597
|
+
content_element = _clone_fragment(item, excluded_children=nested_lists)
|
|
598
|
+
_validate_inline_region(content_element)
|
|
599
|
+
return IndexLeafWireSpec(item_type, block_index, content_element, anchor, level)
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def _collect_anchor_target_ids(
|
|
603
|
+
wrappers: list[tuple[etree._Element, int | None]],
|
|
604
|
+
) -> dict[str, frozenset[str]]:
|
|
605
|
+
"""预收集 renderer 正文和标题 id,供目录 linked carrier 做确定性判定。"""
|
|
606
|
+
collected: dict[str, set[str]] = {}
|
|
607
|
+
for wrapper, _ in wrappers:
|
|
608
|
+
block_type = (wrapper.get("data-block-type") or "").strip()
|
|
609
|
+
if block_type not in {BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE}:
|
|
610
|
+
continue
|
|
611
|
+
anchor = (wrapper.get("data-anchor") or "").strip()
|
|
612
|
+
children = _element_children(wrapper)
|
|
613
|
+
if not anchor or len(children) != 1:
|
|
614
|
+
continue
|
|
615
|
+
identities = {
|
|
616
|
+
identity
|
|
617
|
+
for element in [children[0], *children[0].iterdescendants()]
|
|
618
|
+
if (identity := (element.get("id") or "").strip())
|
|
619
|
+
}
|
|
620
|
+
if identities:
|
|
621
|
+
collected.setdefault(anchor, set()).update(identities)
|
|
622
|
+
return {anchor: frozenset(identities) for anchor, identities in collected.items()}
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _is_canonical_index_link(
|
|
626
|
+
element: etree._Element,
|
|
627
|
+
*,
|
|
628
|
+
item_type: BlockType,
|
|
629
|
+
anchor: str,
|
|
630
|
+
target_ids: dict[str, frozenset[str]],
|
|
631
|
+
) -> bool:
|
|
632
|
+
"""判断直属 anchor 是否为 renderer 生成的目录目标外壳。"""
|
|
633
|
+
if item_type not in {BlockType.TEXT, BlockType.DOC_TITLE, BlockType.PARAGRAPH_TITLE} or local_name(element) != "a":
|
|
634
|
+
return False
|
|
635
|
+
href = (element.get("href") or "").strip()
|
|
636
|
+
if not href.startswith("#"):
|
|
637
|
+
return False
|
|
638
|
+
return unquote(href[1:]).strip() in target_ids.get(anchor, frozenset())
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def _validate_index_leaf_metadata(item_type: BlockType, anchor: str, level: int | None) -> None:
|
|
642
|
+
"""校验目录叶子的 anchor 与标题 level 组合。"""
|
|
643
|
+
if item_type == BlockType.TEXT:
|
|
644
|
+
if level is not None:
|
|
645
|
+
raise NonCanonicalWire
|
|
646
|
+
return
|
|
647
|
+
if not anchor:
|
|
648
|
+
raise NonCanonicalWire
|
|
649
|
+
if item_type == BlockType.DOC_TITLE and level != 1:
|
|
650
|
+
raise NonCanonicalWire
|
|
651
|
+
if item_type == BlockType.PARAGRAPH_TITLE and (level is None or not 2 <= level <= 6):
|
|
652
|
+
raise NonCanonicalWire
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def _validate_inline_region(element: etree._Element) -> None:
|
|
656
|
+
"""校验 canonical 行内区域只含行内节点和可信公式 carrier。"""
|
|
657
|
+
for candidate in element.iterdescendants():
|
|
658
|
+
if not isinstance(candidate.tag, str):
|
|
659
|
+
continue
|
|
660
|
+
name = local_name(candidate)
|
|
661
|
+
marker_type = (candidate.get("data-block-type") or "").strip()
|
|
662
|
+
if name == "math" and marker_type == BlockType.EQUATION:
|
|
663
|
+
_validate_formula_carrier(candidate, expected_display="inline")
|
|
664
|
+
continue
|
|
665
|
+
if name in BLOCK_TAGS or name in {"image", "img"}:
|
|
666
|
+
raise NonCanonicalWire
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _clone_fragment(
|
|
670
|
+
element: etree._Element,
|
|
671
|
+
*,
|
|
672
|
+
after_child: etree._Element | None = None,
|
|
673
|
+
excluded_children: list[etree._Element] | None = None,
|
|
674
|
+
) -> etree._Element:
|
|
675
|
+
"""复制一个不含 renderer 外壳的富内容片段供 materializer 使用。"""
|
|
676
|
+
fragment = etree.Element("div")
|
|
677
|
+
excluded = excluded_children or []
|
|
678
|
+
children = _element_children(element)
|
|
679
|
+
start = 0
|
|
680
|
+
if after_child is None:
|
|
681
|
+
fragment.text = element.text
|
|
682
|
+
else:
|
|
683
|
+
try:
|
|
684
|
+
start = children.index(after_child) + 1
|
|
685
|
+
except ValueError as exc:
|
|
686
|
+
raise NonCanonicalWire from exc
|
|
687
|
+
fragment.text = after_child.tail
|
|
688
|
+
for child in children[start:]:
|
|
689
|
+
if child in excluded:
|
|
690
|
+
continue
|
|
691
|
+
fragment.append(deepcopy(child))
|
|
692
|
+
return fragment
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def _validate_structural_text(element: etree._Element) -> None:
|
|
696
|
+
"""拒绝 renderer 结构容器直属的非空文本和 tail。"""
|
|
697
|
+
if (element.text or "").strip() or any((child.tail or "").strip() for child in element):
|
|
698
|
+
raise NonCanonicalWire
|
|
699
|
+
|
|
700
|
+
|
|
701
|
+
def _page_block_type(wrapper: etree._Element) -> BlockType:
|
|
702
|
+
"""把顶层 data-block-type 转换为公开 PageBlock 类型。"""
|
|
703
|
+
try:
|
|
704
|
+
block_type = BlockType((wrapper.get("data-block-type") or "").strip())
|
|
705
|
+
except ValueError as exc:
|
|
706
|
+
raise NonCanonicalWire from exc
|
|
707
|
+
if block_type not in PAGE_BLOCK_TYPES:
|
|
708
|
+
raise NonCanonicalWire
|
|
709
|
+
return block_type
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def _element_children(element: etree._Element) -> list[etree._Element]:
|
|
713
|
+
"""返回元素直属的真实标签子节点。"""
|
|
714
|
+
return [child for child in element if isinstance(child.tag, str)]
|
|
715
|
+
|
|
716
|
+
|
|
717
|
+
def _class_tokens(element: etree._Element) -> frozenset[str]:
|
|
718
|
+
"""按 HTML class 空白边界返回完整小写 token。"""
|
|
719
|
+
return frozenset((element.get("class") or "").casefold().split())
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
def _non_negative_integer(element: etree._Element, name: str, *, required: bool) -> int | None:
|
|
723
|
+
"""读取 canonical 非负整数 data 属性。"""
|
|
724
|
+
value = element.get(name)
|
|
725
|
+
if value is None:
|
|
726
|
+
if required:
|
|
727
|
+
raise NonCanonicalWire
|
|
728
|
+
return None
|
|
729
|
+
try:
|
|
730
|
+
parsed = int(value)
|
|
731
|
+
except (TypeError, ValueError) as exc:
|
|
732
|
+
raise NonCanonicalWire from exc
|
|
733
|
+
if parsed < 0 or str(parsed) != value.strip():
|
|
734
|
+
raise NonCanonicalWire
|
|
735
|
+
return parsed
|
|
736
|
+
|
|
737
|
+
|
|
738
|
+
def _optional_integer(element: etree._Element, name: str) -> int | None:
|
|
739
|
+
"""读取可选整数属性并拒绝非法文本。"""
|
|
740
|
+
value = element.get(name)
|
|
741
|
+
if value is None:
|
|
742
|
+
return None
|
|
743
|
+
try:
|
|
744
|
+
return int(value)
|
|
745
|
+
except (TypeError, ValueError) as exc:
|
|
746
|
+
raise NonCanonicalWire from exc
|
|
747
|
+
|
|
748
|
+
|
|
749
|
+
def _canonical_list_start(element: etree._Element) -> int:
|
|
750
|
+
"""读取 renderer 生成的合法有序列表起始值。"""
|
|
751
|
+
value = element.get("start")
|
|
752
|
+
if value is None:
|
|
753
|
+
return 1
|
|
754
|
+
try:
|
|
755
|
+
parsed = int(value)
|
|
756
|
+
except ValueError as exc:
|
|
757
|
+
raise NonCanonicalWire from exc
|
|
758
|
+
if parsed < 0 or str(parsed) != value.strip():
|
|
759
|
+
raise NonCanonicalWire
|
|
760
|
+
return parsed
|
|
761
|
+
|
|
762
|
+
|
|
763
|
+
__all__ = ["NonCanonicalWire", "parse_docvortex_html_wire"]
|