docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""HTML wire 物化依赖的资源协议,不依赖具体输入解析器。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Protocol
|
|
6
|
+
from lxml import etree
|
|
7
|
+
|
|
8
|
+
from ...content.markup import MarkupContext
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class WireAnchorResolver(Protocol):
|
|
12
|
+
"""描述精确 HTML 文档提供的锚点解析能力。"""
|
|
13
|
+
|
|
14
|
+
def resolve_fragment(self, fragment: str) -> str | None:
|
|
15
|
+
"""将源 fragment 解析为规范内部链接。"""
|
|
16
|
+
|
|
17
|
+
def heading_anchor(self, heading: etree._Element) -> str | None:
|
|
18
|
+
"""返回标题的规范锚点。"""
|
|
19
|
+
|
|
20
|
+
def heading_label(self, anchor: str) -> str | None:
|
|
21
|
+
"""返回标题锚点对应的文字。"""
|
|
22
|
+
|
|
23
|
+
def note_anchor(self, note: etree._Element) -> str | None:
|
|
24
|
+
"""返回页面脚注的规范锚点。"""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class WireResourceContext(MarkupContext, Protocol):
|
|
28
|
+
"""在共享 markup 能力上增加绑定精确 wire 锚点的操作。"""
|
|
29
|
+
|
|
30
|
+
def bind_anchors(self, anchors: WireAnchorResolver) -> None:
|
|
31
|
+
"""绑定经过整棵 wire 验证的锚点解析器。"""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
__all__ = ["WireAnchorResolver", "WireResourceContext"]
|
docvortex/codecs/json.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""共享文档协议的显式读取入口。"""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
from ..schema import MiddleJson, ModelJson
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def load_model(value: dict[str, Any]) -> ModelJson:
|
|
8
|
+
"""读取新版分析文档,不执行解析、补写来源或历史迁移。"""
|
|
9
|
+
return ModelJson.from_dict(value)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def load_middle(value: dict[str, Any]) -> MiddleJson:
|
|
13
|
+
"""读取新版语义文档,不访问源文件和外部素材。"""
|
|
14
|
+
return MiddleJson.from_dict(value)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
__all__ = ["load_model", "load_middle"]
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Middle JSON 2.0 行内 Span 的规范化、可见文本与段落边界操作。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from copy import deepcopy
|
|
7
|
+
from typing import Callable, Iterable
|
|
8
|
+
|
|
9
|
+
from ..schema import CodeInlineSpan, EquationInlineSpan, HyperlinkSpan, InlineSpan, TextSpan, parse_inline_spans
|
|
10
|
+
from ..foundation.language import detect_lang
|
|
11
|
+
from ..foundation.text import CJK_LANGS, resolve_text_line_boundary
|
|
12
|
+
|
|
13
|
+
_CJK_RE = re.compile(r"[\u3040-\u30ff\u3400-\u9fff\uac00-\ud7af]")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def normalize_inline_spans(spans: Iterable[InlineSpan | dict[str, object]]) -> list[InlineSpan]:
|
|
17
|
+
"""严格解析、递归规范化并合并相邻等样式文字 Span。"""
|
|
18
|
+
parsed = parse_inline_spans(list(spans))
|
|
19
|
+
normalized: list[InlineSpan] = []
|
|
20
|
+
for span in parsed:
|
|
21
|
+
current: InlineSpan
|
|
22
|
+
if isinstance(span, HyperlinkSpan):
|
|
23
|
+
children = normalize_inline_spans(span.content)
|
|
24
|
+
non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
|
|
25
|
+
if not non_link_children:
|
|
26
|
+
continue
|
|
27
|
+
current = span.model_copy(update={"content": non_link_children})
|
|
28
|
+
else:
|
|
29
|
+
current = span.model_copy(deep=True)
|
|
30
|
+
if (
|
|
31
|
+
normalized
|
|
32
|
+
and isinstance(normalized[-1], TextSpan)
|
|
33
|
+
and isinstance(current, TextSpan)
|
|
34
|
+
and normalized[-1].styles == current.styles
|
|
35
|
+
):
|
|
36
|
+
normalized[-1].content += current.content
|
|
37
|
+
continue
|
|
38
|
+
if (
|
|
39
|
+
normalized
|
|
40
|
+
and isinstance(normalized[-1], HyperlinkSpan)
|
|
41
|
+
and isinstance(current, HyperlinkSpan)
|
|
42
|
+
and normalized[-1].url == current.url
|
|
43
|
+
):
|
|
44
|
+
merged_children = normalize_inline_spans([*normalized[-1].content, *current.content])
|
|
45
|
+
normalized[-1].content = [child for child in merged_children if not isinstance(child, HyperlinkSpan)]
|
|
46
|
+
continue
|
|
47
|
+
normalized.append(current)
|
|
48
|
+
return normalized
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def inline_plain_text(spans: Iterable[InlineSpan]) -> str:
|
|
52
|
+
"""提取 Span 列表的完整可见文字,供排序、合并和标题判断使用。"""
|
|
53
|
+
parts: list[str] = []
|
|
54
|
+
for span in spans:
|
|
55
|
+
if isinstance(span, (TextSpan, CodeInlineSpan, EquationInlineSpan)):
|
|
56
|
+
parts.append(span.content)
|
|
57
|
+
elif isinstance(span, HyperlinkSpan):
|
|
58
|
+
parts.append(inline_plain_text(span.content))
|
|
59
|
+
return "".join(parts)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def join_inline_spans(contents: Iterable[Iterable[InlineSpan]]) -> list[InlineSpan]:
|
|
63
|
+
"""按物理段落边界规则合并多组 Span,并保持结构化语义。"""
|
|
64
|
+
merged: list[InlineSpan] = []
|
|
65
|
+
for content in contents:
|
|
66
|
+
current = normalize_inline_spans(list(content))
|
|
67
|
+
if not current:
|
|
68
|
+
continue
|
|
69
|
+
# 边界裁剪最多清空最后一个根节点;保留其前驱以重新合并新相邻的 Span。
|
|
70
|
+
boundary_start = max(0, len(merged) - 2)
|
|
71
|
+
if merged:
|
|
72
|
+
_join_inline_span_sequences(merged, current)
|
|
73
|
+
merged.extend(current)
|
|
74
|
+
merged[boundary_start:] = normalize_inline_spans(_drop_empty_text_spans(merged[boundary_start:]))
|
|
75
|
+
return merged
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def strip_inline_spans(spans: Iterable[InlineSpan]) -> list[InlineSpan]:
|
|
79
|
+
"""删除行内内容首尾空白,同时保留内部 Span 边界和样式。"""
|
|
80
|
+
normalized = normalize_inline_spans(deepcopy(list(spans)))
|
|
81
|
+
first = _first_text_span(normalized)
|
|
82
|
+
last = _last_text_span(normalized)
|
|
83
|
+
if first is not None:
|
|
84
|
+
object.__setattr__(first, "content", first.content.lstrip())
|
|
85
|
+
if last is not None:
|
|
86
|
+
object.__setattr__(last, "content", last.content.rstrip())
|
|
87
|
+
return _drop_empty_text_spans(normalized)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def replace_inline_text(spans: Iterable[InlineSpan], content: str) -> list[InlineSpan]:
|
|
91
|
+
"""把纯文本回填为单一 TextSpan,供确实丢弃原样式的规则使用。"""
|
|
92
|
+
if not content:
|
|
93
|
+
return []
|
|
94
|
+
return [TextSpan(type="text", content=content)]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def slice_inline_spans(spans: Iterable[InlineSpan], start: int = 0, end: int | None = None) -> list[InlineSpan]:
|
|
98
|
+
"""按可见字符偏移裁剪 Span,并保留覆盖范围内的样式和链接。"""
|
|
99
|
+
normalized = normalize_inline_spans(deepcopy(list(spans)))
|
|
100
|
+
visible_length = len(inline_plain_text(normalized))
|
|
101
|
+
resolved_start = min(max(start, 0), visible_length)
|
|
102
|
+
resolved_end = visible_length if end is None else min(max(end, resolved_start), visible_length)
|
|
103
|
+
output: list[InlineSpan] = []
|
|
104
|
+
cursor = 0
|
|
105
|
+
for span in normalized:
|
|
106
|
+
span_length = len(inline_plain_text([span]))
|
|
107
|
+
span_end = cursor + span_length
|
|
108
|
+
overlap_start = max(resolved_start, cursor)
|
|
109
|
+
overlap_end = min(resolved_end, span_end)
|
|
110
|
+
if overlap_start < overlap_end:
|
|
111
|
+
local_start = overlap_start - cursor
|
|
112
|
+
local_end = overlap_end - cursor
|
|
113
|
+
sliced = _slice_inline_span(span, local_start, local_end)
|
|
114
|
+
if sliced is not None:
|
|
115
|
+
output.append(sliced)
|
|
116
|
+
cursor = span_end
|
|
117
|
+
if cursor >= resolved_end:
|
|
118
|
+
break
|
|
119
|
+
return normalize_inline_spans(output)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def map_text_span_content(spans: Iterable[InlineSpan], transform: Callable[[str], str]) -> list[InlineSpan]:
|
|
123
|
+
"""递归转换 TextSpan 正文,同时保留其它 Span 语义。"""
|
|
124
|
+
output: list[InlineSpan] = []
|
|
125
|
+
for span in normalize_inline_spans(deepcopy(list(spans))):
|
|
126
|
+
if isinstance(span, TextSpan):
|
|
127
|
+
content = transform(span.content)
|
|
128
|
+
if content:
|
|
129
|
+
output.append(span.model_copy(update={"content": content}))
|
|
130
|
+
elif isinstance(span, HyperlinkSpan):
|
|
131
|
+
children = map_text_span_content(span.content, transform)
|
|
132
|
+
non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
|
|
133
|
+
if non_link_children:
|
|
134
|
+
output.append(span.model_copy(update={"content": non_link_children}))
|
|
135
|
+
else:
|
|
136
|
+
output.append(span)
|
|
137
|
+
return normalize_inline_spans(output)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _join_inline_span_sequences(previous: list[InlineSpan], current: list[InlineSpan]) -> None:
|
|
141
|
+
"""在两组 Span 之间应用语言相关的换行拼接规则。"""
|
|
142
|
+
previous_visible = inline_plain_text(previous).rstrip()
|
|
143
|
+
current_visible = inline_plain_text(current).lstrip()
|
|
144
|
+
if not previous_visible or not current_visible:
|
|
145
|
+
return
|
|
146
|
+
|
|
147
|
+
last_text = _last_text_span(previous)
|
|
148
|
+
first_text = _first_text_span(current)
|
|
149
|
+
if last_text is not None:
|
|
150
|
+
object.__setattr__(last_text, "content", last_text.content.rstrip())
|
|
151
|
+
if first_text is not None:
|
|
152
|
+
object.__setattr__(first_text, "content", first_text.content.lstrip())
|
|
153
|
+
|
|
154
|
+
language = _detect_boundary_language(f"{previous_visible}{current_visible}")
|
|
155
|
+
if last_text is not None:
|
|
156
|
+
processed, separator = resolve_text_line_boundary(
|
|
157
|
+
last_text.content,
|
|
158
|
+
block_language=language,
|
|
159
|
+
next_content=current_visible,
|
|
160
|
+
)
|
|
161
|
+
object.__setattr__(last_text, "content", processed)
|
|
162
|
+
else:
|
|
163
|
+
separator = "" if language in CJK_LANGS else " "
|
|
164
|
+
if separator:
|
|
165
|
+
previous.append(TextSpan(type="text", content=separator))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _detect_boundary_language(content: str) -> str:
|
|
169
|
+
"""检测段落边界语言,短 CJK 文本优先使用字符范围兜底。"""
|
|
170
|
+
if _CJK_RE.search(content):
|
|
171
|
+
return "zh"
|
|
172
|
+
try:
|
|
173
|
+
return detect_lang(content)
|
|
174
|
+
except Exception:
|
|
175
|
+
return ""
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _first_text_span(spans: list[InlineSpan]) -> TextSpan | None:
|
|
179
|
+
"""返回首个可见叶子为文字时对应的 TextSpan。"""
|
|
180
|
+
for span in spans:
|
|
181
|
+
if not inline_plain_text([span]):
|
|
182
|
+
continue
|
|
183
|
+
if isinstance(span, TextSpan):
|
|
184
|
+
return span
|
|
185
|
+
if isinstance(span, HyperlinkSpan):
|
|
186
|
+
return _first_text_span(list(span.content))
|
|
187
|
+
return None
|
|
188
|
+
return None
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _last_text_span(spans: list[InlineSpan]) -> TextSpan | None:
|
|
192
|
+
"""返回末个可见叶子为文字时对应的 TextSpan。"""
|
|
193
|
+
for span in reversed(spans):
|
|
194
|
+
if not inline_plain_text([span]):
|
|
195
|
+
continue
|
|
196
|
+
if isinstance(span, TextSpan):
|
|
197
|
+
return span
|
|
198
|
+
if isinstance(span, HyperlinkSpan):
|
|
199
|
+
return _last_text_span(list(span.content))
|
|
200
|
+
return None
|
|
201
|
+
return None
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _drop_empty_text_spans(spans: list[InlineSpan]) -> list[InlineSpan]:
|
|
205
|
+
"""删除裁剪后为空的 TextSpan,并递归清理空链接。"""
|
|
206
|
+
result: list[InlineSpan] = []
|
|
207
|
+
for span in spans:
|
|
208
|
+
if isinstance(span, TextSpan) and not span.content:
|
|
209
|
+
continue
|
|
210
|
+
if isinstance(span, HyperlinkSpan):
|
|
211
|
+
children = _drop_empty_text_spans(list(span.content))
|
|
212
|
+
non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
|
|
213
|
+
if not non_link_children:
|
|
214
|
+
continue
|
|
215
|
+
span.content = non_link_children
|
|
216
|
+
result.append(span)
|
|
217
|
+
return result
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _slice_inline_span(span: InlineSpan, start: int, end: int) -> InlineSpan | None:
|
|
221
|
+
"""裁剪单个 Span 的局部可见区间。"""
|
|
222
|
+
if start >= end:
|
|
223
|
+
return None
|
|
224
|
+
if isinstance(span, TextSpan):
|
|
225
|
+
content = span.content[start:end]
|
|
226
|
+
return span.model_copy(update={"content": content}) if content else None
|
|
227
|
+
if isinstance(span, EquationInlineSpan):
|
|
228
|
+
content = span.content[start:end]
|
|
229
|
+
return span.model_copy(update={"content": content}) if content.strip() else None
|
|
230
|
+
if isinstance(span, CodeInlineSpan):
|
|
231
|
+
content = span.content[start:end]
|
|
232
|
+
return span.model_copy(update={"content": content}) if content else None
|
|
233
|
+
if isinstance(span, HyperlinkSpan):
|
|
234
|
+
children = slice_inline_spans(span.content, start, end)
|
|
235
|
+
non_link_children = [child for child in children if not isinstance(child, HyperlinkSpan)]
|
|
236
|
+
return span.model_copy(update={"content": non_link_children}) if non_link_children else None
|
|
237
|
+
return None
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
__all__ = [
|
|
241
|
+
"inline_plain_text",
|
|
242
|
+
"join_inline_spans",
|
|
243
|
+
"map_text_span_content",
|
|
244
|
+
"normalize_inline_spans",
|
|
245
|
+
"replace_inline_text",
|
|
246
|
+
"slice_inline_spans",
|
|
247
|
+
"strip_inline_spans",
|
|
248
|
+
]
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Flash EPUB 与 HTML 共用的静态标记文档投影能力。"""
|
|
2
|
+
|
|
3
|
+
from docvortex.content.markup.anchors import (
|
|
4
|
+
AnchorTextNormalization,
|
|
5
|
+
AnchorVisibilityScope,
|
|
6
|
+
MarkupAnchorDocument,
|
|
7
|
+
MarkupAnchorPolicy,
|
|
8
|
+
MarkupAnchorRegistry,
|
|
9
|
+
canonical_anchor,
|
|
10
|
+
element_id,
|
|
11
|
+
visible_element_text,
|
|
12
|
+
)
|
|
13
|
+
from docvortex.content.markup.formula import (
|
|
14
|
+
FormulaDisplay,
|
|
15
|
+
FormulaExtraction,
|
|
16
|
+
FormulaSourceKind,
|
|
17
|
+
extract_formula,
|
|
18
|
+
strip_formula_delimiters,
|
|
19
|
+
)
|
|
20
|
+
from docvortex.content.markup.projector import MarkupContext, MarkupProjector, ResolvedMarkupImage
|
|
21
|
+
from docvortex.content.markup.styles import ElementStyle, MarkupStylesheet, TextStyle, TextStyleDelta
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"AnchorTextNormalization",
|
|
25
|
+
"AnchorVisibilityScope",
|
|
26
|
+
"ElementStyle",
|
|
27
|
+
"FormulaDisplay",
|
|
28
|
+
"FormulaExtraction",
|
|
29
|
+
"FormulaSourceKind",
|
|
30
|
+
"MarkupAnchorDocument",
|
|
31
|
+
"MarkupAnchorPolicy",
|
|
32
|
+
"MarkupAnchorRegistry",
|
|
33
|
+
"MarkupContext",
|
|
34
|
+
"MarkupProjector",
|
|
35
|
+
"MarkupStylesheet",
|
|
36
|
+
"ResolvedMarkupImage",
|
|
37
|
+
"TextStyle",
|
|
38
|
+
"TextStyleDelta",
|
|
39
|
+
"canonical_anchor",
|
|
40
|
+
"element_id",
|
|
41
|
+
"extract_formula",
|
|
42
|
+
"strip_formula_delimiters",
|
|
43
|
+
"visible_element_text",
|
|
44
|
+
]
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""集中建立静态 HTML/XHTML 标题、脚注与 fragment anchor 索引。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import hashlib
|
|
7
|
+
import html
|
|
8
|
+
import re
|
|
9
|
+
from typing import Literal, Protocol, TypeAlias
|
|
10
|
+
|
|
11
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
12
|
+
|
|
13
|
+
from docvortex.content.markup.projector import local_name, visible_raw_text_with_style
|
|
14
|
+
from docvortex.content.markup.styles import MarkupStylesheet, TextStyle
|
|
15
|
+
from docvortex.foundation.type_identity import preserve_type_module
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
AnchorVisibilityScope: TypeAlias = Literal["all_ancestors", "nearest_body"]
|
|
19
|
+
AnchorTextNormalization: TypeAlias = Literal["unicode_whitespace", "xhtml_whitespace"]
|
|
20
|
+
|
|
21
|
+
_HEADING_TAGS = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"})
|
|
22
|
+
_XHTML_WHITESPACE_RE = re.compile(r"[\t\r\n\f ]+")
|
|
23
|
+
_XML_ID = "{http://www.w3.org/XML/1998/namespace}id"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class MarkupAnchorDocument:
|
|
28
|
+
"""描述一个待建立 anchor 索引的 DOM、样式表与兼容可见性规则。"""
|
|
29
|
+
|
|
30
|
+
key: str
|
|
31
|
+
root: etree._Element
|
|
32
|
+
stylesheet: MarkupStylesheet
|
|
33
|
+
visibility_scope: AnchorVisibilityScope = "all_ancestors"
|
|
34
|
+
text_normalization: AnchorTextNormalization = "unicode_whitespace"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class MarkupAnchorPolicy(Protocol):
|
|
38
|
+
"""定义格式适配器生成标题、脚注 anchor 所需的稳定策略。"""
|
|
39
|
+
|
|
40
|
+
anchor_prefix: str
|
|
41
|
+
register_document_start: bool
|
|
42
|
+
|
|
43
|
+
def heading_identity(self, element: etree._Element, ordinal: int) -> str:
|
|
44
|
+
"""返回当前标题参与稳定摘要的格式专属 identity。"""
|
|
45
|
+
|
|
46
|
+
def is_materializable_note(self, element: etree._Element, document: MarkupAnchorDocument) -> bool:
|
|
47
|
+
"""判断当前元素是否是能够兑现文本 anchor 的格式专属脚注。"""
|
|
48
|
+
|
|
49
|
+
def note_identity(self, element: etree._Element, ordinal: int) -> str:
|
|
50
|
+
"""返回当前脚注参与稳定摘要的格式专属 identity。"""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def element_id(element: etree._Element) -> str | None:
|
|
54
|
+
"""返回元素去除首尾空白后的 HTML id 或 xml:id。"""
|
|
55
|
+
value = (element.get("id") or element.get(_XML_ID) or "").strip()
|
|
56
|
+
return value or None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def visible_element_text(element: etree._Element, document: MarkupAnchorDocument) -> str:
|
|
60
|
+
"""按文档兼容配置解析祖先样式链,并返回最终可输出的纯文本。"""
|
|
61
|
+
inherited = TextStyle()
|
|
62
|
+
visibility_hidden = False
|
|
63
|
+
chain = [ancestor for ancestor in reversed(list(element.iterancestors())) if isinstance(ancestor.tag, str)]
|
|
64
|
+
if document.visibility_scope == "nearest_body":
|
|
65
|
+
body_index = next((index for index, ancestor in enumerate(chain) if local_name(ancestor) == "body"), None)
|
|
66
|
+
if body_index is not None:
|
|
67
|
+
chain = chain[body_index:]
|
|
68
|
+
chain.append(element)
|
|
69
|
+
for current in chain:
|
|
70
|
+
resolved = document.stylesheet.resolve(current, inherited, visibility_hidden)
|
|
71
|
+
if resolved.subtree_hidden:
|
|
72
|
+
return ""
|
|
73
|
+
inherited = resolved.text
|
|
74
|
+
visibility_hidden = resolved.visibility_hidden
|
|
75
|
+
value = visible_raw_text_with_style(element, document.stylesheet, inherited, visibility_hidden)
|
|
76
|
+
if document.text_normalization == "xhtml_whitespace":
|
|
77
|
+
return _XHTML_WHITESPACE_RE.sub(" ", html.unescape(value)).strip()
|
|
78
|
+
return " ".join(value.split())
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def canonical_anchor(prefix: str, document_key: str, identity: str) -> str:
|
|
82
|
+
"""按格式前缀、文档 key 与 identity 生成稳定的二十位摘要 anchor。"""
|
|
83
|
+
digest = hashlib.sha256(f"{document_key}#{identity}".encode()).hexdigest()[:20]
|
|
84
|
+
return f"{prefix}-{digest}"
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class MarkupAnchorRegistry:
|
|
88
|
+
"""统一登记多文档标题、脚注及源 fragment 到实际输出 anchor 的映射。"""
|
|
89
|
+
|
|
90
|
+
def __init__(self, documents: list[MarkupAnchorDocument], policy: MarkupAnchorPolicy) -> None:
|
|
91
|
+
"""按调用方文档顺序建立稳定索引,并保留格式专属 identity 规则。"""
|
|
92
|
+
self._policy = policy
|
|
93
|
+
self._heading_anchors: dict[etree._Element, str] = {}
|
|
94
|
+
self._note_anchors: dict[etree._Element, str] = {}
|
|
95
|
+
self._targets: dict[tuple[str, str | None], str] = {}
|
|
96
|
+
self._heading_labels: dict[str, str] = {}
|
|
97
|
+
for document in documents:
|
|
98
|
+
self._register_document(document)
|
|
99
|
+
|
|
100
|
+
def _register_document(self, document: MarkupAnchorDocument) -> None:
|
|
101
|
+
"""登记单个 DOM 的标题、脚注及全部可解析 fragment 别名。"""
|
|
102
|
+
headings: list[tuple[etree._Element, str]] = []
|
|
103
|
+
for element in document.root.iter():
|
|
104
|
+
if not isinstance(element.tag, str) or local_name(element) not in _HEADING_TAGS:
|
|
105
|
+
continue
|
|
106
|
+
if label := visible_element_text(element, document):
|
|
107
|
+
headings.append((element, label))
|
|
108
|
+
for ordinal, (heading, label) in enumerate(headings):
|
|
109
|
+
identity = self._policy.heading_identity(heading, ordinal)
|
|
110
|
+
anchor = canonical_anchor(self._policy.anchor_prefix, document.key, identity)
|
|
111
|
+
self._heading_anchors[heading] = anchor
|
|
112
|
+
self._heading_labels[anchor] = label
|
|
113
|
+
|
|
114
|
+
notes = [
|
|
115
|
+
element
|
|
116
|
+
for element in document.root.iter()
|
|
117
|
+
if isinstance(element.tag, str) and self._policy.is_materializable_note(element, document)
|
|
118
|
+
]
|
|
119
|
+
for ordinal, note in enumerate(notes):
|
|
120
|
+
identity = self._policy.note_identity(note, ordinal)
|
|
121
|
+
self._note_anchors[note] = canonical_anchor(self._policy.anchor_prefix, document.key, identity)
|
|
122
|
+
|
|
123
|
+
if self._policy.register_document_start and headings:
|
|
124
|
+
self._targets[(document.key, None)] = self._heading_anchors[headings[0][0]]
|
|
125
|
+
for element in document.root.iter():
|
|
126
|
+
if not isinstance(element.tag, str) or not (fragment := element_id(element)):
|
|
127
|
+
continue
|
|
128
|
+
target_key = (document.key, fragment)
|
|
129
|
+
if target_key in self._targets:
|
|
130
|
+
continue
|
|
131
|
+
if anchor := self._target_anchor(element):
|
|
132
|
+
self._targets[target_key] = anchor
|
|
133
|
+
|
|
134
|
+
def _target_anchor(self, element: etree._Element) -> str | None:
|
|
135
|
+
"""把任意 fragment 元素映射到自身、最近祖先或首个后代输出目标。"""
|
|
136
|
+
direct = self._heading_anchors.get(element) or self._note_anchors.get(element)
|
|
137
|
+
if direct is not None:
|
|
138
|
+
return direct
|
|
139
|
+
ancestor = next(
|
|
140
|
+
(parent for parent in element.iterancestors() if parent in self._heading_anchors or parent in self._note_anchors),
|
|
141
|
+
None,
|
|
142
|
+
)
|
|
143
|
+
if ancestor is not None:
|
|
144
|
+
return self._heading_anchors.get(ancestor) or self._note_anchors.get(ancestor)
|
|
145
|
+
descendant = next(
|
|
146
|
+
(
|
|
147
|
+
child
|
|
148
|
+
for child in element.iterdescendants()
|
|
149
|
+
if isinstance(child.tag, str) and (child in self._heading_anchors or child in self._note_anchors)
|
|
150
|
+
),
|
|
151
|
+
None,
|
|
152
|
+
)
|
|
153
|
+
if descendant is None:
|
|
154
|
+
return None
|
|
155
|
+
return self._heading_anchors.get(descendant) or self._note_anchors.get(descendant)
|
|
156
|
+
|
|
157
|
+
def heading_anchor(self, heading: etree._Element) -> str | None:
|
|
158
|
+
"""返回一个已登记标题的规范 anchor。"""
|
|
159
|
+
return self._heading_anchors.get(heading)
|
|
160
|
+
|
|
161
|
+
def heading_label(self, anchor: str) -> str | None:
|
|
162
|
+
"""返回规范标题 anchor 对应的可见标题文本。"""
|
|
163
|
+
return self._heading_labels.get(anchor)
|
|
164
|
+
|
|
165
|
+
def note_anchor(self, note: etree._Element) -> str | None:
|
|
166
|
+
"""返回一个已登记脚注的规范 anchor。"""
|
|
167
|
+
return self._note_anchors.get(note)
|
|
168
|
+
|
|
169
|
+
def resolve_target(self, document_key: str, fragment: str | None) -> str | None:
|
|
170
|
+
"""按文档 key 与可选源 fragment 返回不带井号的规范 anchor。"""
|
|
171
|
+
return self._targets.get((document_key, fragment))
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
__all__ = [
|
|
175
|
+
"AnchorTextNormalization",
|
|
176
|
+
"AnchorVisibilityScope",
|
|
177
|
+
"MarkupAnchorDocument",
|
|
178
|
+
"MarkupAnchorPolicy",
|
|
179
|
+
"MarkupAnchorRegistry",
|
|
180
|
+
"canonical_anchor",
|
|
181
|
+
"element_id",
|
|
182
|
+
"visible_element_text",
|
|
183
|
+
]
|
|
184
|
+
|
|
185
|
+
# 保持既有公开类型的 pickle 路径,所有旧、新入口指向同一个类。
|
|
186
|
+
preserve_type_module(MarkupAnchorDocument, "docvortex.analyzers.native._shared.markup.anchors")
|
|
187
|
+
preserve_type_module(MarkupAnchorPolicy, "docvortex.analyzers.native._shared.markup.anchors")
|
|
188
|
+
preserve_type_module(MarkupAnchorRegistry, "docvortex.analyzers.native._shared.markup.anchors")
|