docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1156 @@
|
|
|
1
|
+
"""严格 MiddleJson 到单正文 EPUB 3.3 的静态 XHTML renderer。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
from functools import lru_cache
|
|
10
|
+
from importlib import resources
|
|
11
|
+
import re
|
|
12
|
+
from urllib.parse import quote, unquote, urlsplit
|
|
13
|
+
from uuid import NAMESPACE_URL, uuid5
|
|
14
|
+
|
|
15
|
+
from bs4 import BeautifulSoup, NavigableString, Tag
|
|
16
|
+
from bs4.element import Comment, Doctype, ProcessingInstruction
|
|
17
|
+
from latex2mathml.converter import convert as latex_to_mathml
|
|
18
|
+
from lxml import etree
|
|
19
|
+
|
|
20
|
+
from ....content.inline import inline_plain_text, join_inline_spans, normalize_inline_spans
|
|
21
|
+
from ....schema import (
|
|
22
|
+
PAGE_AUXILIARY_BLOCK_TYPES,
|
|
23
|
+
RAW_ALGORITHM,
|
|
24
|
+
AlgorithmBodyBlock,
|
|
25
|
+
BlockBase,
|
|
26
|
+
BlockType,
|
|
27
|
+
ChartAnnotationBlock,
|
|
28
|
+
ChartBlock,
|
|
29
|
+
ChartBodyBlock,
|
|
30
|
+
CodeAnnotationBlock,
|
|
31
|
+
CodeBlock,
|
|
32
|
+
CodeBodyBlock,
|
|
33
|
+
CodeInlineSpan,
|
|
34
|
+
DocTitleBlock,
|
|
35
|
+
EquationBlock,
|
|
36
|
+
EquationInlineSpan,
|
|
37
|
+
HyperlinkSpan,
|
|
38
|
+
ImageAnnotationBlock,
|
|
39
|
+
ImageBlock,
|
|
40
|
+
ImageBodyBlock,
|
|
41
|
+
IndexBlock,
|
|
42
|
+
InlineSpan,
|
|
43
|
+
ListBlock,
|
|
44
|
+
MiddleJson,
|
|
45
|
+
PageFootnoteBlock,
|
|
46
|
+
ParagraphTitleBlock,
|
|
47
|
+
RefTextBlock,
|
|
48
|
+
TableAnnotationBlock,
|
|
49
|
+
TableBlock,
|
|
50
|
+
TableBodyBlock,
|
|
51
|
+
TextBlock,
|
|
52
|
+
TextSpan,
|
|
53
|
+
TitleBlockBase,
|
|
54
|
+
)
|
|
55
|
+
from ...contracts import AssetResolver, EpubRenderOptions
|
|
56
|
+
from ..common.index import strip_index_page_tail
|
|
57
|
+
from ..common.list_items import ListItem, parse_list_item_marker, reference_list_needs_bullets
|
|
58
|
+
from ..common.planner import PlannedBlock, build_render_plan
|
|
59
|
+
from .assets import EpubAssetRegistry
|
|
60
|
+
from .package import EpubMetadata, NavigationItem, build_epub_package
|
|
61
|
+
|
|
62
|
+
_EPUB_NS = "http://www.idpf.org/2007/ops"
|
|
63
|
+
_MATHML_NS = "http://www.w3.org/1998/Math/MathML"
|
|
64
|
+
_XHTML_NS = "http://www.w3.org/1999/xhtml"
|
|
65
|
+
_XML_NS = "http://www.w3.org/XML/1998/namespace"
|
|
66
|
+
_STYLE_RESOURCE_NAME = "docvortex.css"
|
|
67
|
+
_INVALID_XML_TEXT_RE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f\ud800-\udfff]")
|
|
68
|
+
_MARKUP_TOKEN_RE = re.compile(
|
|
69
|
+
r"<\s*(?P<closing>/)?\s*(?P<name>[A-Za-z][A-Za-z0-9:-]*)\b(?P<attrs>[^>]*)>",
|
|
70
|
+
re.DOTALL,
|
|
71
|
+
)
|
|
72
|
+
_SAFE_LANGUAGE_RE = re.compile(r"[a-z0-9][a-z0-9-]{0,31}\Z")
|
|
73
|
+
_ALLOWED_MARKUP_TAGS = {
|
|
74
|
+
"a",
|
|
75
|
+
"b",
|
|
76
|
+
"blockquote",
|
|
77
|
+
"br",
|
|
78
|
+
"caption",
|
|
79
|
+
"code",
|
|
80
|
+
"col",
|
|
81
|
+
"colgroup",
|
|
82
|
+
"details",
|
|
83
|
+
"div",
|
|
84
|
+
"em",
|
|
85
|
+
"eq",
|
|
86
|
+
"i",
|
|
87
|
+
"img",
|
|
88
|
+
"kbd",
|
|
89
|
+
"li",
|
|
90
|
+
"mark",
|
|
91
|
+
"ol",
|
|
92
|
+
"p",
|
|
93
|
+
"pre",
|
|
94
|
+
"s",
|
|
95
|
+
"span",
|
|
96
|
+
"strong",
|
|
97
|
+
"sub",
|
|
98
|
+
"summary",
|
|
99
|
+
"summary",
|
|
100
|
+
"sup",
|
|
101
|
+
"table",
|
|
102
|
+
"tbody",
|
|
103
|
+
"td",
|
|
104
|
+
"tfoot",
|
|
105
|
+
"th",
|
|
106
|
+
"thead",
|
|
107
|
+
"tr",
|
|
108
|
+
"u",
|
|
109
|
+
"ul",
|
|
110
|
+
}
|
|
111
|
+
_DROP_CONTENT_TAGS = {
|
|
112
|
+
"audio",
|
|
113
|
+
"button",
|
|
114
|
+
"canvas",
|
|
115
|
+
"embed",
|
|
116
|
+
"form",
|
|
117
|
+
"head",
|
|
118
|
+
"iframe",
|
|
119
|
+
"input",
|
|
120
|
+
"math",
|
|
121
|
+
"noscript",
|
|
122
|
+
"object",
|
|
123
|
+
"script",
|
|
124
|
+
"select",
|
|
125
|
+
"style",
|
|
126
|
+
"svg",
|
|
127
|
+
"template",
|
|
128
|
+
"textarea",
|
|
129
|
+
"video",
|
|
130
|
+
}
|
|
131
|
+
_SOURCE_MARKUP_TAGS = _ALLOWED_MARKUP_TAGS | _DROP_CONTENT_TAGS
|
|
132
|
+
_VOID_MARKUP_TAGS = {"br", "col", "img"}
|
|
133
|
+
_PHRASING_MARKUP_TAGS = {
|
|
134
|
+
"a",
|
|
135
|
+
"b",
|
|
136
|
+
"code",
|
|
137
|
+
"em",
|
|
138
|
+
"i",
|
|
139
|
+
"kbd",
|
|
140
|
+
"mark",
|
|
141
|
+
"p",
|
|
142
|
+
"s",
|
|
143
|
+
"span",
|
|
144
|
+
"strong",
|
|
145
|
+
"sub",
|
|
146
|
+
"sup",
|
|
147
|
+
"u",
|
|
148
|
+
}
|
|
149
|
+
_BLOCK_MARKUP_TAGS = {"blockquote", "details", "div", "ol", "p", "pre", "table", "ul"}
|
|
150
|
+
_TABLE_PARENT_RULES = {
|
|
151
|
+
"caption": {"table"},
|
|
152
|
+
"col": {"colgroup"},
|
|
153
|
+
"colgroup": {"table"},
|
|
154
|
+
"tbody": {"table"},
|
|
155
|
+
"td": {"tr"},
|
|
156
|
+
"tfoot": {"table"},
|
|
157
|
+
"th": {"tr"},
|
|
158
|
+
"thead": {"table"},
|
|
159
|
+
"tr": {"table", "tbody", "tfoot", "thead"},
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
@dataclass(frozen=True, slots=True)
|
|
164
|
+
class _TitleTarget:
|
|
165
|
+
"""保存正文标题的可见文本、层级与 XHTML 目标。"""
|
|
166
|
+
|
|
167
|
+
title: str
|
|
168
|
+
level: int
|
|
169
|
+
target_id: str
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
class _AnchorRegistry:
|
|
173
|
+
"""为正文文本、标题和页面脚注分配文档级唯一 XHTML id。"""
|
|
174
|
+
|
|
175
|
+
def __init__(self, middle_json: MiddleJson) -> None:
|
|
176
|
+
"""按页面与 block 顺序建立目标、来源 anchor 和标题索引。"""
|
|
177
|
+
self._block_targets: dict[tuple[int, int, str], str] = {}
|
|
178
|
+
self._anchor_targets: dict[str, str] = {}
|
|
179
|
+
self._footnote_targets: set[str] = set()
|
|
180
|
+
self.title_targets: list[_TitleTarget] = []
|
|
181
|
+
used_ids: set[str] = {"content-start"}
|
|
182
|
+
heading_position = 0
|
|
183
|
+
text_position = 0
|
|
184
|
+
footnote_position = 0
|
|
185
|
+
for page in middle_json.pages:
|
|
186
|
+
for block in page.blocks:
|
|
187
|
+
if isinstance(block, TitleBlockBase):
|
|
188
|
+
visible = inline_plain_text(block.content).strip()
|
|
189
|
+
if not visible:
|
|
190
|
+
continue
|
|
191
|
+
heading_position += 1
|
|
192
|
+
target_id = _allocate_target_id(
|
|
193
|
+
block.anchor,
|
|
194
|
+
fallback=f"heading-{heading_position}",
|
|
195
|
+
used_ids=used_ids,
|
|
196
|
+
)
|
|
197
|
+
self.title_targets.append(_TitleTarget(visible, block.level, target_id))
|
|
198
|
+
elif isinstance(block, TextBlock):
|
|
199
|
+
visible = inline_plain_text(block.content).strip()
|
|
200
|
+
if not visible or not (block.anchor or "").strip():
|
|
201
|
+
continue
|
|
202
|
+
text_position += 1
|
|
203
|
+
target_id = _allocate_target_id(
|
|
204
|
+
block.anchor,
|
|
205
|
+
fallback=f"text-{text_position}",
|
|
206
|
+
used_ids=used_ids,
|
|
207
|
+
)
|
|
208
|
+
elif isinstance(block, PageFootnoteBlock):
|
|
209
|
+
visible = inline_plain_text(block.content).strip()
|
|
210
|
+
if not visible:
|
|
211
|
+
continue
|
|
212
|
+
footnote_position += 1
|
|
213
|
+
target_id = _allocate_target_id(
|
|
214
|
+
block.anchor,
|
|
215
|
+
fallback=f"footnote-{footnote_position}",
|
|
216
|
+
used_ids=used_ids,
|
|
217
|
+
)
|
|
218
|
+
self._footnote_targets.add(target_id)
|
|
219
|
+
else:
|
|
220
|
+
continue
|
|
221
|
+
assert block.index is not None
|
|
222
|
+
self._block_targets[(page.page_idx, block.index, str(block.type))] = target_id
|
|
223
|
+
anchor_key = _anchor_key(block.anchor)
|
|
224
|
+
if anchor_key and anchor_key not in self._anchor_targets:
|
|
225
|
+
self._anchor_targets[anchor_key] = target_id
|
|
226
|
+
|
|
227
|
+
def target_for_block(self, page_idx: int, block: BlockBase) -> str | None:
|
|
228
|
+
"""按来源页、index 和类型返回标题或脚注的唯一目标。"""
|
|
229
|
+
if block.index is None:
|
|
230
|
+
return None
|
|
231
|
+
return self._block_targets.get((page_idx, block.index, str(block.type)))
|
|
232
|
+
|
|
233
|
+
def target_for_anchor(self, anchor: str | None) -> str | None:
|
|
234
|
+
"""按 producer anchor 返回首次匹配的正文目标。"""
|
|
235
|
+
return self._anchor_targets.get(_anchor_key(anchor))
|
|
236
|
+
|
|
237
|
+
def is_footnote_target(self, target_id: str) -> bool:
|
|
238
|
+
"""判断目标是否对应页面脚注,以便标注 noteref 语义。"""
|
|
239
|
+
return target_id in self._footnote_targets
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class _EpubXhtmlRenderer:
|
|
243
|
+
"""维护单个 EPUB 正文的锚点、素材与 MathML 状态。"""
|
|
244
|
+
|
|
245
|
+
def __init__(
|
|
246
|
+
self,
|
|
247
|
+
middle_json: MiddleJson,
|
|
248
|
+
*,
|
|
249
|
+
metadata: EpubMetadata,
|
|
250
|
+
assets: EpubAssetRegistry,
|
|
251
|
+
anchors: _AnchorRegistry,
|
|
252
|
+
) -> None:
|
|
253
|
+
"""保存严格输入和已规范化的调用状态。"""
|
|
254
|
+
self.middle_json = middle_json
|
|
255
|
+
self.metadata = metadata
|
|
256
|
+
self.assets = assets
|
|
257
|
+
self.anchors = anchors
|
|
258
|
+
self.has_mathml = False
|
|
259
|
+
|
|
260
|
+
def render(self) -> bytes:
|
|
261
|
+
"""把完整 render plan 写成一个无脚本 XHTML content document。"""
|
|
262
|
+
root = etree.Element(
|
|
263
|
+
_xhtml("html"),
|
|
264
|
+
nsmap={None: _XHTML_NS, "epub": _EPUB_NS},
|
|
265
|
+
attrib={f"{{{_XML_NS}}}lang": self.metadata.language, "lang": self.metadata.language},
|
|
266
|
+
)
|
|
267
|
+
head = etree.SubElement(root, _xhtml("head"))
|
|
268
|
+
etree.SubElement(head, _xhtml("meta"), charset="utf-8")
|
|
269
|
+
title = etree.SubElement(head, _xhtml("title"))
|
|
270
|
+
title.text = self.metadata.title
|
|
271
|
+
etree.SubElement(
|
|
272
|
+
head,
|
|
273
|
+
_xhtml("link"),
|
|
274
|
+
rel="stylesheet",
|
|
275
|
+
href="../styles/docvortex.css",
|
|
276
|
+
type="text/css",
|
|
277
|
+
)
|
|
278
|
+
body = etree.SubElement(root, _xhtml("body"), attrib={"class": "docvortex-epub-body"})
|
|
279
|
+
article = etree.SubElement(
|
|
280
|
+
body,
|
|
281
|
+
_xhtml("article"),
|
|
282
|
+
id="content-start",
|
|
283
|
+
attrib={"class": "docvortex-document"},
|
|
284
|
+
)
|
|
285
|
+
self._render_pages(article, build_render_plan(self.middle_json))
|
|
286
|
+
return etree.tostring(
|
|
287
|
+
root,
|
|
288
|
+
encoding="utf-8",
|
|
289
|
+
xml_declaration=True,
|
|
290
|
+
doctype="<!DOCTYPE html>",
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
def _render_pages(self, parent: etree._Element, pages: list[list[PlannedBlock]]) -> None:
|
|
294
|
+
"""把默认计划展平到单个连续阅读容器。"""
|
|
295
|
+
for page in pages:
|
|
296
|
+
for planned in page:
|
|
297
|
+
rendered = self._render_planned_block(planned)
|
|
298
|
+
if rendered is not None:
|
|
299
|
+
parent.append(rendered)
|
|
300
|
+
|
|
301
|
+
def _render_planned_block(self, planned: PlannedBlock) -> etree._Element | None:
|
|
302
|
+
"""过滤计划块、分派具体类型并追加稳定来源属性。"""
|
|
303
|
+
if planned.removed:
|
|
304
|
+
return None
|
|
305
|
+
block = planned.block
|
|
306
|
+
if block.type in PAGE_AUXILIARY_BLOCK_TYPES:
|
|
307
|
+
return None
|
|
308
|
+
content = self._render_block_content(planned)
|
|
309
|
+
if content is None:
|
|
310
|
+
return None
|
|
311
|
+
wrapper = etree.Element(
|
|
312
|
+
_xhtml("div"),
|
|
313
|
+
attrib={
|
|
314
|
+
"class": "docvortex-block",
|
|
315
|
+
"data-page-idx": str(planned.page_idx),
|
|
316
|
+
"data-block-type": str(block.type),
|
|
317
|
+
},
|
|
318
|
+
)
|
|
319
|
+
if block.index is not None:
|
|
320
|
+
wrapper.set("data-block-index", str(block.index))
|
|
321
|
+
wrapper.append(content)
|
|
322
|
+
return wrapper
|
|
323
|
+
|
|
324
|
+
def _render_block_content(self, planned: PlannedBlock) -> etree._Element | None:
|
|
325
|
+
"""把一个具体 PageBlock 映射为静态 XHTML 元素。"""
|
|
326
|
+
block = planned.block
|
|
327
|
+
if isinstance(block, (TextBlock, RefTextBlock)):
|
|
328
|
+
content = join_inline_spans(planned.text_contents or [block.content])
|
|
329
|
+
paragraph = etree.Element(
|
|
330
|
+
_xhtml("p"),
|
|
331
|
+
attrib={"class": "docvortex-ref-text" if isinstance(block, RefTextBlock) else "docvortex-text"},
|
|
332
|
+
)
|
|
333
|
+
if isinstance(block, TextBlock):
|
|
334
|
+
target_id = self.anchors.target_for_block(planned.page_idx, block)
|
|
335
|
+
if target_id:
|
|
336
|
+
paragraph.set("id", target_id)
|
|
337
|
+
self._append_inline_spans(paragraph, content)
|
|
338
|
+
return paragraph if _has_visible_content(paragraph) else None
|
|
339
|
+
if isinstance(block, (DocTitleBlock, ParagraphTitleBlock)):
|
|
340
|
+
return self._render_title(planned.page_idx, block)
|
|
341
|
+
if isinstance(block, PageFootnoteBlock):
|
|
342
|
+
return self._render_page_footnote(planned.page_idx, block)
|
|
343
|
+
if isinstance(block, EquationBlock):
|
|
344
|
+
return self._render_equation(block)
|
|
345
|
+
if isinstance(block, ListBlock):
|
|
346
|
+
return self._render_list(block)
|
|
347
|
+
if isinstance(block, IndexBlock):
|
|
348
|
+
return self._render_index(block)
|
|
349
|
+
if isinstance(block, ImageBlock):
|
|
350
|
+
return self._render_image_block(block)
|
|
351
|
+
if isinstance(block, TableBlock):
|
|
352
|
+
return self._render_table_block(block)
|
|
353
|
+
if isinstance(block, ChartBlock):
|
|
354
|
+
return self._render_chart_block(block)
|
|
355
|
+
if isinstance(block, CodeBlock):
|
|
356
|
+
return self._render_code_block(block)
|
|
357
|
+
raise TypeError(f"Unsupported PageBlock type: {type(block).__name__}")
|
|
358
|
+
|
|
359
|
+
def _render_title(self, page_idx: int, block: TitleBlockBase) -> etree._Element | None:
|
|
360
|
+
"""渲染带文档级唯一 id 的 h1-h6 标题。"""
|
|
361
|
+
if not inline_plain_text(block.content).strip():
|
|
362
|
+
return None
|
|
363
|
+
level = min(max(block.level, 1), 6)
|
|
364
|
+
heading = etree.Element(_xhtml(f"h{level}"), attrib={"class": f"docvortex-heading docvortex-heading--{level}"})
|
|
365
|
+
target_id = self.anchors.target_for_block(page_idx, block)
|
|
366
|
+
if target_id:
|
|
367
|
+
heading.set("id", target_id)
|
|
368
|
+
self._append_inline_spans(heading, block.content)
|
|
369
|
+
return heading
|
|
370
|
+
|
|
371
|
+
def _render_page_footnote(self, page_idx: int, block: PageFootnoteBlock) -> etree._Element | None:
|
|
372
|
+
"""把页面脚注保留为 EPUB footnote aside。"""
|
|
373
|
+
footnote = etree.Element(
|
|
374
|
+
_xhtml("aside"),
|
|
375
|
+
attrib={"class": "docvortex-page-footnote", f"{{{_EPUB_NS}}}type": "footnote", "role": "doc-footnote"},
|
|
376
|
+
)
|
|
377
|
+
target_id = self.anchors.target_for_block(page_idx, block)
|
|
378
|
+
if target_id:
|
|
379
|
+
footnote.set("id", target_id)
|
|
380
|
+
self._append_inline_spans(footnote, block.content)
|
|
381
|
+
return footnote if _has_visible_content(footnote) else None
|
|
382
|
+
|
|
383
|
+
def _render_equation(self, block: EquationBlock) -> etree._Element | None:
|
|
384
|
+
"""优先渲染行间 MathML,空公式时才尝试包内图片。"""
|
|
385
|
+
container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-equation"})
|
|
386
|
+
if block.content.strip():
|
|
387
|
+
self._append_math(container, block.content, display="block")
|
|
388
|
+
elif source := self.assets.resolve_block(block):
|
|
389
|
+
etree.SubElement(container, _xhtml("img"), src=source, alt="formula")
|
|
390
|
+
return container if _has_visible_content(container) else None
|
|
391
|
+
|
|
392
|
+
def _render_list(self, block: ListBlock) -> etree._Element | None:
|
|
393
|
+
"""按共享 marker 语义递归渲染原生有序、无序或显式 marker 列表。"""
|
|
394
|
+
parsed_leaves = [
|
|
395
|
+
parse_list_item_marker(child.content)
|
|
396
|
+
for child in block.content
|
|
397
|
+
if not isinstance(child, ListBlock) and inline_plain_text(child.content).strip()
|
|
398
|
+
]
|
|
399
|
+
add_reference_bullets = reference_list_needs_bullets(block)
|
|
400
|
+
container_tag, list_type, class_name = _classify_list(parsed_leaves, add_reference_bullets)
|
|
401
|
+
container = etree.Element(_xhtml(container_tag), attrib={"class": f"docvortex-list {class_name}"})
|
|
402
|
+
if list_type:
|
|
403
|
+
container.set("type", list_type)
|
|
404
|
+
if container_tag == "ol" and parsed_leaves and parsed_leaves[0].value not in (None, 1):
|
|
405
|
+
container.set("start", str(parsed_leaves[0].value))
|
|
406
|
+
expected_value: int | None = None
|
|
407
|
+
last_item: etree._Element | None = None
|
|
408
|
+
for child in block.content:
|
|
409
|
+
if isinstance(child, ListBlock):
|
|
410
|
+
nested = self._render_list(child)
|
|
411
|
+
if nested is None:
|
|
412
|
+
continue
|
|
413
|
+
if last_item is None:
|
|
414
|
+
last_item = etree.SubElement(container, _xhtml("li"), attrib={"class": "docvortex-list-item--orphan"})
|
|
415
|
+
last_item.append(nested)
|
|
416
|
+
continue
|
|
417
|
+
parsed = parse_list_item_marker(child.content)
|
|
418
|
+
item_content, marker = _list_item_content(
|
|
419
|
+
parsed,
|
|
420
|
+
add_reference_bullets,
|
|
421
|
+
explicit_markers=class_name == "docvortex-list--explicit",
|
|
422
|
+
)
|
|
423
|
+
item = etree.SubElement(container, _xhtml("li"))
|
|
424
|
+
if class_name == "docvortex-list--explicit":
|
|
425
|
+
item.set("class", "docvortex-list-item--explicit")
|
|
426
|
+
if container_tag == "ol" and parsed.kind == "ordered" and parsed.value is not None:
|
|
427
|
+
if expected_value is None:
|
|
428
|
+
expected_value = parsed.value
|
|
429
|
+
if parsed.value != expected_value:
|
|
430
|
+
item.set("value", str(parsed.value))
|
|
431
|
+
expected_value = parsed.value + 1
|
|
432
|
+
if marker or class_name == "docvortex-list--explicit":
|
|
433
|
+
marker_element = etree.SubElement(item, _xhtml("span"), attrib={"class": "docvortex-list-marker"})
|
|
434
|
+
marker_element.text = marker or ""
|
|
435
|
+
content_element = etree.SubElement(item, _xhtml("span"), attrib={"class": "docvortex-list-content"})
|
|
436
|
+
self._append_inline_spans(content_element, item_content)
|
|
437
|
+
last_item = item
|
|
438
|
+
return container if len(container) else None
|
|
439
|
+
|
|
440
|
+
def _render_index(self, block: IndexBlock) -> etree._Element | None:
|
|
441
|
+
"""把源目录保留为正文内导航,并只链接到真实正文目标。"""
|
|
442
|
+
navigation = etree.Element(_xhtml("nav"), attrib={"class": "docvortex-index", "aria-label": "Table of contents"})
|
|
443
|
+
listing = etree.SubElement(navigation, _xhtml("ul"))
|
|
444
|
+
self._append_index_children(listing, block)
|
|
445
|
+
return navigation if len(listing) else None
|
|
446
|
+
|
|
447
|
+
def _append_index_children(self, parent: etree._Element, block: IndexBlock) -> None:
|
|
448
|
+
"""递归渲染 IndexBlock,并把孤立嵌套目录提升到当前层级。"""
|
|
449
|
+
last_item: etree._Element | None = None
|
|
450
|
+
for child in block.content:
|
|
451
|
+
if isinstance(child, IndexBlock):
|
|
452
|
+
nested = etree.Element(_xhtml("ul"))
|
|
453
|
+
self._append_index_children(nested, child)
|
|
454
|
+
if not len(nested):
|
|
455
|
+
continue
|
|
456
|
+
if last_item is None:
|
|
457
|
+
last_item = etree.SubElement(parent, _xhtml("li"), attrib={"class": "docvortex-list-item--orphan"})
|
|
458
|
+
last_item.append(nested)
|
|
459
|
+
continue
|
|
460
|
+
content = strip_index_page_tail(child.content)
|
|
461
|
+
if not inline_plain_text(content).strip():
|
|
462
|
+
continue
|
|
463
|
+
item = etree.SubElement(parent, _xhtml("li"))
|
|
464
|
+
target = self.anchors.target_for_anchor(child.anchor)
|
|
465
|
+
inline_parent = item
|
|
466
|
+
if target:
|
|
467
|
+
inline_parent = etree.SubElement(item, _xhtml("a"), href=f"#{quote(target, safe='-._~')}")
|
|
468
|
+
self._append_inline_spans(inline_parent, content)
|
|
469
|
+
last_item = item
|
|
470
|
+
|
|
471
|
+
def _render_image_block(self, block: ImageBlock) -> etree._Element | None:
|
|
472
|
+
"""按子块顺序渲染图片主体及其标题、脚注。"""
|
|
473
|
+
figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--image"})
|
|
474
|
+
for child in block.content:
|
|
475
|
+
rendered = (
|
|
476
|
+
self._render_image_body(block, child) if isinstance(child, ImageBodyBlock) else self._render_annotation(child)
|
|
477
|
+
)
|
|
478
|
+
if rendered is not None:
|
|
479
|
+
figure.append(rendered)
|
|
480
|
+
return figure if len(figure) else None
|
|
481
|
+
|
|
482
|
+
def _render_image_body(self, parent: ImageBlock, block: ImageBodyBlock) -> etree._Element | None:
|
|
483
|
+
"""渲染包内图片,并在缺图时保留已有结构或可见文字。"""
|
|
484
|
+
container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--image"})
|
|
485
|
+
source = self.assets.resolve_block(block)
|
|
486
|
+
if source:
|
|
487
|
+
alt = _plain_content_text(block.content) or parent.sub_type or "image"
|
|
488
|
+
etree.SubElement(container, _xhtml("img"), src=source, alt=alt, attrib={"class": "docvortex-image"})
|
|
489
|
+
if block.content.strip():
|
|
490
|
+
content = etree.Element(_xhtml("div"), attrib={"class": "docvortex-image-content"})
|
|
491
|
+
self._append_rich_or_text(content, block.content)
|
|
492
|
+
if _has_visible_content(content):
|
|
493
|
+
container.append(content)
|
|
494
|
+
return container if _has_visible_content(container) else None
|
|
495
|
+
|
|
496
|
+
def _render_table_block(self, block: TableBlock) -> etree._Element | None:
|
|
497
|
+
"""按子块顺序渲染结构表格、图片回退及说明。"""
|
|
498
|
+
figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--table"})
|
|
499
|
+
for child in block.content:
|
|
500
|
+
rendered = self._render_table_body(child) if isinstance(child, TableBodyBlock) else self._render_annotation(child)
|
|
501
|
+
if rendered is not None:
|
|
502
|
+
figure.append(rendered)
|
|
503
|
+
return figure if len(figure) else None
|
|
504
|
+
|
|
505
|
+
def _render_table_body(self, block: TableBodyBlock) -> etree._Element | None:
|
|
506
|
+
"""优先输出安全结构内容,无内容时尝试整体表格图片。"""
|
|
507
|
+
container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--table"})
|
|
508
|
+
if block.content.strip():
|
|
509
|
+
if _is_supported_markup(block.content):
|
|
510
|
+
self._append_markup(container, block.content)
|
|
511
|
+
else:
|
|
512
|
+
pre = etree.SubElement(container, _xhtml("pre"), attrib={"class": "docvortex-table-text"})
|
|
513
|
+
pre.text = _normalize_xml_text(block.content)
|
|
514
|
+
if not _has_visible_content(container) and (source := self.assets.resolve_block(block)):
|
|
515
|
+
etree.SubElement(container, _xhtml("img"), src=source, alt="table", attrib={"class": "docvortex-table-image"})
|
|
516
|
+
return container if _has_visible_content(container) else None
|
|
517
|
+
|
|
518
|
+
def _render_chart_block(self, block: ChartBlock) -> etree._Element | None:
|
|
519
|
+
"""按子块顺序渲染图表图片、结构内容及说明。"""
|
|
520
|
+
figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--chart"})
|
|
521
|
+
for child in block.content:
|
|
522
|
+
rendered = (
|
|
523
|
+
self._render_chart_body(block, child) if isinstance(child, ChartBodyBlock) else self._render_annotation(child)
|
|
524
|
+
)
|
|
525
|
+
if rendered is not None:
|
|
526
|
+
figure.append(rendered)
|
|
527
|
+
return figure if len(figure) else None
|
|
528
|
+
|
|
529
|
+
def _render_chart_body(self, parent: ChartBlock, block: ChartBodyBlock) -> etree._Element | None:
|
|
530
|
+
"""渲染包内图表图片,并始终保留并存结构内容。"""
|
|
531
|
+
container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--chart"})
|
|
532
|
+
if source := self.assets.resolve_block(block):
|
|
533
|
+
etree.SubElement(
|
|
534
|
+
container,
|
|
535
|
+
_xhtml("img"),
|
|
536
|
+
src=source,
|
|
537
|
+
alt=parent.sub_type or "chart",
|
|
538
|
+
attrib={"class": "docvortex-chart-image"},
|
|
539
|
+
)
|
|
540
|
+
if block.content.strip():
|
|
541
|
+
content = etree.Element(_xhtml("div"), attrib={"class": "docvortex-chart-content"})
|
|
542
|
+
self._append_rich_or_text(content, block.content, preformatted=True)
|
|
543
|
+
if _has_visible_content(content):
|
|
544
|
+
container.append(content)
|
|
545
|
+
return container if _has_visible_content(container) else None
|
|
546
|
+
|
|
547
|
+
def _render_code_block(self, block: CodeBlock) -> etree._Element | None:
|
|
548
|
+
"""按子块顺序渲染静态代码、算法及其说明。"""
|
|
549
|
+
figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--code"})
|
|
550
|
+
for child in block.content:
|
|
551
|
+
if isinstance(child, (CodeBodyBlock, AlgorithmBodyBlock)):
|
|
552
|
+
rendered = self._render_code_body(block, child)
|
|
553
|
+
else:
|
|
554
|
+
rendered = self._render_annotation(child)
|
|
555
|
+
if rendered is not None:
|
|
556
|
+
figure.append(rendered)
|
|
557
|
+
return figure if len(figure) else None
|
|
558
|
+
|
|
559
|
+
def _render_code_body(self, parent: CodeBlock, block: CodeBodyBlock | AlgorithmBodyBlock) -> etree._Element:
|
|
560
|
+
"""代码使用 pre/code,算法使用保留换行的结构化 Span。"""
|
|
561
|
+
container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--code"})
|
|
562
|
+
if parent.sub_type == BlockType.CODE:
|
|
563
|
+
if not isinstance(block, CodeBodyBlock):
|
|
564
|
+
raise TypeError("code subtype requires CodeBodyBlock")
|
|
565
|
+
pre = etree.SubElement(container, _xhtml("pre"), attrib={"class": "docvortex-code"})
|
|
566
|
+
code = etree.SubElement(pre, _xhtml("code"))
|
|
567
|
+
language = _normalize_code_language(parent.guess_lang)
|
|
568
|
+
if language:
|
|
569
|
+
code.set("class", f"language-{language}")
|
|
570
|
+
code.text = _normalize_xml_text(block.content)
|
|
571
|
+
return container
|
|
572
|
+
if parent.sub_type == RAW_ALGORITHM:
|
|
573
|
+
if not isinstance(block, AlgorithmBodyBlock):
|
|
574
|
+
raise TypeError("algorithm subtype requires AlgorithmBodyBlock")
|
|
575
|
+
algorithm = etree.SubElement(container, _xhtml("div"), attrib={"class": "docvortex-algorithm"})
|
|
576
|
+
self._append_inline_spans(algorithm, block.content, preserve_newlines=True, separate_adjacent_math=True)
|
|
577
|
+
return container
|
|
578
|
+
raise ValueError(f"Unsupported code subtype: {parent.sub_type}")
|
|
579
|
+
|
|
580
|
+
def _render_annotation(
|
|
581
|
+
self,
|
|
582
|
+
block: ImageAnnotationBlock | TableAnnotationBlock | ChartAnnotationBlock | CodeAnnotationBlock,
|
|
583
|
+
) -> etree._Element | None:
|
|
584
|
+
"""按 caption 或 footnote 语义渲染视觉说明。"""
|
|
585
|
+
role = "docvortex-caption" if str(block.type).endswith("caption") else "docvortex-footnote"
|
|
586
|
+
annotation = etree.Element(
|
|
587
|
+
_xhtml("p"),
|
|
588
|
+
attrib={"class": f"{role} {role}--{str(block.type).replace('_', '-')}"},
|
|
589
|
+
)
|
|
590
|
+
self._append_inline_spans(annotation, block.content)
|
|
591
|
+
return annotation if _has_visible_content(annotation) else None
|
|
592
|
+
|
|
593
|
+
def _append_inline_spans(
|
|
594
|
+
self,
|
|
595
|
+
parent: etree._Element,
|
|
596
|
+
spans: list[InlineSpan],
|
|
597
|
+
*,
|
|
598
|
+
preserve_newlines: bool = False,
|
|
599
|
+
separate_adjacent_math: bool = False,
|
|
600
|
+
) -> None:
|
|
601
|
+
"""按结构化 Span 顺序向 XHTML mixed content 追加安全节点。"""
|
|
602
|
+
previous_was_math = False
|
|
603
|
+
for span in spans:
|
|
604
|
+
current_is_math = isinstance(span, EquationInlineSpan)
|
|
605
|
+
if separate_adjacent_math and previous_was_math and current_is_math:
|
|
606
|
+
_append_text(parent, " ", preserve_newlines=True)
|
|
607
|
+
self._append_inline_span(parent, span, preserve_newlines=preserve_newlines)
|
|
608
|
+
previous_was_math = current_is_math
|
|
609
|
+
|
|
610
|
+
def _append_inline_span(self, parent: etree._Element, span: InlineSpan, *, preserve_newlines: bool) -> None:
|
|
611
|
+
"""把单个 Text/Code/Equation/Hyperlink Span 追加到父节点。"""
|
|
612
|
+
if isinstance(span, TextSpan):
|
|
613
|
+
target = _append_text_style_container(parent, span)
|
|
614
|
+
_append_text(target, span.content, preserve_newlines=preserve_newlines)
|
|
615
|
+
return
|
|
616
|
+
if isinstance(span, CodeInlineSpan):
|
|
617
|
+
code = etree.SubElement(parent, _xhtml("code"))
|
|
618
|
+
_append_text(code, span.content, preserve_newlines=preserve_newlines)
|
|
619
|
+
return
|
|
620
|
+
if isinstance(span, EquationInlineSpan):
|
|
621
|
+
self._append_math(parent, span.content, display="inline")
|
|
622
|
+
return
|
|
623
|
+
if isinstance(span, HyperlinkSpan):
|
|
624
|
+
href, target_id = self._resolve_link(span.url)
|
|
625
|
+
link_parent = parent
|
|
626
|
+
if href:
|
|
627
|
+
link_parent = etree.SubElement(parent, _xhtml("a"), href=href)
|
|
628
|
+
if target_id and self.anchors.is_footnote_target(target_id):
|
|
629
|
+
link_parent.set(f"{{{_EPUB_NS}}}type", "noteref")
|
|
630
|
+
link_parent.set("role", "doc-noteref")
|
|
631
|
+
self._append_inline_spans(link_parent, list(span.content), preserve_newlines=preserve_newlines)
|
|
632
|
+
return
|
|
633
|
+
raise TypeError(f"Unsupported inline span: {type(span).__name__}")
|
|
634
|
+
|
|
635
|
+
def _append_math(self, parent: etree._Element, latex: str, *, display: str) -> None:
|
|
636
|
+
"""追加 Presentation MathML,并在转换失败时显示原始 LaTeX。"""
|
|
637
|
+
normalized = latex.strip()
|
|
638
|
+
if not normalized:
|
|
639
|
+
return
|
|
640
|
+
try:
|
|
641
|
+
markup = latex_to_mathml(normalized, display=display)
|
|
642
|
+
parser = etree.XMLParser(resolve_entities=False, load_dtd=False, no_network=True, recover=False, huge_tree=False)
|
|
643
|
+
math = etree.fromstring(markup.encode("utf-8"), parser=parser)
|
|
644
|
+
if math.tag != f"{{{_MATHML_NS}}}math":
|
|
645
|
+
raise ValueError("latex2mathml did not return a MathML root")
|
|
646
|
+
except Exception:
|
|
647
|
+
fallback = etree.SubElement(
|
|
648
|
+
parent,
|
|
649
|
+
_xhtml("code"),
|
|
650
|
+
attrib={"class": f"docvortex-latex-fallback docvortex-latex-fallback--{display}"},
|
|
651
|
+
)
|
|
652
|
+
fallback.text = _normalize_xml_text(normalized)
|
|
653
|
+
return
|
|
654
|
+
parent.append(math)
|
|
655
|
+
self.has_mathml = True
|
|
656
|
+
|
|
657
|
+
def _append_rich_or_text(self, parent: etree._Element, content: str, *, preformatted: bool = False) -> None:
|
|
658
|
+
"""识别安全富 HTML,否则按普通文本或预格式文本输出。"""
|
|
659
|
+
if _is_supported_markup(content):
|
|
660
|
+
self._append_markup(parent, content)
|
|
661
|
+
return
|
|
662
|
+
if preformatted:
|
|
663
|
+
pre = etree.SubElement(parent, _xhtml("pre"))
|
|
664
|
+
pre.text = _normalize_xml_text(content)
|
|
665
|
+
else:
|
|
666
|
+
_append_text(parent, content)
|
|
667
|
+
|
|
668
|
+
def _append_markup(self, parent: etree._Element, markup: str) -> None:
|
|
669
|
+
"""通过 EPUB 专用 allowlist 把不可信 HTML 转为安全 XHTML 节点。"""
|
|
670
|
+
soup = BeautifulSoup(_normalize_xml_text(markup), "html.parser")
|
|
671
|
+
for child in list(soup.contents):
|
|
672
|
+
self._append_soup_node(parent, child)
|
|
673
|
+
|
|
674
|
+
def _append_soup_node(self, parent: etree._Element, node: object) -> None:
|
|
675
|
+
"""递归复制一个 BeautifulSoup 节点,仅创建允许的 XHTML 结构。"""
|
|
676
|
+
if isinstance(node, (Comment, Doctype, ProcessingInstruction)):
|
|
677
|
+
return
|
|
678
|
+
if isinstance(node, NavigableString):
|
|
679
|
+
_append_text(parent, str(node), preserve_newlines=True)
|
|
680
|
+
return
|
|
681
|
+
if not isinstance(node, Tag):
|
|
682
|
+
return
|
|
683
|
+
name = (node.name or "").lower()
|
|
684
|
+
if name in _DROP_CONTENT_TAGS:
|
|
685
|
+
return
|
|
686
|
+
if name not in _ALLOWED_MARKUP_TAGS:
|
|
687
|
+
for child in list(node.children):
|
|
688
|
+
self._append_soup_node(parent, child)
|
|
689
|
+
return
|
|
690
|
+
parent_name = etree.QName(parent).localname
|
|
691
|
+
if name in _TABLE_PARENT_RULES and parent_name not in _TABLE_PARENT_RULES[name]:
|
|
692
|
+
for child in list(node.children):
|
|
693
|
+
self._append_soup_node(parent, child)
|
|
694
|
+
return
|
|
695
|
+
if parent_name in _PHRASING_MARKUP_TAGS and name in _BLOCK_MARKUP_TAGS:
|
|
696
|
+
for child in list(node.children):
|
|
697
|
+
self._append_soup_node(parent, child)
|
|
698
|
+
return
|
|
699
|
+
if name == "a" and parent_name == "a":
|
|
700
|
+
for child in list(node.children):
|
|
701
|
+
self._append_soup_node(parent, child)
|
|
702
|
+
return
|
|
703
|
+
if name == "li" and parent_name not in {"ol", "ul"}:
|
|
704
|
+
if parent_name in _PHRASING_MARKUP_TAGS:
|
|
705
|
+
for child in list(node.children):
|
|
706
|
+
self._append_soup_node(parent, child)
|
|
707
|
+
return
|
|
708
|
+
listing = etree.SubElement(parent, _xhtml("ul"))
|
|
709
|
+
item = etree.SubElement(listing, _xhtml("li"), attrib=_safe_markup_attributes(name, node))
|
|
710
|
+
for child in list(node.children):
|
|
711
|
+
self._append_soup_node(item, child)
|
|
712
|
+
return
|
|
713
|
+
if name in {"ol", "ul"}:
|
|
714
|
+
listing = etree.SubElement(parent, _xhtml(name), attrib=_safe_markup_attributes(name, node))
|
|
715
|
+
for child in list(node.children):
|
|
716
|
+
if isinstance(child, NavigableString) and not str(child).strip():
|
|
717
|
+
continue
|
|
718
|
+
if isinstance(child, Tag) and (child.name or "").lower() == "li":
|
|
719
|
+
self._append_soup_node(listing, child)
|
|
720
|
+
continue
|
|
721
|
+
item = etree.SubElement(listing, _xhtml("li"))
|
|
722
|
+
self._append_soup_node(item, child)
|
|
723
|
+
return
|
|
724
|
+
if name == "eq":
|
|
725
|
+
self._append_math(parent, node.get_text(), display="inline")
|
|
726
|
+
return
|
|
727
|
+
if name == "img":
|
|
728
|
+
source = self.assets.resolve_embedded_source(_attribute_text(node.get("src")))
|
|
729
|
+
alt = _attribute_text(node.get("alt"))
|
|
730
|
+
if source:
|
|
731
|
+
image = etree.SubElement(parent, _xhtml("img"), src=source, alt=_normalize_xml_text(alt))
|
|
732
|
+
title = _attribute_text(node.get("title"))
|
|
733
|
+
if title:
|
|
734
|
+
image.set("title", _normalize_xml_text(title))
|
|
735
|
+
elif alt:
|
|
736
|
+
_append_text(parent, alt)
|
|
737
|
+
return
|
|
738
|
+
if name == "a":
|
|
739
|
+
href, target_id = self._resolve_link(_attribute_text(node.get("href")))
|
|
740
|
+
target_parent = parent
|
|
741
|
+
if href:
|
|
742
|
+
target_parent = etree.SubElement(parent, _xhtml("a"), href=href)
|
|
743
|
+
title = _attribute_text(node.get("title"))
|
|
744
|
+
if title:
|
|
745
|
+
target_parent.set("title", _normalize_xml_text(title))
|
|
746
|
+
if target_id and self.anchors.is_footnote_target(target_id):
|
|
747
|
+
target_parent.set(f"{{{_EPUB_NS}}}type", "noteref")
|
|
748
|
+
target_parent.set("role", "doc-noteref")
|
|
749
|
+
for child in list(node.children):
|
|
750
|
+
self._append_soup_node(target_parent, child)
|
|
751
|
+
return
|
|
752
|
+
attributes = _safe_markup_attributes(name, node)
|
|
753
|
+
if name == "colgroup" and any(isinstance(child, Tag) and child.name == "col" for child in node.children):
|
|
754
|
+
attributes.pop("span", None)
|
|
755
|
+
element = etree.SubElement(parent, _xhtml(name), attrib=attributes)
|
|
756
|
+
if name not in _VOID_MARKUP_TAGS:
|
|
757
|
+
for child in list(node.children):
|
|
758
|
+
self._append_soup_node(element, child)
|
|
759
|
+
|
|
760
|
+
def _resolve_link(self, url: str) -> tuple[str | None, str | None]:
|
|
761
|
+
"""保留安全外链或已登记 fragment,删除无包内目标的相对链接。"""
|
|
762
|
+
normalized = _normalize_xml_text(url).strip()
|
|
763
|
+
if not normalized or normalized.startswith(("//", "\\")):
|
|
764
|
+
return None, None
|
|
765
|
+
if normalized.startswith("#"):
|
|
766
|
+
target = self.anchors.target_for_anchor(unquote(normalized[1:]))
|
|
767
|
+
if target:
|
|
768
|
+
return f"#{quote(target, safe='-._~')}", target
|
|
769
|
+
return None, None
|
|
770
|
+
try:
|
|
771
|
+
parsed = urlsplit(normalized)
|
|
772
|
+
_ = parsed.port
|
|
773
|
+
except ValueError:
|
|
774
|
+
return None, None
|
|
775
|
+
scheme = parsed.scheme.casefold()
|
|
776
|
+
if scheme in {"http", "https"}:
|
|
777
|
+
if not parsed.netloc or parsed.hostname is None or parsed.username is not None or parsed.password is not None:
|
|
778
|
+
return None, None
|
|
779
|
+
elif scheme in {"mailto", "tel"}:
|
|
780
|
+
if not parsed.path:
|
|
781
|
+
return None, None
|
|
782
|
+
else:
|
|
783
|
+
return None, None
|
|
784
|
+
return quote(normalized, safe="/:#?&=%@+~,;!$'*-._"), None
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
def render_epub(
|
|
788
|
+
middle_json: MiddleJson,
|
|
789
|
+
*,
|
|
790
|
+
title: str | None = None,
|
|
791
|
+
authors: tuple[str, ...] = (),
|
|
792
|
+
language: str = "und",
|
|
793
|
+
identifier: str | None = None,
|
|
794
|
+
modified_at: datetime | None = None,
|
|
795
|
+
asset_resolver: AssetResolver | None = None,
|
|
796
|
+
) -> bytes:
|
|
797
|
+
"""把严格 MiddleJson 无副作用地渲染为单正文 EPUB 3.3 字节。"""
|
|
798
|
+
if not isinstance(middle_json, MiddleJson):
|
|
799
|
+
raise TypeError("render_epub expects a MiddleJson instance")
|
|
800
|
+
options = EpubRenderOptions(
|
|
801
|
+
title=title,
|
|
802
|
+
authors=authors,
|
|
803
|
+
language=language,
|
|
804
|
+
identifier=identifier,
|
|
805
|
+
modified_at=modified_at,
|
|
806
|
+
asset_resolver=asset_resolver,
|
|
807
|
+
)
|
|
808
|
+
resolved_title = _resolve_document_title(middle_json, options.title)
|
|
809
|
+
resolved_authors = tuple(_normalize_xml_text(author).strip() for author in options.authors)
|
|
810
|
+
resolved_language = options.language.strip()
|
|
811
|
+
resolved_identifier = (
|
|
812
|
+
options.identifier.strip()
|
|
813
|
+
if options.identifier
|
|
814
|
+
else _stable_identifier(
|
|
815
|
+
middle_json,
|
|
816
|
+
title=resolved_title,
|
|
817
|
+
authors=resolved_authors,
|
|
818
|
+
language=resolved_language,
|
|
819
|
+
)
|
|
820
|
+
)
|
|
821
|
+
resolved_modified = (options.modified_at or datetime.now(timezone.utc)).astimezone(timezone.utc).replace(microsecond=0)
|
|
822
|
+
if resolved_modified.year < 1000:
|
|
823
|
+
raise ValueError("modified_at UTC year must use four digits")
|
|
824
|
+
metadata = EpubMetadata(
|
|
825
|
+
title=resolved_title,
|
|
826
|
+
authors=resolved_authors,
|
|
827
|
+
language=resolved_language,
|
|
828
|
+
identifier=_normalize_xml_text(resolved_identifier),
|
|
829
|
+
modified_at=resolved_modified,
|
|
830
|
+
)
|
|
831
|
+
anchors = _AnchorRegistry(middle_json)
|
|
832
|
+
assets = EpubAssetRegistry(options.asset_resolver)
|
|
833
|
+
renderer = _EpubXhtmlRenderer(
|
|
834
|
+
middle_json,
|
|
835
|
+
metadata=metadata,
|
|
836
|
+
assets=assets,
|
|
837
|
+
anchors=anchors,
|
|
838
|
+
)
|
|
839
|
+
content_xhtml = renderer.render()
|
|
840
|
+
navigation = _build_navigation(middle_json, anchors, resolved_title)
|
|
841
|
+
return build_epub_package(
|
|
842
|
+
metadata=metadata,
|
|
843
|
+
content_xhtml=content_xhtml,
|
|
844
|
+
navigation=navigation,
|
|
845
|
+
stylesheet=_load_epub_stylesheet(),
|
|
846
|
+
assets=assets.assets,
|
|
847
|
+
has_mathml=renderer.has_mathml,
|
|
848
|
+
)
|
|
849
|
+
|
|
850
|
+
|
|
851
|
+
def _build_navigation(middle_json: MiddleJson, anchors: _AnchorRegistry, document_title: str) -> list[NavigationItem]:
|
|
852
|
+
"""优先使用有效 IndexBlock,否则按标题层级或正文起点生成 toc。"""
|
|
853
|
+
for page in middle_json.pages:
|
|
854
|
+
for block in page.blocks:
|
|
855
|
+
if not isinstance(block, IndexBlock):
|
|
856
|
+
continue
|
|
857
|
+
items = _navigation_from_index(block, anchors)
|
|
858
|
+
if items:
|
|
859
|
+
return items
|
|
860
|
+
if anchors.title_targets:
|
|
861
|
+
roots: list[NavigationItem] = []
|
|
862
|
+
stack: list[tuple[int, NavigationItem]] = []
|
|
863
|
+
for target in anchors.title_targets:
|
|
864
|
+
item = NavigationItem(
|
|
865
|
+
title=target.title,
|
|
866
|
+
href=f"text/content.xhtml#{quote(target.target_id, safe='-._~')}",
|
|
867
|
+
)
|
|
868
|
+
while stack and stack[-1][0] >= target.level:
|
|
869
|
+
stack.pop()
|
|
870
|
+
if stack:
|
|
871
|
+
stack[-1][1].children.append(item)
|
|
872
|
+
else:
|
|
873
|
+
roots.append(item)
|
|
874
|
+
stack.append((target.level, item))
|
|
875
|
+
return roots
|
|
876
|
+
return [NavigationItem(title=document_title, href="text/content.xhtml#content-start")]
|
|
877
|
+
|
|
878
|
+
|
|
879
|
+
def _navigation_from_index(block: IndexBlock, anchors: _AnchorRegistry) -> list[NavigationItem]:
|
|
880
|
+
"""从一个 IndexBlock 提取仅包含真实标题目标的层级导航。"""
|
|
881
|
+
result: list[NavigationItem] = []
|
|
882
|
+
last_item: NavigationItem | None = None
|
|
883
|
+
for child in block.content:
|
|
884
|
+
if isinstance(child, IndexBlock):
|
|
885
|
+
nested = _navigation_from_index(child, anchors)
|
|
886
|
+
if last_item is not None:
|
|
887
|
+
last_item.children.extend(nested)
|
|
888
|
+
else:
|
|
889
|
+
result.extend(nested)
|
|
890
|
+
continue
|
|
891
|
+
target = anchors.target_for_anchor(child.anchor)
|
|
892
|
+
title = inline_plain_text(strip_index_page_tail(child.content)).strip()
|
|
893
|
+
if not target or not title:
|
|
894
|
+
continue
|
|
895
|
+
last_item = NavigationItem(
|
|
896
|
+
title=title,
|
|
897
|
+
href=f"text/content.xhtml#{quote(target, safe='-._~')}",
|
|
898
|
+
)
|
|
899
|
+
result.append(last_item)
|
|
900
|
+
return result
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
def _classify_list(items: list[ListItem], add_reference_bullets: bool) -> tuple[str, str | None, str]:
|
|
904
|
+
"""根据直属 marker 选择原生列表类型或显式 marker 模式。"""
|
|
905
|
+
if add_reference_bullets:
|
|
906
|
+
return "ul", None, "docvortex-list--reference"
|
|
907
|
+
if items and all(item.kind == "unordered" for item in items):
|
|
908
|
+
return "ul", None, "docvortex-list--unordered"
|
|
909
|
+
if items and all(item.kind == "ordered" for item in items):
|
|
910
|
+
styles = {item.ordered_style for item in items}
|
|
911
|
+
if len(styles) == 1:
|
|
912
|
+
list_type = {
|
|
913
|
+
"lower-alpha": "a",
|
|
914
|
+
"upper-alpha": "A",
|
|
915
|
+
"lower-roman": "i",
|
|
916
|
+
"upper-roman": "I",
|
|
917
|
+
}.get(next(iter(styles)) or "")
|
|
918
|
+
return "ol", list_type, "docvortex-list--ordered"
|
|
919
|
+
if items and all(item.kind == "none" for item in items):
|
|
920
|
+
return "ul", None, "docvortex-list--unmarked"
|
|
921
|
+
return "ul", None, "docvortex-list--explicit"
|
|
922
|
+
|
|
923
|
+
|
|
924
|
+
def _list_item_content(
|
|
925
|
+
item: ListItem,
|
|
926
|
+
add_reference_bullets: bool,
|
|
927
|
+
*,
|
|
928
|
+
explicit_markers: bool,
|
|
929
|
+
) -> tuple[list[InlineSpan], str | None]:
|
|
930
|
+
"""决定列表项应剥离、保留还是显式显示源 marker。"""
|
|
931
|
+
if add_reference_bullets:
|
|
932
|
+
if item.kind == "unordered":
|
|
933
|
+
return item.body, None
|
|
934
|
+
prefix = f"{item.leading}{item.marker or ''}{item.separator}"
|
|
935
|
+
original = normalize_inline_spans([TextSpan(type="text", content=prefix), *item.body]) if prefix else item.body
|
|
936
|
+
return original, None
|
|
937
|
+
if explicit_markers:
|
|
938
|
+
return item.body, item.marker
|
|
939
|
+
if item.kind in {"unordered", "ordered"}:
|
|
940
|
+
return item.body, None
|
|
941
|
+
return item.body, item.marker
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
def _append_text_style_container(parent: etree._Element, span: TextSpan) -> etree._Element:
|
|
945
|
+
"""按固定样式顺序创建 TextSpan 的 XHTML 包装节点。"""
|
|
946
|
+
target = parent
|
|
947
|
+
if _needs_whitespace_preservation(span.content):
|
|
948
|
+
target = etree.SubElement(target, _xhtml("span"), attrib={"class": "docvortex-preserve-whitespace"})
|
|
949
|
+
wrappers: list[tuple[str, dict[str, str]]] = []
|
|
950
|
+
if "emphasis" in span.styles:
|
|
951
|
+
wrappers.append(("span", {"class": "docvortex-text-emphasis"}))
|
|
952
|
+
if "strikethrough" in span.styles:
|
|
953
|
+
wrappers.append(("s", {}))
|
|
954
|
+
if "italic" in span.styles:
|
|
955
|
+
wrappers.append(("em", {}))
|
|
956
|
+
if "bold" in span.styles:
|
|
957
|
+
wrappers.append(("strong", {}))
|
|
958
|
+
if "underline" in span.styles:
|
|
959
|
+
wrappers.append(("u", {}))
|
|
960
|
+
if "superscript" in span.styles:
|
|
961
|
+
wrappers.append(("sup", {}))
|
|
962
|
+
elif "subscript" in span.styles:
|
|
963
|
+
wrappers.append(("sub", {}))
|
|
964
|
+
for tag, attributes in wrappers:
|
|
965
|
+
target = etree.SubElement(target, _xhtml(tag), attrib=attributes)
|
|
966
|
+
return target
|
|
967
|
+
|
|
968
|
+
|
|
969
|
+
def _append_text(parent: etree._Element, content: str, *, preserve_newlines: bool = False) -> None:
|
|
970
|
+
"""向 mixed content 追加安全文本,并按需把换行转换为 br。"""
|
|
971
|
+
normalized = _normalize_xml_text(content).replace("\r\n", "\n").replace("\r", "\n")
|
|
972
|
+
if preserve_newlines:
|
|
973
|
+
_append_raw_text(parent, normalized)
|
|
974
|
+
return
|
|
975
|
+
parts = normalized.split("\n")
|
|
976
|
+
for position, part in enumerate(parts):
|
|
977
|
+
if position:
|
|
978
|
+
parent.append(etree.Element(_xhtml("br")))
|
|
979
|
+
_append_raw_text(parent, part)
|
|
980
|
+
|
|
981
|
+
|
|
982
|
+
def _append_raw_text(parent: etree._Element, content: str) -> None:
|
|
983
|
+
"""在不破坏既有子节点 tail 的前提下追加一段普通文本。"""
|
|
984
|
+
if not content:
|
|
985
|
+
return
|
|
986
|
+
if len(parent):
|
|
987
|
+
child = parent[-1]
|
|
988
|
+
child.tail = f"{child.tail or ''}{content}"
|
|
989
|
+
else:
|
|
990
|
+
parent.text = f"{parent.text or ''}{content}"
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
def _safe_markup_attributes(name: str, tag: Tag) -> dict[str, str]:
|
|
994
|
+
"""只保留表格和列表语义需要的有界属性。"""
|
|
995
|
+
attributes: dict[str, str] = {}
|
|
996
|
+
if name in {"td", "th"}:
|
|
997
|
+
for attribute in ("colspan", "rowspan"):
|
|
998
|
+
if value := _bounded_integer(_attribute_text(tag.get(attribute)), minimum=1, maximum=1000):
|
|
999
|
+
attributes[attribute] = value
|
|
1000
|
+
scope = _attribute_text(tag.get("scope"))
|
|
1001
|
+
if name == "th" and scope in {"col", "colgroup", "row", "rowgroup"}:
|
|
1002
|
+
attributes["scope"] = scope
|
|
1003
|
+
elif name in {"col", "colgroup"}:
|
|
1004
|
+
if value := _bounded_integer(_attribute_text(tag.get("span")), minimum=1, maximum=1000):
|
|
1005
|
+
attributes["span"] = value
|
|
1006
|
+
elif name == "ol":
|
|
1007
|
+
if value := _bounded_integer(_attribute_text(tag.get("start")), minimum=-1_000_000, maximum=1_000_000):
|
|
1008
|
+
attributes["start"] = value
|
|
1009
|
+
elif name == "li":
|
|
1010
|
+
if value := _bounded_integer(_attribute_text(tag.get("value")), minimum=-1_000_000, maximum=1_000_000):
|
|
1011
|
+
attributes["value"] = value
|
|
1012
|
+
return attributes
|
|
1013
|
+
|
|
1014
|
+
|
|
1015
|
+
def _bounded_integer(value: str, *, minimum: int, maximum: int) -> str | None:
|
|
1016
|
+
"""把十进制属性约束到 EPUB renderer 支持的闭区间。"""
|
|
1017
|
+
if re.fullmatch(r"[+-]?\d+", value) is None:
|
|
1018
|
+
return None
|
|
1019
|
+
number = int(value)
|
|
1020
|
+
return str(number) if minimum <= number <= maximum else None
|
|
1021
|
+
|
|
1022
|
+
|
|
1023
|
+
def _attribute_text(value: object) -> str:
|
|
1024
|
+
"""把 BeautifulSoup 属性值稳定转换为普通字符串。"""
|
|
1025
|
+
if value is None:
|
|
1026
|
+
return ""
|
|
1027
|
+
if isinstance(value, list):
|
|
1028
|
+
return " ".join(str(item) for item in value)
|
|
1029
|
+
return str(value)
|
|
1030
|
+
|
|
1031
|
+
|
|
1032
|
+
def _is_supported_markup(content: str) -> bool:
|
|
1033
|
+
"""仅把白名单或需整段删除的活动标签识别为富 HTML。"""
|
|
1034
|
+
if "<" not in content or ">" not in content:
|
|
1035
|
+
return False
|
|
1036
|
+
tokens = list(_MARKUP_TOKEN_RE.finditer(content))
|
|
1037
|
+
closing_names = {
|
|
1038
|
+
match.group("name").lower()
|
|
1039
|
+
for match in tokens
|
|
1040
|
+
if match.group("closing") and match.group("name").lower() in _SOURCE_MARKUP_TAGS
|
|
1041
|
+
}
|
|
1042
|
+
for match in tokens:
|
|
1043
|
+
if match.group("closing"):
|
|
1044
|
+
continue
|
|
1045
|
+
name = match.group("name").lower()
|
|
1046
|
+
if name not in _SOURCE_MARKUP_TAGS:
|
|
1047
|
+
continue
|
|
1048
|
+
if name in _VOID_MARKUP_TAGS or name in closing_names:
|
|
1049
|
+
return True
|
|
1050
|
+
if name in {"img", "embed"} and re.search(r"\bsrc\s*=", match.group("attrs"), re.IGNORECASE):
|
|
1051
|
+
return True
|
|
1052
|
+
return False
|
|
1053
|
+
|
|
1054
|
+
|
|
1055
|
+
def _resolve_document_title(middle_json: MiddleJson, explicit_title: str | None) -> str:
|
|
1056
|
+
"""按显式值、首个文档标题和固定回退值解析书名。"""
|
|
1057
|
+
if explicit_title:
|
|
1058
|
+
return _normalize_xml_text(explicit_title).strip()
|
|
1059
|
+
for page in middle_json.pages:
|
|
1060
|
+
for block in page.blocks:
|
|
1061
|
+
if isinstance(block, DocTitleBlock):
|
|
1062
|
+
title = _normalize_xml_text(inline_plain_text(block.content)).strip()
|
|
1063
|
+
if title:
|
|
1064
|
+
return title
|
|
1065
|
+
return "DocVortex Document"
|
|
1066
|
+
|
|
1067
|
+
|
|
1068
|
+
def _stable_identifier(middle_json: MiddleJson, *, title: str, authors: tuple[str, ...], language: str) -> str:
|
|
1069
|
+
"""由规范化 MiddleJson 和不随渲染时间变化的元数据生成稳定 UUID URN。"""
|
|
1070
|
+
seed = json.dumps(
|
|
1071
|
+
{
|
|
1072
|
+
"middle_json": middle_json.model_dump(mode="json"),
|
|
1073
|
+
"title": title,
|
|
1074
|
+
"authors": authors,
|
|
1075
|
+
"language": language,
|
|
1076
|
+
},
|
|
1077
|
+
ensure_ascii=False,
|
|
1078
|
+
sort_keys=True,
|
|
1079
|
+
separators=(",", ":"),
|
|
1080
|
+
).encode("utf-8")
|
|
1081
|
+
digest = hashlib.sha256(seed).hexdigest()
|
|
1082
|
+
return f"urn:uuid:{uuid5(NAMESPACE_URL, digest)}"
|
|
1083
|
+
|
|
1084
|
+
|
|
1085
|
+
def _allocate_target_id(anchor: str | None, *, fallback: str, used_ids: set[str]) -> str:
|
|
1086
|
+
"""从 producer anchor 或固定回退值分配无空白且不碰撞的 id。"""
|
|
1087
|
+
base = _safe_id_base(_anchor_key(anchor)) or fallback
|
|
1088
|
+
candidate = base
|
|
1089
|
+
suffix = 2
|
|
1090
|
+
while candidate in used_ids:
|
|
1091
|
+
candidate = f"{base}-{suffix}"
|
|
1092
|
+
suffix += 1
|
|
1093
|
+
used_ids.add(candidate)
|
|
1094
|
+
return candidate
|
|
1095
|
+
|
|
1096
|
+
|
|
1097
|
+
def _safe_id_base(value: str) -> str:
|
|
1098
|
+
"""把 anchor 归一化为适合 XHTML fragment 的稳定 id 基值。"""
|
|
1099
|
+
normalized = _normalize_xml_text(value).strip()
|
|
1100
|
+
normalized = re.sub(r"\s+", "-", normalized)
|
|
1101
|
+
normalized = re.sub(r"[^\w.:-]+", "-", normalized, flags=re.UNICODE).strip("-")
|
|
1102
|
+
return normalized
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _anchor_key(anchor: str | None) -> str:
|
|
1106
|
+
"""保留 producer anchor 身份,仅去除首尾空白。"""
|
|
1107
|
+
return (anchor or "").strip()
|
|
1108
|
+
|
|
1109
|
+
|
|
1110
|
+
def _plain_content_text(content: str) -> str:
|
|
1111
|
+
"""从 body 内容提取图片 alt 所需的可见纯文本。"""
|
|
1112
|
+
if not content:
|
|
1113
|
+
return ""
|
|
1114
|
+
if _is_supported_markup(content):
|
|
1115
|
+
return BeautifulSoup(content, "html.parser").get_text(" ", strip=True)
|
|
1116
|
+
return _normalize_xml_text(content).strip()
|
|
1117
|
+
|
|
1118
|
+
|
|
1119
|
+
def _normalize_code_language(language: str | None) -> str | None:
|
|
1120
|
+
"""把代码语言限制为不会构造危险 class token 的短名称。"""
|
|
1121
|
+
normalized = (language or "").strip().lower().replace("_", "-")
|
|
1122
|
+
return normalized if _SAFE_LANGUAGE_RE.fullmatch(normalized) else None
|
|
1123
|
+
|
|
1124
|
+
|
|
1125
|
+
def _normalize_xml_text(content: str) -> str:
|
|
1126
|
+
"""替换 XML 1.0 禁止的控制字符和孤立 surrogate。"""
|
|
1127
|
+
return _INVALID_XML_TEXT_RE.sub("\ufffd", content)
|
|
1128
|
+
|
|
1129
|
+
|
|
1130
|
+
def _needs_whitespace_preservation(content: str) -> bool:
|
|
1131
|
+
"""判断文本是否含有 XHTML 默认会折叠的有效空白。"""
|
|
1132
|
+
return bool(content and (content != content.strip(" \t\n") or " " in content or "\t" in content or "\n" in content))
|
|
1133
|
+
|
|
1134
|
+
|
|
1135
|
+
def _has_visible_content(element: etree._Element) -> bool:
|
|
1136
|
+
"""判断元素是否包含可见文本或媒体、结构子节点。"""
|
|
1137
|
+
if element.text and element.text.strip():
|
|
1138
|
+
return True
|
|
1139
|
+
if len(element):
|
|
1140
|
+
return True
|
|
1141
|
+
return False
|
|
1142
|
+
|
|
1143
|
+
|
|
1144
|
+
def _xhtml(tag: str) -> str:
|
|
1145
|
+
"""返回 XHTML namespace 下的 Clark notation 标签名。"""
|
|
1146
|
+
return f"{{{_XHTML_NS}}}{tag}"
|
|
1147
|
+
|
|
1148
|
+
|
|
1149
|
+
@lru_cache(maxsize=1)
|
|
1150
|
+
def _load_epub_stylesheet() -> bytes:
|
|
1151
|
+
"""读取随包分发的静态 EPUB 样式表并缓存字节。"""
|
|
1152
|
+
root = resources.files("docvortex").joinpath("resources", "epub")
|
|
1153
|
+
return root.joinpath(_STYLE_RESOURCE_NAME).read_bytes()
|
|
1154
|
+
|
|
1155
|
+
|
|
1156
|
+
__all__ = ["render_epub"]
|