docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from .api import render
|
|
2
|
+
from .contracts import (
|
|
3
|
+
AssetResolver,
|
|
4
|
+
DocxRenderOptions,
|
|
5
|
+
EpubRenderOptions,
|
|
6
|
+
HtmlRenderOptions,
|
|
7
|
+
ImageRenderer,
|
|
8
|
+
LatexRenderOptions,
|
|
9
|
+
MarkdownRenderOptions,
|
|
10
|
+
PdfRenderOptions,
|
|
11
|
+
RenderFormat,
|
|
12
|
+
RenderMode,
|
|
13
|
+
RenderOptions,
|
|
14
|
+
RenderOutput,
|
|
15
|
+
StructuredContentRenderOptions,
|
|
16
|
+
)
|
|
17
|
+
from .docx import DocxRenderError, render_docx
|
|
18
|
+
from .epub import render_epub
|
|
19
|
+
from .html import render_html
|
|
20
|
+
from .latex import render_latex
|
|
21
|
+
from .markdown import render_markdown
|
|
22
|
+
from .pdf import render_pdf
|
|
23
|
+
from .structured_content import render_structured_content
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"AssetResolver",
|
|
27
|
+
"DocxRenderError",
|
|
28
|
+
"DocxRenderOptions",
|
|
29
|
+
"EpubRenderOptions",
|
|
30
|
+
"HtmlRenderOptions",
|
|
31
|
+
"ImageRenderer",
|
|
32
|
+
"LatexRenderOptions",
|
|
33
|
+
"MarkdownRenderOptions",
|
|
34
|
+
"PdfRenderOptions",
|
|
35
|
+
"RenderFormat",
|
|
36
|
+
"RenderMode",
|
|
37
|
+
"RenderOptions",
|
|
38
|
+
"RenderOutput",
|
|
39
|
+
"StructuredContentRenderOptions",
|
|
40
|
+
"render",
|
|
41
|
+
"render_docx",
|
|
42
|
+
"render_epub",
|
|
43
|
+
"render_html",
|
|
44
|
+
"render_latex",
|
|
45
|
+
"render_markdown",
|
|
46
|
+
"render_pdf",
|
|
47
|
+
"render_structured_content",
|
|
48
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""在一次内部渲染调用中传递独占副本,不缓存用户的可变文档。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from contextlib import contextmanager
|
|
6
|
+
from contextvars import ContextVar
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Iterator
|
|
9
|
+
|
|
10
|
+
from ....schema import MiddleJson
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(slots=True)
|
|
14
|
+
class _OwnedRenderDocument:
|
|
15
|
+
"""只允许一个规划器消费当前调用已经隔离的文档。"""
|
|
16
|
+
|
|
17
|
+
document: MiddleJson
|
|
18
|
+
consumed: bool = False
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
_owned_document: ContextVar[_OwnedRenderDocument | None] = ContextVar("docvortex_owned_render_document", default=None)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@contextmanager
|
|
25
|
+
def owned_render_document(document: MiddleJson) -> Iterator[None]:
|
|
26
|
+
"""限定独占文档的调用范围,异常和重入后均恢复外层上下文。"""
|
|
27
|
+
token = _owned_document.set(_OwnedRenderDocument(document))
|
|
28
|
+
try:
|
|
29
|
+
yield
|
|
30
|
+
finally:
|
|
31
|
+
_owned_document.reset(token)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def claim_owned_document(document: MiddleJson) -> bool:
|
|
35
|
+
"""仅复用当前调用的精确对象一次,回调重入必须重新隔离。"""
|
|
36
|
+
context = _owned_document.get()
|
|
37
|
+
if context is None or context.document is not document or context.consumed:
|
|
38
|
+
return False
|
|
39
|
+
context.consumed = True
|
|
40
|
+
return True
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
__all__ = ["owned_render_document", "claim_owned_document"]
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""多格式 renderer 共用的有界 HTML table 占位网格解析。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import re
|
|
7
|
+
from typing import TypeAlias
|
|
8
|
+
|
|
9
|
+
from bs4 import BeautifulSoup, Tag
|
|
10
|
+
|
|
11
|
+
MAX_NESTED_TABLE_DEPTH = 4
|
|
12
|
+
MAX_TABLE_ROWS = 500
|
|
13
|
+
MAX_TABLE_COLUMNS = 100
|
|
14
|
+
MAX_TABLE_SLOTS = 10_000
|
|
15
|
+
_POSITIVE_INTEGER_RE = re.compile(r"[0-9]+")
|
|
16
|
+
|
|
17
|
+
HtmlTableSource: TypeAlias = str | BeautifulSoup | Tag
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class HtmlTableError(ValueError):
|
|
21
|
+
"""表示 HTML table 无法安全解析为严格矩形网格。"""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True, slots=True)
|
|
25
|
+
class HtmlTableCell:
|
|
26
|
+
"""保存原始 HTML 单元格在逻辑占位网格中的位置。"""
|
|
27
|
+
|
|
28
|
+
tag: Tag
|
|
29
|
+
row: int
|
|
30
|
+
column: int
|
|
31
|
+
rowspan: int
|
|
32
|
+
colspan: int
|
|
33
|
+
is_header: bool
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def end_row(self) -> int:
|
|
37
|
+
"""返回单元格占用的末行下标。"""
|
|
38
|
+
return self.row + self.rowspan - 1
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def end_column(self) -> int:
|
|
42
|
+
"""返回单元格占用的末列下标。"""
|
|
43
|
+
return self.column + self.colspan - 1
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True, slots=True)
|
|
47
|
+
class HtmlTableGrid:
|
|
48
|
+
"""保存经过重叠、边界、规模与矩形校验的 HTML 表格网格。"""
|
|
49
|
+
|
|
50
|
+
tag: Tag
|
|
51
|
+
row_count: int
|
|
52
|
+
column_count: int
|
|
53
|
+
cells: tuple[HtmlTableCell, ...]
|
|
54
|
+
header_rows: tuple[int, ...]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def parse_html_tables(source: HtmlTableSource) -> tuple[HtmlTableGrid, ...]:
|
|
58
|
+
"""解析 source 中相对当前上下文的一个或多个顶层 table。"""
|
|
59
|
+
root = BeautifulSoup(source, "html.parser") if isinstance(source, str) else source
|
|
60
|
+
if not isinstance(root, (BeautifulSoup, Tag)):
|
|
61
|
+
raise HtmlTableError("HTML table source must be a string or BeautifulSoup Tag")
|
|
62
|
+
if isinstance(root, Tag) and root.name == "table":
|
|
63
|
+
table_tags = (root,)
|
|
64
|
+
else:
|
|
65
|
+
parent_table = root.find_parent("table") if isinstance(root, Tag) else None
|
|
66
|
+
table_tags = tuple(table for table in root.find_all("table") if table.find_parent("table") is parent_table)
|
|
67
|
+
if not table_tags:
|
|
68
|
+
raise HtmlTableError("HTML does not contain a top-level table")
|
|
69
|
+
return tuple(_parse_html_table(table) for table in table_tags)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _parse_html_table(table: Tag) -> HtmlTableGrid:
|
|
73
|
+
"""把单个 table 标签解析为严格矩形占位网格。"""
|
|
74
|
+
if table.name != "table":
|
|
75
|
+
raise HtmlTableError("Expected a <table> tag")
|
|
76
|
+
rows = tuple(row for row in table.find_all("tr") if row.find_parent("table") is table)
|
|
77
|
+
if not rows or len(rows) > MAX_TABLE_ROWS:
|
|
78
|
+
raise HtmlTableError(f"Table row count must be between 1 and {MAX_TABLE_ROWS}")
|
|
79
|
+
|
|
80
|
+
occupied: dict[tuple[int, int], HtmlTableCell] = {}
|
|
81
|
+
cells: list[HtmlTableCell] = []
|
|
82
|
+
for row_index, row in enumerate(rows):
|
|
83
|
+
column_index = 0
|
|
84
|
+
for source_cell in row.find_all(("td", "th"), recursive=False):
|
|
85
|
+
while (row_index, column_index) in occupied:
|
|
86
|
+
column_index += 1
|
|
87
|
+
rowspan = _parse_span(source_cell, "rowspan")
|
|
88
|
+
colspan = _parse_span(source_cell, "colspan")
|
|
89
|
+
if row_index + rowspan > len(rows):
|
|
90
|
+
raise HtmlTableError(f"rowspan exceeds table bounds at row={row_index}, column={column_index}")
|
|
91
|
+
if column_index + colspan > MAX_TABLE_COLUMNS:
|
|
92
|
+
raise HtmlTableError(f"Table column count exceeds {MAX_TABLE_COLUMNS}")
|
|
93
|
+
coordinates = tuple(
|
|
94
|
+
(target_row, target_column)
|
|
95
|
+
for target_row in range(row_index, row_index + rowspan)
|
|
96
|
+
for target_column in range(column_index, column_index + colspan)
|
|
97
|
+
)
|
|
98
|
+
overlap = next((coordinate for coordinate in coordinates if coordinate in occupied), None)
|
|
99
|
+
if overlap is not None:
|
|
100
|
+
raise HtmlTableError(f"Cell span overlaps row={overlap[0]}, column={overlap[1]}")
|
|
101
|
+
placement = HtmlTableCell(
|
|
102
|
+
tag=source_cell,
|
|
103
|
+
row=row_index,
|
|
104
|
+
column=column_index,
|
|
105
|
+
rowspan=rowspan,
|
|
106
|
+
colspan=colspan,
|
|
107
|
+
is_header=source_cell.name == "th",
|
|
108
|
+
)
|
|
109
|
+
cells.append(placement)
|
|
110
|
+
occupied.update(dict.fromkeys(coordinates, placement))
|
|
111
|
+
if len(occupied) > MAX_TABLE_SLOTS:
|
|
112
|
+
raise HtmlTableError(f"Table occupancy exceeds {MAX_TABLE_SLOTS} slots")
|
|
113
|
+
column_index += colspan
|
|
114
|
+
|
|
115
|
+
if not occupied:
|
|
116
|
+
raise HtmlTableError("Table must contain at least one cell")
|
|
117
|
+
column_count = max(column for _, column in occupied) + 1
|
|
118
|
+
missing = next(
|
|
119
|
+
(
|
|
120
|
+
(row_index, column_index)
|
|
121
|
+
for row_index in range(len(rows))
|
|
122
|
+
for column_index in range(column_count)
|
|
123
|
+
if (row_index, column_index) not in occupied
|
|
124
|
+
),
|
|
125
|
+
None,
|
|
126
|
+
)
|
|
127
|
+
if missing is not None:
|
|
128
|
+
raise HtmlTableError(f"Table occupancy is not rectangular at row={missing[0]}, column={missing[1]}")
|
|
129
|
+
header_rows = tuple(
|
|
130
|
+
row_index
|
|
131
|
+
for row_index, row in enumerate(rows)
|
|
132
|
+
if _row_belongs_to_thead(row, table)
|
|
133
|
+
or all(occupied[(row_index, column_index)].is_header for column_index in range(column_count))
|
|
134
|
+
)
|
|
135
|
+
return HtmlTableGrid(
|
|
136
|
+
tag=table,
|
|
137
|
+
row_count=len(rows),
|
|
138
|
+
column_count=column_count,
|
|
139
|
+
cells=tuple(cells),
|
|
140
|
+
header_rows=header_rows,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _parse_span(cell: Tag, attribute: str) -> int:
|
|
145
|
+
"""读取严格正整数 rowspan/colspan,缺失时返回一。"""
|
|
146
|
+
raw_value = cell.get(attribute, "1")
|
|
147
|
+
if isinstance(raw_value, list):
|
|
148
|
+
raise HtmlTableError(f"Invalid {attribute}: {raw_value!r}")
|
|
149
|
+
value = str(raw_value).strip()
|
|
150
|
+
if _POSITIVE_INTEGER_RE.fullmatch(value) is None:
|
|
151
|
+
raise HtmlTableError(f"Invalid {attribute}: {raw_value!r}")
|
|
152
|
+
span = int(value)
|
|
153
|
+
if span < 1 or span > MAX_TABLE_SLOTS:
|
|
154
|
+
raise HtmlTableError(f"Invalid {attribute}: {raw_value!r}")
|
|
155
|
+
return span
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _row_belongs_to_thead(row: Tag, table: Tag) -> bool:
|
|
159
|
+
"""判断 tr 是否位于当前 table 的 thead 内。"""
|
|
160
|
+
parent = row.parent
|
|
161
|
+
while isinstance(parent, Tag) and parent is not table:
|
|
162
|
+
if parent.name == "thead":
|
|
163
|
+
return True
|
|
164
|
+
parent = parent.parent
|
|
165
|
+
return False
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
__all__ = [
|
|
169
|
+
"HtmlTableCell",
|
|
170
|
+
"HtmlTableError",
|
|
171
|
+
"HtmlTableGrid",
|
|
172
|
+
"HtmlTableSource",
|
|
173
|
+
"MAX_NESTED_TABLE_DEPTH",
|
|
174
|
+
"MAX_TABLE_COLUMNS",
|
|
175
|
+
"MAX_TABLE_ROWS",
|
|
176
|
+
"MAX_TABLE_SLOTS",
|
|
177
|
+
"parse_html_tables",
|
|
178
|
+
]
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""各格式共用的目录页码尾部识别与清理。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from ....content.inline import inline_plain_text, map_text_span_content, normalize_inline_spans, slice_inline_spans
|
|
8
|
+
from ....schema import InlineSpan
|
|
9
|
+
|
|
10
|
+
_INDEX_ROMAN_RE = re.compile(r"[ivxlcdm]+", re.IGNORECASE)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def strip_index_page_tail(content: list[InlineSpan]) -> list[InlineSpan]:
|
|
14
|
+
"""删除目录末尾可信页码,并把其余 tab 转换为普通空格。"""
|
|
15
|
+
content = normalize_inline_spans(content)
|
|
16
|
+
visible_text = inline_plain_text(content)
|
|
17
|
+
if "\t" not in visible_text:
|
|
18
|
+
return content
|
|
19
|
+
tab_offset = visible_text.rfind("\t")
|
|
20
|
+
tail_text = visible_text[tab_offset + 1 :].strip()
|
|
21
|
+
if looks_like_index_page_token(tail_text):
|
|
22
|
+
content = slice_inline_spans(content, 0, tab_offset)
|
|
23
|
+
return map_text_span_content(content, lambda value: value.replace("\t", " "))
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def looks_like_index_page_token(content: str) -> bool:
|
|
27
|
+
"""判断目录 tab 后缀是否为数字、罗马数字或单字母页码。"""
|
|
28
|
+
if not content or len(content) > 12:
|
|
29
|
+
return False
|
|
30
|
+
return bool(content.isdigit() or _INDEX_ROMAN_RE.fullmatch(content) or re.fullmatch(r"[A-Za-z]", content))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
__all__ = ["looks_like_index_page_token", "strip_index_page_tail"]
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""各格式共用的列表 marker 解析与参考文献判定。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
import re
|
|
7
|
+
from typing import Literal, TypeAlias
|
|
8
|
+
|
|
9
|
+
from ....content.inline import inline_plain_text, normalize_inline_spans, slice_inline_spans
|
|
10
|
+
from ....schema import BlockType, InlineSpan, ListBlock
|
|
11
|
+
|
|
12
|
+
ListItemKind: TypeAlias = Literal["unordered", "ordered", "explicit", "none"]
|
|
13
|
+
OrderedListStyle: TypeAlias = Literal["decimal", "lower-alpha", "upper-alpha", "lower-roman", "upper-roman"]
|
|
14
|
+
|
|
15
|
+
_LIST_ITEM_MARKER_RE = re.compile(
|
|
16
|
+
r"^(?P<leading>\s*)(?P<marker>"
|
|
17
|
+
r"(?P<unordered>[-*+])"
|
|
18
|
+
r"|(?P<ordered>\d+\.|[A-Za-z]\.|[IVXLCDMivxlcdm]{2,}\.)"
|
|
19
|
+
r"|(?P<explicit>\d+\)|\(\d+[.)]|[A-Za-z]\)|[IVXLCDMivxlcdm]{2,}\)|\[[^\]\n]+\])"
|
|
20
|
+
r")(?P<separator>\s+)(?P<body>.*)$",
|
|
21
|
+
re.DOTALL,
|
|
22
|
+
)
|
|
23
|
+
_LEADING_WHITESPACE_RE = re.compile(r"^[ \t]*")
|
|
24
|
+
# 去除首部空白后,前五个可见字符内出现 Unicode 数字即视为单项命中。
|
|
25
|
+
_REFERENCE_NUMBER_PREFIX_RE = re.compile(r"^\D{0,4}\d")
|
|
26
|
+
_MARKDOWN_UNORDERED_MARKER_RE = re.compile(r"^[ \t]*-[ \t]+")
|
|
27
|
+
_ROMAN_MARKER_RE = re.compile(r"[IVXLCDM]+", re.IGNORECASE)
|
|
28
|
+
_CANONICAL_ROMAN_RE = re.compile(r"M{0,3}(?:CM|CD|D?C{0,3})(?:XC|XL|L?X{0,3})(?:IX|IV|V?I{0,3})")
|
|
29
|
+
_ROMAN_VALUES = {"I": 1, "V": 5, "X": 10, "L": 50, "C": 100, "D": 500, "M": 1000}
|
|
30
|
+
_MAX_NATIVE_ORDERED_VALUE = 1_000_000
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True, slots=True)
|
|
34
|
+
class ListItem:
|
|
35
|
+
"""保存一个列表条目的原始标记、正文与 HTML 所需分类。"""
|
|
36
|
+
|
|
37
|
+
marker: str | None
|
|
38
|
+
body: list[InlineSpan]
|
|
39
|
+
kind: ListItemKind
|
|
40
|
+
value: int | None
|
|
41
|
+
ordered_style: OrderedListStyle | None
|
|
42
|
+
leading: str
|
|
43
|
+
separator: str
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def parse_list_item_marker(content: list[InlineSpan]) -> ListItem:
|
|
47
|
+
"""解析列表行首 marker;无法识别时仍拆出前导水平空白。"""
|
|
48
|
+
content = normalize_inline_spans(content)
|
|
49
|
+
visible_text = inline_plain_text(content)
|
|
50
|
+
match = _LIST_ITEM_MARKER_RE.match(visible_text)
|
|
51
|
+
if match is None:
|
|
52
|
+
leading_match = _LEADING_WHITESPACE_RE.match(visible_text)
|
|
53
|
+
leading = leading_match.group(0) if leading_match is not None else ""
|
|
54
|
+
return ListItem(
|
|
55
|
+
marker=None,
|
|
56
|
+
body=slice_inline_spans(content, len(leading)),
|
|
57
|
+
kind="none",
|
|
58
|
+
value=None,
|
|
59
|
+
ordered_style=None,
|
|
60
|
+
leading=leading,
|
|
61
|
+
separator="",
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
marker = match.group("marker")
|
|
65
|
+
if match.group("unordered") is not None:
|
|
66
|
+
kind: ListItemKind = "unordered"
|
|
67
|
+
value = None
|
|
68
|
+
ordered_style = None
|
|
69
|
+
elif match.group("ordered") is not None:
|
|
70
|
+
ordered = _ordered_marker_value(marker)
|
|
71
|
+
if ordered is None:
|
|
72
|
+
kind = "explicit"
|
|
73
|
+
value = None
|
|
74
|
+
ordered_style = None
|
|
75
|
+
else:
|
|
76
|
+
kind = "ordered"
|
|
77
|
+
value, ordered_style = ordered
|
|
78
|
+
else:
|
|
79
|
+
kind = "explicit"
|
|
80
|
+
value = None
|
|
81
|
+
ordered_style = None
|
|
82
|
+
return ListItem(
|
|
83
|
+
marker=marker,
|
|
84
|
+
body=slice_inline_spans(content, match.start("body")),
|
|
85
|
+
kind=kind,
|
|
86
|
+
value=value,
|
|
87
|
+
ordered_style=ordered_style,
|
|
88
|
+
leading=match.group("leading"),
|
|
89
|
+
separator=match.group("separator"),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _ordered_marker_value(marker: str) -> tuple[int, OrderedListStyle] | None:
|
|
94
|
+
"""把有界且规范的点号 marker 转成序号;超限或畸形时返回 None。"""
|
|
95
|
+
stem = marker[:-1]
|
|
96
|
+
if re.fullmatch(r"(?:0|[1-9][0-9]*)", stem):
|
|
97
|
+
if len(stem) > 7:
|
|
98
|
+
return None
|
|
99
|
+
value = int(stem)
|
|
100
|
+
return (value, "decimal") if value <= _MAX_NATIVE_ORDERED_VALUE else None
|
|
101
|
+
# 单字符 i/v/x/l/c/d/m 固定按罗马数字解释,消除与字母序号的歧义。
|
|
102
|
+
if _ROMAN_MARKER_RE.fullmatch(stem):
|
|
103
|
+
if _CANONICAL_ROMAN_RE.fullmatch(stem.upper()) is None:
|
|
104
|
+
return None
|
|
105
|
+
style: OrderedListStyle = "upper-roman" if stem.isupper() else "lower-roman"
|
|
106
|
+
return _roman_marker_value(stem), style
|
|
107
|
+
if re.fullmatch(r"[A-Za-z]", stem) is None:
|
|
108
|
+
return None
|
|
109
|
+
style = "upper-alpha" if stem.isupper() else "lower-alpha"
|
|
110
|
+
return ord(stem.lower()) - ord("a") + 1, style
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _roman_marker_value(marker: str) -> int:
|
|
114
|
+
"""按减法记数规则计算罗马 marker 的数值,兼容 producer 的宽松组合。"""
|
|
115
|
+
total = 0
|
|
116
|
+
previous = 0
|
|
117
|
+
for character in reversed(marker.upper()):
|
|
118
|
+
current = _ROMAN_VALUES[character]
|
|
119
|
+
if current < previous:
|
|
120
|
+
total -= current
|
|
121
|
+
else:
|
|
122
|
+
total += current
|
|
123
|
+
previous = current
|
|
124
|
+
return total
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def has_markdown_unordered_marker(content: list[InlineSpan]) -> bool:
|
|
128
|
+
"""判断条目是否已有 Markdown 短横线 marker,保持既有补 bullet 规则。"""
|
|
129
|
+
return _MARKDOWN_UNORDERED_MARKER_RE.match(inline_plain_text(normalize_inline_spans(content))) is not None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def reference_list_needs_bullets(block: ListBlock) -> bool:
|
|
133
|
+
"""按直属非空条目的数字前缀严格多数规则判断是否补无序 marker。"""
|
|
134
|
+
if block.sub_type != BlockType.REF_TEXT:
|
|
135
|
+
return False
|
|
136
|
+
|
|
137
|
+
item_count = 0
|
|
138
|
+
numbered_count = 0
|
|
139
|
+
for child in block.content:
|
|
140
|
+
if isinstance(child, ListBlock):
|
|
141
|
+
continue
|
|
142
|
+
visible_text = inline_plain_text(child.content).lstrip()
|
|
143
|
+
if not visible_text:
|
|
144
|
+
continue
|
|
145
|
+
item_count += 1
|
|
146
|
+
if _REFERENCE_NUMBER_PREFIX_RE.match(visible_text):
|
|
147
|
+
numbered_count += 1
|
|
148
|
+
return item_count > 0 and numbered_count * 2 <= item_count
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
__all__ = [
|
|
152
|
+
"ListItem",
|
|
153
|
+
"ListItemKind",
|
|
154
|
+
"OrderedListStyle",
|
|
155
|
+
"has_markdown_unordered_marker",
|
|
156
|
+
"parse_list_item_marker",
|
|
157
|
+
"reference_list_needs_bullets",
|
|
158
|
+
]
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""各格式共用的逻辑块复制、延续合并与页面规划。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from .context import claim_owned_document
|
|
8
|
+
|
|
9
|
+
from ....content.table import merge_table_content
|
|
10
|
+
from ...contracts import RenderMode
|
|
11
|
+
from ....schema import (
|
|
12
|
+
MERGE_TRANSPARENT_BLOCK_TYPES,
|
|
13
|
+
BlockType,
|
|
14
|
+
ContinuableTextBlockBase,
|
|
15
|
+
ListBlock,
|
|
16
|
+
MiddleJson,
|
|
17
|
+
PageBlock,
|
|
18
|
+
InlineSpan,
|
|
19
|
+
RefTextBlock,
|
|
20
|
+
TableBlock,
|
|
21
|
+
TextBlock,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(slots=True)
|
|
26
|
+
class PlannedBlock:
|
|
27
|
+
"""保存一个待渲染块及其来源页和文本延续片段。"""
|
|
28
|
+
|
|
29
|
+
page_idx: int
|
|
30
|
+
block: PageBlock
|
|
31
|
+
text_contents: list[list[InlineSpan]] = field(default_factory=list)
|
|
32
|
+
removed: bool = False
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def build_render_plan(
|
|
36
|
+
middle_json: MiddleJson,
|
|
37
|
+
mode: RenderMode = RenderMode.DEFAULT,
|
|
38
|
+
) -> list[list[PlannedBlock]]:
|
|
39
|
+
"""深拷贝 MiddleJson,并按模式生成不污染输入的逐页逻辑块计划。"""
|
|
40
|
+
owned = claim_owned_document(middle_json)
|
|
41
|
+
copied = middle_json if owned else middle_json.model_copy(deep=True)
|
|
42
|
+
pages = [
|
|
43
|
+
[
|
|
44
|
+
PlannedBlock(
|
|
45
|
+
page_idx=page.page_idx,
|
|
46
|
+
block=block,
|
|
47
|
+
text_contents=[block.content] if isinstance(block, ContinuableTextBlockBase) else [],
|
|
48
|
+
)
|
|
49
|
+
for block in page.blocks
|
|
50
|
+
]
|
|
51
|
+
for page in copied.pages
|
|
52
|
+
]
|
|
53
|
+
flattened = [planned for page in pages for planned in page]
|
|
54
|
+
_merge_continued_text_blocks(flattened, mode)
|
|
55
|
+
_merge_continued_list_blocks(flattened, mode, copy_on_merge=owned)
|
|
56
|
+
if mode is RenderMode.DEFAULT:
|
|
57
|
+
_merge_continued_table_blocks(flattened)
|
|
58
|
+
return pages
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _merge_continued_text_blocks(blocks: list[PlannedBlock], mode: RenderMode) -> None:
|
|
62
|
+
"""把无独立正文锚点的 continues_prev 文本吸收到最近的前序文本逻辑块。"""
|
|
63
|
+
previous_text: PlannedBlock | None = None
|
|
64
|
+
previous_reference: PlannedBlock | None = None
|
|
65
|
+
for current in blocks:
|
|
66
|
+
is_text = isinstance(current.block, TextBlock)
|
|
67
|
+
is_reference = isinstance(current.block, RefTextBlock)
|
|
68
|
+
previous = previous_text if is_text else previous_reference if is_reference else None
|
|
69
|
+
anchored = is_text and isinstance(current.block.anchor, str) and bool(current.block.anchor.strip())
|
|
70
|
+
if (
|
|
71
|
+
not current.removed
|
|
72
|
+
and isinstance(current.block, ContinuableTextBlockBase)
|
|
73
|
+
and current.block.continues_prev is True
|
|
74
|
+
and not anchored
|
|
75
|
+
and previous is not None
|
|
76
|
+
and not (mode is RenderMode.FULL and previous.page_idx != current.page_idx)
|
|
77
|
+
):
|
|
78
|
+
previous.text_contents.extend(current.text_contents)
|
|
79
|
+
current.removed = True
|
|
80
|
+
if is_text and not current.removed:
|
|
81
|
+
previous_text = current
|
|
82
|
+
if is_reference:
|
|
83
|
+
if not current.removed:
|
|
84
|
+
previous_reference = current
|
|
85
|
+
elif current.block.type not in MERGE_TRANSPARENT_BLOCK_TYPES:
|
|
86
|
+
previous_reference = None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _merge_continued_list_blocks(blocks: list[PlannedBlock], mode: RenderMode, *, copy_on_merge: bool = False) -> None:
|
|
90
|
+
"""把续接列表吸收到子类型一致的前序列表,参考文献可跨过合并透明块。"""
|
|
91
|
+
previous_list: PlannedBlock | None = None
|
|
92
|
+
previous_reference_list: PlannedBlock | None = None
|
|
93
|
+
copied: set[int] = set()
|
|
94
|
+
for current in blocks:
|
|
95
|
+
if not isinstance(current.block, ListBlock):
|
|
96
|
+
previous_list = None
|
|
97
|
+
if current.block.type not in MERGE_TRANSPARENT_BLOCK_TYPES:
|
|
98
|
+
previous_reference_list = None
|
|
99
|
+
continue
|
|
100
|
+
previous = previous_reference_list if current.block.sub_type == BlockType.REF_TEXT else previous_list
|
|
101
|
+
if (
|
|
102
|
+
not current.removed
|
|
103
|
+
and current.block.continues_prev is True
|
|
104
|
+
and previous is not None
|
|
105
|
+
and previous.block.sub_type == current.block.sub_type
|
|
106
|
+
and not (mode is RenderMode.FULL and previous.page_idx != current.page_idx)
|
|
107
|
+
):
|
|
108
|
+
# EPUB 等渲染器仍读取原文档;只有实际修改的列表需要另建副本。
|
|
109
|
+
if copy_on_merge and id(previous) not in copied:
|
|
110
|
+
previous.block = previous.block.model_copy(deep=True)
|
|
111
|
+
copied.add(id(previous))
|
|
112
|
+
previous.block.content.extend(current.block.content)
|
|
113
|
+
current.removed = True
|
|
114
|
+
if not current.removed:
|
|
115
|
+
previous_list = previous_reference_list = current
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _merge_continued_table_blocks(blocks: list[PlannedBlock]) -> None:
|
|
119
|
+
"""在默认模式中把跨页续表合并到最近的前序表格。"""
|
|
120
|
+
previous: PlannedBlock | None = None
|
|
121
|
+
for current in blocks:
|
|
122
|
+
if current.removed or not isinstance(current.block, TableBlock):
|
|
123
|
+
continue
|
|
124
|
+
if current.block.continues_prev is True and previous is not None and previous.page_idx != current.page_idx:
|
|
125
|
+
merged = merge_table_content(
|
|
126
|
+
previous.block.model_dump(mode="python", exclude_none=True),
|
|
127
|
+
current.block.model_dump(mode="python", exclude_none=True),
|
|
128
|
+
)
|
|
129
|
+
if merged is not None:
|
|
130
|
+
try:
|
|
131
|
+
previous.block = TableBlock.model_validate(merged)
|
|
132
|
+
except (TypeError, ValueError):
|
|
133
|
+
pass
|
|
134
|
+
else:
|
|
135
|
+
current.removed = True
|
|
136
|
+
if not current.removed:
|
|
137
|
+
previous = current
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
__all__ = ["PlannedBlock", "build_render_plan"]
|