docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,734 @@
|
|
|
1
|
+
"""严格 MiddleJson 到 ReportLab PDF bytes 的公共渲染实现。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import partial
|
|
6
|
+
import html
|
|
7
|
+
from io import BytesIO
|
|
8
|
+
import re
|
|
9
|
+
from typing import Any, Iterable
|
|
10
|
+
|
|
11
|
+
from bs4 import BeautifulSoup
|
|
12
|
+
from loguru import logger
|
|
13
|
+
from reportlab.lib.pagesizes import A4
|
|
14
|
+
from reportlab.lib.styles import ParagraphStyle
|
|
15
|
+
from reportlab.lib.units import mm
|
|
16
|
+
from reportlab.pdfgen.canvas import Canvas
|
|
17
|
+
from reportlab.platypus import (
|
|
18
|
+
Flowable,
|
|
19
|
+
Image as ReportLabImage,
|
|
20
|
+
Paragraph,
|
|
21
|
+
SimpleDocTemplate,
|
|
22
|
+
Spacer,
|
|
23
|
+
Table,
|
|
24
|
+
TableStyle,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
from ....content.inline import inline_plain_text, join_inline_spans
|
|
28
|
+
from ..common.index import strip_index_page_tail
|
|
29
|
+
from ..common.list_items import parse_list_item_marker, reference_list_needs_bullets
|
|
30
|
+
from ..common.planner import PlannedBlock, build_render_plan
|
|
31
|
+
from ...contracts import AssetResolver
|
|
32
|
+
from ....schema import (
|
|
33
|
+
PAGE_AUXILIARY_BLOCK_TYPES,
|
|
34
|
+
RAW_ALGORITHM,
|
|
35
|
+
AlgorithmBodyBlock,
|
|
36
|
+
BlockBase,
|
|
37
|
+
BlockType,
|
|
38
|
+
ChartAnnotationBlock,
|
|
39
|
+
ChartBlock,
|
|
40
|
+
ChartBodyBlock,
|
|
41
|
+
CodeAnnotationBlock,
|
|
42
|
+
CodeBlock,
|
|
43
|
+
CodeBodyBlock,
|
|
44
|
+
DocTitleBlock,
|
|
45
|
+
EquationBlock,
|
|
46
|
+
HyperlinkSpan,
|
|
47
|
+
ImageAnnotationBlock,
|
|
48
|
+
ImageBlock,
|
|
49
|
+
ImageBodyBlock,
|
|
50
|
+
ImagePayloadBlock,
|
|
51
|
+
IndexBlock,
|
|
52
|
+
InlineSpan,
|
|
53
|
+
ListBlock,
|
|
54
|
+
MiddleJson,
|
|
55
|
+
NonLinkInlineSpan,
|
|
56
|
+
PageFootnoteBlock,
|
|
57
|
+
ParagraphTitleBlock,
|
|
58
|
+
RefTextBlock,
|
|
59
|
+
TableAnnotationBlock,
|
|
60
|
+
TableBlock,
|
|
61
|
+
TableBodyBlock,
|
|
62
|
+
TextBlock,
|
|
63
|
+
TextSpan,
|
|
64
|
+
TitleBlockBase,
|
|
65
|
+
)
|
|
66
|
+
from .assets import PdfAssetError, PreparedImage, prepare_block_image, prepare_html_image
|
|
67
|
+
from .formula import (
|
|
68
|
+
DisplayFormulaFlowable,
|
|
69
|
+
FormulaRenderer,
|
|
70
|
+
InlineFormulaImage,
|
|
71
|
+
PdfFormulaError,
|
|
72
|
+
draw_inline_formula,
|
|
73
|
+
split_formula_tag,
|
|
74
|
+
)
|
|
75
|
+
from .inline import PdfAnchorRegistry, PdfInlineContext, build_pdf_paragraph, render_plain_text_markup
|
|
76
|
+
from .styles import BORDER_COLOR, PAGE_MARGIN, SURFACE_COLOR, build_pdf_styles
|
|
77
|
+
from .table import PdfTableError, build_pdf_tables
|
|
78
|
+
|
|
79
|
+
_HTML_TABLE_RE = re.compile(r"<table\b", re.IGNORECASE)
|
|
80
|
+
_INVALID_METADATA_TEXT_RE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f\ud800-\udfff\ufffe\uffff]")
|
|
81
|
+
_VISIBLE_HTML_TAGS = (
|
|
82
|
+
"a",
|
|
83
|
+
"b",
|
|
84
|
+
"blockquote",
|
|
85
|
+
"br",
|
|
86
|
+
"code",
|
|
87
|
+
"div",
|
|
88
|
+
"em",
|
|
89
|
+
"eq",
|
|
90
|
+
"i",
|
|
91
|
+
"li",
|
|
92
|
+
"ol",
|
|
93
|
+
"p",
|
|
94
|
+
"pre",
|
|
95
|
+
"span",
|
|
96
|
+
"strong",
|
|
97
|
+
"sub",
|
|
98
|
+
"sup",
|
|
99
|
+
"table",
|
|
100
|
+
"u",
|
|
101
|
+
"ul",
|
|
102
|
+
)
|
|
103
|
+
_MAX_PLACEHOLDER_TEXT = 320
|
|
104
|
+
_MIN_IMAGE_WIDTH = 5 * mm
|
|
105
|
+
_PLACEHOLDER_HEIGHT = 18 * mm
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class _PdfCanvas(Canvas):
|
|
109
|
+
"""提供确定性 metadata 与行内 ZiaMath 矢量绘制的 ReportLab Canvas。"""
|
|
110
|
+
|
|
111
|
+
def __init__(self, filename: Any, *args: Any, document_title: str, **kwargs: Any) -> None:
|
|
112
|
+
"""创建启用压缩和 invariant 的 PDF canvas,并写入稳定 metadata。"""
|
|
113
|
+
kwargs.setdefault("pageCompression", 1)
|
|
114
|
+
kwargs["invariant"] = 1
|
|
115
|
+
super().__init__(filename, *args, **kwargs)
|
|
116
|
+
self.setTitle(document_title)
|
|
117
|
+
self.setAuthor("DocVortex")
|
|
118
|
+
self.setCreator("DocVortex PDF Renderer")
|
|
119
|
+
self.setSubject("Semantic document rendering from MiddleJson")
|
|
120
|
+
self.setKeywords("DocVortex, MiddleJson, PDF")
|
|
121
|
+
|
|
122
|
+
def drawImage(
|
|
123
|
+
self,
|
|
124
|
+
image: Any,
|
|
125
|
+
x: float,
|
|
126
|
+
y: float,
|
|
127
|
+
width: float | None = None,
|
|
128
|
+
height: float | None = None,
|
|
129
|
+
mask: Any = None,
|
|
130
|
+
preserveAspectRatio: bool = False,
|
|
131
|
+
anchor: str = "c",
|
|
132
|
+
anchorAtXY: bool = False,
|
|
133
|
+
showBoundary: bool = False,
|
|
134
|
+
) -> Any:
|
|
135
|
+
"""识别 Paragraph 传入的公式代理,否则沿用标准 raster 图片行为。"""
|
|
136
|
+
if isinstance(image, InlineFormulaImage):
|
|
137
|
+
resolved_width = image.vector.width if width is None else float(width)
|
|
138
|
+
resolved_height = image.vector.height if height is None else float(height)
|
|
139
|
+
return draw_inline_formula(self, image, x, y, resolved_width, resolved_height)
|
|
140
|
+
return super().drawImage(
|
|
141
|
+
image,
|
|
142
|
+
x,
|
|
143
|
+
y,
|
|
144
|
+
width,
|
|
145
|
+
height,
|
|
146
|
+
mask=mask,
|
|
147
|
+
preserveAspectRatio=preserveAspectRatio,
|
|
148
|
+
anchor=anchor,
|
|
149
|
+
anchorAtXY=anchorAtXY,
|
|
150
|
+
showBoundary=showBoundary,
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class _PdfRenderer:
|
|
155
|
+
"""维护一次 MiddleJson 到 PDF 渲染所需的样式、公式与素材状态。"""
|
|
156
|
+
|
|
157
|
+
def __init__(
|
|
158
|
+
self,
|
|
159
|
+
middle_json: MiddleJson,
|
|
160
|
+
*,
|
|
161
|
+
asset_resolver: AssetResolver | None,
|
|
162
|
+
document_title: str | None,
|
|
163
|
+
) -> None:
|
|
164
|
+
"""保存严格输入,并预注册标题及页面脚注 anchor。"""
|
|
165
|
+
self.middle_json = middle_json
|
|
166
|
+
self.asset_resolver = asset_resolver
|
|
167
|
+
self.document_title = _resolve_document_title(middle_json, document_title)
|
|
168
|
+
self.styles = build_pdf_styles()
|
|
169
|
+
self.available_width = A4[0] - 2 * PAGE_MARGIN
|
|
170
|
+
self.available_height = A4[1] - 2 * PAGE_MARGIN
|
|
171
|
+
self.inline_context = PdfInlineContext(
|
|
172
|
+
formulas=FormulaRenderer(),
|
|
173
|
+
anchors=PdfAnchorRegistry(_iter_document_anchors(middle_json)),
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
def render(self) -> bytes:
|
|
177
|
+
"""构造逐页 story,并将确定性 ReportLab 文档序列化为 bytes。"""
|
|
178
|
+
story: list[Flowable] = []
|
|
179
|
+
planned_pages = build_render_plan(self.middle_json)
|
|
180
|
+
for planned_blocks in planned_pages:
|
|
181
|
+
for planned in planned_blocks:
|
|
182
|
+
if planned.removed:
|
|
183
|
+
continue
|
|
184
|
+
if planned.block.type in PAGE_AUXILIARY_BLOCK_TYPES:
|
|
185
|
+
continue
|
|
186
|
+
rendered = self._render_planned_block(planned)
|
|
187
|
+
story.extend(rendered)
|
|
188
|
+
if not story:
|
|
189
|
+
story.append(Spacer(1, 1))
|
|
190
|
+
|
|
191
|
+
output = BytesIO()
|
|
192
|
+
document = SimpleDocTemplate(
|
|
193
|
+
output,
|
|
194
|
+
pagesize=A4,
|
|
195
|
+
leftMargin=PAGE_MARGIN,
|
|
196
|
+
rightMargin=PAGE_MARGIN,
|
|
197
|
+
topMargin=PAGE_MARGIN,
|
|
198
|
+
bottomMargin=PAGE_MARGIN,
|
|
199
|
+
title=self.document_title,
|
|
200
|
+
author="DocVortex",
|
|
201
|
+
creator="DocVortex PDF Renderer",
|
|
202
|
+
subject="Semantic document rendering from MiddleJson",
|
|
203
|
+
keywords="DocVortex, MiddleJson, PDF",
|
|
204
|
+
invariant=1,
|
|
205
|
+
pageCompression=1,
|
|
206
|
+
)
|
|
207
|
+
document.build(
|
|
208
|
+
story,
|
|
209
|
+
canvasmaker=partial(_PdfCanvas, document_title=self.document_title),
|
|
210
|
+
)
|
|
211
|
+
return output.getvalue()
|
|
212
|
+
|
|
213
|
+
def _render_planned_block(self, planned: PlannedBlock) -> list[Flowable]:
|
|
214
|
+
"""按严格 PageBlock 具体类型分派 PDF Flowable visitor。"""
|
|
215
|
+
block = planned.block
|
|
216
|
+
if isinstance(block, (TextBlock, RefTextBlock)):
|
|
217
|
+
spans = join_inline_spans(planned.text_contents or [block.content])
|
|
218
|
+
anchor = block.anchor if isinstance(block, TextBlock) else None
|
|
219
|
+
return [self._paragraph(spans, self.styles.body, planned.page_idx, block, anchor=anchor)]
|
|
220
|
+
if isinstance(block, (DocTitleBlock, ParagraphTitleBlock)):
|
|
221
|
+
return [
|
|
222
|
+
self._paragraph(
|
|
223
|
+
block.content,
|
|
224
|
+
self.styles.heading(block.level),
|
|
225
|
+
planned.page_idx,
|
|
226
|
+
block,
|
|
227
|
+
anchor=block.anchor,
|
|
228
|
+
)
|
|
229
|
+
]
|
|
230
|
+
if isinstance(block, PageFootnoteBlock):
|
|
231
|
+
return [
|
|
232
|
+
self._paragraph(
|
|
233
|
+
block.content,
|
|
234
|
+
self.styles.footnote,
|
|
235
|
+
planned.page_idx,
|
|
236
|
+
block,
|
|
237
|
+
anchor=block.anchor,
|
|
238
|
+
)
|
|
239
|
+
]
|
|
240
|
+
if isinstance(block, EquationBlock):
|
|
241
|
+
return self._render_equation(block, planned.page_idx)
|
|
242
|
+
if isinstance(block, ListBlock):
|
|
243
|
+
return self._render_list(block, planned.page_idx, depth=0)
|
|
244
|
+
if isinstance(block, IndexBlock):
|
|
245
|
+
return self._render_index(block, planned.page_idx, depth=0)
|
|
246
|
+
if isinstance(block, ImageBlock):
|
|
247
|
+
return self._render_image_block(block, planned.page_idx)
|
|
248
|
+
if isinstance(block, TableBlock):
|
|
249
|
+
return self._render_table_block(block, planned.page_idx)
|
|
250
|
+
if isinstance(block, ChartBlock):
|
|
251
|
+
return self._render_chart_block(block, planned.page_idx)
|
|
252
|
+
if isinstance(block, CodeBlock):
|
|
253
|
+
return self._render_code_block(block, planned.page_idx)
|
|
254
|
+
raise TypeError(f"Unsupported PageBlock type: {type(block).__name__}")
|
|
255
|
+
|
|
256
|
+
def _paragraph(
|
|
257
|
+
self,
|
|
258
|
+
spans: list[InlineSpan],
|
|
259
|
+
style: ParagraphStyle,
|
|
260
|
+
page_idx: int,
|
|
261
|
+
block: BlockBase,
|
|
262
|
+
*,
|
|
263
|
+
anchor: str | None = None,
|
|
264
|
+
preserve_newlines: bool = False,
|
|
265
|
+
max_width: float | None = None,
|
|
266
|
+
) -> Paragraph:
|
|
267
|
+
"""为当前 block 构造带定位、anchor 与矢量公式的 Paragraph。"""
|
|
268
|
+
return build_pdf_paragraph(
|
|
269
|
+
spans,
|
|
270
|
+
style,
|
|
271
|
+
context=self.inline_context,
|
|
272
|
+
page_idx=page_idx,
|
|
273
|
+
block_index=block.index,
|
|
274
|
+
block_type=str(block.type),
|
|
275
|
+
max_width=self.available_width if max_width is None else max_width,
|
|
276
|
+
anchor=anchor,
|
|
277
|
+
preserve_newlines=preserve_newlines,
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
def _render_equation(self, block: EquationBlock, page_idx: int) -> list[Flowable]:
|
|
281
|
+
"""优先输出 ZiaMath display 矢量,失败后使用图片、LaTeX 或占位。"""
|
|
282
|
+
content = block.content.strip()
|
|
283
|
+
if content:
|
|
284
|
+
formula, tag = split_formula_tag(content)
|
|
285
|
+
try:
|
|
286
|
+
vector = self.inline_context.formulas.render(formula or content, inline=False, font_size=14)
|
|
287
|
+
tag_vector = None
|
|
288
|
+
if tag:
|
|
289
|
+
tag_vector = self.inline_context.formulas.render(f"({tag})", inline=True, font_size=9)
|
|
290
|
+
flowable = DisplayFormulaFlowable(vector, tag_vector)
|
|
291
|
+
flowable.spaceBefore = 5
|
|
292
|
+
flowable.spaceAfter = 7
|
|
293
|
+
return [flowable]
|
|
294
|
+
except PdfFormulaError as exc:
|
|
295
|
+
logger.warning("PDF display formula fallback: {} ({})", exc, self._location(page_idx, block))
|
|
296
|
+
if _has_image_payload(block):
|
|
297
|
+
image = self._try_prepared_block_image(block, page_idx)
|
|
298
|
+
if image is not None:
|
|
299
|
+
return [self._prepared_image_flowable(image, block)]
|
|
300
|
+
if content:
|
|
301
|
+
return [
|
|
302
|
+
self._paragraph(
|
|
303
|
+
[TextSpan(type="text", content=content)],
|
|
304
|
+
self.styles.formula_fallback,
|
|
305
|
+
page_idx,
|
|
306
|
+
block,
|
|
307
|
+
preserve_newlines=True,
|
|
308
|
+
)
|
|
309
|
+
]
|
|
310
|
+
return [self._placeholder("formula unavailable", block=block, page_idx=page_idx)]
|
|
311
|
+
|
|
312
|
+
def _render_list(self, block: ListBlock, page_idx: int, *, depth: int) -> list[Flowable]:
|
|
313
|
+
"""保留 producer marker,以缩进和悬挂缩进表达递归列表。"""
|
|
314
|
+
rendered: list[Flowable] = []
|
|
315
|
+
add_reference_bullets = reference_list_needs_bullets(block)
|
|
316
|
+
for child in block.content:
|
|
317
|
+
if isinstance(child, ListBlock):
|
|
318
|
+
rendered.extend(self._render_list(child, page_idx, depth=depth + 1))
|
|
319
|
+
continue
|
|
320
|
+
spans = list(child.content)
|
|
321
|
+
parsed = parse_list_item_marker(spans)
|
|
322
|
+
if add_reference_bullets and parsed.marker is None and inline_plain_text(spans).strip():
|
|
323
|
+
spans = [TextSpan(type="text", content="- "), *spans]
|
|
324
|
+
parsed = parse_list_item_marker(spans)
|
|
325
|
+
style = ParagraphStyle(
|
|
326
|
+
f"DocVortex PDF List {depth} {len(rendered)}",
|
|
327
|
+
parent=self.styles.body,
|
|
328
|
+
leftIndent=(depth + (1 if parsed.marker else 0)) * 14,
|
|
329
|
+
firstLineIndent=-14 if parsed.marker else 0,
|
|
330
|
+
spaceAfter=3,
|
|
331
|
+
)
|
|
332
|
+
rendered.append(self._paragraph(spans, style, page_idx, child))
|
|
333
|
+
return rendered
|
|
334
|
+
|
|
335
|
+
def _render_index(self, block: IndexBlock, page_idx: int, *, depth: int) -> list[Flowable]:
|
|
336
|
+
"""递归输出目录叶子,并把已注册标题 anchor 写成内部链接。"""
|
|
337
|
+
rendered: list[Flowable] = []
|
|
338
|
+
for child in block.content:
|
|
339
|
+
if isinstance(child, IndexBlock):
|
|
340
|
+
rendered.extend(self._render_index(child, page_idx, depth=depth + 1))
|
|
341
|
+
continue
|
|
342
|
+
content = strip_index_page_tail(child.content)
|
|
343
|
+
if not content:
|
|
344
|
+
continue
|
|
345
|
+
spans: list[InlineSpan] = [TextSpan(type="text", content="• ")]
|
|
346
|
+
if child.anchor:
|
|
347
|
+
link_content = _flatten_non_link_spans(content)
|
|
348
|
+
if link_content:
|
|
349
|
+
spans.append(HyperlinkSpan(type="hyperlink", url=f"#{child.anchor}", content=link_content))
|
|
350
|
+
else:
|
|
351
|
+
spans.extend(content)
|
|
352
|
+
style = ParagraphStyle(
|
|
353
|
+
f"DocVortex PDF Index {depth} {len(rendered)}",
|
|
354
|
+
parent=self.styles.body,
|
|
355
|
+
leftIndent=(depth + 1) * 14,
|
|
356
|
+
firstLineIndent=-10,
|
|
357
|
+
spaceAfter=3,
|
|
358
|
+
)
|
|
359
|
+
rendered.append(self._paragraph(spans, style, page_idx, child))
|
|
360
|
+
return rendered
|
|
361
|
+
|
|
362
|
+
def _render_image_block(self, block: ImageBlock, page_idx: int) -> list[Flowable]:
|
|
363
|
+
"""按原始子块顺序输出图片主体、宽松占位与说明。"""
|
|
364
|
+
rendered: list[Flowable] = []
|
|
365
|
+
for child in block.content:
|
|
366
|
+
if isinstance(child, ImageBodyBlock):
|
|
367
|
+
alt_text = _plain_html_text(child.content) or block.sub_type or "image"
|
|
368
|
+
flowable, succeeded = self._image_or_placeholder(child, page_idx=page_idx, alt_text=alt_text)
|
|
369
|
+
rendered.append(flowable)
|
|
370
|
+
if not succeeded and child.content.strip():
|
|
371
|
+
rendered.append(
|
|
372
|
+
self._paragraph(
|
|
373
|
+
[TextSpan(type="text", content=_plain_html_text(child.content))],
|
|
374
|
+
self.styles.footnote,
|
|
375
|
+
page_idx,
|
|
376
|
+
child,
|
|
377
|
+
)
|
|
378
|
+
)
|
|
379
|
+
elif isinstance(child, ImageAnnotationBlock):
|
|
380
|
+
rendered.append(self._render_annotation(child, page_idx))
|
|
381
|
+
else:
|
|
382
|
+
raise TypeError(f"Unsupported image child: {type(child).__name__}")
|
|
383
|
+
return rendered
|
|
384
|
+
|
|
385
|
+
def _render_table_block(self, block: TableBlock, page_idx: int) -> list[Flowable]:
|
|
386
|
+
"""优先输出原生 HTML table,再回退空间文本、图片或占位。"""
|
|
387
|
+
rendered: list[Flowable] = []
|
|
388
|
+
for child in block.content:
|
|
389
|
+
if isinstance(child, TableBodyBlock):
|
|
390
|
+
content = child.content.strip()
|
|
391
|
+
if content and _HTML_TABLE_RE.search(content):
|
|
392
|
+
try:
|
|
393
|
+
rendered.extend(self._html_tables(content, page_idx=page_idx, block=child))
|
|
394
|
+
continue
|
|
395
|
+
except PdfTableError as exc:
|
|
396
|
+
logger.warning("PDF HTML table fallback: {} ({})", exc, self._location(page_idx, child))
|
|
397
|
+
if _has_image_payload(child):
|
|
398
|
+
image = self._try_prepared_block_image(child, page_idx)
|
|
399
|
+
if image is not None:
|
|
400
|
+
rendered.append(self._prepared_image_flowable(image, child))
|
|
401
|
+
continue
|
|
402
|
+
plain = _plain_html_text(content)
|
|
403
|
+
if plain:
|
|
404
|
+
rendered.append(self._preformatted(plain, page_idx, child, self.styles.spatial_table))
|
|
405
|
+
rendered.append(self._placeholder("table unavailable", block=child, page_idx=page_idx))
|
|
406
|
+
elif content:
|
|
407
|
+
rendered.append(self._preformatted(child.content, page_idx, child, self.styles.spatial_table))
|
|
408
|
+
else:
|
|
409
|
+
flowable, _succeeded = self._image_or_placeholder(child, page_idx=page_idx, alt_text="table")
|
|
410
|
+
rendered.append(flowable)
|
|
411
|
+
elif isinstance(child, TableAnnotationBlock):
|
|
412
|
+
rendered.append(self._render_annotation(child, page_idx))
|
|
413
|
+
else:
|
|
414
|
+
raise TypeError(f"Unsupported table child: {type(child).__name__}")
|
|
415
|
+
return rendered
|
|
416
|
+
|
|
417
|
+
def _render_chart_block(self, block: ChartBlock, page_idx: int) -> list[Flowable]:
|
|
418
|
+
"""输出 chart 图片或占位,并继续保留可物化的结构化内容。"""
|
|
419
|
+
rendered: list[Flowable] = []
|
|
420
|
+
for child in block.content:
|
|
421
|
+
if isinstance(child, ChartBodyBlock):
|
|
422
|
+
has_image = _has_image_payload(child)
|
|
423
|
+
image_succeeded = False
|
|
424
|
+
if has_image:
|
|
425
|
+
flowable, image_succeeded = self._image_or_placeholder(
|
|
426
|
+
child,
|
|
427
|
+
page_idx=page_idx,
|
|
428
|
+
alt_text=block.sub_type or "chart",
|
|
429
|
+
)
|
|
430
|
+
rendered.append(flowable)
|
|
431
|
+
content = child.content.strip()
|
|
432
|
+
if content and _HTML_TABLE_RE.search(content):
|
|
433
|
+
try:
|
|
434
|
+
rendered.extend(self._html_tables(content, page_idx=page_idx, block=child))
|
|
435
|
+
except PdfTableError as exc:
|
|
436
|
+
logger.warning("PDF chart table omitted: {} ({})", exc, self._location(page_idx, child))
|
|
437
|
+
if not image_succeeded:
|
|
438
|
+
rendered.append(self._preformatted(_plain_html_text(content), page_idx, child, self.styles.body))
|
|
439
|
+
elif content and not image_succeeded:
|
|
440
|
+
rendered.append(
|
|
441
|
+
self._paragraph(
|
|
442
|
+
[TextSpan(type="text", content=_plain_html_text(content))],
|
|
443
|
+
self.styles.body,
|
|
444
|
+
page_idx,
|
|
445
|
+
child,
|
|
446
|
+
)
|
|
447
|
+
)
|
|
448
|
+
elif not has_image and not content:
|
|
449
|
+
rendered.append(self._placeholder("chart unavailable", block=child, page_idx=page_idx))
|
|
450
|
+
elif isinstance(child, ChartAnnotationBlock):
|
|
451
|
+
rendered.append(self._render_annotation(child, page_idx))
|
|
452
|
+
else:
|
|
453
|
+
raise TypeError(f"Unsupported chart child: {type(child).__name__}")
|
|
454
|
+
return rendered
|
|
455
|
+
|
|
456
|
+
def _render_code_block(self, block: CodeBlock, page_idx: int) -> list[Flowable]:
|
|
457
|
+
"""使用等宽浅色块输出代码或带矢量公式的算法正文。"""
|
|
458
|
+
rendered: list[Flowable] = []
|
|
459
|
+
for child in block.content:
|
|
460
|
+
if isinstance(child, CodeBodyBlock):
|
|
461
|
+
if block.sub_type != BlockType.CODE:
|
|
462
|
+
raise TypeError("code_body requires code subtype")
|
|
463
|
+
rendered.append(
|
|
464
|
+
self._paragraph(
|
|
465
|
+
[TextSpan(type="text", content=child.content or " ")],
|
|
466
|
+
self.styles.code,
|
|
467
|
+
page_idx,
|
|
468
|
+
child,
|
|
469
|
+
preserve_newlines=True,
|
|
470
|
+
)
|
|
471
|
+
)
|
|
472
|
+
elif isinstance(child, AlgorithmBodyBlock):
|
|
473
|
+
if block.sub_type != RAW_ALGORITHM:
|
|
474
|
+
raise TypeError("algorithm_body requires algorithm subtype")
|
|
475
|
+
rendered.append(
|
|
476
|
+
self._paragraph(
|
|
477
|
+
child.content,
|
|
478
|
+
self.styles.code,
|
|
479
|
+
page_idx,
|
|
480
|
+
child,
|
|
481
|
+
preserve_newlines=True,
|
|
482
|
+
)
|
|
483
|
+
)
|
|
484
|
+
elif isinstance(child, CodeAnnotationBlock):
|
|
485
|
+
rendered.append(self._render_annotation(child, page_idx))
|
|
486
|
+
else:
|
|
487
|
+
raise TypeError(f"Unsupported code child: {type(child).__name__}")
|
|
488
|
+
return rendered
|
|
489
|
+
|
|
490
|
+
def _render_annotation(
|
|
491
|
+
self,
|
|
492
|
+
block: ImageAnnotationBlock | TableAnnotationBlock | ChartAnnotationBlock | CodeAnnotationBlock,
|
|
493
|
+
page_idx: int,
|
|
494
|
+
) -> Paragraph:
|
|
495
|
+
"""根据 caption/footnote discriminator 选择弱化说明样式。"""
|
|
496
|
+
style = self.styles.caption if str(block.type).endswith("caption") else self.styles.footnote
|
|
497
|
+
return self._paragraph(block.content, style, page_idx, block)
|
|
498
|
+
|
|
499
|
+
def _preformatted(
|
|
500
|
+
self,
|
|
501
|
+
content: str,
|
|
502
|
+
page_idx: int,
|
|
503
|
+
block: BlockBase,
|
|
504
|
+
style: ParagraphStyle,
|
|
505
|
+
) -> Paragraph:
|
|
506
|
+
"""把需要保留换行和空白的普通字符串写为 Paragraph。"""
|
|
507
|
+
return self._paragraph(
|
|
508
|
+
[TextSpan(type="text", content=content or " ")],
|
|
509
|
+
style,
|
|
510
|
+
page_idx,
|
|
511
|
+
block,
|
|
512
|
+
preserve_newlines=True,
|
|
513
|
+
)
|
|
514
|
+
|
|
515
|
+
def _html_tables(self, content: str, *, page_idx: int, block: BlockBase) -> list[Table]:
|
|
516
|
+
"""使用当前 block 上下文把 HTML table 物化为 ReportLab 表格。"""
|
|
517
|
+
|
|
518
|
+
def build_paragraph(spans: list[InlineSpan], style: object, max_width: float) -> Paragraph:
|
|
519
|
+
"""为表格单元格构造支持公式和链接的 Paragraph。"""
|
|
520
|
+
if not isinstance(style, ParagraphStyle):
|
|
521
|
+
raise TypeError("PDF table paragraph style must be a ParagraphStyle")
|
|
522
|
+
return self._paragraph(
|
|
523
|
+
spans,
|
|
524
|
+
style,
|
|
525
|
+
page_idx,
|
|
526
|
+
block,
|
|
527
|
+
preserve_newlines=True,
|
|
528
|
+
max_width=max_width,
|
|
529
|
+
)
|
|
530
|
+
|
|
531
|
+
def build_image(source: str, max_width: float, alt_text: str) -> Flowable:
|
|
532
|
+
"""为表格单元格构造离线图片或宽松占位。"""
|
|
533
|
+
try:
|
|
534
|
+
prepared = prepare_html_image(source, self.asset_resolver)
|
|
535
|
+
except PdfAssetError as exc:
|
|
536
|
+
logger.warning("PDF table image placeholder: {} ({})", exc, self._location(page_idx, block))
|
|
537
|
+
return self._placeholder(
|
|
538
|
+
f"image unavailable: {alt_text}",
|
|
539
|
+
block=block,
|
|
540
|
+
page_idx=page_idx,
|
|
541
|
+
width=max_width,
|
|
542
|
+
url=source if _is_remote_url(source) else None,
|
|
543
|
+
)
|
|
544
|
+
return self._prepared_image_flowable(prepared, None, max_width=max_width)
|
|
545
|
+
|
|
546
|
+
return list(
|
|
547
|
+
build_pdf_tables(
|
|
548
|
+
content,
|
|
549
|
+
available_width=self.available_width,
|
|
550
|
+
styles=self.styles,
|
|
551
|
+
build_paragraph=build_paragraph,
|
|
552
|
+
build_image=build_image,
|
|
553
|
+
)
|
|
554
|
+
)
|
|
555
|
+
|
|
556
|
+
def _image_or_placeholder(
|
|
557
|
+
self,
|
|
558
|
+
block: ImagePayloadBlock,
|
|
559
|
+
*,
|
|
560
|
+
page_idx: int,
|
|
561
|
+
alt_text: str,
|
|
562
|
+
) -> tuple[Flowable, bool]:
|
|
563
|
+
"""加载 block 图片;任何离线失败都转换为可见占位而不抛出。"""
|
|
564
|
+
prepared = self._try_prepared_block_image(block, page_idx)
|
|
565
|
+
if prepared is not None:
|
|
566
|
+
return self._prepared_image_flowable(prepared, block), True
|
|
567
|
+
remote_url = block.image_url if block.image_path is None and block.image_base64 is None else None
|
|
568
|
+
return (
|
|
569
|
+
self._placeholder(
|
|
570
|
+
f"image unavailable: {alt_text}",
|
|
571
|
+
block=block,
|
|
572
|
+
page_idx=page_idx,
|
|
573
|
+
url=remote_url,
|
|
574
|
+
),
|
|
575
|
+
False,
|
|
576
|
+
)
|
|
577
|
+
|
|
578
|
+
def _try_prepared_block_image(self, block: ImagePayloadBlock, page_idx: int) -> PreparedImage | None:
|
|
579
|
+
"""尝试离线准备图片并把所有素材错误降级为 warning。"""
|
|
580
|
+
try:
|
|
581
|
+
return prepare_block_image(block, self.asset_resolver)
|
|
582
|
+
except PdfAssetError as exc:
|
|
583
|
+
logger.warning("PDF image placeholder: {} ({})", exc, self._location(page_idx, block))
|
|
584
|
+
return None
|
|
585
|
+
|
|
586
|
+
def _prepared_image_flowable(
|
|
587
|
+
self,
|
|
588
|
+
prepared: PreparedImage,
|
|
589
|
+
block: ImagePayloadBlock | None,
|
|
590
|
+
*,
|
|
591
|
+
max_width: float | None = None,
|
|
592
|
+
) -> ReportLabImage:
|
|
593
|
+
"""按自然尺寸或 bbox 宽度限制图片,并保持宽高比和左对齐。"""
|
|
594
|
+
available_width = self.available_width if max_width is None else max(1.0, max_width)
|
|
595
|
+
natural_width = prepared.width_px / 96 * 72
|
|
596
|
+
desired_width = natural_width
|
|
597
|
+
if block is not None and block.bbox is not None:
|
|
598
|
+
desired_width = available_width * (block.bbox[2] - block.bbox[0])
|
|
599
|
+
desired_width = max(min(_MIN_IMAGE_WIDTH, available_width), min(desired_width, available_width))
|
|
600
|
+
desired_height = desired_width * prepared.height_px / max(prepared.width_px, 1)
|
|
601
|
+
max_height = self.available_height
|
|
602
|
+
if desired_height > max_height:
|
|
603
|
+
scale = max_height / desired_height
|
|
604
|
+
desired_width *= scale
|
|
605
|
+
desired_height *= scale
|
|
606
|
+
image = ReportLabImage(BytesIO(prepared.data), width=desired_width, height=desired_height)
|
|
607
|
+
image.hAlign = "LEFT"
|
|
608
|
+
image.spaceBefore = 5
|
|
609
|
+
image.spaceAfter = 5
|
|
610
|
+
return image
|
|
611
|
+
|
|
612
|
+
def _placeholder(
|
|
613
|
+
self,
|
|
614
|
+
label: str,
|
|
615
|
+
*,
|
|
616
|
+
block: BlockBase,
|
|
617
|
+
page_idx: int,
|
|
618
|
+
width: float | None = None,
|
|
619
|
+
url: str | None = None,
|
|
620
|
+
) -> Table:
|
|
621
|
+
"""创建带浅色边框、可选远程链接和定位文本的稳定占位框。"""
|
|
622
|
+
normalized = re.sub(r"\s+", " ", label).strip()[:_MAX_PLACEHOLDER_TEXT] or "content unavailable"
|
|
623
|
+
markup = render_plain_text_markup(normalized)
|
|
624
|
+
if url:
|
|
625
|
+
markup += f'<br/><a href="{html.escape(url, quote=True)}" color="#0b6fc2">{html.escape(url)}</a>'
|
|
626
|
+
location = self._location(page_idx, block)
|
|
627
|
+
markup += f'<br/><font size="7" color="#6b7280">{html.escape(location)}</font>'
|
|
628
|
+
paragraph = Paragraph(markup, self.styles.placeholder)
|
|
629
|
+
target_width = self.available_width if width is None else max(1.0, min(width, self.available_width))
|
|
630
|
+
table = Table([[paragraph]], colWidths=[target_width], rowHeights=[_PLACEHOLDER_HEIGHT], hAlign="LEFT")
|
|
631
|
+
table.setStyle(
|
|
632
|
+
TableStyle(
|
|
633
|
+
[
|
|
634
|
+
("BACKGROUND", (0, 0), (-1, -1), SURFACE_COLOR),
|
|
635
|
+
("BOX", (0, 0), (-1, -1), 0.6, BORDER_COLOR),
|
|
636
|
+
("VALIGN", (0, 0), (-1, -1), "MIDDLE"),
|
|
637
|
+
("LEFTPADDING", (0, 0), (-1, -1), 8),
|
|
638
|
+
("RIGHTPADDING", (0, 0), (-1, -1), 8),
|
|
639
|
+
("TOPPADDING", (0, 0), (-1, -1), 6),
|
|
640
|
+
("BOTTOMPADDING", (0, 0), (-1, -1), 6),
|
|
641
|
+
]
|
|
642
|
+
)
|
|
643
|
+
)
|
|
644
|
+
table.spaceBefore = 5
|
|
645
|
+
table.spaceAfter = 5
|
|
646
|
+
return table
|
|
647
|
+
|
|
648
|
+
@staticmethod
|
|
649
|
+
def _location(page_idx: int, block: BlockBase) -> str:
|
|
650
|
+
"""返回 PDF 告警与占位使用的稳定 page/block 定位。"""
|
|
651
|
+
return f"page_idx={page_idx}, block_index={block.index}, block_type={block.type}"
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def render_pdf(
|
|
655
|
+
middle_json: MiddleJson,
|
|
656
|
+
*,
|
|
657
|
+
asset_resolver: AssetResolver | None = None,
|
|
658
|
+
document_title: str | None = None,
|
|
659
|
+
) -> bytes:
|
|
660
|
+
"""把严格 MiddleJson 无副作用地渲染为完整 PDF bytes。"""
|
|
661
|
+
if not isinstance(middle_json, MiddleJson):
|
|
662
|
+
raise TypeError("render_pdf expects a MiddleJson instance")
|
|
663
|
+
if asset_resolver is not None and not callable(asset_resolver):
|
|
664
|
+
raise TypeError("asset_resolver must be callable or None")
|
|
665
|
+
if document_title is not None and not isinstance(document_title, str):
|
|
666
|
+
raise TypeError("document_title must be a string or None")
|
|
667
|
+
return _PdfRenderer(
|
|
668
|
+
middle_json,
|
|
669
|
+
asset_resolver=asset_resolver,
|
|
670
|
+
document_title=document_title,
|
|
671
|
+
).render()
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def _resolve_document_title(middle_json: MiddleJson, explicit: str | None) -> str:
|
|
675
|
+
"""按显式标题、首个文档标题和固定回退的顺序生成 metadata title。"""
|
|
676
|
+
if explicit is not None:
|
|
677
|
+
normalized = re.sub(r"\s+", " ", _INVALID_METADATA_TEXT_RE.sub("\ufffd", explicit)).strip()
|
|
678
|
+
return normalized or "DocVortex Document"
|
|
679
|
+
for page in middle_json.pages:
|
|
680
|
+
for block in page.blocks:
|
|
681
|
+
if isinstance(block, DocTitleBlock):
|
|
682
|
+
normalized = re.sub(
|
|
683
|
+
r"\s+",
|
|
684
|
+
" ",
|
|
685
|
+
_INVALID_METADATA_TEXT_RE.sub("\ufffd", inline_plain_text(block.content)),
|
|
686
|
+
).strip()
|
|
687
|
+
if normalized:
|
|
688
|
+
return normalized
|
|
689
|
+
return "DocVortex Document"
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def _iter_document_anchors(middle_json: MiddleJson) -> Iterable[str]:
|
|
693
|
+
"""按文档顺序枚举正文、标题和页面脚注的非空 anchor。"""
|
|
694
|
+
for page in middle_json.pages:
|
|
695
|
+
for block in page.blocks:
|
|
696
|
+
if isinstance(block, (TextBlock, TitleBlockBase)) and block.anchor:
|
|
697
|
+
yield block.anchor
|
|
698
|
+
elif isinstance(block, PageFootnoteBlock) and block.anchor:
|
|
699
|
+
yield block.anchor
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
def _plain_html_text(content: str) -> str:
|
|
703
|
+
"""把视觉 body 的 HTML 或普通字符串压缩为可见文本。"""
|
|
704
|
+
if not content:
|
|
705
|
+
return ""
|
|
706
|
+
soup = BeautifulSoup(content, "html.parser")
|
|
707
|
+
if soup.find(_VISIBLE_HTML_TAGS) is None:
|
|
708
|
+
return re.sub(r"[ \t]+", " ", content).strip()
|
|
709
|
+
return re.sub(r"[ \t]+", " ", soup.get_text("\n")).strip()
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def _flatten_non_link_spans(spans: list[InlineSpan]) -> list[NonLinkInlineSpan]:
|
|
713
|
+
"""递归移除已有 hyperlink 包装,供目录目标建立单层内部链接。"""
|
|
714
|
+
flattened: list[NonLinkInlineSpan] = []
|
|
715
|
+
for span in spans:
|
|
716
|
+
if isinstance(span, HyperlinkSpan):
|
|
717
|
+
flattened.extend(span.content)
|
|
718
|
+
else:
|
|
719
|
+
flattened.append(span)
|
|
720
|
+
return flattened
|
|
721
|
+
|
|
722
|
+
|
|
723
|
+
def _has_image_payload(block: ImagePayloadBlock) -> bool:
|
|
724
|
+
"""判断统一图片载荷是否声明 sidecar、data URI 或远程 URL。"""
|
|
725
|
+
return block.image_path is not None or block.image_base64 is not None or block.image_url is not None
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
def _is_remote_url(source: str) -> bool:
|
|
729
|
+
"""判断图片 source 是否是不会被 PDF renderer 下载的 HTTP(S) URL。"""
|
|
730
|
+
normalized = source.strip().casefold()
|
|
731
|
+
return normalized.startswith("http://") or normalized.startswith("https://")
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
__all__ = ["render_pdf"]
|