docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,521 @@
|
|
|
1
|
+
"""从 TextObject/TextCode 恢复语义文字与页面几何。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import html
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
from collections import deque
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from io import BytesIO, StringIO
|
|
11
|
+
|
|
12
|
+
from fontTools.pens.boundsPen import BoundsPen
|
|
13
|
+
from loguru import logger
|
|
14
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
15
|
+
|
|
16
|
+
from ....content.spans import text_spans
|
|
17
|
+
from ....schema import BBox
|
|
18
|
+
from .constants import MAX_DELTA_TOKENS, MAX_EXPANDED_GLYPHS, MAX_EXPANDED_TEXT_BYTES, MAX_FONT_BYTES, MAX_GLYPH_TOKENS
|
|
19
|
+
from .errors import OfdResourceLimitError
|
|
20
|
+
from .geometry import (
|
|
21
|
+
Affine,
|
|
22
|
+
bbox_intersection,
|
|
23
|
+
bbox_union,
|
|
24
|
+
canonical_angle,
|
|
25
|
+
parse_affine,
|
|
26
|
+
parse_st_box,
|
|
27
|
+
quad_bbox,
|
|
28
|
+
rect_quad,
|
|
29
|
+
transform_angle,
|
|
30
|
+
transform_bbox,
|
|
31
|
+
transform_quad,
|
|
32
|
+
)
|
|
33
|
+
from .models import FontResource, GlyphItem, ResourceRegistry, TextLine
|
|
34
|
+
from .package import OfdPackage, element_text, local_name, parse_int
|
|
35
|
+
|
|
36
|
+
_HEX_ESCAPE_RE = re.compile(r"\\([0-9A-Fa-f]{4})")
|
|
37
|
+
_GLYPH_TOKEN_RE = re.compile(r"\S+")
|
|
38
|
+
_DELTA_TOKEN_RE = re.compile(r"[^,\s]+")
|
|
39
|
+
_HEX_DIGITS = frozenset("0123456789abcdefABCDEF")
|
|
40
|
+
_TEXT_CODE_DECODE_CHUNK_SIZE = 64 * 1024
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(slots=True)
|
|
44
|
+
class OfdTextBudget:
|
|
45
|
+
"""累计限制 TextCode 文字与展开字形数量。"""
|
|
46
|
+
|
|
47
|
+
text_bytes: int = 0
|
|
48
|
+
glyph_count: int = 0
|
|
49
|
+
glyph_mapping_count: int = 0
|
|
50
|
+
glyph_token_count: int = 0
|
|
51
|
+
delta_token_count: int = 0
|
|
52
|
+
|
|
53
|
+
def charge(self, text: str) -> None:
|
|
54
|
+
"""为一次 TextCode 展开计费。"""
|
|
55
|
+
glyph_count = self.glyph_count + len(text)
|
|
56
|
+
if glyph_count > MAX_EXPANDED_GLYPHS:
|
|
57
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_expanded_glyphs={MAX_EXPANDED_GLYPHS}")
|
|
58
|
+
text_bytes = self.text_bytes + len(text.encode("utf-8"))
|
|
59
|
+
if text_bytes > MAX_EXPANDED_TEXT_BYTES:
|
|
60
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_expanded_text_bytes={MAX_EXPANDED_TEXT_BYTES}")
|
|
61
|
+
self.glyph_count = glyph_count
|
|
62
|
+
self.text_bytes = text_bytes
|
|
63
|
+
|
|
64
|
+
def charge_glyph_mapping(self, count: int) -> None:
|
|
65
|
+
"""累计 CGTransform 的有效字符映射数量并限制全文展开量。"""
|
|
66
|
+
self.glyph_mapping_count += count
|
|
67
|
+
if self.glyph_mapping_count > MAX_EXPANDED_GLYPHS:
|
|
68
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_expanded_glyphs={MAX_EXPANDED_GLYPHS}")
|
|
69
|
+
|
|
70
|
+
def charge_glyph_token(self) -> None:
|
|
71
|
+
"""累计实际扫描的 Glyphs token 数量并限制全文解析量。"""
|
|
72
|
+
self.glyph_token_count += 1
|
|
73
|
+
if self.glyph_token_count > MAX_GLYPH_TOKENS:
|
|
74
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_glyph_tokens={MAX_GLYPH_TOKENS}")
|
|
75
|
+
|
|
76
|
+
def charge_delta_token(self) -> None:
|
|
77
|
+
"""累计实际扫描的 Delta token 数量并限制全文解析量。"""
|
|
78
|
+
self.delta_token_count += 1
|
|
79
|
+
if self.delta_token_count > MAX_DELTA_TOKENS:
|
|
80
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_delta_tokens={MAX_DELTA_TOKENS}")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass(slots=True)
|
|
84
|
+
class _LoadedFont:
|
|
85
|
+
"""缓存 FontTools 中与字形几何相关的只读表。"""
|
|
86
|
+
|
|
87
|
+
font: object
|
|
88
|
+
glyph_order: list[str]
|
|
89
|
+
glyph_set: object
|
|
90
|
+
units_per_em: float
|
|
91
|
+
advances: dict[str, tuple[int, int]]
|
|
92
|
+
char_to_name: dict[int, str]
|
|
93
|
+
name_to_char: dict[str, str]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class FontMetricResolver:
|
|
97
|
+
"""按 OFD 字体资源惰性读取内嵌 OpenType 指标。"""
|
|
98
|
+
|
|
99
|
+
def __init__(self, package: OfdPackage) -> None:
|
|
100
|
+
"""绑定当前包并创建字体解析缓存。"""
|
|
101
|
+
self.package = package
|
|
102
|
+
self._cache: dict[str, _LoadedFont | None] = {}
|
|
103
|
+
|
|
104
|
+
def _load(self, resource: FontResource | None) -> _LoadedFont | None:
|
|
105
|
+
"""读取一个受限字体成员,失败时缓存空结果。"""
|
|
106
|
+
if resource is None or resource.font_part is None:
|
|
107
|
+
return None
|
|
108
|
+
if resource.font_part in self._cache:
|
|
109
|
+
return self._cache[resource.font_part]
|
|
110
|
+
data = self.package.read_part(resource.font_part, asset=True)
|
|
111
|
+
if data is None or len(data) > MAX_FONT_BYTES:
|
|
112
|
+
logger.warning(
|
|
113
|
+
f"OFD_FONT_UNAVAILABLE: part={resource.font_part!r}, reason={'missing' if data is None else 'too_large'}"
|
|
114
|
+
)
|
|
115
|
+
self._cache[resource.font_part] = None
|
|
116
|
+
return None
|
|
117
|
+
try:
|
|
118
|
+
from fontTools.ttLib import TTFont
|
|
119
|
+
|
|
120
|
+
font = TTFont(BytesIO(data), lazy=True)
|
|
121
|
+
glyph_order = list(font.getGlyphOrder())
|
|
122
|
+
glyph_set = font.getGlyphSet()
|
|
123
|
+
units_per_em = float(font["head"].unitsPerEm) if "head" in font else 1000.0
|
|
124
|
+
advances = dict(font["hmtx"].metrics) if "hmtx" in font else {}
|
|
125
|
+
char_to_name: dict[int, str] = {}
|
|
126
|
+
if "cmap" in font:
|
|
127
|
+
for table in font["cmap"].tables:
|
|
128
|
+
char_to_name.update(table.cmap)
|
|
129
|
+
name_to_char: dict[str, str] = {}
|
|
130
|
+
for codepoint, glyph_name in char_to_name.items():
|
|
131
|
+
if glyph_name not in name_to_char and 0 <= codepoint <= 0x10FFFF:
|
|
132
|
+
name_to_char[glyph_name] = chr(codepoint)
|
|
133
|
+
loaded = _LoadedFont(
|
|
134
|
+
font=font,
|
|
135
|
+
glyph_order=glyph_order,
|
|
136
|
+
glyph_set=glyph_set,
|
|
137
|
+
units_per_em=max(units_per_em, 1.0),
|
|
138
|
+
advances=advances,
|
|
139
|
+
char_to_name=char_to_name,
|
|
140
|
+
name_to_char=name_to_char,
|
|
141
|
+
)
|
|
142
|
+
except Exception as exc:
|
|
143
|
+
logger.warning(f"OFD_FONT_INVALID: part={resource.font_part!r}, error={type(exc).__name__}")
|
|
144
|
+
loaded = None
|
|
145
|
+
self._cache[resource.font_part] = loaded
|
|
146
|
+
return loaded
|
|
147
|
+
|
|
148
|
+
def resolve_character(self, resource: FontResource | None, glyph_id: int | None, fallback: str) -> str:
|
|
149
|
+
"""在 TextCode 使用占位符时尝试由 glyph cmap 恢复字符。"""
|
|
150
|
+
if fallback != "¤" or glyph_id is None:
|
|
151
|
+
return fallback
|
|
152
|
+
loaded = self._load(resource)
|
|
153
|
+
if loaded is None or not (0 <= glyph_id < len(loaded.glyph_order)):
|
|
154
|
+
return fallback
|
|
155
|
+
return loaded.name_to_char.get(loaded.glyph_order[glyph_id], fallback)
|
|
156
|
+
|
|
157
|
+
def glyph_bbox(
|
|
158
|
+
self,
|
|
159
|
+
resource: FontResource | None,
|
|
160
|
+
glyph_id: int | None,
|
|
161
|
+
character: str,
|
|
162
|
+
*,
|
|
163
|
+
size: float,
|
|
164
|
+
hscale: float,
|
|
165
|
+
advance_hint: float | None,
|
|
166
|
+
) -> BBox:
|
|
167
|
+
"""返回以 glyph origin 为基准的确定性局部字形框。"""
|
|
168
|
+
loaded = self._load(resource)
|
|
169
|
+
glyph_name: str | None = None
|
|
170
|
+
if loaded is not None:
|
|
171
|
+
if glyph_id is not None and 0 <= glyph_id < len(loaded.glyph_order):
|
|
172
|
+
glyph_name = loaded.glyph_order[glyph_id]
|
|
173
|
+
elif character:
|
|
174
|
+
glyph_name = loaded.char_to_name.get(ord(character[0]))
|
|
175
|
+
if loaded is not None and glyph_name is not None and glyph_name in loaded.glyph_set:
|
|
176
|
+
scale = size / loaded.units_per_em
|
|
177
|
+
try:
|
|
178
|
+
pen = BoundsPen(loaded.glyph_set)
|
|
179
|
+
loaded.glyph_set[glyph_name].draw(pen)
|
|
180
|
+
bounds = pen.bounds
|
|
181
|
+
except Exception:
|
|
182
|
+
bounds = None
|
|
183
|
+
advance = float(loaded.advances.get(glyph_name, (loaded.units_per_em, 0))[0]) * scale * hscale
|
|
184
|
+
if advance_hint is not None and advance_hint > 0:
|
|
185
|
+
advance = advance_hint
|
|
186
|
+
if bounds is not None:
|
|
187
|
+
x0, y0, x1, y1 = bounds
|
|
188
|
+
bbox = (x0 * scale * hscale, -y1 * scale, x1 * scale * hscale, -y0 * scale)
|
|
189
|
+
if bbox[2] > bbox[0] and bbox[3] > bbox[1]:
|
|
190
|
+
return bbox
|
|
191
|
+
return (0.0, -0.85 * size, max(advance, 0.2 * size), 0.2 * size)
|
|
192
|
+
fallback_advance = (
|
|
193
|
+
advance_hint if advance_hint is not None and advance_hint > 0 else max(0.5 * size * hscale, 0.2 * size)
|
|
194
|
+
)
|
|
195
|
+
return (0.0, -0.85 * size, fallback_advance, 0.2 * size)
|
|
196
|
+
|
|
197
|
+
def close(self) -> None:
|
|
198
|
+
"""关闭已经打开的 FontTools 字体对象。"""
|
|
199
|
+
for loaded in self._cache.values():
|
|
200
|
+
if loaded is None:
|
|
201
|
+
continue
|
|
202
|
+
close = getattr(loaded.font, "close", None)
|
|
203
|
+
if callable(close):
|
|
204
|
+
close()
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def decode_text_code(value: str) -> str:
|
|
208
|
+
"""解码 OFD TextCode 中的反斜杠四位十六进制字符。"""
|
|
209
|
+
return _HEX_ESCAPE_RE.sub(lambda match: chr(int(match.group(1), 16)), value)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _incomplete_hex_escape_length(value: str) -> int:
|
|
213
|
+
"""返回末尾可能跨分片的反斜杠十六进制前缀长度。"""
|
|
214
|
+
for length in range(min(4, len(value)), 0, -1):
|
|
215
|
+
suffix = value[-length:]
|
|
216
|
+
if suffix[0] == "\\" and all(character in _HEX_DIGITS for character in suffix[1:]):
|
|
217
|
+
return length
|
|
218
|
+
return 0
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _decode_text_code_element(text_code: etree._Element, budget: OfdTextBudget) -> str:
|
|
222
|
+
"""分片解码一个 TextCode,并在写入完整字符串前累计文字预算。"""
|
|
223
|
+
output = StringIO()
|
|
224
|
+
carry = ""
|
|
225
|
+
for part in text_code.itertext():
|
|
226
|
+
for offset in range(0, len(part), _TEXT_CODE_DECODE_CHUNK_SIZE):
|
|
227
|
+
chunk = part[offset : offset + _TEXT_CODE_DECODE_CHUNK_SIZE]
|
|
228
|
+
value = f"{carry}{chunk}" if carry else chunk
|
|
229
|
+
carry_length = _incomplete_hex_escape_length(value)
|
|
230
|
+
if carry_length:
|
|
231
|
+
complete = value[:-carry_length]
|
|
232
|
+
carry = value[-carry_length:]
|
|
233
|
+
else:
|
|
234
|
+
complete = value
|
|
235
|
+
carry = ""
|
|
236
|
+
if not complete:
|
|
237
|
+
continue
|
|
238
|
+
decoded = decode_text_code(complete)
|
|
239
|
+
budget.charge(decoded)
|
|
240
|
+
output.write(decoded)
|
|
241
|
+
if carry:
|
|
242
|
+
budget.charge(carry)
|
|
243
|
+
output.write(carry)
|
|
244
|
+
return output.getvalue()
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def parse_delta(value: str | None, count: int, budget: OfdTextBudget) -> list[float]:
|
|
248
|
+
"""流式展开普通与 g-count-value 压缩 Delta,并补齐不足项。"""
|
|
249
|
+
if count <= 0:
|
|
250
|
+
return []
|
|
251
|
+
output: list[float] = []
|
|
252
|
+
matches = iter(_DELTA_TOKEN_RE.finditer(value or ""))
|
|
253
|
+
pending: deque[str] = deque()
|
|
254
|
+
|
|
255
|
+
def next_token() -> str | None:
|
|
256
|
+
"""返回下一个 Delta token,并只对首次扫描计费。"""
|
|
257
|
+
if pending:
|
|
258
|
+
return pending.popleft()
|
|
259
|
+
try:
|
|
260
|
+
match = next(matches)
|
|
261
|
+
except StopIteration:
|
|
262
|
+
return None
|
|
263
|
+
budget.charge_delta_token()
|
|
264
|
+
return match.group()
|
|
265
|
+
|
|
266
|
+
while len(output) < count:
|
|
267
|
+
token = next_token()
|
|
268
|
+
if token is None:
|
|
269
|
+
break
|
|
270
|
+
if token.casefold() == "g":
|
|
271
|
+
repeat_token = next_token()
|
|
272
|
+
value_token = next_token()
|
|
273
|
+
if repeat_token is None or value_token is None:
|
|
274
|
+
if repeat_token is not None:
|
|
275
|
+
pending.appendleft(repeat_token)
|
|
276
|
+
continue
|
|
277
|
+
try:
|
|
278
|
+
repeat = max(0, int(repeat_token))
|
|
279
|
+
repeated_value = float(value_token)
|
|
280
|
+
except ValueError:
|
|
281
|
+
pending.appendleft(value_token)
|
|
282
|
+
pending.appendleft(repeat_token)
|
|
283
|
+
continue
|
|
284
|
+
if math.isfinite(repeated_value):
|
|
285
|
+
output.extend([repeated_value] * min(repeat, count - len(output)))
|
|
286
|
+
continue
|
|
287
|
+
try:
|
|
288
|
+
parsed = float(token)
|
|
289
|
+
except ValueError:
|
|
290
|
+
continue
|
|
291
|
+
if math.isfinite(parsed):
|
|
292
|
+
output.append(parsed)
|
|
293
|
+
if len(output) < count:
|
|
294
|
+
output.extend([output[-1] if output else 0.0] * (count - len(output)))
|
|
295
|
+
return output[:count]
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _bounded_glyph_ids(glyphs_element: etree._Element, limit: int, budget: OfdTextBudget) -> list[int]:
|
|
299
|
+
"""按需迭代 Glyphs token,只保留有效映射所需的有限 ID。"""
|
|
300
|
+
glyph_ids: list[int] = []
|
|
301
|
+
for match in _GLYPH_TOKEN_RE.finditer(element_text(glyphs_element)):
|
|
302
|
+
budget.charge_glyph_token()
|
|
303
|
+
glyph_id = parse_int(match.group())
|
|
304
|
+
if glyph_id is None:
|
|
305
|
+
continue
|
|
306
|
+
glyph_ids.append(glyph_id)
|
|
307
|
+
if len(glyph_ids) >= limit:
|
|
308
|
+
break
|
|
309
|
+
return glyph_ids
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _glyph_map(text_object: etree._Element, position_count: int, budget: OfdTextBudget) -> dict[int, int]:
|
|
313
|
+
"""把实际 TextCode 字符位置映射到 glyph ID,并限制累计展开量。"""
|
|
314
|
+
result: dict[int, int] = {}
|
|
315
|
+
for element in text_object:
|
|
316
|
+
if local_name(element.tag) != "CGTransform":
|
|
317
|
+
continue
|
|
318
|
+
code_position = parse_int(element.get("CodePosition"))
|
|
319
|
+
code_count = parse_int(element.get("CodeCount"))
|
|
320
|
+
glyphs_element = next((child for child in element if local_name(child.tag) == "Glyphs"), None)
|
|
321
|
+
if code_position is None or code_count is None or glyphs_element is None:
|
|
322
|
+
continue
|
|
323
|
+
effective_count = min(code_count, max(0, position_count - code_position))
|
|
324
|
+
if effective_count <= 0:
|
|
325
|
+
continue
|
|
326
|
+
glyph_ids = _bounded_glyph_ids(glyphs_element, effective_count, budget)
|
|
327
|
+
if not glyph_ids:
|
|
328
|
+
continue
|
|
329
|
+
budget.charge_glyph_mapping(effective_count)
|
|
330
|
+
for offset in range(effective_count):
|
|
331
|
+
glyph_index = min(offset, len(glyph_ids) - 1)
|
|
332
|
+
result[code_position + offset] = glyph_ids[glyph_index]
|
|
333
|
+
return result
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _styles(
|
|
337
|
+
text_object: etree._Element,
|
|
338
|
+
font: FontResource | None,
|
|
339
|
+
resolved_style: dict[str, str],
|
|
340
|
+
) -> tuple[str, ...]:
|
|
341
|
+
"""从字体资源和 TextObject 属性恢复可投影行内样式。"""
|
|
342
|
+
styles: list[str] = []
|
|
343
|
+
weight = parse_int(resolved_style.get("Weight") or text_object.get("Weight"))
|
|
344
|
+
if (font is not None and font.bold) or (weight is not None and weight >= 600):
|
|
345
|
+
styles.append("bold")
|
|
346
|
+
if (font is not None and font.italic) or (resolved_style.get("Italic") or text_object.get("Italic") or "").casefold() in {
|
|
347
|
+
"true",
|
|
348
|
+
"1",
|
|
349
|
+
}:
|
|
350
|
+
styles.append("italic")
|
|
351
|
+
return tuple(styles)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def format_line_spans(text: str, styles: tuple[str, ...]) -> list[dict[str, object]]:
|
|
355
|
+
"""把 OFD 原生文字和样式直接投影为结构化 Span。"""
|
|
356
|
+
return text_spans(text, styles)
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def format_line_html(text: str, styles: tuple[str, ...]) -> str:
|
|
360
|
+
"""把 OFD 表格单元格文字序列化为安全 HTML。"""
|
|
361
|
+
rendered = html.escape(text, quote=False)
|
|
362
|
+
if "superscript" in styles:
|
|
363
|
+
rendered = f"<sup>{rendered}</sup>"
|
|
364
|
+
elif "subscript" in styles:
|
|
365
|
+
rendered = f"<sub>{rendered}</sub>"
|
|
366
|
+
if "underline" in styles:
|
|
367
|
+
rendered = f"<u>{rendered}</u>"
|
|
368
|
+
if "bold" in styles:
|
|
369
|
+
rendered = f"<strong>{rendered}</strong>"
|
|
370
|
+
if "italic" in styles:
|
|
371
|
+
rendered = f"<em>{rendered}</em>"
|
|
372
|
+
if "strikethrough" in styles:
|
|
373
|
+
rendered = f"<s>{rendered}</s>"
|
|
374
|
+
return rendered
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def build_text_lines(
|
|
378
|
+
text_object: etree._Element,
|
|
379
|
+
*,
|
|
380
|
+
parent_transform: Affine,
|
|
381
|
+
parent_clip: BBox,
|
|
382
|
+
resources: ResourceRegistry,
|
|
383
|
+
package: OfdPackage,
|
|
384
|
+
font_metrics: FontMetricResolver,
|
|
385
|
+
budget: OfdTextBudget,
|
|
386
|
+
paint_order: int,
|
|
387
|
+
layer_type: str,
|
|
388
|
+
template_id: int | None,
|
|
389
|
+
resolved_style: dict[str, str] | None = None,
|
|
390
|
+
) -> list[TextLine]:
|
|
391
|
+
"""把一个 TextObject 展开为按 TextCode 划分的页面文字行。"""
|
|
392
|
+
style = resolved_style or {}
|
|
393
|
+
if (style.get("Visible") or text_object.get("Visible") or "true").casefold() in {"false", "0"}:
|
|
394
|
+
return []
|
|
395
|
+
if (style.get("Alpha") or text_object.get("Alpha") or "255").strip() == "0":
|
|
396
|
+
return []
|
|
397
|
+
boundary = parse_st_box(text_object.get("Boundary"))
|
|
398
|
+
if boundary is None:
|
|
399
|
+
logger.warning(f"OFD_TEXT_INVALID_BOUNDARY: object_id={text_object.get('ID')!r}")
|
|
400
|
+
return []
|
|
401
|
+
boundary_page = transform_bbox(boundary, parent_transform)
|
|
402
|
+
if boundary_page is None:
|
|
403
|
+
return []
|
|
404
|
+
object_clip = bbox_intersection(parent_clip, boundary_page)
|
|
405
|
+
if object_clip is None:
|
|
406
|
+
return []
|
|
407
|
+
translation = Affine.translation(boundary[0], boundary[1])
|
|
408
|
+
object_transform = parent_transform.compose(translation).compose(parse_affine(text_object.get("CTM")))
|
|
409
|
+
read_direction = float(parse_int(text_object.get("ReadDirection")) or 0)
|
|
410
|
+
char_direction = float(parse_int(text_object.get("CharDirection")) or 0)
|
|
411
|
+
direction_transform = object_transform.compose(Affine.rotation(read_direction))
|
|
412
|
+
font_id = parse_int(text_object.get("Font"))
|
|
413
|
+
font = resources.fonts.get(font_id) if font_id is not None else None
|
|
414
|
+
try:
|
|
415
|
+
size = max(0.1, float(text_object.get("Size") or 1.0))
|
|
416
|
+
hscale = max(0.01, float(text_object.get("HScale") or 1.0))
|
|
417
|
+
except ValueError:
|
|
418
|
+
size, hscale = 1.0, 1.0
|
|
419
|
+
styles = _styles(text_object, font, style)
|
|
420
|
+
decoded_text_codes: list[tuple[etree._Element, str]] = []
|
|
421
|
+
position_count = 0
|
|
422
|
+
for text_code in (element for element in text_object if local_name(element.tag) == "TextCode"):
|
|
423
|
+
text = _decode_text_code_element(text_code, budget)
|
|
424
|
+
decoded_text_codes.append((text_code, text))
|
|
425
|
+
if text:
|
|
426
|
+
position_count += len(text)
|
|
427
|
+
glyph_by_position = _glyph_map(text_object, position_count, budget)
|
|
428
|
+
global_position = 0
|
|
429
|
+
inherited_x: float | None = None
|
|
430
|
+
inherited_y: float | None = None
|
|
431
|
+
lines: list[TextLine] = []
|
|
432
|
+
object_id = parse_int(text_object.get("ID"))
|
|
433
|
+
for code_index, (text_code, text) in enumerate(decoded_text_codes):
|
|
434
|
+
if not text:
|
|
435
|
+
continue
|
|
436
|
+
try:
|
|
437
|
+
if text_code.get("X") is not None:
|
|
438
|
+
inherited_x = float(text_code.get("X") or "")
|
|
439
|
+
if text_code.get("Y") is not None:
|
|
440
|
+
inherited_y = float(text_code.get("Y") or "")
|
|
441
|
+
except ValueError:
|
|
442
|
+
continue
|
|
443
|
+
if inherited_x is None or inherited_y is None:
|
|
444
|
+
logger.warning(f"OFD_TEXT_MISSING_ORIGIN: object_id={object_id}, text_code={code_index}")
|
|
445
|
+
global_position += len(text)
|
|
446
|
+
continue
|
|
447
|
+
delta_count = max(0, len(text) - 1)
|
|
448
|
+
delta_x = parse_delta(text_code.get("DeltaX"), delta_count, budget)
|
|
449
|
+
delta_y = parse_delta(text_code.get("DeltaY"), delta_count, budget)
|
|
450
|
+
origins: list[tuple[float, float]] = [(inherited_x, inherited_y)]
|
|
451
|
+
for index in range(delta_count):
|
|
452
|
+
previous = origins[-1]
|
|
453
|
+
origins.append((previous[0] + delta_x[index], previous[1] + delta_y[index]))
|
|
454
|
+
glyph_items: list[GlyphItem] = []
|
|
455
|
+
for index, (character, origin) in enumerate(zip(text, origins, strict=True)):
|
|
456
|
+
glyph_id = glyph_by_position.get(global_position + index)
|
|
457
|
+
resolved_character = font_metrics.resolve_character(font, glyph_id, character)
|
|
458
|
+
advance_hint = None
|
|
459
|
+
if index + 1 < len(origins):
|
|
460
|
+
advance_hint = math.dist(origin, origins[index + 1])
|
|
461
|
+
local_bbox = font_metrics.glyph_bbox(
|
|
462
|
+
font,
|
|
463
|
+
glyph_id,
|
|
464
|
+
resolved_character,
|
|
465
|
+
size=size,
|
|
466
|
+
hscale=hscale,
|
|
467
|
+
advance_hint=advance_hint,
|
|
468
|
+
)
|
|
469
|
+
char_transform = direction_transform.compose(Affine.translation(origin[0], origin[1])).compose(
|
|
470
|
+
Affine.rotation(char_direction)
|
|
471
|
+
)
|
|
472
|
+
quad = transform_quad(rect_quad(local_bbox), char_transform)
|
|
473
|
+
glyph_bbox = quad_bbox(quad)
|
|
474
|
+
if glyph_bbox is None:
|
|
475
|
+
continue
|
|
476
|
+
clipped_glyph_bbox = bbox_intersection(glyph_bbox, object_clip)
|
|
477
|
+
if clipped_glyph_bbox is None:
|
|
478
|
+
continue
|
|
479
|
+
glyph_items.append(
|
|
480
|
+
GlyphItem(
|
|
481
|
+
text=resolved_character,
|
|
482
|
+
bbox=clipped_glyph_bbox,
|
|
483
|
+
quad=quad,
|
|
484
|
+
origin=char_transform.apply((0.0, 0.0)),
|
|
485
|
+
glyph_id=glyph_id,
|
|
486
|
+
)
|
|
487
|
+
)
|
|
488
|
+
global_position += len(text)
|
|
489
|
+
if not glyph_items:
|
|
490
|
+
continue
|
|
491
|
+
line_bbox = bbox_union(item.bbox for item in glyph_items)
|
|
492
|
+
if line_bbox is None:
|
|
493
|
+
continue
|
|
494
|
+
if line_bbox[2] <= line_bbox[0] or line_bbox[3] <= line_bbox[1]:
|
|
495
|
+
continue
|
|
496
|
+
lines.append(
|
|
497
|
+
TextLine(
|
|
498
|
+
text="".join(item.text for item in glyph_items),
|
|
499
|
+
bbox=line_bbox,
|
|
500
|
+
glyphs=glyph_items,
|
|
501
|
+
angle=canonical_angle(transform_angle(direction_transform)),
|
|
502
|
+
font_size=size * math.hypot(direction_transform.a, direction_transform.b),
|
|
503
|
+
paint_order=paint_order + code_index,
|
|
504
|
+
object_id=object_id,
|
|
505
|
+
layer_type=layer_type,
|
|
506
|
+
template_id=template_id,
|
|
507
|
+
styles=styles,
|
|
508
|
+
)
|
|
509
|
+
)
|
|
510
|
+
return lines
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
__all__ = [
|
|
514
|
+
"FontMetricResolver",
|
|
515
|
+
"OfdTextBudget",
|
|
516
|
+
"build_text_lines",
|
|
517
|
+
"decode_text_code",
|
|
518
|
+
"format_line_html",
|
|
519
|
+
"format_line_spans",
|
|
520
|
+
"parse_delta",
|
|
521
|
+
]
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""解析 DOC 标准书签名称及其主文档 CP 范围。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ..legacy.binary import bounded_slice, get_u16
|
|
6
|
+
from .records import DocBudget, parse_plc
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _parse_string_table(data: bytes, budget: DocBudget) -> list[str]:
|
|
10
|
+
"""解析扩展或单字节 STTB 字符串表。"""
|
|
11
|
+
|
|
12
|
+
if len(data) < 2:
|
|
13
|
+
return []
|
|
14
|
+
extended = get_u16(data, 0) == 0xFFFF
|
|
15
|
+
if extended:
|
|
16
|
+
count = get_u16(data, 2)
|
|
17
|
+
extra = get_u16(data, 4)
|
|
18
|
+
cursor = 6
|
|
19
|
+
else:
|
|
20
|
+
count = get_u16(data, 0)
|
|
21
|
+
extra = get_u16(data, 2)
|
|
22
|
+
cursor = 4
|
|
23
|
+
if count is None or extra is None:
|
|
24
|
+
return []
|
|
25
|
+
strings: list[str] = []
|
|
26
|
+
for _ in range(count):
|
|
27
|
+
if extended:
|
|
28
|
+
length = get_u16(data, cursor)
|
|
29
|
+
cursor += 2
|
|
30
|
+
width = 2
|
|
31
|
+
else:
|
|
32
|
+
length = data[cursor] if cursor < len(data) else None
|
|
33
|
+
cursor += 1
|
|
34
|
+
width = 1
|
|
35
|
+
if length is None or length == 0xFFFF:
|
|
36
|
+
strings.append("")
|
|
37
|
+
continue
|
|
38
|
+
payload = bounded_slice(data, cursor, length * width)
|
|
39
|
+
if payload is None:
|
|
40
|
+
break
|
|
41
|
+
cursor += length * width
|
|
42
|
+
strings.append(payload.decode("utf-16le" if extended else "cp1252", errors="replace"))
|
|
43
|
+
cursor += extra
|
|
44
|
+
budget.charge()
|
|
45
|
+
return strings
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def parse_bookmarks(
|
|
49
|
+
table_stream: bytes,
|
|
50
|
+
*,
|
|
51
|
+
names_offset: int,
|
|
52
|
+
names_size: int,
|
|
53
|
+
starts_offset: int,
|
|
54
|
+
starts_size: int,
|
|
55
|
+
ends_offset: int,
|
|
56
|
+
ends_size: int,
|
|
57
|
+
budget: DocBudget,
|
|
58
|
+
) -> dict[int, list[str]]:
|
|
59
|
+
"""返回主文档中书签起始 CP 到名称列表的映射。"""
|
|
60
|
+
|
|
61
|
+
names_payload = bounded_slice(table_stream, names_offset, names_size)
|
|
62
|
+
starts_payload = bounded_slice(table_stream, starts_offset, starts_size)
|
|
63
|
+
ends_payload = bounded_slice(table_stream, ends_offset, ends_size)
|
|
64
|
+
if names_payload is None or starts_payload is None or ends_payload is None:
|
|
65
|
+
return {}
|
|
66
|
+
names = _parse_string_table(names_payload, budget)
|
|
67
|
+
start_cps, start_items = parse_plc(starts_payload, item_size=4, budget=budget)
|
|
68
|
+
end_cps, _ = parse_plc(ends_payload, item_size=0, budget=budget)
|
|
69
|
+
result: dict[int, list[str]] = {}
|
|
70
|
+
for index, (name, item) in enumerate(zip(names, start_items, strict=False)):
|
|
71
|
+
if not name or name.startswith("_GoBack") or index >= len(start_cps):
|
|
72
|
+
continue
|
|
73
|
+
end_index = get_u16(item, 0)
|
|
74
|
+
if end_index is None or end_index >= max(len(end_cps) - 1, 0):
|
|
75
|
+
continue
|
|
76
|
+
start = start_cps[index]
|
|
77
|
+
end = end_cps[end_index]
|
|
78
|
+
if end < start:
|
|
79
|
+
continue
|
|
80
|
+
result.setdefault(start, []).append(name)
|
|
81
|
+
return result
|