docvortex 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docvortex/__init__.py +21 -0
- docvortex/analyzers/__init__.py +3 -0
- docvortex/analyzers/native/__init__.py +37 -0
- docvortex/analyzers/native/_shared/__init__.py +3 -0
- docvortex/analyzers/native/_shared/hyperlink.py +16 -0
- docvortex/analyzers/native/_shared/image.py +8 -0
- docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
- docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
- docvortex/analyzers/native/_shared/markup/formula.py +37 -0
- docvortex/analyzers/native/_shared/markup/projector.py +57 -0
- docvortex/analyzers/native/_shared/markup/styles.py +19 -0
- docvortex/analyzers/native/_shared/mathml.py +17 -0
- docvortex/analyzers/native/_shared/names.py +7 -0
- docvortex/analyzers/native/_shared/xycut.py +414 -0
- docvortex/analyzers/native/contracts.py +44 -0
- docvortex/analyzers/native/csv.py +351 -0
- docvortex/analyzers/native/epub/__init__.py +18 -0
- docvortex/analyzers/native/epub/constants.py +45 -0
- docvortex/analyzers/native/epub/converter.py +80 -0
- docvortex/analyzers/native/epub/errors.py +20 -0
- docvortex/analyzers/native/epub/metadata.py +31 -0
- docvortex/analyzers/native/epub/package.py +649 -0
- docvortex/analyzers/native/epub/xhtml.py +306 -0
- docvortex/analyzers/native/html/__init__.py +6 -0
- docvortex/analyzers/native/html/anchors.py +271 -0
- docvortex/analyzers/native/html/constants.py +25 -0
- docvortex/analyzers/native/html/contracts.py +7 -0
- docvortex/analyzers/native/html/converter.py +117 -0
- docvortex/analyzers/native/html/document.py +389 -0
- docvortex/analyzers/native/html/errors.py +12 -0
- docvortex/analyzers/native/html/resources.py +365 -0
- docvortex/analyzers/native/html/selector.py +421 -0
- docvortex/analyzers/native/models.py +209 -0
- docvortex/analyzers/native/ofd/__init__.py +14 -0
- docvortex/analyzers/native/ofd/constants.py +59 -0
- docvortex/analyzers/native/ofd/converter.py +31 -0
- docvortex/analyzers/native/ofd/errors.py +16 -0
- docvortex/analyzers/native/ofd/geometry.py +215 -0
- docvortex/analyzers/native/ofd/images.py +124 -0
- docvortex/analyzers/native/ofd/metadata.py +53 -0
- docvortex/analyzers/native/ofd/models.py +170 -0
- docvortex/analyzers/native/ofd/package.py +346 -0
- docvortex/analyzers/native/ofd/path.py +201 -0
- docvortex/analyzers/native/ofd/reading_order.py +293 -0
- docvortex/analyzers/native/ofd/resources.py +122 -0
- docvortex/analyzers/native/ofd/scene.py +347 -0
- docvortex/analyzers/native/ofd/table.py +240 -0
- docvortex/analyzers/native/ofd/text.py +521 -0
- docvortex/analyzers/native/office/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/__init__.py +3 -0
- docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
- docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
- docvortex/analyzers/native/office/doc/fib.py +261 -0
- docvortex/analyzers/native/office/doc/fields.py +101 -0
- docvortex/analyzers/native/office/doc/formatting.py +171 -0
- docvortex/analyzers/native/office/doc/images.py +162 -0
- docvortex/analyzers/native/office/doc/lists.py +356 -0
- docvortex/analyzers/native/office/doc/models.py +164 -0
- docvortex/analyzers/native/office/doc/parser.py +851 -0
- docvortex/analyzers/native/office/doc/pieces.py +256 -0
- docvortex/analyzers/native/office/doc/records.py +40 -0
- docvortex/analyzers/native/office/doc/sprm.py +269 -0
- docvortex/analyzers/native/office/doc/styles.py +215 -0
- docvortex/analyzers/native/office/docx/__init__.py +3 -0
- docvortex/analyzers/native/office/docx/context.py +71 -0
- docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
- docvortex/analyzers/native/office/docx/equationxml.py +119 -0
- docvortex/analyzers/native/office/docx/fields.py +816 -0
- docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
- docvortex/analyzers/native/office/docx/main.py +50 -0
- docvortex/analyzers/native/office/docx/numbering.py +491 -0
- docvortex/analyzers/native/office/docx/office_xml.py +57 -0
- docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
- docvortex/analyzers/native/office/docx/resources.py +526 -0
- docvortex/analyzers/native/office/docx/styles.py +693 -0
- docvortex/analyzers/native/office/docx/tables.py +515 -0
- docvortex/analyzers/native/office/equation/__init__.py +3 -0
- docvortex/analyzers/native/office/equation/image.py +470 -0
- docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
- docvortex/analyzers/native/office/equation/mtef.py +885 -0
- docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
- docvortex/analyzers/native/office/equation/omml.py +561 -0
- docvortex/analyzers/native/office/equation/ooxml.py +62 -0
- docvortex/analyzers/native/office/errors.py +33 -0
- docvortex/analyzers/native/office/image.py +307 -0
- docvortex/analyzers/native/office/legacy/__init__.py +3 -0
- docvortex/analyzers/native/office/legacy/binary.py +43 -0
- docvortex/analyzers/native/office/legacy/officeart.py +362 -0
- docvortex/analyzers/native/office/legacy/ole.py +110 -0
- docvortex/analyzers/native/office/limits.py +12 -0
- docvortex/analyzers/native/office/odf/__init__.py +3 -0
- docvortex/analyzers/native/office/odf/chart.py +76 -0
- docvortex/analyzers/native/office/odf/constants.py +70 -0
- docvortex/analyzers/native/office/odf/converters.py +403 -0
- docvortex/analyzers/native/office/odf/errors.py +18 -0
- docvortex/analyzers/native/office/odf/metadata.py +91 -0
- docvortex/analyzers/native/office/odf/models.py +176 -0
- docvortex/analyzers/native/office/odf/package.py +270 -0
- docvortex/analyzers/native/office/odf/styles.py +329 -0
- docvortex/analyzers/native/office/odf/table.py +469 -0
- docvortex/analyzers/native/office/odf/text.py +1002 -0
- docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
- docvortex/analyzers/native/office/opc.py +38 -0
- docvortex/analyzers/native/office/ppt/__init__.py +3 -0
- docvortex/analyzers/native/office/ppt/models.py +118 -0
- docvortex/analyzers/native/office/ppt/parser.py +1895 -0
- docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
- docvortex/analyzers/native/office/ppt/records.py +131 -0
- docvortex/analyzers/native/office/ppt/style_text.py +247 -0
- docvortex/analyzers/native/office/pptx/__init__.py +3 -0
- docvortex/analyzers/native/office/pptx/context.py +113 -0
- docvortex/analyzers/native/office/pptx/lists.py +558 -0
- docvortex/analyzers/native/office/pptx/main.py +20 -0
- docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
- docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
- docvortex/analyzers/native/office/pptx/resources.py +329 -0
- docvortex/analyzers/native/office/pptx/shapes.py +393 -0
- docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
- docvortex/analyzers/native/office/pptx/titles.py +178 -0
- docvortex/analyzers/native/office/rich_text.py +420 -0
- docvortex/analyzers/native/office/rtf/__init__.py +3 -0
- docvortex/analyzers/native/office/rtf/converter.py +708 -0
- docvortex/analyzers/native/office/rtf/lexer.py +213 -0
- docvortex/analyzers/native/office/rtf/math.py +339 -0
- docvortex/analyzers/native/office/rtf/models.py +185 -0
- docvortex/analyzers/native/office/rtf/parser.py +1553 -0
- docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
- docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
- docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
- docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
- docvortex/analyzers/native/office/streams.py +18 -0
- docvortex/analyzers/native/office/xls/__init__.py +3 -0
- docvortex/analyzers/native/office/xls/chart.py +132 -0
- docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
- docvortex/analyzers/native/office/xls/models.py +109 -0
- docvortex/analyzers/native/office/xls/number_format.py +521 -0
- docvortex/analyzers/native/office/xls/parser.py +1145 -0
- docvortex/analyzers/native/office/xls/records.py +201 -0
- docvortex/analyzers/native/office/xls/strings.py +205 -0
- docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
- docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
- docvortex/analyzers/native/office/xlsx/main.py +20 -0
- docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
- docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
- docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
- docvortex/analyzers/native/pdf/__init__.py +3 -0
- docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
- docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
- docvortex/analyzers/native/pdf/code_blocks.py +535 -0
- docvortex/analyzers/native/pdf/formulas.py +1985 -0
- docvortex/analyzers/native/pdf/geometry.py +281 -0
- docvortex/analyzers/native/pdf/graphics.py +1501 -0
- docvortex/analyzers/native/pdf/index_blocks.py +268 -0
- docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
- docvortex/analyzers/native/pdf/inline/common.py +112 -0
- docvortex/analyzers/native/pdf/inline/detection.py +590 -0
- docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
- docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
- docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
- docvortex/analyzers/native/pdf/inline/types.py +385 -0
- docvortex/analyzers/native/pdf/line_layout.py +1106 -0
- docvortex/analyzers/native/pdf/line_merging.py +1223 -0
- docvortex/analyzers/native/pdf/models.py +246 -0
- docvortex/analyzers/native/pdf/native_text.py +1004 -0
- docvortex/analyzers/native/pdf/pipeline.py +1390 -0
- docvortex/analyzers/native/pdf/script_geometry.py +636 -0
- docvortex/analyzers/native/pdf/shared.py +30 -0
- docvortex/analyzers/native/pdf/spatial_text.py +383 -0
- docvortex/analyzers/native/pdf/table_annotations.py +446 -0
- docvortex/analyzers/native/pdf/table_constants.py +48 -0
- docvortex/analyzers/native/pdf/table_detection.py +227 -0
- docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
- docvortex/analyzers/native/pdf/table_geometry.py +40 -0
- docvortex/analyzers/native/pdf/table_materialization.py +444 -0
- docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
- docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
- docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
- docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
- docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
- docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
- docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
- docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
- docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
- docvortex/analyzers/native/pdf/table_rows.py +34 -0
- docvortex/analyzers/native/pdf/table_rules.py +1129 -0
- docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
- docvortex/analyzers/native/pdf/tables.py +147 -0
- docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
- docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
- docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
- docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
- docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
- docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
- docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
- docvortex/analyzers/native/pdf/text_blocks.py +82 -0
- docvortex/analyzers/native/pdf/text_styles.py +55 -0
- docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
- docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
- docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
- docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
- docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
- docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
- docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
- docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
- docvortex/analyzers/native/pdf/titles.py +75 -0
- docvortex/analyzers/native/pdf/typography.py +19 -0
- docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
- docvortex/api.py +180 -0
- docvortex/assets/__init__.py +5 -0
- docvortex/assets/store.py +51 -0
- docvortex/cli.py +54 -0
- docvortex/codecs/__init__.py +3 -0
- docvortex/codecs/html/__init__.py +22 -0
- docvortex/codecs/html/contracts.py +236 -0
- docvortex/codecs/html/materializer.py +331 -0
- docvortex/codecs/html/parser.py +763 -0
- docvortex/codecs/html/resources.py +34 -0
- docvortex/codecs/json.py +17 -0
- docvortex/content/__init__.py +5 -0
- docvortex/content/inline.py +248 -0
- docvortex/content/markup/__init__.py +44 -0
- docvortex/content/markup/anchors.py +188 -0
- docvortex/content/markup/formula.py +280 -0
- docvortex/content/markup/projector.py +1237 -0
- docvortex/content/markup/styles.py +327 -0
- docvortex/content/mathml.py +167 -0
- docvortex/content/normalization.py +188 -0
- docvortex/content/spans.py +183 -0
- docvortex/content/table/__init__.py +18 -0
- docvortex/content/table/blocks.py +152 -0
- docvortex/content/table/content.py +425 -0
- docvortex/content/table/document.py +104 -0
- docvortex/content/table/html.py +399 -0
- docvortex/content/table/models.py +76 -0
- docvortex/content/table/rules.py +42 -0
- docvortex/content/table/structure.py +221 -0
- docvortex/content/tree.py +5 -0
- docvortex/document/__init__.py +3 -0
- docvortex/document/contracts.py +22 -0
- docvortex/document/detection.py +389 -0
- docvortex/document/filetypes.py +175 -0
- docvortex/document/page_range.py +167 -0
- docvortex/document/pdf/__init__.py +19 -0
- docvortex/document/pdf/classify.py +1138 -0
- docvortex/document/pdf/constants.py +132 -0
- docvortex/document/pdf/diagnostics.py +350 -0
- docvortex/document/pdf/document.py +582 -0
- docvortex/document/pdf/font_runtime.py +335 -0
- docvortex/document/pdf/geometry.py +31 -0
- docvortex/document/pdf/images.py +654 -0
- docvortex/document/pdf/native_annotations.py +367 -0
- docvortex/document/pdf/native_contracts.py +169 -0
- docvortex/document/pdf/native_coordinates.py +216 -0
- docvortex/document/pdf/native_lifecycle.py +16 -0
- docvortex/document/pdf/native_objects.py +902 -0
- docvortex/document/pdf/native_text_geometry.py +315 -0
- docvortex/document/pdf/pdfium.py +325 -0
- docvortex/document/pdf/raster.py +46 -0
- docvortex/document/pdf/text/__init__.py +62 -0
- docvortex/document/pdf/text/contracts.py +211 -0
- docvortex/document/pdf/text/extract.py +165 -0
- docvortex/document/pdf/text/geometry.py +16 -0
- docvortex/document/pdf/text/groups.py +162 -0
- docvortex/document/pdf/visual_geometry.py +201 -0
- docvortex/document/pdf/visuals.py +343 -0
- docvortex/document/source.py +92 -0
- docvortex/errors.py +21 -0
- docvortex/export/__init__.py +3 -0
- docvortex/export/bundle.py +98 -0
- docvortex/export/files.py +66 -0
- docvortex/export/middle.py +208 -0
- docvortex/foundation/__init__.py +3 -0
- docvortex/foundation/geometry.py +125 -0
- docvortex/foundation/hyperlink.py +65 -0
- docvortex/foundation/image.py +48 -0
- docvortex/foundation/image_encoding.py +30 -0
- docvortex/foundation/image_payload.py +280 -0
- docvortex/foundation/language.py +92 -0
- docvortex/foundation/platform.py +38 -0
- docvortex/foundation/text.py +153 -0
- docvortex/foundation/type_identity.py +20 -0
- docvortex/foundation/xml_names.py +20 -0
- docvortex/options.py +30 -0
- docvortex/postprocess/__init__.py +3 -0
- docvortex/postprocess/content.py +53 -0
- docvortex/postprocess/document.py +19 -0
- docvortex/postprocess/lists.py +236 -0
- docvortex/postprocess/page_blocks.py +214 -0
- docvortex/postprocess/pages.py +95 -0
- docvortex/postprocess/paragraphs.py +580 -0
- docvortex/postprocess/visual.py +715 -0
- docvortex/render/__init__.py +48 -0
- docvortex/render/_internal/__init__.py +3 -0
- docvortex/render/_internal/common/__init__.py +3 -0
- docvortex/render/_internal/common/context.py +43 -0
- docvortex/render/_internal/common/html_table.py +178 -0
- docvortex/render/_internal/common/index.py +33 -0
- docvortex/render/_internal/common/list_items.py +158 -0
- docvortex/render/_internal/common/planner.py +140 -0
- docvortex/render/_internal/docx/__init__.py +3 -0
- docvortex/render/_internal/docx/assets.py +202 -0
- docvortex/render/_internal/docx/inline.py +434 -0
- docvortex/render/_internal/docx/math.py +220 -0
- docvortex/render/_internal/docx/renderer.py +905 -0
- docvortex/render/_internal/docx/styles.py +195 -0
- docvortex/render/_internal/docx/table.py +442 -0
- docvortex/render/_internal/epub/__init__.py +5 -0
- docvortex/render/_internal/epub/assets.py +173 -0
- docvortex/render/_internal/epub/package.py +249 -0
- docvortex/render/_internal/epub/renderer.py +1156 -0
- docvortex/render/_internal/html/__init__.py +3 -0
- docvortex/render/_internal/html/inline.py +346 -0
- docvortex/render/_internal/html/renderer.py +1041 -0
- docvortex/render/_internal/html/sanitizer.py +478 -0
- docvortex/render/_internal/html/table.py +121 -0
- docvortex/render/_internal/latex/__init__.py +1 -0
- docvortex/render/_internal/latex/assets.py +85 -0
- docvortex/render/_internal/latex/inline.py +145 -0
- docvortex/render/_internal/latex/renderer.py +506 -0
- docvortex/render/_internal/latex/table.py +347 -0
- docvortex/render/_internal/markdown/__init__.py +3 -0
- docvortex/render/_internal/markdown/assets.py +78 -0
- docvortex/render/_internal/markdown/blocks.py +635 -0
- docvortex/render/_internal/markdown/escaping.py +51 -0
- docvortex/render/_internal/markdown/inline.py +260 -0
- docvortex/render/_internal/markdown/renderer.py +93 -0
- docvortex/render/_internal/markdown/table.py +281 -0
- docvortex/render/_internal/pdf/__init__.py +3 -0
- docvortex/render/_internal/pdf/assets.py +197 -0
- docvortex/render/_internal/pdf/formula.py +417 -0
- docvortex/render/_internal/pdf/inline.py +343 -0
- docvortex/render/_internal/pdf/renderer.py +734 -0
- docvortex/render/_internal/pdf/styles.py +206 -0
- docvortex/render/_internal/pdf/table.py +272 -0
- docvortex/render/_internal/structured_content/__init__.py +3 -0
- docvortex/render/_internal/structured_content/renderer.py +193 -0
- docvortex/render/api.py +199 -0
- docvortex/render/contracts.py +205 -0
- docvortex/render/docx.py +38 -0
- docvortex/render/epub.py +35 -0
- docvortex/render/fragments.py +70 -0
- docvortex/render/html.py +29 -0
- docvortex/render/latex.py +24 -0
- docvortex/render/markdown.py +50 -0
- docvortex/render/pdf.py +25 -0
- docvortex/render/structured_content.py +23 -0
- docvortex/resources/epub/docvortex.css +91 -0
- docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
- docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
- docvortex/resources/fonts/NOTICE +190 -0
- docvortex/resources/fonts/manifest.json +9 -0
- docvortex/resources/html/docvortex.css +601 -0
- docvortex/resources/html/docvortex.min.css +1 -0
- docvortex/result.py +91 -0
- docvortex/schema.py +1141 -0
- docvortex/version.py +3 -0
- docvortex-0.2.1.dist-info/METADATA +193 -0
- docvortex-0.2.1.dist-info/RECORD +364 -0
- docvortex-0.2.1.dist-info/WHEEL +5 -0
- docvortex-0.2.1.dist-info/entry_points.txt +2 -0
- docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
- docvortex-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
"""在固定预算内读取 OFD ZIP/XML 包。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import posixpath
|
|
6
|
+
import re
|
|
7
|
+
from io import BytesIO
|
|
8
|
+
from pathlib import Path, PurePosixPath
|
|
9
|
+
from urllib.parse import unquote, urlsplit
|
|
10
|
+
from zipfile import BadZipFile, ZIP_DEFLATED, ZIP_STORED, ZipFile, ZipInfo
|
|
11
|
+
|
|
12
|
+
from loguru import logger
|
|
13
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
14
|
+
|
|
15
|
+
from .constants import (
|
|
16
|
+
MAX_ASSET_TOTAL_BYTES,
|
|
17
|
+
MAX_DOCUMENT_COUNT,
|
|
18
|
+
MAX_ENTRY_BYTES,
|
|
19
|
+
MAX_ENTRY_COUNT,
|
|
20
|
+
MAX_TOTAL_BYTES,
|
|
21
|
+
MAX_XML_DEPTH,
|
|
22
|
+
MAX_XML_NODES,
|
|
23
|
+
OFD_KNOWN_VERSIONS,
|
|
24
|
+
OFD_NAMESPACES,
|
|
25
|
+
)
|
|
26
|
+
from .errors import OfdEncryptedError, OfdParseError, OfdResourceLimitError
|
|
27
|
+
from .models import OfdDocumentRef
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def local_name(tag: object) -> str:
|
|
31
|
+
"""返回 XML 标签不含命名空间的本地名。"""
|
|
32
|
+
return tag.rsplit("}", 1)[-1] if isinstance(tag, str) else ""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def namespace_name(tag: object) -> str:
|
|
36
|
+
"""返回 Clark notation 标签中的命名空间。"""
|
|
37
|
+
if not isinstance(tag, str) or not tag.startswith("{"):
|
|
38
|
+
return ""
|
|
39
|
+
return tag[1:].split("}", 1)[0]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def first_child(element: etree._Element | None, name: str) -> etree._Element | None:
|
|
43
|
+
"""返回指定本地名的首个直接子元素。"""
|
|
44
|
+
if element is None:
|
|
45
|
+
return None
|
|
46
|
+
return next((child for child in element if local_name(child.tag) == name), None)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def first_descendant(element: etree._Element | None, name: str) -> etree._Element | None:
|
|
50
|
+
"""返回指定本地名的首个后代元素。"""
|
|
51
|
+
if element is None:
|
|
52
|
+
return None
|
|
53
|
+
return next((child for child in element.iter() if local_name(child.tag) == name), None)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def element_text(element: etree._Element | None) -> str:
|
|
57
|
+
"""返回元素折叠首尾空白后的完整文本。"""
|
|
58
|
+
return "" if element is None else "".join(element.itertext()).strip()
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def parse_int(value: object) -> int | None:
|
|
62
|
+
"""把非负整数字段安全解析为 Python int。"""
|
|
63
|
+
try:
|
|
64
|
+
parsed = int(str(value))
|
|
65
|
+
except (TypeError, ValueError):
|
|
66
|
+
return None
|
|
67
|
+
return parsed if parsed >= 0 else None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _xml_parser() -> etree.XMLParser:
|
|
71
|
+
"""为每个 OFD XML part 创建禁用实体、DTD 和网络的解析器。"""
|
|
72
|
+
return etree.XMLParser(
|
|
73
|
+
resolve_entities=False,
|
|
74
|
+
load_dtd=False,
|
|
75
|
+
no_network=True,
|
|
76
|
+
recover=False,
|
|
77
|
+
remove_blank_text=False,
|
|
78
|
+
huge_tree=False,
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class OfdPackage:
|
|
83
|
+
"""负责 OFD 包身份、成员访问和受限 XML 解析。"""
|
|
84
|
+
|
|
85
|
+
def __init__(self, file_bytes: bytes) -> None:
|
|
86
|
+
"""打开内存包并在读取正文前校验中央目录。"""
|
|
87
|
+
if len(file_bytes) > MAX_TOTAL_BYTES:
|
|
88
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_total_bytes={MAX_TOTAL_BYTES}")
|
|
89
|
+
try:
|
|
90
|
+
self._zip = ZipFile(BytesIO(file_bytes))
|
|
91
|
+
except (BadZipFile, OSError, ValueError) as exc:
|
|
92
|
+
raise OfdParseError(f"Malformed OFD package: {exc}") from exc
|
|
93
|
+
try:
|
|
94
|
+
self._infos = self._validate_members(self._zip.infolist())
|
|
95
|
+
except Exception:
|
|
96
|
+
self._zip.close()
|
|
97
|
+
raise
|
|
98
|
+
self._cache: dict[str, bytes] = {}
|
|
99
|
+
self._total_read = 0
|
|
100
|
+
self._asset_parts: set[str] = set()
|
|
101
|
+
self._asset_bytes = 0
|
|
102
|
+
self._root: etree._Element | None = None
|
|
103
|
+
|
|
104
|
+
@staticmethod
|
|
105
|
+
def _is_safe_member_name(name: str) -> bool:
|
|
106
|
+
"""判断 ZIP 成员是否为包内安全 POSIX 路径。"""
|
|
107
|
+
if not name or "\x00" in name or "\\" in name or name.startswith("/"):
|
|
108
|
+
return False
|
|
109
|
+
parts = PurePosixPath(name).parts
|
|
110
|
+
return bool(parts) and all(part not in {"", ".", ".."} for part in parts)
|
|
111
|
+
|
|
112
|
+
@classmethod
|
|
113
|
+
def _validate_members(cls, infos: list[ZipInfo]) -> dict[str, ZipInfo]:
|
|
114
|
+
"""校验成员数量、路径、加密、压缩方式和声明体积。"""
|
|
115
|
+
if len(infos) > MAX_ENTRY_COUNT:
|
|
116
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_entry_count={MAX_ENTRY_COUNT}")
|
|
117
|
+
members: dict[str, ZipInfo] = {}
|
|
118
|
+
total_size = 0
|
|
119
|
+
for info in infos:
|
|
120
|
+
name = info.filename
|
|
121
|
+
if not cls._is_safe_member_name(name):
|
|
122
|
+
raise OfdParseError(f"Malformed OFD package: unsafe member path {name!r}")
|
|
123
|
+
if name in members:
|
|
124
|
+
raise OfdParseError(f"Malformed OFD package: duplicate member {name!r}")
|
|
125
|
+
if info.flag_bits & 0x1:
|
|
126
|
+
raise OfdEncryptedError(f"Encrypted OFD ZIP member is not supported: {name!r}")
|
|
127
|
+
if info.compress_type not in {ZIP_STORED, ZIP_DEFLATED}:
|
|
128
|
+
raise OfdParseError(f"Malformed OFD package: unsupported ZIP compression for {name!r}")
|
|
129
|
+
if info.file_size > MAX_ENTRY_BYTES:
|
|
130
|
+
raise OfdResourceLimitError(
|
|
131
|
+
f"OFD resource limit exceeded: member {name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}"
|
|
132
|
+
)
|
|
133
|
+
total_size += info.file_size
|
|
134
|
+
if total_size > MAX_TOTAL_BYTES:
|
|
135
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_total_bytes={MAX_TOTAL_BYTES}")
|
|
136
|
+
members[name] = info
|
|
137
|
+
if "OFD.xml" not in members:
|
|
138
|
+
raise OfdParseError("Malformed OFD package: missing required root 'OFD.xml'")
|
|
139
|
+
return members
|
|
140
|
+
|
|
141
|
+
def has_part(self, part_name: str) -> bool:
|
|
142
|
+
"""返回包内是否存在指定规范成员。"""
|
|
143
|
+
return part_name in self._infos
|
|
144
|
+
|
|
145
|
+
def read_part(self, part_name: str, *, required: bool = False, asset: bool = False) -> bytes | None:
|
|
146
|
+
"""在成员和累计预算内读取一个包内 part。"""
|
|
147
|
+
info = self._infos.get(part_name)
|
|
148
|
+
if info is None:
|
|
149
|
+
if required:
|
|
150
|
+
raise OfdParseError(f"Malformed OFD package: missing required part {part_name!r}")
|
|
151
|
+
return None
|
|
152
|
+
if part_name in self._cache:
|
|
153
|
+
data = self._cache[part_name]
|
|
154
|
+
if asset:
|
|
155
|
+
self._charge_asset(part_name, len(data))
|
|
156
|
+
return data
|
|
157
|
+
try:
|
|
158
|
+
with self._zip.open(info) as source:
|
|
159
|
+
data = source.read(MAX_ENTRY_BYTES + 1)
|
|
160
|
+
except (BadZipFile, OSError, RuntimeError, ValueError) as exc:
|
|
161
|
+
if required:
|
|
162
|
+
raise OfdParseError(f"Malformed OFD package: cannot read {part_name!r}: {exc}") from exc
|
|
163
|
+
return None
|
|
164
|
+
if len(data) > MAX_ENTRY_BYTES:
|
|
165
|
+
raise OfdResourceLimitError(
|
|
166
|
+
f"OFD resource limit exceeded: member {part_name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}"
|
|
167
|
+
)
|
|
168
|
+
self._total_read += len(data)
|
|
169
|
+
if self._total_read > MAX_TOTAL_BYTES:
|
|
170
|
+
raise OfdResourceLimitError(
|
|
171
|
+
f"OFD resource limit exceeded while reading {part_name!r}: max_total_bytes={MAX_TOTAL_BYTES}"
|
|
172
|
+
)
|
|
173
|
+
if asset:
|
|
174
|
+
self._charge_asset(part_name, len(data))
|
|
175
|
+
self._cache[part_name] = data
|
|
176
|
+
return data
|
|
177
|
+
|
|
178
|
+
def _charge_asset(self, part_name: str, byte_count: int) -> None:
|
|
179
|
+
"""按唯一成员累计保留的资源字节。"""
|
|
180
|
+
if part_name in self._asset_parts:
|
|
181
|
+
return
|
|
182
|
+
self._asset_parts.add(part_name)
|
|
183
|
+
self._asset_bytes += byte_count
|
|
184
|
+
if self._asset_bytes > MAX_ASSET_TOTAL_BYTES:
|
|
185
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_asset_total_bytes={MAX_ASSET_TOTAL_BYTES}")
|
|
186
|
+
|
|
187
|
+
def xml_part(self, part_name: str, *, required: bool = False) -> etree._Element | None:
|
|
188
|
+
"""禁用 DTD/实体后解析 XML,并校验节点数量与深度。"""
|
|
189
|
+
data = self.read_part(part_name, required=required)
|
|
190
|
+
if data is None:
|
|
191
|
+
return None
|
|
192
|
+
try:
|
|
193
|
+
root = etree.fromstring(data, parser=_xml_parser())
|
|
194
|
+
except (etree.XMLSyntaxError, ValueError) as exc:
|
|
195
|
+
if required:
|
|
196
|
+
raise OfdParseError(f"Malformed OFD package: invalid XML part {part_name!r}: {exc}") from exc
|
|
197
|
+
return None
|
|
198
|
+
if root.getroottree().docinfo.doctype:
|
|
199
|
+
raise OfdParseError(f"Malformed OFD package: DTD is not allowed in {part_name!r}")
|
|
200
|
+
self._validate_xml_shape(root, part_name)
|
|
201
|
+
return root
|
|
202
|
+
|
|
203
|
+
@staticmethod
|
|
204
|
+
def _validate_xml_shape(root: etree._Element, part_name: str) -> None:
|
|
205
|
+
"""迭代校验 XML 节点数和最大深度。"""
|
|
206
|
+
node_count = 0
|
|
207
|
+
stack: list[tuple[etree._Element, int]] = [(root, 1)]
|
|
208
|
+
while stack:
|
|
209
|
+
element, depth = stack.pop()
|
|
210
|
+
node_count += 1
|
|
211
|
+
if node_count > MAX_XML_NODES:
|
|
212
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: {part_name!r} exceeds max_xml_nodes={MAX_XML_NODES}")
|
|
213
|
+
if depth > MAX_XML_DEPTH:
|
|
214
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: {part_name!r} exceeds max_xml_depth={MAX_XML_DEPTH}")
|
|
215
|
+
stack.extend((child, depth + 1) for child in element if isinstance(child.tag, str))
|
|
216
|
+
|
|
217
|
+
def root(self) -> etree._Element:
|
|
218
|
+
"""读取并验证 OFD.xml 根节点、命名空间、版本和文档类型。"""
|
|
219
|
+
if self._root is not None:
|
|
220
|
+
return self._root
|
|
221
|
+
root = self.xml_part("OFD.xml", required=True)
|
|
222
|
+
assert root is not None
|
|
223
|
+
namespace = namespace_name(root.tag)
|
|
224
|
+
if local_name(root.tag) != "OFD" or namespace not in OFD_NAMESPACES:
|
|
225
|
+
raise OfdParseError(f"Malformed OFD package: unsupported root namespace {namespace!r}")
|
|
226
|
+
version = (root.get("Version") or "").strip()
|
|
227
|
+
if not re.fullmatch(r"1(?:\.\d+)?", version):
|
|
228
|
+
raise OfdParseError(f"Unsupported OFD version: {version or '<missing>'}")
|
|
229
|
+
if version not in OFD_KNOWN_VERSIONS:
|
|
230
|
+
logger.warning(f"OFD_COMPAT_VERSION: parsing unrecognized 1.x version {version!r}")
|
|
231
|
+
doc_type = (root.get("DocType") or "OFD").strip().upper()
|
|
232
|
+
if doc_type not in {"OFD", "OFD-A"}:
|
|
233
|
+
raise OfdParseError(f"Unsupported OFD DocType: {doc_type!r}")
|
|
234
|
+
if not any(local_name(child.tag) == "DocBody" for child in root):
|
|
235
|
+
raise OfdParseError("Malformed OFD package: OFD.xml has no DocBody")
|
|
236
|
+
self._root = root
|
|
237
|
+
return root
|
|
238
|
+
|
|
239
|
+
def resolve_reference(self, base_part: str, location: str | None) -> str | None:
|
|
240
|
+
"""解析大小写敏感的 ST_Loc,并拒绝包外与网络位置。"""
|
|
241
|
+
raw = unquote((location or "").strip()).replace("\\", "/")
|
|
242
|
+
if not raw:
|
|
243
|
+
return None
|
|
244
|
+
try:
|
|
245
|
+
parsed = urlsplit(raw)
|
|
246
|
+
except ValueError:
|
|
247
|
+
return None
|
|
248
|
+
if parsed.scheme or parsed.netloc or parsed.query:
|
|
249
|
+
return None
|
|
250
|
+
raw_path = parsed.path
|
|
251
|
+
if raw_path.startswith("/"):
|
|
252
|
+
resolved = posixpath.normpath(raw_path).lstrip("/")
|
|
253
|
+
else:
|
|
254
|
+
resolved = posixpath.normpath(posixpath.join(posixpath.dirname(base_part), raw_path))
|
|
255
|
+
if resolved in {"", ".", ".."} or resolved.startswith("../") or resolved.startswith("/"):
|
|
256
|
+
return None
|
|
257
|
+
return resolved.removeprefix("./")
|
|
258
|
+
|
|
259
|
+
def document_refs(self) -> list[OfdDocumentRef]:
|
|
260
|
+
"""按 DocBody 声明顺序返回全部文档入口。"""
|
|
261
|
+
refs: list[OfdDocumentRef] = []
|
|
262
|
+
for body in self.root():
|
|
263
|
+
if local_name(body.tag) != "DocBody":
|
|
264
|
+
continue
|
|
265
|
+
if len(refs) >= MAX_DOCUMENT_COUNT:
|
|
266
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_document_count={MAX_DOCUMENT_COUNT}")
|
|
267
|
+
doc_root = first_child(body, "DocRoot")
|
|
268
|
+
document_part = self.resolve_reference("OFD.xml", element_text(doc_root))
|
|
269
|
+
if document_part is None:
|
|
270
|
+
raise OfdParseError("Malformed OFD package: DocBody has invalid DocRoot")
|
|
271
|
+
signatures = first_child(body, "Signatures")
|
|
272
|
+
signatures_part = self.resolve_reference("OFD.xml", element_text(signatures)) if signatures is not None else None
|
|
273
|
+
metadata: dict[str, str] = {}
|
|
274
|
+
doc_info = first_child(body, "DocInfo")
|
|
275
|
+
if doc_info is not None:
|
|
276
|
+
for key in ("Title", "Author", "Subject", "Keywords", "DocUsage", "Creator", "CreatorVersion"):
|
|
277
|
+
value = element_text(first_child(doc_info, key))
|
|
278
|
+
if value:
|
|
279
|
+
metadata[key] = value
|
|
280
|
+
refs.append(OfdDocumentRef(document_part=document_part, signatures_part=signatures_part, metadata=metadata))
|
|
281
|
+
return refs
|
|
282
|
+
|
|
283
|
+
def close(self) -> None:
|
|
284
|
+
"""关闭底层 ZipFile。"""
|
|
285
|
+
self._zip.close()
|
|
286
|
+
|
|
287
|
+
def __enter__(self) -> OfdPackage:
|
|
288
|
+
"""返回当前包以支持 with 生命周期。"""
|
|
289
|
+
return self
|
|
290
|
+
|
|
291
|
+
def __exit__(self, _exc_type: object, _exc: object, _traceback: object) -> None:
|
|
292
|
+
"""退出 with 块时关闭包。"""
|
|
293
|
+
self.close()
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def detect_ofd(file_bytes: bytes) -> bool:
|
|
297
|
+
"""按受限 OFD 包身份识别内存字节。"""
|
|
298
|
+
try:
|
|
299
|
+
with OfdPackage(file_bytes) as package:
|
|
300
|
+
package.root()
|
|
301
|
+
return True
|
|
302
|
+
except (BadZipFile, OfdParseError, OSError, ValueError):
|
|
303
|
+
return False
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def detect_ofd_path(file_path: str | Path) -> bool:
|
|
307
|
+
"""直接从 ZIP 路径读取有限 OFD.xml 并验证包身份。"""
|
|
308
|
+
try:
|
|
309
|
+
path = Path(file_path)
|
|
310
|
+
if path.stat().st_size > MAX_TOTAL_BYTES:
|
|
311
|
+
return False
|
|
312
|
+
with ZipFile(path) as package:
|
|
313
|
+
info = package.getinfo("OFD.xml")
|
|
314
|
+
if info.file_size > MAX_ENTRY_BYTES or info.flag_bits & 0x1 or info.compress_type not in {ZIP_STORED, ZIP_DEFLATED}:
|
|
315
|
+
return False
|
|
316
|
+
with package.open(info) as source:
|
|
317
|
+
data = source.read(MAX_ENTRY_BYTES + 1)
|
|
318
|
+
if len(data) > MAX_ENTRY_BYTES:
|
|
319
|
+
return False
|
|
320
|
+
root = etree.fromstring(data, parser=_xml_parser())
|
|
321
|
+
namespace = namespace_name(root.tag)
|
|
322
|
+
version = (root.get("Version") or "").strip()
|
|
323
|
+
doc_type = (root.get("DocType") or "OFD").strip().upper()
|
|
324
|
+
return (
|
|
325
|
+
not root.getroottree().docinfo.doctype
|
|
326
|
+
and local_name(root.tag) == "OFD"
|
|
327
|
+
and namespace in OFD_NAMESPACES
|
|
328
|
+
and re.fullmatch(r"1(?:\.\d+)?", version) is not None
|
|
329
|
+
and doc_type in {"OFD", "OFD-A"}
|
|
330
|
+
and any(local_name(child.tag) == "DocBody" for child in root)
|
|
331
|
+
)
|
|
332
|
+
except (BadZipFile, KeyError, OSError, RuntimeError, ValueError, etree.XMLSyntaxError):
|
|
333
|
+
return False
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
__all__ = [
|
|
337
|
+
"OfdPackage",
|
|
338
|
+
"detect_ofd",
|
|
339
|
+
"detect_ofd_path",
|
|
340
|
+
"element_text",
|
|
341
|
+
"first_child",
|
|
342
|
+
"first_descendant",
|
|
343
|
+
"local_name",
|
|
344
|
+
"namespace_name",
|
|
345
|
+
"parse_int",
|
|
346
|
+
]
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""解析 OFD PathObject 并提取表格可用的轴向线段。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
from loguru import logger
|
|
10
|
+
from lxml import etree # type: ignore[reportMissingImports]
|
|
11
|
+
|
|
12
|
+
from ....schema import BBox
|
|
13
|
+
from .constants import MAX_PATH_COMMANDS, MAX_PATH_TOKENS
|
|
14
|
+
from .errors import OfdResourceLimitError
|
|
15
|
+
from .geometry import Affine, bbox_intersection, parse_affine, parse_st_box, transform_bbox
|
|
16
|
+
from .models import AxisLine
|
|
17
|
+
from .package import element_text, first_descendant, parse_int
|
|
18
|
+
|
|
19
|
+
_TOKEN_RE = re.compile(r"CM|[SMLQBAC]|[-+]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][-+]?\d+)?")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(slots=True)
|
|
23
|
+
class OfdPathBudget:
|
|
24
|
+
"""累计限制紧缩路径 token 与命令数量。"""
|
|
25
|
+
|
|
26
|
+
command_count: int = 0
|
|
27
|
+
token_count: int = 0
|
|
28
|
+
|
|
29
|
+
def charge_token(self) -> None:
|
|
30
|
+
"""累计实际扫描的路径 token 并在超限时失败。"""
|
|
31
|
+
self.token_count += 1
|
|
32
|
+
if self.token_count > MAX_PATH_TOKENS:
|
|
33
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_path_tokens={MAX_PATH_TOKENS}")
|
|
34
|
+
|
|
35
|
+
def charge_command(self) -> None:
|
|
36
|
+
"""累计实际扫描的路径命令并在超限时失败。"""
|
|
37
|
+
self.command_count += 1
|
|
38
|
+
if self.command_count > MAX_PATH_COMMANDS:
|
|
39
|
+
raise OfdResourceLimitError(f"OFD resource limit exceeded: max_path_commands={MAX_PATH_COMMANDS}")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _finite_float(value: str) -> float | None:
|
|
43
|
+
"""把路径 token 转换为有限浮点数。"""
|
|
44
|
+
try:
|
|
45
|
+
parsed = float(value)
|
|
46
|
+
except ValueError:
|
|
47
|
+
return None
|
|
48
|
+
return parsed if math.isfinite(parsed) else None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _line_bbox(first: tuple[float, float], second: tuple[float, float], width: float) -> tuple[BBox, str] | None:
|
|
52
|
+
"""把近似水平或垂直线段转换为非退化 bbox。"""
|
|
53
|
+
dx = second[0] - first[0]
|
|
54
|
+
dy = second[1] - first[1]
|
|
55
|
+
length = math.hypot(dx, dy)
|
|
56
|
+
if length <= 0:
|
|
57
|
+
return None
|
|
58
|
+
tolerance = max(0.2, 0.02 * length)
|
|
59
|
+
thickness = max(width, 0.1)
|
|
60
|
+
if abs(dy) <= tolerance:
|
|
61
|
+
return (
|
|
62
|
+
(
|
|
63
|
+
min(first[0], second[0]),
|
|
64
|
+
min(first[1], second[1]) - thickness / 2,
|
|
65
|
+
max(first[0], second[0]),
|
|
66
|
+
max(first[1], second[1]) + thickness / 2,
|
|
67
|
+
),
|
|
68
|
+
"horizontal",
|
|
69
|
+
)
|
|
70
|
+
if abs(dx) <= tolerance:
|
|
71
|
+
return (
|
|
72
|
+
(
|
|
73
|
+
min(first[0], second[0]) - thickness / 2,
|
|
74
|
+
min(first[1], second[1]),
|
|
75
|
+
max(first[0], second[0]) + thickness / 2,
|
|
76
|
+
max(first[1], second[1]),
|
|
77
|
+
),
|
|
78
|
+
"vertical",
|
|
79
|
+
)
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _segments(value: str, transform: Affine, budget: OfdPathBudget) -> list[tuple[tuple[float, float], tuple[float, float]]]:
|
|
84
|
+
"""流式提取 S/M/L/CM 直线段,曲线命令只推进当前位置。"""
|
|
85
|
+
result: list[tuple[tuple[float, float], tuple[float, float]]] = []
|
|
86
|
+
current: tuple[float, float] | None = None
|
|
87
|
+
subpath_start: tuple[float, float] | None = None
|
|
88
|
+
command: str | None = None
|
|
89
|
+
parameters: list[float] = []
|
|
90
|
+
commands = {"S", "M", "L", "Q", "B", "A", "C", "CM"}
|
|
91
|
+
arity = {"S": 2, "M": 2, "L": 2, "Q": 4, "B": 6, "A": 7, "CM": 2}
|
|
92
|
+
for match in _TOKEN_RE.finditer(value):
|
|
93
|
+
budget.charge_token()
|
|
94
|
+
token = match.group()
|
|
95
|
+
if token in commands:
|
|
96
|
+
if parameters:
|
|
97
|
+
return []
|
|
98
|
+
budget.charge_command()
|
|
99
|
+
command = token
|
|
100
|
+
if command == "C":
|
|
101
|
+
if current is not None and subpath_start is not None and current != subpath_start:
|
|
102
|
+
result.append((transform.apply(current), transform.apply(subpath_start)))
|
|
103
|
+
current = subpath_start
|
|
104
|
+
command = None
|
|
105
|
+
continue
|
|
106
|
+
if command is None or command == "C":
|
|
107
|
+
return []
|
|
108
|
+
parsed = _finite_float(token)
|
|
109
|
+
if parsed is None:
|
|
110
|
+
return []
|
|
111
|
+
parameters.append(parsed)
|
|
112
|
+
if len(parameters) < arity[command]:
|
|
113
|
+
continue
|
|
114
|
+
endpoint = (parameters[-2], parameters[-1])
|
|
115
|
+
if command in {"S", "M"}:
|
|
116
|
+
current = endpoint
|
|
117
|
+
subpath_start = endpoint
|
|
118
|
+
elif command in {"L", "CM"}:
|
|
119
|
+
if current is not None:
|
|
120
|
+
result.append((transform.apply(current), transform.apply(endpoint)))
|
|
121
|
+
current = endpoint
|
|
122
|
+
else:
|
|
123
|
+
current = endpoint
|
|
124
|
+
parameters.clear()
|
|
125
|
+
return [] if parameters else result
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def build_axis_lines(
|
|
129
|
+
path_object: etree._Element,
|
|
130
|
+
*,
|
|
131
|
+
parent_transform: Affine,
|
|
132
|
+
parent_clip: BBox,
|
|
133
|
+
paint_order: int,
|
|
134
|
+
template_id: int | None,
|
|
135
|
+
budget: OfdPathBudget,
|
|
136
|
+
resolved_style: dict[str, str] | None = None,
|
|
137
|
+
) -> list[AxisLine]:
|
|
138
|
+
"""从一个 PathObject 提取可见轴向线段。"""
|
|
139
|
+
style = resolved_style or {}
|
|
140
|
+
if (style.get("Visible") or path_object.get("Visible") or "true").casefold() in {"false", "0"}:
|
|
141
|
+
return []
|
|
142
|
+
if (style.get("Alpha") or path_object.get("Alpha") or "255").strip() == "0":
|
|
143
|
+
return []
|
|
144
|
+
boundary = parse_st_box(path_object.get("Boundary"))
|
|
145
|
+
if boundary is None:
|
|
146
|
+
return []
|
|
147
|
+
boundary_page = transform_bbox(boundary, parent_transform)
|
|
148
|
+
if boundary_page is None:
|
|
149
|
+
return []
|
|
150
|
+
object_clip = bbox_intersection(boundary_page, parent_clip)
|
|
151
|
+
if object_clip is None:
|
|
152
|
+
return []
|
|
153
|
+
try:
|
|
154
|
+
width = max(0.1, float(style.get("LineWidth") or path_object.get("LineWidth") or 0.353))
|
|
155
|
+
except ValueError:
|
|
156
|
+
width = 0.353
|
|
157
|
+
object_transform = parent_transform.compose(Affine.translation(boundary[0], boundary[1])).compose(
|
|
158
|
+
parse_affine(path_object.get("CTM"))
|
|
159
|
+
)
|
|
160
|
+
extracted: list[AxisLine] = []
|
|
161
|
+
data = first_descendant(path_object, "AbbreviatedData")
|
|
162
|
+
segments = _segments(element_text(data), object_transform, budget) if data is not None else []
|
|
163
|
+
for first, second in segments:
|
|
164
|
+
line = _line_bbox(first, second, width)
|
|
165
|
+
if line is None:
|
|
166
|
+
continue
|
|
167
|
+
line_bbox, orientation = line
|
|
168
|
+
clipped = bbox_intersection(line_bbox, object_clip)
|
|
169
|
+
if clipped is not None:
|
|
170
|
+
extracted.append(
|
|
171
|
+
AxisLine(
|
|
172
|
+
bbox=clipped,
|
|
173
|
+
orientation=orientation,
|
|
174
|
+
width=width,
|
|
175
|
+
paint_order=paint_order,
|
|
176
|
+
template_id=template_id,
|
|
177
|
+
)
|
|
178
|
+
)
|
|
179
|
+
boundary_width = boundary_page[2] - boundary_page[0]
|
|
180
|
+
boundary_height = boundary_page[3] - boundary_page[1]
|
|
181
|
+
if (
|
|
182
|
+
not extracted
|
|
183
|
+
and not segments
|
|
184
|
+
and max(boundary_width, boundary_height) >= 5 * max(min(boundary_width, boundary_height), 0.01)
|
|
185
|
+
):
|
|
186
|
+
orientation = "horizontal" if boundary_width >= boundary_height else "vertical"
|
|
187
|
+
extracted.append(
|
|
188
|
+
AxisLine(
|
|
189
|
+
bbox=object_clip,
|
|
190
|
+
orientation=orientation,
|
|
191
|
+
width=width,
|
|
192
|
+
paint_order=paint_order,
|
|
193
|
+
template_id=template_id,
|
|
194
|
+
)
|
|
195
|
+
)
|
|
196
|
+
if not extracted and parse_int(path_object.get("ID")) is None:
|
|
197
|
+
logger.debug("OFD_PATH_SKIPPED: path without stable ID produced no axis lines")
|
|
198
|
+
return extracted
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
__all__ = ["OfdPathBudget", "build_axis_lines"]
|