docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,346 @@
1
+ """在固定预算内读取 OFD ZIP/XML 包。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import posixpath
6
+ import re
7
+ from io import BytesIO
8
+ from pathlib import Path, PurePosixPath
9
+ from urllib.parse import unquote, urlsplit
10
+ from zipfile import BadZipFile, ZIP_DEFLATED, ZIP_STORED, ZipFile, ZipInfo
11
+
12
+ from loguru import logger
13
+ from lxml import etree # type: ignore[reportMissingImports]
14
+
15
+ from .constants import (
16
+ MAX_ASSET_TOTAL_BYTES,
17
+ MAX_DOCUMENT_COUNT,
18
+ MAX_ENTRY_BYTES,
19
+ MAX_ENTRY_COUNT,
20
+ MAX_TOTAL_BYTES,
21
+ MAX_XML_DEPTH,
22
+ MAX_XML_NODES,
23
+ OFD_KNOWN_VERSIONS,
24
+ OFD_NAMESPACES,
25
+ )
26
+ from .errors import OfdEncryptedError, OfdParseError, OfdResourceLimitError
27
+ from .models import OfdDocumentRef
28
+
29
+
30
+ def local_name(tag: object) -> str:
31
+ """返回 XML 标签不含命名空间的本地名。"""
32
+ return tag.rsplit("}", 1)[-1] if isinstance(tag, str) else ""
33
+
34
+
35
+ def namespace_name(tag: object) -> str:
36
+ """返回 Clark notation 标签中的命名空间。"""
37
+ if not isinstance(tag, str) or not tag.startswith("{"):
38
+ return ""
39
+ return tag[1:].split("}", 1)[0]
40
+
41
+
42
+ def first_child(element: etree._Element | None, name: str) -> etree._Element | None:
43
+ """返回指定本地名的首个直接子元素。"""
44
+ if element is None:
45
+ return None
46
+ return next((child for child in element if local_name(child.tag) == name), None)
47
+
48
+
49
+ def first_descendant(element: etree._Element | None, name: str) -> etree._Element | None:
50
+ """返回指定本地名的首个后代元素。"""
51
+ if element is None:
52
+ return None
53
+ return next((child for child in element.iter() if local_name(child.tag) == name), None)
54
+
55
+
56
+ def element_text(element: etree._Element | None) -> str:
57
+ """返回元素折叠首尾空白后的完整文本。"""
58
+ return "" if element is None else "".join(element.itertext()).strip()
59
+
60
+
61
+ def parse_int(value: object) -> int | None:
62
+ """把非负整数字段安全解析为 Python int。"""
63
+ try:
64
+ parsed = int(str(value))
65
+ except (TypeError, ValueError):
66
+ return None
67
+ return parsed if parsed >= 0 else None
68
+
69
+
70
+ def _xml_parser() -> etree.XMLParser:
71
+ """为每个 OFD XML part 创建禁用实体、DTD 和网络的解析器。"""
72
+ return etree.XMLParser(
73
+ resolve_entities=False,
74
+ load_dtd=False,
75
+ no_network=True,
76
+ recover=False,
77
+ remove_blank_text=False,
78
+ huge_tree=False,
79
+ )
80
+
81
+
82
+ class OfdPackage:
83
+ """负责 OFD 包身份、成员访问和受限 XML 解析。"""
84
+
85
+ def __init__(self, file_bytes: bytes) -> None:
86
+ """打开内存包并在读取正文前校验中央目录。"""
87
+ if len(file_bytes) > MAX_TOTAL_BYTES:
88
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_total_bytes={MAX_TOTAL_BYTES}")
89
+ try:
90
+ self._zip = ZipFile(BytesIO(file_bytes))
91
+ except (BadZipFile, OSError, ValueError) as exc:
92
+ raise OfdParseError(f"Malformed OFD package: {exc}") from exc
93
+ try:
94
+ self._infos = self._validate_members(self._zip.infolist())
95
+ except Exception:
96
+ self._zip.close()
97
+ raise
98
+ self._cache: dict[str, bytes] = {}
99
+ self._total_read = 0
100
+ self._asset_parts: set[str] = set()
101
+ self._asset_bytes = 0
102
+ self._root: etree._Element | None = None
103
+
104
+ @staticmethod
105
+ def _is_safe_member_name(name: str) -> bool:
106
+ """判断 ZIP 成员是否为包内安全 POSIX 路径。"""
107
+ if not name or "\x00" in name or "\\" in name or name.startswith("/"):
108
+ return False
109
+ parts = PurePosixPath(name).parts
110
+ return bool(parts) and all(part not in {"", ".", ".."} for part in parts)
111
+
112
+ @classmethod
113
+ def _validate_members(cls, infos: list[ZipInfo]) -> dict[str, ZipInfo]:
114
+ """校验成员数量、路径、加密、压缩方式和声明体积。"""
115
+ if len(infos) > MAX_ENTRY_COUNT:
116
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_entry_count={MAX_ENTRY_COUNT}")
117
+ members: dict[str, ZipInfo] = {}
118
+ total_size = 0
119
+ for info in infos:
120
+ name = info.filename
121
+ if not cls._is_safe_member_name(name):
122
+ raise OfdParseError(f"Malformed OFD package: unsafe member path {name!r}")
123
+ if name in members:
124
+ raise OfdParseError(f"Malformed OFD package: duplicate member {name!r}")
125
+ if info.flag_bits & 0x1:
126
+ raise OfdEncryptedError(f"Encrypted OFD ZIP member is not supported: {name!r}")
127
+ if info.compress_type not in {ZIP_STORED, ZIP_DEFLATED}:
128
+ raise OfdParseError(f"Malformed OFD package: unsupported ZIP compression for {name!r}")
129
+ if info.file_size > MAX_ENTRY_BYTES:
130
+ raise OfdResourceLimitError(
131
+ f"OFD resource limit exceeded: member {name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}"
132
+ )
133
+ total_size += info.file_size
134
+ if total_size > MAX_TOTAL_BYTES:
135
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_total_bytes={MAX_TOTAL_BYTES}")
136
+ members[name] = info
137
+ if "OFD.xml" not in members:
138
+ raise OfdParseError("Malformed OFD package: missing required root 'OFD.xml'")
139
+ return members
140
+
141
+ def has_part(self, part_name: str) -> bool:
142
+ """返回包内是否存在指定规范成员。"""
143
+ return part_name in self._infos
144
+
145
+ def read_part(self, part_name: str, *, required: bool = False, asset: bool = False) -> bytes | None:
146
+ """在成员和累计预算内读取一个包内 part。"""
147
+ info = self._infos.get(part_name)
148
+ if info is None:
149
+ if required:
150
+ raise OfdParseError(f"Malformed OFD package: missing required part {part_name!r}")
151
+ return None
152
+ if part_name in self._cache:
153
+ data = self._cache[part_name]
154
+ if asset:
155
+ self._charge_asset(part_name, len(data))
156
+ return data
157
+ try:
158
+ with self._zip.open(info) as source:
159
+ data = source.read(MAX_ENTRY_BYTES + 1)
160
+ except (BadZipFile, OSError, RuntimeError, ValueError) as exc:
161
+ if required:
162
+ raise OfdParseError(f"Malformed OFD package: cannot read {part_name!r}: {exc}") from exc
163
+ return None
164
+ if len(data) > MAX_ENTRY_BYTES:
165
+ raise OfdResourceLimitError(
166
+ f"OFD resource limit exceeded: member {part_name!r} exceeds max_entry_bytes={MAX_ENTRY_BYTES}"
167
+ )
168
+ self._total_read += len(data)
169
+ if self._total_read > MAX_TOTAL_BYTES:
170
+ raise OfdResourceLimitError(
171
+ f"OFD resource limit exceeded while reading {part_name!r}: max_total_bytes={MAX_TOTAL_BYTES}"
172
+ )
173
+ if asset:
174
+ self._charge_asset(part_name, len(data))
175
+ self._cache[part_name] = data
176
+ return data
177
+
178
+ def _charge_asset(self, part_name: str, byte_count: int) -> None:
179
+ """按唯一成员累计保留的资源字节。"""
180
+ if part_name in self._asset_parts:
181
+ return
182
+ self._asset_parts.add(part_name)
183
+ self._asset_bytes += byte_count
184
+ if self._asset_bytes > MAX_ASSET_TOTAL_BYTES:
185
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_asset_total_bytes={MAX_ASSET_TOTAL_BYTES}")
186
+
187
+ def xml_part(self, part_name: str, *, required: bool = False) -> etree._Element | None:
188
+ """禁用 DTD/实体后解析 XML,并校验节点数量与深度。"""
189
+ data = self.read_part(part_name, required=required)
190
+ if data is None:
191
+ return None
192
+ try:
193
+ root = etree.fromstring(data, parser=_xml_parser())
194
+ except (etree.XMLSyntaxError, ValueError) as exc:
195
+ if required:
196
+ raise OfdParseError(f"Malformed OFD package: invalid XML part {part_name!r}: {exc}") from exc
197
+ return None
198
+ if root.getroottree().docinfo.doctype:
199
+ raise OfdParseError(f"Malformed OFD package: DTD is not allowed in {part_name!r}")
200
+ self._validate_xml_shape(root, part_name)
201
+ return root
202
+
203
+ @staticmethod
204
+ def _validate_xml_shape(root: etree._Element, part_name: str) -> None:
205
+ """迭代校验 XML 节点数和最大深度。"""
206
+ node_count = 0
207
+ stack: list[tuple[etree._Element, int]] = [(root, 1)]
208
+ while stack:
209
+ element, depth = stack.pop()
210
+ node_count += 1
211
+ if node_count > MAX_XML_NODES:
212
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: {part_name!r} exceeds max_xml_nodes={MAX_XML_NODES}")
213
+ if depth > MAX_XML_DEPTH:
214
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: {part_name!r} exceeds max_xml_depth={MAX_XML_DEPTH}")
215
+ stack.extend((child, depth + 1) for child in element if isinstance(child.tag, str))
216
+
217
+ def root(self) -> etree._Element:
218
+ """读取并验证 OFD.xml 根节点、命名空间、版本和文档类型。"""
219
+ if self._root is not None:
220
+ return self._root
221
+ root = self.xml_part("OFD.xml", required=True)
222
+ assert root is not None
223
+ namespace = namespace_name(root.tag)
224
+ if local_name(root.tag) != "OFD" or namespace not in OFD_NAMESPACES:
225
+ raise OfdParseError(f"Malformed OFD package: unsupported root namespace {namespace!r}")
226
+ version = (root.get("Version") or "").strip()
227
+ if not re.fullmatch(r"1(?:\.\d+)?", version):
228
+ raise OfdParseError(f"Unsupported OFD version: {version or '<missing>'}")
229
+ if version not in OFD_KNOWN_VERSIONS:
230
+ logger.warning(f"OFD_COMPAT_VERSION: parsing unrecognized 1.x version {version!r}")
231
+ doc_type = (root.get("DocType") or "OFD").strip().upper()
232
+ if doc_type not in {"OFD", "OFD-A"}:
233
+ raise OfdParseError(f"Unsupported OFD DocType: {doc_type!r}")
234
+ if not any(local_name(child.tag) == "DocBody" for child in root):
235
+ raise OfdParseError("Malformed OFD package: OFD.xml has no DocBody")
236
+ self._root = root
237
+ return root
238
+
239
+ def resolve_reference(self, base_part: str, location: str | None) -> str | None:
240
+ """解析大小写敏感的 ST_Loc,并拒绝包外与网络位置。"""
241
+ raw = unquote((location or "").strip()).replace("\\", "/")
242
+ if not raw:
243
+ return None
244
+ try:
245
+ parsed = urlsplit(raw)
246
+ except ValueError:
247
+ return None
248
+ if parsed.scheme or parsed.netloc or parsed.query:
249
+ return None
250
+ raw_path = parsed.path
251
+ if raw_path.startswith("/"):
252
+ resolved = posixpath.normpath(raw_path).lstrip("/")
253
+ else:
254
+ resolved = posixpath.normpath(posixpath.join(posixpath.dirname(base_part), raw_path))
255
+ if resolved in {"", ".", ".."} or resolved.startswith("../") or resolved.startswith("/"):
256
+ return None
257
+ return resolved.removeprefix("./")
258
+
259
+ def document_refs(self) -> list[OfdDocumentRef]:
260
+ """按 DocBody 声明顺序返回全部文档入口。"""
261
+ refs: list[OfdDocumentRef] = []
262
+ for body in self.root():
263
+ if local_name(body.tag) != "DocBody":
264
+ continue
265
+ if len(refs) >= MAX_DOCUMENT_COUNT:
266
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_document_count={MAX_DOCUMENT_COUNT}")
267
+ doc_root = first_child(body, "DocRoot")
268
+ document_part = self.resolve_reference("OFD.xml", element_text(doc_root))
269
+ if document_part is None:
270
+ raise OfdParseError("Malformed OFD package: DocBody has invalid DocRoot")
271
+ signatures = first_child(body, "Signatures")
272
+ signatures_part = self.resolve_reference("OFD.xml", element_text(signatures)) if signatures is not None else None
273
+ metadata: dict[str, str] = {}
274
+ doc_info = first_child(body, "DocInfo")
275
+ if doc_info is not None:
276
+ for key in ("Title", "Author", "Subject", "Keywords", "DocUsage", "Creator", "CreatorVersion"):
277
+ value = element_text(first_child(doc_info, key))
278
+ if value:
279
+ metadata[key] = value
280
+ refs.append(OfdDocumentRef(document_part=document_part, signatures_part=signatures_part, metadata=metadata))
281
+ return refs
282
+
283
+ def close(self) -> None:
284
+ """关闭底层 ZipFile。"""
285
+ self._zip.close()
286
+
287
+ def __enter__(self) -> OfdPackage:
288
+ """返回当前包以支持 with 生命周期。"""
289
+ return self
290
+
291
+ def __exit__(self, _exc_type: object, _exc: object, _traceback: object) -> None:
292
+ """退出 with 块时关闭包。"""
293
+ self.close()
294
+
295
+
296
+ def detect_ofd(file_bytes: bytes) -> bool:
297
+ """按受限 OFD 包身份识别内存字节。"""
298
+ try:
299
+ with OfdPackage(file_bytes) as package:
300
+ package.root()
301
+ return True
302
+ except (BadZipFile, OfdParseError, OSError, ValueError):
303
+ return False
304
+
305
+
306
+ def detect_ofd_path(file_path: str | Path) -> bool:
307
+ """直接从 ZIP 路径读取有限 OFD.xml 并验证包身份。"""
308
+ try:
309
+ path = Path(file_path)
310
+ if path.stat().st_size > MAX_TOTAL_BYTES:
311
+ return False
312
+ with ZipFile(path) as package:
313
+ info = package.getinfo("OFD.xml")
314
+ if info.file_size > MAX_ENTRY_BYTES or info.flag_bits & 0x1 or info.compress_type not in {ZIP_STORED, ZIP_DEFLATED}:
315
+ return False
316
+ with package.open(info) as source:
317
+ data = source.read(MAX_ENTRY_BYTES + 1)
318
+ if len(data) > MAX_ENTRY_BYTES:
319
+ return False
320
+ root = etree.fromstring(data, parser=_xml_parser())
321
+ namespace = namespace_name(root.tag)
322
+ version = (root.get("Version") or "").strip()
323
+ doc_type = (root.get("DocType") or "OFD").strip().upper()
324
+ return (
325
+ not root.getroottree().docinfo.doctype
326
+ and local_name(root.tag) == "OFD"
327
+ and namespace in OFD_NAMESPACES
328
+ and re.fullmatch(r"1(?:\.\d+)?", version) is not None
329
+ and doc_type in {"OFD", "OFD-A"}
330
+ and any(local_name(child.tag) == "DocBody" for child in root)
331
+ )
332
+ except (BadZipFile, KeyError, OSError, RuntimeError, ValueError, etree.XMLSyntaxError):
333
+ return False
334
+
335
+
336
+ __all__ = [
337
+ "OfdPackage",
338
+ "detect_ofd",
339
+ "detect_ofd_path",
340
+ "element_text",
341
+ "first_child",
342
+ "first_descendant",
343
+ "local_name",
344
+ "namespace_name",
345
+ "parse_int",
346
+ ]
@@ -0,0 +1,201 @@
1
+ """解析 OFD PathObject 并提取表格可用的轴向线段。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ import re
7
+ from dataclasses import dataclass
8
+
9
+ from loguru import logger
10
+ from lxml import etree # type: ignore[reportMissingImports]
11
+
12
+ from ....schema import BBox
13
+ from .constants import MAX_PATH_COMMANDS, MAX_PATH_TOKENS
14
+ from .errors import OfdResourceLimitError
15
+ from .geometry import Affine, bbox_intersection, parse_affine, parse_st_box, transform_bbox
16
+ from .models import AxisLine
17
+ from .package import element_text, first_descendant, parse_int
18
+
19
+ _TOKEN_RE = re.compile(r"CM|[SMLQBAC]|[-+]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][-+]?\d+)?")
20
+
21
+
22
+ @dataclass(slots=True)
23
+ class OfdPathBudget:
24
+ """累计限制紧缩路径 token 与命令数量。"""
25
+
26
+ command_count: int = 0
27
+ token_count: int = 0
28
+
29
+ def charge_token(self) -> None:
30
+ """累计实际扫描的路径 token 并在超限时失败。"""
31
+ self.token_count += 1
32
+ if self.token_count > MAX_PATH_TOKENS:
33
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_path_tokens={MAX_PATH_TOKENS}")
34
+
35
+ def charge_command(self) -> None:
36
+ """累计实际扫描的路径命令并在超限时失败。"""
37
+ self.command_count += 1
38
+ if self.command_count > MAX_PATH_COMMANDS:
39
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_path_commands={MAX_PATH_COMMANDS}")
40
+
41
+
42
+ def _finite_float(value: str) -> float | None:
43
+ """把路径 token 转换为有限浮点数。"""
44
+ try:
45
+ parsed = float(value)
46
+ except ValueError:
47
+ return None
48
+ return parsed if math.isfinite(parsed) else None
49
+
50
+
51
+ def _line_bbox(first: tuple[float, float], second: tuple[float, float], width: float) -> tuple[BBox, str] | None:
52
+ """把近似水平或垂直线段转换为非退化 bbox。"""
53
+ dx = second[0] - first[0]
54
+ dy = second[1] - first[1]
55
+ length = math.hypot(dx, dy)
56
+ if length <= 0:
57
+ return None
58
+ tolerance = max(0.2, 0.02 * length)
59
+ thickness = max(width, 0.1)
60
+ if abs(dy) <= tolerance:
61
+ return (
62
+ (
63
+ min(first[0], second[0]),
64
+ min(first[1], second[1]) - thickness / 2,
65
+ max(first[0], second[0]),
66
+ max(first[1], second[1]) + thickness / 2,
67
+ ),
68
+ "horizontal",
69
+ )
70
+ if abs(dx) <= tolerance:
71
+ return (
72
+ (
73
+ min(first[0], second[0]) - thickness / 2,
74
+ min(first[1], second[1]),
75
+ max(first[0], second[0]) + thickness / 2,
76
+ max(first[1], second[1]),
77
+ ),
78
+ "vertical",
79
+ )
80
+ return None
81
+
82
+
83
+ def _segments(value: str, transform: Affine, budget: OfdPathBudget) -> list[tuple[tuple[float, float], tuple[float, float]]]:
84
+ """流式提取 S/M/L/CM 直线段,曲线命令只推进当前位置。"""
85
+ result: list[tuple[tuple[float, float], tuple[float, float]]] = []
86
+ current: tuple[float, float] | None = None
87
+ subpath_start: tuple[float, float] | None = None
88
+ command: str | None = None
89
+ parameters: list[float] = []
90
+ commands = {"S", "M", "L", "Q", "B", "A", "C", "CM"}
91
+ arity = {"S": 2, "M": 2, "L": 2, "Q": 4, "B": 6, "A": 7, "CM": 2}
92
+ for match in _TOKEN_RE.finditer(value):
93
+ budget.charge_token()
94
+ token = match.group()
95
+ if token in commands:
96
+ if parameters:
97
+ return []
98
+ budget.charge_command()
99
+ command = token
100
+ if command == "C":
101
+ if current is not None and subpath_start is not None and current != subpath_start:
102
+ result.append((transform.apply(current), transform.apply(subpath_start)))
103
+ current = subpath_start
104
+ command = None
105
+ continue
106
+ if command is None or command == "C":
107
+ return []
108
+ parsed = _finite_float(token)
109
+ if parsed is None:
110
+ return []
111
+ parameters.append(parsed)
112
+ if len(parameters) < arity[command]:
113
+ continue
114
+ endpoint = (parameters[-2], parameters[-1])
115
+ if command in {"S", "M"}:
116
+ current = endpoint
117
+ subpath_start = endpoint
118
+ elif command in {"L", "CM"}:
119
+ if current is not None:
120
+ result.append((transform.apply(current), transform.apply(endpoint)))
121
+ current = endpoint
122
+ else:
123
+ current = endpoint
124
+ parameters.clear()
125
+ return [] if parameters else result
126
+
127
+
128
+ def build_axis_lines(
129
+ path_object: etree._Element,
130
+ *,
131
+ parent_transform: Affine,
132
+ parent_clip: BBox,
133
+ paint_order: int,
134
+ template_id: int | None,
135
+ budget: OfdPathBudget,
136
+ resolved_style: dict[str, str] | None = None,
137
+ ) -> list[AxisLine]:
138
+ """从一个 PathObject 提取可见轴向线段。"""
139
+ style = resolved_style or {}
140
+ if (style.get("Visible") or path_object.get("Visible") or "true").casefold() in {"false", "0"}:
141
+ return []
142
+ if (style.get("Alpha") or path_object.get("Alpha") or "255").strip() == "0":
143
+ return []
144
+ boundary = parse_st_box(path_object.get("Boundary"))
145
+ if boundary is None:
146
+ return []
147
+ boundary_page = transform_bbox(boundary, parent_transform)
148
+ if boundary_page is None:
149
+ return []
150
+ object_clip = bbox_intersection(boundary_page, parent_clip)
151
+ if object_clip is None:
152
+ return []
153
+ try:
154
+ width = max(0.1, float(style.get("LineWidth") or path_object.get("LineWidth") or 0.353))
155
+ except ValueError:
156
+ width = 0.353
157
+ object_transform = parent_transform.compose(Affine.translation(boundary[0], boundary[1])).compose(
158
+ parse_affine(path_object.get("CTM"))
159
+ )
160
+ extracted: list[AxisLine] = []
161
+ data = first_descendant(path_object, "AbbreviatedData")
162
+ segments = _segments(element_text(data), object_transform, budget) if data is not None else []
163
+ for first, second in segments:
164
+ line = _line_bbox(first, second, width)
165
+ if line is None:
166
+ continue
167
+ line_bbox, orientation = line
168
+ clipped = bbox_intersection(line_bbox, object_clip)
169
+ if clipped is not None:
170
+ extracted.append(
171
+ AxisLine(
172
+ bbox=clipped,
173
+ orientation=orientation,
174
+ width=width,
175
+ paint_order=paint_order,
176
+ template_id=template_id,
177
+ )
178
+ )
179
+ boundary_width = boundary_page[2] - boundary_page[0]
180
+ boundary_height = boundary_page[3] - boundary_page[1]
181
+ if (
182
+ not extracted
183
+ and not segments
184
+ and max(boundary_width, boundary_height) >= 5 * max(min(boundary_width, boundary_height), 0.01)
185
+ ):
186
+ orientation = "horizontal" if boundary_width >= boundary_height else "vertical"
187
+ extracted.append(
188
+ AxisLine(
189
+ bbox=object_clip,
190
+ orientation=orientation,
191
+ width=width,
192
+ paint_order=paint_order,
193
+ template_id=template_id,
194
+ )
195
+ )
196
+ if not extracted and parse_int(path_object.get("ID")) is None:
197
+ logger.debug("OFD_PATH_SKIPPED: path without stable ID produced no axis lines")
198
+ return extracted
199
+
200
+
201
+ __all__ = ["OfdPathBudget", "build_axis_lines"]