docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,693 @@
1
+ """DOCX 富文本与样式处理;共享当前 Converter 的单文档状态。"""
2
+
3
+ from pathlib import Path
4
+ from typing import Any, Iterator, Optional, Union
5
+ from docx.enum.style import WD_STYLE_TYPE
6
+ from docx.oxml.xmlchemy import BaseOxmlElement
7
+ from docx.text.hyperlink import Hyperlink
8
+ from docx.text.paragraph import Paragraph
9
+ from docx.text.run import Run
10
+ from pydantic import AnyUrl
11
+ from .formatting_types import Formatting, Script
12
+ from .....content.spans import (
13
+ append_equation_span,
14
+ extend_inline_spans,
15
+ inline_span_plain_text,
16
+ slice_span_dicts,
17
+ strip_span_dicts,
18
+ )
19
+ from ..rich_text import (
20
+ append_rich_text_element,
21
+ build_spans_from_elements,
22
+ formatting_to_style_str,
23
+ has_non_visible_text_style,
24
+ has_visible_style,
25
+ normalize_format_for_text,
26
+ should_keep_group_text,
27
+ )
28
+
29
+ from .context import _DocxConstants, _ParagraphElement
30
+
31
+
32
+ class _DocxStyles:
33
+ """集中维护富文本与样式,不自行创建文档或持有跨文档缓存。"""
34
+
35
+ def _get_style_id_from_property(
36
+ self,
37
+ xml_element: Optional[BaseOxmlElement],
38
+ property_tag: str,
39
+ style_tag: str,
40
+ ) -> Optional[str]:
41
+ """从段落或 run 的直接属性节点读取样式 ID,避免触发 python-docx 样式查找。"""
42
+ if xml_element is None:
43
+ return None
44
+
45
+ property_element = xml_element.find(
46
+ property_tag,
47
+ namespaces=_DocxConstants._BLIP_NAMESPACES,
48
+ )
49
+ if property_element is None:
50
+ return None
51
+
52
+ style_element = property_element.find(
53
+ style_tag,
54
+ namespaces=_DocxConstants._BLIP_NAMESPACES,
55
+ )
56
+ if style_element is None:
57
+ return None
58
+
59
+ return style_element.get(self.XML_KEY) or None
60
+
61
+ def _get_cached_docx_style(
62
+ self,
63
+ part: Any,
64
+ style_id: Optional[str],
65
+ style_type: Any,
66
+ ) -> Any:
67
+ """按 style id 和类型缓存 python-docx 样式对象,避免大 styles.xml 被反复线性扫描。"""
68
+ if part is None:
69
+ return None
70
+
71
+ cache_key = (style_type, style_id)
72
+ if cache_key not in self._style_lookup_cache:
73
+ self._style_lookup_cache[cache_key] = part.get_style(
74
+ style_id,
75
+ style_type,
76
+ )
77
+ return self._style_lookup_cache[cache_key]
78
+
79
+ def _get_paragraph_style(self, paragraph: Optional[Paragraph]) -> Any:
80
+ """读取段落样式;无显式 pStyle 时缓存默认段落样式查询结果。"""
81
+ if paragraph is None:
82
+ return None
83
+ style_id = self._get_style_id_from_property(
84
+ paragraph._element,
85
+ "w:pPr",
86
+ "w:pStyle",
87
+ )
88
+ return self._get_cached_docx_style(
89
+ paragraph.part,
90
+ style_id,
91
+ WD_STYLE_TYPE.PARAGRAPH,
92
+ )
93
+
94
+ def _get_run_style(self, run: Optional[Run]) -> Any:
95
+ """读取 run 字符样式;无显式 rStyle 时缓存默认字符样式查询结果。"""
96
+ if run is None:
97
+ return None
98
+ style_id = self._get_style_id_from_property(
99
+ run._element,
100
+ "w:rPr",
101
+ "w:rStyle",
102
+ )
103
+ return self._get_cached_docx_style(
104
+ run.part,
105
+ style_id,
106
+ WD_STYLE_TYPE.CHARACTER,
107
+ )
108
+
109
+ @staticmethod
110
+ def _escape_hyperlink_text(text: str) -> str:
111
+ """
112
+ 转义超链接文本中的方括号。
113
+
114
+ Args:
115
+ text: 要转义的文本
116
+
117
+ Returns:
118
+ str: 转义后的文本
119
+ """
120
+ if not text:
121
+ return text
122
+ # 转义方括号
123
+ text = text.replace("[", "\\[").replace("]", "\\]")
124
+ return text
125
+
126
+ @staticmethod
127
+ def _escape_hyperlink_url(url: str) -> str:
128
+ """
129
+ 转义超链接 URL 中的括号。
130
+
131
+ Args:
132
+ url: 要转义的 URL
133
+
134
+ Returns:
135
+ str: 转义后的 URL
136
+ """
137
+ if not url:
138
+ return url
139
+ # 对括号进行 URL 编码
140
+ url = url.replace("(", "%28").replace(")", "%29")
141
+ return url
142
+
143
+ @staticmethod
144
+ def _get_style_str_from_format(format_obj) -> Optional[str]:
145
+ """
146
+ 从 Formatting 对象提取样式字符串。
147
+
148
+ Args:
149
+ format_obj: Formatting 对象
150
+
151
+ Returns:
152
+ Optional[str]: 样式字符串(如 "bold,italic"),无样式时返回 None
153
+ """
154
+ return formatting_to_style_str(format_obj)
155
+
156
+ @staticmethod
157
+ def _has_visible_style(format_obj) -> bool:
158
+ """
159
+ 检查格式是否包含可见样式(下划线或删除线)。
160
+
161
+ 空白文本在有这些样式时仍然是可见的,应当保留。
162
+
163
+ Args:
164
+ format_obj: Formatting 对象
165
+
166
+ Returns:
167
+ bool: 是否包含可见样式
168
+ """
169
+ return has_visible_style(format_obj)
170
+
171
+ @staticmethod
172
+ def _has_non_visible_text_style(format_obj) -> bool:
173
+ """判断格式是否只有空白文本不可见的字形样式。"""
174
+ return has_non_visible_text_style(format_obj)
175
+
176
+ @classmethod
177
+ def _normalize_format_for_text(
178
+ cls,
179
+ format_obj: Optional[Formatting],
180
+ text: str,
181
+ *,
182
+ preserve_blank_non_visible_style: bool = False,
183
+ ) -> Optional[Formatting]:
184
+ """按文本内容收敛 run 格式,避免空白 run 把不可见样式传给输出。
185
+
186
+ preserve_blank_non_visible_style 用于保留同一文本片段内空白 run 的
187
+ bold/italic:这些样式自身不让空格可见,但可能是连续同样式文本的一部分。
188
+ """
189
+ return normalize_format_for_text(
190
+ format_obj,
191
+ text,
192
+ preserve_blank_non_visible_style=preserve_blank_non_visible_style,
193
+ )
194
+
195
+ def _find_adjacent_non_blank_run_format(
196
+ self,
197
+ inline_contents: list[Any],
198
+ current_index: int,
199
+ step: int,
200
+ ) -> Optional[Formatting]:
201
+ """查找相邻方向上最近的非空白普通 run 格式,用于判断空白 run 是否属于同一段样式文本。"""
202
+ index = current_index + step
203
+ while 0 <= index < len(inline_contents):
204
+ content = inline_contents[index]
205
+ # 超链接是独立输出边界,不跨越超链接借用样式上下文。
206
+ if isinstance(content, Hyperlink):
207
+ return None
208
+ if not isinstance(content, Run):
209
+ index += step
210
+ continue
211
+ if self._is_hidden_run(content):
212
+ index += step
213
+ continue
214
+ text = content.text or ""
215
+ if text.strip():
216
+ return self._get_format_from_run(content)
217
+ index += step
218
+ return None
219
+
220
+ def _should_preserve_blank_non_visible_style(
221
+ self,
222
+ inline_contents: list[Any],
223
+ current_index: int,
224
+ text: str,
225
+ format_obj: Optional[Formatting],
226
+ ) -> bool:
227
+ """判断空白 run 的 bold/italic 是否应保留,以便连续同样式文本合并成一个 span。"""
228
+ if not text or text.strip():
229
+ return False
230
+ if not self._has_non_visible_text_style(format_obj):
231
+ return False
232
+
233
+ previous_format = self._find_adjacent_non_blank_run_format(
234
+ inline_contents,
235
+ current_index,
236
+ -1,
237
+ )
238
+ if format_obj == previous_format:
239
+ return True
240
+
241
+ next_format = self._find_adjacent_non_blank_run_format(
242
+ inline_contents,
243
+ current_index,
244
+ 1,
245
+ )
246
+ return format_obj == next_format
247
+
248
+ @classmethod
249
+ def _should_keep_group_text(
250
+ cls,
251
+ text: str,
252
+ format_obj: Optional[Formatting],
253
+ *,
254
+ preserve_plain_blank: bool = False,
255
+ ) -> bool:
256
+ """判断当前累积 run 是否需要输出,保留夹在可见样式之间的普通空白。"""
257
+ return should_keep_group_text(
258
+ text,
259
+ format_obj,
260
+ preserve_plain_blank=preserve_plain_blank,
261
+ )
262
+
263
+ @staticmethod
264
+ def _append_paragraph_element(
265
+ paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
266
+ text: str,
267
+ format_obj: Optional[Formatting],
268
+ hyperlink: Optional[Union[AnyUrl, Path, str]],
269
+ ) -> None:
270
+ """追加段落元素;相邻同超链接且同格式的 run 合并为一个元素。"""
271
+ append_rich_text_element(paragraph_elements, text, format_obj, hyperlink)
272
+
273
+ @staticmethod
274
+ def _normalize_hyperlink_group_boundaries(
275
+ paragraph_elements: list[_ParagraphElement],
276
+ ) -> list[_ParagraphElement]:
277
+ """把链接组边界空白移为普通文本,仅裁剪整段首尾并保留组内空白。"""
278
+
279
+ output: list[_ParagraphElement] = []
280
+ index = 0
281
+ while index < len(paragraph_elements):
282
+ element = paragraph_elements[index]
283
+ hyperlink = element[2]
284
+ if hyperlink is None:
285
+ output.append(element)
286
+ index += 1
287
+ continue
288
+ group_end = index + 1
289
+ while (
290
+ group_end < len(paragraph_elements)
291
+ and paragraph_elements[group_end][2] is not None
292
+ and str(paragraph_elements[group_end][2]) == str(hyperlink)
293
+ ):
294
+ group_end += 1
295
+ group = list(paragraph_elements[index:group_end])
296
+ first_text, first_format, first_hyperlink = group[0]
297
+ last_text, last_format, last_hyperlink = group[-1]
298
+ if len(group) == 1 and not first_text.strip():
299
+ leading_space = ""
300
+ trailing_space = first_text
301
+ group[0] = ("", first_format, first_hyperlink)
302
+ else:
303
+ leading_length = len(first_text) - len(first_text.lstrip())
304
+ trailing_length = len(last_text) - len(last_text.rstrip())
305
+ leading_space = first_text[:leading_length]
306
+ trailing_space = last_text[len(last_text) - trailing_length :] if trailing_length else ""
307
+ if len(group) == 1:
308
+ group[0] = (
309
+ first_text[leading_length : len(first_text) - trailing_length if trailing_length else len(first_text)],
310
+ first_format,
311
+ first_hyperlink,
312
+ )
313
+ else:
314
+ group[0] = (
315
+ first_text[leading_length:],
316
+ first_format,
317
+ first_hyperlink,
318
+ )
319
+ group[-1] = (
320
+ last_text[: len(last_text) - trailing_length] if trailing_length else last_text,
321
+ last_format,
322
+ last_hyperlink,
323
+ )
324
+ if leading_space and output:
325
+ output.append((leading_space, first_format, None))
326
+ output.extend(item for item in group if item[0])
327
+ if trailing_space and group_end < len(paragraph_elements):
328
+ output.append((trailing_space, last_format, None))
329
+ index = group_end
330
+ return output
331
+
332
+ @staticmethod
333
+ def _is_hidden_run(run: Run) -> bool:
334
+ """Check whether a run is marked as hidden text in Word."""
335
+ _W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
336
+ rpr = run._element.find(f"{{{_W}}}rPr")
337
+ if rpr is None:
338
+ return False
339
+ # webHidden: commonly used by TOC page-number field runs
340
+ if rpr.find(f"{{{_W}}}webHidden") is not None:
341
+ return True
342
+ # vanish: generic hidden text
343
+ if rpr.find(f"{{{_W}}}vanish") is not None:
344
+ return True
345
+ return False
346
+
347
+ @classmethod
348
+ def _build_spans_from_elements(
349
+ cls,
350
+ paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
351
+ ) -> list[dict[str, Any]]:
352
+ """按连续同 URL hyperlink 分组,直接生成结构化 Span。"""
353
+ return build_spans_from_elements(paragraph_elements)
354
+
355
+ def _build_text_from_elements(
356
+ self,
357
+ paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
358
+ ) -> list[dict[str, Any]]:
359
+ """
360
+ 从 paragraph_elements 重组文本,应用超链接格式和字体样式。
361
+
362
+ Args:
363
+ paragraph_elements: 段落元素列表
364
+
365
+ Returns:
366
+ list[dict]: 重组后的 Span
367
+ """
368
+ return self._build_spans_from_elements(paragraph_elements)
369
+
370
+ @staticmethod
371
+ def _normalize_text_block_content(content: list[dict[str, Any]]) -> list[dict[str, Any]]:
372
+ """
373
+ 规范化普通文本块导出内容。
374
+
375
+ DOCX 常用段首/段尾空格模拟版式对齐,导出普通文本块前去除这些前后空白。
376
+ """
377
+ return strip_span_dicts(content)
378
+
379
+ def _build_text_with_equations_and_hyperlinks(
380
+ self,
381
+ paragraph_elements: list[tuple[str, Optional[Formatting], Optional[Union[AnyUrl, Path, str]]]],
382
+ text_with_equations: str,
383
+ equations: list[tuple[str, str]],
384
+ ) -> list[dict[str, Any]]:
385
+ """
386
+ 构建同时包含公式、超链接和字体样式的文本。
387
+
388
+ Args:
389
+ paragraph_elements: 段落元素列表,包含格式和超链接信息
390
+ text_with_equations: 不含公式的原始可见文本
391
+ equations: 按源顺序排列的 text/equation token
392
+
393
+ Returns:
394
+ list[dict]: 包含公式、超链接和字体样式的 Span
395
+ """
396
+ if not equations:
397
+ return self._build_text_from_elements(paragraph_elements)
398
+ styled_spans = self._build_text_from_elements(paragraph_elements)
399
+ styled_visible = inline_span_plain_text(styled_spans)
400
+ plain_visible = "".join(value for kind, value in equations if kind == "text")
401
+ if plain_visible != styled_visible:
402
+ return styled_spans
403
+
404
+ output: list[dict[str, Any]] = []
405
+ visible_cursor = 0
406
+ for kind, value in equations:
407
+ if kind == "equation":
408
+ append_equation_span(output, value)
409
+ continue
410
+ next_cursor = visible_cursor + len(value)
411
+ extend_inline_spans(output, slice_span_dicts(styled_spans, visible_cursor, next_cursor))
412
+ visible_cursor = next_cursor
413
+ return strip_span_dicts(output)
414
+
415
+ @staticmethod
416
+ def _get_paragraph_text_from_contents(
417
+ inner_contents: list[Union[Run, Hyperlink]],
418
+ ) -> str:
419
+ """Rebuild paragraph plain text from visible inline containers."""
420
+ return "".join(content.text or "" for content in inner_contents)
421
+
422
+ def _get_paragraph_text(self, paragraph: Paragraph) -> str:
423
+ """Return paragraph plain text, including inline ``w:sdt`` content."""
424
+ return self._get_paragraph_text_from_contents(list(self._iter_paragraph_inner_content(paragraph)))
425
+
426
+ def _resolve_style_chain_bool(
427
+ self,
428
+ style_obj,
429
+ attr_name: str,
430
+ ) -> Optional[bool]:
431
+ """从样式继承链中解析布尔字体属性。"""
432
+ if style_obj is None:
433
+ return None
434
+
435
+ cache_key = (id(style_obj), attr_name)
436
+ if cache_key in self._style_bool_cache:
437
+ return self._style_bool_cache[cache_key]
438
+
439
+ style = style_obj
440
+ result = None
441
+ visited_styles = set()
442
+ while style is not None:
443
+ style_element = getattr(style, "_element", None)
444
+ style_id = getattr(style, "style_id", None)
445
+ style_type = getattr(style, "type", None)
446
+ # DOCX 可能存在 basedOn 自引用或环形引用,记录已访问样式避免继承链死循环。
447
+ style_marker = (
448
+ id(style_element) if style_element is not None else id(style),
449
+ style_type,
450
+ style_id,
451
+ )
452
+ if style_marker in visited_styles:
453
+ break
454
+ visited_styles.add(style_marker)
455
+
456
+ font = getattr(style, "font", None)
457
+ if font is not None:
458
+ if attr_name == "underline":
459
+ value = font.underline
460
+ elif attr_name == "strikethrough":
461
+ value = font.strike
462
+ else:
463
+ value = getattr(font, attr_name, None)
464
+ if value is not None:
465
+ result = bool(value)
466
+ break
467
+ style = getattr(style, "base_style", None)
468
+ self._style_bool_cache[cache_key] = result
469
+ return result
470
+
471
+ def _resolve_run_bool_with_inheritance(
472
+ self,
473
+ run: Run,
474
+ attr_name: str,
475
+ ) -> bool:
476
+ """解析 run 的字体属性,支持 run/字符样式/段落样式继承。"""
477
+ if attr_name == "underline":
478
+ direct_value = run.underline
479
+ elif attr_name == "strikethrough":
480
+ direct_value = run.font.strike
481
+ else:
482
+ direct_value = getattr(run, attr_name, None)
483
+
484
+ if direct_value is not None:
485
+ return bool(direct_value)
486
+
487
+ # 先看 run 级字符样式链(跳过 Hyperlink 默认字符样式,避免把默认下划线
488
+ # 误当作正文强调样式注入到解析结果中)
489
+ run_style = self._get_run_style(run)
490
+ run_style_id = str(getattr(run_style, "style_id", "") or "").lower()
491
+ run_style_name = str(getattr(run_style, "name", "") or "").lower()
492
+ is_hyperlink_style = run_style_id == "hyperlink" or "hyperlink" in run_style_name
493
+ if not is_hyperlink_style:
494
+ inherited = self._resolve_style_chain_bool(run_style, attr_name)
495
+ if inherited is not None:
496
+ return inherited
497
+
498
+ # 再看所在段落样式链
499
+ parent = getattr(run, "_parent", None)
500
+ inherited = self._resolve_style_chain_bool(
501
+ self._get_paragraph_style(parent),
502
+ attr_name,
503
+ )
504
+ if inherited is not None:
505
+ return inherited
506
+
507
+ return False
508
+
509
+ @staticmethod
510
+ def _get_direct_underline_style(run: Run) -> str:
511
+ """读取 run 级下划线类型,用于区分 words 这类不作用于空格的下划线。"""
512
+ _W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
513
+ rPr = run._element.find(f"{{{_W}}}rPr")
514
+ if rPr is None:
515
+ return ""
516
+ underline = rPr.find(f"{{{_W}}}u")
517
+ if underline is None:
518
+ return ""
519
+ return underline.get(f"{{{_W}}}val", "single")
520
+
521
+ def _get_format_from_run(self, run: Run) -> Optional[Formatting]:
522
+ """
523
+ 从 Run 对象获取格式信息。
524
+
525
+ Args:
526
+ run: Run 对象
527
+
528
+ Returns:
529
+ Optional[Formatting]: 格式对象
530
+ """
531
+ is_bold = self._resolve_run_bool_with_inheritance(run, "bold")
532
+ is_italic = self._resolve_run_bool_with_inheritance(run, "italic")
533
+ is_strikethrough = self._resolve_run_bool_with_inheritance(run, "strikethrough")
534
+ is_underline = self._resolve_run_bool_with_inheritance(run, "underline")
535
+ underline_style = self._get_direct_underline_style(run)
536
+
537
+ # 检测着重符号 (w:em):独立保留为 emphasis,避免和真实下划线混淆。
538
+ is_emphasis = False
539
+ _W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
540
+ rPr = run._element.find(f"{{{_W}}}rPr")
541
+ if rPr is not None:
542
+ em = rPr.find(f"{{{_W}}}em")
543
+ if em is not None:
544
+ em_val = em.get(f"{{{_W}}}val", "")
545
+ if em_val and em_val != "none":
546
+ is_emphasis = True
547
+
548
+ is_sub = run.font.subscript or False
549
+ is_sup = run.font.superscript or False
550
+ script = Script.SUB if is_sub else Script.SUPER if is_sup else Script.BASELINE
551
+
552
+ return Formatting(
553
+ bold=is_bold,
554
+ italic=is_italic,
555
+ underline=is_underline,
556
+ underline_style=underline_style,
557
+ emphasis=is_emphasis,
558
+ strikethrough=is_strikethrough,
559
+ script=script,
560
+ )
561
+
562
+ def _handle_equations_in_text(
563
+ self,
564
+ element: Any,
565
+ text: str,
566
+ *,
567
+ part: Any | None = None,
568
+ ) -> tuple[str, list[tuple[str, str]]]:
569
+ """
570
+ 处理文本中的公式。
571
+
572
+ Args:
573
+ element: 元素对象
574
+ text: 文本内容
575
+ part: 当前段落所属的 OOXML part,用于解析局部 relationship
576
+
577
+ Returns:
578
+ tuple: (原始可见文本, 含公式时的有序 text/equation token)
579
+ """
580
+ source_part = part or self._require_document_part()
581
+ only_texts: list[str] = []
582
+ formula_values: list[str] = []
583
+ tokens: list[tuple[str, str]] = []
584
+ for token_kind, value in self._docx_formula_tokens(element, source_part):
585
+ if token_kind == "text":
586
+ only_texts.append(value)
587
+ tokens.append(("text", value))
588
+ continue
589
+ formula_values.append(value)
590
+ tokens.append(("equation", value))
591
+
592
+ if not formula_values:
593
+ return text, []
594
+
595
+ if "".join(only_texts) != text:
596
+ # 如果我们无法重构初始原始文本
597
+ # 不要尝试解析公式并返回原始文本
598
+ return text, []
599
+
600
+ return text, tokens
601
+
602
+ def _get_label_and_level(self, paragraph: Paragraph) -> tuple[str, Optional[int]]:
603
+ """
604
+ 获取段落的标签和层级。
605
+
606
+ Args:
607
+ paragraph: 段落对象
608
+
609
+ Returns:
610
+ tuple[str, Optional[int]]: (标签, 层级) 元组
611
+ """
612
+ paragraph_style = self._get_paragraph_style(paragraph)
613
+ if paragraph_style is None:
614
+ return "Normal", None
615
+
616
+ label = paragraph_style.style_id
617
+ name = paragraph_style.name
618
+
619
+ if label is None:
620
+ return "Normal", None
621
+
622
+ for style in self._iter_style_chain(paragraph_style):
623
+ style_label = getattr(style, "style_id", None)
624
+ style_name = getattr(style, "name", None)
625
+
626
+ if style_label and ":" in style_label:
627
+ parts = style_label.split(":")
628
+ if len(parts) == 2:
629
+ return parts[0], self._str_to_int(parts[1], None)
630
+
631
+ for candidate in (style_label, style_name):
632
+ if candidate and "heading" in candidate.lower():
633
+ return self._get_heading_and_level(candidate)
634
+
635
+ outline_level = self._get_effective_outline_level(paragraph)
636
+ if outline_level is not None:
637
+ return "Heading", outline_level + 1
638
+
639
+ return name or label or "Normal", None
640
+
641
+ def _iter_style_chain(self, style: Any) -> Iterator[Any]:
642
+ """Yield a style and its base-style chain once each."""
643
+ seen: set[int] = set()
644
+ current = style
645
+ while current is not None:
646
+ current_id = id(current)
647
+ if current_id in seen:
648
+ break
649
+ seen.add(current_id)
650
+ yield current
651
+ current = getattr(current, "base_style", None)
652
+
653
+ def _get_paragraph_property_child(
654
+ self, xml_element: Optional[BaseOxmlElement], child_tag: str
655
+ ) -> Optional[BaseOxmlElement]:
656
+ """Read a direct child from w:pPr without matching nested descendants."""
657
+ if xml_element is None:
658
+ return None
659
+
660
+ namespaces = getattr(xml_element, "nsmap", None) or _DocxConstants._BLIP_NAMESPACES
661
+ pPr = xml_element.find("w:pPr", namespaces=namespaces)
662
+ if pPr is None:
663
+ return None
664
+ return pPr.find(child_tag, namespaces=namespaces)
665
+
666
+ def _get_effective_numPr(self, paragraph: Paragraph) -> Optional[BaseOxmlElement]:
667
+ """Resolve paragraph numbering from direct properties, then style inheritance."""
668
+ numPr = self._get_paragraph_property_child(paragraph._element, "w:numPr")
669
+ if numPr is not None:
670
+ return numPr
671
+
672
+ for style in self._iter_style_chain(self._get_paragraph_style(paragraph)):
673
+ style_element = getattr(style, "element", None)
674
+ numPr = self._get_paragraph_property_child(style_element, "w:numPr")
675
+ if numPr is not None:
676
+ return numPr
677
+
678
+ return None
679
+
680
+ def _get_effective_outline_level(self, paragraph: Paragraph) -> Optional[int]:
681
+ """Resolve outline level from paragraph properties or inherited styles."""
682
+ outline_lvl = self._get_paragraph_property_child(paragraph._element, "w:outlineLvl")
683
+ if outline_lvl is None:
684
+ for style in self._iter_style_chain(self._get_paragraph_style(paragraph)):
685
+ style_element = getattr(style, "element", None)
686
+ outline_lvl = self._get_paragraph_property_child(style_element, "w:outlineLvl")
687
+ if outline_lvl is not None:
688
+ break
689
+
690
+ if outline_lvl is None:
691
+ return None
692
+
693
+ return self._str_to_int(outline_lvl.get(self.XML_KEY), None)