docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,614 @@
1
+ from io import BytesIO
2
+ from typing import Any, BinaryIO, Optional
3
+
4
+ from docx import Document
5
+ from docx.document import Document as DocxDocument
6
+ from docx.oxml.xmlchemy import BaseOxmlElement
7
+ from docx.text.paragraph import Paragraph
8
+ from loguru import logger
9
+ from lxml import etree
10
+
11
+ from ..equation.ooxml import OoxmlEquationDecoder
12
+ from ..equation.image import OfficeImageEquationDecoder
13
+ from .....schema import RAW_CAPTION
14
+ from .equationxml import DocxEquationXmlDecoder
15
+ from .package_normalizer import normalize_docx_package
16
+ from .....schema import BlockType
17
+ from .....content.spans import (
18
+ append_text_span,
19
+ extend_inline_spans,
20
+ inline_span_plain_text,
21
+ text_spans,
22
+ )
23
+
24
+
25
+ from .context import (
26
+ _DocxConstants,
27
+ _DocxComplexFieldFrame as _DocxComplexFieldFrame,
28
+ _ParagraphElement as _ParagraphElement,
29
+ _ParagraphHyperlink as _ParagraphHyperlink,
30
+ )
31
+ from .resources import _DocxResources
32
+ from .styles import _DocxStyles
33
+ from .numbering import _DocxNumbering
34
+ from .fields import _DocxFields
35
+ from .tables import _DocxTables
36
+
37
+
38
+ class DocxConverter(_DocxConstants, _DocxResources, _DocxStyles, _DocxNumbering, _DocxFields, _DocxTables):
39
+ """编排 DOCX 文档生命周期、段落遍历与各项职责处理。"""
40
+
41
+ def __init__(self):
42
+ """配置固定 XML 能力并建立本次文档转换的独立状态。"""
43
+ self.XML_KEY = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}val"
44
+ self.xml_namespaces = {"w": "http://schemas.microsoft.com/office/word/2003/wordml"}
45
+ self.picture_xpath_expr = etree.XPath(".//a:blip | .//v:imagedata", namespaces=DocxConverter._BLIP_NAMESPACES)
46
+ self.equation_bookends: str = "<eq>{EQ}</eq>" # 公式标记格式
47
+ self._reset_document_state()
48
+
49
+ def _reset_document_state(self) -> None:
50
+ """每次转换重新创建可变状态及解码器,避免失败或复用实例残留旧文档。"""
51
+ self.docx_obj = None
52
+ self.pages = []
53
+ self.cur_page = []
54
+ self._mammoth_tables_html: list = [] # 与正文顶层表格对齐的 mammoth 预解析 HTML,None 表示回退解析
55
+ self._mammoth_table_idx: int = 0 # 当前预解析表格游标
56
+ self.pre_num_id: int = -1 # 上一个处理元素的 numId
57
+ self.pre_ilevel: int = -1 # 上一个处理元素的缩进等级, 用于判断列表层级
58
+ self.list_block_stack: list = [] # 列表块堆栈
59
+ self.list_counters: dict[tuple[int, int], int] = {} # 列表计数器 (numId, ilvl) -> count
60
+ self.index_block_stack: list = [] # 目录索引块堆栈
61
+ self.pre_index_ilevel: int = -1 # 上一个目录项的缩进等级
62
+ self.plain_toc_base_level: Optional[int] = None # 普通目录段落的起始层级
63
+ self.heading_list_numids: set = set() # 用作章节标题的列表numId集合
64
+ self.processed_textbox_elements: list = []
65
+ self.toc_anchor_set: set[str] = set() # TOC 超链接目标锚点集合
66
+ self.toc_anchor_aliases: dict[str, str] = {} # 同一正文段落内 TOC bookmark 到唯一公开 anchor 的映射
67
+ self._numbering_root: Optional[BaseOxmlElement] = None
68
+ self._numbering_root_loaded: bool = False
69
+ self._numbering_level_cache: dict[tuple[int, int], Optional[BaseOxmlElement]] = {}
70
+ self._numbering_start_cache: dict[tuple[int, int], int] = {}
71
+ self._style_lookup_cache: dict[tuple[Any, Optional[str]], Any] = {}
72
+ self._style_bool_cache: dict[tuple[int, str], Optional[bool]] = {}
73
+ self._ooxml_equation_decoder = OoxmlEquationDecoder()
74
+ self._mtef_warned_relations: set[tuple[str, str]] = set()
75
+ self._equationxml_decoder = DocxEquationXmlDecoder()
76
+ self._equationxml_warned_shapes: set[tuple[str, str, str]] = set()
77
+ self._image_equation_decoder = OfficeImageEquationDecoder()
78
+
79
+ @staticmethod
80
+ def _local_name(element: Any) -> Optional[str]:
81
+ """安全获取 XML 元素本地标签名,遇到注释或处理指令等非元素节点时返回 None。"""
82
+ tag = getattr(element, "tag", None)
83
+ if not isinstance(tag, str):
84
+ return None
85
+ try:
86
+ return etree.QName(tag).localname
87
+ except ValueError:
88
+ return None
89
+
90
+ def _reset_style_caches(self) -> None:
91
+ """重置样式查询缓存,避免同一 converter 实例多次转换时复用旧文档样式。"""
92
+ self._style_lookup_cache = {}
93
+ self._style_bool_cache = {}
94
+
95
+ @staticmethod
96
+ def _docx_part_key(part: Any) -> str:
97
+ """返回用于公式关系缓存和告警去重的 DOCX part 名称。"""
98
+
99
+ return str(getattr(part, "partname", ""))
100
+
101
+ def _require_document_part(self) -> Any:
102
+ """返回已加载的主文档 part,生命周期异常时立即失败。"""
103
+
104
+ if self.docx_obj is None:
105
+ raise ValueError("DOCX document part is not initialized")
106
+ return self.docx_obj.part
107
+
108
+ def _sanitize_missing_internal_relationships(self, file_bytes: bytes) -> bytes:
109
+ """规范化 DOCX 包,兼容缺失内部关系和损坏图片成员。"""
110
+ return normalize_docx_package(file_bytes)
111
+
112
+ def _start_new_page(self) -> None:
113
+ self.cur_page = []
114
+ self.pages.append(self.cur_page)
115
+
116
+ def _is_layout_only_section_break(self, element: BaseOxmlElement) -> bool:
117
+ w_ns = DocxConverter._BLIP_NAMESPACES["w"]
118
+ p_pr = element.find(f"{{{w_ns}}}pPr")
119
+ sect_pr = p_pr.find(f"{{{w_ns}}}sectPr") if p_pr is not None else None
120
+ if sect_pr is None:
121
+ return False
122
+
123
+ paragraph = Paragraph(element, self.docx_obj)
124
+ if self._get_paragraph_text(paragraph).strip():
125
+ return False
126
+
127
+ if self.picture_xpath_expr(element):
128
+ return False
129
+
130
+ sect_type = sect_pr.find(f"{{{w_ns}}}type")
131
+ sect_val = sect_type.get(f"{{{w_ns}}}val", "continuous") if sect_type is not None else "continuous"
132
+ if sect_val != "continuous":
133
+ return False
134
+
135
+ pg_mar = sect_pr.find(f"{{{w_ns}}}pgMar")
136
+ if pg_mar is None:
137
+ return False
138
+
139
+ for attr in ("header", "footer", "top", "bottom", "left", "right"):
140
+ if pg_mar.get(f"{{{w_ns}}}{attr}", "0") != "0":
141
+ return False
142
+ return True
143
+
144
+ def convert(
145
+ self,
146
+ file_stream: BinaryIO,
147
+ ):
148
+ """重置文档状态后按原顺序读取、预解析表格并遍历正文。"""
149
+ self._reset_document_state()
150
+ # 读取文件字节,以便 mammoth 和 python-docx 各自使用独立读取流
151
+ file_bytes = self._sanitize_missing_internal_relationships(file_stream.read())
152
+ # 使用完整 DOCX 上下文预解析顶层表格,避免转换非表格正文带来的资源浪费
153
+ self._mammoth_tables_html = self._preparse_tables_with_mammoth(file_bytes)
154
+ self._mammoth_table_idx = 0
155
+ self.docx_obj = Document(BytesIO(file_bytes))
156
+ self.toc_anchor_set = self._collect_toc_anchor_set()
157
+ self.toc_anchor_aliases = self._collect_toc_anchor_aliases(
158
+ self.toc_anchor_set,
159
+ )
160
+ # 预扫描文档,识别用作章节标题的列表numId
161
+ self.heading_list_numids = self._detect_heading_list_numids()
162
+ self.pages.append(self.cur_page)
163
+ self._walk_linear(self.docx_obj.element.body)
164
+ self._add_header_footer(self.docx_obj)
165
+
166
+ def _close_active_list(self) -> None:
167
+ """关闭当前活跃列表块,但保留 Word numId 的连续编号计数。"""
168
+ self.pre_num_id = -1
169
+ self.pre_ilevel = -1
170
+ self.list_block_stack = []
171
+
172
+ def _reset_index_state(self) -> None:
173
+ """重置目录索引栈,避免相隔的多个目录块被错误合并。"""
174
+ self.index_block_stack = []
175
+ self.pre_index_ilevel = -1
176
+ self.plain_toc_base_level = None
177
+
178
+ def _walk_linear(
179
+ self,
180
+ body: BaseOxmlElement,
181
+ ):
182
+ for element in body:
183
+ # 获取元素的标签名(去除命名空间前缀)
184
+ tag_name = self._local_name(element)
185
+ if tag_name is None:
186
+ continue
187
+ # 检查是否存在内联图像(blip元素)
188
+ picture_refs = self.picture_xpath_expr(element)
189
+
190
+ # 查找所有绘图元素(用于处理DrawingML)
191
+ drawingml_els = element.findall(".//w:drawing", namespaces=DocxConverter._BLIP_NAMESPACES)
192
+ if drawingml_els:
193
+ self._handle_drawingml(drawingml_els)
194
+
195
+ # 检查文本框内容(支持多种文本框格式)
196
+ # 仅当该元素之前未被处理时才处理
197
+ if element not in self.processed_textbox_elements:
198
+ # 现代 Word 文本框
199
+ txbx_xpath = etree.XPath(
200
+ ".//w:txbxContent|.//v:textbox//w:p",
201
+ namespaces=DocxConverter._BLIP_NAMESPACES,
202
+ )
203
+ textbox_elements = txbx_xpath(element)
204
+
205
+ # 未找到现代文本框,检查替代/旧版文本框格式
206
+ if not textbox_elements and tag_name in ["drawing", "pict"]:
207
+ # 额外检查 DrawingML 和 VML 格式中的文本框
208
+ alt_txbx_xpath = etree.XPath(
209
+ ".//wps:txbx//w:p|.//w10:wrap//w:p|.//a:p//a:t",
210
+ namespaces=DocxConverter._BLIP_NAMESPACES,
211
+ )
212
+ textbox_elements = alt_txbx_xpath(element)
213
+
214
+ # 检查不在标准文本框内的形状文本
215
+ if not textbox_elements:
216
+ shape_text_xpath = etree.XPath(
217
+ ".//a:bodyPr/ancestor::*//a:t|.//a:txBody//a:t",
218
+ namespaces=DocxConverter._BLIP_NAMESPACES,
219
+ )
220
+ shape_text_elements = shape_text_xpath(element)
221
+ if shape_text_elements:
222
+ # 从形状文本创建自定义文本元素
223
+ raw_text = " ".join([t.text for t in shape_text_elements if t.text])
224
+ text_content = self._normalize_text_block_content(text_spans(raw_text))
225
+ visible_text = inline_span_plain_text(text_content)
226
+ if visible_text:
227
+ logger.debug(f"Found shape text: {visible_text[:50]}...")
228
+ self.cur_page.append(
229
+ {
230
+ "type": BlockType.TEXT,
231
+ "content": text_content,
232
+ }
233
+ )
234
+ if textbox_elements:
235
+ self.processed_textbox_elements.append(element)
236
+ for tb_element in textbox_elements:
237
+ self.processed_textbox_elements.append(tb_element)
238
+
239
+ logger.debug(f"Found textbox content with {len(textbox_elements)} elements")
240
+ self._handle_textbox_content(textbox_elements)
241
+
242
+ if tag_name == "tbl":
243
+ # 表格是顶层块级元素,会中断活跃列表的上下文。
244
+ # 若不重置列表状态,后续列表项会被追加到表格之前创建的列表块中,
245
+ # 导致表格在 cur_page 中出现在那些列表项之后,产生顺序错乱。
246
+ if self.pre_num_id != -1:
247
+ self._close_active_list()
248
+ try:
249
+ # 处理表格元素
250
+ self._handle_tables(element)
251
+ except Exception as e:
252
+ # 表格解析失败会导致整表丢失,需以 warning 级别暴露异常详情。
253
+ logger.warning(f"Could not parse a table, broken docx table: {e}")
254
+ # 检查图片元素
255
+ elif picture_refs:
256
+ # 判断图片是否为锚定(浮动)图片
257
+ is_anchored = bool(
258
+ element.findall(
259
+ ".//wp:anchor",
260
+ namespaces=DocxConverter._BLIP_NAMESPACES,
261
+ )
262
+ )
263
+ # 锚定图片在段落中浮动定位,段落文本应出现在图片之前
264
+ if is_anchored and tag_name == "p":
265
+ self._handle_text_elements(element)
266
+ self._handle_pictures(
267
+ picture_refs,
268
+ part=self._require_document_part(),
269
+ )
270
+ else:
271
+ # 处理图片元素
272
+ self._handle_pictures(
273
+ picture_refs,
274
+ part=self._require_document_part(),
275
+ )
276
+ # 如果是段落元素,同时处理其中的文本内容(如描述性文字)
277
+ if tag_name == "p":
278
+ self._handle_text_elements(element)
279
+ # 检查 sdt 元素
280
+ elif tag_name == "sdt":
281
+ sdt_content = element.find(".//w:sdtContent", namespaces=DocxConverter._BLIP_NAMESPACES)
282
+ if sdt_content is not None:
283
+ if self._is_toc_sdt(element):
284
+ # 处理目录SDT,转换为INDEX块
285
+ self._handle_sdt_as_index(sdt_content)
286
+ else:
287
+ # 其他SDT元素,按普通文本处理
288
+ paragraphs = sdt_content.findall(".//w:p", namespaces=DocxConverter._BLIP_NAMESPACES)
289
+ for p in paragraphs:
290
+ self._handle_text_elements(p)
291
+ # 检查文本段落元素
292
+ elif tag_name == "p":
293
+ # 处理文本元素(包括段落属性如"tcPr", "sectPr"等)
294
+ self._handle_text_elements(element)
295
+
296
+ # 忽略其他未知元素并记录日志
297
+ else:
298
+ logger.debug(f"Ignoring element in DOCX with tag: {tag_name}")
299
+
300
+ def _handle_text_elements(
301
+ self,
302
+ element: BaseOxmlElement,
303
+ ):
304
+ """
305
+ 处理文本元素。
306
+
307
+ Args:
308
+ element: 元素对象
309
+ doc: DoclingDocument 对象
310
+
311
+ Returns:
312
+
313
+ """
314
+ is_section_end = False
315
+ has_section_break = element.find(".//w:sectPr", namespaces=DocxConverter._BLIP_NAMESPACES) is not None
316
+ if has_section_break and not self._is_layout_only_section_break(element):
317
+ # 如果没有text内容
318
+ if element.text == "":
319
+ self._start_new_page()
320
+ else:
321
+ # 标记本节结束,处理完文本之后再分节
322
+ is_section_end = True
323
+ paragraph = Paragraph(element, self.docx_obj)
324
+ paragraph_elements = self._get_paragraph_elements(paragraph)
325
+ paragraph_text = self._get_paragraph_text(paragraph)
326
+ paragraph_anchor = self._extract_paragraph_bookmark(element)
327
+ text, equations = self._handle_equations_in_text(
328
+ element=element,
329
+ text=paragraph_text,
330
+ part=paragraph.part,
331
+ )
332
+
333
+ if text is None:
334
+ return None
335
+ text = text.strip()
336
+
337
+ if self._handle_plain_toc_paragraph_as_index(
338
+ paragraph=paragraph,
339
+ paragraph_element=element,
340
+ paragraph_elements=paragraph_elements,
341
+ text=text,
342
+ equations=equations,
343
+ ):
344
+ # 普通 TOC 是列表边界,避免后续同 numId 列表项继续合并到目录前的列表块。
345
+ if self.pre_num_id != -1:
346
+ self._close_active_list()
347
+ # 普通 TOC 段落被转换为 INDEX 后,也要保留段落末尾分节分页语义。
348
+ if is_section_end:
349
+ self._start_new_page()
350
+ return None
351
+ self._reset_index_state()
352
+
353
+ # 常见的项目符号和编号列表样式。
354
+ # "List Bullet", "List Number", "List Paragraph"
355
+ # 识别列表是否为编号列表
356
+ p_style_id, p_level = self._get_label_and_level(paragraph)
357
+ p_style_id = p_style_id or "Normal"
358
+ numid, ilevel = self._get_numId_and_ilvl(paragraph)
359
+
360
+ if numid == 0:
361
+ numid = None
362
+
363
+ # 处理列表
364
+ if numid is not None and ilevel is not None and p_style_id not in ["Title", "Heading"]:
365
+ # 通过检查 numFmt 来确认这是否实际上是编号列表
366
+ is_numbered = self._is_numbered_list(numid, ilevel)
367
+
368
+ if numid in self.heading_list_numids:
369
+ # 该列表被用作章节标题(列表项间穿插了正文内容),直接转换为title block
370
+ # 先关闭任何活跃的普通列表
371
+ if self.pre_num_id != -1:
372
+ self._close_active_list()
373
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
374
+ if content_text:
375
+ title_block = {
376
+ "type": BlockType.PARAGRAPH_TITLE,
377
+ "level": ilevel + 2,
378
+ "is_numbered_style": is_numbered,
379
+ "content": content_text,
380
+ }
381
+ if paragraph_anchor:
382
+ title_block["anchor"] = paragraph_anchor
383
+ self.cur_page.append(title_block)
384
+ else:
385
+ self._add_list_item(
386
+ numid=numid,
387
+ ilevel=ilevel,
388
+ elements=paragraph_elements,
389
+ is_numbered=is_numbered,
390
+ text=text,
391
+ equations=equations,
392
+ )
393
+ # 列表项已处理,返回
394
+ return None
395
+ elif ( # 列表结束处理
396
+ numid is None and self.pre_num_id != -1 and p_style_id not in ["Title", "Heading"]
397
+ ): # 关闭列表
398
+ # 重置列表状态
399
+ self._close_active_list()
400
+
401
+ if p_style_id in ["Title"]:
402
+ # 构建包含公式和超链接的文本
403
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
404
+ if content_text:
405
+ title_block = {
406
+ "type": BlockType.DOC_TITLE,
407
+ "level": 1,
408
+ "content": content_text,
409
+ }
410
+ if paragraph_anchor:
411
+ title_block["anchor"] = paragraph_anchor
412
+ self.cur_page.append(title_block)
413
+
414
+ elif "Heading" in p_style_id:
415
+ is_numbered_style = numid is not None and ilevel is not None and self._is_numbered_list(numid, ilevel)
416
+ # 构建包含公式和超链接的文本
417
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
418
+ if content_text:
419
+ h_block = {
420
+ "type": BlockType.PARAGRAPH_TITLE,
421
+ "level": max(p_level + 1, 2) if p_level is not None else 2,
422
+ "is_numbered_style": is_numbered_style,
423
+ "content": content_text,
424
+ }
425
+ if paragraph_anchor:
426
+ h_block["anchor"] = paragraph_anchor
427
+ self.cur_page.append(h_block)
428
+
429
+ elif equations:
430
+ equation_values = [value for kind, value in equations if kind == "equation"]
431
+ if (paragraph_text is None or len(paragraph_text.strip()) == 0) and equation_values:
432
+ # 独立公式
433
+ eq_block = {
434
+ "type": BlockType.EQUATION,
435
+ "content": equation_values[0] if len(equation_values) == 1 else "\n".join(equation_values),
436
+ }
437
+ self.cur_page.append(eq_block)
438
+ else:
439
+ # 包含行内公式的文本块,同时支持超链接
440
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
441
+ content_text = self._normalize_text_block_content(content_text)
442
+ if content_text:
443
+ text_with_inline_eq_block = {
444
+ "type": BlockType.TEXT,
445
+ "content": content_text,
446
+ }
447
+ if paragraph_anchor:
448
+ text_with_inline_eq_block["anchor"] = paragraph_anchor
449
+ self.cur_page.append(text_with_inline_eq_block)
450
+ elif p_style_id in [
451
+ "Paragraph",
452
+ "Normal",
453
+ "Subtitle",
454
+ "Author",
455
+ "DefaultText",
456
+ "ListParagraph",
457
+ "ListBullet",
458
+ "Quote",
459
+ ]:
460
+ # 构建包含公式和超链接的文本
461
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
462
+ content_text = self._normalize_text_block_content(content_text)
463
+ if content_text:
464
+ text_block = {
465
+ "type": BlockType.TEXT,
466
+ "content": content_text,
467
+ }
468
+ if paragraph_anchor:
469
+ text_block["anchor"] = paragraph_anchor
470
+ self.cur_page.append(text_block)
471
+ # 判断是否是 Caption
472
+ elif self._is_caption(element):
473
+ # 构建包含公式和超链接的文本
474
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
475
+ if content_text:
476
+ caption_block = {
477
+ "type": RAW_CAPTION,
478
+ "content": content_text,
479
+ }
480
+ self.cur_page.append(caption_block)
481
+ else:
482
+ # 文本样式名称不仅有默认值,还可能有用户自定义值
483
+ # 因此我们将所有其他标签视为纯文本
484
+ # 构建包含公式和超链接的文本
485
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
486
+ content_text = self._normalize_text_block_content(content_text)
487
+ if content_text:
488
+ text_block = {
489
+ "type": BlockType.TEXT,
490
+ "content": content_text,
491
+ }
492
+ if paragraph_anchor:
493
+ text_block["anchor"] = paragraph_anchor
494
+ self.cur_page.append(text_block)
495
+
496
+ if is_section_end:
497
+ self._start_new_page()
498
+
499
+ def _process_header_footer_paragraph(self, paragraph: Paragraph) -> list[dict[str, Any]]:
500
+ """
501
+ 处理页眉/页脚中的单个段落,支持行内公式和超链接。
502
+
503
+ Args:
504
+ paragraph: 段落对象
505
+
506
+ Returns:
507
+ list[dict]: 处理后的结构化 Span
508
+ """
509
+ paragraph_elements = self._get_paragraph_elements(paragraph)
510
+ paragraph_text = self._get_paragraph_text(paragraph)
511
+ text, equations = self._handle_equations_in_text(
512
+ element=paragraph._element,
513
+ text=paragraph_text,
514
+ part=paragraph.part,
515
+ )
516
+
517
+ text = text.strip()
518
+ if not text and not equations:
519
+ return []
520
+
521
+ # 构建包含公式和超链接的文本
522
+ content_text = self._build_text_with_equations_and_hyperlinks(paragraph_elements, text, equations)
523
+
524
+ return content_text
525
+
526
+ def _add_header_footer(self, docx_obj: DocxDocument) -> None:
527
+ """
528
+ 处理页眉和页脚,按照分节顺序添加到 pages 列表中,过滤掉空字符串和纯数字内容
529
+ 分为整个文档是否启用奇偶页不同和每一节是否启用首页不同两种情况,
530
+ 支持行内公式和超链接,并根据类型去重
531
+ """
532
+ is_odd_even_different = docx_obj.settings.odd_and_even_pages_header_footer
533
+ for sec_idx, section in enumerate(docx_obj.sections):
534
+ # 用于去重的集合
535
+ added_headers = set()
536
+ added_footers = set()
537
+
538
+ hdrs = [section.header]
539
+ if is_odd_even_different:
540
+ hdrs.append(section.even_page_header)
541
+ if section.different_first_page_header_footer:
542
+ hdrs.append(section.first_page_header)
543
+ for hdr in hdrs:
544
+ # 处理每个段落,支持公式和超链接
545
+ processed_parts: list[list[dict[str, Any]]] = []
546
+ for par in hdr.paragraphs:
547
+ content = self._process_header_footer_paragraph(par)
548
+ if content:
549
+ processed_parts.append(content)
550
+ spans: list[dict[str, Any]] = []
551
+ for part in processed_parts:
552
+ if spans:
553
+ append_text_span(spans, " ")
554
+ extend_inline_spans(spans, part)
555
+ visible = inline_span_plain_text(spans)
556
+ if spans and not visible.isdigit() and visible not in added_headers:
557
+ added_headers.add(visible)
558
+ try:
559
+ self.pages[sec_idx].append(
560
+ {
561
+ "type": BlockType.HEADER,
562
+ "content": spans,
563
+ }
564
+ )
565
+ except IndexError:
566
+ logger.error("Section index out of range when adding header.")
567
+
568
+ ftrs = [section.footer]
569
+ if is_odd_even_different:
570
+ ftrs.append(section.even_page_footer)
571
+ if section.different_first_page_header_footer:
572
+ ftrs.append(section.first_page_footer)
573
+ for ftr in ftrs:
574
+ # 处理每个段落,支持公式和超链接
575
+ processed_parts = []
576
+ for par in ftr.paragraphs:
577
+ content = self._process_header_footer_paragraph(par)
578
+ if content:
579
+ processed_parts.append(content)
580
+ spans = []
581
+ for part in processed_parts:
582
+ if spans:
583
+ append_text_span(spans, " ")
584
+ extend_inline_spans(spans, part)
585
+ visible = inline_span_plain_text(spans)
586
+ if spans and not visible.isdigit() and visible not in added_footers:
587
+ added_footers.add(visible)
588
+ try:
589
+ self.pages[sec_idx].append(
590
+ {
591
+ "type": BlockType.FOOTER,
592
+ "content": spans,
593
+ }
594
+ )
595
+ except IndexError:
596
+ logger.error("Section index out of range when adding footer.")
597
+
598
+ def _is_caption(self, element: BaseOxmlElement) -> bool:
599
+ """
600
+ 根据 insertText 中是否有 SEQ 字段来判断是否为 caption
601
+
602
+ Args:
603
+ element: 段落元素对象
604
+
605
+ Returns:
606
+ bool: 如果是标题返回 True,否则返回 False
607
+ """
608
+ instr_texts = element.findall(".//w:instrText", namespaces=DocxConverter._BLIP_NAMESPACES)
609
+
610
+ for instr in instr_texts:
611
+ if instr.text and "SEQ" in instr.text:
612
+ return True
613
+
614
+ return False