docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1002 @@
1
+ """把 ODF 文本、列表、表格和嵌入对象投影为 DocVortex raw blocks。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import base64
6
+ import html
7
+ import re
8
+ from collections.abc import Sequence
9
+ from dataclasses import dataclass
10
+ from typing import Any
11
+
12
+ from lxml import etree # type: ignore[reportMissingImports]
13
+
14
+ from .....schema import BlockType
15
+ from ..._shared.hyperlink import sanitize_hyperlink_target
16
+ from .....content.spans import (
17
+ append_equation_span,
18
+ append_text_span,
19
+ extend_inline_spans,
20
+ inline_span_plain_text,
21
+ strip_span_dicts,
22
+ )
23
+ from docvortex.content.mathml import mathml_to_latex
24
+ from docvortex.foundation.image_encoding import image_to_b64str
25
+ from ..image import create_text_placeholder, serialize_office_image
26
+ from ..rich_text import OfficeRichTextSegment, build_rich_text_from_segments
27
+ from .chart import parse_chart_block
28
+ from .constants import MAX_EXPANSION_TEXT_BYTES, qname
29
+ from .errors import OdfResourceLimitError
30
+ from .models import InlineAtom, InlineBlockGroup, InlineBreak, InlineImage, InlineMath, InlineNote, InlineText, TextStyle
31
+ from .package import OdfPackage
32
+ from .styles import OdfStyles
33
+ from .table import OdfTableExpansionBudget, parse_table_grid, table_grid_to_html
34
+
35
+
36
+ _WHITESPACE_RE = re.compile(r"[\t\r\n ]+")
37
+ _MAX_EXPLICIT_SPACE_COUNT = 10_000
38
+
39
+
40
+ @dataclass(slots=True)
41
+ class OdfTextExpansionBudget:
42
+ """记录单个 ODF 文档显式文本膨胀的累计字节数。"""
43
+
44
+ used_bytes: int = 0
45
+
46
+ def charge(self, byte_count: int) -> None:
47
+ """在分配膨胀文本前计费,超过固定上限时立即失败。"""
48
+ if byte_count < 0 or self.used_bytes > MAX_EXPANSION_TEXT_BYTES - byte_count:
49
+ raise OdfResourceLimitError(f"ODF resource limit exceeded: max_expansion_text_bytes={MAX_EXPANSION_TEXT_BYTES}")
50
+ self.used_bytes += byte_count
51
+
52
+
53
+ @dataclass(frozen=True, slots=True)
54
+ class OdfMasterPageChange:
55
+ """表示列表流中由段落样式请求的 master-page 变化。"""
56
+
57
+ master_page_name: str
58
+
59
+
60
+ RawFlowItem = dict[str, Any] | InlineNote
61
+ OdfListFlowItem = dict[str, Any] | InlineNote | OdfMasterPageChange
62
+
63
+
64
+ def _clean_xml_text(value: str | None) -> str:
65
+ """折叠 XML 排版空白,显式多空格由 text:s 单独恢复。"""
66
+ if not value:
67
+ return ""
68
+ return _WHITESPACE_RE.sub(" ", value)
69
+
70
+
71
+ def _paragraph_anchor(paragraph: etree._Element) -> str | None:
72
+ """返回段落最终能够挂载到输出 block 的首个 bookmark 名称。"""
73
+ for tag in (qname("text", "bookmark"), qname("text", "bookmark-start")):
74
+ bookmark = next(paragraph.iter(tag), None)
75
+ if bookmark is not None and (name := bookmark.get(qname("text", "name"))):
76
+ return name
77
+ return None
78
+
79
+
80
+ def collect_emittable_anchor_targets(root: etree._Element, styles: OdfStyles) -> frozenset[str]:
81
+ """收集 ODT 标题类 block 实际能够公开的 bookmark target。"""
82
+ targets: set[str] = set()
83
+ for paragraph in root.iter():
84
+ if paragraph.tag not in {qname("text", "p"), qname("text", "h")}:
85
+ continue
86
+ style_name = paragraph.get(qname("text", "style-name"))
87
+ if paragraph.tag != qname("text", "h") and not styles.is_document_title(style_name):
88
+ continue
89
+ if anchor := _paragraph_anchor(paragraph):
90
+ targets.add(anchor)
91
+ return frozenset(targets)
92
+
93
+
94
+ def _style_html(text: str, style: TextStyle) -> str:
95
+ """按稳定顺序把已转义文本包裹为 HTML 行内样式。"""
96
+ rendered = text
97
+ wrappers = [
98
+ (style.bold, "strong"),
99
+ (style.italic, "em"),
100
+ (style.underline, "u"),
101
+ (style.strikethrough, "s"),
102
+ (style.superscript, "sup"),
103
+ (style.subscript, "sub"),
104
+ ]
105
+ for enabled, tag in wrappers:
106
+ if enabled:
107
+ rendered = f"<{tag}>{rendered}</{tag}>"
108
+ return rendered
109
+
110
+
111
+ def _serialize_odf_image(
112
+ image_bytes: bytes,
113
+ *,
114
+ part_name: str | None,
115
+ content_type: str | None,
116
+ ) -> str | None:
117
+ """序列化 ODF 图片;SVG、SVM 和 GDIMeta 使用安全占位图保留对象位置。"""
118
+ normalized_type = (content_type or "").split(";", 1)[0].strip().casefold()
119
+ suffix = (part_name or "").rsplit(".", 1)[-1].casefold() if "." in (part_name or "") else ""
120
+ if normalized_type == "image/svg+xml" or suffix == "svg":
121
+ placeholder = create_text_placeholder((320, 180), ["SVG image", "Preview unavailable"])
122
+ return image_to_b64str(placeholder, image_format="JPEG")
123
+ if suffix == "svm" or "gdimetafile" in normalized_type or image_bytes.startswith(b"VCLMTF"):
124
+ placeholder = create_text_placeholder((320, 180), ["ODF vector image", "Preview unavailable"])
125
+ return image_to_b64str(placeholder, image_format="JPEG")
126
+ return serialize_office_image(image_bytes, part_name=part_name, content_type=content_type)
127
+
128
+
129
+ def render_atoms_to_html(atoms: Sequence[InlineAtom]) -> str:
130
+ """把 ODF 行内语义序列安全渲染为表格单元格 HTML。"""
131
+ parts: list[str] = []
132
+ for atom in atoms:
133
+ if isinstance(atom, InlineText):
134
+ rendered = _style_html(html.escape(atom.text), atom.style)
135
+ if atom.hyperlink:
136
+ rendered = f'<a href="{html.escape(atom.hyperlink, quote=True)}">{rendered}</a>'
137
+ parts.append(rendered)
138
+ elif isinstance(atom, InlineMath):
139
+ parts.append(f"<eq>{html.escape(atom.latex)}</eq>")
140
+ elif isinstance(atom, InlineBreak):
141
+ parts.append("<br/>")
142
+ elif isinstance(atom, InlineImage):
143
+ parts.append(f'<img src="{html.escape(atom.data_uri, quote=True)}" alt="{html.escape(atom.alt, quote=True)}"/>')
144
+ elif isinstance(atom, (InlineBlockGroup, InlineNote)):
145
+ continue
146
+ return "".join(parts)
147
+
148
+
149
+ def render_atoms_to_model(atoms: Sequence[InlineAtom], *, trim_edges: bool = False) -> list[dict[str, Any]]:
150
+ """把 ODF 行内语义序列直接转换为结构化 Span。"""
151
+ spans: list[dict[str, Any]] = []
152
+ segments: list[OfficeRichTextSegment] = []
153
+ text_fragments: list[str] = []
154
+ fragment_style: tuple[str, ...] | None = None
155
+ fragment_hyperlink: str | None = None
156
+
157
+ def flush_text_fragments() -> None:
158
+ """线性合并连续同样式文本,避免逐片段重复复制前缀。"""
159
+ nonlocal fragment_style, fragment_hyperlink
160
+ if not text_fragments:
161
+ return
162
+ segments.append(OfficeRichTextSegment("".join(text_fragments), fragment_style, fragment_hyperlink))
163
+ text_fragments.clear()
164
+ fragment_style = None
165
+ fragment_hyperlink = None
166
+
167
+ def flush_segments() -> None:
168
+ """把连续文本片段批量写入富文本结果。"""
169
+ flush_text_fragments()
170
+ if not segments:
171
+ return
172
+ extend_inline_spans(spans, build_rich_text_from_segments(list(segments), trim_plain_edges=trim_edges and not spans))
173
+ segments.clear()
174
+
175
+ for atom in atoms:
176
+ if isinstance(atom, InlineText):
177
+ style_names = atom.style.names()
178
+ hyperlink = atom.hyperlink
179
+ if text_fragments and (style_names != fragment_style or hyperlink != fragment_hyperlink):
180
+ flush_text_fragments()
181
+ if not text_fragments:
182
+ fragment_style = style_names
183
+ fragment_hyperlink = hyperlink
184
+ text_fragments.append(atom.text)
185
+ continue
186
+ flush_segments()
187
+ if isinstance(atom, InlineMath):
188
+ append_equation_span(spans, atom.latex)
189
+ elif isinstance(atom, InlineBreak):
190
+ append_text_span(spans, "\n")
191
+ elif isinstance(atom, InlineImage) and atom.alt:
192
+ append_text_span(spans, atom.alt)
193
+ elif isinstance(atom, (InlineBlockGroup, InlineNote)):
194
+ continue
195
+ flush_segments()
196
+ return strip_span_dicts(spans) if trim_edges else spans
197
+
198
+
199
+ class OdfBlockParser:
200
+ """在单个 ODF 包上下文中解析正文、表格和嵌入资源。"""
201
+
202
+ def __init__(
203
+ self,
204
+ package: OdfPackage,
205
+ styles: OdfStyles,
206
+ *,
207
+ base_part: str = "content.xml",
208
+ shared_notes: list[str] | None = None,
209
+ list_counters: dict[tuple[str, int], int] | None = None,
210
+ list_ids: dict[str, int] | None = None,
211
+ collect_cell_visuals: bool = False,
212
+ shared_cell_visuals: list[dict[str, Any]] | None = None,
213
+ anchor_targets: frozenset[str] | None = None,
214
+ text_expansion_budget: OdfTextExpansionBudget | None = None,
215
+ table_expansion_budget: OdfTableExpansionBudget | None = None,
216
+ ) -> None:
217
+ """绑定单次解析包、样式、子文档路径及可跨 parser 共享的状态。"""
218
+ self.package = package
219
+ self.styles = styles
220
+ self.base_part = base_part
221
+ self.notes = shared_notes if shared_notes is not None else []
222
+ self._list_counters = list_counters if list_counters is not None else {}
223
+ self._list_ids = list_ids if list_ids is not None else {}
224
+ self._collect_cell_visuals = collect_cell_visuals
225
+ self._cell_visuals = shared_cell_visuals if shared_cell_visuals is not None else []
226
+ self._anchor_targets = anchor_targets or frozenset()
227
+ self._text_expansion_budget = text_expansion_budget or OdfTextExpansionBudget()
228
+ self.table_expansion_budget = table_expansion_budget or OdfTableExpansionBudget()
229
+
230
+ def _append_text_atom(
231
+ self,
232
+ atoms: list[InlineAtom],
233
+ value: str | None,
234
+ *,
235
+ style: TextStyle,
236
+ hyperlink: str | None,
237
+ preserve_whitespace: bool = False,
238
+ ) -> None:
239
+ """清理并追加文本节点;显式 ODF 空格可跳过普通 XML 空白折叠。"""
240
+ text = value if preserve_whitespace else _clean_xml_text(value)
241
+ if not text:
242
+ return
243
+ atoms.append(InlineText(text=text, style=style, hyperlink=hyperlink))
244
+
245
+ def _walk_inlines(
246
+ self,
247
+ element: etree._Element,
248
+ *,
249
+ style: TextStyle,
250
+ hyperlink: str | None,
251
+ atoms: list[InlineAtom],
252
+ ) -> None:
253
+ """递归遍历段落行内节点,并把 frame 视觉对象旁路为 block。"""
254
+ self._append_text_atom(atoms, element.text, style=style, hyperlink=hyperlink)
255
+ for child in element:
256
+ if not isinstance(child.tag, str):
257
+ self._append_text_atom(atoms, child.tail, style=style, hyperlink=hyperlink)
258
+ continue
259
+ if child.tag == qname("text", "span"):
260
+ span_style = self.styles.text_style(
261
+ child.get(qname("text", "style-name")),
262
+ family="text",
263
+ inherited=style,
264
+ )
265
+ self._walk_inlines(
266
+ child,
267
+ style=span_style,
268
+ hyperlink=hyperlink,
269
+ atoms=atoms,
270
+ )
271
+ elif child.tag == qname("text", "a"):
272
+ target = sanitize_hyperlink_target(
273
+ child.get(qname("xlink", "href")),
274
+ allow_relative=True,
275
+ allow_fragment=True,
276
+ )
277
+ if target is not None and target.startswith("#") and target[1:] not in self._anchor_targets:
278
+ target = None
279
+ self._walk_inlines(
280
+ child,
281
+ style=style,
282
+ hyperlink=target or hyperlink,
283
+ atoms=atoms,
284
+ )
285
+ elif child.tag == qname("text", "s"):
286
+ count = _positive_space_count(child.get(qname("text", "c")))
287
+ self._text_expansion_budget.charge(count)
288
+ self._append_text_atom(
289
+ atoms,
290
+ " " * count,
291
+ style=style,
292
+ hyperlink=hyperlink,
293
+ preserve_whitespace=True,
294
+ )
295
+ elif child.tag == qname("text", "tab"):
296
+ self._append_text_atom(atoms, " ", style=style, hyperlink=hyperlink)
297
+ elif child.tag == qname("text", "line-break"):
298
+ atoms.append(InlineBreak())
299
+ elif child.tag == qname("text", "soft-page-break"):
300
+ pass
301
+ elif child.tag == qname("text", "note"):
302
+ self._parse_note(child, style=style, hyperlink=hyperlink, atoms=atoms)
303
+ elif child.tag == qname("office", "annotation"):
304
+ if annotation_text := self._annotation_text(child):
305
+ atoms.append(InlineNote(annotation_text))
306
+ elif child.tag == qname("office", "annotation-end"):
307
+ pass
308
+ elif child.tag == qname("draw", "frame"):
309
+ inline_atom, blocks = self._parse_frame(child)
310
+ if inline_atom is not None:
311
+ atoms.append(inline_atom)
312
+ if blocks:
313
+ atoms.append(
314
+ InlineBlockGroup(
315
+ tuple(blocks),
316
+ inline_image_rendered=isinstance(inline_atom, InlineImage),
317
+ )
318
+ )
319
+ elif child.tag == qname("math", "math"):
320
+ if latex := mathml_to_latex(child):
321
+ atoms.append(InlineMath(latex))
322
+ elif child.tag in {
323
+ qname("text", "bookmark"),
324
+ qname("text", "bookmark-start"),
325
+ qname("text", "bookmark-end"),
326
+ }:
327
+ pass
328
+ else:
329
+ self._walk_inlines(
330
+ child,
331
+ style=style,
332
+ hyperlink=hyperlink,
333
+ atoms=atoms,
334
+ )
335
+ self._append_text_atom(atoms, child.tail, style=style, hyperlink=hyperlink)
336
+
337
+ def _annotation_text(self, annotation: etree._Element) -> str:
338
+ """只提取 ODF annotation 的正文段落与列表,不混入作者日期元数据。"""
339
+ return flatten_block_text(self.parse_container(annotation)).strip()
340
+
341
+ def _parse_note(
342
+ self,
343
+ note: etree._Element,
344
+ *,
345
+ style: TextStyle,
346
+ hyperlink: str | None,
347
+ atoms: list[InlineAtom],
348
+ ) -> None:
349
+ """保留脚注标记,并把 note-body 内容排入当前逻辑页脚注队列。"""
350
+ citation = note.find(qname("text", "note-citation"))
351
+ citation_text = (
352
+ "".join(citation.itertext()).strip()
353
+ if citation is not None
354
+ else str(len(self.notes) + sum(isinstance(atom, InlineNote) for atom in atoms) + 1)
355
+ )
356
+ self._append_text_atom(atoms, f"[{citation_text}]", style=style, hyperlink=hyperlink)
357
+ body = note.find(qname("text", "note-body"))
358
+ if body is None:
359
+ return
360
+ blocks = self.parse_container(body)
361
+ visible = flatten_block_text(blocks)
362
+ if visible:
363
+ atoms.append(InlineNote(f"[{citation_text}] {visible}"))
364
+
365
+ def parse_inline_atoms(self, paragraph: etree._Element) -> list[InlineAtom]:
366
+ """解析一个段落的行内语义,并用原位 marker 保留段外 block。"""
367
+ paragraph_style = self.styles.text_style(
368
+ paragraph.get(qname("text", "style-name")),
369
+ family="paragraph",
370
+ )
371
+ atoms: list[InlineAtom] = []
372
+ self._walk_inlines(
373
+ paragraph,
374
+ style=paragraph_style,
375
+ hyperlink=None,
376
+ atoms=atoms,
377
+ )
378
+ return atoms
379
+
380
+ def parse_paragraph(self, paragraph: etree._Element) -> list[RawFlowItem]:
381
+ """把 text:p/text:h 转为标题、正文、公式和段外内容。"""
382
+ atoms = self.parse_inline_atoms(paragraph)
383
+ results: list[RawFlowItem] = []
384
+ is_heading = paragraph.tag == qname("text", "h")
385
+ style_name = paragraph.get(qname("text", "style-name"))
386
+ content_atoms = [atom for atom in atoms if not isinstance(atom, (InlineBlockGroup, InlineNote))]
387
+ content = render_atoms_to_model(content_atoms, trim_edges=True)
388
+ math_atoms = [atom for atom in content_atoms if isinstance(atom, InlineMath)]
389
+ visible_text = "".join(atom.text for atom in content_atoms if isinstance(atom, InlineText)).strip()
390
+ if content:
391
+ if math_atoms and not visible_text and len(math_atoms) == 1 and len(content_atoms) == 1:
392
+ results.append({"type": BlockType.EQUATION, "content": math_atoms[0].latex})
393
+ elif self.styles.is_document_title(style_name):
394
+ block: dict[str, Any] = {"type": BlockType.DOC_TITLE, "level": 1, "content": content}
395
+ if anchor := _paragraph_anchor(paragraph):
396
+ block["anchor"] = anchor
397
+ results.append(block)
398
+ elif is_heading:
399
+ try:
400
+ outline_level = int(paragraph.get(qname("text", "outline-level"), "1"))
401
+ except ValueError:
402
+ outline_level = 1
403
+ block = {
404
+ "type": BlockType.PARAGRAPH_TITLE,
405
+ "level": min(max(outline_level + 1, 2), 6),
406
+ "is_numbered_style": False,
407
+ "content": content,
408
+ }
409
+ if anchor := _paragraph_anchor(paragraph):
410
+ block["anchor"] = anchor
411
+ results.append(block)
412
+ else:
413
+ results.append({"type": BlockType.TEXT, "content": content})
414
+ for atom in atoms:
415
+ if isinstance(atom, InlineBlockGroup):
416
+ results.extend(atom.blocks)
417
+ elif isinstance(atom, InlineNote):
418
+ results.append(atom)
419
+ return results
420
+
421
+ def parse_list(
422
+ self,
423
+ element: etree._Element,
424
+ *,
425
+ depth: int = 0,
426
+ inherited_style: str | None = None,
427
+ emit_master_page_changes: bool = False,
428
+ ) -> list[OdfListFlowItem]:
429
+ """递归构造严格 LIST 分片,并把不允许嵌套的 block 提升为有序兄弟。"""
430
+ items = [
431
+ item
432
+ for item in element
433
+ if isinstance(item.tag, str) and item.tag in {qname("text", "list-item"), qname("text", "list-header")}
434
+ ]
435
+ return self._parse_list_items(
436
+ element,
437
+ items,
438
+ depth=depth,
439
+ inherited_style=inherited_style,
440
+ emit_master_page_changes=emit_master_page_changes,
441
+ )
442
+
443
+ def _parse_list_items(
444
+ self,
445
+ element: etree._Element,
446
+ items: Sequence[etree._Element],
447
+ *,
448
+ depth: int,
449
+ inherited_style: str | None,
450
+ emit_master_page_changes: bool,
451
+ ) -> list[OdfListFlowItem]:
452
+ """按源条目构造 LIST 分片,每个条目只保留一个文本叶子和一个 marker。"""
453
+ style_name = element.get(qname("text", "style-name")) or inherited_style
454
+ level = self.styles.list_level(style_name, depth)
455
+ key = (style_name or "", depth)
456
+ start = level.start
457
+ continue_list = element.get(qname("text", "continue-list"))
458
+ if continue_list and continue_list in self._list_ids:
459
+ start = self._list_ids[continue_list]
460
+ elif element.get(qname("text", "continue-numbering")) == "true" and key in self._list_counters:
461
+ start = self._list_counters[key]
462
+ results: list[OdfListFlowItem] = []
463
+ content: list[dict[str, Any]] = []
464
+ fragment_notes: list[InlineNote] = []
465
+ item_count = 0
466
+ active_master: str | None = None
467
+
468
+ def flush_content(fragment_start: int) -> None:
469
+ """把当前合法子块冻结为一个 LIST 分片。"""
470
+ if content:
471
+ block: dict[str, Any] = {
472
+ "type": BlockType.LIST,
473
+ "attribute": "ordered" if level.ordered else "unordered",
474
+ "ilevel": depth,
475
+ "content": list(content),
476
+ }
477
+ if level.ordered:
478
+ block["start"] = fragment_start
479
+ results.append(block)
480
+ content.clear()
481
+ if fragment_notes:
482
+ results.extend(fragment_notes)
483
+ fragment_notes.clear()
484
+
485
+ fragment_start = start
486
+ for item in items:
487
+ is_header = item.tag == qname("text", "list-header")
488
+ if not is_header:
489
+ if item_count == 0:
490
+ try:
491
+ item_start = int(item.get(qname("text", "start-value"), str(start)))
492
+ start = max(0, item_start)
493
+ fragment_start = start
494
+ except ValueError:
495
+ pass
496
+ # 统一 LIST 只支持列表级起始值,后续逐项重启按连续序号投影。
497
+ item_count += 1
498
+ first_paragraph = next(
499
+ (
500
+ child
501
+ for child in item
502
+ if isinstance(child.tag, str) and child.tag in {qname("text", "p"), qname("text", "h")}
503
+ ),
504
+ None,
505
+ )
506
+ requested_master = (
507
+ self.styles.paragraph_master_page_name(first_paragraph.get(qname("text", "style-name")))
508
+ if first_paragraph is not None
509
+ else None
510
+ )
511
+ if emit_master_page_changes and requested_master is not None and requested_master != active_master:
512
+ flush_content(fragment_start)
513
+ results.append(OdfMasterPageChange(requested_master))
514
+ active_master = requested_master
515
+ fragment_start = start + item_count - (0 if is_header else 1)
516
+ text_content: list[dict[str, Any]] = []
517
+ nested_blocks: list[dict[str, Any]] = []
518
+ lifted_blocks: list[OdfListFlowItem] = []
519
+
520
+ def consume_flow(flow: Sequence[RawFlowItem]) -> None:
521
+ """把段落子流投影到列表文本、嵌套列表或提升块。"""
522
+ for block in flow:
523
+ if isinstance(block, InlineNote):
524
+ if emit_master_page_changes:
525
+ fragment_notes.append(block)
526
+ else:
527
+ self.notes.append(block.content)
528
+ continue
529
+ block_type = block.get("type")
530
+ block_content = block.get("content")
531
+ if block_type in {
532
+ BlockType.TEXT,
533
+ BlockType.REF_TEXT,
534
+ BlockType.DOC_TITLE,
535
+ BlockType.PARAGRAPH_TITLE,
536
+ } and isinstance(block_content, list):
537
+ block_spans = [span for span in block_content if isinstance(span, dict)]
538
+ if block_spans:
539
+ if text_content:
540
+ append_text_span(text_content, "\n")
541
+ extend_inline_spans(text_content, block_spans)
542
+ elif block_type == BlockType.LIST:
543
+ nested_blocks.append(block)
544
+ else:
545
+ lifted_blocks.append(block)
546
+
547
+ for child in item:
548
+ if not isinstance(child.tag, str):
549
+ continue
550
+ if child.tag in {qname("text", "p"), qname("text", "h")}:
551
+ consume_flow(self.parse_paragraph(child))
552
+ elif child.tag == qname("text", "list"):
553
+ nested_flow = self.parse_list_blocks(
554
+ child,
555
+ depth=depth + 1,
556
+ inherited_style=style_name,
557
+ emit_master_page_changes=emit_master_page_changes,
558
+ )
559
+ if not any(isinstance(block, OdfMasterPageChange) for block in nested_flow) and all(
560
+ isinstance(block, InlineNote) or block.get("type") == BlockType.LIST for block in nested_flow
561
+ ):
562
+ nested_blocks.extend(block for block in nested_flow if isinstance(block, dict))
563
+ fragment_notes.extend(block for block in nested_flow if isinstance(block, InlineNote))
564
+ else:
565
+ lifted_blocks.extend(nested_flow)
566
+ else:
567
+ consume_flow(self.parse_container(child))
568
+ if text_content:
569
+ content.append({"type": BlockType.TEXT, "content": text_content})
570
+ content.extend(nested_blocks)
571
+ if lifted_blocks:
572
+ flush_content(fragment_start)
573
+ results.extend(lifted_blocks)
574
+ fragment_start = start + item_count
575
+
576
+ flush_content(fragment_start)
577
+ next_value = start + item_count
578
+ self._list_counters[key] = next_value
579
+ if list_id := element.get(qname("xml", "id")):
580
+ self._list_ids[list_id] = next_value
581
+ return results
582
+
583
+ def parse_list_blocks(
584
+ self,
585
+ element: etree._Element,
586
+ *,
587
+ depth: int = 0,
588
+ inherited_style: str | None = None,
589
+ emit_master_page_changes: bool = False,
590
+ ) -> list[OdfListFlowItem]:
591
+ """把含 text:h 的编号章节提升为标题,并保留其余连续列表。"""
592
+ if next(element.iter(qname("text", "h")), None) is None:
593
+ return self.parse_list(
594
+ element,
595
+ depth=depth,
596
+ inherited_style=inherited_style,
597
+ emit_master_page_changes=emit_master_page_changes,
598
+ )
599
+ results: list[OdfListFlowItem] = []
600
+ pending_items: list[etree._Element] = []
601
+ active_master: str | None = None
602
+
603
+ def flush_pending() -> None:
604
+ """把标题之间积累的普通列表项写为独立连续 LIST block。"""
605
+ if not pending_items:
606
+ return
607
+ blocks = self._parse_list_items(
608
+ element,
609
+ list(pending_items),
610
+ depth=depth,
611
+ inherited_style=inherited_style,
612
+ emit_master_page_changes=emit_master_page_changes,
613
+ )
614
+ pending_items.clear()
615
+ results.extend(blocks)
616
+
617
+ for item in element:
618
+ if not isinstance(item.tag, str) or item.tag not in {
619
+ qname("text", "list-item"),
620
+ qname("text", "list-header"),
621
+ }:
622
+ continue
623
+ if next(item.iter(qname("text", "h")), None) is None:
624
+ pending_items.append(item)
625
+ continue
626
+ flush_pending()
627
+ for child in item:
628
+ if not isinstance(child.tag, str):
629
+ continue
630
+ if child.tag in {qname("text", "p"), qname("text", "h")}:
631
+ requested_master = self.styles.paragraph_master_page_name(child.get(qname("text", "style-name")))
632
+ if emit_master_page_changes and requested_master is not None and requested_master != active_master:
633
+ results.append(OdfMasterPageChange(requested_master))
634
+ active_master = requested_master
635
+ for parsed in self.parse_paragraph(child):
636
+ if isinstance(parsed, InlineNote):
637
+ if emit_master_page_changes:
638
+ results.append(parsed)
639
+ else:
640
+ self.notes.append(parsed.content)
641
+ continue
642
+ if child.tag == qname("text", "h") and parsed.get("type") == BlockType.PARAGRAPH_TITLE:
643
+ parsed["is_numbered_style"] = True
644
+ results.append(parsed)
645
+ elif child.tag == qname("text", "list"):
646
+ results.extend(
647
+ self.parse_list_blocks(
648
+ child,
649
+ depth=depth + 1,
650
+ inherited_style=element.get(qname("text", "style-name")) or inherited_style,
651
+ emit_master_page_changes=emit_master_page_changes,
652
+ )
653
+ )
654
+ else:
655
+ results.extend(self.parse_element(child))
656
+ flush_pending()
657
+ return results
658
+
659
+ def _parse_index(self, element: etree._Element) -> dict[str, Any] | None:
660
+ """把 ODF 已存储目录正文转换为扁平 INDEX 子项。"""
661
+ leaves: list[dict[str, Any]] = []
662
+ for paragraph in element.iter():
663
+ if paragraph.tag not in {qname("text", "p"), qname("text", "h")}:
664
+ continue
665
+ for block in self.parse_paragraph(paragraph):
666
+ if isinstance(block, InlineNote):
667
+ self.notes.append(block.content)
668
+ continue
669
+ if isinstance(block, dict) and block.get("content"):
670
+ leaves.append({"type": BlockType.TEXT, "content": block["content"]})
671
+ if not leaves:
672
+ return None
673
+ return {"type": BlockType.INDEX, "ilevel": 0, "content": leaves}
674
+
675
+ def parse_table(self, element: etree._Element) -> dict[str, Any] | None:
676
+ """把一个 ODF table 转为包含合并语义的 TABLE raw block。"""
677
+ grid = parse_table_grid(element, self.render_cell_html, expansion_budget=self.table_expansion_budget)
678
+ content = table_grid_to_html(grid)
679
+ return {"type": BlockType.TABLE, "content": content} if content else None
680
+
681
+ def _load_image(self, image: etree._Element) -> tuple[str | None, str]:
682
+ """读取 draw:image 的包内或内联载荷并复用 Office 图片序列化。"""
683
+ href = image.get(qname("xlink", "href"), "")
684
+ part_name = self.package.resolve_reference(href, base_part=self.base_part) if href else None
685
+ image_bytes: bytes | None = None
686
+ content_type: str | None = None
687
+ if part_name:
688
+ image_bytes = self.package.read_part(part_name, asset=True)
689
+ content_type = self.package.content_type_for(part_name)
690
+ if image_bytes is None:
691
+ binary = image.find(f".//{qname('office', 'binary-data')}")
692
+ if binary is not None and (binary.text or "").strip():
693
+ try:
694
+ image_bytes = base64.b64decode("".join((binary.text or "").split()), validate=True)
695
+ except (ValueError, TypeError):
696
+ image_bytes = None
697
+ alt = ""
698
+ parent = image.getparent()
699
+ if parent is not None:
700
+ title = parent.find(qname("svg", "title"))
701
+ description = parent.find(qname("svg", "desc"))
702
+ alt = " ".join(
703
+ text.strip()
704
+ for text in (
705
+ "".join(title.itertext()) if title is not None else "",
706
+ "".join(description.itertext()) if description is not None else "",
707
+ )
708
+ if text.strip()
709
+ )
710
+ if not image_bytes:
711
+ return None, alt
712
+ return _serialize_odf_image(image_bytes, part_name=part_name, content_type=content_type), alt
713
+
714
+ def _object_root(self, object_element: etree._Element) -> tuple[etree._Element | None, str | None]:
715
+ """读取 draw:object 指向的子文档内容树和成员路径。"""
716
+ inline_math = next(object_element.iter(qname("math", "math")), None)
717
+ if inline_math is not None:
718
+ return inline_math, self.base_part
719
+ href = object_element.get(qname("xlink", "href"), "")
720
+ part_name = self.package.resolve_object_content(href, base_part=self.base_part)
721
+ if part_name is None:
722
+ return None, None
723
+ return self.package.xml_part(part_name), part_name
724
+
725
+ def _parse_frame(self, frame: etree._Element) -> tuple[InlineAtom | None, list[dict[str, Any]]]:
726
+ """按公式、图表、文本框、表格、图片优先级解析一个 draw:frame。"""
727
+ image_element = next(frame.iter(qname("draw", "image")), None)
728
+ preview_uri: str | None = None
729
+ preview_alt = ""
730
+
731
+ def load_preview() -> tuple[str | None, str]:
732
+ """只在对象需要图片回退或图表预览时读取 sibling draw:image。"""
733
+ nonlocal preview_uri, preview_alt
734
+ if image_element is not None and preview_uri is None:
735
+ preview_uri, preview_alt = self._load_image(image_element)
736
+ return preview_uri, preview_alt
737
+
738
+ object_element = next(frame.iter(qname("draw", "object")), None)
739
+ if object_element is not None:
740
+ object_root, object_part = self._object_root(object_element)
741
+ if object_root is not None:
742
+ math_element = (
743
+ object_root
744
+ if object_root.tag == qname("math", "math")
745
+ else next(
746
+ object_root.iter(qname("math", "math")),
747
+ None,
748
+ )
749
+ )
750
+ if math_element is not None and (latex := mathml_to_latex(math_element)):
751
+ return InlineMath(latex), []
752
+ load_preview()
753
+ object_parser = OdfBlockParser(
754
+ self.package,
755
+ self.styles,
756
+ base_part=object_part or self.base_part,
757
+ shared_notes=self.notes,
758
+ list_counters=self._list_counters,
759
+ list_ids=self._list_ids,
760
+ collect_cell_visuals=self._collect_cell_visuals,
761
+ shared_cell_visuals=self._cell_visuals,
762
+ anchor_targets=self._anchor_targets,
763
+ text_expansion_budget=self._text_expansion_budget,
764
+ table_expansion_budget=self.table_expansion_budget,
765
+ )
766
+ chart = parse_chart_block(
767
+ object_root,
768
+ render_cell=object_parser.render_cell_html,
769
+ preview_data_uri=preview_uri,
770
+ table_expansion_budget=self.table_expansion_budget,
771
+ )
772
+ if chart is not None:
773
+ return None, [chart]
774
+ text_box = next(frame.iter(qname("draw", "text-box")), None)
775
+ if text_box is not None:
776
+ return None, self.parse_container(text_box)
777
+ table = next(frame.iter(qname("table", "table")), None)
778
+ if table is not None and (table_block := self.parse_table(table)) is not None:
779
+ return None, [table_block]
780
+ load_preview()
781
+ if preview_uri:
782
+ return InlineImage(preview_uri, preview_alt), [{"type": BlockType.IMAGE, "image_base64": preview_uri}]
783
+ if preview_alt:
784
+ return InlineText(preview_alt), []
785
+ return None, []
786
+
787
+ def parse_frame_blocks(self, frame: etree._Element) -> list[dict[str, Any]]:
788
+ """把 frame 的内联结果提升为页面级 block,避免正文重复图片。"""
789
+ inline, blocks = self._parse_frame(frame)
790
+ if blocks:
791
+ return blocks
792
+ if isinstance(inline, InlineMath):
793
+ return [{"type": BlockType.EQUATION, "content": inline.latex}]
794
+ if isinstance(inline, InlineImage):
795
+ return [{"type": BlockType.IMAGE, "image_base64": inline.data_uri}]
796
+ if isinstance(inline, InlineText):
797
+ content = render_atoms_to_model([inline], trim_edges=True)
798
+ return [{"type": BlockType.TEXT, "content": content}] if content else []
799
+ return []
800
+
801
+ def parse_element(self, element: etree._Element) -> list[dict[str, Any]]:
802
+ """解析一个 ODF block 元素,不移动或修改原始 XML 节点。"""
803
+ if element.tag in {qname("text", "p"), qname("text", "h")}:
804
+ blocks: list[dict[str, Any]] = []
805
+ for item in self.parse_paragraph(element):
806
+ if isinstance(item, dict):
807
+ blocks.append(item)
808
+ elif isinstance(item, InlineNote):
809
+ self.notes.append(item.content)
810
+ return blocks
811
+ if element.tag == qname("text", "list"):
812
+ return [item for item in self.parse_list_blocks(element) if isinstance(item, dict)]
813
+ if element.tag == qname("office", "annotation"):
814
+ if annotation_text := self._annotation_text(element):
815
+ self.notes.append(annotation_text)
816
+ return []
817
+ if element.tag == qname("office", "annotation-end"):
818
+ return []
819
+ if element.tag == qname("table", "table"):
820
+ table_block = self.parse_table(element)
821
+ return [table_block] if table_block is not None else []
822
+ if element.tag == qname("draw", "frame"):
823
+ return self.parse_frame_blocks(element)
824
+ if element.tag in {
825
+ qname("text", "section"),
826
+ qname("text", "index-body"),
827
+ qname("text", "index-title"),
828
+ qname("draw", "g"),
829
+ qname("draw", "custom-shape"),
830
+ }:
831
+ return self.parse_container(element)
832
+ if element.tag in {
833
+ qname("text", "table-of-content"),
834
+ qname("text", "alphabetical-index"),
835
+ qname("text", "bibliography"),
836
+ qname("text", "illustration-index"),
837
+ }:
838
+ index = self._parse_index(element)
839
+ return [index] if index is not None else []
840
+ return []
841
+
842
+ def parse_container(self, parent: etree._Element) -> list[dict[str, Any]]:
843
+ """按文档顺序解析普通 ODF block 容器,不建立页面边界。"""
844
+ blocks: list[dict[str, Any]] = []
845
+ for child in parent:
846
+ if isinstance(child.tag, str):
847
+ blocks.extend(self.parse_element(child))
848
+ return blocks
849
+
850
+ def _append_cell_blocks(
851
+ self,
852
+ parts: list[str],
853
+ blocks: Sequence[dict[str, Any]],
854
+ *,
855
+ inline_image_rendered: bool,
856
+ ) -> None:
857
+ """按单元格视觉策略收集或内联 block,并避免重复输出配对图片。"""
858
+ for block in blocks:
859
+ block_type = block.get("type")
860
+ if self._collect_cell_visuals and block_type in {
861
+ BlockType.IMAGE,
862
+ BlockType.CHART,
863
+ BlockType.EQUATION,
864
+ }:
865
+ self._cell_visuals.append(block)
866
+ continue
867
+ if inline_image_rendered and block_type == BlockType.IMAGE:
868
+ continue
869
+ if block_type in {BlockType.TABLE, BlockType.CHART} and block.get("content"):
870
+ parts.append(str(block["content"]))
871
+ elif block.get("image_base64"):
872
+ parts.append(f'<img src="{html.escape(str(block["image_base64"]), quote=True)}"/>')
873
+
874
+ def _queue_inline_notes(self, atoms: Sequence[InlineAtom]) -> None:
875
+ """把单元格行内流中的 note marker 排入当前逻辑页队列。"""
876
+ self.notes.extend(atom.content for atom in atoms if isinstance(atom, InlineNote))
877
+
878
+ def render_cell_html(self, cell: etree._Element) -> str:
879
+ """把表格单元格中的段落、列表、嵌套表和 frame 转为 HTML。"""
880
+ parts: list[str] = []
881
+ for child in cell:
882
+ if not isinstance(child.tag, str):
883
+ continue
884
+ if child.tag in {qname("text", "p"), qname("text", "h")}:
885
+ atoms = self.parse_inline_atoms(child)
886
+ self._queue_inline_notes(atoms)
887
+ rendered_atoms = (
888
+ [atom for atom in atoms if not isinstance(atom, InlineImage)] if self._collect_cell_visuals else atoms
889
+ )
890
+ parts.append(f"<p>{render_atoms_to_html(rendered_atoms)}</p>")
891
+ for atom in atoms:
892
+ if isinstance(atom, InlineBlockGroup):
893
+ self._append_cell_blocks(
894
+ parts,
895
+ atom.blocks,
896
+ inline_image_rendered=atom.inline_image_rendered and not self._collect_cell_visuals,
897
+ )
898
+ elif child.tag == qname("text", "list"):
899
+ parts.append(self._render_list_html(child))
900
+ elif child.tag == qname("table", "table"):
901
+ nested = parse_table_grid(child, self.render_cell_html, expansion_budget=self.table_expansion_budget)
902
+ parts.append(table_grid_to_html(nested))
903
+ elif child.tag == qname("draw", "frame"):
904
+ inline, blocks = self._parse_frame(child)
905
+ if inline is not None and not (self._collect_cell_visuals and isinstance(inline, InlineImage)):
906
+ parts.append(render_atoms_to_html([inline]))
907
+ self._append_cell_blocks(
908
+ parts,
909
+ blocks,
910
+ inline_image_rendered=isinstance(inline, InlineImage) and not self._collect_cell_visuals,
911
+ )
912
+ return "".join(part for part in parts if part)
913
+
914
+ def _render_list_html(self, element: etree._Element, *, depth: int = 0, inherited_style: str | None = None) -> str:
915
+ """把单元格内 ODF 列表递归渲染为 ol/ul HTML。"""
916
+ style_name = element.get(qname("text", "style-name")) or inherited_style
917
+ level = self.styles.list_level(style_name, depth)
918
+ tag = "ol" if level.ordered else "ul"
919
+ start = f' start="{level.start}"' if level.ordered and level.start != 1 else ""
920
+ parts = [f"<{tag}{start}>"]
921
+ for item in element:
922
+ if item.tag not in {qname("text", "list-item"), qname("text", "list-header")}:
923
+ continue
924
+ if item.tag == qname("text", "list-header"):
925
+ for child in item:
926
+ if child.tag in {qname("text", "p"), qname("text", "h")}:
927
+ atoms = self.parse_inline_atoms(child)
928
+ self._queue_inline_notes(atoms)
929
+ parts.append(f"<li>{render_atoms_to_html(atoms)}</li>")
930
+ continue
931
+ parts.append("<li>")
932
+ for child in item:
933
+ if child.tag in {qname("text", "p"), qname("text", "h")}:
934
+ atoms = self.parse_inline_atoms(child)
935
+ self._queue_inline_notes(atoms)
936
+ parts.append(render_atoms_to_html(atoms))
937
+ elif child.tag == qname("text", "list"):
938
+ parts.append(self._render_list_html(child, depth=depth + 1, inherited_style=style_name))
939
+ parts.append("</li>")
940
+ parts.append(f"</{tag}>")
941
+ return "".join(parts)
942
+
943
+ def drain_notes(self) -> list[str]:
944
+ """取出当前累计脚注并清空共享队列。"""
945
+ values = list(self.notes)
946
+ self.notes.clear()
947
+ return values
948
+
949
+ def drain_cell_visuals(self) -> list[dict[str, Any]]:
950
+ """取出 ODS 单元格解析期间收集的视觉对象并清空队列。"""
951
+ values = list(self._cell_visuals)
952
+ self._cell_visuals.clear()
953
+ return values
954
+
955
+
956
+ def _positive_space_count(value: str | None) -> int:
957
+ """在整数转换前校验并限制 text:s 重复空格数,非法值按一处理。"""
958
+ normalized = (value or "").strip()
959
+ if normalized.startswith("+"):
960
+ normalized = normalized[1:]
961
+ if not normalized or not normalized.isascii() or not normalized.isdigit():
962
+ return 1
963
+ significant = normalized.lstrip("0")
964
+ if not significant:
965
+ return 1
966
+ max_digits = len(str(_MAX_EXPLICIT_SPACE_COUNT))
967
+ if len(significant) > max_digits:
968
+ return _MAX_EXPLICIT_SPACE_COUNT
969
+ return min(max(1, int(significant)), _MAX_EXPLICIT_SPACE_COUNT)
970
+
971
+
972
+ def flatten_block_text(blocks: list[dict[str, Any]]) -> str:
973
+ """递归提取 raw block 的可见字符串,供标题和备注聚合。"""
974
+ parts: list[str] = []
975
+ for block in blocks:
976
+ content = block.get("content")
977
+ if isinstance(content, str):
978
+ if content.strip():
979
+ visible = re.sub(r"<[^>]+>", "", content)
980
+ parts.append(html.unescape(visible).strip())
981
+ elif isinstance(content, list):
982
+ children = [child for child in content if isinstance(child, dict)]
983
+ if children and all(
984
+ child.get("type") in {"text", "equation_inline", "code_inline", "hyperlink"} for child in children
985
+ ):
986
+ nested = inline_span_plain_text(children)
987
+ else:
988
+ nested = flatten_block_text(children)
989
+ if nested:
990
+ parts.append(nested)
991
+ return "\n".join(parts)
992
+
993
+
994
+ __all__ = [
995
+ "OdfBlockParser",
996
+ "OdfMasterPageChange",
997
+ "OdfTextExpansionBudget",
998
+ "RawFlowItem",
999
+ "flatten_block_text",
1000
+ "render_atoms_to_html",
1001
+ "render_atoms_to_model",
1002
+ ]