docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,905 @@
1
+ """严格 MiddleJson 到可编辑 DOCX bytes 的公共渲染实现。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterable
6
+ from io import BytesIO
7
+ import re
8
+
9
+ from bs4 import BeautifulSoup, NavigableString, Tag
10
+ from docx import Document
11
+ from docx.document import Document as DocumentType
12
+ from docx.enum.table import WD_CELL_VERTICAL_ALIGNMENT
13
+ from docx.enum.text import WD_ALIGN_PARAGRAPH, WD_TAB_ALIGNMENT
14
+ from docx.oxml import OxmlElement
15
+ from docx.oxml.ns import qn
16
+ from docx.opc.constants import RELATIONSHIP_TYPE as RT
17
+ from docx.opc.part import Part
18
+ from docx.shared import Emu, Mm, Pt, Twips
19
+ from docx.table import _Cell
20
+ from docx.text.paragraph import Paragraph
21
+ from lxml import etree
22
+ from loguru import logger
23
+
24
+ from ..common.index import strip_index_page_tail
25
+ from ....content.inline import normalize_inline_spans
26
+ from ..common.list_items import parse_list_item_marker
27
+ from ..common.planner import PlannedBlock, build_render_plan
28
+ from .assets import DocxAssetError, PreparedImage, prepare_block_image, prepare_html_image
29
+ from .inline import (
30
+ BookmarkRegistry,
31
+ InlineRenderContext,
32
+ append_inline_content,
33
+ append_inline_spans,
34
+ append_internal_link,
35
+ append_joined_inline_contents,
36
+ sanitize_xml_text,
37
+ )
38
+ from .math import DocxFormulaError, latex_to_omml, split_formula_tag
39
+ from .styles import (
40
+ BODY_STYLE,
41
+ CAPTION_STYLE,
42
+ CODE_STYLE,
43
+ FOOTNOTE_STYLE,
44
+ FORMULA_FALLBACK_STYLE,
45
+ SPATIAL_TABLE_STYLE,
46
+ configure_document,
47
+ usable_width_emu,
48
+ usable_width_twips,
49
+ )
50
+ from .table import DocxTableError, NestedTableWriter, materialize_docx_tables
51
+ from ...contracts import AssetResolver
52
+ from ...docx import DocxRenderError
53
+ from ....schema import (
54
+ PAGE_AUXILIARY_BLOCK_TYPES,
55
+ RAW_ALGORITHM,
56
+ AlgorithmBodyBlock,
57
+ BlockBase,
58
+ BlockType,
59
+ BBox,
60
+ ChartAnnotationBlock,
61
+ ChartBlock,
62
+ ChartBodyBlock,
63
+ CodeAnnotationBlock,
64
+ CodeBlock,
65
+ CodeBodyBlock,
66
+ DocTitleBlock,
67
+ EquationBlock,
68
+ ImageAnnotationBlock,
69
+ ImageBlock,
70
+ ImageBodyBlock,
71
+ ImagePayloadBlock,
72
+ IndexBlock,
73
+ InlineSpan,
74
+ ListBlock,
75
+ MiddleJson,
76
+ PageFootnoteBlock,
77
+ ParagraphTitleBlock,
78
+ RefTextBlock,
79
+ TableAnnotationBlock,
80
+ TableBlock,
81
+ TableBodyBlock,
82
+ TextBlock,
83
+ TextSpan,
84
+ EquationInlineSpan,
85
+ HyperlinkSpan,
86
+ TitleBlockBase,
87
+ )
88
+ from ....foundation.image_payload import validate_remote_image_url
89
+
90
+ _SVG_BLIP_NAMESPACE = "http://schemas.microsoft.com/office/drawing/2016/SVG/main"
91
+ _SVG_BLIP_EXTENSION_URI = "{96DAC541-7B7A-43D3-8B79-37D633B846F1}"
92
+
93
+ _HTML_TABLE_RE = re.compile(r"<table\b", re.IGNORECASE)
94
+ _ANNOTATION_CAPTION_TYPES = {
95
+ BlockType.IMAGE_CAPTION,
96
+ BlockType.TABLE_CAPTION,
97
+ BlockType.CHART_CAPTION,
98
+ BlockType.CODE_CAPTION,
99
+ }
100
+ _ANNOTATION_FOOTNOTE_TYPES = {
101
+ BlockType.IMAGE_FOOTNOTE,
102
+ BlockType.TABLE_FOOTNOTE,
103
+ BlockType.CHART_FOOTNOTE,
104
+ BlockType.CODE_FOOTNOTE,
105
+ }
106
+
107
+
108
+ def _has_block_image_payload(block: ImagePayloadBlock) -> bool:
109
+ """判断统一图片载荷是否包含 sidecar、data URI 或远程 URL。"""
110
+ return block.image_path is not None or block.image_base64 is not None or block.image_url is not None
111
+
112
+
113
+ class _DocxRenderer:
114
+ """持有单次 DOCX 渲染所需的 document、素材解析器和书签状态。"""
115
+
116
+ def __init__(
117
+ self,
118
+ middle_json: MiddleJson,
119
+ *,
120
+ asset_resolver: AssetResolver | None,
121
+ ) -> None:
122
+ """初始化 renderer,并预注册标题与默认可见页面脚注 anchor。"""
123
+ self.middle_json = middle_json
124
+ self.asset_resolver = asset_resolver
125
+ self.document = Document()
126
+ configure_document(self.document)
127
+ self.bookmarks = BookmarkRegistry(_iter_document_anchors(middle_json))
128
+ self.usable_width_emu = usable_width_emu(self.document)
129
+ self.usable_width_twips = usable_width_twips(self.document)
130
+
131
+ def render(self) -> bytes:
132
+ """执行逐页 visitor,并把 python-docx document 序列化为 bytes。"""
133
+ planned_pages = build_render_plan(self.middle_json)
134
+ for planned_blocks in planned_pages:
135
+ for planned in planned_blocks:
136
+ if planned.removed:
137
+ continue
138
+ if planned.block.type in PAGE_AUXILIARY_BLOCK_TYPES:
139
+ continue
140
+ self._render_planned_block(planned)
141
+
142
+ output = BytesIO()
143
+ self.document.save(output)
144
+ return output.getvalue()
145
+
146
+ def _render_planned_block(self, planned: PlannedBlock) -> None:
147
+ """按严格 PageBlock 具体类型分派 Word 写入逻辑。"""
148
+ block = planned.block
149
+ context = self._context(planned.page_idx, block)
150
+ if isinstance(block, (TextBlock, RefTextBlock)):
151
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
152
+ append_joined_inline_contents(
153
+ paragraph,
154
+ planned.text_contents or [block.content],
155
+ context=context,
156
+ )
157
+ if isinstance(block, TextBlock):
158
+ self.bookmarks.attach(paragraph, block.anchor)
159
+ return
160
+ if isinstance(block, (DocTitleBlock, ParagraphTitleBlock)):
161
+ self._render_title(block, context)
162
+ return
163
+ if isinstance(block, PageFootnoteBlock):
164
+ paragraph = self.document.add_paragraph(style=FOOTNOTE_STYLE)
165
+ append_inline_content(paragraph, block.content, context=context)
166
+ self.bookmarks.attach(paragraph, block.anchor)
167
+ return
168
+ if isinstance(block, EquationBlock):
169
+ self._render_equation(block, context)
170
+ return
171
+ if isinstance(block, ListBlock):
172
+ self._render_list(block, context, depth=0)
173
+ return
174
+ if isinstance(block, IndexBlock):
175
+ self._render_index(block, context, depth=0)
176
+ return
177
+ if isinstance(block, ImageBlock):
178
+ self._render_image_block(block, context)
179
+ return
180
+ if isinstance(block, TableBlock):
181
+ self._render_table_block(block, context)
182
+ return
183
+ if isinstance(block, ChartBlock):
184
+ self._render_chart_block(block, context)
185
+ return
186
+ if isinstance(block, CodeBlock):
187
+ self._render_code_block(block, context)
188
+ return
189
+ raise TypeError(f"Unsupported PageBlock type: {type(block).__name__}")
190
+
191
+ def _context(self, page_idx: int, block: BlockBase) -> InlineRenderContext:
192
+ """为当前 block 构造行内渲染上下文。"""
193
+ return InlineRenderContext(
194
+ bookmarks=self.bookmarks,
195
+ page_idx=page_idx,
196
+ block_index=block.index,
197
+ block_type=block.type,
198
+ )
199
+
200
+ def _render_title(
201
+ self,
202
+ block: DocTitleBlock | ParagraphTitleBlock,
203
+ context: InlineRenderContext,
204
+ ) -> None:
205
+ """用 Heading 1–6 写严格标题,并在段落内容外添加 bookmark。"""
206
+ level = min(max(block.level, 1), 9)
207
+ paragraph = self.document.add_paragraph(style=f"Heading {level}")
208
+ append_inline_content(paragraph, block.content, context=context)
209
+ self.bookmarks.attach(paragraph, block.anchor)
210
+
211
+ def _render_equation(self, block: EquationBlock, context: InlineRenderContext) -> None:
212
+ """优先写可编辑 OMML,失败时按图片、可见 LaTeX 的顺序回退。"""
213
+ content = block.content.strip()
214
+ if content:
215
+ formula, tag = split_formula_tag(content)
216
+ try:
217
+ math_element = latex_to_omml(formula, display=True)
218
+ self._append_display_formula(math_element, tag, context=context)
219
+ return
220
+ except DocxFormulaError as exc:
221
+ logger.warning("DOCX display formula fallback: {} ({})", exc, context.location())
222
+
223
+ if _has_block_image_payload(block):
224
+ try:
225
+ self._append_block_image(block, context=context, alt_text="formula")
226
+ return
227
+ except DocxRenderError:
228
+ if not content:
229
+ raise
230
+ logger.warning("DOCX formula image fallback failed; preserving LaTeX ({})", context.location())
231
+
232
+ if content:
233
+ paragraph = self.document.add_paragraph(style=FORMULA_FALLBACK_STYLE)
234
+ paragraph.add_run(sanitize_xml_text(content, context=context))
235
+ return
236
+ raise self._render_error("Equation does not contain usable LaTeX or image", context)
237
+
238
+ def _append_display_formula(
239
+ self,
240
+ math_element: etree._Element,
241
+ tag: str | None,
242
+ *,
243
+ context: InlineRenderContext,
244
+ ) -> None:
245
+ """把块公式居中写入;带 tag 时使用中心与右对齐 tab。"""
246
+ if tag is None:
247
+ properties = OxmlElement("m:oMathParaPr")
248
+ justification = OxmlElement("m:jc")
249
+ justification.set(qn("m:val"), "center")
250
+ properties.append(justification)
251
+ math_element.insert(0, properties)
252
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
253
+ paragraph.paragraph_format.space_before = Pt(4)
254
+ paragraph.paragraph_format.space_after = Pt(5)
255
+ paragraph._p.append(math_element)
256
+ return
257
+
258
+ equation = math_element.find(qn("m:oMath"))
259
+ if equation is None:
260
+ raise ValueError("Display OMML does not contain m:oMath")
261
+ math_element.remove(equation)
262
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
263
+ paragraph.paragraph_format.space_before = Pt(4)
264
+ paragraph.paragraph_format.space_after = Pt(5)
265
+ paragraph.paragraph_format.tab_stops.add_tab_stop(
266
+ Twips(self.usable_width_twips // 2),
267
+ WD_TAB_ALIGNMENT.CENTER,
268
+ )
269
+ paragraph.paragraph_format.tab_stops.add_tab_stop(
270
+ Twips(self.usable_width_twips),
271
+ WD_TAB_ALIGNMENT.RIGHT,
272
+ )
273
+ paragraph.add_run().add_tab()
274
+ paragraph._p.append(equation)
275
+ paragraph.add_run().add_tab()
276
+ rendered_tag = tag if tag.startswith("(") and tag.endswith(")") else f"({tag})"
277
+ paragraph.add_run(sanitize_xml_text(rendered_tag, context=context))
278
+
279
+ def _render_list(
280
+ self,
281
+ block: ListBlock,
282
+ context: InlineRenderContext,
283
+ *,
284
+ depth: int,
285
+ ) -> None:
286
+ """保留叶子原 marker,并仅用缩进表达递归列表层级。"""
287
+ for child in block.content:
288
+ if isinstance(child, ListBlock):
289
+ self._render_list(child, context, depth=depth + 1)
290
+ continue
291
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
292
+ item = parse_list_item_marker(child.content)
293
+ if item.marker is None:
294
+ paragraph.paragraph_format.left_indent = Mm(depth * 6)
295
+ else:
296
+ paragraph.paragraph_format.left_indent = Mm((depth + 1) * 6)
297
+ paragraph.paragraph_format.first_line_indent = Mm(-6)
298
+ append_inline_content(paragraph, child.content, context=context)
299
+
300
+ def _render_index(
301
+ self,
302
+ block: IndexBlock,
303
+ context: InlineRenderContext,
304
+ *,
305
+ depth: int,
306
+ ) -> None:
307
+ """递归输出目录项,并把标题叶子链接到预注册 bookmark。"""
308
+ for child in block.content:
309
+ if isinstance(child, IndexBlock):
310
+ self._render_index(child, context, depth=depth + 1)
311
+ continue
312
+ content = strip_index_page_tail(child.content)
313
+ if not content:
314
+ continue
315
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
316
+ paragraph.paragraph_format.left_indent = Mm((depth + 1) * 6)
317
+ paragraph.paragraph_format.first_line_indent = Mm(-4)
318
+ paragraph.add_run("• ")
319
+ append_internal_link(paragraph, content, anchor=child.anchor, context=context)
320
+
321
+ def _render_image_block(self, block: ImageBlock, context: InlineRenderContext) -> None:
322
+ """按原始子块顺序写图片主体、caption 与 footnote。"""
323
+ for child in block.content:
324
+ if isinstance(child, ImageBodyBlock):
325
+ alt_text = _plain_html_text(child.content) or block.sub_type or "image"
326
+ self._append_block_image(child, context=context, alt_text=alt_text)
327
+ elif isinstance(child, ImageAnnotationBlock):
328
+ self._render_annotation(child, context)
329
+ else:
330
+ raise TypeError(f"Unsupported image child: {type(child).__name__}")
331
+
332
+ def _render_table_block(self, block: TableBlock, context: InlineRenderContext) -> None:
333
+ """按原始子块顺序写原生 HTML table 或预格式空间投影文本。"""
334
+ for child in block.content:
335
+ if isinstance(child, TableBodyBlock):
336
+ if _HTML_TABLE_RE.search(child.content):
337
+ try:
338
+ self._append_html_tables(child.content, context=context, depth=0)
339
+ except DocxRenderError as exc:
340
+ logger.warning("DOCX HTML table fallback: {}", exc)
341
+ if not _has_block_image_payload(child):
342
+ raise self._render_error(
343
+ "HTML table cannot be materialized and has no image fallback",
344
+ context,
345
+ ) from exc
346
+ self._append_block_image(child, context=context, alt_text="table")
347
+ else:
348
+ if child.content and not child.content.isspace():
349
+ paragraph = self.document.add_paragraph(style=SPATIAL_TABLE_STYLE)
350
+ paragraph.add_run(sanitize_xml_text(child.content, context=context))
351
+ elif _has_block_image_payload(child):
352
+ self._append_block_image(child, context=context, alt_text="table")
353
+ else:
354
+ raise self._render_error(
355
+ "Spatial table does not contain text content or image",
356
+ context,
357
+ )
358
+ elif isinstance(child, TableAnnotationBlock):
359
+ self._render_annotation(child, context)
360
+ else:
361
+ raise TypeError(f"Unsupported table child: {type(child).__name__}")
362
+
363
+ def _render_chart_block(self, block: ChartBlock, context: InlineRenderContext) -> None:
364
+ """先写图表图片,再把 HTML 结构化数据追加为原生表格。"""
365
+ for child in block.content:
366
+ if isinstance(child, ChartBodyBlock):
367
+ has_image = _has_block_image_payload(child)
368
+ if has_image:
369
+ self._append_block_image(
370
+ child,
371
+ context=context,
372
+ alt_text=_plain_html_text(child.content) or block.sub_type or "chart",
373
+ )
374
+ if _HTML_TABLE_RE.search(child.content):
375
+ try:
376
+ self._append_html_tables(child.content, context=context, depth=0)
377
+ except DocxRenderError as exc:
378
+ if not has_image:
379
+ raise
380
+ logger.warning("DOCX chart data table omitted after image fallback: {}", exc)
381
+ elif child.content.strip() and not has_image:
382
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
383
+ append_inline_content(paragraph, [TextSpan(type="text", content=child.content)], context=context)
384
+ elif not has_image and not child.content.strip():
385
+ raise self._render_error("Chart does not contain image or structured content", context)
386
+ elif isinstance(child, ChartAnnotationBlock):
387
+ self._render_annotation(child, context)
388
+ else:
389
+ raise TypeError(f"Unsupported chart child: {type(child).__name__}")
390
+
391
+ def _render_code_block(self, block: CodeBlock, context: InlineRenderContext) -> None:
392
+ """按原始子块顺序写代码或算法主体及说明。"""
393
+ for child in block.content:
394
+ if isinstance(child, (CodeBodyBlock, AlgorithmBodyBlock)):
395
+ paragraph = self.document.add_paragraph(style=CODE_STYLE)
396
+ if block.sub_type == BlockType.CODE:
397
+ if not isinstance(child, CodeBodyBlock):
398
+ raise TypeError("code subtype requires CodeBodyBlock")
399
+ paragraph.add_run(sanitize_xml_text(child.content, context=context))
400
+ elif block.sub_type == RAW_ALGORITHM:
401
+ if not isinstance(child, AlgorithmBodyBlock):
402
+ raise TypeError("algorithm subtype requires AlgorithmBodyBlock")
403
+ append_inline_content(paragraph, child.content, context=context)
404
+ else:
405
+ raise ValueError(f"Unsupported code subtype: {block.sub_type}")
406
+ elif isinstance(child, CodeAnnotationBlock):
407
+ self._render_annotation(child, context)
408
+ else:
409
+ raise TypeError(f"Unsupported code child: {type(child).__name__}")
410
+
411
+ def _render_annotation(
412
+ self,
413
+ block: ImageAnnotationBlock | TableAnnotationBlock | ChartAnnotationBlock | CodeAnnotationBlock,
414
+ context: InlineRenderContext,
415
+ ) -> None:
416
+ """根据说明类型选择 Caption 或 Footnote 样式并写入行内内容。"""
417
+ if block.type in _ANNOTATION_CAPTION_TYPES:
418
+ style = CAPTION_STYLE
419
+ elif block.type in _ANNOTATION_FOOTNOTE_TYPES:
420
+ style = FOOTNOTE_STYLE
421
+ else:
422
+ raise TypeError(f"Unsupported annotation type: {block.type}")
423
+ paragraph = self.document.add_paragraph(style=style)
424
+ append_inline_content(paragraph, block.content, context=context)
425
+
426
+ def _append_block_image(
427
+ self,
428
+ block: ImagePayloadBlock,
429
+ *,
430
+ context: InlineRenderContext,
431
+ alt_text: str,
432
+ ) -> None:
433
+ """安全加载 block 图片,并按 bbox/自然尺寸限制到可用页面范围。"""
434
+ if block.image_base64 is None and block.image_path is None and block.image_url:
435
+ paragraph = self.document.add_paragraph(style=BODY_STYLE)
436
+ append_inline_spans(
437
+ paragraph,
438
+ [
439
+ HyperlinkSpan(
440
+ type="hyperlink",
441
+ url=block.image_url,
442
+ content=[TextSpan(type="text", content=alt_text or "remote image")],
443
+ )
444
+ ],
445
+ context=context,
446
+ )
447
+ return
448
+ try:
449
+ prepared = prepare_block_image(block, self.asset_resolver)
450
+ except DocxAssetError as exc:
451
+ raise self._render_error(str(exc), context) from exc
452
+ self._append_prepared_image(
453
+ prepared,
454
+ bbox=block.bbox,
455
+ alt_text=alt_text,
456
+ max_width_emu=self.usable_width_emu,
457
+ context=context,
458
+ )
459
+
460
+ def _append_prepared_image(
461
+ self,
462
+ prepared: PreparedImage,
463
+ *,
464
+ bbox: BBox | None,
465
+ alt_text: str,
466
+ max_width_emu: int,
467
+ context: InlineRenderContext,
468
+ paragraph: Paragraph | None = None,
469
+ ) -> None:
470
+ """把准备好的图片按比例插入居中段落并写入 alt description。"""
471
+ target_paragraph = paragraph or self.document.add_paragraph(style=BODY_STYLE)
472
+ target_paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
473
+ natural_width = round(prepared.width_px / 96 * 914400)
474
+ desired_width = natural_width
475
+ if bbox is not None:
476
+ desired_width = round(max_width_emu * max(0.0, min(1.0, bbox[2] - bbox[0])))
477
+ minimum_width = min(int(Mm(5)), max_width_emu)
478
+ desired_width = max(minimum_width, min(desired_width, max_width_emu))
479
+
480
+ usable_height = int(Mm(257))
481
+ rendered_height = round(desired_width * prepared.height_px / max(prepared.width_px, 1))
482
+ if rendered_height > usable_height:
483
+ desired_width = round(usable_height * prepared.width_px / max(prepared.height_px, 1))
484
+
485
+ shape = target_paragraph.add_run().add_picture(
486
+ BytesIO(prepared.data),
487
+ width=Emu(max(1, desired_width)),
488
+ )
489
+ if prepared.svg_data is not None:
490
+ self._attach_native_svg(shape._inline, target_paragraph.part, prepared.svg_data)
491
+ description = sanitize_xml_text(
492
+ re.sub(r"\s+", " ", alt_text).strip()[:2048],
493
+ context=context,
494
+ )
495
+ if description:
496
+ shape._inline.docPr.set("descr", description)
497
+
498
+ @staticmethod
499
+ def _attach_native_svg(inline: etree._Element, target_part: Part, svg_data: bytes) -> None:
500
+ """为 fallback PNG 的 a:blip 添加 Office 2016 原生 SVG relationship。"""
501
+ package = target_part.package
502
+ svg_part = Part(
503
+ package.next_partname("/word/media/image%d.svg"),
504
+ "image/svg+xml",
505
+ svg_data,
506
+ package,
507
+ )
508
+ relationship_id = target_part.relate_to(svg_part, RT.IMAGE)
509
+ blip = inline.find(f".//{qn('a:blip')}")
510
+ if blip is None:
511
+ raise ValueError("DOCX picture does not contain an a:blip element")
512
+ extension_list = blip.find(qn("a:extLst"))
513
+ if extension_list is None:
514
+ extension_list = OxmlElement("a:extLst")
515
+ blip.append(extension_list)
516
+ extension = OxmlElement("a:ext")
517
+ extension.set("uri", _SVG_BLIP_EXTENSION_URI)
518
+ svg_blip = etree.SubElement(
519
+ extension,
520
+ f"{{{_SVG_BLIP_NAMESPACE}}}svgBlip",
521
+ nsmap={"asvg": _SVG_BLIP_NAMESPACE},
522
+ )
523
+ svg_blip.set(qn("r:embed"), relationship_id)
524
+ extension_list.append(extension)
525
+
526
+ def _append_html_tables(
527
+ self,
528
+ markup: str,
529
+ *,
530
+ context: InlineRenderContext,
531
+ depth: int,
532
+ container: DocumentType | _Cell | None = None,
533
+ ) -> None:
534
+ """物化 HTML 表格,并通过共享行内/素材逻辑填充 origin cell。"""
535
+ if depth >= 4:
536
+ raise self._render_error("Nested HTML table depth exceeds 4", context)
537
+ target_container = container or self.document
538
+ target_xml = target_container._element.body if isinstance(target_container, DocumentType) else target_container._tc
539
+ existing_child_count = len(target_xml)
540
+ insertion_index = existing_child_count - 1 if isinstance(target_container, DocumentType) else existing_child_count
541
+ existing_relationship_ids = set(self.document.part.rels)
542
+
543
+ def fill_cell(cell: _Cell, source: Tag, write_nested: NestedTableWriter) -> None:
544
+ """填充一个 origin cell,并在遇到 nested table 时调用递归 writer。"""
545
+ self._fill_html_cell(
546
+ cell,
547
+ source,
548
+ write_nested=write_nested,
549
+ context=context,
550
+ )
551
+
552
+ try:
553
+ materialize_docx_tables(
554
+ target_container,
555
+ markup,
556
+ width_twips=self.usable_width_twips,
557
+ fill_cell=fill_cell,
558
+ )
559
+ except DocxRenderError:
560
+ self._rollback_table_materialization(
561
+ target_xml,
562
+ existing_child_count=existing_child_count,
563
+ insertion_index=insertion_index,
564
+ existing_relationship_ids=existing_relationship_ids,
565
+ )
566
+ raise
567
+ except DocxTableError as exc:
568
+ self._rollback_table_materialization(
569
+ target_xml,
570
+ existing_child_count=existing_child_count,
571
+ insertion_index=insertion_index,
572
+ existing_relationship_ids=existing_relationship_ids,
573
+ )
574
+ raise self._render_error(str(exc), context) from exc
575
+
576
+ def _rollback_table_materialization(
577
+ self,
578
+ target_xml: etree._Element,
579
+ *,
580
+ existing_child_count: int,
581
+ insertion_index: int,
582
+ existing_relationship_ids: set[str],
583
+ ) -> None:
584
+ """回滚本轮表格新增的 XML 根节点和 relationship,避免残留半成品。"""
585
+ while len(target_xml) > existing_child_count:
586
+ target_xml.remove(target_xml[insertion_index])
587
+ for relationship_id in tuple(self.document.part.rels):
588
+ if relationship_id not in existing_relationship_ids:
589
+ self.document.part.drop_rel(relationship_id)
590
+
591
+ def _fill_html_cell(
592
+ self,
593
+ cell: _Cell,
594
+ source: Tag,
595
+ *,
596
+ write_nested: NestedTableWriter,
597
+ context: InlineRenderContext,
598
+ ) -> None:
599
+ """按源顺序填充单元格文本、图片和任意包装层中的嵌套表格。"""
600
+ cell.text = ""
601
+ cell.vertical_alignment = WD_CELL_VERTICAL_ALIGNMENT.CENTER
602
+ _set_cell_margins(cell, top=100, start=120, bottom=100, end=120)
603
+ max_image_width_emu = _cell_content_width_emu(cell, horizontal_margin_twips=240)
604
+ paragraph = cell.paragraphs[0]
605
+ paragraph.paragraph_format.space_after = Pt(0)
606
+ paragraph.paragraph_format.line_spacing = 1.0
607
+
608
+ for child in source.children:
609
+ if not isinstance(child, (NavigableString, Tag)):
610
+ continue
611
+ if isinstance(child, Tag) and child.name.lower() in {"p", "div"} and _paragraph_has_content(paragraph):
612
+ paragraph = cell.add_paragraph()
613
+ paragraph.paragraph_format.space_after = Pt(0)
614
+ paragraph.paragraph_format.line_spacing = 1.0
615
+ styles = ("bold",) if source.name == "th" else ()
616
+ paragraph, _ = self._append_html_node(
617
+ paragraph,
618
+ child,
619
+ cell=cell,
620
+ write_nested=write_nested,
621
+ context=context,
622
+ inherited_styles=styles,
623
+ max_image_width_emu=max_image_width_emu,
624
+ )
625
+
626
+ def _append_html_node(
627
+ self,
628
+ paragraph: Paragraph,
629
+ node: NavigableString | Tag,
630
+ *,
631
+ cell: _Cell,
632
+ write_nested: NestedTableWriter,
633
+ context: InlineRenderContext,
634
+ inherited_styles: tuple[str, ...],
635
+ max_image_width_emu: int,
636
+ ) -> tuple[Paragraph, bool]:
637
+ """按 HTML 源顺序写入单元格行内文本、样式、公式和图片。"""
638
+ if isinstance(node, NavigableString):
639
+ text = str(node)
640
+ if not text:
641
+ return paragraph, False
642
+ spans: list[InlineSpan] = [TextSpan(type="text", content=text, styles=list(inherited_styles))]
643
+ append_inline_spans(paragraph, spans, context=context)
644
+ return paragraph, bool(text.strip())
645
+
646
+ name = node.name.lower()
647
+ if name == "table":
648
+ write_nested(node)
649
+ trailing_paragraph = cell.paragraphs[-1]
650
+ trailing_paragraph.paragraph_format.space_after = Pt(0)
651
+ trailing_paragraph.paragraph_format.line_spacing = 1.0
652
+ return trailing_paragraph, True
653
+ if name == "img":
654
+ self._append_html_image(
655
+ paragraph,
656
+ str(node.get("src", "")),
657
+ context=context,
658
+ alt_text=str(node.get("alt", "")),
659
+ max_width_emu=max_image_width_emu,
660
+ )
661
+ return paragraph, True
662
+ if name == "br":
663
+ paragraph.add_run().add_break()
664
+ return paragraph, True
665
+ if name == "eq":
666
+ latex = node.get_text().strip()
667
+ spans = [EquationInlineSpan(type="equation_inline", content=latex)] if latex else []
668
+ append_inline_spans(paragraph, spans, context=context)
669
+ return paragraph, bool(spans)
670
+ if name == "a":
671
+ children = _html_inline_spans(node, inherited_styles=inherited_styles)
672
+ append_inline_spans(
673
+ paragraph,
674
+ [HyperlinkSpan(type="hyperlink", content=children, url=str(node.get("href", "")).strip())] if children else [],
675
+ context=context,
676
+ )
677
+ return paragraph, bool(children)
678
+ if name in {"ul", "ol"}:
679
+ start_value = node.get("start", "1")
680
+ try:
681
+ ordered_number = int(str(start_value))
682
+ except ValueError:
683
+ ordered_number = 1
684
+ rendered = False
685
+ for item in node.find_all("li", recursive=False):
686
+ if _paragraph_has_content(paragraph) or rendered:
687
+ paragraph.add_run().add_break()
688
+ marker = f"{ordered_number}. " if name == "ol" else "- "
689
+ paragraph.add_run(marker)
690
+ if name == "ol":
691
+ ordered_number += 1
692
+ for child in item.children:
693
+ if isinstance(child, (NavigableString, Tag)):
694
+ paragraph, child_rendered = self._append_html_node(
695
+ paragraph,
696
+ child,
697
+ cell=cell,
698
+ write_nested=write_nested,
699
+ context=context,
700
+ inherited_styles=inherited_styles,
701
+ max_image_width_emu=max_image_width_emu,
702
+ )
703
+ rendered = child_rendered or rendered
704
+ return paragraph, rendered
705
+ if name == "li":
706
+ rendered = False
707
+ for child in node.children:
708
+ if isinstance(child, (NavigableString, Tag)):
709
+ paragraph, child_rendered = self._append_html_node(
710
+ paragraph,
711
+ child,
712
+ cell=cell,
713
+ write_nested=write_nested,
714
+ context=context,
715
+ inherited_styles=inherited_styles,
716
+ max_image_width_emu=max_image_width_emu,
717
+ )
718
+ rendered = child_rendered or rendered
719
+ return paragraph, rendered
720
+ style = {
721
+ "strong": "bold",
722
+ "b": "bold",
723
+ "em": "italic",
724
+ "i": "italic",
725
+ "u": "underline",
726
+ "s": "strikethrough",
727
+ "sup": "superscript",
728
+ "sub": "subscript",
729
+ }.get(name)
730
+ styles = tuple(dict.fromkeys((*inherited_styles, style))) if style else inherited_styles
731
+ rendered = False
732
+ for child in node.children:
733
+ if isinstance(child, (NavigableString, Tag)):
734
+ paragraph, child_rendered = self._append_html_node(
735
+ paragraph,
736
+ child,
737
+ cell=cell,
738
+ write_nested=write_nested,
739
+ context=context,
740
+ inherited_styles=styles,
741
+ max_image_width_emu=max_image_width_emu,
742
+ )
743
+ rendered = child_rendered or rendered
744
+ return paragraph, rendered
745
+
746
+ def _append_html_image(
747
+ self,
748
+ paragraph: Paragraph,
749
+ source: str,
750
+ *,
751
+ context: InlineRenderContext,
752
+ alt_text: str,
753
+ max_width_emu: int,
754
+ ) -> None:
755
+ """安全加载表格单元格 img,并限制到紧凑的单元格宽度。"""
756
+ try:
757
+ remote_source = validate_remote_image_url(source)
758
+ except ValueError:
759
+ remote_source = None
760
+ if remote_source is not None:
761
+ append_inline_spans(
762
+ paragraph,
763
+ [
764
+ HyperlinkSpan(
765
+ type="hyperlink",
766
+ url=remote_source,
767
+ content=[TextSpan(type="text", content=alt_text.strip() or "remote image")],
768
+ )
769
+ ],
770
+ context=context,
771
+ )
772
+ return
773
+ try:
774
+ prepared = prepare_html_image(source, self.asset_resolver)
775
+ except DocxAssetError as exc:
776
+ raise self._render_error(str(exc), context) from exc
777
+ self._append_prepared_image(
778
+ prepared,
779
+ bbox=None,
780
+ alt_text=alt_text or "table image",
781
+ max_width_emu=min(self.usable_width_emu, max_width_emu),
782
+ context=context,
783
+ paragraph=paragraph,
784
+ )
785
+
786
+ def _render_error(self, message: str, context: InlineRenderContext) -> DocxRenderError:
787
+ """用当前 context 构造带 page/block 定位的公共异常。"""
788
+ return DocxRenderError(
789
+ message,
790
+ page_idx=context.page_idx,
791
+ block_index=context.block_index,
792
+ block_type=context.block_type,
793
+ )
794
+
795
+
796
+ def render_docx(
797
+ middle_json: MiddleJson,
798
+ *,
799
+ asset_resolver: AssetResolver | None = None,
800
+ ) -> bytes:
801
+ """把严格 MiddleJson 无副作用地渲染为完整 DOCX bytes。"""
802
+ if not isinstance(middle_json, MiddleJson):
803
+ raise TypeError("render_docx expects a MiddleJson instance")
804
+ return _DocxRenderer(
805
+ middle_json,
806
+ asset_resolver=asset_resolver,
807
+ ).render()
808
+
809
+
810
+ def _iter_document_anchors(middle_json: MiddleJson) -> Iterable[str]:
811
+ """遍历正文、标题和默认可见页面脚注实际会写入的 bookmark anchor。"""
812
+ for page in middle_json.pages:
813
+ for block in page.blocks:
814
+ if isinstance(block, (TextBlock, TitleBlockBase)) and block.anchor:
815
+ yield block.anchor
816
+ elif isinstance(block, PageFootnoteBlock) and block.anchor:
817
+ yield block.anchor
818
+
819
+
820
+ def _plain_html_text(content: str) -> str:
821
+ """把图片或图表 body 内容收敛为适合 alt description 的纯文本。"""
822
+ if not content.strip():
823
+ return ""
824
+ return BeautifulSoup(content, "html.parser").get_text(" ", strip=True)
825
+
826
+
827
+ def _html_inline_spans(node: Tag, *, inherited_styles: tuple[str, ...] = ()) -> list[InlineSpan]:
828
+ """把表格单元格中的安全 HTML 行内标签转换为结构化 Span。"""
829
+ spans: list[InlineSpan] = []
830
+ for child in node.children:
831
+ if isinstance(child, NavigableString):
832
+ if str(child):
833
+ spans.append(TextSpan(type="text", content=str(child), styles=list(inherited_styles)))
834
+ continue
835
+ if not isinstance(child, Tag):
836
+ continue
837
+ name = child.name.lower()
838
+ if name == "eq":
839
+ latex = child.get_text().strip()
840
+ if latex:
841
+ spans.append(EquationInlineSpan(type="equation_inline", content=latex))
842
+ continue
843
+ if name == "br":
844
+ spans.append(TextSpan(type="text", content="\n", styles=list(inherited_styles)))
845
+ continue
846
+ style = {
847
+ "strong": "bold",
848
+ "b": "bold",
849
+ "em": "italic",
850
+ "i": "italic",
851
+ "u": "underline",
852
+ "s": "strikethrough",
853
+ "sup": "superscript",
854
+ "sub": "subscript",
855
+ }.get(name)
856
+ styles = tuple(dict.fromkeys((*inherited_styles, style))) if style else inherited_styles
857
+ children = _html_inline_spans(child, inherited_styles=styles)
858
+ if name == "a":
859
+ non_link_children = [item for item in children if not isinstance(item, HyperlinkSpan)]
860
+ url = str(child.get("href", "")).strip()
861
+ if non_link_children and url:
862
+ spans.append(HyperlinkSpan(type="hyperlink", content=non_link_children, url=url))
863
+ else:
864
+ spans.extend(non_link_children)
865
+ elif name not in {"img", "table"}:
866
+ spans.extend(children)
867
+ return normalize_inline_spans(spans)
868
+
869
+
870
+ def _set_cell_margins(
871
+ cell: _Cell,
872
+ *,
873
+ top: int,
874
+ start: int,
875
+ bottom: int,
876
+ end: int,
877
+ ) -> None:
878
+ """用 DXA 为 Word 单元格写入四边内边距。"""
879
+ cell_properties = cell._tc.get_or_add_tcPr()
880
+ margins = cell_properties.find(qn("w:tcMar"))
881
+ if margins is None:
882
+ margins = OxmlElement("w:tcMar")
883
+ cell_properties.append(margins)
884
+ for side, value in (("top", top), ("start", start), ("bottom", bottom), ("end", end)):
885
+ element = margins.find(qn(f"w:{side}"))
886
+ if element is None:
887
+ element = OxmlElement(f"w:{side}")
888
+ margins.append(element)
889
+ element.set(qn("w:w"), str(value))
890
+ element.set(qn("w:type"), "dxa")
891
+
892
+
893
+ def _cell_content_width_emu(cell: _Cell, *, horizontal_margin_twips: int) -> int:
894
+ """从当前物理或合并单元格 tcW 计算扣除左右内边距后的可用宽度。"""
895
+ cell_width = cell._tc.get_or_add_tcPr().get_or_add_tcW().w
896
+ width_twips = int(cell_width) if cell_width is not None else 1
897
+ return int(Twips(max(1, width_twips - horizontal_margin_twips)))
898
+
899
+
900
+ def _paragraph_has_content(paragraph: Paragraph) -> bool:
901
+ """判断段落是否包含 pPr 之外的 run、公式、链接或 drawing。"""
902
+ return any(child.tag != qn("w:pPr") for child in paragraph._p)
903
+
904
+
905
+ __all__ = ["render_docx"]