docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,48 @@
1
+ from .api import render
2
+ from .contracts import (
3
+ AssetResolver,
4
+ DocxRenderOptions,
5
+ EpubRenderOptions,
6
+ HtmlRenderOptions,
7
+ ImageRenderer,
8
+ LatexRenderOptions,
9
+ MarkdownRenderOptions,
10
+ PdfRenderOptions,
11
+ RenderFormat,
12
+ RenderMode,
13
+ RenderOptions,
14
+ RenderOutput,
15
+ StructuredContentRenderOptions,
16
+ )
17
+ from .docx import DocxRenderError, render_docx
18
+ from .epub import render_epub
19
+ from .html import render_html
20
+ from .latex import render_latex
21
+ from .markdown import render_markdown
22
+ from .pdf import render_pdf
23
+ from .structured_content import render_structured_content
24
+
25
+ __all__ = [
26
+ "AssetResolver",
27
+ "DocxRenderError",
28
+ "DocxRenderOptions",
29
+ "EpubRenderOptions",
30
+ "HtmlRenderOptions",
31
+ "ImageRenderer",
32
+ "LatexRenderOptions",
33
+ "MarkdownRenderOptions",
34
+ "PdfRenderOptions",
35
+ "RenderFormat",
36
+ "RenderMode",
37
+ "RenderOptions",
38
+ "RenderOutput",
39
+ "StructuredContentRenderOptions",
40
+ "render",
41
+ "render_docx",
42
+ "render_epub",
43
+ "render_html",
44
+ "render_latex",
45
+ "render_markdown",
46
+ "render_pdf",
47
+ "render_structured_content",
48
+ ]
@@ -0,0 +1,3 @@
1
+ """DocVortex renderer 的非公共实现模块。"""
2
+
3
+ __all__: list[str] = []
@@ -0,0 +1,3 @@
1
+ """各输出格式共用的渲染语义与规划实现。"""
2
+
3
+ __all__: list[str] = []
@@ -0,0 +1,43 @@
1
+ """在一次内部渲染调用中传递独占副本,不缓存用户的可变文档。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from contextlib import contextmanager
6
+ from contextvars import ContextVar
7
+ from dataclasses import dataclass
8
+ from typing import Iterator
9
+
10
+ from ....schema import MiddleJson
11
+
12
+
13
+ @dataclass(slots=True)
14
+ class _OwnedRenderDocument:
15
+ """只允许一个规划器消费当前调用已经隔离的文档。"""
16
+
17
+ document: MiddleJson
18
+ consumed: bool = False
19
+
20
+
21
+ _owned_document: ContextVar[_OwnedRenderDocument | None] = ContextVar("docvortex_owned_render_document", default=None)
22
+
23
+
24
+ @contextmanager
25
+ def owned_render_document(document: MiddleJson) -> Iterator[None]:
26
+ """限定独占文档的调用范围,异常和重入后均恢复外层上下文。"""
27
+ token = _owned_document.set(_OwnedRenderDocument(document))
28
+ try:
29
+ yield
30
+ finally:
31
+ _owned_document.reset(token)
32
+
33
+
34
+ def claim_owned_document(document: MiddleJson) -> bool:
35
+ """仅复用当前调用的精确对象一次,回调重入必须重新隔离。"""
36
+ context = _owned_document.get()
37
+ if context is None or context.document is not document or context.consumed:
38
+ return False
39
+ context.consumed = True
40
+ return True
41
+
42
+
43
+ __all__ = ["owned_render_document", "claim_owned_document"]
@@ -0,0 +1,178 @@
1
+ """多格式 renderer 共用的有界 HTML table 占位网格解析。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import re
7
+ from typing import TypeAlias
8
+
9
+ from bs4 import BeautifulSoup, Tag
10
+
11
+ MAX_NESTED_TABLE_DEPTH = 4
12
+ MAX_TABLE_ROWS = 500
13
+ MAX_TABLE_COLUMNS = 100
14
+ MAX_TABLE_SLOTS = 10_000
15
+ _POSITIVE_INTEGER_RE = re.compile(r"[0-9]+")
16
+
17
+ HtmlTableSource: TypeAlias = str | BeautifulSoup | Tag
18
+
19
+
20
+ class HtmlTableError(ValueError):
21
+ """表示 HTML table 无法安全解析为严格矩形网格。"""
22
+
23
+
24
+ @dataclass(frozen=True, slots=True)
25
+ class HtmlTableCell:
26
+ """保存原始 HTML 单元格在逻辑占位网格中的位置。"""
27
+
28
+ tag: Tag
29
+ row: int
30
+ column: int
31
+ rowspan: int
32
+ colspan: int
33
+ is_header: bool
34
+
35
+ @property
36
+ def end_row(self) -> int:
37
+ """返回单元格占用的末行下标。"""
38
+ return self.row + self.rowspan - 1
39
+
40
+ @property
41
+ def end_column(self) -> int:
42
+ """返回单元格占用的末列下标。"""
43
+ return self.column + self.colspan - 1
44
+
45
+
46
+ @dataclass(frozen=True, slots=True)
47
+ class HtmlTableGrid:
48
+ """保存经过重叠、边界、规模与矩形校验的 HTML 表格网格。"""
49
+
50
+ tag: Tag
51
+ row_count: int
52
+ column_count: int
53
+ cells: tuple[HtmlTableCell, ...]
54
+ header_rows: tuple[int, ...]
55
+
56
+
57
+ def parse_html_tables(source: HtmlTableSource) -> tuple[HtmlTableGrid, ...]:
58
+ """解析 source 中相对当前上下文的一个或多个顶层 table。"""
59
+ root = BeautifulSoup(source, "html.parser") if isinstance(source, str) else source
60
+ if not isinstance(root, (BeautifulSoup, Tag)):
61
+ raise HtmlTableError("HTML table source must be a string or BeautifulSoup Tag")
62
+ if isinstance(root, Tag) and root.name == "table":
63
+ table_tags = (root,)
64
+ else:
65
+ parent_table = root.find_parent("table") if isinstance(root, Tag) else None
66
+ table_tags = tuple(table for table in root.find_all("table") if table.find_parent("table") is parent_table)
67
+ if not table_tags:
68
+ raise HtmlTableError("HTML does not contain a top-level table")
69
+ return tuple(_parse_html_table(table) for table in table_tags)
70
+
71
+
72
+ def _parse_html_table(table: Tag) -> HtmlTableGrid:
73
+ """把单个 table 标签解析为严格矩形占位网格。"""
74
+ if table.name != "table":
75
+ raise HtmlTableError("Expected a <table> tag")
76
+ rows = tuple(row for row in table.find_all("tr") if row.find_parent("table") is table)
77
+ if not rows or len(rows) > MAX_TABLE_ROWS:
78
+ raise HtmlTableError(f"Table row count must be between 1 and {MAX_TABLE_ROWS}")
79
+
80
+ occupied: dict[tuple[int, int], HtmlTableCell] = {}
81
+ cells: list[HtmlTableCell] = []
82
+ for row_index, row in enumerate(rows):
83
+ column_index = 0
84
+ for source_cell in row.find_all(("td", "th"), recursive=False):
85
+ while (row_index, column_index) in occupied:
86
+ column_index += 1
87
+ rowspan = _parse_span(source_cell, "rowspan")
88
+ colspan = _parse_span(source_cell, "colspan")
89
+ if row_index + rowspan > len(rows):
90
+ raise HtmlTableError(f"rowspan exceeds table bounds at row={row_index}, column={column_index}")
91
+ if column_index + colspan > MAX_TABLE_COLUMNS:
92
+ raise HtmlTableError(f"Table column count exceeds {MAX_TABLE_COLUMNS}")
93
+ coordinates = tuple(
94
+ (target_row, target_column)
95
+ for target_row in range(row_index, row_index + rowspan)
96
+ for target_column in range(column_index, column_index + colspan)
97
+ )
98
+ overlap = next((coordinate for coordinate in coordinates if coordinate in occupied), None)
99
+ if overlap is not None:
100
+ raise HtmlTableError(f"Cell span overlaps row={overlap[0]}, column={overlap[1]}")
101
+ placement = HtmlTableCell(
102
+ tag=source_cell,
103
+ row=row_index,
104
+ column=column_index,
105
+ rowspan=rowspan,
106
+ colspan=colspan,
107
+ is_header=source_cell.name == "th",
108
+ )
109
+ cells.append(placement)
110
+ occupied.update(dict.fromkeys(coordinates, placement))
111
+ if len(occupied) > MAX_TABLE_SLOTS:
112
+ raise HtmlTableError(f"Table occupancy exceeds {MAX_TABLE_SLOTS} slots")
113
+ column_index += colspan
114
+
115
+ if not occupied:
116
+ raise HtmlTableError("Table must contain at least one cell")
117
+ column_count = max(column for _, column in occupied) + 1
118
+ missing = next(
119
+ (
120
+ (row_index, column_index)
121
+ for row_index in range(len(rows))
122
+ for column_index in range(column_count)
123
+ if (row_index, column_index) not in occupied
124
+ ),
125
+ None,
126
+ )
127
+ if missing is not None:
128
+ raise HtmlTableError(f"Table occupancy is not rectangular at row={missing[0]}, column={missing[1]}")
129
+ header_rows = tuple(
130
+ row_index
131
+ for row_index, row in enumerate(rows)
132
+ if _row_belongs_to_thead(row, table)
133
+ or all(occupied[(row_index, column_index)].is_header for column_index in range(column_count))
134
+ )
135
+ return HtmlTableGrid(
136
+ tag=table,
137
+ row_count=len(rows),
138
+ column_count=column_count,
139
+ cells=tuple(cells),
140
+ header_rows=header_rows,
141
+ )
142
+
143
+
144
+ def _parse_span(cell: Tag, attribute: str) -> int:
145
+ """读取严格正整数 rowspan/colspan,缺失时返回一。"""
146
+ raw_value = cell.get(attribute, "1")
147
+ if isinstance(raw_value, list):
148
+ raise HtmlTableError(f"Invalid {attribute}: {raw_value!r}")
149
+ value = str(raw_value).strip()
150
+ if _POSITIVE_INTEGER_RE.fullmatch(value) is None:
151
+ raise HtmlTableError(f"Invalid {attribute}: {raw_value!r}")
152
+ span = int(value)
153
+ if span < 1 or span > MAX_TABLE_SLOTS:
154
+ raise HtmlTableError(f"Invalid {attribute}: {raw_value!r}")
155
+ return span
156
+
157
+
158
+ def _row_belongs_to_thead(row: Tag, table: Tag) -> bool:
159
+ """判断 tr 是否位于当前 table 的 thead 内。"""
160
+ parent = row.parent
161
+ while isinstance(parent, Tag) and parent is not table:
162
+ if parent.name == "thead":
163
+ return True
164
+ parent = parent.parent
165
+ return False
166
+
167
+
168
+ __all__ = [
169
+ "HtmlTableCell",
170
+ "HtmlTableError",
171
+ "HtmlTableGrid",
172
+ "HtmlTableSource",
173
+ "MAX_NESTED_TABLE_DEPTH",
174
+ "MAX_TABLE_COLUMNS",
175
+ "MAX_TABLE_ROWS",
176
+ "MAX_TABLE_SLOTS",
177
+ "parse_html_tables",
178
+ ]
@@ -0,0 +1,33 @@
1
+ """各格式共用的目录页码尾部识别与清理。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ from ....content.inline import inline_plain_text, map_text_span_content, normalize_inline_spans, slice_inline_spans
8
+ from ....schema import InlineSpan
9
+
10
+ _INDEX_ROMAN_RE = re.compile(r"[ivxlcdm]+", re.IGNORECASE)
11
+
12
+
13
+ def strip_index_page_tail(content: list[InlineSpan]) -> list[InlineSpan]:
14
+ """删除目录末尾可信页码,并把其余 tab 转换为普通空格。"""
15
+ content = normalize_inline_spans(content)
16
+ visible_text = inline_plain_text(content)
17
+ if "\t" not in visible_text:
18
+ return content
19
+ tab_offset = visible_text.rfind("\t")
20
+ tail_text = visible_text[tab_offset + 1 :].strip()
21
+ if looks_like_index_page_token(tail_text):
22
+ content = slice_inline_spans(content, 0, tab_offset)
23
+ return map_text_span_content(content, lambda value: value.replace("\t", " "))
24
+
25
+
26
+ def looks_like_index_page_token(content: str) -> bool:
27
+ """判断目录 tab 后缀是否为数字、罗马数字或单字母页码。"""
28
+ if not content or len(content) > 12:
29
+ return False
30
+ return bool(content.isdigit() or _INDEX_ROMAN_RE.fullmatch(content) or re.fullmatch(r"[A-Za-z]", content))
31
+
32
+
33
+ __all__ = ["looks_like_index_page_token", "strip_index_page_tail"]
@@ -0,0 +1,158 @@
1
+ """各格式共用的列表 marker 解析与参考文献判定。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ import re
7
+ from typing import Literal, TypeAlias
8
+
9
+ from ....content.inline import inline_plain_text, normalize_inline_spans, slice_inline_spans
10
+ from ....schema import BlockType, InlineSpan, ListBlock
11
+
12
+ ListItemKind: TypeAlias = Literal["unordered", "ordered", "explicit", "none"]
13
+ OrderedListStyle: TypeAlias = Literal["decimal", "lower-alpha", "upper-alpha", "lower-roman", "upper-roman"]
14
+
15
+ _LIST_ITEM_MARKER_RE = re.compile(
16
+ r"^(?P<leading>\s*)(?P<marker>"
17
+ r"(?P<unordered>[-*+])"
18
+ r"|(?P<ordered>\d+\.|[A-Za-z]\.|[IVXLCDMivxlcdm]{2,}\.)"
19
+ r"|(?P<explicit>\d+\)|\(\d+[.)]|[A-Za-z]\)|[IVXLCDMivxlcdm]{2,}\)|\[[^\]\n]+\])"
20
+ r")(?P<separator>\s+)(?P<body>.*)$",
21
+ re.DOTALL,
22
+ )
23
+ _LEADING_WHITESPACE_RE = re.compile(r"^[ \t]*")
24
+ # 去除首部空白后,前五个可见字符内出现 Unicode 数字即视为单项命中。
25
+ _REFERENCE_NUMBER_PREFIX_RE = re.compile(r"^\D{0,4}\d")
26
+ _MARKDOWN_UNORDERED_MARKER_RE = re.compile(r"^[ \t]*-[ \t]+")
27
+ _ROMAN_MARKER_RE = re.compile(r"[IVXLCDM]+", re.IGNORECASE)
28
+ _CANONICAL_ROMAN_RE = re.compile(r"M{0,3}(?:CM|CD|D?C{0,3})(?:XC|XL|L?X{0,3})(?:IX|IV|V?I{0,3})")
29
+ _ROMAN_VALUES = {"I": 1, "V": 5, "X": 10, "L": 50, "C": 100, "D": 500, "M": 1000}
30
+ _MAX_NATIVE_ORDERED_VALUE = 1_000_000
31
+
32
+
33
+ @dataclass(frozen=True, slots=True)
34
+ class ListItem:
35
+ """保存一个列表条目的原始标记、正文与 HTML 所需分类。"""
36
+
37
+ marker: str | None
38
+ body: list[InlineSpan]
39
+ kind: ListItemKind
40
+ value: int | None
41
+ ordered_style: OrderedListStyle | None
42
+ leading: str
43
+ separator: str
44
+
45
+
46
+ def parse_list_item_marker(content: list[InlineSpan]) -> ListItem:
47
+ """解析列表行首 marker;无法识别时仍拆出前导水平空白。"""
48
+ content = normalize_inline_spans(content)
49
+ visible_text = inline_plain_text(content)
50
+ match = _LIST_ITEM_MARKER_RE.match(visible_text)
51
+ if match is None:
52
+ leading_match = _LEADING_WHITESPACE_RE.match(visible_text)
53
+ leading = leading_match.group(0) if leading_match is not None else ""
54
+ return ListItem(
55
+ marker=None,
56
+ body=slice_inline_spans(content, len(leading)),
57
+ kind="none",
58
+ value=None,
59
+ ordered_style=None,
60
+ leading=leading,
61
+ separator="",
62
+ )
63
+
64
+ marker = match.group("marker")
65
+ if match.group("unordered") is not None:
66
+ kind: ListItemKind = "unordered"
67
+ value = None
68
+ ordered_style = None
69
+ elif match.group("ordered") is not None:
70
+ ordered = _ordered_marker_value(marker)
71
+ if ordered is None:
72
+ kind = "explicit"
73
+ value = None
74
+ ordered_style = None
75
+ else:
76
+ kind = "ordered"
77
+ value, ordered_style = ordered
78
+ else:
79
+ kind = "explicit"
80
+ value = None
81
+ ordered_style = None
82
+ return ListItem(
83
+ marker=marker,
84
+ body=slice_inline_spans(content, match.start("body")),
85
+ kind=kind,
86
+ value=value,
87
+ ordered_style=ordered_style,
88
+ leading=match.group("leading"),
89
+ separator=match.group("separator"),
90
+ )
91
+
92
+
93
+ def _ordered_marker_value(marker: str) -> tuple[int, OrderedListStyle] | None:
94
+ """把有界且规范的点号 marker 转成序号;超限或畸形时返回 None。"""
95
+ stem = marker[:-1]
96
+ if re.fullmatch(r"(?:0|[1-9][0-9]*)", stem):
97
+ if len(stem) > 7:
98
+ return None
99
+ value = int(stem)
100
+ return (value, "decimal") if value <= _MAX_NATIVE_ORDERED_VALUE else None
101
+ # 单字符 i/v/x/l/c/d/m 固定按罗马数字解释,消除与字母序号的歧义。
102
+ if _ROMAN_MARKER_RE.fullmatch(stem):
103
+ if _CANONICAL_ROMAN_RE.fullmatch(stem.upper()) is None:
104
+ return None
105
+ style: OrderedListStyle = "upper-roman" if stem.isupper() else "lower-roman"
106
+ return _roman_marker_value(stem), style
107
+ if re.fullmatch(r"[A-Za-z]", stem) is None:
108
+ return None
109
+ style = "upper-alpha" if stem.isupper() else "lower-alpha"
110
+ return ord(stem.lower()) - ord("a") + 1, style
111
+
112
+
113
+ def _roman_marker_value(marker: str) -> int:
114
+ """按减法记数规则计算罗马 marker 的数值,兼容 producer 的宽松组合。"""
115
+ total = 0
116
+ previous = 0
117
+ for character in reversed(marker.upper()):
118
+ current = _ROMAN_VALUES[character]
119
+ if current < previous:
120
+ total -= current
121
+ else:
122
+ total += current
123
+ previous = current
124
+ return total
125
+
126
+
127
+ def has_markdown_unordered_marker(content: list[InlineSpan]) -> bool:
128
+ """判断条目是否已有 Markdown 短横线 marker,保持既有补 bullet 规则。"""
129
+ return _MARKDOWN_UNORDERED_MARKER_RE.match(inline_plain_text(normalize_inline_spans(content))) is not None
130
+
131
+
132
+ def reference_list_needs_bullets(block: ListBlock) -> bool:
133
+ """按直属非空条目的数字前缀严格多数规则判断是否补无序 marker。"""
134
+ if block.sub_type != BlockType.REF_TEXT:
135
+ return False
136
+
137
+ item_count = 0
138
+ numbered_count = 0
139
+ for child in block.content:
140
+ if isinstance(child, ListBlock):
141
+ continue
142
+ visible_text = inline_plain_text(child.content).lstrip()
143
+ if not visible_text:
144
+ continue
145
+ item_count += 1
146
+ if _REFERENCE_NUMBER_PREFIX_RE.match(visible_text):
147
+ numbered_count += 1
148
+ return item_count > 0 and numbered_count * 2 <= item_count
149
+
150
+
151
+ __all__ = [
152
+ "ListItem",
153
+ "ListItemKind",
154
+ "OrderedListStyle",
155
+ "has_markdown_unordered_marker",
156
+ "parse_list_item_marker",
157
+ "reference_list_needs_bullets",
158
+ ]
@@ -0,0 +1,140 @@
1
+ """各格式共用的逻辑块复制、延续合并与页面规划。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+
7
+ from .context import claim_owned_document
8
+
9
+ from ....content.table import merge_table_content
10
+ from ...contracts import RenderMode
11
+ from ....schema import (
12
+ MERGE_TRANSPARENT_BLOCK_TYPES,
13
+ BlockType,
14
+ ContinuableTextBlockBase,
15
+ ListBlock,
16
+ MiddleJson,
17
+ PageBlock,
18
+ InlineSpan,
19
+ RefTextBlock,
20
+ TableBlock,
21
+ TextBlock,
22
+ )
23
+
24
+
25
+ @dataclass(slots=True)
26
+ class PlannedBlock:
27
+ """保存一个待渲染块及其来源页和文本延续片段。"""
28
+
29
+ page_idx: int
30
+ block: PageBlock
31
+ text_contents: list[list[InlineSpan]] = field(default_factory=list)
32
+ removed: bool = False
33
+
34
+
35
+ def build_render_plan(
36
+ middle_json: MiddleJson,
37
+ mode: RenderMode = RenderMode.DEFAULT,
38
+ ) -> list[list[PlannedBlock]]:
39
+ """深拷贝 MiddleJson,并按模式生成不污染输入的逐页逻辑块计划。"""
40
+ owned = claim_owned_document(middle_json)
41
+ copied = middle_json if owned else middle_json.model_copy(deep=True)
42
+ pages = [
43
+ [
44
+ PlannedBlock(
45
+ page_idx=page.page_idx,
46
+ block=block,
47
+ text_contents=[block.content] if isinstance(block, ContinuableTextBlockBase) else [],
48
+ )
49
+ for block in page.blocks
50
+ ]
51
+ for page in copied.pages
52
+ ]
53
+ flattened = [planned for page in pages for planned in page]
54
+ _merge_continued_text_blocks(flattened, mode)
55
+ _merge_continued_list_blocks(flattened, mode, copy_on_merge=owned)
56
+ if mode is RenderMode.DEFAULT:
57
+ _merge_continued_table_blocks(flattened)
58
+ return pages
59
+
60
+
61
+ def _merge_continued_text_blocks(blocks: list[PlannedBlock], mode: RenderMode) -> None:
62
+ """把无独立正文锚点的 continues_prev 文本吸收到最近的前序文本逻辑块。"""
63
+ previous_text: PlannedBlock | None = None
64
+ previous_reference: PlannedBlock | None = None
65
+ for current in blocks:
66
+ is_text = isinstance(current.block, TextBlock)
67
+ is_reference = isinstance(current.block, RefTextBlock)
68
+ previous = previous_text if is_text else previous_reference if is_reference else None
69
+ anchored = is_text and isinstance(current.block.anchor, str) and bool(current.block.anchor.strip())
70
+ if (
71
+ not current.removed
72
+ and isinstance(current.block, ContinuableTextBlockBase)
73
+ and current.block.continues_prev is True
74
+ and not anchored
75
+ and previous is not None
76
+ and not (mode is RenderMode.FULL and previous.page_idx != current.page_idx)
77
+ ):
78
+ previous.text_contents.extend(current.text_contents)
79
+ current.removed = True
80
+ if is_text and not current.removed:
81
+ previous_text = current
82
+ if is_reference:
83
+ if not current.removed:
84
+ previous_reference = current
85
+ elif current.block.type not in MERGE_TRANSPARENT_BLOCK_TYPES:
86
+ previous_reference = None
87
+
88
+
89
+ def _merge_continued_list_blocks(blocks: list[PlannedBlock], mode: RenderMode, *, copy_on_merge: bool = False) -> None:
90
+ """把续接列表吸收到子类型一致的前序列表,参考文献可跨过合并透明块。"""
91
+ previous_list: PlannedBlock | None = None
92
+ previous_reference_list: PlannedBlock | None = None
93
+ copied: set[int] = set()
94
+ for current in blocks:
95
+ if not isinstance(current.block, ListBlock):
96
+ previous_list = None
97
+ if current.block.type not in MERGE_TRANSPARENT_BLOCK_TYPES:
98
+ previous_reference_list = None
99
+ continue
100
+ previous = previous_reference_list if current.block.sub_type == BlockType.REF_TEXT else previous_list
101
+ if (
102
+ not current.removed
103
+ and current.block.continues_prev is True
104
+ and previous is not None
105
+ and previous.block.sub_type == current.block.sub_type
106
+ and not (mode is RenderMode.FULL and previous.page_idx != current.page_idx)
107
+ ):
108
+ # EPUB 等渲染器仍读取原文档;只有实际修改的列表需要另建副本。
109
+ if copy_on_merge and id(previous) not in copied:
110
+ previous.block = previous.block.model_copy(deep=True)
111
+ copied.add(id(previous))
112
+ previous.block.content.extend(current.block.content)
113
+ current.removed = True
114
+ if not current.removed:
115
+ previous_list = previous_reference_list = current
116
+
117
+
118
+ def _merge_continued_table_blocks(blocks: list[PlannedBlock]) -> None:
119
+ """在默认模式中把跨页续表合并到最近的前序表格。"""
120
+ previous: PlannedBlock | None = None
121
+ for current in blocks:
122
+ if current.removed or not isinstance(current.block, TableBlock):
123
+ continue
124
+ if current.block.continues_prev is True and previous is not None and previous.page_idx != current.page_idx:
125
+ merged = merge_table_content(
126
+ previous.block.model_dump(mode="python", exclude_none=True),
127
+ current.block.model_dump(mode="python", exclude_none=True),
128
+ )
129
+ if merged is not None:
130
+ try:
131
+ previous.block = TableBlock.model_validate(merged)
132
+ except (TypeError, ValueError):
133
+ pass
134
+ else:
135
+ current.removed = True
136
+ if not current.removed:
137
+ previous = current
138
+
139
+
140
+ __all__ = ["PlannedBlock", "build_render_plan"]
@@ -0,0 +1,3 @@
1
+ """DOCX renderer 的内部实现。"""
2
+
3
+ __all__: list[str] = []