docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,206 @@
1
+ """PDF renderer 的页面几何、字体与打印样式。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from importlib import resources
7
+ from threading import RLock
8
+
9
+ from reportlab.lib import colors
10
+ from reportlab.lib.enums import TA_CENTER, TA_LEFT
11
+ from reportlab.lib.styles import ParagraphStyle
12
+ from reportlab.lib.units import mm
13
+ from reportlab.pdfbase import cidfonts, pdfmetrics
14
+ from reportlab.pdfbase.ttfonts import TTFont
15
+
16
+ PAGE_MARGIN = 20 * mm
17
+ BODY_FONT = "Helvetica"
18
+ BODY_BOLD_FONT = "Helvetica-Bold"
19
+ MONO_FONT = "Courier"
20
+ HAN_FONT = "STSong-Light"
21
+ JAPANESE_FONT = "HeiseiMin-W3"
22
+ KOREAN_FONT = "HYSMyeongJo-Medium"
23
+ UNICODE_FALLBACK_FONT = "DocVortex-DejaVuSans"
24
+
25
+ ACCENT_COLOR = colors.HexColor("#0b6fc2")
26
+ BACKGROUND_COLOR = colors.HexColor("#ffffff")
27
+ BORDER_COLOR = colors.HexColor("#d1d5db")
28
+ MUTED_COLOR = colors.HexColor("#6b7280")
29
+ SURFACE_COLOR = colors.HexColor("#f8fafc")
30
+ TEXT_COLOR = colors.HexColor("#1f2937")
31
+
32
+ _FONT_LOCK = RLock()
33
+ _HEADING_SIZES = (20.0, 16.0, 14.0, 12.0, 11.0, 10.0)
34
+
35
+
36
+ @dataclass(frozen=True, slots=True)
37
+ class PdfStyleSet:
38
+ """保存 PDF renderer 使用的全部稳定段落样式。"""
39
+
40
+ body: ParagraphStyle
41
+ headings: tuple[ParagraphStyle, ...]
42
+ caption: ParagraphStyle
43
+ footnote: ParagraphStyle
44
+ code: ParagraphStyle
45
+ spatial_table: ParagraphStyle
46
+ formula_fallback: ParagraphStyle
47
+ placeholder: ParagraphStyle
48
+ table_cell: ParagraphStyle
49
+ table_header: ParagraphStyle
50
+
51
+ def heading(self, level: int) -> ParagraphStyle:
52
+ """按公开一到六级标题返回对应的 PDF 样式。"""
53
+ return self.headings[min(max(level, 1), len(self.headings)) - 1]
54
+
55
+
56
+ def build_pdf_styles() -> PdfStyleSet:
57
+ """注册标准 CID 字体并构造无外部字体依赖的打印样式集。"""
58
+ _register_pdf_fonts()
59
+ body = ParagraphStyle(
60
+ "DocVortex PDF Body",
61
+ fontName=BODY_FONT,
62
+ fontSize=10.5,
63
+ leading=16,
64
+ textColor=TEXT_COLOR,
65
+ alignment=TA_LEFT,
66
+ spaceBefore=0,
67
+ spaceAfter=7,
68
+ splitLongWords=True,
69
+ allowWidows=1,
70
+ allowOrphans=1,
71
+ )
72
+ headings = tuple(
73
+ ParagraphStyle(
74
+ f"DocVortex PDF Heading {level}",
75
+ parent=body,
76
+ fontName=BODY_BOLD_FONT,
77
+ fontSize=size,
78
+ leading=size * 1.3,
79
+ spaceBefore=10 if level <= 3 else 8,
80
+ spaceAfter=6 if level <= 3 else 4,
81
+ keepWithNext=True,
82
+ )
83
+ for level, size in enumerate(_HEADING_SIZES, start=1)
84
+ )
85
+ caption = ParagraphStyle(
86
+ "DocVortex PDF Caption",
87
+ parent=body,
88
+ fontSize=9,
89
+ leading=12,
90
+ textColor=MUTED_COLOR,
91
+ spaceBefore=3,
92
+ spaceAfter=5,
93
+ )
94
+ footnote = ParagraphStyle(
95
+ "DocVortex PDF Footnote",
96
+ parent=body,
97
+ fontSize=8.5,
98
+ leading=11,
99
+ textColor=MUTED_COLOR,
100
+ spaceBefore=2,
101
+ spaceAfter=4,
102
+ )
103
+ code = ParagraphStyle(
104
+ "DocVortex PDF Code",
105
+ parent=body,
106
+ fontName=MONO_FONT,
107
+ fontSize=8.5,
108
+ leading=11,
109
+ backColor=SURFACE_COLOR,
110
+ borderColor=BORDER_COLOR,
111
+ borderWidth=0.5,
112
+ borderPadding=7,
113
+ borderRadius=3,
114
+ spaceBefore=5,
115
+ spaceAfter=7,
116
+ )
117
+ spatial_table = ParagraphStyle(
118
+ "DocVortex PDF Spatial Table",
119
+ parent=code,
120
+ backColor=BACKGROUND_COLOR,
121
+ borderPadding=5,
122
+ )
123
+ formula_fallback = ParagraphStyle(
124
+ "DocVortex PDF Formula Fallback",
125
+ parent=body,
126
+ fontName=MONO_FONT,
127
+ fontSize=9,
128
+ leading=13,
129
+ textColor=MUTED_COLOR,
130
+ alignment=TA_CENTER,
131
+ spaceBefore=5,
132
+ spaceAfter=6,
133
+ )
134
+ placeholder = ParagraphStyle(
135
+ "DocVortex PDF Placeholder",
136
+ parent=body,
137
+ fontSize=9,
138
+ leading=12,
139
+ textColor=MUTED_COLOR,
140
+ spaceBefore=0,
141
+ spaceAfter=0,
142
+ )
143
+ table_cell = ParagraphStyle(
144
+ "DocVortex PDF Table Cell",
145
+ parent=body,
146
+ fontSize=8.5,
147
+ leading=11,
148
+ spaceBefore=0,
149
+ spaceAfter=0,
150
+ )
151
+ table_header = ParagraphStyle(
152
+ "DocVortex PDF Table Header",
153
+ parent=table_cell,
154
+ fontName=BODY_BOLD_FONT,
155
+ )
156
+ return PdfStyleSet(
157
+ body=body,
158
+ headings=headings,
159
+ caption=caption,
160
+ footnote=footnote,
161
+ code=code,
162
+ spatial_table=spatial_table,
163
+ formula_fallback=formula_fallback,
164
+ placeholder=placeholder,
165
+ table_cell=table_cell,
166
+ table_header=table_header,
167
+ )
168
+
169
+
170
+ def _register_pdf_fonts() -> None:
171
+ """幂等注册中日韩 CID 字体与 ziafont 自带 Unicode 回退字体。"""
172
+ with _FONT_LOCK:
173
+ registered = set(pdfmetrics.getRegisteredFontNames())
174
+ for font_name in (HAN_FONT, JAPANESE_FONT, KOREAN_FONT):
175
+ if font_name not in registered:
176
+ pdfmetrics.registerFont(cidfonts.UnicodeCIDFont(font_name))
177
+ if UNICODE_FALLBACK_FONT not in registered:
178
+ font_path = resources.files("ziafont").joinpath("fonts", "DejaVuSans.ttf")
179
+ pdfmetrics.registerFont(TTFont(UNICODE_FALLBACK_FONT, str(font_path)))
180
+ pdfmetrics.registerFontFamily(
181
+ UNICODE_FALLBACK_FONT,
182
+ normal=UNICODE_FALLBACK_FONT,
183
+ bold=UNICODE_FALLBACK_FONT,
184
+ italic=UNICODE_FALLBACK_FONT,
185
+ boldItalic=UNICODE_FALLBACK_FONT,
186
+ )
187
+
188
+
189
+ __all__ = [
190
+ "ACCENT_COLOR",
191
+ "BACKGROUND_COLOR",
192
+ "BODY_BOLD_FONT",
193
+ "BODY_FONT",
194
+ "BORDER_COLOR",
195
+ "HAN_FONT",
196
+ "JAPANESE_FONT",
197
+ "KOREAN_FONT",
198
+ "MONO_FONT",
199
+ "MUTED_COLOR",
200
+ "PAGE_MARGIN",
201
+ "PdfStyleSet",
202
+ "SURFACE_COLOR",
203
+ "TEXT_COLOR",
204
+ "UNICODE_FALLBACK_FONT",
205
+ "build_pdf_styles",
206
+ ]
@@ -0,0 +1,272 @@
1
+ """HTML table 到可分页 ReportLab 原生表格的安全物化。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Protocol
6
+
7
+ from bs4 import NavigableString, Tag
8
+ from pydantic import ValidationError
9
+ from reportlab.platypus import Flowable, LongTable, Paragraph, Table, TableStyle
10
+
11
+ from ....schema import CodeInlineSpan, EquationInlineSpan, HyperlinkSpan, InlineSpan, InlineStyle, TextSpan, parse_inline_spans
12
+ from ....foundation.hyperlink import OFFICE_EXTERNAL_HYPERLINK_SCHEMES, sanitize_hyperlink_target
13
+ from ..common.html_table import (
14
+ HtmlTableCell,
15
+ HtmlTableError,
16
+ HtmlTableGrid,
17
+ HtmlTableSource,
18
+ MAX_NESTED_TABLE_DEPTH,
19
+ parse_html_tables as _parse_common_html_tables,
20
+ )
21
+ from .styles import BORDER_COLOR, SURFACE_COLOR, PdfStyleSet
22
+
23
+ _BLOCK_TAGS = {"address", "article", "blockquote", "div", "figcaption", "footer", "header", "li", "p", "section"}
24
+ _SKIPPED_TAGS = {"script", "style", "template", "noscript"}
25
+
26
+
27
+ class PdfTableError(HtmlTableError):
28
+ """表示 HTML 表格结构或 PDF 表格几何无法安全物化。"""
29
+
30
+
31
+ class ParagraphBuilder(Protocol):
32
+ """定义表格单元格创建富文本 Paragraph 的回调。"""
33
+
34
+ def __call__(self, spans: list[InlineSpan], style: object, max_width: float) -> Paragraph:
35
+ """把单元格行内 span 构造成指定宽度的 Paragraph。"""
36
+ ...
37
+
38
+
39
+ class HtmlImageBuilder(Protocol):
40
+ """定义表格单元格创建离线图片或占位 Flowable 的回调。"""
41
+
42
+ def __call__(self, source: str, max_width: float, alt_text: str) -> Flowable:
43
+ """把 HTML img source 转换为图片或宽松占位。"""
44
+ ...
45
+
46
+
47
+ def parse_html_tables(source: HtmlTableSource) -> tuple[HtmlTableGrid, ...]:
48
+ """复用共用网格解析,并保持 PDF 私有异常类型不变。"""
49
+ try:
50
+ return _parse_common_html_tables(source)
51
+ except HtmlTableError as exc:
52
+ raise PdfTableError(str(exc)) from exc
53
+
54
+
55
+ def build_pdf_tables(
56
+ source: HtmlTableSource,
57
+ *,
58
+ available_width: float,
59
+ styles: PdfStyleSet,
60
+ build_paragraph: ParagraphBuilder,
61
+ build_image: HtmlImageBuilder,
62
+ depth: int = 1,
63
+ ) -> tuple[Table, ...]:
64
+ """把 HTML 表格递归转换为支持合并单元格与重复表头的 PDF 表格。"""
65
+ if depth > MAX_NESTED_TABLE_DEPTH:
66
+ raise PdfTableError(f"Nested table depth exceeds {MAX_NESTED_TABLE_DEPTH}")
67
+ if available_width <= 0:
68
+ raise PdfTableError("available_width must be positive")
69
+ grids = parse_html_tables(source)
70
+ return tuple(
71
+ _build_pdf_table(
72
+ grid,
73
+ available_width=available_width,
74
+ styles=styles,
75
+ build_paragraph=build_paragraph,
76
+ build_image=build_image,
77
+ depth=depth,
78
+ )
79
+ for grid in grids
80
+ )
81
+
82
+
83
+ def _build_pdf_table(
84
+ grid: HtmlTableGrid,
85
+ *,
86
+ available_width: float,
87
+ styles: PdfStyleSet,
88
+ build_paragraph: ParagraphBuilder,
89
+ build_image: HtmlImageBuilder,
90
+ depth: int,
91
+ ) -> Table:
92
+ """物化一个网格,写入单元格内容、合并区域与固定打印样式。"""
93
+ column_widths = _column_widths(available_width, grid.column_count)
94
+ data: list[list[object]] = [["" for _ in range(grid.column_count)] for _ in range(grid.row_count)]
95
+ commands: list[tuple[object, ...]] = [
96
+ ("GRID", (0, 0), (-1, -1), 0.5, BORDER_COLOR),
97
+ ("VALIGN", (0, 0), (-1, -1), "TOP"),
98
+ ("LEFTPADDING", (0, 0), (-1, -1), 5),
99
+ ("RIGHTPADDING", (0, 0), (-1, -1), 5),
100
+ ("TOPPADDING", (0, 0), (-1, -1), 4),
101
+ ("BOTTOMPADDING", (0, 0), (-1, -1), 4),
102
+ ]
103
+ for row_index in grid.header_rows:
104
+ commands.append(("BACKGROUND", (0, row_index), (-1, row_index), SURFACE_COLOR))
105
+ for placement in grid.cells:
106
+ cell_width = sum(column_widths[placement.column : placement.column + placement.colspan]) - 10
107
+ style = styles.table_header if placement.is_header else styles.table_cell
108
+ data[placement.row][placement.column] = _cell_flowables(
109
+ placement.tag,
110
+ max_width=max(1.0, cell_width),
111
+ style=style,
112
+ styles=styles,
113
+ build_paragraph=build_paragraph,
114
+ build_image=build_image,
115
+ depth=depth,
116
+ )
117
+ if placement.rowspan > 1 or placement.colspan > 1:
118
+ commands.append(
119
+ (
120
+ "SPAN",
121
+ (placement.column, placement.row),
122
+ (placement.end_column, placement.end_row),
123
+ )
124
+ )
125
+ repeat_rows = 0
126
+ while repeat_rows in grid.header_rows:
127
+ repeat_rows += 1
128
+ table_class = LongTable if depth == 1 else Table
129
+ table = table_class(
130
+ data,
131
+ colWidths=column_widths,
132
+ repeatRows=repeat_rows,
133
+ splitByRow=1,
134
+ splitInRow=1,
135
+ hAlign="LEFT",
136
+ )
137
+ table.setStyle(TableStyle(commands))
138
+ return table
139
+
140
+
141
+ def _cell_flowables(
142
+ cell: Tag,
143
+ *,
144
+ max_width: float,
145
+ style: object,
146
+ styles: PdfStyleSet,
147
+ build_paragraph: ParagraphBuilder,
148
+ build_image: HtmlImageBuilder,
149
+ depth: int,
150
+ ) -> list[Flowable]:
151
+ """把单元格文本、图片与直接嵌套表按安全顺序转换为 Flowable。"""
152
+ flowables: list[Flowable] = []
153
+ spans = _html_cell_spans(cell)
154
+ if spans:
155
+ flowables.append(build_paragraph(spans, style, max_width))
156
+ for image in cell.find_all("img"):
157
+ if image.find_parent("table") is not cell.find_parent("table"):
158
+ continue
159
+ source = image.get("src", "")
160
+ if isinstance(source, list):
161
+ source = ""
162
+ alt = image.get("alt", "image")
163
+ if isinstance(alt, list):
164
+ alt = " ".join(str(item) for item in alt)
165
+ flowables.append(build_image(str(source), max_width, str(alt)))
166
+ for nested in cell.find_all("table"):
167
+ if nested.find_parent(("td", "th")) is not cell:
168
+ continue
169
+ flowables.extend(
170
+ build_pdf_tables(
171
+ nested,
172
+ available_width=max_width,
173
+ styles=styles,
174
+ build_paragraph=build_paragraph,
175
+ build_image=build_image,
176
+ depth=depth + 1,
177
+ )
178
+ )
179
+ if not flowables:
180
+ flowables.append(build_paragraph([TextSpan(type="text", content=" ")], style, max_width))
181
+ return flowables
182
+
183
+
184
+ def _html_cell_spans(cell: Tag) -> list[InlineSpan]:
185
+ """把单元格中非图片、非嵌套表内容转换为严格 InlineSpan。"""
186
+ spans: list[InlineSpan] = []
187
+ for child in cell.children:
188
+ spans.extend(_html_node_spans(child, styles=(), allow_links=True))
189
+ if not spans:
190
+ return []
191
+ try:
192
+ return parse_inline_spans(spans)
193
+ except (TypeError, ValidationError, ValueError) as exc:
194
+ raise PdfTableError("HTML table cell inline content is invalid") from exc
195
+
196
+
197
+ def _html_node_spans(node: object, *, styles: tuple[InlineStyle, ...], allow_links: bool) -> list[InlineSpan]:
198
+ """递归解析安全 HTML 富文本标签,忽略活动内容与独立视觉节点。"""
199
+ if isinstance(node, NavigableString):
200
+ text = str(node)
201
+ return [TextSpan(type="text", content=text, styles=list(styles))] if text else []
202
+ if not isinstance(node, Tag):
203
+ return []
204
+ name = (node.name or "").lower()
205
+ if name in _SKIPPED_TAGS or name in {"img", "table"}:
206
+ return []
207
+ if name == "br":
208
+ return [TextSpan(type="text", content="\n", styles=list(styles))]
209
+ if name == "eq":
210
+ content = node.get_text()
211
+ return [EquationInlineSpan(type="equation_inline", content=content)] if content.strip() else []
212
+ if name == "code":
213
+ content = node.get_text()
214
+ return [CodeInlineSpan(type="code_inline", content=content)] if content else []
215
+ style_name: InlineStyle | None = {
216
+ "b": "bold",
217
+ "strong": "bold",
218
+ "i": "italic",
219
+ "em": "italic",
220
+ "u": "underline",
221
+ "s": "strikethrough",
222
+ "del": "strikethrough",
223
+ "sup": "superscript",
224
+ "sub": "subscript",
225
+ }.get(name)
226
+ child_styles = tuple(dict.fromkeys((*styles, style_name))) if style_name is not None else styles
227
+ if name == "a" and allow_links:
228
+ children = [
229
+ span
230
+ for child in node.children
231
+ for span in _html_node_spans(child, styles=child_styles, allow_links=False)
232
+ if not isinstance(span, HyperlinkSpan)
233
+ ]
234
+ target = sanitize_hyperlink_target(
235
+ node.get("href"),
236
+ allowed_schemes=OFFICE_EXTERNAL_HYPERLINK_SCHEMES,
237
+ allow_relative=True,
238
+ allow_fragment=True,
239
+ )
240
+ if target is not None and children:
241
+ try:
242
+ return [HyperlinkSpan(type="hyperlink", url=target, content=children)] # type: ignore[arg-type]
243
+ except ValidationError:
244
+ pass
245
+ return children
246
+ spans = [span for child in node.children for span in _html_node_spans(child, styles=child_styles, allow_links=allow_links)]
247
+ if name in _BLOCK_TAGS and spans and not _spans_end_with_newline(spans):
248
+ spans.append(TextSpan(type="text", content="\n"))
249
+ return spans
250
+
251
+
252
+ def _spans_end_with_newline(spans: list[InlineSpan]) -> bool:
253
+ """判断当前 span 序列是否已经以普通文本换行结束。"""
254
+ return bool(spans and isinstance(spans[-1], TextSpan) and spans[-1].content.endswith("\n"))
255
+
256
+
257
+ def _column_widths(total_width: float, column_count: int) -> list[float]:
258
+ """把可用宽度确定性地均分到全部逻辑列。"""
259
+ if column_count <= 0:
260
+ raise PdfTableError("Table must contain at least one column")
261
+ base = total_width / column_count
262
+ return [base for _ in range(column_count)]
263
+
264
+
265
+ __all__ = [
266
+ "HtmlTableCell",
267
+ "HtmlTableGrid",
268
+ "MAX_NESTED_TABLE_DEPTH",
269
+ "PdfTableError",
270
+ "build_pdf_tables",
271
+ "parse_html_tables",
272
+ ]
@@ -0,0 +1,3 @@
1
+ """Structured Content 私有渲染实现。"""
2
+
3
+ __all__: list[str] = []
@@ -0,0 +1,193 @@
1
+ """严格 MiddleJson 到树形 Markdown structured_content 的公共渲染实现。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any, TypeAlias
6
+
7
+ from ....options import LatexDelimitersConfig
8
+ from ..markdown.blocks import (
9
+ render_single_block,
10
+ render_title_inline_content,
11
+ render_visual_annotation,
12
+ render_visual_body_content,
13
+ )
14
+ from ..markdown.assets import normalize_image_source, resolve_image_source
15
+ from ..markdown.escaping import escape_standalone_marker_rule, escape_text_block_markdown_prefix
16
+ from ..markdown.inline import render_inline_content
17
+ from ....schema import (
18
+ BlockType,
19
+ ChartAnnotationBlock,
20
+ ChartBlock,
21
+ ChartBodyBlock,
22
+ CodeAnnotationBlock,
23
+ CodeBlock,
24
+ DocTitleBlock,
25
+ EquationBlock,
26
+ ImageAnnotationBlock,
27
+ ImageBlock,
28
+ ImageBodyBlock,
29
+ ImagePayloadBlock,
30
+ MiddleJson,
31
+ PageAuxTextBlock,
32
+ PageFootnoteBlock,
33
+ PageBlock,
34
+ ParagraphTitleBlock,
35
+ TextBlock,
36
+ TableAnnotationBlock,
37
+ TableBlock,
38
+ TableBodyBlock,
39
+ )
40
+
41
+ VisualBlock: TypeAlias = ImageBlock | TableBlock | ChartBlock | CodeBlock
42
+ VisualAnnotationBlock: TypeAlias = ImageAnnotationBlock | TableAnnotationBlock | ChartAnnotationBlock | CodeAnnotationBlock
43
+
44
+ _CAPTION_TYPES = {
45
+ BlockType.IMAGE_CAPTION,
46
+ BlockType.TABLE_CAPTION,
47
+ BlockType.CHART_CAPTION,
48
+ BlockType.CODE_CAPTION,
49
+ }
50
+ _FOOTNOTE_TYPES = {
51
+ BlockType.IMAGE_FOOTNOTE,
52
+ BlockType.TABLE_FOOTNOTE,
53
+ BlockType.CHART_FOOTNOTE,
54
+ BlockType.CODE_FOOTNOTE,
55
+ }
56
+ _REMOVED_BLOCK_FIELDS = {"content", "index", "guess_lang", "image_path", "image_base64", "image_url"}
57
+
58
+
59
+ def render_structured_content(
60
+ middle_json: MiddleJson,
61
+ *,
62
+ asset_base_url: str = "",
63
+ latex_delimiters: LatexDelimitersConfig | None = None,
64
+ ) -> dict[str, Any]:
65
+ """把严格 MiddleJson 无副作用地渲染为树形 Markdown structured_content。"""
66
+ if not isinstance(middle_json, MiddleJson):
67
+ raise TypeError("render_structured_content expects a MiddleJson instance")
68
+
69
+ delimiters = latex_delimiters or LatexDelimitersConfig()
70
+ document_fields = middle_json.model_dump(
71
+ mode="json",
72
+ exclude={"pages", "schema_id", "schema_version"},
73
+ exclude_defaults=True,
74
+ )
75
+ pages = [
76
+ {
77
+ "page_idx": page.page_idx,
78
+ "blocks": [
79
+ _render_content_block(
80
+ block,
81
+ delimiters=delimiters,
82
+ asset_base_url=asset_base_url,
83
+ )
84
+ for block in page.blocks
85
+ ],
86
+ }
87
+ for page in middle_json.pages
88
+ ]
89
+ return {"pages": pages, **document_fields}
90
+
91
+
92
+ def _render_content_block(
93
+ block: PageBlock,
94
+ *,
95
+ delimiters: LatexDelimitersConfig,
96
+ asset_base_url: str,
97
+ ) -> dict[str, Any]:
98
+ """保留父块元数据,并把 block 内容收敛为 Markdown 字符串。"""
99
+ payload = block.model_dump(
100
+ mode="json",
101
+ exclude=_REMOVED_BLOCK_FIELDS,
102
+ exclude_defaults=True,
103
+ )
104
+ if isinstance(block, (DocTitleBlock, ParagraphTitleBlock)):
105
+ content = render_title_inline_content(block, delimiters)
106
+ payload["content"] = escape_standalone_marker_rule(escape_text_block_markdown_prefix(content))
107
+ return payload
108
+ if isinstance(block, TextBlock):
109
+ content = render_inline_content(block.content, delimiters)
110
+ payload["content"] = escape_standalone_marker_rule(escape_text_block_markdown_prefix(content))
111
+ return payload
112
+ if isinstance(block, EquationBlock):
113
+ payload["content"] = block.content.strip()
114
+ image_source = _resolve_content_image_source(block, asset_base_url)
115
+ if image_source is not None:
116
+ payload["image_source"] = image_source
117
+ return payload
118
+ if isinstance(block, (PageAuxTextBlock, PageFootnoteBlock)):
119
+ content = escape_text_block_markdown_prefix(render_inline_content(block.content, delimiters))
120
+ payload["content"] = escape_standalone_marker_rule(content)
121
+ return payload
122
+ if not isinstance(block, (ImageBlock, TableBlock, ChartBlock, CodeBlock)):
123
+ payload["content"] = render_single_block(
124
+ block,
125
+ delimiters=delimiters,
126
+ asset_base_url=asset_base_url,
127
+ )
128
+ return payload
129
+
130
+ payload["content"] = render_visual_body_content(
131
+ block,
132
+ delimiters=delimiters,
133
+ asset_base_url=asset_base_url,
134
+ )
135
+ payload["captions"] = _render_annotation_group(block, _CAPTION_TYPES, delimiters)
136
+ payload["footnotes"] = _render_annotation_group(block, _FOOTNOTE_TYPES, delimiters)
137
+ image_source = _resolve_visual_image_source(block, asset_base_url)
138
+ if image_source is not None:
139
+ payload["image_source"] = image_source
140
+ return payload
141
+
142
+
143
+ def _render_annotation_group(
144
+ block: VisualBlock,
145
+ accepted_types: set[str],
146
+ delimiters: LatexDelimitersConfig,
147
+ ) -> list[dict[str, Any]]:
148
+ """按源 index 稳定排序视觉说明,并保留可用 bbox 与 Markdown 内容。"""
149
+ annotations: list[tuple[int, VisualAnnotationBlock]] = []
150
+ for position, child in enumerate(block.content):
151
+ if (
152
+ isinstance(
153
+ child,
154
+ (ImageAnnotationBlock, TableAnnotationBlock, ChartAnnotationBlock, CodeAnnotationBlock),
155
+ )
156
+ and child.type in accepted_types
157
+ ):
158
+ annotations.append((position, child))
159
+ annotations.sort(key=_annotation_sort_key)
160
+ rendered_annotations: list[dict[str, Any]] = []
161
+ for _, child in annotations:
162
+ annotation_payload: dict[str, Any] = {}
163
+ if child.bbox is not None:
164
+ annotation_payload["bbox"] = list(child.bbox)
165
+ annotation_payload["content"] = render_visual_annotation(child, delimiters)
166
+ rendered_annotations.append(annotation_payload)
167
+ return rendered_annotations
168
+
169
+
170
+ def _annotation_sort_key(
171
+ item: tuple[int, VisualAnnotationBlock],
172
+ ) -> tuple[bool, int, int]:
173
+ """让有 index 的说明升序优先,缺失 index 的说明稳定排在末尾。"""
174
+ position, block = item
175
+ return block.index is None, block.index if block.index is not None else 0, position
176
+
177
+
178
+ def _resolve_visual_image_source(block: VisualBlock, asset_base_url: str) -> str | None:
179
+ """解析视觉 body 实际选择的图片来源,并返回安全的 Markdown 地址。"""
180
+ for child in block.content:
181
+ if not isinstance(child, (ImageBodyBlock, TableBodyBlock, ChartBodyBlock)):
182
+ continue
183
+ return _resolve_content_image_source(child, asset_base_url)
184
+ return None
185
+
186
+
187
+ def _resolve_content_image_source(block: ImagePayloadBlock, asset_base_url: str) -> str | None:
188
+ """把图片载荷收敛为 structured_content 中唯一且安全的 image_source。"""
189
+ source = resolve_image_source(block, asset_base_url)
190
+ return normalize_image_source(source) if source else None
191
+
192
+
193
+ __all__ = ["render_structured_content"]