docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1156 @@
1
+ """严格 MiddleJson 到单正文 EPUB 3.3 的静态 XHTML renderer。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from datetime import datetime, timezone
7
+ import hashlib
8
+ import json
9
+ from functools import lru_cache
10
+ from importlib import resources
11
+ import re
12
+ from urllib.parse import quote, unquote, urlsplit
13
+ from uuid import NAMESPACE_URL, uuid5
14
+
15
+ from bs4 import BeautifulSoup, NavigableString, Tag
16
+ from bs4.element import Comment, Doctype, ProcessingInstruction
17
+ from latex2mathml.converter import convert as latex_to_mathml
18
+ from lxml import etree
19
+
20
+ from ....content.inline import inline_plain_text, join_inline_spans, normalize_inline_spans
21
+ from ....schema import (
22
+ PAGE_AUXILIARY_BLOCK_TYPES,
23
+ RAW_ALGORITHM,
24
+ AlgorithmBodyBlock,
25
+ BlockBase,
26
+ BlockType,
27
+ ChartAnnotationBlock,
28
+ ChartBlock,
29
+ ChartBodyBlock,
30
+ CodeAnnotationBlock,
31
+ CodeBlock,
32
+ CodeBodyBlock,
33
+ CodeInlineSpan,
34
+ DocTitleBlock,
35
+ EquationBlock,
36
+ EquationInlineSpan,
37
+ HyperlinkSpan,
38
+ ImageAnnotationBlock,
39
+ ImageBlock,
40
+ ImageBodyBlock,
41
+ IndexBlock,
42
+ InlineSpan,
43
+ ListBlock,
44
+ MiddleJson,
45
+ PageFootnoteBlock,
46
+ ParagraphTitleBlock,
47
+ RefTextBlock,
48
+ TableAnnotationBlock,
49
+ TableBlock,
50
+ TableBodyBlock,
51
+ TextBlock,
52
+ TextSpan,
53
+ TitleBlockBase,
54
+ )
55
+ from ...contracts import AssetResolver, EpubRenderOptions
56
+ from ..common.index import strip_index_page_tail
57
+ from ..common.list_items import ListItem, parse_list_item_marker, reference_list_needs_bullets
58
+ from ..common.planner import PlannedBlock, build_render_plan
59
+ from .assets import EpubAssetRegistry
60
+ from .package import EpubMetadata, NavigationItem, build_epub_package
61
+
62
+ _EPUB_NS = "http://www.idpf.org/2007/ops"
63
+ _MATHML_NS = "http://www.w3.org/1998/Math/MathML"
64
+ _XHTML_NS = "http://www.w3.org/1999/xhtml"
65
+ _XML_NS = "http://www.w3.org/XML/1998/namespace"
66
+ _STYLE_RESOURCE_NAME = "docvortex.css"
67
+ _INVALID_XML_TEXT_RE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f\ud800-\udfff]")
68
+ _MARKUP_TOKEN_RE = re.compile(
69
+ r"<\s*(?P<closing>/)?\s*(?P<name>[A-Za-z][A-Za-z0-9:-]*)\b(?P<attrs>[^>]*)>",
70
+ re.DOTALL,
71
+ )
72
+ _SAFE_LANGUAGE_RE = re.compile(r"[a-z0-9][a-z0-9-]{0,31}\Z")
73
+ _ALLOWED_MARKUP_TAGS = {
74
+ "a",
75
+ "b",
76
+ "blockquote",
77
+ "br",
78
+ "caption",
79
+ "code",
80
+ "col",
81
+ "colgroup",
82
+ "details",
83
+ "div",
84
+ "em",
85
+ "eq",
86
+ "i",
87
+ "img",
88
+ "kbd",
89
+ "li",
90
+ "mark",
91
+ "ol",
92
+ "p",
93
+ "pre",
94
+ "s",
95
+ "span",
96
+ "strong",
97
+ "sub",
98
+ "summary",
99
+ "summary",
100
+ "sup",
101
+ "table",
102
+ "tbody",
103
+ "td",
104
+ "tfoot",
105
+ "th",
106
+ "thead",
107
+ "tr",
108
+ "u",
109
+ "ul",
110
+ }
111
+ _DROP_CONTENT_TAGS = {
112
+ "audio",
113
+ "button",
114
+ "canvas",
115
+ "embed",
116
+ "form",
117
+ "head",
118
+ "iframe",
119
+ "input",
120
+ "math",
121
+ "noscript",
122
+ "object",
123
+ "script",
124
+ "select",
125
+ "style",
126
+ "svg",
127
+ "template",
128
+ "textarea",
129
+ "video",
130
+ }
131
+ _SOURCE_MARKUP_TAGS = _ALLOWED_MARKUP_TAGS | _DROP_CONTENT_TAGS
132
+ _VOID_MARKUP_TAGS = {"br", "col", "img"}
133
+ _PHRASING_MARKUP_TAGS = {
134
+ "a",
135
+ "b",
136
+ "code",
137
+ "em",
138
+ "i",
139
+ "kbd",
140
+ "mark",
141
+ "p",
142
+ "s",
143
+ "span",
144
+ "strong",
145
+ "sub",
146
+ "sup",
147
+ "u",
148
+ }
149
+ _BLOCK_MARKUP_TAGS = {"blockquote", "details", "div", "ol", "p", "pre", "table", "ul"}
150
+ _TABLE_PARENT_RULES = {
151
+ "caption": {"table"},
152
+ "col": {"colgroup"},
153
+ "colgroup": {"table"},
154
+ "tbody": {"table"},
155
+ "td": {"tr"},
156
+ "tfoot": {"table"},
157
+ "th": {"tr"},
158
+ "thead": {"table"},
159
+ "tr": {"table", "tbody", "tfoot", "thead"},
160
+ }
161
+
162
+
163
+ @dataclass(frozen=True, slots=True)
164
+ class _TitleTarget:
165
+ """保存正文标题的可见文本、层级与 XHTML 目标。"""
166
+
167
+ title: str
168
+ level: int
169
+ target_id: str
170
+
171
+
172
+ class _AnchorRegistry:
173
+ """为正文文本、标题和页面脚注分配文档级唯一 XHTML id。"""
174
+
175
+ def __init__(self, middle_json: MiddleJson) -> None:
176
+ """按页面与 block 顺序建立目标、来源 anchor 和标题索引。"""
177
+ self._block_targets: dict[tuple[int, int, str], str] = {}
178
+ self._anchor_targets: dict[str, str] = {}
179
+ self._footnote_targets: set[str] = set()
180
+ self.title_targets: list[_TitleTarget] = []
181
+ used_ids: set[str] = {"content-start"}
182
+ heading_position = 0
183
+ text_position = 0
184
+ footnote_position = 0
185
+ for page in middle_json.pages:
186
+ for block in page.blocks:
187
+ if isinstance(block, TitleBlockBase):
188
+ visible = inline_plain_text(block.content).strip()
189
+ if not visible:
190
+ continue
191
+ heading_position += 1
192
+ target_id = _allocate_target_id(
193
+ block.anchor,
194
+ fallback=f"heading-{heading_position}",
195
+ used_ids=used_ids,
196
+ )
197
+ self.title_targets.append(_TitleTarget(visible, block.level, target_id))
198
+ elif isinstance(block, TextBlock):
199
+ visible = inline_plain_text(block.content).strip()
200
+ if not visible or not (block.anchor or "").strip():
201
+ continue
202
+ text_position += 1
203
+ target_id = _allocate_target_id(
204
+ block.anchor,
205
+ fallback=f"text-{text_position}",
206
+ used_ids=used_ids,
207
+ )
208
+ elif isinstance(block, PageFootnoteBlock):
209
+ visible = inline_plain_text(block.content).strip()
210
+ if not visible:
211
+ continue
212
+ footnote_position += 1
213
+ target_id = _allocate_target_id(
214
+ block.anchor,
215
+ fallback=f"footnote-{footnote_position}",
216
+ used_ids=used_ids,
217
+ )
218
+ self._footnote_targets.add(target_id)
219
+ else:
220
+ continue
221
+ assert block.index is not None
222
+ self._block_targets[(page.page_idx, block.index, str(block.type))] = target_id
223
+ anchor_key = _anchor_key(block.anchor)
224
+ if anchor_key and anchor_key not in self._anchor_targets:
225
+ self._anchor_targets[anchor_key] = target_id
226
+
227
+ def target_for_block(self, page_idx: int, block: BlockBase) -> str | None:
228
+ """按来源页、index 和类型返回标题或脚注的唯一目标。"""
229
+ if block.index is None:
230
+ return None
231
+ return self._block_targets.get((page_idx, block.index, str(block.type)))
232
+
233
+ def target_for_anchor(self, anchor: str | None) -> str | None:
234
+ """按 producer anchor 返回首次匹配的正文目标。"""
235
+ return self._anchor_targets.get(_anchor_key(anchor))
236
+
237
+ def is_footnote_target(self, target_id: str) -> bool:
238
+ """判断目标是否对应页面脚注,以便标注 noteref 语义。"""
239
+ return target_id in self._footnote_targets
240
+
241
+
242
+ class _EpubXhtmlRenderer:
243
+ """维护单个 EPUB 正文的锚点、素材与 MathML 状态。"""
244
+
245
+ def __init__(
246
+ self,
247
+ middle_json: MiddleJson,
248
+ *,
249
+ metadata: EpubMetadata,
250
+ assets: EpubAssetRegistry,
251
+ anchors: _AnchorRegistry,
252
+ ) -> None:
253
+ """保存严格输入和已规范化的调用状态。"""
254
+ self.middle_json = middle_json
255
+ self.metadata = metadata
256
+ self.assets = assets
257
+ self.anchors = anchors
258
+ self.has_mathml = False
259
+
260
+ def render(self) -> bytes:
261
+ """把完整 render plan 写成一个无脚本 XHTML content document。"""
262
+ root = etree.Element(
263
+ _xhtml("html"),
264
+ nsmap={None: _XHTML_NS, "epub": _EPUB_NS},
265
+ attrib={f"{{{_XML_NS}}}lang": self.metadata.language, "lang": self.metadata.language},
266
+ )
267
+ head = etree.SubElement(root, _xhtml("head"))
268
+ etree.SubElement(head, _xhtml("meta"), charset="utf-8")
269
+ title = etree.SubElement(head, _xhtml("title"))
270
+ title.text = self.metadata.title
271
+ etree.SubElement(
272
+ head,
273
+ _xhtml("link"),
274
+ rel="stylesheet",
275
+ href="../styles/docvortex.css",
276
+ type="text/css",
277
+ )
278
+ body = etree.SubElement(root, _xhtml("body"), attrib={"class": "docvortex-epub-body"})
279
+ article = etree.SubElement(
280
+ body,
281
+ _xhtml("article"),
282
+ id="content-start",
283
+ attrib={"class": "docvortex-document"},
284
+ )
285
+ self._render_pages(article, build_render_plan(self.middle_json))
286
+ return etree.tostring(
287
+ root,
288
+ encoding="utf-8",
289
+ xml_declaration=True,
290
+ doctype="<!DOCTYPE html>",
291
+ )
292
+
293
+ def _render_pages(self, parent: etree._Element, pages: list[list[PlannedBlock]]) -> None:
294
+ """把默认计划展平到单个连续阅读容器。"""
295
+ for page in pages:
296
+ for planned in page:
297
+ rendered = self._render_planned_block(planned)
298
+ if rendered is not None:
299
+ parent.append(rendered)
300
+
301
+ def _render_planned_block(self, planned: PlannedBlock) -> etree._Element | None:
302
+ """过滤计划块、分派具体类型并追加稳定来源属性。"""
303
+ if planned.removed:
304
+ return None
305
+ block = planned.block
306
+ if block.type in PAGE_AUXILIARY_BLOCK_TYPES:
307
+ return None
308
+ content = self._render_block_content(planned)
309
+ if content is None:
310
+ return None
311
+ wrapper = etree.Element(
312
+ _xhtml("div"),
313
+ attrib={
314
+ "class": "docvortex-block",
315
+ "data-page-idx": str(planned.page_idx),
316
+ "data-block-type": str(block.type),
317
+ },
318
+ )
319
+ if block.index is not None:
320
+ wrapper.set("data-block-index", str(block.index))
321
+ wrapper.append(content)
322
+ return wrapper
323
+
324
+ def _render_block_content(self, planned: PlannedBlock) -> etree._Element | None:
325
+ """把一个具体 PageBlock 映射为静态 XHTML 元素。"""
326
+ block = planned.block
327
+ if isinstance(block, (TextBlock, RefTextBlock)):
328
+ content = join_inline_spans(planned.text_contents or [block.content])
329
+ paragraph = etree.Element(
330
+ _xhtml("p"),
331
+ attrib={"class": "docvortex-ref-text" if isinstance(block, RefTextBlock) else "docvortex-text"},
332
+ )
333
+ if isinstance(block, TextBlock):
334
+ target_id = self.anchors.target_for_block(planned.page_idx, block)
335
+ if target_id:
336
+ paragraph.set("id", target_id)
337
+ self._append_inline_spans(paragraph, content)
338
+ return paragraph if _has_visible_content(paragraph) else None
339
+ if isinstance(block, (DocTitleBlock, ParagraphTitleBlock)):
340
+ return self._render_title(planned.page_idx, block)
341
+ if isinstance(block, PageFootnoteBlock):
342
+ return self._render_page_footnote(planned.page_idx, block)
343
+ if isinstance(block, EquationBlock):
344
+ return self._render_equation(block)
345
+ if isinstance(block, ListBlock):
346
+ return self._render_list(block)
347
+ if isinstance(block, IndexBlock):
348
+ return self._render_index(block)
349
+ if isinstance(block, ImageBlock):
350
+ return self._render_image_block(block)
351
+ if isinstance(block, TableBlock):
352
+ return self._render_table_block(block)
353
+ if isinstance(block, ChartBlock):
354
+ return self._render_chart_block(block)
355
+ if isinstance(block, CodeBlock):
356
+ return self._render_code_block(block)
357
+ raise TypeError(f"Unsupported PageBlock type: {type(block).__name__}")
358
+
359
+ def _render_title(self, page_idx: int, block: TitleBlockBase) -> etree._Element | None:
360
+ """渲染带文档级唯一 id 的 h1-h6 标题。"""
361
+ if not inline_plain_text(block.content).strip():
362
+ return None
363
+ level = min(max(block.level, 1), 6)
364
+ heading = etree.Element(_xhtml(f"h{level}"), attrib={"class": f"docvortex-heading docvortex-heading--{level}"})
365
+ target_id = self.anchors.target_for_block(page_idx, block)
366
+ if target_id:
367
+ heading.set("id", target_id)
368
+ self._append_inline_spans(heading, block.content)
369
+ return heading
370
+
371
+ def _render_page_footnote(self, page_idx: int, block: PageFootnoteBlock) -> etree._Element | None:
372
+ """把页面脚注保留为 EPUB footnote aside。"""
373
+ footnote = etree.Element(
374
+ _xhtml("aside"),
375
+ attrib={"class": "docvortex-page-footnote", f"{{{_EPUB_NS}}}type": "footnote", "role": "doc-footnote"},
376
+ )
377
+ target_id = self.anchors.target_for_block(page_idx, block)
378
+ if target_id:
379
+ footnote.set("id", target_id)
380
+ self._append_inline_spans(footnote, block.content)
381
+ return footnote if _has_visible_content(footnote) else None
382
+
383
+ def _render_equation(self, block: EquationBlock) -> etree._Element | None:
384
+ """优先渲染行间 MathML,空公式时才尝试包内图片。"""
385
+ container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-equation"})
386
+ if block.content.strip():
387
+ self._append_math(container, block.content, display="block")
388
+ elif source := self.assets.resolve_block(block):
389
+ etree.SubElement(container, _xhtml("img"), src=source, alt="formula")
390
+ return container if _has_visible_content(container) else None
391
+
392
+ def _render_list(self, block: ListBlock) -> etree._Element | None:
393
+ """按共享 marker 语义递归渲染原生有序、无序或显式 marker 列表。"""
394
+ parsed_leaves = [
395
+ parse_list_item_marker(child.content)
396
+ for child in block.content
397
+ if not isinstance(child, ListBlock) and inline_plain_text(child.content).strip()
398
+ ]
399
+ add_reference_bullets = reference_list_needs_bullets(block)
400
+ container_tag, list_type, class_name = _classify_list(parsed_leaves, add_reference_bullets)
401
+ container = etree.Element(_xhtml(container_tag), attrib={"class": f"docvortex-list {class_name}"})
402
+ if list_type:
403
+ container.set("type", list_type)
404
+ if container_tag == "ol" and parsed_leaves and parsed_leaves[0].value not in (None, 1):
405
+ container.set("start", str(parsed_leaves[0].value))
406
+ expected_value: int | None = None
407
+ last_item: etree._Element | None = None
408
+ for child in block.content:
409
+ if isinstance(child, ListBlock):
410
+ nested = self._render_list(child)
411
+ if nested is None:
412
+ continue
413
+ if last_item is None:
414
+ last_item = etree.SubElement(container, _xhtml("li"), attrib={"class": "docvortex-list-item--orphan"})
415
+ last_item.append(nested)
416
+ continue
417
+ parsed = parse_list_item_marker(child.content)
418
+ item_content, marker = _list_item_content(
419
+ parsed,
420
+ add_reference_bullets,
421
+ explicit_markers=class_name == "docvortex-list--explicit",
422
+ )
423
+ item = etree.SubElement(container, _xhtml("li"))
424
+ if class_name == "docvortex-list--explicit":
425
+ item.set("class", "docvortex-list-item--explicit")
426
+ if container_tag == "ol" and parsed.kind == "ordered" and parsed.value is not None:
427
+ if expected_value is None:
428
+ expected_value = parsed.value
429
+ if parsed.value != expected_value:
430
+ item.set("value", str(parsed.value))
431
+ expected_value = parsed.value + 1
432
+ if marker or class_name == "docvortex-list--explicit":
433
+ marker_element = etree.SubElement(item, _xhtml("span"), attrib={"class": "docvortex-list-marker"})
434
+ marker_element.text = marker or ""
435
+ content_element = etree.SubElement(item, _xhtml("span"), attrib={"class": "docvortex-list-content"})
436
+ self._append_inline_spans(content_element, item_content)
437
+ last_item = item
438
+ return container if len(container) else None
439
+
440
+ def _render_index(self, block: IndexBlock) -> etree._Element | None:
441
+ """把源目录保留为正文内导航,并只链接到真实正文目标。"""
442
+ navigation = etree.Element(_xhtml("nav"), attrib={"class": "docvortex-index", "aria-label": "Table of contents"})
443
+ listing = etree.SubElement(navigation, _xhtml("ul"))
444
+ self._append_index_children(listing, block)
445
+ return navigation if len(listing) else None
446
+
447
+ def _append_index_children(self, parent: etree._Element, block: IndexBlock) -> None:
448
+ """递归渲染 IndexBlock,并把孤立嵌套目录提升到当前层级。"""
449
+ last_item: etree._Element | None = None
450
+ for child in block.content:
451
+ if isinstance(child, IndexBlock):
452
+ nested = etree.Element(_xhtml("ul"))
453
+ self._append_index_children(nested, child)
454
+ if not len(nested):
455
+ continue
456
+ if last_item is None:
457
+ last_item = etree.SubElement(parent, _xhtml("li"), attrib={"class": "docvortex-list-item--orphan"})
458
+ last_item.append(nested)
459
+ continue
460
+ content = strip_index_page_tail(child.content)
461
+ if not inline_plain_text(content).strip():
462
+ continue
463
+ item = etree.SubElement(parent, _xhtml("li"))
464
+ target = self.anchors.target_for_anchor(child.anchor)
465
+ inline_parent = item
466
+ if target:
467
+ inline_parent = etree.SubElement(item, _xhtml("a"), href=f"#{quote(target, safe='-._~')}")
468
+ self._append_inline_spans(inline_parent, content)
469
+ last_item = item
470
+
471
+ def _render_image_block(self, block: ImageBlock) -> etree._Element | None:
472
+ """按子块顺序渲染图片主体及其标题、脚注。"""
473
+ figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--image"})
474
+ for child in block.content:
475
+ rendered = (
476
+ self._render_image_body(block, child) if isinstance(child, ImageBodyBlock) else self._render_annotation(child)
477
+ )
478
+ if rendered is not None:
479
+ figure.append(rendered)
480
+ return figure if len(figure) else None
481
+
482
+ def _render_image_body(self, parent: ImageBlock, block: ImageBodyBlock) -> etree._Element | None:
483
+ """渲染包内图片,并在缺图时保留已有结构或可见文字。"""
484
+ container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--image"})
485
+ source = self.assets.resolve_block(block)
486
+ if source:
487
+ alt = _plain_content_text(block.content) or parent.sub_type or "image"
488
+ etree.SubElement(container, _xhtml("img"), src=source, alt=alt, attrib={"class": "docvortex-image"})
489
+ if block.content.strip():
490
+ content = etree.Element(_xhtml("div"), attrib={"class": "docvortex-image-content"})
491
+ self._append_rich_or_text(content, block.content)
492
+ if _has_visible_content(content):
493
+ container.append(content)
494
+ return container if _has_visible_content(container) else None
495
+
496
+ def _render_table_block(self, block: TableBlock) -> etree._Element | None:
497
+ """按子块顺序渲染结构表格、图片回退及说明。"""
498
+ figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--table"})
499
+ for child in block.content:
500
+ rendered = self._render_table_body(child) if isinstance(child, TableBodyBlock) else self._render_annotation(child)
501
+ if rendered is not None:
502
+ figure.append(rendered)
503
+ return figure if len(figure) else None
504
+
505
+ def _render_table_body(self, block: TableBodyBlock) -> etree._Element | None:
506
+ """优先输出安全结构内容,无内容时尝试整体表格图片。"""
507
+ container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--table"})
508
+ if block.content.strip():
509
+ if _is_supported_markup(block.content):
510
+ self._append_markup(container, block.content)
511
+ else:
512
+ pre = etree.SubElement(container, _xhtml("pre"), attrib={"class": "docvortex-table-text"})
513
+ pre.text = _normalize_xml_text(block.content)
514
+ if not _has_visible_content(container) and (source := self.assets.resolve_block(block)):
515
+ etree.SubElement(container, _xhtml("img"), src=source, alt="table", attrib={"class": "docvortex-table-image"})
516
+ return container if _has_visible_content(container) else None
517
+
518
+ def _render_chart_block(self, block: ChartBlock) -> etree._Element | None:
519
+ """按子块顺序渲染图表图片、结构内容及说明。"""
520
+ figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--chart"})
521
+ for child in block.content:
522
+ rendered = (
523
+ self._render_chart_body(block, child) if isinstance(child, ChartBodyBlock) else self._render_annotation(child)
524
+ )
525
+ if rendered is not None:
526
+ figure.append(rendered)
527
+ return figure if len(figure) else None
528
+
529
+ def _render_chart_body(self, parent: ChartBlock, block: ChartBodyBlock) -> etree._Element | None:
530
+ """渲染包内图表图片,并始终保留并存结构内容。"""
531
+ container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--chart"})
532
+ if source := self.assets.resolve_block(block):
533
+ etree.SubElement(
534
+ container,
535
+ _xhtml("img"),
536
+ src=source,
537
+ alt=parent.sub_type or "chart",
538
+ attrib={"class": "docvortex-chart-image"},
539
+ )
540
+ if block.content.strip():
541
+ content = etree.Element(_xhtml("div"), attrib={"class": "docvortex-chart-content"})
542
+ self._append_rich_or_text(content, block.content, preformatted=True)
543
+ if _has_visible_content(content):
544
+ container.append(content)
545
+ return container if _has_visible_content(container) else None
546
+
547
+ def _render_code_block(self, block: CodeBlock) -> etree._Element | None:
548
+ """按子块顺序渲染静态代码、算法及其说明。"""
549
+ figure = etree.Element(_xhtml("figure"), attrib={"class": "docvortex-figure docvortex-figure--code"})
550
+ for child in block.content:
551
+ if isinstance(child, (CodeBodyBlock, AlgorithmBodyBlock)):
552
+ rendered = self._render_code_body(block, child)
553
+ else:
554
+ rendered = self._render_annotation(child)
555
+ if rendered is not None:
556
+ figure.append(rendered)
557
+ return figure if len(figure) else None
558
+
559
+ def _render_code_body(self, parent: CodeBlock, block: CodeBodyBlock | AlgorithmBodyBlock) -> etree._Element:
560
+ """代码使用 pre/code,算法使用保留换行的结构化 Span。"""
561
+ container = etree.Element(_xhtml("div"), attrib={"class": "docvortex-visual-body docvortex-visual-body--code"})
562
+ if parent.sub_type == BlockType.CODE:
563
+ if not isinstance(block, CodeBodyBlock):
564
+ raise TypeError("code subtype requires CodeBodyBlock")
565
+ pre = etree.SubElement(container, _xhtml("pre"), attrib={"class": "docvortex-code"})
566
+ code = etree.SubElement(pre, _xhtml("code"))
567
+ language = _normalize_code_language(parent.guess_lang)
568
+ if language:
569
+ code.set("class", f"language-{language}")
570
+ code.text = _normalize_xml_text(block.content)
571
+ return container
572
+ if parent.sub_type == RAW_ALGORITHM:
573
+ if not isinstance(block, AlgorithmBodyBlock):
574
+ raise TypeError("algorithm subtype requires AlgorithmBodyBlock")
575
+ algorithm = etree.SubElement(container, _xhtml("div"), attrib={"class": "docvortex-algorithm"})
576
+ self._append_inline_spans(algorithm, block.content, preserve_newlines=True, separate_adjacent_math=True)
577
+ return container
578
+ raise ValueError(f"Unsupported code subtype: {parent.sub_type}")
579
+
580
+ def _render_annotation(
581
+ self,
582
+ block: ImageAnnotationBlock | TableAnnotationBlock | ChartAnnotationBlock | CodeAnnotationBlock,
583
+ ) -> etree._Element | None:
584
+ """按 caption 或 footnote 语义渲染视觉说明。"""
585
+ role = "docvortex-caption" if str(block.type).endswith("caption") else "docvortex-footnote"
586
+ annotation = etree.Element(
587
+ _xhtml("p"),
588
+ attrib={"class": f"{role} {role}--{str(block.type).replace('_', '-')}"},
589
+ )
590
+ self._append_inline_spans(annotation, block.content)
591
+ return annotation if _has_visible_content(annotation) else None
592
+
593
+ def _append_inline_spans(
594
+ self,
595
+ parent: etree._Element,
596
+ spans: list[InlineSpan],
597
+ *,
598
+ preserve_newlines: bool = False,
599
+ separate_adjacent_math: bool = False,
600
+ ) -> None:
601
+ """按结构化 Span 顺序向 XHTML mixed content 追加安全节点。"""
602
+ previous_was_math = False
603
+ for span in spans:
604
+ current_is_math = isinstance(span, EquationInlineSpan)
605
+ if separate_adjacent_math and previous_was_math and current_is_math:
606
+ _append_text(parent, " ", preserve_newlines=True)
607
+ self._append_inline_span(parent, span, preserve_newlines=preserve_newlines)
608
+ previous_was_math = current_is_math
609
+
610
+ def _append_inline_span(self, parent: etree._Element, span: InlineSpan, *, preserve_newlines: bool) -> None:
611
+ """把单个 Text/Code/Equation/Hyperlink Span 追加到父节点。"""
612
+ if isinstance(span, TextSpan):
613
+ target = _append_text_style_container(parent, span)
614
+ _append_text(target, span.content, preserve_newlines=preserve_newlines)
615
+ return
616
+ if isinstance(span, CodeInlineSpan):
617
+ code = etree.SubElement(parent, _xhtml("code"))
618
+ _append_text(code, span.content, preserve_newlines=preserve_newlines)
619
+ return
620
+ if isinstance(span, EquationInlineSpan):
621
+ self._append_math(parent, span.content, display="inline")
622
+ return
623
+ if isinstance(span, HyperlinkSpan):
624
+ href, target_id = self._resolve_link(span.url)
625
+ link_parent = parent
626
+ if href:
627
+ link_parent = etree.SubElement(parent, _xhtml("a"), href=href)
628
+ if target_id and self.anchors.is_footnote_target(target_id):
629
+ link_parent.set(f"{{{_EPUB_NS}}}type", "noteref")
630
+ link_parent.set("role", "doc-noteref")
631
+ self._append_inline_spans(link_parent, list(span.content), preserve_newlines=preserve_newlines)
632
+ return
633
+ raise TypeError(f"Unsupported inline span: {type(span).__name__}")
634
+
635
+ def _append_math(self, parent: etree._Element, latex: str, *, display: str) -> None:
636
+ """追加 Presentation MathML,并在转换失败时显示原始 LaTeX。"""
637
+ normalized = latex.strip()
638
+ if not normalized:
639
+ return
640
+ try:
641
+ markup = latex_to_mathml(normalized, display=display)
642
+ parser = etree.XMLParser(resolve_entities=False, load_dtd=False, no_network=True, recover=False, huge_tree=False)
643
+ math = etree.fromstring(markup.encode("utf-8"), parser=parser)
644
+ if math.tag != f"{{{_MATHML_NS}}}math":
645
+ raise ValueError("latex2mathml did not return a MathML root")
646
+ except Exception:
647
+ fallback = etree.SubElement(
648
+ parent,
649
+ _xhtml("code"),
650
+ attrib={"class": f"docvortex-latex-fallback docvortex-latex-fallback--{display}"},
651
+ )
652
+ fallback.text = _normalize_xml_text(normalized)
653
+ return
654
+ parent.append(math)
655
+ self.has_mathml = True
656
+
657
+ def _append_rich_or_text(self, parent: etree._Element, content: str, *, preformatted: bool = False) -> None:
658
+ """识别安全富 HTML,否则按普通文本或预格式文本输出。"""
659
+ if _is_supported_markup(content):
660
+ self._append_markup(parent, content)
661
+ return
662
+ if preformatted:
663
+ pre = etree.SubElement(parent, _xhtml("pre"))
664
+ pre.text = _normalize_xml_text(content)
665
+ else:
666
+ _append_text(parent, content)
667
+
668
+ def _append_markup(self, parent: etree._Element, markup: str) -> None:
669
+ """通过 EPUB 专用 allowlist 把不可信 HTML 转为安全 XHTML 节点。"""
670
+ soup = BeautifulSoup(_normalize_xml_text(markup), "html.parser")
671
+ for child in list(soup.contents):
672
+ self._append_soup_node(parent, child)
673
+
674
+ def _append_soup_node(self, parent: etree._Element, node: object) -> None:
675
+ """递归复制一个 BeautifulSoup 节点,仅创建允许的 XHTML 结构。"""
676
+ if isinstance(node, (Comment, Doctype, ProcessingInstruction)):
677
+ return
678
+ if isinstance(node, NavigableString):
679
+ _append_text(parent, str(node), preserve_newlines=True)
680
+ return
681
+ if not isinstance(node, Tag):
682
+ return
683
+ name = (node.name or "").lower()
684
+ if name in _DROP_CONTENT_TAGS:
685
+ return
686
+ if name not in _ALLOWED_MARKUP_TAGS:
687
+ for child in list(node.children):
688
+ self._append_soup_node(parent, child)
689
+ return
690
+ parent_name = etree.QName(parent).localname
691
+ if name in _TABLE_PARENT_RULES and parent_name not in _TABLE_PARENT_RULES[name]:
692
+ for child in list(node.children):
693
+ self._append_soup_node(parent, child)
694
+ return
695
+ if parent_name in _PHRASING_MARKUP_TAGS and name in _BLOCK_MARKUP_TAGS:
696
+ for child in list(node.children):
697
+ self._append_soup_node(parent, child)
698
+ return
699
+ if name == "a" and parent_name == "a":
700
+ for child in list(node.children):
701
+ self._append_soup_node(parent, child)
702
+ return
703
+ if name == "li" and parent_name not in {"ol", "ul"}:
704
+ if parent_name in _PHRASING_MARKUP_TAGS:
705
+ for child in list(node.children):
706
+ self._append_soup_node(parent, child)
707
+ return
708
+ listing = etree.SubElement(parent, _xhtml("ul"))
709
+ item = etree.SubElement(listing, _xhtml("li"), attrib=_safe_markup_attributes(name, node))
710
+ for child in list(node.children):
711
+ self._append_soup_node(item, child)
712
+ return
713
+ if name in {"ol", "ul"}:
714
+ listing = etree.SubElement(parent, _xhtml(name), attrib=_safe_markup_attributes(name, node))
715
+ for child in list(node.children):
716
+ if isinstance(child, NavigableString) and not str(child).strip():
717
+ continue
718
+ if isinstance(child, Tag) and (child.name or "").lower() == "li":
719
+ self._append_soup_node(listing, child)
720
+ continue
721
+ item = etree.SubElement(listing, _xhtml("li"))
722
+ self._append_soup_node(item, child)
723
+ return
724
+ if name == "eq":
725
+ self._append_math(parent, node.get_text(), display="inline")
726
+ return
727
+ if name == "img":
728
+ source = self.assets.resolve_embedded_source(_attribute_text(node.get("src")))
729
+ alt = _attribute_text(node.get("alt"))
730
+ if source:
731
+ image = etree.SubElement(parent, _xhtml("img"), src=source, alt=_normalize_xml_text(alt))
732
+ title = _attribute_text(node.get("title"))
733
+ if title:
734
+ image.set("title", _normalize_xml_text(title))
735
+ elif alt:
736
+ _append_text(parent, alt)
737
+ return
738
+ if name == "a":
739
+ href, target_id = self._resolve_link(_attribute_text(node.get("href")))
740
+ target_parent = parent
741
+ if href:
742
+ target_parent = etree.SubElement(parent, _xhtml("a"), href=href)
743
+ title = _attribute_text(node.get("title"))
744
+ if title:
745
+ target_parent.set("title", _normalize_xml_text(title))
746
+ if target_id and self.anchors.is_footnote_target(target_id):
747
+ target_parent.set(f"{{{_EPUB_NS}}}type", "noteref")
748
+ target_parent.set("role", "doc-noteref")
749
+ for child in list(node.children):
750
+ self._append_soup_node(target_parent, child)
751
+ return
752
+ attributes = _safe_markup_attributes(name, node)
753
+ if name == "colgroup" and any(isinstance(child, Tag) and child.name == "col" for child in node.children):
754
+ attributes.pop("span", None)
755
+ element = etree.SubElement(parent, _xhtml(name), attrib=attributes)
756
+ if name not in _VOID_MARKUP_TAGS:
757
+ for child in list(node.children):
758
+ self._append_soup_node(element, child)
759
+
760
+ def _resolve_link(self, url: str) -> tuple[str | None, str | None]:
761
+ """保留安全外链或已登记 fragment,删除无包内目标的相对链接。"""
762
+ normalized = _normalize_xml_text(url).strip()
763
+ if not normalized or normalized.startswith(("//", "\\")):
764
+ return None, None
765
+ if normalized.startswith("#"):
766
+ target = self.anchors.target_for_anchor(unquote(normalized[1:]))
767
+ if target:
768
+ return f"#{quote(target, safe='-._~')}", target
769
+ return None, None
770
+ try:
771
+ parsed = urlsplit(normalized)
772
+ _ = parsed.port
773
+ except ValueError:
774
+ return None, None
775
+ scheme = parsed.scheme.casefold()
776
+ if scheme in {"http", "https"}:
777
+ if not parsed.netloc or parsed.hostname is None or parsed.username is not None or parsed.password is not None:
778
+ return None, None
779
+ elif scheme in {"mailto", "tel"}:
780
+ if not parsed.path:
781
+ return None, None
782
+ else:
783
+ return None, None
784
+ return quote(normalized, safe="/:#?&=%@+~,;!$'*-._"), None
785
+
786
+
787
+ def render_epub(
788
+ middle_json: MiddleJson,
789
+ *,
790
+ title: str | None = None,
791
+ authors: tuple[str, ...] = (),
792
+ language: str = "und",
793
+ identifier: str | None = None,
794
+ modified_at: datetime | None = None,
795
+ asset_resolver: AssetResolver | None = None,
796
+ ) -> bytes:
797
+ """把严格 MiddleJson 无副作用地渲染为单正文 EPUB 3.3 字节。"""
798
+ if not isinstance(middle_json, MiddleJson):
799
+ raise TypeError("render_epub expects a MiddleJson instance")
800
+ options = EpubRenderOptions(
801
+ title=title,
802
+ authors=authors,
803
+ language=language,
804
+ identifier=identifier,
805
+ modified_at=modified_at,
806
+ asset_resolver=asset_resolver,
807
+ )
808
+ resolved_title = _resolve_document_title(middle_json, options.title)
809
+ resolved_authors = tuple(_normalize_xml_text(author).strip() for author in options.authors)
810
+ resolved_language = options.language.strip()
811
+ resolved_identifier = (
812
+ options.identifier.strip()
813
+ if options.identifier
814
+ else _stable_identifier(
815
+ middle_json,
816
+ title=resolved_title,
817
+ authors=resolved_authors,
818
+ language=resolved_language,
819
+ )
820
+ )
821
+ resolved_modified = (options.modified_at or datetime.now(timezone.utc)).astimezone(timezone.utc).replace(microsecond=0)
822
+ if resolved_modified.year < 1000:
823
+ raise ValueError("modified_at UTC year must use four digits")
824
+ metadata = EpubMetadata(
825
+ title=resolved_title,
826
+ authors=resolved_authors,
827
+ language=resolved_language,
828
+ identifier=_normalize_xml_text(resolved_identifier),
829
+ modified_at=resolved_modified,
830
+ )
831
+ anchors = _AnchorRegistry(middle_json)
832
+ assets = EpubAssetRegistry(options.asset_resolver)
833
+ renderer = _EpubXhtmlRenderer(
834
+ middle_json,
835
+ metadata=metadata,
836
+ assets=assets,
837
+ anchors=anchors,
838
+ )
839
+ content_xhtml = renderer.render()
840
+ navigation = _build_navigation(middle_json, anchors, resolved_title)
841
+ return build_epub_package(
842
+ metadata=metadata,
843
+ content_xhtml=content_xhtml,
844
+ navigation=navigation,
845
+ stylesheet=_load_epub_stylesheet(),
846
+ assets=assets.assets,
847
+ has_mathml=renderer.has_mathml,
848
+ )
849
+
850
+
851
+ def _build_navigation(middle_json: MiddleJson, anchors: _AnchorRegistry, document_title: str) -> list[NavigationItem]:
852
+ """优先使用有效 IndexBlock,否则按标题层级或正文起点生成 toc。"""
853
+ for page in middle_json.pages:
854
+ for block in page.blocks:
855
+ if not isinstance(block, IndexBlock):
856
+ continue
857
+ items = _navigation_from_index(block, anchors)
858
+ if items:
859
+ return items
860
+ if anchors.title_targets:
861
+ roots: list[NavigationItem] = []
862
+ stack: list[tuple[int, NavigationItem]] = []
863
+ for target in anchors.title_targets:
864
+ item = NavigationItem(
865
+ title=target.title,
866
+ href=f"text/content.xhtml#{quote(target.target_id, safe='-._~')}",
867
+ )
868
+ while stack and stack[-1][0] >= target.level:
869
+ stack.pop()
870
+ if stack:
871
+ stack[-1][1].children.append(item)
872
+ else:
873
+ roots.append(item)
874
+ stack.append((target.level, item))
875
+ return roots
876
+ return [NavigationItem(title=document_title, href="text/content.xhtml#content-start")]
877
+
878
+
879
+ def _navigation_from_index(block: IndexBlock, anchors: _AnchorRegistry) -> list[NavigationItem]:
880
+ """从一个 IndexBlock 提取仅包含真实标题目标的层级导航。"""
881
+ result: list[NavigationItem] = []
882
+ last_item: NavigationItem | None = None
883
+ for child in block.content:
884
+ if isinstance(child, IndexBlock):
885
+ nested = _navigation_from_index(child, anchors)
886
+ if last_item is not None:
887
+ last_item.children.extend(nested)
888
+ else:
889
+ result.extend(nested)
890
+ continue
891
+ target = anchors.target_for_anchor(child.anchor)
892
+ title = inline_plain_text(strip_index_page_tail(child.content)).strip()
893
+ if not target or not title:
894
+ continue
895
+ last_item = NavigationItem(
896
+ title=title,
897
+ href=f"text/content.xhtml#{quote(target, safe='-._~')}",
898
+ )
899
+ result.append(last_item)
900
+ return result
901
+
902
+
903
+ def _classify_list(items: list[ListItem], add_reference_bullets: bool) -> tuple[str, str | None, str]:
904
+ """根据直属 marker 选择原生列表类型或显式 marker 模式。"""
905
+ if add_reference_bullets:
906
+ return "ul", None, "docvortex-list--reference"
907
+ if items and all(item.kind == "unordered" for item in items):
908
+ return "ul", None, "docvortex-list--unordered"
909
+ if items and all(item.kind == "ordered" for item in items):
910
+ styles = {item.ordered_style for item in items}
911
+ if len(styles) == 1:
912
+ list_type = {
913
+ "lower-alpha": "a",
914
+ "upper-alpha": "A",
915
+ "lower-roman": "i",
916
+ "upper-roman": "I",
917
+ }.get(next(iter(styles)) or "")
918
+ return "ol", list_type, "docvortex-list--ordered"
919
+ if items and all(item.kind == "none" for item in items):
920
+ return "ul", None, "docvortex-list--unmarked"
921
+ return "ul", None, "docvortex-list--explicit"
922
+
923
+
924
+ def _list_item_content(
925
+ item: ListItem,
926
+ add_reference_bullets: bool,
927
+ *,
928
+ explicit_markers: bool,
929
+ ) -> tuple[list[InlineSpan], str | None]:
930
+ """决定列表项应剥离、保留还是显式显示源 marker。"""
931
+ if add_reference_bullets:
932
+ if item.kind == "unordered":
933
+ return item.body, None
934
+ prefix = f"{item.leading}{item.marker or ''}{item.separator}"
935
+ original = normalize_inline_spans([TextSpan(type="text", content=prefix), *item.body]) if prefix else item.body
936
+ return original, None
937
+ if explicit_markers:
938
+ return item.body, item.marker
939
+ if item.kind in {"unordered", "ordered"}:
940
+ return item.body, None
941
+ return item.body, item.marker
942
+
943
+
944
+ def _append_text_style_container(parent: etree._Element, span: TextSpan) -> etree._Element:
945
+ """按固定样式顺序创建 TextSpan 的 XHTML 包装节点。"""
946
+ target = parent
947
+ if _needs_whitespace_preservation(span.content):
948
+ target = etree.SubElement(target, _xhtml("span"), attrib={"class": "docvortex-preserve-whitespace"})
949
+ wrappers: list[tuple[str, dict[str, str]]] = []
950
+ if "emphasis" in span.styles:
951
+ wrappers.append(("span", {"class": "docvortex-text-emphasis"}))
952
+ if "strikethrough" in span.styles:
953
+ wrappers.append(("s", {}))
954
+ if "italic" in span.styles:
955
+ wrappers.append(("em", {}))
956
+ if "bold" in span.styles:
957
+ wrappers.append(("strong", {}))
958
+ if "underline" in span.styles:
959
+ wrappers.append(("u", {}))
960
+ if "superscript" in span.styles:
961
+ wrappers.append(("sup", {}))
962
+ elif "subscript" in span.styles:
963
+ wrappers.append(("sub", {}))
964
+ for tag, attributes in wrappers:
965
+ target = etree.SubElement(target, _xhtml(tag), attrib=attributes)
966
+ return target
967
+
968
+
969
+ def _append_text(parent: etree._Element, content: str, *, preserve_newlines: bool = False) -> None:
970
+ """向 mixed content 追加安全文本,并按需把换行转换为 br。"""
971
+ normalized = _normalize_xml_text(content).replace("\r\n", "\n").replace("\r", "\n")
972
+ if preserve_newlines:
973
+ _append_raw_text(parent, normalized)
974
+ return
975
+ parts = normalized.split("\n")
976
+ for position, part in enumerate(parts):
977
+ if position:
978
+ parent.append(etree.Element(_xhtml("br")))
979
+ _append_raw_text(parent, part)
980
+
981
+
982
+ def _append_raw_text(parent: etree._Element, content: str) -> None:
983
+ """在不破坏既有子节点 tail 的前提下追加一段普通文本。"""
984
+ if not content:
985
+ return
986
+ if len(parent):
987
+ child = parent[-1]
988
+ child.tail = f"{child.tail or ''}{content}"
989
+ else:
990
+ parent.text = f"{parent.text or ''}{content}"
991
+
992
+
993
+ def _safe_markup_attributes(name: str, tag: Tag) -> dict[str, str]:
994
+ """只保留表格和列表语义需要的有界属性。"""
995
+ attributes: dict[str, str] = {}
996
+ if name in {"td", "th"}:
997
+ for attribute in ("colspan", "rowspan"):
998
+ if value := _bounded_integer(_attribute_text(tag.get(attribute)), minimum=1, maximum=1000):
999
+ attributes[attribute] = value
1000
+ scope = _attribute_text(tag.get("scope"))
1001
+ if name == "th" and scope in {"col", "colgroup", "row", "rowgroup"}:
1002
+ attributes["scope"] = scope
1003
+ elif name in {"col", "colgroup"}:
1004
+ if value := _bounded_integer(_attribute_text(tag.get("span")), minimum=1, maximum=1000):
1005
+ attributes["span"] = value
1006
+ elif name == "ol":
1007
+ if value := _bounded_integer(_attribute_text(tag.get("start")), minimum=-1_000_000, maximum=1_000_000):
1008
+ attributes["start"] = value
1009
+ elif name == "li":
1010
+ if value := _bounded_integer(_attribute_text(tag.get("value")), minimum=-1_000_000, maximum=1_000_000):
1011
+ attributes["value"] = value
1012
+ return attributes
1013
+
1014
+
1015
+ def _bounded_integer(value: str, *, minimum: int, maximum: int) -> str | None:
1016
+ """把十进制属性约束到 EPUB renderer 支持的闭区间。"""
1017
+ if re.fullmatch(r"[+-]?\d+", value) is None:
1018
+ return None
1019
+ number = int(value)
1020
+ return str(number) if minimum <= number <= maximum else None
1021
+
1022
+
1023
+ def _attribute_text(value: object) -> str:
1024
+ """把 BeautifulSoup 属性值稳定转换为普通字符串。"""
1025
+ if value is None:
1026
+ return ""
1027
+ if isinstance(value, list):
1028
+ return " ".join(str(item) for item in value)
1029
+ return str(value)
1030
+
1031
+
1032
+ def _is_supported_markup(content: str) -> bool:
1033
+ """仅把白名单或需整段删除的活动标签识别为富 HTML。"""
1034
+ if "<" not in content or ">" not in content:
1035
+ return False
1036
+ tokens = list(_MARKUP_TOKEN_RE.finditer(content))
1037
+ closing_names = {
1038
+ match.group("name").lower()
1039
+ for match in tokens
1040
+ if match.group("closing") and match.group("name").lower() in _SOURCE_MARKUP_TAGS
1041
+ }
1042
+ for match in tokens:
1043
+ if match.group("closing"):
1044
+ continue
1045
+ name = match.group("name").lower()
1046
+ if name not in _SOURCE_MARKUP_TAGS:
1047
+ continue
1048
+ if name in _VOID_MARKUP_TAGS or name in closing_names:
1049
+ return True
1050
+ if name in {"img", "embed"} and re.search(r"\bsrc\s*=", match.group("attrs"), re.IGNORECASE):
1051
+ return True
1052
+ return False
1053
+
1054
+
1055
+ def _resolve_document_title(middle_json: MiddleJson, explicit_title: str | None) -> str:
1056
+ """按显式值、首个文档标题和固定回退值解析书名。"""
1057
+ if explicit_title:
1058
+ return _normalize_xml_text(explicit_title).strip()
1059
+ for page in middle_json.pages:
1060
+ for block in page.blocks:
1061
+ if isinstance(block, DocTitleBlock):
1062
+ title = _normalize_xml_text(inline_plain_text(block.content)).strip()
1063
+ if title:
1064
+ return title
1065
+ return "DocVortex Document"
1066
+
1067
+
1068
+ def _stable_identifier(middle_json: MiddleJson, *, title: str, authors: tuple[str, ...], language: str) -> str:
1069
+ """由规范化 MiddleJson 和不随渲染时间变化的元数据生成稳定 UUID URN。"""
1070
+ seed = json.dumps(
1071
+ {
1072
+ "middle_json": middle_json.model_dump(mode="json"),
1073
+ "title": title,
1074
+ "authors": authors,
1075
+ "language": language,
1076
+ },
1077
+ ensure_ascii=False,
1078
+ sort_keys=True,
1079
+ separators=(",", ":"),
1080
+ ).encode("utf-8")
1081
+ digest = hashlib.sha256(seed).hexdigest()
1082
+ return f"urn:uuid:{uuid5(NAMESPACE_URL, digest)}"
1083
+
1084
+
1085
+ def _allocate_target_id(anchor: str | None, *, fallback: str, used_ids: set[str]) -> str:
1086
+ """从 producer anchor 或固定回退值分配无空白且不碰撞的 id。"""
1087
+ base = _safe_id_base(_anchor_key(anchor)) or fallback
1088
+ candidate = base
1089
+ suffix = 2
1090
+ while candidate in used_ids:
1091
+ candidate = f"{base}-{suffix}"
1092
+ suffix += 1
1093
+ used_ids.add(candidate)
1094
+ return candidate
1095
+
1096
+
1097
+ def _safe_id_base(value: str) -> str:
1098
+ """把 anchor 归一化为适合 XHTML fragment 的稳定 id 基值。"""
1099
+ normalized = _normalize_xml_text(value).strip()
1100
+ normalized = re.sub(r"\s+", "-", normalized)
1101
+ normalized = re.sub(r"[^\w.:-]+", "-", normalized, flags=re.UNICODE).strip("-")
1102
+ return normalized
1103
+
1104
+
1105
+ def _anchor_key(anchor: str | None) -> str:
1106
+ """保留 producer anchor 身份,仅去除首尾空白。"""
1107
+ return (anchor or "").strip()
1108
+
1109
+
1110
+ def _plain_content_text(content: str) -> str:
1111
+ """从 body 内容提取图片 alt 所需的可见纯文本。"""
1112
+ if not content:
1113
+ return ""
1114
+ if _is_supported_markup(content):
1115
+ return BeautifulSoup(content, "html.parser").get_text(" ", strip=True)
1116
+ return _normalize_xml_text(content).strip()
1117
+
1118
+
1119
+ def _normalize_code_language(language: str | None) -> str | None:
1120
+ """把代码语言限制为不会构造危险 class token 的短名称。"""
1121
+ normalized = (language or "").strip().lower().replace("_", "-")
1122
+ return normalized if _SAFE_LANGUAGE_RE.fullmatch(normalized) else None
1123
+
1124
+
1125
+ def _normalize_xml_text(content: str) -> str:
1126
+ """替换 XML 1.0 禁止的控制字符和孤立 surrogate。"""
1127
+ return _INVALID_XML_TEXT_RE.sub("\ufffd", content)
1128
+
1129
+
1130
+ def _needs_whitespace_preservation(content: str) -> bool:
1131
+ """判断文本是否含有 XHTML 默认会折叠的有效空白。"""
1132
+ return bool(content and (content != content.strip(" \t\n") or " " in content or "\t" in content or "\n" in content))
1133
+
1134
+
1135
+ def _has_visible_content(element: etree._Element) -> bool:
1136
+ """判断元素是否包含可见文本或媒体、结构子节点。"""
1137
+ if element.text and element.text.strip():
1138
+ return True
1139
+ if len(element):
1140
+ return True
1141
+ return False
1142
+
1143
+
1144
+ def _xhtml(tag: str) -> str:
1145
+ """返回 XHTML namespace 下的 Clark notation 标签名。"""
1146
+ return f"{{{_XHTML_NS}}}{tag}"
1147
+
1148
+
1149
+ @lru_cache(maxsize=1)
1150
+ def _load_epub_stylesheet() -> bytes:
1151
+ """读取随包分发的静态 EPUB 样式表并缓存字节。"""
1152
+ root = resources.files("docvortex").joinpath("resources", "epub")
1153
+ return root.joinpath(_STYLE_RESOURCE_NAME).read_bytes()
1154
+
1155
+
1156
+ __all__ = ["render_epub"]