docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,280 @@
1
+ from __future__ import annotations
2
+
3
+ import base64
4
+ import binascii
5
+ from pathlib import Path, PureWindowsPath
6
+ import re
7
+ import xml.etree.ElementTree as ElementTree
8
+ from typing import Final
9
+ from urllib.parse import urlsplit
10
+
11
+ INLINE_IMAGE_DATA_URI_RE = re.compile(r"data:image/([^;\"']+);base64,([^\"']+)", re.DOTALL)
12
+ _SVG_NAMESPACE = "http://www.w3.org/2000/svg"
13
+ _METAFILE_SVG_MARKER = "wmf-emf"
14
+ _METAFILE_SVG_FALLBACK_ID = "metafile-render-raster-fallback"
15
+ MAX_IMAGE_PAYLOAD_BYTES: Final = 64 * 1024 * 1024
16
+ MAX_GENERATED_SVG_BYTES: Final = MAX_IMAGE_PAYLOAD_BYTES
17
+ MAX_RASTER_IMAGE_BYTES: Final = MAX_IMAGE_PAYLOAD_BYTES
18
+ MAX_IMAGE_DATA_URI_BYTES: Final = len("data:image/svg+xml;base64,") + ((MAX_IMAGE_PAYLOAD_BYTES + 2) // 3) * 4
19
+ MAX_DECODED_RASTER_DIMENSION: Final = 8_192
20
+ MAX_DECODED_RASTER_PIXELS: Final = 16_000_000
21
+ _MAX_GENERATED_SVG_NODES = 100_000
22
+ _SAFE_SVG_ATTRIBUTES: dict[str, set[str]] = {
23
+ "svg": {"width", "height", "viewBox", "data-metafile-render"},
24
+ "metadata": {"id", "data-mime"},
25
+ "defs": set(),
26
+ "clipPath": {"id"},
27
+ "path": {
28
+ "d",
29
+ "fill",
30
+ "fill-rule",
31
+ "fill-opacity",
32
+ "opacity",
33
+ "stroke",
34
+ "stroke-width",
35
+ "stroke-linecap",
36
+ "stroke-linejoin",
37
+ "stroke-miterlimit",
38
+ "stroke-opacity",
39
+ "stroke-dasharray",
40
+ },
41
+ "g": {"clip-path"},
42
+ "image": {"width", "height", "transform", "href"},
43
+ "text": {
44
+ "x",
45
+ "y",
46
+ "fill",
47
+ "opacity",
48
+ "font-family",
49
+ "font-size",
50
+ "font-weight",
51
+ "font-style",
52
+ "text-anchor",
53
+ "dominant-baseline",
54
+ "text-decoration",
55
+ "transform",
56
+ },
57
+ "rect": {"x", "y", "width", "height", "fill", "opacity"},
58
+ }
59
+
60
+
61
+ class _RejectingSvgTreeBuilder(ElementTree.TreeBuilder):
62
+ """构造拒绝任何 DTD 的 SVG XML 树。"""
63
+
64
+ def doctype(self, _name: str, _pubid: str | None, _system: str | None) -> None:
65
+ """在实体声明被处理前拒绝任意偏移和编码的 DOCTYPE。"""
66
+ raise ValueError("Generated SVG must not contain a DTD or entity declaration")
67
+
68
+
69
+ def normalize_image_extension(fmt: str) -> str:
70
+ """规范化图片扩展名,保证同一图片格式生成稳定文件名。"""
71
+ normalized = fmt.lower().split("+", 1)[0]
72
+ return "jpg" if normalized in {"jpeg", "jpg"} else normalized
73
+
74
+
75
+ def validate_decoded_raster_size(width: int, height: int) -> None:
76
+ """在 Pillow 解码像素前校验 raster 单边尺寸与总像素预算。"""
77
+ if (
78
+ width <= 0
79
+ or height <= 0
80
+ or width > MAX_DECODED_RASTER_DIMENSION
81
+ or height > MAX_DECODED_RASTER_DIMENSION
82
+ or width * height > MAX_DECODED_RASTER_PIXELS
83
+ ):
84
+ raise ValueError(
85
+ f"Decoded raster image exceeds limits: {width}x{height}; "
86
+ f"max_dimension={MAX_DECODED_RASTER_DIMENSION}, max_pixels={MAX_DECODED_RASTER_PIXELS}"
87
+ )
88
+
89
+
90
+ def _parse_svg_root_strict(payload: bytes) -> ElementTree.Element:
91
+ """在固定字节预算内解析 SVG 根节点,并在实体展开前拒绝 DTD。"""
92
+ if len(payload) > MAX_GENERATED_SVG_BYTES:
93
+ raise ValueError("SVG image payload exceeds its byte limit")
94
+ try:
95
+ parser = ElementTree.XMLParser(target=_RejectingSvgTreeBuilder())
96
+ root = ElementTree.fromstring(payload, parser=parser)
97
+ except ElementTree.ParseError as exc:
98
+ raise ValueError("Invalid SVG image payload") from exc
99
+ if not isinstance(root.tag, str) or root.tag.rsplit("}", 1)[-1].lower() != "svg":
100
+ raise ValueError("Image signature does not match MIME subtype: svg+xml")
101
+ return root
102
+
103
+
104
+ def parse_image_data_uri_strict(data_uri: str) -> tuple[bytes, str]:
105
+ """严格解析图片 data URI,并同时校验 MIME 与文件签名是否一致。"""
106
+ if len(data_uri) > MAX_IMAGE_DATA_URI_BYTES:
107
+ raise ValueError("Image data URI exceeds its byte limit")
108
+ match = re.fullmatch(r"data:image/([^;]+);base64,([A-Za-z0-9+/]*={0,2})", data_uri)
109
+ if match is None:
110
+ raise ValueError("Invalid image data URI")
111
+
112
+ mime_subtype = match.group(1).lower()
113
+ extension = normalize_image_extension(mime_subtype)
114
+ encoded_payload = match.group(2)
115
+ payload_limit = MAX_GENERATED_SVG_BYTES if extension == "svg" else MAX_RASTER_IMAGE_BYTES
116
+ max_encoded_bytes = ((payload_limit + 2) // 3) * 4
117
+ if len(encoded_payload) > max_encoded_bytes:
118
+ payload_kind = "SVG" if extension == "svg" else "Raster"
119
+ raise ValueError(f"{payload_kind} image payload exceeds its byte limit")
120
+ if extension == "svg":
121
+ if mime_subtype != "svg+xml":
122
+ raise ValueError(f"Unsupported image MIME subtype: {mime_subtype}")
123
+ try:
124
+ payload = base64.b64decode(encoded_payload, validate=True)
125
+ except (binascii.Error, ValueError) as exc:
126
+ raise ValueError("Invalid base64 image payload") from exc
127
+ if not payload:
128
+ raise ValueError("Image payload must not be empty")
129
+ if len(payload) > payload_limit:
130
+ payload_kind = "SVG" if extension == "svg" else "Raster"
131
+ raise ValueError(f"{payload_kind} image payload exceeds its byte limit")
132
+
133
+ if extension == "svg":
134
+ _parse_svg_root_strict(payload)
135
+ return payload, extension
136
+
137
+ signatures: dict[str, tuple[bytes, ...]] = {
138
+ "jpg": (b"\xff\xd8\xff",),
139
+ "png": (b"\x89PNG\r\n\x1a\n",),
140
+ "gif": (b"GIF87a", b"GIF89a"),
141
+ "webp": (b"RIFF",),
142
+ "bmp": (b"BM",),
143
+ "tiff": (b"II*\x00", b"MM\x00*"),
144
+ }
145
+ expected = signatures.get(extension)
146
+ if expected is None:
147
+ raise ValueError(f"Unsupported image MIME subtype: {mime_subtype}")
148
+ if not any(payload.startswith(signature) for signature in expected):
149
+ raise ValueError(f"Image signature does not match MIME subtype: {mime_subtype}")
150
+ if extension == "webp" and (len(payload) < 12 or payload[8:12] != b"WEBP"):
151
+ raise ValueError("Image signature does not match MIME subtype: webp")
152
+ return payload, extension
153
+
154
+
155
+ def _decode_png_data_uri(value: str) -> bytes:
156
+ """严格解码生成 SVG 内嵌的 PNG data URI。"""
157
+ match = re.fullmatch(r"data:image/png;base64,([A-Za-z0-9+/]*={0,2})", value)
158
+ if match is None:
159
+ raise ValueError("Generated SVG image href must contain a PNG data URI")
160
+ try:
161
+ payload = base64.b64decode(match.group(1), validate=True)
162
+ except (binascii.Error, ValueError) as exc:
163
+ raise ValueError("Generated SVG contains invalid PNG base64") from exc
164
+ if not payload.startswith(b"\x89PNG\r\n\x1a\n"):
165
+ raise ValueError("Generated SVG fallback does not contain a PNG")
166
+ return payload
167
+
168
+
169
+ def _validate_generated_svg_attribute(tag: str, name: str, value: str) -> None:
170
+ """校验生成 SVG 属性只使用静态数值、颜色、本地 clip 或 PNG。"""
171
+ if name not in _SAFE_SVG_ATTRIBUTES[tag] or "\x00" in value or name.lower().startswith("on"):
172
+ raise ValueError(f"Generated SVG contains an unsafe attribute: {tag}.{name}")
173
+ normalized = value.strip().casefold()
174
+ if "javascript:" in normalized or "data:text" in normalized or "url(" in normalized and name != "clip-path":
175
+ raise ValueError(f"Generated SVG contains an unsafe attribute value: {tag}.{name}")
176
+ if name == "href":
177
+ _decode_png_data_uri(value)
178
+ elif name == "clip-path" and re.fullmatch(r"url\(#metafile-render-clip-\d+\)", value) is None:
179
+ raise ValueError("Generated SVG clip-path must reference a local metafile-render clip")
180
+ elif name == "transform" and re.fullmatch(r"(?:matrix|rotate)\([0-9eE+.,\- ]+\)", value) is None:
181
+ raise ValueError("Generated SVG transform is outside the supported subset")
182
+ elif name == "d" and re.fullmatch(r"[MmLlCcZz0-9eE+.,\- ]*", value) is None:
183
+ raise ValueError("Generated SVG path data is outside the supported subset")
184
+
185
+
186
+ def extract_generated_svg_fallback(payload: bytes) -> tuple[bytes, int, int]:
187
+ """验证 metafile-render 生成 SVG,并返回 PNG fallback 与逻辑像素尺寸。"""
188
+ if not isinstance(payload, bytes) or not payload:
189
+ raise ValueError("Generated SVG payload is empty or exceeds its byte limit")
190
+ root = _parse_svg_root_strict(payload)
191
+ if root.tag != f"{{{_SVG_NAMESPACE}}}svg" or root.get("data-metafile-render") != _METAFILE_SVG_MARKER:
192
+ raise ValueError("SVG is not marked as a metafile-render generated image")
193
+ try:
194
+ width = int(root.get("width", ""))
195
+ height = int(root.get("height", ""))
196
+ except ValueError as exc:
197
+ raise ValueError("Generated SVG dimensions must be integers") from exc
198
+ if width <= 0 or height <= 0 or width > 8192 or height > 8192 or root.get("viewBox") != f"0 0 {width} {height}":
199
+ raise ValueError("Generated SVG dimensions or viewBox are outside the supported bounds")
200
+
201
+ fallback: bytes | None = None
202
+ node_count = 0
203
+ for element in root.iter():
204
+ node_count += 1
205
+ if node_count > _MAX_GENERATED_SVG_NODES:
206
+ raise ValueError("Generated SVG exceeds its node limit")
207
+ if not isinstance(element.tag, str) or not element.tag.startswith(f"{{{_SVG_NAMESPACE}}}"):
208
+ raise ValueError("Generated SVG contains a foreign namespace")
209
+ tag = element.tag.rsplit("}", 1)[-1]
210
+ if tag not in _SAFE_SVG_ATTRIBUTES:
211
+ raise ValueError(f"Generated SVG contains an unsupported element: {tag}")
212
+ for attribute_name, attribute_value in element.attrib.items():
213
+ if "}" in attribute_name:
214
+ raise ValueError("Generated SVG contains a namespaced attribute")
215
+ _validate_generated_svg_attribute(tag, attribute_name, attribute_value)
216
+ if element.tail and element.tail.strip():
217
+ raise ValueError("Generated SVG contains unexpected tail text")
218
+ if tag == "metadata":
219
+ if (
220
+ element.get("id") != _METAFILE_SVG_FALLBACK_ID
221
+ or element.get("data-mime") != "image/png"
222
+ or fallback is not None
223
+ ):
224
+ raise ValueError("Generated SVG fallback metadata is invalid or duplicated")
225
+ encoded = (element.text or "").strip()
226
+ fallback = _decode_png_data_uri(f"data:image/png;base64,{encoded}")
227
+ elif tag != "text" and element.text and element.text.strip():
228
+ raise ValueError(f"Generated SVG element must not contain text: {tag}")
229
+ if fallback is None:
230
+ raise ValueError("Generated SVG does not contain a PNG fallback")
231
+ return fallback, width, height
232
+
233
+
234
+ def validate_image_sidecar_path(image_path: str) -> str:
235
+ """校验图片 sidecar 路径只能是安全的相对子路径,并返回规范化 POSIX 路径。"""
236
+ if (
237
+ not image_path
238
+ or image_path == "."
239
+ or "\\" in image_path
240
+ or any(ord(char) < 0x20 or ord(char) == 0x7F for char in image_path)
241
+ ):
242
+ raise ValueError(f"Unsafe image sidecar path: {image_path}")
243
+ parsed = urlsplit(image_path)
244
+ if parsed.scheme or parsed.netloc:
245
+ raise ValueError(f"Unsafe image sidecar path: {image_path}")
246
+ posix_path = Path(image_path)
247
+ windows_path = PureWindowsPath(image_path)
248
+ if (
249
+ posix_path.is_absolute()
250
+ or windows_path.is_absolute()
251
+ or windows_path.drive
252
+ or windows_path.root
253
+ or ".." in posix_path.parts
254
+ or ".." in windows_path.parts
255
+ ):
256
+ raise ValueError(f"Unsafe image sidecar path: {image_path}")
257
+ return posix_path.as_posix()
258
+
259
+
260
+ def validate_remote_image_url(image_url: str) -> str:
261
+ """校验远程图片只能使用无凭据的 HTTP(S) 绝对地址。"""
262
+ normalized = image_url.strip()
263
+ if (
264
+ not normalized
265
+ or any(ord(char) < 0x20 or ord(char) == 0x7F for char in normalized)
266
+ or any(char in normalized for char in ("<", ">", "\\"))
267
+ or normalized.startswith("//")
268
+ ):
269
+ raise ValueError(f"Unsafe remote image URL: {image_url}")
270
+ try:
271
+ parsed = urlsplit(normalized)
272
+ _ = parsed.port
273
+ hostname = parsed.hostname
274
+ except ValueError as exc:
275
+ raise ValueError(f"Unsafe remote image URL: {image_url}") from exc
276
+ if parsed.scheme.casefold() not in {"http", "https"} or not parsed.netloc or hostname is None:
277
+ raise ValueError(f"Unsafe remote image URL: {image_url}")
278
+ if parsed.username is not None or parsed.password is not None:
279
+ raise ValueError(f"Remote image URL must not contain credentials: {image_url}")
280
+ return normalized
@@ -0,0 +1,92 @@
1
+ import os
2
+ import re
3
+ import unicodedata
4
+
5
+ DEFAULT_CODE_LANGUAGE = "txt"
6
+ _INVALID_SURROGATES = re.compile("[\ud800-\udfff]")
7
+
8
+
9
+ def _detect_language(text: str) -> object:
10
+ """首次检测时配置本地模型缓存并惰性加载语言识别器。"""
11
+ from fast_langdetect import detect_language
12
+
13
+ return detect_language(text, low_memory=True)
14
+
15
+
16
+ def remove_invalid_surrogates(text: str) -> str:
17
+ """等价移除代理码点;常见合法 Unicode 文本无需逐字符 Python 扫描。"""
18
+ return _INVALID_SURROGATES.sub("", text)
19
+
20
+
21
+ def detect_lang(text: str) -> str:
22
+ if len(text) == 0:
23
+ return ""
24
+
25
+ text = text.replace("\n", "")
26
+ text = remove_invalid_surrogates(text)
27
+
28
+ try:
29
+ lang_upper = _detect_language(text)
30
+ except Exception:
31
+ html_no_ctrl_chars = "".join([c for c in text if unicodedata.category(c)[0] not in ["C"]])
32
+ lang_upper = _detect_language(html_no_ctrl_chars)
33
+
34
+ try:
35
+ lang = lang_upper.lower()
36
+ except Exception:
37
+ lang = ""
38
+ return lang
39
+
40
+
41
+ def _normalize_text_for_language_guess(code: str) -> str:
42
+ """移除孤立代理字符并还原合法代理对,供代码语言识别使用。"""
43
+ if not code:
44
+ return ""
45
+ normalized: list[str] = []
46
+ index = 0
47
+ while index < len(code):
48
+ current_char = code[index]
49
+ current_ord = ord(current_char)
50
+ if 0xD800 <= current_ord <= 0xDBFF:
51
+ if index + 1 < len(code):
52
+ next_char = code[index + 1]
53
+ next_ord = ord(next_char)
54
+ if 0xDC00 <= next_ord <= 0xDFFF:
55
+ pair = current_char + next_char
56
+ normalized.append(pair.encode("utf-16", "surrogatepass").decode("utf-16"))
57
+ index += 2
58
+ continue
59
+ index += 1
60
+ continue
61
+ if 0xDC00 <= current_ord <= 0xDFFF:
62
+ index += 1
63
+ continue
64
+ normalized.append(current_char)
65
+ index += 1
66
+ return "".join(normalized)
67
+
68
+
69
+ def guess_code_language(code: str) -> str:
70
+ """使用 Magika 推断代码块语言,失败时返回纯文本类型。"""
71
+ normalized_code = _normalize_text_for_language_guess(code)
72
+ if not normalized_code:
73
+ return DEFAULT_CODE_LANGUAGE
74
+ try:
75
+ from magika import Magika
76
+
77
+ lang = Magika().identify_bytes(normalized_code.encode("utf-8", errors="replace")).prediction.output.label
78
+ except Exception:
79
+ return DEFAULT_CODE_LANGUAGE
80
+ return lang if lang != "unknown" else DEFAULT_CODE_LANGUAGE
81
+
82
+
83
+ if __name__ == "__main__":
84
+ print(os.getenv("FTLANG_CACHE"))
85
+ print(detect_lang("This is a test."))
86
+ print(detect_lang("<html>This is a test</html>"))
87
+ print(detect_lang("这个是中文测试。"))
88
+ print(detect_lang("<html>这个是中文测试。</html>"))
89
+ print(detect_lang("〖\ud835\udc46\ud835〗这是个包含utf-16的中文测试"))
90
+
91
+
92
+ __all__ = ["DEFAULT_CODE_LANGUAGE", "detect_lang", "guess_code_language", "remove_invalid_surrogates"]
@@ -0,0 +1,38 @@
1
+ import platform
2
+
3
+ from packaging import version
4
+
5
+
6
+ def is_windows_environment() -> bool:
7
+ return platform.system() == "Windows"
8
+
9
+
10
+ # Detect if the current environment is a Mac computer
11
+ def is_mac_environment() -> bool:
12
+ return platform.system() == "Darwin"
13
+
14
+
15
+ def is_linux_environment() -> bool:
16
+ return platform.system() == "Linux"
17
+
18
+
19
+ # Detect if CPU is Apple Silicon architecture
20
+ def is_apple_silicon_cpu() -> bool:
21
+ return platform.machine() in ["arm64", "aarch64"]
22
+
23
+
24
+ # If Mac computer with Apple Silicon architecture, check if macOS version is 13.5 or above
25
+ def is_mac_os_version_supported(min_version: str = "13.5") -> bool:
26
+ if not is_mac_environment() or not is_apple_silicon_cpu():
27
+ return False
28
+ mac_version = platform.mac_ver()[0]
29
+ if not mac_version:
30
+ return False
31
+ # print("Mac OS Version:", mac_version)
32
+ return version.parse(mac_version) >= version.parse(min_version)
33
+
34
+
35
+ if __name__ == "__main__":
36
+ print("Is Mac Environment:", is_mac_environment())
37
+ print("Is Apple Silicon CPU:", is_apple_silicon_cpu())
38
+ print("Is Mac OS Version Supported (>=13.5):", is_mac_os_version_supported())
@@ -0,0 +1,153 @@
1
+ """跨模型共享的文本字符规范化与换行连接规则。"""
2
+
3
+ import re
4
+ from collections.abc import Sequence
5
+
6
+ # 中日韩文本的物理换行通常不需要插入额外空格,集中定义以供各后端和渲染器共享。
7
+ CJK_LANGS = frozenset({"zh", "ja", "ko"})
8
+
9
+ # PDF 文本抽取时,英文跨行断词可能被编码为多种 hyphen 字符。
10
+ # 这里只用于判断“行末英文断词符”,不要扩展到 en/em dash 等普通破折号。
11
+ LINE_END_HYPHEN_CHARS = "-\u00ad\u2010\u2011\u2043"
12
+ LINE_END_HYPHEN_RE = re.compile(rf"[A-Za-z]+[{re.escape(LINE_END_HYPHEN_CHARS)}]\s*$")
13
+
14
+ # URL 候选仅允许 RFC 3986 常见 ASCII 字符,避免把中文正文吞入链接。
15
+ _URL_CANDIDATE_RE = re.compile(
16
+ r"(?<![A-Za-z0-9])(?:(?:https?|ftp)://|www\.)[A-Za-z0-9._~:/?#\[\]@!$&'()*+,;=%-]+",
17
+ re.ASCII | re.IGNORECASE,
18
+ )
19
+ # 下一行自身以完整 URL 开头时,必须保留边界,避免两条独立链接相连。
20
+ _URL_AT_LINE_START_RE = re.compile(
21
+ r"(?:(?:https?|ftp)://|www\.)",
22
+ re.ASCII | re.IGNORECASE,
23
+ )
24
+
25
+
26
+ def is_hyphen_at_line_end(line: str) -> bool:
27
+ """判断文本行是否以英文单词的跨行断词符结尾。
28
+
29
+ 只识别字母后紧跟行末 hyphen 的断词场景,不处理词内连字符或普通破折号。
30
+ """
31
+ return bool(LINE_END_HYPHEN_RE.search(line))
32
+
33
+
34
+ def _url_spans_line_boundary(previous_content: str, next_content: str) -> bool:
35
+ """判断无空格候选中是否存在严格横跨当前物理行边界的 URL。"""
36
+ stripped_previous = previous_content.rstrip()
37
+ stripped_next = next_content.lstrip()
38
+ if not stripped_previous or not stripped_next:
39
+ return False
40
+ candidate = f"{stripped_previous}{stripped_next}"
41
+ boundary = len(stripped_previous)
42
+ return any(match.start() < boundary < match.end() for match in _URL_CANDIDATE_RE.finditer(candidate))
43
+
44
+
45
+ def resolve_text_line_boundary(
46
+ previous_content: str,
47
+ *,
48
+ block_language: str,
49
+ next_content: str,
50
+ ) -> tuple[str, str]:
51
+ """返回处理后的上一行内容和本次物理行边界分隔符。
52
+
53
+ 严格横跨边界的 URL 候选直接连接,但下一行自身为完整 URL 时保留空格。
54
+ 其余 CJK 文本直接连接物理行,普通西文行插入一个空格。西文行末如果是
55
+ 合法的 hyphen,则始终直接连接下一行,并仅在下一行以小写字母开头时删除
56
+ hyphen。
57
+ """
58
+ processed_content = previous_content.rstrip()
59
+ if not processed_content:
60
+ return "", ""
61
+ stripped_next = next_content.lstrip()
62
+ if _url_spans_line_boundary(processed_content, stripped_next):
63
+ if _URL_AT_LINE_START_RE.match(stripped_next):
64
+ return processed_content, " "
65
+ return processed_content, ""
66
+ if block_language in CJK_LANGS:
67
+ return processed_content, ""
68
+ if not is_hyphen_at_line_end(processed_content):
69
+ return processed_content, " "
70
+ if stripped_next and stripped_next[0].islower():
71
+ return processed_content[:-1], ""
72
+ return processed_content, ""
73
+
74
+
75
+ def merge_text_line_contents(
76
+ line_contents: Sequence[str],
77
+ *,
78
+ block_language: str,
79
+ ) -> str:
80
+ """按累计文本上下文折叠物理行,支持跨越三行以上的 URL 连续拼接。"""
81
+
82
+ normalized_lines = [str(content) for content in line_contents if str(content)]
83
+ if not normalized_lines:
84
+ return ""
85
+ merged_content = normalized_lines[0]
86
+ for current_line in normalized_lines[1:]:
87
+ merged_content, separator = resolve_text_line_boundary(
88
+ merged_content,
89
+ block_language=block_language,
90
+ next_content=current_line,
91
+ )
92
+ merged_content = f"{merged_content}{separator}{current_line}"
93
+ return merged_content.strip()
94
+
95
+
96
+ def full_to_half_exclude_marks(text: str) -> str:
97
+ """将全角英文字母和数字转换为半角形式,同时保留全角标点。"""
98
+ result = []
99
+ for char in text:
100
+ code = ord(char)
101
+ # Full-width letters and numbers (FF21-FF3A for A-Z, FF41-FF5A for a-z, FF10-FF19 for 0-9)
102
+ if (0xFF21 <= code <= 0xFF3A) or (0xFF41 <= code <= 0xFF5A) or (0xFF10 <= code <= 0xFF19):
103
+ result.append(chr(code - 0xFEE0)) # Shift to ASCII range
104
+ else:
105
+ result.append(char)
106
+ return "".join(result)
107
+
108
+
109
+ def full_to_half(text: str) -> str:
110
+ """将全角 ASCII 字母、数字和标点统一转换为半角形式。"""
111
+ result = []
112
+ for char in text:
113
+ code = ord(char)
114
+ # Full-width letters, numbers and punctuation (FF01-FF5E)
115
+ if 0xFF01 <= code <= 0xFF5E:
116
+ result.append(chr(code - 0xFEE0)) # Shift to ASCII range
117
+ else:
118
+ result.append(char)
119
+ return "".join(result)
120
+
121
+
122
+ def clean_isolated_formula(content: str) -> str:
123
+ """移除行间公式外层的反斜杠方括号并清理首尾空白。"""
124
+ latex = content[:]
125
+ if latex.startswith("\\["):
126
+ latex = latex[2:]
127
+ if latex.endswith("\\]"):
128
+ latex = latex[:-2]
129
+ return latex.strip()
130
+
131
+
132
+ def normalize_formula_tag_content(tag_content: str) -> str:
133
+ """归一化公式编号文本,去掉全角字符和包裹括号后用于 \\tag{}。"""
134
+ tag_content = full_to_half(str(tag_content or "").strip())
135
+ if tag_content.startswith(("(", "﹙")):
136
+ tag_content = tag_content[1:].strip()
137
+ if tag_content.endswith((")", "﹚")):
138
+ tag_content = tag_content[:-1].strip()
139
+ return tag_content
140
+
141
+
142
+ def normalize_formula_content_for_tag(formula_content: str) -> str:
143
+ """归一化待合并编号的公式正文,去掉模型可能携带的展示公式分隔符。"""
144
+ return clean_isolated_formula(str(formula_content or ""))
145
+
146
+
147
+ def build_tagged_formula_content(formula_content: str, tag_content: str) -> str | None:
148
+ """将公式正文和编号文本合成为带 LaTeX tag 的纯公式内容。"""
149
+ formula_content = normalize_formula_content_for_tag(formula_content)
150
+ tag_content = normalize_formula_tag_content(tag_content)
151
+ if not formula_content or not tag_content:
152
+ return None
153
+ return f"{formula_content}\\tag{{{tag_content}}}"
@@ -0,0 +1,20 @@
1
+ """保持搬迁后类型的历史导入身份及可独立解析的注解。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import fields, is_dataclass
6
+ from typing import get_type_hints
7
+
8
+
9
+ def preserve_type_module(cls: type, module: str) -> None:
10
+ """先在定义模块解析注解,再保留旧 pickle 路径,避免依赖尚未导入的门面。"""
11
+ annotations = get_type_hints(cls)
12
+ cls.__annotations__ = annotations
13
+ if is_dataclass(cls):
14
+ for field in fields(cls):
15
+ if field.name in annotations:
16
+ field.type = annotations[field.name]
17
+ cls.__module__ = module
18
+
19
+
20
+ __all__ = ["preserve_type_module"]
@@ -0,0 +1,20 @@
1
+ """安全归一化 XML namespace 与 legacy HTML 前缀标签名。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from lxml import etree # type: ignore[reportMissingImports]
6
+
7
+
8
+ def local_name(element: etree._Element) -> str:
9
+ """返回 Clark notation、普通或冒号前缀标签的小写本地名。"""
10
+ tag = element.tag
11
+ if not isinstance(tag, str):
12
+ return ""
13
+ if tag.startswith("{") and "}" in tag:
14
+ tag = tag.split("}", 1)[1]
15
+ elif ":" in tag:
16
+ tag = tag.rsplit(":", 1)[1]
17
+ return tag.casefold()
18
+
19
+
20
+ __all__ = ["local_name"]
docvortex/options.py ADDED
@@ -0,0 +1,30 @@
1
+ """独立引擎的显式渲染选项,不读取环境或宿主配置。"""
2
+
3
+ from pydantic import BaseModel, Field
4
+
5
+
6
+ class LatexDelimiterConfig(BaseModel):
7
+ """单组 LaTeX 左右定界符配置。"""
8
+
9
+ left: str = Field(min_length=1)
10
+ right: str = Field(min_length=1)
11
+
12
+
13
+ def _default_display_latex_delimiter() -> LatexDelimiterConfig:
14
+ """构造缺省行间公式定界符。"""
15
+ return LatexDelimiterConfig(left="$$", right="$$")
16
+
17
+
18
+ def _default_inline_latex_delimiter() -> LatexDelimiterConfig:
19
+ """构造缺省行内公式定界符。"""
20
+ return LatexDelimiterConfig(left="$", right="$")
21
+
22
+
23
+ class LatexDelimitersConfig(BaseModel):
24
+ """Markdown 行内与行间公式定界符配置。"""
25
+
26
+ display: LatexDelimiterConfig = Field(default_factory=_default_display_latex_delimiter)
27
+ inline: LatexDelimiterConfig = Field(default_factory=_default_inline_latex_delimiter)
28
+
29
+
30
+ __all__ = ["LatexDelimiterConfig", "LatexDelimitersConfig"]
@@ -0,0 +1,3 @@
1
+ """将严格 ModelJson 转换并增强为 MiddleJson 的后处理层。"""
2
+
3
+ __all__: list[str] = []