docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,201 @@
1
+ """PDF 页面坐标、方向和裁图使用的无状态几何原语。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import base64
6
+ from typing import Any
7
+
8
+ import cv2
9
+ import numpy as np
10
+ from loguru import logger
11
+
12
+ from ...schema import BBox
13
+ from ...foundation.geometry import normalize_to_int_bbox
14
+
15
+
16
+ def _normalize_page_size(page_image: Any) -> tuple[int, int]:
17
+ """从PIL或numpy图像中读取页面宽高,供归一化bbox还原为像素bbox。"""
18
+ if hasattr(page_image, "size"):
19
+ return page_image.size
20
+
21
+ height, width = page_image.shape[:2]
22
+ return width, height
23
+
24
+
25
+ def _bbox_to_pixel_bbox(bbox: BBox | None, page_size: tuple[int, int]) -> BBox | None:
26
+ """将归一化或像素bbox统一成像素bbox,异常bbox返回None。"""
27
+ if bbox is None or len(bbox) != 4:
28
+ return None
29
+
30
+ try:
31
+ x0, y0, x1, y1 = [float(v) for v in bbox]
32
+ except (TypeError, ValueError):
33
+ return None
34
+
35
+ width, height = page_size
36
+ if all(0.0 <= value <= 1.0 for value in [x0, y0, x1, y1]):
37
+ x0, y0, x1, y1 = x0 * width, y0 * height, x1 * width, y1 * height
38
+
39
+ left, right = sorted([x0, x1])
40
+ top, bottom = sorted([y0, y1])
41
+ if right <= left or bottom <= top:
42
+ return None
43
+ return (left, top, right, bottom)
44
+
45
+
46
+ def _normalize_layout_bbox_to_unit(bbox: BBox | None, page_size: tuple[int, int]) -> list[float] | None:
47
+ """将 layout 像素 bbox 归一化为 VLM ContentBlock 需要的 0-1 坐标。"""
48
+ pixel_bbox = _bbox_to_pixel_bbox(bbox, page_size)
49
+ if pixel_bbox is None:
50
+ return None
51
+
52
+ page_width, page_height = page_size
53
+ if page_width <= 0 or page_height <= 0:
54
+ return None
55
+
56
+ x0, y0, x1, y1 = pixel_bbox
57
+ unit_bbox = [
58
+ round(max(0.0, min(1.0, float(x0) / page_width)), 3),
59
+ round(max(0.0, min(1.0, float(y0) / page_height)), 3),
60
+ round(max(0.0, min(1.0, float(x1) / page_width)), 3),
61
+ round(max(0.0, min(1.0, float(y1) / page_height)), 3),
62
+ ]
63
+ if unit_bbox[2] <= unit_bbox[0] or unit_bbox[3] <= unit_bbox[1]:
64
+ return None
65
+ return unit_bbox
66
+
67
+
68
+ def _medium_bbox_to_quad(bbox: list[float] | tuple[float, ...]) -> np.ndarray:
69
+ """将普通 bbox 转为表格模型 OCR token 使用的四点框。"""
70
+ x0, y0, x1, y1 = [float(v) for v in bbox]
71
+ return np.asarray([[x0, y0], [x1, y0], [x1, y1], [x0, y1]], dtype=np.float32)
72
+
73
+
74
+ def _normalize_medium_content(value: Any) -> str:
75
+ """将 medium 本地模型输出的文本字段规范成 Hybrid block 可消费的字符串。"""
76
+ if isinstance(value, list):
77
+ return "\n".join(str(item) for item in value if str(item).strip())
78
+ if isinstance(value, str):
79
+ return value.strip()
80
+ return ""
81
+
82
+
83
+ def _table_bbox_center(bbox: BBox) -> tuple[float, float]:
84
+ """计算 bbox 中心点,用于判断图片或公式应归属哪个表格。"""
85
+ return (float(bbox[0]) + float(bbox[2])) / 2.0, (float(bbox[1]) + float(bbox[3])) / 2.0
86
+
87
+
88
+ def _normalize_visual_block_angle(angle: Any) -> int:
89
+ """规范视觉块角度为 0/90/180/270,无法识别的角度按 0 处理。"""
90
+ try:
91
+ normalized_angle = int(float(angle or 0)) % 360
92
+ except (TypeError, ValueError):
93
+ logger.warning(f"Unsupported visual block angle: {angle}, using 0")
94
+ return 0
95
+ if normalized_angle not in {0, 90, 180, 270}:
96
+ logger.warning(f"Unsupported visual block angle: {angle}, using 0")
97
+ return 0
98
+ return normalized_angle
99
+
100
+
101
+ def _rotate_visual_block_image_to_upright(image: np.ndarray, angle: int) -> np.ndarray:
102
+ """按 layout 视觉块角度把裁图旋转至正向,角度语义与方向分类模型保持一致。"""
103
+ if angle == 270:
104
+ return cv2.rotate(image, cv2.ROTATE_90_CLOCKWISE)
105
+ if angle == 90:
106
+ return cv2.rotate(image, cv2.ROTATE_90_COUNTERCLOCKWISE)
107
+ if angle == 180:
108
+ return cv2.rotate(image, cv2.ROTATE_180)
109
+ return image
110
+
111
+
112
+ def _rotate_medium_table_bbox(
113
+ bbox: BBox,
114
+ image_width: float,
115
+ image_height: float,
116
+ angle: int,
117
+ ) -> BBox:
118
+ """把原表格裁图中的 bbox 同步转换到旋转后裁图坐标系。"""
119
+ x0, y0, x1, y1 = [float(value) for value in bbox]
120
+ if angle == 270:
121
+ # 顺时针旋转 90 度后,新 x 轴来自原 y 轴的反方向。
122
+ return (image_height - y1, x0, image_height - y0, x1)
123
+ if angle == 90:
124
+ # 逆时针旋转 90 度后,新 y 轴来自原 x 轴的反方向。
125
+ return (y0, image_width - x1, y1, image_width - x0)
126
+ if angle == 180:
127
+ return (image_width - x1, image_height - y1, image_width - x0, image_height - y0)
128
+ return (x0, y0, x1, y1)
129
+
130
+
131
+ def _get_medium_table_virtual_image_bbox(
132
+ bbox: BBox,
133
+ image_size: tuple[int, int],
134
+ box_size: float = 10.0,
135
+ ) -> BBox:
136
+ """在图片中心生成小 OCR token 框,避免图片大框干扰单元格匹配。"""
137
+ image_width, image_height = image_size
138
+ center_x, center_y = _table_bbox_center(bbox)
139
+ half_size = box_size / 2.0
140
+ return (
141
+ max(0.0, center_x - half_size),
142
+ max(0.0, center_y - half_size),
143
+ min(float(image_width), center_x + half_size),
144
+ min(float(image_height), center_y + half_size),
145
+ )
146
+
147
+
148
+ def _encode_page_crop_as_jpeg_data_uri(
149
+ np_image: np.ndarray,
150
+ page_bbox: BBox,
151
+ angle: int,
152
+ ) -> str:
153
+ """从页面原图按像素框裁剪,按视觉块方向回正后编码为 JPEG data URI。"""
154
+ image_h, image_w = np_image.shape[:2]
155
+ image_bbox = normalize_to_int_bbox(page_bbox, image_size=(image_h, image_w))
156
+ if image_bbox is None:
157
+ return ""
158
+ x0, y0, x1, y1 = image_bbox
159
+ crop_rgb = np_image[y0:y1, x0:x1].copy()
160
+ if crop_rgb.size == 0:
161
+ return ""
162
+
163
+ crop_rgb = _rotate_visual_block_image_to_upright(crop_rgb, angle)
164
+ crop_bgr = cv2.cvtColor(crop_rgb, cv2.COLOR_RGB2BGR)
165
+ success, encoded = cv2.imencode(".jpg", crop_bgr)
166
+ if not success:
167
+ return ""
168
+ return f"data:image/jpeg;base64,{base64.b64encode(encoded.tobytes()).decode('ascii')}"
169
+
170
+
171
+ def _sidecar_bbox_to_page_bbox(
172
+ bbox: BBox | None,
173
+ page_size: tuple[float, float],
174
+ render_scale: float,
175
+ ) -> BBox | None:
176
+ """将公式或 OCR sidecar bbox 转为 PDF point 坐标,供原生字符匹配和组行。"""
177
+ if bbox is None or len(bbox) != 4 or render_scale <= 0:
178
+ return None
179
+ try:
180
+ x0, y0, x1, y1 = [float(value) for value in bbox]
181
+ except (TypeError, ValueError):
182
+ return None
183
+
184
+ page_width, page_height = page_size
185
+ if page_width <= 0 or page_height <= 0:
186
+ return None
187
+ if all(0.0 <= value <= 1.0 for value in [x0, y0, x1, y1]):
188
+ x0, y0, x1, y1 = x0 * page_width, y0 * page_height, x1 * page_width, y1 * page_height
189
+ else:
190
+ x0, y0, x1, y1 = (
191
+ x0 / render_scale,
192
+ y0 / render_scale,
193
+ x1 / render_scale,
194
+ y1 / render_scale,
195
+ )
196
+
197
+ left, right = sorted([max(0.0, min(page_width, x0)), max(0.0, min(page_width, x1))])
198
+ top, bottom = sorted([max(0.0, min(page_height, y0)), max(0.0, min(page_height, y1))])
199
+ if right <= left or bottom <= top:
200
+ return None
201
+ return (left, top, right, bottom)
@@ -0,0 +1,343 @@
1
+ """视觉块容器补全、方向归一化与页面裁图。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from typing import TYPE_CHECKING, Any
7
+
8
+ import numpy as np
9
+ from loguru import logger
10
+
11
+ from ...schema import BBox, BlockType
12
+ from ...foundation.geometry import calculate_overlap_area_2_minbox_area_ratio, calculate_overlap_area_in_bbox1_area_ratio
13
+
14
+ from .constants import (
15
+ IMAGE_BLOCK_CONTAINMENT_THRESHOLD,
16
+ IMAGE_BLOCK_LAYOUT_COVERAGE_THRESHOLD,
17
+ IMAGE_BLOCK_LAYOUT_MIN_VISUAL_COUNT,
18
+ LOCAL_LAYOUT_IMAGE_BLOCK_AREA_TYPES,
19
+ LOCAL_LAYOUT_IMAGE_BLOCK_BODY_TYPES,
20
+ MODEL_JSON_VISUAL_BLOCK_TYPES,
21
+ )
22
+ from .visual_geometry import (
23
+ _bbox_to_pixel_bbox,
24
+ _encode_page_crop_as_jpeg_data_uri,
25
+ _normalize_page_size,
26
+ _normalize_visual_block_angle,
27
+ )
28
+
29
+ if TYPE_CHECKING:
30
+ from .document import PDFDocument
31
+
32
+
33
+ def _normalize_model_bbox_for_containment(raw_bbox: Any) -> BBox | None:
34
+ """校验模型 block 的四点框,返回可用于面积包含判断的浮点坐标。"""
35
+ try:
36
+ if raw_bbox is None or len(raw_bbox) != 4:
37
+ return None
38
+ bbox = tuple(float(value) for value in raw_bbox)
39
+ except (TypeError, ValueError):
40
+ return None
41
+
42
+ if not all(math.isfinite(value) for value in bbox):
43
+ return None
44
+ if bbox[2] <= bbox[0] or bbox[3] <= bbox[1]:
45
+ return None
46
+ return bbox
47
+
48
+
49
+ def _collapse_image_blocks(
50
+ page_model_list: list[dict[str, Any]],
51
+ containment_threshold: float = IMAGE_BLOCK_CONTAINMENT_THRESHOLD,
52
+ ) -> None:
53
+ """将 image_block 折叠为单个图片,并删除面积上被其包裹的非容器子块。"""
54
+ image_blocks = [block for block in page_model_list if block.get("type") == "image_block"]
55
+ if not image_blocks:
56
+ return
57
+
58
+ image_block_ids = {id(block) for block in image_blocks}
59
+ image_block_bboxes = [
60
+ bbox for block in image_blocks if (bbox := _normalize_model_bbox_for_containment(block.get("bbox"))) is not None
61
+ ]
62
+
63
+ retained_blocks: list[dict[str, Any]] = []
64
+ for block in page_model_list:
65
+ if id(block) in image_block_ids:
66
+ block["type"] = BlockType.IMAGE
67
+ retained_blocks.append(block)
68
+ continue
69
+
70
+ block_bbox = _normalize_model_bbox_for_containment(block.get("bbox"))
71
+ is_contained = block_bbox is not None and any(
72
+ calculate_overlap_area_in_bbox1_area_ratio(block_bbox, image_block_bbox) >= containment_threshold
73
+ for image_block_bbox in image_block_bboxes
74
+ )
75
+ if not is_contained:
76
+ retained_blocks.append(block)
77
+
78
+ page_model_list[:] = retained_blocks
79
+
80
+
81
+ def _supplement_missing_image_block_containers(
82
+ model_list: list[list[dict[str, Any]]],
83
+ layout_blocks_list: list[list[dict[str, Any]]],
84
+ containment_threshold: float = IMAGE_BLOCK_CONTAINMENT_THRESHOLD,
85
+ coverage_threshold: float = IMAGE_BLOCK_LAYOUT_COVERAGE_THRESHOLD,
86
+ min_visual_count: int = IMAGE_BLOCK_LAYOUT_MIN_VISUAL_COUNT,
87
+ ) -> None:
88
+ """用本地 layout 整图框为 xhigh 结果补充缺失的 image_block 容器。"""
89
+ if len(model_list) != len(layout_blocks_list):
90
+ raise ValueError(
91
+ "Hybrid image-block fallback page count mismatch: "
92
+ f"model_list={len(model_list)}, layout_blocks={len(layout_blocks_list)}"
93
+ )
94
+
95
+ for page_model_list, page_layout_blocks in zip(model_list, layout_blocks_list):
96
+ existing_image_block_bboxes = [
97
+ bbox
98
+ for block in page_model_list
99
+ if block.get("type") == "image_block"
100
+ if (bbox := _normalize_model_bbox_for_containment(block.get("bbox"))) is not None
101
+ ]
102
+
103
+ existing_claimed_block_ids: set[int] = set()
104
+ if existing_image_block_bboxes:
105
+ for block in page_model_list:
106
+ if block.get("type") == "image_block":
107
+ continue
108
+ block_bbox = _normalize_model_bbox_for_containment(block.get("bbox"))
109
+ if block_bbox is not None and any(
110
+ calculate_overlap_area_in_bbox1_area_ratio(block_bbox, image_block_bbox) >= containment_threshold
111
+ for image_block_bbox in existing_image_block_bboxes
112
+ ):
113
+ existing_claimed_block_ids.add(id(block))
114
+
115
+ candidates: list[tuple[int, float, int, int, dict[str, Any], set[int]]] = []
116
+ for layout_order, layout_block in enumerate(page_layout_blocks):
117
+ if layout_block.get("type") != BlockType.IMAGE or layout_block.get("sub_type") == "seal":
118
+ continue
119
+
120
+ layout_bbox = _normalize_model_bbox_for_containment(layout_block.get("bbox"))
121
+ if layout_bbox is None:
122
+ continue
123
+ if any(
124
+ calculate_overlap_area_2_minbox_area_ratio(layout_bbox, image_block_bbox) >= containment_threshold
125
+ for image_block_bbox in existing_image_block_bboxes
126
+ ):
127
+ continue
128
+
129
+ contained_blocks: list[tuple[int, dict[str, Any], BBox]] = []
130
+ for block_index, block in enumerate(page_model_list):
131
+ if block.get("type") == "image_block" or id(block) in existing_claimed_block_ids:
132
+ continue
133
+ block_bbox = _normalize_model_bbox_for_containment(block.get("bbox"))
134
+ if block_bbox is None:
135
+ continue
136
+ if calculate_overlap_area_in_bbox1_area_ratio(block_bbox, layout_bbox) >= containment_threshold:
137
+ contained_blocks.append((block_index, block, block_bbox))
138
+
139
+ contained_visuals = [
140
+ (block_index, block)
141
+ for block_index, block, _ in contained_blocks
142
+ if block.get("type") in LOCAL_LAYOUT_IMAGE_BLOCK_BODY_TYPES
143
+ ]
144
+ if len(contained_visuals) < min_visual_count:
145
+ continue
146
+
147
+ layout_area = (layout_bbox[2] - layout_bbox[0]) * (layout_bbox[3] - layout_bbox[1])
148
+ contained_area = sum(
149
+ (block_bbox[2] - block_bbox[0]) * (block_bbox[3] - block_bbox[1])
150
+ for _, block, block_bbox in contained_blocks
151
+ if block.get("type") in LOCAL_LAYOUT_IMAGE_BLOCK_AREA_TYPES
152
+ )
153
+ coverage_ratio = contained_area / layout_area
154
+ if coverage_ratio < coverage_threshold and not math.isclose(
155
+ coverage_ratio,
156
+ coverage_threshold,
157
+ rel_tol=0.0,
158
+ abs_tol=1e-12,
159
+ ):
160
+ continue
161
+
162
+ contained_block_ids = {id(block) for _, block, _ in contained_blocks}
163
+ first_block_index = min(block_index for block_index, _, _ in contained_blocks)
164
+ candidates.append(
165
+ (
166
+ -len(contained_visuals),
167
+ layout_area,
168
+ layout_order,
169
+ first_block_index,
170
+ layout_block,
171
+ contained_block_ids,
172
+ )
173
+ )
174
+
175
+ claimed_block_ids: set[int] = set()
176
+ selected_containers: list[tuple[int, dict[str, Any]]] = []
177
+ for _, _, _, first_block_index, layout_block, block_ids in sorted(candidates):
178
+ if claimed_block_ids.intersection(block_ids):
179
+ continue
180
+ claimed_block_ids.update(block_ids)
181
+ selected_containers.append(
182
+ (
183
+ first_block_index,
184
+ {
185
+ "type": "image_block",
186
+ "bbox": list(layout_block["bbox"]),
187
+ "angle": layout_block.get("angle", 0),
188
+ "content": None,
189
+ },
190
+ )
191
+ )
192
+
193
+ for insert_index, image_block in sorted(selected_containers, reverse=True):
194
+ page_model_list.insert(insert_index, image_block)
195
+
196
+
197
+ def _attach_visual_block_images(
198
+ model_list: list[list[dict[str, Any]]],
199
+ images_list: list[dict[str, Any]],
200
+ page_start_index: int = 0,
201
+ ) -> None:
202
+ """在窗口页图释放前,为最终 model_list 视觉块写入回正后的页面裁图。"""
203
+ if len(model_list) != len(images_list):
204
+ raise ValueError(f"Hybrid visual crop page count mismatch: model_list={len(model_list)}, images={len(images_list)}")
205
+
206
+ for page_offset, (page_model_list, image_dict) in enumerate(zip(model_list, images_list)):
207
+ _attach_prepared_visual_block_images(
208
+ [_prepare_page_visual_blocks(page_model_list)], [image_dict], page_start_index + page_offset
209
+ )
210
+
211
+
212
+ def _prepare_page_visual_blocks(page_model_list: list[dict[str, Any]]) -> list[tuple[int, dict[str, Any]]]:
213
+ """先折叠视觉容器,再记录原始块索引,供按需页图任务复用。"""
214
+ _collapse_image_blocks(page_model_list)
215
+ return [(index, block) for index, block in enumerate(page_model_list) if block.get("type") in MODEL_JSON_VISUAL_BLOCK_TYPES]
216
+
217
+
218
+ def _visual_page_ranges(
219
+ prepared_pages: list[list[tuple[int, dict[str, Any]]]],
220
+ image_bytes_by_page: dict[int, int] | None = None,
221
+ *,
222
+ window_size: int = 64,
223
+ ) -> list[tuple[int, int]]:
224
+ """在指定窗口和 32MiB 像素预算内合并需裁图页,不改变单页清晰度。"""
225
+ ranges: list[tuple[int, int]] = []
226
+ batch_bytes = 0
227
+ for page_index, blocks in enumerate(prepared_pages):
228
+ if not blocks:
229
+ continue
230
+ page_bytes = (image_bytes_by_page or {}).get(page_index, 0)
231
+ if (
232
+ ranges
233
+ and page_index == ranges[-1][1] + 1
234
+ and page_index // window_size == ranges[-1][0] // window_size
235
+ and batch_bytes + page_bytes <= 32 * 1024 * 1024
236
+ ):
237
+ ranges[-1] = (ranges[-1][0], page_index)
238
+ batch_bytes += page_bytes
239
+ else:
240
+ ranges.append((page_index, page_index))
241
+ batch_bytes = page_bytes
242
+ return ranges
243
+
244
+
245
+ def attach_visual_block_images_from_pdf(
246
+ document: PDFDocument,
247
+ model_list: list[list[dict[str, Any]]],
248
+ *,
249
+ window_size: int = 64,
250
+ timeout: int | None = None,
251
+ threads: int | None = None,
252
+ ) -> None:
253
+ """按当前 PDF 全部物理页的视觉块需求原地补图;页图由本函数释放,文档仍归调用方。"""
254
+ from .images import load_images_from_pdf_bytes_range
255
+ from .raster import estimate_page_image_bytes
256
+
257
+ if isinstance(window_size, bool) or not isinstance(window_size, int) or window_size <= 0:
258
+ raise ValueError("window_size must be a positive integer")
259
+ if len(model_list) != document.page_count:
260
+ raise ValueError(f"PDF visual crop page count mismatch: model_list={len(model_list)}, document={document.page_count}")
261
+
262
+ prepared_visuals = [_prepare_page_visual_blocks(page) for page in model_list]
263
+ image_bytes = {
264
+ index: estimate_page_image_bytes(document.page_size(index)) for index, blocks in enumerate(prepared_visuals) if blocks
265
+ }
266
+ for start, end in _visual_page_ranges(prepared_visuals, image_bytes, window_size=window_size):
267
+ images = load_images_from_pdf_bytes_range(
268
+ document.bytes,
269
+ start_page_id=start,
270
+ end_page_id=end,
271
+ image_type="pil_img",
272
+ timeout=timeout,
273
+ threads=threads,
274
+ )
275
+ try:
276
+ _attach_prepared_visual_block_images(prepared_visuals[start : end + 1], images, page_start_index=start)
277
+ finally:
278
+ for item in images:
279
+ if item.get("img_pil") is not None:
280
+ item["img_pil"].close()
281
+
282
+
283
+ def _attach_prepared_visual_block_images(
284
+ prepared_pages: list[list[tuple[int, dict[str, Any]]]],
285
+ images_list: list[dict[str, Any]],
286
+ page_start_index: int = 0,
287
+ ) -> None:
288
+ """按所选 PDF 的物理页索引裁图,输入块已经整理且无需再次折叠。"""
289
+ if len(prepared_pages) != len(images_list):
290
+ raise ValueError(f"Hybrid visual crop page count mismatch: model_list={len(prepared_pages)}, images={len(images_list)}")
291
+ for page_offset, (visual_blocks, image_dict) in enumerate(zip(prepared_pages, images_list)):
292
+ if not visual_blocks:
293
+ continue
294
+
295
+ page_index = page_start_index + page_offset
296
+ page_pil_image = image_dict.get("img_pil")
297
+ if page_pil_image is None:
298
+ logger.warning(f"Skipping model visual block crops without page image: page={page_index}")
299
+ continue
300
+
301
+ converted_page_image = None
302
+ try:
303
+ if getattr(page_pil_image, "mode", None) == "RGB":
304
+ page_rgb_image = page_pil_image
305
+ else:
306
+ converted_page_image = page_pil_image.convert("RGB")
307
+ page_rgb_image = converted_page_image
308
+
309
+ page_size = _normalize_page_size(page_rgb_image)
310
+ np_image = np.asarray(page_rgb_image)
311
+ for block_idx, block in visual_blocks:
312
+ try:
313
+ pixel_bbox = _bbox_to_pixel_bbox(block.get("bbox"), page_size)
314
+ if pixel_bbox is None:
315
+ raise ValueError("invalid bbox")
316
+ angle = _normalize_visual_block_angle(block.get("angle", 0))
317
+ image_base64 = _encode_page_crop_as_jpeg_data_uri(
318
+ np_image,
319
+ pixel_bbox,
320
+ angle,
321
+ )
322
+ if not image_base64:
323
+ raise ValueError("empty crop or JPEG encoding failure")
324
+ block["image_base64"] = image_base64
325
+ except Exception as exc:
326
+ logger.warning(
327
+ "Skipping invalid model visual block crop: "
328
+ f"page={page_index}, block={block_idx}, type={block.get('type')}, "
329
+ f"bbox={block.get('bbox')}, error={exc}"
330
+ )
331
+ finally:
332
+ if converted_page_image is not None:
333
+ converted_page_image.close()
334
+
335
+
336
+ attach_visual_block_images = _attach_visual_block_images
337
+ supplement_missing_image_block_containers = _supplement_missing_image_block_containers
338
+
339
+ __all__ = [
340
+ "attach_visual_block_images",
341
+ "attach_visual_block_images_from_pdf",
342
+ "supplement_missing_image_block_containers",
343
+ ]
@@ -0,0 +1,92 @@
1
+ """文件输入准备与有效页范围,不包含档位或 OCR 路由。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+ from typing import TYPE_CHECKING, cast
8
+
9
+ from docvortex.document.contracts import HtmlSourceContext
10
+ from ..errors import InvalidRequestError
11
+ from ..schema import FILE_SUFFIXES, FileSuffix
12
+
13
+ if TYPE_CHECKING:
14
+ from .pdf.document import PDFDocument
15
+
16
+
17
+ @dataclass(slots=True)
18
+ class PreparedSource:
19
+ """保存输入字节、有效页面映射及底层文档的资源所有权。"""
20
+
21
+ data: bytes
22
+ file_suffix: FileSuffix
23
+ source_context: HtmlSourceContext | None = None
24
+ page_index_map: list[int] | None = None
25
+ broken_page_indices: tuple[int, ...] = ()
26
+ document: PDFDocument | None = None
27
+ owns_document: bool = False
28
+
29
+
30
+ def prepare_source(
31
+ source: str | Path | bytes | PDFDocument,
32
+ *,
33
+ file_suffix: FileSuffix | None = None,
34
+ page_range: str = "",
35
+ source_context: HtmlSourceContext | None = None,
36
+ ) -> PreparedSource:
37
+ """准备原生解析输入,调用者持有的 PDFDocument 不由引擎关闭。"""
38
+ from .page_range import normalize_page_range_input, parse_page_range
39
+
40
+ document = None
41
+ path = Path(source) if isinstance(source, (str, Path)) else None
42
+ if path is not None:
43
+ data = path.read_bytes()
44
+ elif isinstance(source, bytes):
45
+ data = source
46
+ else:
47
+ from .pdf.document import PDFDocument
48
+
49
+ if not isinstance(source, PDFDocument):
50
+ raise TypeError("source must be a path, bytes, or PDFDocument")
51
+ document, data, file_suffix = source, source.bytes, "pdf"
52
+ if file_suffix:
53
+ suffix = file_suffix
54
+ else:
55
+ from .detection import guess_suffix_by_bytes
56
+
57
+ suffix = guess_suffix_by_bytes(data, str(path) if path else None)
58
+ if suffix not in FILE_SUFFIXES:
59
+ raise InvalidRequestError("file_type_unsupported", f"Unsupported native input format: {suffix}", "file_suffix")
60
+ page_range = normalize_page_range_input(page_range)
61
+ if suffix != "pdf" and page_range:
62
+ raise InvalidRequestError("page_range_invalid", "Page ranges are supported only for PDF input", "page_range")
63
+ if suffix == "html" and source_context is None and path is not None:
64
+ absolute = path.resolve()
65
+ source_context = HtmlSourceContext(source_uri=absolute.as_uri(), local_resource_root=absolute.parent)
66
+ prepared = PreparedSource(data, cast(FileSuffix, suffix), source_context=source_context, document=document)
67
+ if suffix == "pdf":
68
+ from .pdf.document import PDFDocument
69
+ from .pdf.pdfium import safe_rewrite_pdf_bytes_with_pdfium_result
70
+
71
+ if document is None:
72
+ document = PDFDocument(data)
73
+ prepared.document, prepared.owns_document = document, True
74
+ try:
75
+ indices = parse_page_range(page_range, document.page_count)
76
+ if indices != list(range(document.page_count)):
77
+ rewrite = safe_rewrite_pdf_bytes_with_pdfium_result(data, page_indices=indices)
78
+ prepared.data = rewrite.pdf_bytes or data
79
+ prepared.page_index_map = None if rewrite.used_original else rewrite.retained_page_indices
80
+ prepared.broken_page_indices = tuple(rewrite.broken_page_indices)
81
+ if prepared.owns_document:
82
+ document.close()
83
+ prepared.document = PDFDocument(prepared.data)
84
+ prepared.owns_document = True
85
+ except Exception:
86
+ if prepared.owns_document and prepared.document is not None:
87
+ prepared.document.close()
88
+ raise
89
+ return prepared
90
+
91
+
92
+ __all__ = ["PreparedSource", "prepare_source", "HtmlSourceContext"]
docvortex/errors.py ADDED
@@ -0,0 +1,21 @@
1
+ """独立文档引擎的可定位错误,不依赖宿主 HTTP 或任务状态。"""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class DocumentError(ValueError):
7
+ """携带稳定错误码的文档输入或处理错误。"""
8
+
9
+ def __init__(self, code: str, message: str, param: str | None = None) -> None:
10
+ """保存供命令行及上层应用映射的错误信息。"""
11
+ super().__init__(message)
12
+ self.code = code
13
+ self.message = message
14
+ self.param = param
15
+
16
+
17
+ class InvalidRequestError(DocumentError):
18
+ """表示页范围或格式选项不符合文档输入契约。"""
19
+
20
+
21
+ __all__ = ["DocumentError", "InvalidRequestError"]
@@ -0,0 +1,3 @@
1
+ """DocVortex 模块边界。"""
2
+
3
+ __all__ = []