docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,315 @@
1
+ """PDF 字符去重及原始文字几何提取,保持原生提取算法与资源语义。"""
2
+
3
+ from __future__ import annotations
4
+ import logging
5
+ import math
6
+ from typing import Any, Iterator, cast
7
+ import pypdfium2 as pdfium
8
+ import pypdfium2.raw as pdfium_c
9
+ from .text.extract import get_chars, deduplicate_chars
10
+ from .text.contracts import Char
11
+ from .text.geometry import char_bbox_values as _char_bbox_values
12
+
13
+ from .native_contracts import (
14
+ NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE,
15
+ OFFSET_DUPLICATE_CHAR_BBOX_TOLERANCE,
16
+ OFFSET_DUPLICATE_MIN_BBOX_OVERLAP_RATIO,
17
+ OFFSET_DUPLICATE_TRANSLATION_TOLERANCE,
18
+ PDFPageImage,
19
+ PDFPageTextGeometry,
20
+ )
21
+ from .native_lifecycle import _try_close
22
+
23
+ logger = logging.getLogger("docvortex.document.pdf.document")
24
+
25
+
26
+ def _get_visible_char_signature(
27
+ char: Char,
28
+ ) -> tuple[str, tuple[Any, Any, Any, Any], float]:
29
+ """生成可见字符去重签名,不把 bbox 放入签名以便单独做近重合判断。"""
30
+ font = char.get("font") or {}
31
+ font_key = (
32
+ font.get("name"),
33
+ font.get("flags"),
34
+ font.get("size"),
35
+ font.get("weight"),
36
+ )
37
+ rotation_key = round(float(char.get("rotation") or 0.0), 3)
38
+ return str(char.get("char", "")), font_key, rotation_key
39
+
40
+
41
+ def _is_near_identical_bbox(
42
+ bbox_a: tuple[float, float, float, float],
43
+ bbox_b: tuple[float, float, float, float],
44
+ ) -> bool:
45
+ """判断两个字符 bbox 是否属于同一视觉位置的一点内抖动。"""
46
+ return all(abs(coord_a - coord_b) <= NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE for coord_a, coord_b in zip(bbox_a, bbox_b))
47
+
48
+
49
+ def _calculate_bbox_overlap_in_smaller_area(
50
+ bbox_a: tuple[float, float, float, float],
51
+ bbox_b: tuple[float, float, float, float],
52
+ ) -> float:
53
+ """计算两个字符框交集占较小字符框面积的比例。"""
54
+ intersection_width = max(
55
+ 0.0,
56
+ min(bbox_a[2], bbox_b[2]) - max(bbox_a[0], bbox_b[0]),
57
+ )
58
+ intersection_height = max(
59
+ 0.0,
60
+ min(bbox_a[3], bbox_b[3]) - max(bbox_a[1], bbox_b[1]),
61
+ )
62
+ bbox_a_area = max(0.0, bbox_a[2] - bbox_a[0]) * max(
63
+ 0.0,
64
+ bbox_a[3] - bbox_a[1],
65
+ )
66
+ bbox_b_area = max(0.0, bbox_b[2] - bbox_b[0]) * max(
67
+ 0.0,
68
+ bbox_b[3] - bbox_b[1],
69
+ )
70
+ smaller_area = min(bbox_a_area, bbox_b_area)
71
+ if smaller_area == 0:
72
+ return 0.0
73
+ return intersection_width * intersection_height / smaller_area
74
+
75
+
76
+ def _is_adjacent_offset_duplicate_char(
77
+ previous_char: Char,
78
+ current_char: Char,
79
+ ) -> bool:
80
+ """识别相邻字符中由对角平移阴影产生的第二个重复字符。"""
81
+ if _get_visible_char_signature(previous_char) != _get_visible_char_signature(current_char):
82
+ return False
83
+
84
+ previous_bbox = _char_bbox_values(previous_char.get("bbox"))
85
+ current_bbox = _char_bbox_values(current_char.get("bbox"))
86
+ if previous_bbox is None or current_bbox is None:
87
+ return False
88
+
89
+ x_start_offset = current_bbox[0] - previous_bbox[0]
90
+ y_start_offset = current_bbox[1] - previous_bbox[1]
91
+ x_end_offset = current_bbox[2] - previous_bbox[2]
92
+ y_end_offset = current_bbox[3] - previous_bbox[3]
93
+
94
+ # 阴影层应是同一字符框的刚性平移,避免把大小不同的相邻同字误判为重复。
95
+ if (
96
+ abs(x_start_offset - x_end_offset) > OFFSET_DUPLICATE_TRANSLATION_TOLERANCE
97
+ or abs(y_start_offset - y_end_offset) > OFFSET_DUPLICATE_TRANSLATION_TOLERANCE
98
+ ):
99
+ return False
100
+
101
+ if not (
102
+ NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE < abs(x_start_offset) <= OFFSET_DUPLICATE_CHAR_BBOX_TOLERANCE
103
+ and NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE < abs(y_start_offset) <= OFFSET_DUPLICATE_CHAR_BBOX_TOLERANCE
104
+ ):
105
+ return False
106
+
107
+ return _calculate_bbox_overlap_in_smaller_area(previous_bbox, current_bbox) >= OFFSET_DUPLICATE_MIN_BBOX_OVERLAP_RATIO
108
+
109
+
110
+ def _get_near_identical_bbox_bucket_key(
111
+ bbox_coords: tuple[float, float, float, float],
112
+ ) -> tuple[int, int]:
113
+ """按字符 bbox 左上角生成空间桶 key,缩小近重合判断的候选范围。"""
114
+ return (
115
+ math.floor(bbox_coords[0] / NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE),
116
+ math.floor(bbox_coords[1] / NEAR_IDENTICAL_CHAR_BBOX_TOLERANCE),
117
+ )
118
+
119
+
120
+ def _iter_neighbor_bbox_bucket_keys(
121
+ bucket_key: tuple[int, int],
122
+ ) -> Iterator[tuple[int, int]]:
123
+ """遍历当前桶及周围 8 个邻近桶,覆盖 bbox 容差范围内的候选字符。"""
124
+ bucket_x, bucket_y = bucket_key
125
+ for offset_x in (-1, 0, 1):
126
+ for offset_y in (-1, 0, 1):
127
+ yield bucket_x + offset_x, bucket_y + offset_y
128
+
129
+
130
+ def _deduplicate_near_identical_chars(chars: list[Char]) -> list[Char]:
131
+ """清理 PDFium 文本层边界处同字符、同位置及对角阴影重复字符。"""
132
+ seen_visible_char_bboxes: dict[
133
+ tuple[str, tuple[Any, Any, Any, Any], float],
134
+ dict[tuple[int, int], list[tuple[float, float, float, float]]],
135
+ ] = {}
136
+ deduplicated_chars: list[Char] = []
137
+
138
+ for char in chars:
139
+ text = str(char.get("char", ""))
140
+ if not text or text.isspace():
141
+ deduplicated_chars.append(char)
142
+ continue
143
+
144
+ visible_char_key = _get_visible_char_signature(char)
145
+ bbox_coords = _char_bbox_values(char.get("bbox"))
146
+ if bbox_coords is None:
147
+ deduplicated_chars.append(char)
148
+ continue
149
+
150
+ if deduplicated_chars and _is_adjacent_offset_duplicate_char(
151
+ deduplicated_chars[-1],
152
+ char,
153
+ ):
154
+ continue
155
+
156
+ bbox_bucket_key = _get_near_identical_bbox_bucket_key(bbox_coords)
157
+ visible_char_bbox_buckets = seen_visible_char_bboxes.setdefault(
158
+ visible_char_key,
159
+ {},
160
+ )
161
+ if any(
162
+ _is_near_identical_bbox(bbox_coords, seen_bbox)
163
+ for neighbor_bucket_key in _iter_neighbor_bbox_bucket_keys(bbox_bucket_key)
164
+ for seen_bbox in visible_char_bbox_buckets.get(neighbor_bucket_key, [])
165
+ ):
166
+ continue
167
+
168
+ visible_char_bbox_buckets.setdefault(bbox_bucket_key, []).append(bbox_coords)
169
+ deduplicated_chars.append(char)
170
+
171
+ return deduplicated_chars
172
+
173
+
174
+ def _restore_pdfium_surrogate_pairs(
175
+ chars: list[Char],
176
+ textpage: pdfium.PdfTextPage,
177
+ *,
178
+ raw_codes: dict[int, int] | None = None,
179
+ ) -> list[Char]:
180
+ """利用 PDFium 原始 UTF-16 code unit 恢复 pdftext 丢失的补充平面字符。"""
181
+ if not any(
182
+ len(text := str(char.get("char", ""))) == 1 and (text == "\ufffd" or 0xD800 <= ord(text) <= 0xDFFF) for char in chars
183
+ ):
184
+ return chars
185
+
186
+ try:
187
+ textpage_raw = textpage.raw
188
+ char_count = int(textpage.count_chars())
189
+ except Exception:
190
+ textpage_raw = None
191
+ char_count = 0
192
+
193
+ restored_chars: list[Char] = []
194
+ consumed_char_indices: set[int] = set()
195
+
196
+ def get_unicode(handle: object, index: int) -> int:
197
+ """优先复用首次读取的原始码值,仅为独立辅助调用读取原生接口。"""
198
+ return raw_codes[index] if raw_codes is not None else int(pdfium_c.FPDFText_GetUnicode(handle, index))
199
+
200
+ for char in chars:
201
+ text = str(char.get("char", ""))
202
+ raw_char_idx = char.get("char_idx")
203
+ try:
204
+ char_idx = int(raw_char_idx) if raw_char_idx is not None else -1
205
+ except (TypeError, ValueError):
206
+ char_idx = -1
207
+
208
+ if char_idx in consumed_char_indices:
209
+ continue
210
+
211
+ raw_code = None
212
+ if (
213
+ textpage_raw is not None
214
+ and 0 <= char_idx < char_count
215
+ and len(text) == 1
216
+ and (text == "\ufffd" or 0xD800 <= ord(text) <= 0xDFFF)
217
+ ):
218
+ raw_code = int(get_unicode(textpage_raw, char_idx))
219
+
220
+ high_surrogate = None
221
+ low_surrogate = None
222
+ if raw_code is not None and 0xD800 <= raw_code <= 0xDBFF and char_idx + 1 < char_count:
223
+ next_code = int(get_unicode(textpage_raw, char_idx + 1))
224
+ if 0xDC00 <= next_code <= 0xDFFF:
225
+ high_surrogate = raw_code
226
+ low_surrogate = next_code
227
+ consumed_char_indices.add(char_idx + 1)
228
+ elif raw_code is not None and 0xDC00 <= raw_code <= 0xDFFF and char_idx > 0:
229
+ previous_code = int(get_unicode(textpage_raw, char_idx - 1))
230
+ if 0xD800 <= previous_code <= 0xDBFF:
231
+ high_surrogate = previous_code
232
+ low_surrogate = raw_code
233
+
234
+ if high_surrogate is not None and low_surrogate is not None:
235
+ restored_char = cast(Char, dict(char))
236
+ restored_char["char"] = chr(0x10000 + ((high_surrogate - 0xD800) << 10) + (low_surrogate - 0xDC00))
237
+ restored_char["source_indices"] = tuple(
238
+ sorted(
239
+ set(
240
+ (
241
+ *char.get("source_indices", (char_idx,)),
242
+ char_idx + 1 if raw_code is not None and raw_code <= 0xDBFF else char_idx - 1,
243
+ )
244
+ )
245
+ )
246
+ )
247
+ restored_chars.append(restored_char)
248
+ continue
249
+
250
+ if len(text) == 1 and 0xD800 <= ord(text) <= 0xDFFF:
251
+ restored_char = cast(Char, dict(char))
252
+ restored_char["char"] = "\ufffd"
253
+ restored_chars.append(restored_char)
254
+ continue
255
+
256
+ restored_chars.append(char)
257
+
258
+ return restored_chars
259
+
260
+
261
+ def _page_to_image(page: pdfium.PdfPage, scale: float, max_edge: int) -> PDFPageImage:
262
+ """按原缩放与长边上限复制页面像素,并返回独立持有的图片。"""
263
+ long_edge_length = max(*page.get_size())
264
+ if (long_edge_length * scale) > max_edge:
265
+ scale = max_edge / long_edge_length
266
+
267
+ bitmap = None
268
+ try:
269
+ bitmap = page.render(scale=scale) # type: ignore
270
+ bitmap = cast(pdfium.PdfBitmap, bitmap)
271
+ pil_image = bitmap.to_pil()
272
+ finally:
273
+ _try_close(bitmap)
274
+
275
+ return PDFPageImage(pil_image=pil_image, scale=scale)
276
+
277
+
278
+ def _extract_page_text_geometry(
279
+ page: pdfium.PdfPage,
280
+ *,
281
+ include_extended_geometry: bool,
282
+ ) -> PDFPageTextGeometry:
283
+ """在调用方持有的页面和锁内读取字符,使批量提取与独立接口共用实现。"""
284
+ textpage = None
285
+ try:
286
+ textpage = page.get_textpage()
287
+ raw_page_bbox: list[float] = list(page.get_bbox())
288
+ page_rotation: int = 0
289
+ try:
290
+ page_rotation = page.get_rotation()
291
+ except Exception:
292
+ pass
293
+ chars = get_chars(textpage, raw_page_bbox, page_rotation, include_geometry=include_extended_geometry)
294
+ raw_codes = {char["char_idx"]: char["raw_code"] for char in chars}
295
+ chars = deduplicate_chars(chars)
296
+ chars = _restore_pdfium_surrogate_pairs(chars, textpage, raw_codes=raw_codes)
297
+ chars = _deduplicate_near_identical_chars(chars)
298
+ if include_extended_geometry:
299
+ loose_bboxes = {
300
+ char["char_idx"]: char["loose_bbox"]
301
+ for char in chars
302
+ if char.get("loose_bbox") is not None and abs(char["rotation"]) > 1e-9
303
+ }
304
+ tight_bboxes = {char["char_idx"]: char["tight_bbox"] for char in chars if char.get("tight_bbox") is not None}
305
+ origins = {char["char_idx"]: char["origin"] for char in chars if char.get("origin") is not None}
306
+ else:
307
+ loose_bboxes, tight_bboxes, origins = {}, {}, {}
308
+ finally:
309
+ _try_close(textpage)
310
+ return PDFPageTextGeometry(
311
+ chars=chars,
312
+ tight_bboxes=tight_bboxes,
313
+ origins=origins,
314
+ loose_bboxes=loose_bboxes,
315
+ )
@@ -0,0 +1,325 @@
1
+ import os
2
+ import threading
3
+ from contextlib import contextmanager
4
+ from dataclasses import dataclass, field
5
+ from io import BytesIO
6
+ from typing import Any, Iterator, Sequence, TypeVar
7
+
8
+ from loguru import logger
9
+
10
+ from .font_runtime import PdfiumFontError, PdfiumRuntimeInfo, _FontProvider
11
+
12
+ _pdfium_lock = threading.RLock()
13
+ _font_provider: _FontProvider | None = None
14
+
15
+ T = TypeVar("T")
16
+
17
+
18
+ def _check_runtime_process() -> None:
19
+ """拒绝复用 fork 继承的字体接口;渲染 worker 应通过 spawn/forkserver 独立初始化。"""
20
+ if _font_provider is not None and _font_provider.pid != os.getpid():
21
+ raise PdfiumFontError("Inherited PDFium font runtime: use spawn/forkserver instead of forking after PDF use")
22
+
23
+
24
+ def initialize_pdfium_runtime() -> PdfiumRuntimeInfo:
25
+ """幂等安装本进程的固定 CJK 字体提供器;必须先于首次 PDF 字体使用。"""
26
+ global _font_provider
27
+ _check_runtime_process()
28
+ with _pdfium_lock:
29
+ if _font_provider is None:
30
+ import pypdfium2 as pdfium
31
+ import pypdfium2.raw as raw
32
+
33
+ provider = _FontProvider(raw, str(pdfium.PDFIUM_INFO))
34
+ # 注册前保留强引用,保证即使安装阶段回调失败,C 指针也不会指向被回收的对象。
35
+ _font_provider = provider
36
+ provider.install()
37
+ _font_provider.raise_if_failed()
38
+ return _font_provider.info
39
+
40
+
41
+ @dataclass
42
+ class PdfiumRewriteResult:
43
+ """记录 PDFium 安全重写结果,供调用方按实际保留页修正原始页号。"""
44
+
45
+ pdf_bytes: bytes
46
+ retained_page_indices: list[int] | None = None
47
+ broken_page_indices: list[int] = field(default_factory=list)
48
+ used_original: bool = False
49
+
50
+
51
+ @contextmanager
52
+ def pdfium_guard() -> Iterator[None]:
53
+ """串行化访问并确保字体运行时就绪,在原生栈退出后传播字体故障。"""
54
+ _check_runtime_process()
55
+ with _pdfium_lock:
56
+ initialize_pdfium_runtime()
57
+ assert _font_provider is not None
58
+ try:
59
+ yield
60
+ finally:
61
+ _font_provider.raise_if_failed()
62
+
63
+
64
+ def close_pdfium_document(pdf_doc: Any) -> None:
65
+ """清理时只持锁,不安装字体、不重建已销毁的 PDFium 运行时。"""
66
+ if pdf_doc is None:
67
+ return
68
+ with _pdfium_lock:
69
+ pdf_doc.close()
70
+
71
+
72
+ def close_pdfium_child(pdfium_obj: Any) -> None:
73
+ """显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。"""
74
+ if pdfium_obj is None:
75
+ return
76
+ close = getattr(pdfium_obj, "close", None)
77
+ if callable(close):
78
+ with _pdfium_lock:
79
+ close()
80
+
81
+
82
+ def close_pdfium_objects_safely(*pdfium_objs: object, owner: str = "pdfium cleanup") -> None:
83
+ """清理多个 PDFium 对象时逐个尝试关闭,避免前一个关闭失败阻断后续对象释放。"""
84
+ for pdfium_obj in pdfium_objs:
85
+ if pdfium_obj is None:
86
+ continue
87
+ try:
88
+ close_pdfium_child(pdfium_obj)
89
+ except Exception as exc:
90
+ logger.warning(f"Failed to close PDFium object during {owner}: {exc}")
91
+
92
+
93
+ def get_loadable_pdfium_page_indices(
94
+ src_pdf_bytes: bytes,
95
+ start_page_id: int = 0,
96
+ end_page_id: int | None = None,
97
+ ) -> tuple[list[int], list[int]]:
98
+ """逐页探测 PDFium 可加载页面,返回可保留页和损坏页的 0-based 索引。"""
99
+ import pypdfium2 as pdfium
100
+
101
+ loadable_page_indices = []
102
+ broken_page_indices = []
103
+ pdf_doc = None
104
+
105
+ try:
106
+ with pdfium_guard():
107
+ pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
108
+ total_page_count = len(pdf_doc)
109
+ if total_page_count == 0:
110
+ return [], []
111
+
112
+ normalized_start_page_id = max(0, start_page_id)
113
+ normalized_end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else total_page_count - 1
114
+ if normalized_end_page_id > total_page_count - 1:
115
+ normalized_end_page_id = total_page_count - 1
116
+ if normalized_start_page_id > normalized_end_page_id:
117
+ return [], []
118
+
119
+ for page_index in range(
120
+ normalized_start_page_id,
121
+ normalized_end_page_id + 1,
122
+ ):
123
+ page = None
124
+ try:
125
+ page = pdf_doc[page_index]
126
+ page.get_size()
127
+ loadable_page_indices.append(page_index)
128
+ except Exception:
129
+ broken_page_indices.append(page_index)
130
+ finally:
131
+ close_pdfium_child(page)
132
+ finally:
133
+ close_pdfium_document(pdf_doc)
134
+
135
+ return loadable_page_indices, broken_page_indices
136
+
137
+
138
+ def _normalize_rewrite_page_indices(
139
+ total_page_count: int,
140
+ start_page_id: int = 0,
141
+ end_page_id: int | None = None,
142
+ page_indices: Sequence[int] | None = None,
143
+ ) -> list[int]:
144
+ """按 rewrite_pdf_bytes_with_pdfium 的规则归一化实际导出的 0-based 页号。"""
145
+ if total_page_count == 0:
146
+ return []
147
+
148
+ if page_indices is not None:
149
+ return sorted({int(page_index) for page_index in page_indices if 0 <= int(page_index) < total_page_count})
150
+
151
+ normalized_start_page_id = max(0, start_page_id)
152
+ normalized_end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else total_page_count - 1
153
+ if normalized_end_page_id > total_page_count - 1:
154
+ normalized_end_page_id = total_page_count - 1
155
+ if normalized_start_page_id > normalized_end_page_id:
156
+ return []
157
+ return list(range(normalized_start_page_id, normalized_end_page_id + 1))
158
+
159
+
160
+ def _get_rewrite_page_indices_from_pdf(
161
+ src_pdf_bytes: bytes,
162
+ start_page_id: int = 0,
163
+ end_page_id: int | None = None,
164
+ page_indices: Sequence[int] | None = None,
165
+ ) -> list[int]:
166
+ """读取源 PDF 页数并计算本次重写会保留的原始页号。"""
167
+ import pypdfium2 as pdfium
168
+
169
+ pdf_doc = None
170
+ try:
171
+ with pdfium_guard():
172
+ pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
173
+ return _normalize_rewrite_page_indices(
174
+ len(pdf_doc),
175
+ start_page_id=start_page_id,
176
+ end_page_id=end_page_id,
177
+ page_indices=page_indices,
178
+ )
179
+ finally:
180
+ close_pdfium_document(pdf_doc)
181
+
182
+
183
+ def rewrite_pdf_bytes_with_pdfium(
184
+ src_pdf_bytes: bytes,
185
+ start_page_id: int = 0,
186
+ end_page_id: int | None = None,
187
+ page_indices: Sequence[int] | None = None,
188
+ ) -> bytes:
189
+ import pypdfium2 as pdfium
190
+
191
+ pdf_doc = None
192
+ output_doc = None
193
+ try:
194
+ with pdfium_guard():
195
+ pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
196
+ total_page_count = len(pdf_doc)
197
+ if total_page_count == 0:
198
+ return b""
199
+
200
+ normalized_page_indices = _normalize_rewrite_page_indices(
201
+ total_page_count,
202
+ start_page_id=start_page_id,
203
+ end_page_id=end_page_id,
204
+ page_indices=page_indices,
205
+ )
206
+ if not normalized_page_indices:
207
+ return b""
208
+
209
+ output_doc = pdfium.PdfDocument.new()
210
+ output_doc.import_pages(pdf_doc, normalized_page_indices)
211
+
212
+ output_buffer = BytesIO()
213
+ output_doc.save(output_buffer)
214
+ return output_buffer.getvalue()
215
+ finally:
216
+ close_pdfium_objects_safely(
217
+ output_doc,
218
+ pdf_doc,
219
+ owner="rewrite_pdf_bytes_with_pdfium",
220
+ )
221
+
222
+
223
+ def safe_rewrite_pdf_bytes_with_pdfium(
224
+ src_pdf_bytes: bytes,
225
+ start_page_id: int = 0,
226
+ end_page_id: int | None = None,
227
+ page_indices: Sequence[int] | None = None,
228
+ ) -> bytes:
229
+ """安全重写 PDF 字节;常规重写失败时跳过损坏页并保留可加载页面。"""
230
+ return safe_rewrite_pdf_bytes_with_pdfium_result(
231
+ src_pdf_bytes,
232
+ start_page_id=start_page_id,
233
+ end_page_id=end_page_id,
234
+ page_indices=page_indices,
235
+ ).pdf_bytes
236
+
237
+
238
+ def safe_rewrite_pdf_bytes_with_pdfium_result(
239
+ src_pdf_bytes: bytes,
240
+ start_page_id: int = 0,
241
+ end_page_id: int | None = None,
242
+ page_indices: Sequence[int] | None = None,
243
+ ) -> PdfiumRewriteResult:
244
+ """安全重写 PDF 字节,并返回重写后 PDF 对应的原始页号映射。"""
245
+ try:
246
+ rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium(
247
+ src_pdf_bytes,
248
+ start_page_id=start_page_id,
249
+ end_page_id=end_page_id,
250
+ page_indices=page_indices,
251
+ )
252
+ if rebuilt_pdf_bytes:
253
+ retained_page_indices = _get_rewrite_page_indices_from_pdf(
254
+ src_pdf_bytes,
255
+ start_page_id=start_page_id,
256
+ end_page_id=end_page_id,
257
+ page_indices=page_indices,
258
+ )
259
+ return PdfiumRewriteResult(
260
+ pdf_bytes=rebuilt_pdf_bytes,
261
+ retained_page_indices=retained_page_indices,
262
+ )
263
+ logger.warning("PDFium rewrite returned empty bytes, trying to skip broken pages.")
264
+ except PdfiumFontError:
265
+ raise
266
+ except Exception as fallback_error:
267
+ logger.warning(f"Error in converting PDF bytes with pdfium: {fallback_error}, trying to skip broken pages.")
268
+
269
+ try:
270
+ if page_indices is not None:
271
+ requested_page_indices = sorted({int(page_index) for page_index in page_indices if int(page_index) >= 0})
272
+ if not requested_page_indices:
273
+ logger.warning("PDFium safe rewrite received no valid requested pages, using original PDF bytes.")
274
+ return PdfiumRewriteResult(pdf_bytes=src_pdf_bytes, retained_page_indices=None, used_original=True)
275
+ probe_start_page_id = requested_page_indices[0]
276
+ probe_end_page_id = requested_page_indices[-1]
277
+ else:
278
+ requested_page_indices = None
279
+ probe_start_page_id = start_page_id
280
+ probe_end_page_id = end_page_id
281
+
282
+ loadable_page_indices, broken_page_indices = get_loadable_pdfium_page_indices(
283
+ src_pdf_bytes,
284
+ start_page_id=probe_start_page_id,
285
+ end_page_id=probe_end_page_id,
286
+ )
287
+ if requested_page_indices is not None:
288
+ requested_page_index_set = set(requested_page_indices)
289
+ loadable_page_indices = [
290
+ page_index for page_index in loadable_page_indices if page_index in requested_page_index_set
291
+ ]
292
+ broken_page_indices = [page_index for page_index in broken_page_indices if page_index in requested_page_index_set]
293
+
294
+ if broken_page_indices:
295
+ skipped_pages = [page_index + 1 for page_index in broken_page_indices]
296
+ logger.warning(f"Skipped broken PDF pages during PDFium rewrite: {skipped_pages}")
297
+ if not loadable_page_indices:
298
+ logger.warning("PDFium skip-broken-page rewrite found no loadable pages, using original PDF bytes.")
299
+ return PdfiumRewriteResult(
300
+ pdf_bytes=src_pdf_bytes,
301
+ retained_page_indices=None,
302
+ broken_page_indices=broken_page_indices,
303
+ used_original=True,
304
+ )
305
+
306
+ rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium(
307
+ src_pdf_bytes,
308
+ start_page_id=probe_start_page_id,
309
+ end_page_id=probe_end_page_id,
310
+ page_indices=loadable_page_indices,
311
+ )
312
+ if rebuilt_pdf_bytes:
313
+ return PdfiumRewriteResult(
314
+ pdf_bytes=rebuilt_pdf_bytes,
315
+ retained_page_indices=loadable_page_indices,
316
+ broken_page_indices=broken_page_indices,
317
+ )
318
+ logger.warning("PDFium skip-broken-page rewrite returned empty bytes, using original PDF bytes.")
319
+ except PdfiumFontError:
320
+ raise
321
+ except Exception as fallback_error:
322
+ logger.warning(
323
+ f"Error in converting PDF bytes with skip-broken-page fallback: {fallback_error}, using original PDF bytes."
324
+ )
325
+ return PdfiumRewriteResult(pdf_bytes=src_pdf_bytes, retained_page_indices=None, used_original=True)
@@ -0,0 +1,46 @@
1
+ from loguru import logger
2
+ from PIL import Image
3
+ from pypdfium2 import PdfBitmap, PdfPage
4
+
5
+ from .pdfium import pdfium_guard
6
+
7
+ DEFAULT_PDF_IMAGE_DPI = 200
8
+ DEFAULT_MAX_RENDER_EDGE = 3500
9
+
10
+
11
+ def estimate_page_image_bytes(page_size: tuple[float, float], dpi: int = DEFAULT_PDF_IMAGE_DPI) -> int:
12
+ """按现有渲染缩放估算四通道页图字节,用于限制批量驻留内存。"""
13
+ import math
14
+
15
+ width, height = page_size
16
+ scale = min(dpi / 72, DEFAULT_MAX_RENDER_EDGE / max(width, height))
17
+ return math.ceil(width * scale) * math.ceil(height * scale) * 4
18
+
19
+
20
+ def page_to_image(
21
+ page: PdfPage,
22
+ dpi: int = DEFAULT_PDF_IMAGE_DPI,
23
+ max_width_or_height: int = DEFAULT_MAX_RENDER_EDGE,
24
+ ) -> tuple[Image.Image, float]:
25
+ """按既有 DPI 与长边上限渲染页面,并独立持有返回图片的像素。"""
26
+ with pdfium_guard():
27
+ scale = dpi / 72
28
+
29
+ long_side_length = max(*page.get_size())
30
+ if (long_side_length * scale) > max_width_or_height:
31
+ scale = max_width_or_height / long_side_length
32
+
33
+ bitmap: PdfBitmap | None = None
34
+ try:
35
+ bitmap = page.render(scale=scale) # type: ignore
36
+ image = bitmap.to_pil().copy()
37
+ finally:
38
+ if bitmap is not None:
39
+ try:
40
+ bitmap.close()
41
+ except Exception as e:
42
+ logger.error(f"Failed to close bitmap: {e}")
43
+ return image, scale
44
+
45
+
46
+ __all__ = ["page_to_image"]