docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1138 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections import Counter
5
+ from ctypes import byref, c_int, create_string_buffer
6
+ from io import BytesIO
7
+ from typing import Any
8
+
9
+ import pypdfium2 as pdfium
10
+ import pypdfium2.raw as pdfium_c
11
+ from loguru import logger
12
+ from pypdf import PdfReader
13
+ from pypdf.generic import ContentStream
14
+
15
+ from .pdfium import PdfiumFontError, close_pdfium_child, pdfium_guard
16
+
17
+ MAX_SAMPLE_PAGES = 10
18
+ CHARS_THRESHOLD = 50
19
+ HIGH_IMAGE_COVERAGE_THRESHOLD = 0.8
20
+ TEXT_QUALITY_MIN_CHARS = 300
21
+ TEXT_QUALITY_BAD_THRESHOLD = 0.03
22
+ UNICODE_MAP_ERROR_RATIO_THRESHOLD = 0.04
23
+ CID_FONT_USAGE_RATIO_THRESHOLD = 0.01
24
+ CID_FONT_USAGE_COUNT_THRESHOLD = 30
25
+ LATIN_CJK_FONT_USAGE_RATIO_THRESHOLD = 0.01
26
+ LATIN_CJK_FONT_USAGE_COUNT_THRESHOLD = 30
27
+ LATIN_CJK_FONT_CJK_RATIO_THRESHOLD = 0.8
28
+ LATIN_CHARSET_MIN_LATIN_GLYPHS = 10
29
+ LATIN_CHARSET_MIN_LATIN_RATIO = 0.5
30
+ MAX_PAGE_ASPECT_RATIO = 10.0
31
+ SUSPICIOUS_CJK_72XX_START = 0x7280
32
+ SUSPICIOUS_CJK_72XX_END = 0x72DF
33
+ SUSPICIOUS_CJK_72XX_COUNT_THRESHOLD = 30
34
+ SUSPICIOUS_CJK_72XX_CJK_RATIO_THRESHOLD = 0.026
35
+ SUSPICIOUS_CJK_72XX_WHITELIST = set("犀犁犄犊犒犟犬犯状犷犹狂狄狈狐狗狙狞")
36
+ ASCII_PUNCT_CHARS = set("!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~")
37
+ ASCII_PUNCT_RUN_MIN_LENGTH = 4
38
+ SUSPICIOUS_ASCII_PUNCT_MIN_TEXT_CHARS = 100
39
+ SUSPICIOUS_ASCII_PUNCT_RATIO_THRESHOLD = 0.25
40
+ SUSPICIOUS_ASCII_PUNCT_RUN_RATIO_THRESHOLD = 0.10
41
+ SUSPICIOUS_CROSS_SCRIPT_MIN_TEXT_CHARS = 300
42
+ SUSPICIOUS_CROSS_SCRIPT_MIN_CJK_CHARS = 100
43
+ SUSPICIOUS_CROSS_SCRIPT_MIN_OTHER_SCRIPT_CHARS = 120
44
+ SUSPICIOUS_CROSS_SCRIPT_OTHER_SCRIPT_RATIO = 0.18
45
+ SUSPICIOUS_CROSS_SCRIPT_MIN_DENSE_SCRIPTS = 3
46
+ SUSPICIOUS_CROSS_SCRIPT_DENSE_SCRIPT_CHARS = 5
47
+ SUSPICIOUS_CROSS_SCRIPT_RANGES = (
48
+ (0x0370, 0x03FF, "Greek"),
49
+ (0x0400, 0x052F, "Cyrillic"),
50
+ (0x0600, 0x06FF, "Arabic"),
51
+ (0x0700, 0x074F, "Syriac"),
52
+ (0x0750, 0x077F, "Arabic Supplement"),
53
+ (0x0780, 0x07BF, "Thaana"),
54
+ (0x07C0, 0x07FF, "NKo"),
55
+ (0x0800, 0x083F, "Samaritan"),
56
+ (0x0840, 0x085F, "Mandaic"),
57
+ (0x0860, 0x086F, "Syriac Supplement"),
58
+ (0x0870, 0x089F, "Arabic Extended-B"),
59
+ (0x0900, 0x097F, "Devanagari"),
60
+ (0x0C80, 0x0CFF, "Kannada"),
61
+ (0x0E00, 0x0E7F, "Thai"),
62
+ (0x1000, 0x109F, "Myanmar"),
63
+ (0x1100, 0x11FF, "Hangul Jamo"),
64
+ (0x1200, 0x137F, "Ethiopic"),
65
+ (0x13A0, 0x13FF, "Cherokee"),
66
+ (0x1400, 0x167F, "Canadian Syllabics"),
67
+ (0x1800, 0x18AF, "Mongolian"),
68
+ (0x1A20, 0x1AAF, "Tai Tham"),
69
+ (0x2C00, 0x2C5F, "Glagolitic"),
70
+ (0xA000, 0xA48F, "Yi"),
71
+ )
72
+ CJK_TEXT_RANGES = (
73
+ (0x3400, 0x4DBF),
74
+ (0x4E00, 0x9FFF),
75
+ (0xF900, 0xFAFF),
76
+ (0x20000, 0x2EBEF),
77
+ )
78
+
79
+ _ALLOWED_CONTROL_CODES = {9, 10, 13}
80
+ _PRIVATE_USE_AREA_START = 0xE000
81
+ _PRIVATE_USE_AREA_END = 0xF8FF
82
+
83
+
84
+ def _is_disallowed_control_unicode(unicode_code: int) -> bool:
85
+ return (0 <= unicode_code < 32 or 127 <= unicode_code <= 159) and unicode_code not in _ALLOWED_CONTROL_CODES
86
+
87
+
88
+ def classify(pdf_doc: pdfium.PdfDocument, pdf_bytes: bytes) -> str:
89
+ """
90
+ Fast PDF classification path.
91
+
92
+ The path uses pdfium + pypdf to detect text PDFs and garbled PDFs.
93
+
94
+ Returns:
95
+ "txt" if the PDF can be parsed as text, otherwise "ocr".
96
+ """
97
+
98
+ try:
99
+ with pdfium_guard():
100
+ page_count = len(pdf_doc)
101
+ if page_count == 0:
102
+ return "ocr"
103
+
104
+ page_indices = get_sample_page_indices(page_count, MAX_SAMPLE_PAGES)
105
+ if not page_indices:
106
+ return "ocr"
107
+
108
+ extreme_page_index, extreme_ratio = get_extreme_aspect_ratio_page_pdfium(
109
+ pdf_doc,
110
+ page_indices,
111
+ )
112
+ if extreme_page_index is not None:
113
+ logger.debug(
114
+ "Classify PDF as OCR due to extreme sampled-page aspect ratio: "
115
+ f"page={extreme_page_index + 1}, ratio={extreme_ratio:.2f}"
116
+ )
117
+ return "ocr"
118
+
119
+ text_samples = _collect_pdfium_text_samples(pdf_doc, page_indices)
120
+ avg_cleaned_chars_per_page = _get_avg_cleaned_chars_per_page_from_samples(text_samples)
121
+ if avg_cleaned_chars_per_page < CHARS_THRESHOLD:
122
+ return "ocr"
123
+
124
+ unicode_map_error_signal = _get_unicode_map_error_signal_from_samples(text_samples)
125
+ if unicode_map_error_signal["unicode_map_error_ratio"] >= UNICODE_MAP_ERROR_RATIO_THRESHOLD:
126
+ logger.debug(
127
+ "Classify PDF as OCR due to PDFium Unicode map errors: "
128
+ f"errors={unicode_map_error_signal['unicode_map_error_count']}, "
129
+ f"total={unicode_map_error_signal['total_chars']}, "
130
+ f"ratio={unicode_map_error_signal['unicode_map_error_ratio']:.4f}"
131
+ )
132
+ return "ocr"
133
+
134
+ font_resource_signals = _get_font_resource_signals_pypdf(
135
+ pdf_bytes,
136
+ page_indices,
137
+ )
138
+ cid_font_usage_signal = _get_cid_font_usage_signal_from_samples(
139
+ text_samples,
140
+ font_resource_signals["cid_without_to_unicode_usage"],
141
+ )
142
+ if cid_font_usage_signal["triggered"]:
143
+ logger.debug(
144
+ "Classify PDF as OCR due to high CID font usage without ToUnicode: "
145
+ f"page={cid_font_usage_signal['page_index'] + 1}, "
146
+ f"fonts={cid_font_usage_signal['font_names']}, "
147
+ f"chars={cid_font_usage_signal['cid_font_char_count']}, "
148
+ f"total={cid_font_usage_signal['total_chars']}, "
149
+ f"ratio={cid_font_usage_signal['cid_font_usage_ratio']:.4f}"
150
+ )
151
+ return "ocr"
152
+
153
+ latin_cjk_font_usage_signal = _get_latin_font_cjk_usage_signal_from_samples(
154
+ text_samples,
155
+ font_resource_signals["latin_charset_with_to_unicode"],
156
+ count_threshold=LATIN_CJK_FONT_USAGE_COUNT_THRESHOLD,
157
+ usage_ratio_threshold=LATIN_CJK_FONT_USAGE_RATIO_THRESHOLD,
158
+ cjk_ratio_threshold=LATIN_CJK_FONT_CJK_RATIO_THRESHOLD,
159
+ )
160
+ if latin_cjk_font_usage_signal["triggered"]:
161
+ logger.debug(
162
+ "Classify PDF as OCR due to Latin CharSet font decoding as CJK: "
163
+ f"page={latin_cjk_font_usage_signal['page_index'] + 1}, "
164
+ f"fonts={latin_cjk_font_usage_signal['font_names']}, "
165
+ f"chars={latin_cjk_font_usage_signal['font_char_count']}, "
166
+ f"cjk={latin_cjk_font_usage_signal['cjk_char_count']}, "
167
+ f"total={latin_cjk_font_usage_signal['total_chars']}, "
168
+ f"usage_ratio={latin_cjk_font_usage_signal['font_usage_ratio']:.4f}, "
169
+ f"cjk_ratio={latin_cjk_font_usage_signal['font_cjk_ratio']:.4f}"
170
+ )
171
+ return "ocr"
172
+
173
+ text_quality_signal = _get_text_quality_signal_from_samples(text_samples)
174
+ total_chars = text_quality_signal["total_chars"]
175
+ abnormal_ratio = text_quality_signal["abnormal_ratio"]
176
+
177
+ if total_chars >= TEXT_QUALITY_MIN_CHARS and abnormal_ratio >= TEXT_QUALITY_BAD_THRESHOLD:
178
+ return "ocr"
179
+
180
+ u72xx_signal = _get_u72xx_text_signal_from_samples(text_samples)
181
+ if (
182
+ u72xx_signal["u72xx_count"] >= SUSPICIOUS_CJK_72XX_COUNT_THRESHOLD
183
+ and u72xx_signal["u72xx_cjk_ratio"] >= SUSPICIOUS_CJK_72XX_CJK_RATIO_THRESHOLD
184
+ ):
185
+ logger.debug(
186
+ "Classify PDF as OCR due to suspicious U+7280-U+72DF text: "
187
+ f"count={u72xx_signal['u72xx_count']}, "
188
+ f"cjk_ratio={u72xx_signal['u72xx_cjk_ratio']:.4f}"
189
+ )
190
+ return "ocr"
191
+
192
+ cross_script_signal = _get_cross_script_text_signal_from_samples(text_samples)
193
+ if cross_script_signal["triggered"]:
194
+ logger.debug(
195
+ "Classify PDF as OCR due to suspicious cross-script text: "
196
+ f"chars={cross_script_signal['total_chars']}, "
197
+ f"cjk={cross_script_signal['cjk_chars']}, "
198
+ f"suspicious={cross_script_signal['suspicious_chars']}, "
199
+ f"ratio={cross_script_signal['suspicious_ratio']:.4f}, "
200
+ f"scripts={cross_script_signal['top_scripts']}"
201
+ )
202
+ return "ocr"
203
+
204
+ ascii_punct_signal = _get_sampled_ascii_punct_signal_from_samples(text_samples)
205
+ if ascii_punct_signal["triggered"]:
206
+ logger.debug(
207
+ "Classify PDF as OCR due to suspicious sampled-page ASCII punctuation "
208
+ f"text: page={ascii_punct_signal['page_index'] + 1}, "
209
+ f"text_chars={ascii_punct_signal['cleaned_text_chars']}, "
210
+ f"ascii_punct_ratio="
211
+ f"{ascii_punct_signal['ascii_punct_ratio']:.4f}, "
212
+ f"punct_run_ratio={ascii_punct_signal['punct_run_ratio']:.4f}"
213
+ )
214
+ return "ocr"
215
+
216
+ if get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices) >= HIGH_IMAGE_COVERAGE_THRESHOLD:
217
+ return "ocr"
218
+
219
+ except PdfiumFontError:
220
+ raise
221
+ except Exception as e:
222
+ logger.error(f"Failed to classify PDF: {e}")
223
+ return "ocr"
224
+
225
+ return "txt"
226
+
227
+
228
+ def get_sample_page_indices(page_count: int, max_pages: int = MAX_SAMPLE_PAGES) -> list[int]:
229
+ if page_count <= 0 or max_pages <= 0:
230
+ return []
231
+
232
+ sample_count = min(page_count, max_pages)
233
+ if sample_count == page_count:
234
+ return list(range(page_count))
235
+ if sample_count == 1:
236
+ return [0]
237
+
238
+ indices = []
239
+ seen = set()
240
+ for i in range(sample_count):
241
+ page_index = round(i * (page_count - 1) / (sample_count - 1))
242
+ page_index = max(0, min(page_count - 1, page_index))
243
+ if page_index not in seen:
244
+ indices.append(page_index)
245
+ seen.add(page_index)
246
+
247
+ if len(indices) < sample_count:
248
+ for page_index in range(page_count):
249
+ if page_index in seen:
250
+ continue
251
+ indices.append(page_index)
252
+ seen.add(page_index)
253
+ if len(indices) == sample_count:
254
+ break
255
+
256
+ return sorted(indices)
257
+
258
+
259
+ def get_extreme_aspect_ratio_page_pdfium(
260
+ pdf_doc: pdfium.PdfDocument,
261
+ page_indices: list[int],
262
+ max_page_aspect_ratio: float = MAX_PAGE_ASPECT_RATIO,
263
+ ) -> tuple[Any, Any]:
264
+ with pdfium_guard():
265
+ for page_index in page_indices:
266
+ page = None
267
+ try:
268
+ page = pdf_doc[page_index]
269
+ page_width, page_height = page.get_size()
270
+ if page_width <= 0 or page_height <= 0:
271
+ continue
272
+
273
+ aspect_ratio = max(page_width / page_height, page_height / page_width)
274
+ if aspect_ratio > max_page_aspect_ratio:
275
+ return page_index, aspect_ratio
276
+ finally:
277
+ close_pdfium_child(page)
278
+
279
+ return None, None
280
+
281
+
282
+ def _collect_pdfium_text_sample_from_page(page_index: int, page: Any) -> dict[str, Any]:
283
+ """从单页 PDFium 对象提取纯 Python 文本统计,并在调用方释放子对象。"""
284
+ text_page = None
285
+ try:
286
+ text_page = page.get_textpage()
287
+ text = text_page.get_text_bounded()
288
+ char_count = text_page.count_chars()
289
+ null_char_count = 0
290
+ replacement_char_count = 0
291
+ control_char_count = 0
292
+ private_use_char_count = 0
293
+ unicode_map_error_count = 0
294
+ font_name_counts = {}
295
+ non_generated_char_count = 0
296
+ font_non_generated_char_counts = {}
297
+ font_non_generated_cjk_char_counts = {}
298
+
299
+ for char_index in range(char_count):
300
+ unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
301
+ is_generated = pdfium_c.FPDFText_IsGenerated(text_page, char_index) == 1
302
+ if not is_generated:
303
+ non_generated_char_count += 1
304
+
305
+ if unicode_code == 0:
306
+ null_char_count += 1
307
+ elif unicode_code == 0xFFFD:
308
+ replacement_char_count += 1
309
+ elif _is_disallowed_control_unicode(unicode_code):
310
+ control_char_count += 1
311
+ elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END:
312
+ private_use_char_count += 1
313
+
314
+ if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index):
315
+ unicode_map_error_count += 1
316
+
317
+ font_name = _normalize_pdf_font_name(_get_pdfium_char_font_name(text_page, char_index))
318
+ if font_name:
319
+ font_name_counts[font_name] = font_name_counts.get(font_name, 0) + 1
320
+ if not is_generated:
321
+ font_non_generated_char_counts[font_name] = font_non_generated_char_counts.get(font_name, 0) + 1
322
+ if _is_cjk_unicode_code(unicode_code):
323
+ font_non_generated_cjk_char_counts[font_name] = font_non_generated_cjk_char_counts.get(font_name, 0) + 1
324
+
325
+ return {
326
+ "page_index": page_index,
327
+ "text": text,
328
+ "cleaned_text": re.sub(r"\s+", "", text),
329
+ "char_count": char_count,
330
+ "null_char_count": null_char_count,
331
+ "replacement_char_count": replacement_char_count,
332
+ "control_char_count": control_char_count,
333
+ "private_use_char_count": private_use_char_count,
334
+ "unicode_map_error_count": unicode_map_error_count,
335
+ "font_name_counts": font_name_counts,
336
+ "non_generated_char_count": non_generated_char_count,
337
+ "font_non_generated_char_counts": font_non_generated_char_counts,
338
+ "font_non_generated_cjk_char_counts": font_non_generated_cjk_char_counts,
339
+ }
340
+ finally:
341
+ close_pdfium_child(text_page)
342
+
343
+
344
+ def _collect_pdfium_text_samples(pdf_doc: pdfium.PdfDocument, page_indices: list[int]) -> list[dict[str, Any]]:
345
+ """一次性收集抽样页文本统计,返回纯 Python 数据,避免缓存 PDFium 子对象。"""
346
+ text_samples = []
347
+
348
+ with pdfium_guard():
349
+ for page_index in page_indices:
350
+ page = None
351
+ try:
352
+ page = pdf_doc[page_index]
353
+ text_samples.append(_collect_pdfium_text_sample_from_page(page_index, page))
354
+ finally:
355
+ close_pdfium_child(page)
356
+
357
+ return text_samples
358
+
359
+
360
+ def _get_avg_cleaned_chars_per_page_from_samples(text_samples: list[dict[str, Any]]) -> float:
361
+ """基于已缓存的抽样页文本计算平均有效字符数。"""
362
+ cleaned_total_chars = 0
363
+
364
+ for text_sample in text_samples:
365
+ cleaned_total_chars += len(text_sample["cleaned_text"])
366
+
367
+ if not text_samples:
368
+ return 0.0
369
+ return cleaned_total_chars / len(text_samples)
370
+
371
+
372
+ def _get_text_quality_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
373
+ """基于已缓存的抽样页字符计数统计异常字符质量信号。"""
374
+ total_chars = 0
375
+ null_char_count = 0
376
+ replacement_char_count = 0
377
+ control_char_count = 0
378
+ private_use_char_count = 0
379
+
380
+ for text_sample in text_samples:
381
+ total_chars += text_sample["char_count"]
382
+ null_char_count += text_sample["null_char_count"]
383
+ replacement_char_count += text_sample["replacement_char_count"]
384
+ control_char_count += text_sample["control_char_count"]
385
+ private_use_char_count += text_sample["private_use_char_count"]
386
+
387
+ abnormal_chars = null_char_count + replacement_char_count + control_char_count + private_use_char_count
388
+
389
+ abnormal_ratio = 0.0
390
+ if total_chars > 0:
391
+ abnormal_ratio = abnormal_chars / total_chars
392
+
393
+ return {
394
+ "total_chars": total_chars,
395
+ "abnormal_ratio": abnormal_ratio,
396
+ "null_char_count": null_char_count,
397
+ "replacement_char_count": replacement_char_count,
398
+ "control_char_count": control_char_count,
399
+ "private_use_char_count": private_use_char_count,
400
+ }
401
+
402
+
403
+ def _get_unicode_map_error_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
404
+ """统计 PDFium 字符级 Unicode 映射失败比例,用于识别无法可靠抽取的乱码文本。"""
405
+ total_chars = 0
406
+ unicode_map_error_count = 0
407
+
408
+ for text_sample in text_samples:
409
+ total_chars += text_sample["char_count"]
410
+ unicode_map_error_count += text_sample["unicode_map_error_count"]
411
+
412
+ unicode_map_error_ratio = 0.0
413
+ if total_chars > 0:
414
+ unicode_map_error_ratio = unicode_map_error_count / total_chars
415
+
416
+ return {
417
+ "total_chars": total_chars,
418
+ "unicode_map_error_count": unicode_map_error_count,
419
+ "unicode_map_error_ratio": unicode_map_error_ratio,
420
+ }
421
+
422
+
423
+ def _is_cjk_unicode_code(unicode_code: int) -> bool:
424
+ """判断 Unicode 码点是否属于分类器认可的 CJK 文本范围。"""
425
+ return any(start <= unicode_code <= end for start, end in CJK_TEXT_RANGES)
426
+
427
+
428
+ def _get_cjk_glyph_name_code(glyph_name: str) -> int | None:
429
+ """解析 uniXXXX/uXXXXX 形式的 CJK glyph name,其他名称返回 None。"""
430
+ match = re.fullmatch(
431
+ r"(?:uni([0-9A-Fa-f]{4,6})|u([0-9A-Fa-f]{4,6}))",
432
+ glyph_name,
433
+ )
434
+ if match is None:
435
+ return None
436
+ unicode_code = int(match.group(1) or match.group(2), 16)
437
+ if not _is_cjk_unicode_code(unicode_code):
438
+ return None
439
+ return unicode_code
440
+
441
+
442
+ def _get_empty_latin_charset_with_to_unicode_signal() -> dict[str, Any]:
443
+ """构造未触发的 Type1 Latin CharSet 字体候选信号。"""
444
+ return {
445
+ "triggered": False,
446
+ "charset_glyph_count": 0,
447
+ "latin_glyph_count": 0,
448
+ "latin_glyph_ratio": 0.0,
449
+ "cjk_charset_glyph_count": 0,
450
+ "cjk_charset_glyph_ratio": 0.0,
451
+ }
452
+
453
+
454
+ def _get_latin_charset_with_to_unicode_signal(font: Any) -> dict[str, Any]:
455
+ """识别带 ToUnicode 且 CharSet 明显为 Latin 的 Type1 字体候选。"""
456
+ signal = _get_empty_latin_charset_with_to_unicode_signal()
457
+ if str(font.get("/Subtype")) != "/Type1":
458
+ return signal
459
+
460
+ descriptor = _resolve_pdf_object(font.get("/FontDescriptor"))
461
+ to_unicode = _resolve_pdf_object(font.get("/ToUnicode"))
462
+ if descriptor is None or to_unicode is None:
463
+ return signal
464
+
465
+ charset = descriptor.get("/CharSet")
466
+ if charset is None:
467
+ return signal
468
+
469
+ glyph_names = set(re.findall(r"/([^/\s]+)", str(charset)))
470
+ charset_glyph_count = len(glyph_names)
471
+ latin_glyph_count = sum(1 for glyph_name in glyph_names if re.fullmatch(r"[A-Za-z]", glyph_name))
472
+ cjk_charset_glyph_count = sum(1 for glyph_name in glyph_names if _get_cjk_glyph_name_code(glyph_name) is not None)
473
+ latin_glyph_ratio = latin_glyph_count / charset_glyph_count if charset_glyph_count else 0.0
474
+ cjk_charset_glyph_ratio = cjk_charset_glyph_count / charset_glyph_count if charset_glyph_count else 0.0
475
+
476
+ signal.update(
477
+ {
478
+ "charset_glyph_count": charset_glyph_count,
479
+ "latin_glyph_count": latin_glyph_count,
480
+ "latin_glyph_ratio": latin_glyph_ratio,
481
+ "cjk_charset_glyph_count": cjk_charset_glyph_count,
482
+ "cjk_charset_glyph_ratio": cjk_charset_glyph_ratio,
483
+ }
484
+ )
485
+ signal["triggered"] = (
486
+ latin_glyph_count >= LATIN_CHARSET_MIN_LATIN_GLYPHS
487
+ and latin_glyph_ratio >= LATIN_CHARSET_MIN_LATIN_RATIO
488
+ and cjk_charset_glyph_count == 0
489
+ )
490
+ return signal
491
+
492
+
493
+ def _normalize_pdf_font_name(font_name: Any) -> str:
494
+ """规范化 PDF 字体名,统一 pypdf 的 NameObject 和 PDFium 返回值格式。"""
495
+ if font_name is None:
496
+ return ""
497
+ normalized_name = str(font_name).strip().lstrip("/")
498
+ return re.sub(r"^[A-Z]{6}\+", "", normalized_name, count=1)
499
+
500
+
501
+ def _get_pdfium_char_font_name(text_page: Any, char_index: int) -> str:
502
+ """读取 PDFium 字符级字体名,用于统计可疑 CID 字体的实际使用比例。"""
503
+ flags = c_int()
504
+ buffer_length = pdfium_c.FPDFText_GetFontInfo(
505
+ text_page,
506
+ char_index,
507
+ None,
508
+ 0,
509
+ byref(flags),
510
+ )
511
+ if buffer_length <= 0:
512
+ return ""
513
+
514
+ font_name_buffer = create_string_buffer(buffer_length)
515
+ actual_length = pdfium_c.FPDFText_GetFontInfo(
516
+ text_page,
517
+ char_index,
518
+ font_name_buffer,
519
+ buffer_length,
520
+ byref(flags),
521
+ )
522
+ if actual_length <= 0:
523
+ return ""
524
+
525
+ return font_name_buffer.value.decode("utf-8", errors="ignore")
526
+
527
+
528
+ def _get_cid_font_usage_signal_from_samples(
529
+ text_samples: list[dict[str, Any]], cid_font_usage: dict[int, dict[str, Any]]
530
+ ) -> dict[str, Any]:
531
+ """结合内容流精确计数与 PDFium 总字符数计算可疑 CID 字体使用比例。"""
532
+ best_signal = {
533
+ "triggered": False,
534
+ "page_index": None,
535
+ "font_names": [],
536
+ "cid_font_char_count": 0,
537
+ "total_chars": 0,
538
+ "cid_font_usage_ratio": 0.0,
539
+ }
540
+
541
+ for text_sample in text_samples:
542
+ page_index = text_sample.get("page_index")
543
+ total_chars = text_sample["char_count"]
544
+ if total_chars <= 0:
545
+ continue
546
+
547
+ page_usage = cid_font_usage.get(page_index) or {}
548
+ cid_font_char_count = int(page_usage.get("cid_font_char_count", 0))
549
+ matched_font_names = sorted(page_usage.get("font_names") or [])
550
+
551
+ cid_font_usage_ratio = cid_font_char_count / total_chars
552
+ signal = {
553
+ "triggered": False,
554
+ "page_index": page_index,
555
+ "font_names": matched_font_names,
556
+ "cid_font_char_count": cid_font_char_count,
557
+ "total_chars": total_chars,
558
+ "cid_font_usage_ratio": cid_font_usage_ratio,
559
+ }
560
+ if cid_font_char_count >= CID_FONT_USAGE_COUNT_THRESHOLD and cid_font_usage_ratio >= CID_FONT_USAGE_RATIO_THRESHOLD:
561
+ signal["triggered"] = True
562
+ return signal
563
+
564
+ if (
565
+ signal["cid_font_usage_ratio"],
566
+ signal["cid_font_char_count"],
567
+ ) > (
568
+ best_signal["cid_font_usage_ratio"],
569
+ best_signal["cid_font_char_count"],
570
+ ):
571
+ best_signal = signal
572
+
573
+ return best_signal
574
+
575
+
576
+ def _get_latin_font_cjk_usage_signal_from_samples(
577
+ text_samples: list[dict[str, Any]],
578
+ font_signal: dict[str, Any],
579
+ count_threshold: int,
580
+ usage_ratio_threshold: float,
581
+ cjk_ratio_threshold: float,
582
+ ) -> dict[str, Any]:
583
+ """按单个 Latin 候选字体统计实际使用量及 PDFium 解码后的 CJK 比例。"""
584
+ best_signal = {
585
+ "triggered": False,
586
+ "page_index": None,
587
+ "font_names": [],
588
+ "font_char_count": 0,
589
+ "cjk_char_count": 0,
590
+ "total_chars": 0,
591
+ "font_usage_ratio": 0.0,
592
+ "font_cjk_ratio": 0.0,
593
+ }
594
+ if not font_signal or not font_signal.get("triggered"):
595
+ return best_signal
596
+
597
+ page_fonts = font_signal.get("page_fonts") or {}
598
+ for text_sample in text_samples:
599
+ page_index = text_sample.get("page_index")
600
+ total_chars = text_sample.get("non_generated_char_count", 0)
601
+ if total_chars <= 0:
602
+ continue
603
+
604
+ font_name_counts = text_sample.get("font_non_generated_char_counts") or {}
605
+ font_cjk_char_counts = text_sample.get("font_non_generated_cjk_char_counts") or {}
606
+ candidate_font_names = {_normalize_pdf_font_name(font_name) for font_name in page_fonts.get(page_index, set())}
607
+ candidate_font_names.discard("")
608
+
609
+ for font_name in sorted(candidate_font_names):
610
+ font_char_count = font_name_counts.get(font_name, 0)
611
+ cjk_char_count = font_cjk_char_counts.get(font_name, 0)
612
+ font_usage_ratio = font_char_count / total_chars
613
+ font_cjk_ratio = cjk_char_count / font_char_count if font_char_count else 0.0
614
+ signal = {
615
+ "triggered": False,
616
+ "page_index": page_index,
617
+ "font_names": [font_name] if font_char_count else [],
618
+ "font_char_count": font_char_count,
619
+ "cjk_char_count": cjk_char_count,
620
+ "total_chars": total_chars,
621
+ "font_usage_ratio": font_usage_ratio,
622
+ "font_cjk_ratio": font_cjk_ratio,
623
+ }
624
+ if (
625
+ font_char_count >= count_threshold
626
+ and font_usage_ratio >= usage_ratio_threshold
627
+ and font_cjk_ratio >= cjk_ratio_threshold
628
+ ):
629
+ signal["triggered"] = True
630
+ return signal
631
+
632
+ if (
633
+ signal["font_cjk_ratio"],
634
+ signal["font_usage_ratio"],
635
+ signal["font_char_count"],
636
+ ) > (
637
+ best_signal["font_cjk_ratio"],
638
+ best_signal["font_usage_ratio"],
639
+ best_signal["font_char_count"],
640
+ ):
641
+ best_signal = signal
642
+
643
+ return best_signal
644
+
645
+
646
+ def _get_u72xx_text_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
647
+ """基于已缓存的抽样页文本统计扣除常用字后的 U+7280-U+72DF 字符占比。"""
648
+ cjk_chars = 0
649
+ u72xx_count = 0
650
+
651
+ for text_sample in text_samples:
652
+ for char in text_sample["cleaned_text"]:
653
+ unicode_code = ord(char)
654
+ if 0x4E00 <= unicode_code <= 0x9FFF:
655
+ cjk_chars += 1
656
+ if (
657
+ SUSPICIOUS_CJK_72XX_START <= unicode_code <= SUSPICIOUS_CJK_72XX_END
658
+ and char not in SUSPICIOUS_CJK_72XX_WHITELIST
659
+ ):
660
+ u72xx_count += 1
661
+
662
+ u72xx_cjk_ratio = 0.0
663
+ if cjk_chars > 0:
664
+ u72xx_cjk_ratio = u72xx_count / cjk_chars
665
+
666
+ return {
667
+ "cjk_chars": cjk_chars,
668
+ "u72xx_count": u72xx_count,
669
+ "u72xx_cjk_ratio": u72xx_cjk_ratio,
670
+ }
671
+
672
+
673
+ def _get_sample_cleaned_text(text_sample: Any) -> str:
674
+ """兼容 dict 和测试替身对象,读取抽样页的 cleaned_text 字段。"""
675
+ if isinstance(text_sample, dict):
676
+ return str(text_sample.get("cleaned_text", ""))
677
+ return str(getattr(text_sample, "cleaned_text", ""))
678
+
679
+
680
+ def _is_cjk_text_char(char: str) -> bool:
681
+ """判断字符是否属于中文文档中可接受的 CJK 文字范围。"""
682
+ return _is_cjk_unicode_code(ord(char))
683
+
684
+
685
+ def _get_cross_script_name(char: str) -> str | None:
686
+ """识别中文文档乱码中常见的跨脚本字符块名称。"""
687
+ unicode_code = ord(char)
688
+ for start, end, script_name in SUSPICIOUS_CROSS_SCRIPT_RANGES:
689
+ if start <= unicode_code <= end:
690
+ return script_name
691
+ return None
692
+
693
+
694
+ def _get_cross_script_text_signal_from_samples(text_samples: list[Any]) -> dict[str, Any]:
695
+ """统计中文文档文本层中大比例跨脚本混入信号,用于识别合法 Unicode 错码。"""
696
+ total_chars = 0
697
+ cjk_chars = 0
698
+ suspicious_chars = 0
699
+ script_counts: dict[str, int] = {}
700
+
701
+ for text_sample in text_samples:
702
+ for char in _get_sample_cleaned_text(text_sample):
703
+ total_chars += 1
704
+ if _is_cjk_text_char(char):
705
+ cjk_chars += 1
706
+
707
+ script_name = _get_cross_script_name(char)
708
+ if script_name is None:
709
+ continue
710
+
711
+ suspicious_chars += 1
712
+ script_counts[script_name] = script_counts.get(script_name, 0) + 1
713
+
714
+ suspicious_ratio = 0.0
715
+ if total_chars > 0:
716
+ suspicious_ratio = suspicious_chars / total_chars
717
+ dense_script_count = sum(1 for count in script_counts.values() if count >= SUSPICIOUS_CROSS_SCRIPT_DENSE_SCRIPT_CHARS)
718
+ top_scripts = sorted(
719
+ script_counts.items(),
720
+ key=lambda item: (-item[1], item[0]),
721
+ )[:5]
722
+ triggered = (
723
+ total_chars >= SUSPICIOUS_CROSS_SCRIPT_MIN_TEXT_CHARS
724
+ and cjk_chars >= SUSPICIOUS_CROSS_SCRIPT_MIN_CJK_CHARS
725
+ and suspicious_chars >= SUSPICIOUS_CROSS_SCRIPT_MIN_OTHER_SCRIPT_CHARS
726
+ and suspicious_ratio >= SUSPICIOUS_CROSS_SCRIPT_OTHER_SCRIPT_RATIO
727
+ and dense_script_count >= SUSPICIOUS_CROSS_SCRIPT_MIN_DENSE_SCRIPTS
728
+ )
729
+
730
+ return {
731
+ "triggered": triggered,
732
+ "total_chars": total_chars,
733
+ "cjk_chars": cjk_chars,
734
+ "suspicious_chars": suspicious_chars,
735
+ "suspicious_ratio": suspicious_ratio,
736
+ "script_counts": script_counts,
737
+ "top_scripts": top_scripts,
738
+ "dense_script_count": dense_script_count,
739
+ }
740
+
741
+
742
+ def _count_ascii_punct_run_chars(text: str) -> int:
743
+ """统计连续 ASCII 标点字符数,仅累计长度达到阈值的 run。"""
744
+ run_chars = 0
745
+ current_run = 0
746
+ current_run_types: set[str] = set()
747
+
748
+ for char in text:
749
+ if char in ASCII_PUNCT_CHARS:
750
+ current_run += 1
751
+ current_run_types.add(char)
752
+ continue
753
+
754
+ if current_run >= ASCII_PUNCT_RUN_MIN_LENGTH and len(current_run_types) >= 2:
755
+ run_chars += current_run
756
+ current_run = 0
757
+ current_run_types.clear()
758
+
759
+ if current_run >= ASCII_PUNCT_RUN_MIN_LENGTH and len(current_run_types) >= 2:
760
+ run_chars += current_run
761
+
762
+ return run_chars
763
+
764
+
765
+ def _get_sampled_ascii_punct_signal_from_samples(text_samples: list[dict[str, Any]]) -> dict[str, Any]:
766
+ """检查所有抽样页的 ASCII 标点密集度,用于识别无 ToUnicode 的乱码文本。"""
767
+ best_signal = {
768
+ "triggered": False,
769
+ "page_index": None,
770
+ "cleaned_text_chars": 0,
771
+ "ascii_punct_count": 0,
772
+ "ascii_punct_ratio": 0.0,
773
+ "ascii_punct_run_chars": 0,
774
+ "punct_run_ratio": 0.0,
775
+ }
776
+
777
+ for text_sample in text_samples:
778
+ page_index = text_sample.get("page_index")
779
+ cleaned_text = text_sample["cleaned_text"]
780
+ cleaned_text_chars = len(cleaned_text)
781
+ ascii_punct_count = sum(1 for char in cleaned_text if char in ASCII_PUNCT_CHARS)
782
+ ascii_punct_run_chars = _count_ascii_punct_run_chars(cleaned_text)
783
+
784
+ ascii_punct_ratio = 0.0
785
+ punct_run_ratio = 0.0
786
+ if cleaned_text_chars > 0:
787
+ ascii_punct_ratio = ascii_punct_count / cleaned_text_chars
788
+ punct_run_ratio = ascii_punct_run_chars / cleaned_text_chars
789
+
790
+ signal = {
791
+ "triggered": False,
792
+ "page_index": page_index,
793
+ "cleaned_text_chars": cleaned_text_chars,
794
+ "ascii_punct_count": ascii_punct_count,
795
+ "ascii_punct_ratio": ascii_punct_ratio,
796
+ "ascii_punct_run_chars": ascii_punct_run_chars,
797
+ "punct_run_ratio": punct_run_ratio,
798
+ }
799
+ if (
800
+ cleaned_text_chars >= SUSPICIOUS_ASCII_PUNCT_MIN_TEXT_CHARS
801
+ and ascii_punct_ratio >= SUSPICIOUS_ASCII_PUNCT_RATIO_THRESHOLD
802
+ and punct_run_ratio >= SUSPICIOUS_ASCII_PUNCT_RUN_RATIO_THRESHOLD
803
+ ):
804
+ signal["triggered"] = True
805
+ return signal
806
+
807
+ # 未触发时保留最可疑的抽样页指标,方便日志扩展和后续排查阈值边界。
808
+ if (
809
+ signal["punct_run_ratio"],
810
+ signal["ascii_punct_ratio"],
811
+ signal["cleaned_text_chars"],
812
+ ) > (
813
+ best_signal["punct_run_ratio"],
814
+ best_signal["ascii_punct_ratio"],
815
+ best_signal["cleaned_text_chars"],
816
+ ):
817
+ best_signal = signal
818
+
819
+ return best_signal
820
+
821
+
822
+ def _get_pdf_object_cache_key(obj_ref: Any, obj: Any) -> tuple[Any, ...]:
823
+ """为 pypdf 间接或直接对象生成可复用的身份键。"""
824
+ idnum = getattr(obj_ref, "idnum", None)
825
+ generation = getattr(obj_ref, "generation", None)
826
+ if idnum is not None:
827
+ return "indirect", idnum, generation
828
+ return "direct", id(obj)
829
+
830
+
831
+ def _get_font_resource_analysis(
832
+ font_ref: Any,
833
+ font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]],
834
+ ) -> tuple[Any, dict[str, bool]]:
835
+ """按字体资源对象身份缓存 CID 与 Type1 字体语义分析结果。"""
836
+ font = _resolve_pdf_object(font_ref)
837
+ if not font:
838
+ raise ValueError("Unable to resolve PDF font resource")
839
+
840
+ cache_key = _get_pdf_object_cache_key(font_ref, font)
841
+ analysis = font_analysis_cache.get(cache_key)
842
+ if analysis is None:
843
+ subtype = str(font.get("/Subtype"))
844
+ encoding = str(font.get("/Encoding"))
845
+ cid_without_to_unicode = (
846
+ subtype == "/Type0"
847
+ and encoding in ("/Identity-H", "/Identity-V")
848
+ and "/DescendantFonts" in font
849
+ and "/ToUnicode" not in font
850
+ )
851
+ latin_charset_signal = _get_latin_charset_with_to_unicode_signal(font)
852
+ analysis = {
853
+ "cid_without_to_unicode": cid_without_to_unicode,
854
+ "latin_charset_with_to_unicode": latin_charset_signal["triggered"],
855
+ }
856
+ font_analysis_cache[cache_key] = analysis
857
+ return font, analysis
858
+
859
+
860
+ def _get_pdf_string_raw_bytes(value: Any) -> bytes:
861
+ """读取 pypdf 字符串对象的原始字节,禁止用已解码文本替代。"""
862
+ raw_bytes = getattr(value, "original_bytes", None)
863
+ if isinstance(raw_bytes, bytes):
864
+ return raw_bytes
865
+ if isinstance(value, bytes):
866
+ return value
867
+ raise ValueError("PDF text string does not expose original bytes")
868
+
869
+
870
+ def _count_identity_cid_string(value: Any) -> int:
871
+ """按 Identity-H/V 的双字节编码统计文本字符串中的 CID 数量。"""
872
+ raw_bytes = _get_pdf_string_raw_bytes(value)
873
+ if len(raw_bytes) % 2:
874
+ raise ValueError("Identity CID text string has an odd byte length")
875
+ return len(raw_bytes) // 2
876
+
877
+
878
+ def _resource_graph_has_cid_without_to_unicode(
879
+ resources: Any,
880
+ font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]],
881
+ active_form_keys: frozenset[tuple[Any, ...]] = frozenset(),
882
+ ) -> bool:
883
+ """递归检查页面及 Form 资源图中是否存在缺少 ToUnicode 的 Identity CID 字体。"""
884
+ resources = _resolve_pdf_object(resources)
885
+ if not resources:
886
+ return False
887
+
888
+ fonts = _resolve_pdf_object(resources.get("/Font")) or {}
889
+ for font_ref in fonts.values():
890
+ _font, analysis = _get_font_resource_analysis(
891
+ font_ref,
892
+ font_analysis_cache,
893
+ )
894
+ if analysis["cid_without_to_unicode"]:
895
+ return True
896
+
897
+ xobjects = _resolve_pdf_object(resources.get("/XObject")) or {}
898
+ for xobject_ref in xobjects.values():
899
+ xobject = _resolve_pdf_object(xobject_ref)
900
+ if not xobject or str(xobject.get("/Subtype")) != "/Form":
901
+ continue
902
+
903
+ form_key = _get_pdf_object_cache_key(xobject_ref, xobject)
904
+ if form_key in active_form_keys:
905
+ continue
906
+ form_resources = xobject.get("/Resources")
907
+ if form_resources is None:
908
+ continue
909
+ if _resource_graph_has_cid_without_to_unicode(
910
+ form_resources,
911
+ font_analysis_cache,
912
+ active_form_keys | {form_key},
913
+ ):
914
+ return True
915
+ return False
916
+
917
+
918
+ def _count_cid_font_usage_in_content(
919
+ reader: PdfReader,
920
+ content: Any,
921
+ resources: Any,
922
+ font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]],
923
+ *,
924
+ inherited_font: tuple[str, bool] | None = None,
925
+ active_form_keys: frozenset[tuple[Any, ...]] = frozenset(),
926
+ ) -> Counter[str]:
927
+ """按实际 Tf 资源递归统计内容流中缺少 ToUnicode 的 Identity CID 字形。"""
928
+ counts: Counter[str] = Counter()
929
+ if content is None:
930
+ return counts
931
+
932
+ resources = _resolve_pdf_object(resources)
933
+ if resources is None:
934
+ raise ValueError("PDF content stream has no resolvable resources")
935
+
936
+ fonts = _resolve_pdf_object(resources.get("/Font")) or {}
937
+ xobjects = _resolve_pdf_object(resources.get("/XObject")) or {}
938
+ current_font = inherited_font
939
+ font_stack: list[tuple[str, bool] | None] = []
940
+
941
+ for operands, operator in ContentStream(content, reader).operations:
942
+ if operator == b"q":
943
+ font_stack.append(current_font)
944
+ continue
945
+ if operator == b"Q":
946
+ current_font = font_stack.pop() if font_stack else inherited_font
947
+ continue
948
+ if operator == b"Tf":
949
+ if not operands:
950
+ raise ValueError("PDF Tf operator has no font resource name")
951
+ font_key = operands[0]
952
+ font_ref = fonts.get(font_key)
953
+ if font_ref is None:
954
+ raise ValueError(f"Unable to resolve PDF font resource {font_key}")
955
+ font, analysis = _get_font_resource_analysis(
956
+ font_ref,
957
+ font_analysis_cache,
958
+ )
959
+ font_name = _normalize_pdf_font_name(font.get("/BaseFont") or font_key)
960
+ current_font = (
961
+ font_name,
962
+ analysis["cid_without_to_unicode"],
963
+ )
964
+ continue
965
+
966
+ if operator in (b"Tj", b"'", b'"'):
967
+ if current_font is None:
968
+ raise ValueError("PDF text is shown before selecting a font")
969
+ if current_font[1]:
970
+ counts[current_font[0]] += _count_identity_cid_string(operands[-1])
971
+ continue
972
+ if operator == b"TJ":
973
+ if current_font is None:
974
+ raise ValueError("PDF text is shown before selecting a font")
975
+ if current_font[1]:
976
+ for value in operands[0]:
977
+ if isinstance(value, (int, float)):
978
+ continue
979
+ counts[current_font[0]] += _count_identity_cid_string(value)
980
+ continue
981
+ if operator != b"Do":
982
+ continue
983
+
984
+ if not operands:
985
+ raise ValueError("PDF Do operator has no XObject resource name")
986
+ xobject_key = operands[0]
987
+ xobject_ref = xobjects.get(xobject_key)
988
+ if xobject_ref is None:
989
+ raise ValueError(f"Unable to resolve PDF XObject resource {xobject_key}")
990
+ xobject = _resolve_pdf_object(xobject_ref)
991
+ if not xobject:
992
+ raise ValueError(f"Unable to resolve PDF XObject {xobject_key}")
993
+ if str(xobject.get("/Subtype")) != "/Form":
994
+ continue
995
+
996
+ form_key = _get_pdf_object_cache_key(xobject_ref, xobject)
997
+ if form_key in active_form_keys:
998
+ raise ValueError(f"Cyclic PDF Form XObject reference {xobject_key}")
999
+ form_resources = xobject.get("/Resources")
1000
+ child_resources = resources if form_resources is None else form_resources
1001
+ counts.update(
1002
+ _count_cid_font_usage_in_content(
1003
+ reader,
1004
+ xobject,
1005
+ child_resources,
1006
+ font_analysis_cache,
1007
+ inherited_font=current_font,
1008
+ active_form_keys=active_form_keys | {form_key},
1009
+ )
1010
+ )
1011
+ return counts
1012
+
1013
+
1014
+ def _get_font_resource_signals_pypdf(
1015
+ pdf_bytes: bytes,
1016
+ page_indices: list[int],
1017
+ ) -> dict[str, Any]:
1018
+ """一次扫描抽样页字体资源,收集 CID 缺映射和 Type1 Latin 候选字体。"""
1019
+ reader = PdfReader(BytesIO(pdf_bytes))
1020
+ cid_page_fonts: dict[int, set[str]] = {}
1021
+ cid_page_usage: dict[int, dict[str, Any]] = {}
1022
+ latin_charset_page_fonts: dict[int, set[str]] = {}
1023
+ font_analysis_cache: dict[tuple[Any, ...], dict[str, bool]] = {}
1024
+
1025
+ for page_index in page_indices:
1026
+ page = reader.pages[page_index]
1027
+ resources = _resolve_pdf_object(page.get("/Resources"))
1028
+ if not resources:
1029
+ continue
1030
+
1031
+ fonts = _resolve_pdf_object(resources.get("/Font")) or {}
1032
+
1033
+ # Type1 Latin 信号仍按字体名使用 PDFium 统计;CID 用量在后续按资源对象精确计算。
1034
+ page_latin_font_resources: dict[str, dict[tuple[Any, ...], bool]] = {}
1035
+ for font_key, font_ref in fonts.items():
1036
+ font, analysis = _get_font_resource_analysis(
1037
+ font_ref,
1038
+ font_analysis_cache,
1039
+ )
1040
+ font_name = _normalize_pdf_font_name(font.get("/BaseFont") or font_key)
1041
+ if not font_name:
1042
+ continue
1043
+
1044
+ cache_key = _get_pdf_object_cache_key(font_ref, font)
1045
+
1046
+ if analysis["cid_without_to_unicode"]:
1047
+ cid_page_fonts.setdefault(page_index, set()).add(font_name)
1048
+
1049
+ page_latin_font_resources.setdefault(font_name, {})[cache_key] = analysis["latin_charset_with_to_unicode"]
1050
+
1051
+ for font_name, resource_states in page_latin_font_resources.items():
1052
+ if len(resource_states) == 1 and set(resource_states.values()) == {True}:
1053
+ latin_charset_page_fonts.setdefault(page_index, set()).add(font_name)
1054
+
1055
+ if _resource_graph_has_cid_without_to_unicode(
1056
+ resources,
1057
+ font_analysis_cache,
1058
+ ):
1059
+ usage_counts = _count_cid_font_usage_in_content(
1060
+ reader,
1061
+ page.get_contents(),
1062
+ resources,
1063
+ font_analysis_cache,
1064
+ )
1065
+ cid_page_usage[page_index] = {
1066
+ "font_names": sorted(font_name for font_name, char_count in usage_counts.items() if char_count > 0),
1067
+ "cid_font_char_count": sum(usage_counts.values()),
1068
+ }
1069
+
1070
+ return {
1071
+ "cid_without_to_unicode": {
1072
+ "triggered": bool(cid_page_fonts),
1073
+ "page_fonts": cid_page_fonts,
1074
+ },
1075
+ "cid_without_to_unicode_usage": cid_page_usage,
1076
+ "latin_charset_with_to_unicode": {
1077
+ "triggered": bool(latin_charset_page_fonts),
1078
+ "page_fonts": latin_charset_page_fonts,
1079
+ },
1080
+ }
1081
+
1082
+
1083
+ def _resolve_pdf_object(obj: Any) -> Any:
1084
+ if hasattr(obj, "get_object"):
1085
+ return obj.get_object()
1086
+ return obj
1087
+
1088
+
1089
+ def _get_pdfium_page_object_bounds(page_object: Any) -> tuple[float, float, float, float]:
1090
+ """兼容 pypdfium2 4.x/5.x,统一获取页面对象的边界坐标。"""
1091
+ get_bounds = getattr(page_object, "get_bounds", None)
1092
+ if callable(get_bounds):
1093
+ return get_bounds()
1094
+
1095
+ get_pos = getattr(page_object, "get_pos", None)
1096
+ if callable(get_pos):
1097
+ return get_pos()
1098
+
1099
+ raise AttributeError("PDFium page object has neither get_bounds() nor get_pos()")
1100
+
1101
+
1102
+ def get_high_image_coverage_ratio_pdfium(pdf_doc: pdfium.PdfDocument, page_indices: list[int]) -> float:
1103
+ high_image_coverage_pages = 0
1104
+
1105
+ with pdfium_guard():
1106
+ for page_index in page_indices:
1107
+ page = None
1108
+ try:
1109
+ page = pdf_doc[page_index]
1110
+ page_bbox: tuple[float, float, float, float] = page.get_bbox()
1111
+ page_area = abs((page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1]))
1112
+ image_area = 0.0
1113
+
1114
+ for page_object in page.get_objects(filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3):
1115
+ try:
1116
+ left, bottom, right, top = _get_pdfium_page_object_bounds(page_object)
1117
+ image_area += max(0.0, right - left) * max(0.0, top - bottom)
1118
+ finally:
1119
+ close_pdfium_child(page_object)
1120
+
1121
+ coverage_ratio = min(image_area / page_area, 1.0) if page_area > 0 else 0.0
1122
+ if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD:
1123
+ high_image_coverage_pages += 1
1124
+ finally:
1125
+ close_pdfium_child(page)
1126
+
1127
+ if not page_indices:
1128
+ return 0.0
1129
+ return high_image_coverage_pages / len(page_indices)
1130
+
1131
+
1132
+ if __name__ == "__main__":
1133
+ from .document import PDFDocument
1134
+
1135
+ with open("/Users/myhloli/pdf/luanma2x10.pdf", "rb") as f:
1136
+ p_bytes = f.read()
1137
+ pdf_doc = PDFDocument(p_bytes)
1138
+ logger.info(f"PDF classify result: {pdf_doc.classify()}")