docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,636 @@
1
+ """提供 PDF 字符 loose/tight/origin 驱动的通用上下标几何分类。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ import statistics
7
+ from dataclasses import dataclass
8
+ from typing import Any, Literal
9
+
10
+ from ....document.pdf.text.contracts import Char
11
+
12
+ from ....schema import BBox
13
+
14
+
15
+ SCRIPT_BODY_COMPARABLE_HEIGHT_RATIO = 0.9
16
+ SCRIPT_BASELINE_ABSOLUTE_TOLERANCE = 0.35
17
+ SCRIPT_BASELINE_LOOSE_HEIGHT_RATIO = 0.04
18
+ SCRIPT_ORIGIN_MIN_SHIFT_ABSOLUTE = 0.5
19
+ SCRIPT_ORIGIN_MIN_SHIFT_RATIO = 0.12
20
+ SCRIPT_TIGHT_HEIGHT_RATIO = 0.9
21
+ SCRIPT_STRONG_SHIFT_RATIO = 0.3
22
+ SCRIPT_STRONG_MAX_HEIGHT_RATIO = 1.1
23
+ SCRIPT_CONSENSUS_TIGHT_HEIGHT_RATIO = 0.88
24
+ SCRIPT_CONSENSUS_ORIGIN_SHIFT_RATIO = 0.22
25
+ SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO = 0.3
26
+ SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO = 0.15
27
+ SCRIPT_LOOSE_HEIGHT_ANOMALY_RATIO = 1.35
28
+ SCRIPT_COMPONENT_FORWARD_GAP_RATIO = 1.5
29
+ SCRIPT_COMPONENT_X_BACKTRACK_RATIO = 0.5
30
+ SCRIPT_COMPONENT_SEED_GAP_RATIO = 0.5
31
+ SCRIPT_COMPONENT_SEED_MAX_POSITION_DISTANCE = 4
32
+ SCRIPT_COMPONENT_MIN_OFFSET_RATIO = 0.08
33
+ SCRIPT_COMPONENT_MAX_HEIGHT_RATIO = 1.1
34
+ CONTROL_LINE_BREAK_CHARS = {"\r", "\n"}
35
+
36
+ ScriptRole = Literal["body", "sup", "sub"]
37
+ ScriptMarkRole = Literal["sup", "sub"]
38
+ _ScriptFontKey = tuple[str, int | None, int | None]
39
+
40
+
41
+ @dataclass(frozen=True, slots=True)
42
+ class ScriptCharFeature:
43
+ """保存单个字符参与纯几何上下标判定所需的只读特征。"""
44
+
45
+ index: int
46
+ text: str
47
+ loose_bbox: BBox
48
+ tight_bbox: BBox | None
49
+ origin: tuple[float, float] | None
50
+ is_valid: bool
51
+ is_body_anchor: bool
52
+
53
+ @property
54
+ def loose_height(self) -> float:
55
+ """返回 loose bbox 高度。"""
56
+ return self.loose_bbox[3] - self.loose_bbox[1]
57
+
58
+ @property
59
+ def loose_center_y(self) -> float:
60
+ """返回 loose bbox 中心 y。"""
61
+ return (self.loose_bbox[1] + self.loose_bbox[3]) / 2
62
+
63
+ @property
64
+ def tight_height(self) -> float:
65
+ """返回 tight bbox 高度,无有效框时返回零。"""
66
+ if self.tight_bbox is None:
67
+ return 0.0
68
+ return self.tight_bbox[3] - self.tight_bbox[1]
69
+
70
+ @property
71
+ def tight_center_y(self) -> float | None:
72
+ """返回 tight bbox 中心 y。"""
73
+ if self.tight_bbox is None:
74
+ return None
75
+ return (self.tight_bbox[1] + self.tight_bbox[3]) / 2
76
+
77
+
78
+ @dataclass(frozen=True, slots=True)
79
+ class ScriptBodyBand:
80
+ """表示当前视觉组件的正文 origin 基线与双 bbox 参考高度。"""
81
+
82
+ baseline: float
83
+ tight_height: float
84
+ loose_height: float
85
+ member_indices: frozenset[int]
86
+
87
+
88
+ @dataclass(frozen=True, slots=True)
89
+ class ScriptBaselineCluster:
90
+ """表示共享近似字符 origin 的正文或角标基线簇。"""
91
+
92
+ baseline: float
93
+ member_indices: tuple[int, ...]
94
+
95
+
96
+ def _coerce_finite_bbox(value: Any) -> BBox | None:
97
+ """把可迭代四元组收敛为合法有限 bbox。"""
98
+ try:
99
+ raw_bbox = getattr(value, "bbox", value)
100
+ if raw_bbox is None or len(raw_bbox) != 4:
101
+ return None
102
+ bbox = tuple(float(item) for item in raw_bbox)
103
+ except (TypeError, ValueError):
104
+ return None
105
+ if not all(math.isfinite(item) for item in bbox) or bbox[2] <= bbox[0] or bbox[3] <= bbox[1]:
106
+ return None
107
+ return bbox # type: ignore[return-value]
108
+
109
+
110
+ def _char_geometry_key(char: Char) -> int | None:
111
+ """返回可用于 side-map 查询的合法 PDFium char_idx。"""
112
+ char_idx = char.get("char_idx")
113
+ if isinstance(char_idx, bool) or not isinstance(char_idx, int):
114
+ return None
115
+ return char_idx
116
+
117
+
118
+ def _script_role(offset: float) -> ScriptMarkRole:
119
+ """把相对正文 origin 基线的纵向偏移转换为上下标角色。"""
120
+ return "sub" if offset > 0 else "sup"
121
+
122
+
123
+ def build_script_features(
124
+ chars: list[Char],
125
+ tight_bboxes: dict[int, BBox],
126
+ origins: dict[int, tuple[float, float]],
127
+ protected_body_indices: set[int],
128
+ ) -> list[ScriptCharFeature]:
129
+ """一次性构造 loose/tight/origin 三类字符几何特征。"""
130
+ features = []
131
+ for index, char in enumerate(chars):
132
+ text = str(char.get("char", ""))
133
+ char_idx = _char_geometry_key(char)
134
+ loose_bbox = _coerce_finite_bbox(char.get("bbox")) or (0.0, 0.0, 0.0, 0.0)
135
+ tight_bbox = _coerce_finite_bbox(tight_bboxes.get(char_idx)) if char_idx is not None else None
136
+ raw_origin = origins.get(char_idx) if char_idx is not None else None
137
+ origin = None
138
+ if raw_origin is not None:
139
+ try:
140
+ candidate_origin = (float(raw_origin[0]), float(raw_origin[1]))
141
+ except (IndexError, TypeError, ValueError):
142
+ candidate_origin = None
143
+ if candidate_origin is not None and all(math.isfinite(value) for value in candidate_origin):
144
+ origin = candidate_origin
145
+ is_valid = text not in CONTROL_LINE_BREAK_CHARS and not text.isspace() and tight_bbox is not None and origin is not None
146
+ features.append(
147
+ ScriptCharFeature(
148
+ index=index,
149
+ text=text,
150
+ loose_bbox=loose_bbox,
151
+ tight_bbox=tight_bbox,
152
+ origin=origin,
153
+ is_valid=is_valid,
154
+ is_body_anchor=is_valid and text.isalnum() and index not in protected_body_indices,
155
+ )
156
+ )
157
+ return features
158
+
159
+
160
+ def split_script_visual_components(features: list[ScriptCharFeature]) -> list[list[int]]:
161
+ """按换行、x 回退和大间隙切分独立视觉组件。"""
162
+ valid_heights = [feature.loose_height for feature in features if feature.loose_height > 0]
163
+ scale = statistics.median(valid_heights) if valid_heights else 1.0
164
+ components: list[list[int]] = []
165
+ current: list[int] = []
166
+ previous_visible: ScriptCharFeature | None = None
167
+ for feature in features:
168
+ if feature.text in CONTROL_LINE_BREAK_CHARS or feature.text.isspace():
169
+ if current:
170
+ components.append(current)
171
+ current = []
172
+ previous_visible = None
173
+ continue
174
+ if feature.is_valid and previous_visible is not None:
175
+ x_backtrack = previous_visible.loose_bbox[0] - feature.loose_bbox[0]
176
+ forward_gap = feature.loose_bbox[0] - previous_visible.loose_bbox[2]
177
+ if (
178
+ x_backtrack > scale * SCRIPT_COMPONENT_X_BACKTRACK_RATIO
179
+ or forward_gap > scale * SCRIPT_COMPONENT_FORWARD_GAP_RATIO
180
+ ):
181
+ if current:
182
+ components.append(current)
183
+ current = []
184
+ previous_visible = None
185
+ current.append(feature.index)
186
+ if feature.is_valid:
187
+ previous_visible = feature
188
+ if current:
189
+ components.append(current)
190
+ return components
191
+
192
+
193
+ def _quantile(values: list[float], fraction: float) -> float:
194
+ """返回适合小样本基线簇的稳定分位数。"""
195
+ if not values:
196
+ return 0.0
197
+ ordered = sorted(values)
198
+ index = min(len(ordered) - 1, max(0, round((len(ordered) - 1) * fraction)))
199
+ return ordered[index]
200
+
201
+
202
+ def _cluster_baselines(
203
+ features: list[ScriptCharFeature],
204
+ component_indices: list[int],
205
+ ) -> tuple[list[ScriptBaselineCluster], float]:
206
+ """按字符 origin y 聚类当前组件的字母数字基线。"""
207
+ anchors = [features[index] for index in component_indices if features[index].is_body_anchor]
208
+ if not anchors:
209
+ return [], 0.0
210
+ median_loose_height = statistics.median(feature.loose_height for feature in anchors)
211
+ tolerance = max(SCRIPT_BASELINE_ABSOLUTE_TOLERANCE, median_loose_height * SCRIPT_BASELINE_LOOSE_HEIGHT_RATIO)
212
+ groups: list[list[int]] = []
213
+ for feature in sorted(anchors, key=lambda item: item.origin[1] if item.origin is not None else 0.0):
214
+ baseline = feature.origin[1] if feature.origin is not None else 0.0
215
+ if not groups:
216
+ groups.append([feature.index])
217
+ continue
218
+ previous_baseline = statistics.median(
219
+ features[index].origin[1] for index in groups[-1] if features[index].origin is not None
220
+ )
221
+ if abs(baseline - previous_baseline) <= tolerance:
222
+ groups[-1].append(feature.index)
223
+ else:
224
+ groups.append([feature.index])
225
+ return (
226
+ [
227
+ ScriptBaselineCluster(
228
+ baseline=statistics.median(features[index].origin[1] for index in group if features[index].origin is not None),
229
+ member_indices=tuple(group),
230
+ )
231
+ for group in groups
232
+ ],
233
+ tolerance,
234
+ )
235
+
236
+
237
+ def _cluster_tight_height(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
238
+ """返回基线簇的高分位 tight 高度。"""
239
+ return _quantile(
240
+ [features[index].tight_height for index in cluster.member_indices if features[index].tight_height > 0], 0.9
241
+ )
242
+
243
+
244
+ def _cluster_loose_height(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
245
+ """返回基线簇的高分位 loose 高度。"""
246
+ return _quantile(
247
+ [features[index].loose_height for index in cluster.member_indices if features[index].loose_height > 0], 0.9
248
+ )
249
+
250
+
251
+ def _cluster_tight_center(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
252
+ """返回基线簇的 tight bbox 中心中位数。"""
253
+ return statistics.median(
254
+ features[index].tight_center_y for index in cluster.member_indices if features[index].tight_center_y is not None
255
+ )
256
+
257
+
258
+ def _cluster_loose_center(features: list[ScriptCharFeature], cluster: ScriptBaselineCluster) -> float:
259
+ """返回基线簇的 loose bbox 中心中位数。"""
260
+ return statistics.median(features[index].loose_center_y for index in cluster.member_indices)
261
+
262
+
263
+ def _cluster_has_consistent_displacement(
264
+ features: list[ScriptCharFeature],
265
+ cluster: ScriptBaselineCluster,
266
+ body_cluster: ScriptBaselineCluster,
267
+ body_band: ScriptBodyBand,
268
+ ) -> bool:
269
+ """要求 origin 与双 bbox 至少两项同向,并排除普通混合字体的弱中心偏移。"""
270
+ origin_shift = cluster.baseline - body_band.baseline
271
+ tight_shift = _cluster_tight_center(features, cluster) - _cluster_tight_center(features, body_cluster)
272
+ loose_shift = _cluster_loose_center(features, cluster) - _cluster_loose_center(features, body_cluster)
273
+ shifts = (origin_shift, tight_shift, loose_shift)
274
+ expected_positive = origin_shift > 0
275
+ if sum(value > 0.05 if expected_positive else value < -0.05 for value in shifts) < 2:
276
+ return False
277
+ origin_ratio = abs(origin_shift) / max(body_band.tight_height, 1e-6)
278
+ tight_ratio = abs(tight_shift) / max(body_band.tight_height, 1e-6)
279
+ loose_ratio = abs(loose_shift) / max(body_band.loose_height, 1e-6)
280
+ return origin_ratio >= SCRIPT_CONSENSUS_ORIGIN_SHIFT_RATIO or (
281
+ tight_ratio >= SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO and loose_ratio >= SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO
282
+ )
283
+
284
+
285
+ def _choose_body_band(
286
+ features: list[ScriptCharFeature],
287
+ clusters: list[ScriptBaselineCluster],
288
+ ) -> tuple[ScriptBodyBand, ScriptBaselineCluster] | None:
289
+ """在接近最高字形的基线簇中按字符数选择正文基线。"""
290
+ if not clusters:
291
+ return None
292
+ maximum_height = max(_cluster_tight_height(features, cluster) for cluster in clusters)
293
+ comparable = [
294
+ cluster
295
+ for cluster in clusters
296
+ if _cluster_tight_height(features, cluster) >= maximum_height * SCRIPT_BODY_COMPARABLE_HEIGHT_RATIO
297
+ ]
298
+ body_cluster = max(
299
+ comparable,
300
+ key=lambda cluster: (
301
+ len(cluster.member_indices),
302
+ _cluster_tight_height(features, cluster),
303
+ _cluster_loose_height(features, cluster),
304
+ ),
305
+ )
306
+ return (
307
+ ScriptBodyBand(
308
+ baseline=body_cluster.baseline,
309
+ tight_height=_cluster_tight_height(features, body_cluster),
310
+ loose_height=_cluster_loose_height(features, body_cluster),
311
+ member_indices=frozenset(body_cluster.member_indices),
312
+ ),
313
+ body_cluster,
314
+ )
315
+
316
+
317
+ def _nearest_cluster(
318
+ feature: ScriptCharFeature,
319
+ clusters: list[ScriptBaselineCluster],
320
+ tolerance: float,
321
+ ) -> ScriptBaselineCluster | None:
322
+ """把非字母数字字符附着到最近 origin 基线簇。"""
323
+ if feature.origin is None or not clusters:
324
+ return None
325
+ nearest = min(clusters, key=lambda cluster: abs(feature.origin[1] - cluster.baseline))
326
+ return nearest if abs(feature.origin[1] - nearest.baseline) <= tolerance else None
327
+
328
+
329
+ def _horizontal_gap(first: BBox, second: BBox) -> float:
330
+ """返回两个字符 tight bbox 的水平间隙。"""
331
+ return max(0.0, first[0] - second[2], second[0] - first[2])
332
+
333
+
334
+ def _drop_unseeded_punctuation(
335
+ features: list[ScriptCharFeature],
336
+ component_indices: list[int],
337
+ roles: list[ScriptRole],
338
+ body_height: float,
339
+ ) -> None:
340
+ """移除未邻近同类字母数字种子的标点。"""
341
+ seeded = [
342
+ features[index]
343
+ for index in component_indices
344
+ if roles[index] != "body" and features[index].text.isalnum() and features[index].tight_bbox is not None
345
+ ]
346
+ for index in component_indices:
347
+ feature = features[index]
348
+ role = roles[index]
349
+ if role == "body" or feature.text.isalnum() or feature.tight_bbox is None:
350
+ continue
351
+ nearby = any(
352
+ roles[seed.index] == role
353
+ and _horizontal_gap(feature.tight_bbox, seed.tight_bbox) <= max(2.0, body_height * SCRIPT_COMPONENT_SEED_GAP_RATIO)
354
+ and abs(feature.index - seed.index) <= SCRIPT_COMPONENT_SEED_MAX_POSITION_DISTANCE
355
+ for seed in seeded
356
+ )
357
+ if not nearby:
358
+ roles[index] = "body"
359
+
360
+
361
+ def _apply_consensus_candidates(
362
+ features: list[ScriptCharFeature],
363
+ component_indices: list[int],
364
+ body_band: ScriptBodyBand,
365
+ roles: list[ScriptRole],
366
+ ) -> None:
367
+ """用 origin/tight/loose 三证据一致性补充孤立边界字符。"""
368
+ body_members = [
369
+ features[index]
370
+ for index in body_band.member_indices
371
+ if features[index].tight_bbox is not None
372
+ and features[index].tight_center_y is not None
373
+ and features[index].origin is not None
374
+ ]
375
+ for index in component_indices:
376
+ feature = features[index]
377
+ if (
378
+ roles[index] != "body"
379
+ or index in body_band.member_indices
380
+ or not feature.is_valid
381
+ or feature.tight_bbox is None
382
+ or feature.tight_center_y is None
383
+ or feature.origin is None
384
+ or not body_members
385
+ ):
386
+ continue
387
+ reference = min(
388
+ body_members,
389
+ key=lambda member: (_horizontal_gap(feature.tight_bbox, member.tight_bbox), abs(feature.index - member.index)),
390
+ )
391
+ if _horizontal_gap(feature.tight_bbox, reference.tight_bbox) > max(2.5, reference.tight_height * 1.2):
392
+ continue
393
+ if feature.tight_height / max(reference.tight_height, 1e-6) > SCRIPT_CONSENSUS_TIGHT_HEIGHT_RATIO:
394
+ continue
395
+ shifts = (
396
+ feature.origin[1] - reference.origin[1],
397
+ feature.tight_center_y - reference.tight_center_y,
398
+ feature.loose_center_y - reference.loose_center_y,
399
+ )
400
+ positive_votes = sum(value > 0.05 for value in shifts)
401
+ negative_votes = sum(value < -0.05 for value in shifts)
402
+ if max(positive_votes, negative_votes) < 2:
403
+ continue
404
+ role = "sub" if positive_votes > negative_votes else "sup"
405
+ if _script_role(shifts[0]) != role:
406
+ continue
407
+ origin_ratio = abs(shifts[0]) / max(reference.tight_height, 1e-6)
408
+ tight_ratio = abs(shifts[1]) / max(reference.tight_height, 1e-6)
409
+ loose_ratio = abs(shifts[2]) / max(reference.loose_height, 1e-6)
410
+ if not feature.text.isalnum():
411
+ expected_positive = role == "sub"
412
+ if not all(value > 0.05 if expected_positive else value < -0.05 for value in shifts) or not (
413
+ origin_ratio >= SCRIPT_STRONG_SHIFT_RATIO
414
+ and tight_ratio >= SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO
415
+ and loose_ratio >= SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO
416
+ ):
417
+ continue
418
+ if origin_ratio >= SCRIPT_CONSENSUS_ORIGIN_SHIFT_RATIO or (
419
+ tight_ratio >= SCRIPT_CONSENSUS_TIGHT_CENTER_SHIFT_RATIO
420
+ and loose_ratio >= SCRIPT_CONSENSUS_LOOSE_CENTER_SHIFT_RATIO
421
+ ):
422
+ roles[index] = role
423
+
424
+
425
+ def _expand_component_neighbors(
426
+ features: list[ScriptCharFeature],
427
+ component_indices: list[int],
428
+ body_band: ScriptBodyBand,
429
+ protected_body_indices: set[int],
430
+ roles: list[ScriptRole],
431
+ ) -> None:
432
+ """把同侧连续字符并入已有角标 run。"""
433
+ component_set = set(component_indices)
434
+ changed = True
435
+ while changed:
436
+ changed = False
437
+ for index in component_indices:
438
+ feature = features[index]
439
+ if (
440
+ roles[index] != "body"
441
+ or index in protected_body_indices
442
+ or not feature.is_valid
443
+ or feature.origin is None
444
+ or feature.text.isspace()
445
+ or feature.tight_height > body_band.tight_height * SCRIPT_COMPONENT_MAX_HEIGHT_RATIO
446
+ ):
447
+ continue
448
+ neighbor_roles = {
449
+ roles[neighbor]
450
+ for neighbor in (index - 1, index + 1)
451
+ if neighbor in component_set and roles[neighbor] != "body"
452
+ }
453
+ if len(neighbor_roles) != 1:
454
+ continue
455
+ role = next(iter(neighbor_roles))
456
+ shift = feature.origin[1] - body_band.baseline
457
+ if abs(shift) < body_band.tight_height * SCRIPT_COMPONENT_MIN_OFFSET_RATIO or _script_role(shift) != role:
458
+ continue
459
+ roles[index] = role
460
+ changed = True
461
+
462
+
463
+ def _script_font_key(char: Char) -> _ScriptFontKey | None:
464
+ """读取原始字体身份,不合并子集名称,也不把原始字号当作有效字形高度。"""
465
+ font = char.get("font")
466
+ if not isinstance(font, dict):
467
+ return None
468
+ name = font.get("name")
469
+ if not isinstance(name, str) or not name:
470
+ return None
471
+ flags, weight = font.get("flags"), font.get("weight")
472
+ return name, flags if type(flags) is int else None, weight if type(weight) is int else None
473
+
474
+
475
+ def _local_font_run(
476
+ features: list[ScriptCharFeature],
477
+ font_keys: list[_ScriptFontKey | None],
478
+ component: set[int],
479
+ cluster: ScriptBaselineCluster,
480
+ ) -> set[int]:
481
+ """限定同一视觉组件内连续的同字体 ASCII/全角英数字串。"""
482
+ keys = {font_keys[index] for index in cluster.member_indices}
483
+ if len(keys) != 1 or None in keys:
484
+ return set()
485
+ key = next(iter(keys))
486
+
487
+ def belongs(index: int) -> bool:
488
+ """让空白、括号、运算符、CJK、缺失几何和字体边界阻断局部参考。"""
489
+ if index not in component or not features[index].is_valid or font_keys[index] != key:
490
+ return False
491
+ text = features[index].text
492
+ return len(text) == 1 and (
493
+ (text.isascii() and text.isalnum()) or "0" <= text <= "9" or "A" <= text <= "Z" or "a" <= text <= "z"
494
+ )
495
+
496
+ left = min(cluster.member_indices)
497
+ right = left + 1
498
+ while belongs(left - 1):
499
+ left -= 1
500
+ while belongs(right):
501
+ right += 1
502
+ if not all(left <= index < right and belongs(index) for index in cluster.member_indices):
503
+ return set()
504
+ return set(range(left, right))
505
+
506
+
507
+ def _recheck_weak_clusters_with_local_font_body(
508
+ features: list[ScriptCharFeature],
509
+ font_keys: list[_ScriptFontKey | None],
510
+ component_indices: list[int],
511
+ body_band: ScriptBodyBand,
512
+ cluster_roles: dict[ScriptBaselineCluster, ScriptRole],
513
+ ) -> None:
514
+ """用同字体局部正文撤销弱误判;不制造角标,不让撤销结果成为新参考。"""
515
+ initial_roles = dict(cluster_roles)
516
+ component = set(component_indices)
517
+ for cluster, role in initial_roles.items():
518
+ if role == "body" or abs(cluster.baseline - body_band.baseline) >= body_band.tight_height * SCRIPT_STRONG_SHIFT_RATIO:
519
+ continue
520
+ run = _local_font_run(features, font_keys, component, cluster)
521
+ if not run:
522
+ continue
523
+ references: list[tuple[float, int, ScriptBaselineCluster]] = []
524
+ for reference, reference_role in initial_roles.items():
525
+ if reference_role != "body":
526
+ continue
527
+ indices = tuple(index for index in reference.member_indices if index in run)
528
+ if len(indices) < 2:
529
+ continue
530
+ gap = min(
531
+ _horizontal_gap(features[first].tight_bbox, features[second].tight_bbox)
532
+ for first in cluster.member_indices
533
+ for second in indices
534
+ if features[first].tight_bbox is not None and features[second].tight_bbox is not None
535
+ )
536
+ if gap > max(2.5, body_band.tight_height * 1.2):
537
+ continue
538
+ baseline = statistics.median(features[index].origin[1] for index in indices if features[index].origin is not None)
539
+ references.append((gap, -len(indices), ScriptBaselineCluster(baseline, indices)))
540
+ if not references:
541
+ continue
542
+ reference = min(references, key=lambda item: item[:2])[2]
543
+ height = _cluster_tight_height(features, reference)
544
+ if height <= 0:
545
+ continue
546
+ shift = cluster.baseline - reference.baseline
547
+ strong_shift = abs(shift) >= height * SCRIPT_STRONG_SHIFT_RATIO
548
+ tight_ratio = _cluster_tight_height(features, cluster) / height
549
+ if abs(shift) < max(SCRIPT_ORIGIN_MIN_SHIFT_ABSOLUTE, height * SCRIPT_ORIGIN_MIN_SHIFT_RATIO) or (
550
+ tight_ratio > SCRIPT_TIGHT_HEIGHT_RATIO and not (strong_shift and tight_ratio <= SCRIPT_STRONG_MAX_HEIGHT_RATIO)
551
+ ):
552
+ cluster_roles[cluster] = "body"
553
+
554
+
555
+ def _assign_component(
556
+ features: list[ScriptCharFeature],
557
+ component_indices: list[int],
558
+ protected_body_indices: set[int],
559
+ roles: list[ScriptRole],
560
+ font_keys: list[_ScriptFontKey | None],
561
+ ) -> None:
562
+ """在单个视觉组件内按 origin 基线簇和双 bbox 一致性分配角色。"""
563
+ clusters, tolerance = _cluster_baselines(features, component_indices)
564
+ body_result = _choose_body_band(features, clusters)
565
+ if body_result is None:
566
+ return
567
+ body_band, body_cluster = body_result
568
+ if body_band.tight_height <= 0 or body_band.loose_height <= 0:
569
+ return
570
+ cluster_roles: dict[ScriptBaselineCluster, ScriptRole] = {body_cluster: "body"}
571
+ for cluster in clusters:
572
+ if cluster is body_cluster:
573
+ continue
574
+ shift = cluster.baseline - body_band.baseline
575
+ tight_ratio = _cluster_tight_height(features, cluster) / body_band.tight_height
576
+ loose_ratio = _cluster_loose_height(features, cluster) / body_band.loose_height
577
+ minimum_shift = max(SCRIPT_ORIGIN_MIN_SHIFT_ABSOLUTE, body_band.tight_height * SCRIPT_ORIGIN_MIN_SHIFT_RATIO)
578
+ strong_shift = abs(shift) >= body_band.tight_height * SCRIPT_STRONG_SHIFT_RATIO
579
+ if (
580
+ abs(shift) < minimum_shift
581
+ or (
582
+ tight_ratio > SCRIPT_TIGHT_HEIGHT_RATIO and not (strong_shift and tight_ratio <= SCRIPT_STRONG_MAX_HEIGHT_RATIO)
583
+ )
584
+ or (
585
+ loose_ratio > SCRIPT_LOOSE_HEIGHT_ANOMALY_RATIO
586
+ and not _cluster_has_consistent_displacement(features, cluster, body_cluster, body_band)
587
+ )
588
+ ):
589
+ cluster_roles[cluster] = "body"
590
+ else:
591
+ cluster_roles[cluster] = _script_role(shift)
592
+ _recheck_weak_clusters_with_local_font_body(features, font_keys, component_indices, body_band, cluster_roles)
593
+ for index in component_indices:
594
+ feature = features[index]
595
+ if (
596
+ index in protected_body_indices
597
+ or feature.origin is None
598
+ or feature.text.isspace()
599
+ or feature.text in CONTROL_LINE_BREAK_CHARS
600
+ ):
601
+ continue
602
+ cluster = _nearest_cluster(feature, clusters, tolerance)
603
+ if cluster is not None:
604
+ roles[index] = cluster_roles.get(cluster, "body")
605
+ _drop_unseeded_punctuation(features, component_indices, roles, body_band.tight_height)
606
+ _apply_consensus_candidates(features, component_indices, body_band, roles)
607
+ _expand_component_neighbors(features, component_indices, body_band, protected_body_indices, roles)
608
+ for index in protected_body_indices.intersection(component_indices):
609
+ roles[index] = "body"
610
+
611
+
612
+ def classify_char_script_roles(
613
+ chars: list[Char],
614
+ *,
615
+ tight_bboxes: dict[int, BBox],
616
+ origins: dict[int, tuple[float, float]],
617
+ protected_body_indices: set[int] | None = None,
618
+ ) -> list[ScriptRole]:
619
+ """按视觉组件、origin 基线簇和双 bbox 一致性识别上下标。"""
620
+ protected = protected_body_indices or set()
621
+ features = build_script_features(chars, tight_bboxes, origins, protected)
622
+ font_keys = [_script_font_key(char) for char in chars]
623
+ roles: list[ScriptRole] = ["body"] * len(features)
624
+ for component_indices in split_script_visual_components(features):
625
+ _assign_component(features, component_indices, protected, roles, font_keys)
626
+ return roles
627
+
628
+
629
+ __all__ = [
630
+ "CONTROL_LINE_BREAK_CHARS",
631
+ "ScriptCharFeature",
632
+ "ScriptRole",
633
+ "build_script_features",
634
+ "classify_char_script_roles",
635
+ "split_script_visual_components",
636
+ ]
@@ -0,0 +1,30 @@
1
+ """原生 PDF 的公共复用能力;模型区域融合可独立使用这些纯数据操作。"""
2
+
3
+ from .native_text import _build_native_line_items as build_native_line_items
4
+ from .line_merging import (
5
+ _merge_overlapping_inline_text_clusters as merge_overlapping_inline_text_clusters,
6
+ _merge_same_baseline_text_lines as merge_same_baseline_text_lines,
7
+ )
8
+ from .script_geometry import ScriptRole, classify_char_script_roles
9
+ from .spatial_text import project_ocr_table_text as project_table_text
10
+ from .table_text_styles import render_native_table_html_with_scripts
11
+ from .table_recovery import (
12
+ NativeTableInput,
13
+ coerce_native_table_rectangles,
14
+ coerce_native_table_rules,
15
+ recover_native_pdf_table,
16
+ )
17
+
18
+ __all__ = [
19
+ "build_native_line_items",
20
+ "merge_overlapping_inline_text_clusters",
21
+ "merge_same_baseline_text_lines",
22
+ "ScriptRole",
23
+ "classify_char_script_roles",
24
+ "project_table_text",
25
+ "render_native_table_html_with_scripts",
26
+ "NativeTableInput",
27
+ "coerce_native_table_rectangles",
28
+ "coerce_native_table_rules",
29
+ "recover_native_pdf_table",
30
+ ]