docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,62 @@
1
+ """PDF 字符到文本片段及行的共享纯数据接口。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from .contracts import Bbox, Char, Line, Span
6
+ from .groups import assign_scripts, get_lines
7
+
8
+
9
+ def get_spans(chars: list[Char], superscript_height_threshold: float = 0.8, line_distance_threshold: float = 0.1) -> list[Span]:
10
+ """直接从自有字符记录构建片段,避免容器与数组往返转换。"""
11
+ spans: list[Span] = []
12
+ for char in chars:
13
+ current = spans[-1] if spans else None
14
+ box = char["bbox"]
15
+ new_span = current is None
16
+ if current is not None:
17
+ previous = current["chars"][-1]
18
+ height = current["bbox"].height
19
+ new_span = (
20
+ char["font"] != current["font"]
21
+ or char["rotation"] != current["rotation"]
22
+ or previous["char"] in {"\x02", "\n"}
23
+ or (
24
+ box.y_start < current["bbox"].y_start - height * line_distance_threshold
25
+ and box.y_end < height * superscript_height_threshold + current["bbox"].y_start
26
+ and box.x_start > current["bbox"].x_end
27
+ )
28
+ )
29
+ if new_span:
30
+ spans.append(
31
+ {
32
+ "bbox": box.copy(),
33
+ "text": char["char"],
34
+ "font": char["font"],
35
+ "chars": [char],
36
+ "char_start_idx": char["char_idx"],
37
+ "char_end_idx": char["char_idx"],
38
+ "rotation": char["rotation"],
39
+ "url": "",
40
+ "superscript": False,
41
+ "subscript": False,
42
+ }
43
+ )
44
+ else:
45
+ current["bbox"].merge_inplace(box)
46
+ current["text"] += char["char"]
47
+ current["chars"].append(char)
48
+ current["char_end_idx"] = char["char_idx"]
49
+ return spans
50
+
51
+
52
+ def get_lines_from_chars(
53
+ chars: list[Char], superscript_height_threshold: float = 0.7, line_distance_threshold: float = 0.1
54
+ ) -> list[Line]:
55
+ """由已物化字符生成基础文本行,不访问 PDFium 或源文档。"""
56
+ spans = get_spans(chars, superscript_height_threshold, line_distance_threshold)
57
+ lines = get_lines(spans)
58
+ assign_scripts(lines, superscript_height_threshold, line_distance_threshold)
59
+ return lines
60
+
61
+
62
+ __all__ = ["Bbox", "Char", "Line", "Span", "get_lines_from_chars"]
@@ -0,0 +1,211 @@
1
+ # Portions derived from pdftext 0.7.1, Copyright Vik Paruchuri, Apache-2.0.
2
+ # Changed in DocVortex: owned character types and source-index mappings replace upstream containers.
3
+ """DocVortex 自有 PDF 字符与几何数据,不携带 PDFium 句柄。"""
4
+
5
+ from __future__ import annotations
6
+ from typing import Any, TypedDict
7
+
8
+
9
+ class Bbox:
10
+ __slots__ = ("bbox", "ensure_nonzero_area")
11
+
12
+ def __init__(self, bbox: list[float], ensure_nonzero_area: bool = False) -> None:
13
+ """建立独立矩形对象并按需保证面积。"""
14
+ if ensure_nonzero_area:
15
+ bbox = list(bbox)
16
+ bbox[2] = max(bbox[0], bbox[2] + 1)
17
+ bbox[3] = max(bbox[1], bbox[3] + 1)
18
+ self.bbox = bbox
19
+ self.ensure_nonzero_area = ensure_nonzero_area
20
+
21
+ def __getitem__(self, item: int | slice) -> float | list[float]:
22
+ """读取矩形坐标。"""
23
+ return self.bbox[item]
24
+
25
+ def __repr__(self) -> str:
26
+ """返回矩形的可读表示。"""
27
+ return f"Bbox({self.bbox})"
28
+
29
+ def __reduce__(self) -> tuple:
30
+ # ensure_nonzero_area is already applied at construction; don't re-apply on unpickle
31
+ """保存矩形数值以支持跨进程序列化。"""
32
+ return (Bbox, (self.bbox,))
33
+
34
+ def copy(self) -> Bbox:
35
+ """复制矩形避免共享累加状态。"""
36
+ return Bbox(list(self.bbox))
37
+
38
+ @property
39
+ def height(self) -> float:
40
+ """计算矩形的 height 几何属性。"""
41
+ return self.bbox[3] - self.bbox[1]
42
+
43
+ @property
44
+ def width(self) -> float:
45
+ """计算矩形的 width 几何属性。"""
46
+ return self.bbox[2] - self.bbox[0]
47
+
48
+ @property
49
+ def area(self) -> float:
50
+ """计算矩形的 area 几何属性。"""
51
+ return self.width * self.height
52
+
53
+ @property
54
+ def center(self) -> list[float]:
55
+ """计算矩形的 center 几何属性。"""
56
+ return [(self.bbox[0] + self.bbox[2]) / 2, (self.bbox[1] + self.bbox[3]) / 2]
57
+
58
+ @property
59
+ def size(self) -> list[float]:
60
+ """计算矩形的 size 几何属性。"""
61
+ return [self.width, self.height]
62
+
63
+ @property
64
+ def x_start(self) -> float:
65
+ """计算矩形的 x_start 几何属性。"""
66
+ return self.bbox[0]
67
+
68
+ @property
69
+ def y_start(self) -> float:
70
+ """计算矩形的 y_start 几何属性。"""
71
+ return self.bbox[1]
72
+
73
+ @property
74
+ def x_end(self) -> float:
75
+ """计算矩形的 x_end 几何属性。"""
76
+ return self.bbox[2]
77
+
78
+ @property
79
+ def y_end(self) -> float:
80
+ """计算矩形的 y_end 几何属性。"""
81
+ return self.bbox[3]
82
+
83
+ def merge(self, other: Bbox) -> Bbox:
84
+ """返回覆盖两个矩形的新对象。"""
85
+ self_bbox = self.bbox
86
+ other_bbox = other.bbox
87
+ return Bbox(
88
+ [
89
+ min(self_bbox[0], other_bbox[0]),
90
+ min(self_bbox[1], other_bbox[1]),
91
+ max(self_bbox[2], other_bbox[2]),
92
+ max(self_bbox[3], other_bbox[3]),
93
+ ]
94
+ )
95
+
96
+ def merge_inplace(self, other: Bbox) -> Bbox:
97
+ # Mutates this bbox; only safe on accumulator bboxes that aren't shared
98
+ """仅修改当前累加矩形。"""
99
+ self_bbox = self.bbox
100
+ other_bbox = other.bbox
101
+ if other_bbox[0] < self_bbox[0]:
102
+ self_bbox[0] = other_bbox[0]
103
+ if other_bbox[1] < self_bbox[1]:
104
+ self_bbox[1] = other_bbox[1]
105
+ if other_bbox[2] > self_bbox[2]:
106
+ self_bbox[2] = other_bbox[2]
107
+ if other_bbox[3] > self_bbox[3]:
108
+ self_bbox[3] = other_bbox[3]
109
+ return self
110
+
111
+ def overlap_x(self, other: Bbox) -> float:
112
+ """计算矩形的 overlap_x 几何属性。"""
113
+ return max(0, min(self.bbox[2], other.bbox[2]) - max(self.bbox[0], other.bbox[0]))
114
+
115
+ def overlap_y(self, other: Bbox) -> float:
116
+ """计算矩形的 overlap_y 几何属性。"""
117
+ return max(0, min(self.bbox[3], other.bbox[3]) - max(self.bbox[1], other.bbox[1]))
118
+
119
+ def intersection_area(self, other: Bbox) -> float:
120
+ """计算矩形的 intersection_area 几何属性。"""
121
+ return self.overlap_x(other) * self.overlap_y(other)
122
+
123
+ def intersection_pct(self, other: Bbox) -> float:
124
+ """计算矩形的 intersection_pct 几何属性。"""
125
+ if self.area <= 0:
126
+ return 0
127
+
128
+ intersection = self.intersection_area(other)
129
+ return intersection / self.area
130
+
131
+ def rotate(self, page_width: float, page_height: float, rotation: int) -> Bbox:
132
+ """将矩形转换到旋转后的页面坐标。"""
133
+ if rotation not in [0, 90, 180, 270]:
134
+ raise ValueError("Rotation must be one of [0, 90, 180, 270] degrees.")
135
+
136
+ x_min, y_min, x_max, y_max = self.bbox
137
+
138
+ if rotation == 0:
139
+ return Bbox(list(self.bbox))
140
+ elif rotation == 90:
141
+ new_x_min = page_height - y_max
142
+ new_y_min = x_min
143
+ new_x_max = page_height - y_min
144
+ new_y_max = x_max
145
+ elif rotation == 180:
146
+ new_x_min = page_width - x_max
147
+ new_y_min = page_height - y_max
148
+ new_x_max = page_width - x_min
149
+ new_y_max = page_height - y_min
150
+ elif rotation == 270:
151
+ new_x_min = y_min
152
+ new_y_min = page_width - x_max
153
+ new_x_max = y_max
154
+ new_y_max = page_width - x_min
155
+
156
+ # Ensure that x_min < x_max and y_min < y_max; must stay a list so
157
+ # merge_inplace can mutate it
158
+ rotated_bbox = [
159
+ min(new_x_min, new_x_max),
160
+ min(new_y_min, new_y_max),
161
+ max(new_x_min, new_x_max),
162
+ max(new_y_min, new_y_max),
163
+ ]
164
+
165
+ return Bbox(rotated_bbox)
166
+
167
+
168
+ class _CharValue(TypedDict):
169
+ bbox: Bbox
170
+ char: str
171
+ rotation: float
172
+ font: dict[str, Any]
173
+ char_idx: int
174
+
175
+
176
+ class Char(_CharValue, total=False):
177
+ """字符及原始索引映射,几何缺失保留为显式空值。"""
178
+
179
+ source_indices: tuple[int, ...]
180
+ raw_code: int
181
+ loose_bbox: tuple[float, float, float, float] | None
182
+ tight_bbox: tuple[float, float, float, float] | None
183
+ origin: tuple[float, float] | None
184
+
185
+
186
+ class Span(TypedDict):
187
+ """基础字体片段,包含已解码文本及原始字符引用。"""
188
+
189
+ bbox: Bbox
190
+ text: str
191
+ font: dict[str, Any]
192
+ chars: list[Char]
193
+ char_start_idx: int
194
+ char_end_idx: int
195
+ rotation: float
196
+ url: str
197
+ superscript: bool
198
+ subscript: bool
199
+
200
+
201
+ class Line(TypedDict):
202
+ """基础文本行,几何与片段顺序均保持可追溯。"""
203
+
204
+ spans: list[Span]
205
+ bbox: Bbox
206
+ rotation: float
207
+
208
+
209
+ Spans = list[Span]
210
+ Lines = list[Line]
211
+ __all__ = ["Bbox", "Char", "Span", "Line", "Spans", "Lines"]
@@ -0,0 +1,165 @@
1
+ # Portions derived from pdftext 0.7.1, Copyright Vik Paruchuri, Apache-2.0.
2
+ # Changed in DocVortex: direct PDFium extraction collects character and extended geometry together.
3
+ """在一次 PDFium 字符遍历内收集原始码值、字体及可选几何。"""
4
+
5
+ from __future__ import annotations
6
+
7
+ from ctypes import byref, c_double, c_int, create_string_buffer
8
+ import math
9
+ from typing import Any
10
+
11
+ import pypdfium2 as pdfium
12
+ import pypdfium2.raw as raw
13
+
14
+ from .contracts import Bbox, Char
15
+
16
+
17
+ def transform_point(
18
+ point: tuple[float, float], page_bbox: tuple[float, float, float, float], rotation: int
19
+ ) -> tuple[float, float]:
20
+ """保留浮点页面框,将 PDF 原始坐标转换到视觉页面坐标。"""
21
+ left, bottom, right, top = page_bbox
22
+ width, height = abs(right - left), abs(top - bottom)
23
+ x, y = point[0] - min(left, right), max(bottom, top) - point[1]
24
+ rotation %= 360
25
+ if rotation == 90:
26
+ return height - y, x
27
+ if rotation == 180:
28
+ return width - x, height - y
29
+ if rotation == 270:
30
+ return y, width - x
31
+ return x, y
32
+
33
+
34
+ def visual_bbox(
35
+ box: tuple[float, float, float, float], page_bbox: tuple[float, float, float, float], rotation: int
36
+ ) -> tuple[float, float, float, float] | None:
37
+ """转换原始矩形,几何缺失或零面积时返回空值而非伪造坐标。"""
38
+ left, bottom, right, top = box
39
+ points = [
40
+ transform_point(point, page_bbox, rotation) for point in ((left, bottom), (left, top), (right, bottom), (right, top))
41
+ ]
42
+ result = (min(p[0] for p in points), min(p[1] for p in points), max(p[0] for p in points), max(p[1] for p in points))
43
+ return result if all(math.isfinite(v) for v in result) and result[2] > result[0] and result[3] > result[1] else None
44
+
45
+
46
+ def _font_name(handle: Any, index: int, buffer: Any, flags: c_int) -> tuple[str, int]:
47
+ """复用字体缓冲区,超长名称按 PDFium 返回长度重新读取。"""
48
+ try:
49
+ length = raw.FPDFText_GetFontInfo(handle, index, buffer, len(buffer), byref(flags))
50
+ if length > len(buffer):
51
+ buffer = create_string_buffer(length)
52
+ raw.FPDFText_GetFontInfo(handle, index, buffer, length, byref(flags))
53
+ return (buffer.value.decode("utf-8", errors="replace"), flags.value) if length > 0 else ("", 0)
54
+ except pdfium.PdfiumError:
55
+ return "", 0
56
+
57
+
58
+ def get_chars(
59
+ textpage: pdfium.PdfTextPage, page_bbox: list[float], page_rotation: int, *, include_geometry: bool = False
60
+ ) -> list[Char]:
61
+ """读取原始字符记录;原始码值始终保留,随后统一解码和去重。"""
62
+ handle = textpage.raw
63
+ left, bottom, right, top = page_bbox
64
+ width, height = math.ceil(abs(right - left)), math.ceil(abs(top - bottom))
65
+ rect = raw.FS_RECTF()
66
+ tight_left, tight_right, tight_bottom, tight_top = c_double(), c_double(), c_double(), c_double()
67
+ origin_x, origin_y = c_double(), c_double()
68
+ font_buffer, font_flags = create_string_buffer(256), c_int()
69
+ fonts: dict[tuple[Any, ...], dict[str, Any]] = {}
70
+ chars: list[Char] = []
71
+ for index in range(textpage.count_chars()):
72
+ code = int(raw.FPDFText_GetUnicode(handle, index))
73
+ rotation = float(raw.FPDFText_GetCharAngle(handle, index))
74
+ loose: tuple[float, float, float, float] | None = None
75
+ tight: tuple[float, float, float, float] | None = None
76
+ if rotation == 0 or include_geometry:
77
+ try:
78
+ if raw.FPDFText_GetLooseCharBox(handle, index, rect):
79
+ loose = (float(rect.left), float(rect.bottom), float(rect.right), float(rect.top))
80
+ except Exception:
81
+ if rotation == 0:
82
+ raise
83
+ if rotation != 0 or include_geometry:
84
+ try:
85
+ if raw.FPDFText_GetCharBox(handle, index, tight_left, tight_right, tight_bottom, tight_top):
86
+ tight = (tight_left.value, tight_bottom.value, tight_right.value, tight_top.value)
87
+ except Exception:
88
+ if rotation != 0:
89
+ raise
90
+ selected = loose if rotation == 0 else tight
91
+ if selected is None:
92
+ raise pdfium.PdfiumError("Failed to get charbox.")
93
+ x0, y0, x1, y1 = selected
94
+ # 布局框保留基线的整数页面高度,原始扩展几何另用浮点页面框。
95
+ ys = (height - (y0 - bottom), height - (y1 - bottom))
96
+ box = Bbox([min(x0, x1) - left, min(ys), max(x0, x1) - left, max(ys)])
97
+ if page_rotation:
98
+ box = box.rotate(width, height, page_rotation)
99
+ name, flags = _font_name(handle, index, font_buffer, font_flags)
100
+ size, weight = raw.FPDFText_GetFontSize(handle, index), raw.FPDFText_GetFontWeight(handle, index)
101
+ font = fonts.setdefault((name, flags, size, weight), {"name": name, "flags": flags, "size": size, "weight": weight})
102
+ char: Char = {
103
+ "bbox": box,
104
+ "char": chr(code) if not 0xD800 <= code <= 0xDFFF else "\ufffd",
105
+ "rotation": rotation,
106
+ "font": font,
107
+ "char_idx": index,
108
+ "source_indices": (index,),
109
+ "raw_code": code,
110
+ }
111
+ if include_geometry:
112
+ char["loose_bbox"] = visual_bbox(loose, tuple(page_bbox), page_rotation) if loose else None
113
+ char["tight_bbox"] = visual_bbox(tight, tuple(page_bbox), page_rotation) if tight else None
114
+ char["origin"] = None
115
+ try:
116
+ if raw.FPDFText_GetCharOrigin(handle, index, origin_x, origin_y):
117
+ origin = transform_point((origin_x.value, origin_y.value), tuple(page_bbox), page_rotation)
118
+ if all(math.isfinite(v) for v in origin):
119
+ char["origin"] = origin
120
+ except Exception:
121
+ pass
122
+ chars.append(char)
123
+ return chars
124
+
125
+
126
+ def deduplicate_chars(chars: list[Char]) -> list[Char]:
127
+ """按词文本、字体、方向及取整位置去重,保留最早原始索引。"""
128
+ if not chars:
129
+ return []
130
+ groups: list[list[Char]] = [[chars[0]]]
131
+ for char in chars[1:]:
132
+ previous = groups[-1][-1]
133
+ if (
134
+ previous["char"] in {"\x02", "\n", " "}
135
+ or char["font"] != previous["font"]
136
+ or char["rotation"] != previous["rotation"]
137
+ ):
138
+ groups.append([])
139
+ groups[-1].append(char)
140
+ seen: dict[tuple[Any, ...], list[Char]] = {}
141
+ result: list[Char] = []
142
+ for group in groups:
143
+ box = group[0]["bbox"].copy()
144
+ for char in group[1:]:
145
+ box.merge_inplace(char["bbox"])
146
+ font = group[0]["font"]
147
+ key = (
148
+ tuple(round(float(v), 0) for v in box.bbox),
149
+ "".join(c["char"] for c in group),
150
+ group[0]["rotation"],
151
+ tuple(font.get(k) for k in ("name", "flags", "size", "weight")),
152
+ )
153
+ if key not in seen:
154
+ seen[key] = group
155
+ result.extend(group)
156
+ else:
157
+ for retained, duplicate in zip(seen[key], group):
158
+ retained["source_indices"] = (
159
+ *retained.get("source_indices", (retained["char_idx"],)),
160
+ *duplicate.get("source_indices", (duplicate["char_idx"],)),
161
+ )
162
+ return result
163
+
164
+
165
+ __all__ = ["transform_point", "visual_bbox"]
@@ -0,0 +1,16 @@
1
+ """几何载荷与独立矩形类型之间的纯值转换。"""
2
+
3
+ from __future__ import annotations
4
+ from .contracts import Bbox
5
+
6
+
7
+ def char_bbox_values(bbox: object) -> tuple[float, float, float, float] | None:
8
+ """读取固定四元坐标,拒绝结构不完整的几何对象。"""
9
+ if isinstance(bbox, Bbox):
10
+ bbox = bbox.bbox
11
+ if isinstance(bbox, (tuple, list)) and len(bbox) == 4:
12
+ return tuple(float(value) for value in bbox)
13
+ return None
14
+
15
+
16
+ __all__ = ["char_bbox_values"]
@@ -0,0 +1,162 @@
1
+ # Portions derived from pdftext 0.7.1, Copyright Vik Paruchuri, Apache-2.0.
2
+ # Changed in DocVortex: grouping uses owned dictionaries without upstream container adapters.
3
+ """基础文本行与上下标分组;保留已验证的几何判断。"""
4
+
5
+ from __future__ import annotations
6
+ import math
7
+ import unicodedata
8
+ from .contracts import Line, Lines, Spans
9
+
10
+
11
+ def is_math_symbol(char: str) -> bool:
12
+ """判断单字符数学符号。"""
13
+ if len(char) != 1:
14
+ return False
15
+
16
+ category = unicodedata.category(char)
17
+ return category == "Sm"
18
+
19
+
20
+ def _top2(values: list[float]) -> tuple[float, int, float]:
21
+ # Returns (max1, max1_idx, max2) so that max-excluding-index can be answered in O(1)
22
+ """在线性时间内找到两个最大值。"""
23
+ max1 = max2 = float("-inf")
24
+ max1_idx = -1
25
+ for idx, v in enumerate(values):
26
+ if v > max1:
27
+ max2 = max1
28
+ max1 = v
29
+ max1_idx = idx
30
+ elif v > max2:
31
+ max2 = v
32
+ return max1, max1_idx, max2
33
+
34
+
35
+ def _bottom2(values: list[float]) -> tuple[float, int, float]:
36
+ """在线性时间内找到两个最小值。"""
37
+ min1 = min2 = float("inf")
38
+ min1_idx = -1
39
+ for idx, v in enumerate(values):
40
+ if v < min1:
41
+ min2 = min1
42
+ min1 = v
43
+ min1_idx = idx
44
+ elif v < min2:
45
+ min2 = v
46
+ return min1, min1_idx, min2
47
+
48
+
49
+ def assign_scripts(lines: Lines, height_threshold: float = 0.8, line_distance_threshold: float = 0.1) -> None:
50
+ """根据邻接片段几何设置基础上下标提示。"""
51
+ for line in lines:
52
+ spans = line["spans"]
53
+ if len(spans) < 2:
54
+ continue
55
+
56
+ line_bbox = line["bbox"].bbox
57
+ line_height = line_bbox[3] - line_bbox[1]
58
+ # Skip vertical lines
59
+ if line_height > line_bbox[2] - line_bbox[0]:
60
+ continue
61
+
62
+ # Precompute per-span geometry once; the loop below would otherwise
63
+ # recompute these via Bbox properties O(n^2) times per line
64
+ heights = []
65
+ y_starts = []
66
+ y_ends = []
67
+ v_above = []
68
+ v_below = []
69
+ for s in spans:
70
+ bbox = s["bbox"].bbox
71
+ height = bbox[3] - bbox[1]
72
+ heights.append(height)
73
+ y_starts.append(bbox[1])
74
+ y_ends.append(bbox[3])
75
+ v_above.append(bbox[1] - height * line_distance_threshold)
76
+ v_below.append(bbox[3] + height * line_distance_threshold)
77
+
78
+ above_max1, above_max1_idx, above_max2 = _top2(v_above)
79
+ below_min1, below_min1_idx, below_min2 = _bottom2(v_below)
80
+ max_line_height = max(1, line_height)
81
+ last_idx = len(spans) - 1
82
+
83
+ for i, span in enumerate(spans):
84
+ is_first = i == 0 or not spans[i - 1]["text"].strip()
85
+ is_last = i == last_idx or not spans[i + 1]["text"].strip()
86
+ span_height = heights[i]
87
+ span_top = y_starts[i]
88
+ span_bottom = y_ends[i]
89
+
90
+ line_fullheight = span_height / max_line_height <= height_threshold
91
+ next_fullheight = is_last or span_height / max(1, heights[i + 1]) <= height_threshold
92
+ prev_fullheight = is_first or span_height / max(1, heights[i - 1]) <= height_threshold
93
+
94
+ # any(span_top < v_above[j] for j != i) == span_top < max(v_above excluding i)
95
+ above = span_top < (above_max2 if i == above_max1_idx else above_max1)
96
+ prev_above = is_first or span_top < y_starts[i - 1]
97
+ next_above = is_last or span_top < y_starts[i + 1]
98
+
99
+ below = span_bottom > (below_min2 if i == below_min1_idx else below_min1)
100
+ prev_below = is_first or span_bottom > y_ends[i - 1]
101
+ next_below = is_last or span_bottom > y_ends[i + 1]
102
+
103
+ span_text = span["text"].strip()
104
+ span_text_okay = all(
105
+ [
106
+ (len(span_text) == 1 or span_text.isdigit()), # Ensure that the span text is a single char or a number
107
+ span_text.isalnum()
108
+ or is_math_symbol(span_text), # Ensure that the span text is an alphanumeric or a math symbol
109
+ ]
110
+ )
111
+
112
+ if all([(prev_fullheight or next_fullheight), (prev_above or next_above), above, line_fullheight, span_text_okay]):
113
+ span["superscript"] = True
114
+ elif all(
115
+ [(prev_fullheight or next_fullheight), (prev_below or next_below), below, line_fullheight, span_text_okay]
116
+ ):
117
+ span["subscript"] = True
118
+
119
+
120
+ def get_lines(spans: Spans) -> Lines:
121
+ """按换行、角度和位置将片段聚合为行。"""
122
+ lines: Lines = []
123
+ line: Line = None
124
+
125
+ def line_break() -> None:
126
+ """以当前片段开始一个新的文本行。"""
127
+ lines.append({"spans": [span], "bbox": span["bbox"].copy(), "rotation": span["rotation"]})
128
+
129
+ for span in spans:
130
+ if lines:
131
+ line = lines[-1]
132
+
133
+ if not line:
134
+ line_break()
135
+ continue
136
+
137
+ # we break if the previous span ends with a linebreak
138
+ last_text = line["spans"][-1]["text"]
139
+ if any(last_text.endswith(suffix) for suffix in ["\n", "\x02"]):
140
+ line_break()
141
+ continue
142
+
143
+ # rotations are radians from FPDFText_GetCharAngle; compare circularly.
144
+ # Only break on roughly perpendicular text: pdfium reports a 180-degree
145
+ # flip for ordinary text rendered with negative-scale matrices, which
146
+ # still belongs to the same visual line
147
+ if span["rotation"] != line["rotation"]:
148
+ rotation_diff = abs(span["rotation"] - line["rotation"]) % (2 * math.pi)
149
+ rotation_diff = min(rotation_diff, 2 * math.pi - rotation_diff)
150
+ if math.radians(45) <= rotation_diff <= math.radians(135):
151
+ line_break()
152
+ continue
153
+
154
+ # sometimes pdfium doesn't inject a linebreak, so we check the span positions
155
+ if span["bbox"].y_start > line["bbox"].y_end:
156
+ line_break()
157
+ continue
158
+
159
+ line["spans"].append(span)
160
+ line["bbox"].merge_inplace(span["bbox"])
161
+
162
+ return lines