docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,692 @@
1
+ """依据栏带、缩进和排版重置寻找正文行分组边界。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import statistics
7
+ from typing import Sequence
8
+
9
+ from .....schema import BBox
10
+ from .....foundation.text import is_hyphen_at_line_end
11
+ from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_x, _rotate_bbox_to_upright
12
+ from ..line_layout import (
13
+ _connection_crosses_table,
14
+ _effective_body_text_row_gap,
15
+ _effective_text_row_gap,
16
+ _horizontal_rule_separates_rows,
17
+ _line_effective_height,
18
+ _line_tight_output_bbox,
19
+ _title_fonts_compatible,
20
+ )
21
+ from ..models import _LineItem, _LocalAxisLine, _TextLane
22
+ from .common import (
23
+ _ABSTRACT_METADATA_RE,
24
+ _BULLET_ITEM_RE,
25
+ _EMAIL_METADATA_RE,
26
+ _FRONT_MATTER_FIELD_RE,
27
+ _LABELLED_METADATA_RE,
28
+ _LIST_ITEM_RE,
29
+ _REFERENCE_ENTRY_RE,
30
+ _URL_LINE_RE,
31
+ )
32
+
33
+
34
+ def _local_tight_output_line_bboxes(
35
+ lines: Sequence[_LineItem],
36
+ page_size: tuple[float, float],
37
+ angle: int,
38
+ ) -> tuple[list[BBox], bool]:
39
+ """返回与原行顺序一致的 tight+1pt 局部框及是否存在可靠候选。"""
40
+
41
+ output = []
42
+ changed = False
43
+ for line in lines:
44
+ candidate = _line_tight_output_bbox(line, page_size)
45
+ output.append(
46
+ _rotate_bbox_to_upright(
47
+ candidate or line.bbox,
48
+ page_size,
49
+ angle,
50
+ )
51
+ )
52
+ changed = changed or candidate is not None
53
+ return output, changed
54
+
55
+
56
+ def _starts_structural_reference_entry(
57
+ previous: tuple[_LineItem, BBox],
58
+ current: tuple[_LineItem, BBox],
59
+ ) -> bool:
60
+ """仅在编号行相对续行明显左突时确认新的参考文献条目。"""
61
+
62
+ if _REFERENCE_ENTRY_RE.match(current[0].text.strip()) is None:
63
+ return False
64
+ previous_height = _line_effective_height(*previous)
65
+ current_height = _line_effective_height(*current)
66
+ pair_height = max(previous_height, current_height)
67
+ return (
68
+ current[1][0] <= previous[1][0] - max(5.0, 0.6 * min(previous_height, current_height))
69
+ and -0.75 * pair_height <= _effective_text_row_gap(previous, current) <= 1.5 * pair_height
70
+ )
71
+
72
+
73
+ def _build_hanging_indent_group_map(
74
+ lane: _TextLane,
75
+ table_bboxes: list[BBox],
76
+ axis_lines: list[_LocalAxisLine],
77
+ ) -> dict[int, int]:
78
+ """仅按重复的左突首行和稳定续行缩进识别悬挂缩进条目。"""
79
+
80
+ if len(lane.lines) < 4:
81
+ return {}
82
+ rows = sorted(
83
+ (item for item in lane.lines if item[0].semantic_type is None),
84
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
85
+ )
86
+ if len(rows) < 4:
87
+ return {}
88
+ median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in rows)
89
+ start_tolerance = max(5.0, 0.65 * median_height)
90
+ minimum_indent = max(7.0, 0.8 * median_height)
91
+ continuation_tolerance = max(4.0, 0.55 * median_height)
92
+
93
+ def rows_are_adjacent(
94
+ previous: tuple[_LineItem, BBox],
95
+ current: tuple[_LineItem, BBox],
96
+ ) -> bool:
97
+ """检查相邻行的净空和几何障碍是否允许组成同一缩进序列。"""
98
+
99
+ effective_gap = _effective_text_row_gap(previous, current)
100
+ top_pitch = current[1][1] - previous[1][1]
101
+ robust_pitch_fallback = 0.5 * median_height <= top_pitch <= 1.8 * median_height
102
+ if not -0.6 * median_height <= effective_gap <= 1.3 * median_height and not robust_pitch_fallback:
103
+ return False
104
+ if _connection_crosses_table(
105
+ previous[0].bbox,
106
+ current[0].bbox,
107
+ table_bboxes,
108
+ ):
109
+ return False
110
+ return not _horizontal_rule_separates_rows(
111
+ previous[1],
112
+ current[1],
113
+ lane,
114
+ axis_lines,
115
+ )
116
+
117
+ def consume_entry(
118
+ start_index: int,
119
+ start_left: float,
120
+ expected_continuation_left: float | None,
121
+ *,
122
+ require_next_start: bool,
123
+ ) -> tuple[int, float] | None:
124
+ """消费一个左突首行及其续行,并返回下一条首行位置。"""
125
+
126
+ lane_width = max(0.1, lane.right - lane.left)
127
+ full_width_midparagraph_entry = (
128
+ rows[start_index][1][2] - rows[start_index][1][0] >= 0.8 * lane_width
129
+ and start_index + 1 < len(rows)
130
+ and rows[start_index + 1][1][2] - rows[start_index + 1][1][0] <= 0.75 * lane_width
131
+ )
132
+ if (
133
+ start_index > 0
134
+ and rows_are_adjacent(
135
+ rows[start_index - 1],
136
+ rows[start_index],
137
+ )
138
+ and abs(rows[start_index - 1][1][0] - start_left) <= start_tolerance
139
+ and not full_width_midparagraph_entry
140
+ ):
141
+ # 同左缘正文仍在连续时不能从段落中部启动悬挂条目序列。
142
+ return None
143
+ if (
144
+ start_index > 0
145
+ and is_hyphen_at_line_end(rows[start_index - 1][0].text)
146
+ and rows_are_adjacent(rows[start_index - 1], rows[start_index])
147
+ ):
148
+ # 排版断词后的下一物理行属于前文,不能被缩进几何误当成新条目首行。
149
+ return None
150
+ continuation_index = start_index + 1
151
+ if continuation_index >= len(rows):
152
+ return None
153
+ first_continuation = rows[continuation_index]
154
+ if not rows_are_adjacent(rows[start_index], first_continuation):
155
+ return None
156
+ continuation_left = first_continuation[1][0]
157
+ if continuation_left < start_left + minimum_indent:
158
+ return None
159
+ if (
160
+ expected_continuation_left is not None
161
+ and abs(continuation_left - expected_continuation_left) > continuation_tolerance
162
+ ):
163
+ return None
164
+
165
+ continuation_index += 1
166
+ while continuation_index < len(rows):
167
+ previous = rows[continuation_index - 1]
168
+ current = rows[continuation_index]
169
+ current_left = current[1][0]
170
+ if not rows_are_adjacent(previous, current):
171
+ break
172
+ if current_left < start_left + minimum_indent:
173
+ break
174
+ if abs(current_left - continuation_left) > continuation_tolerance:
175
+ break
176
+ continuation_index += 1
177
+
178
+ if not require_next_start:
179
+ return continuation_index, continuation_left
180
+ if continuation_index >= len(rows):
181
+ return None
182
+ if not rows_are_adjacent(rows[continuation_index - 1], rows[continuation_index]):
183
+ return None
184
+ if abs(rows[continuation_index][1][0] - start_left) > start_tolerance:
185
+ return None
186
+ return continuation_index, continuation_left
187
+
188
+ group_map: dict[int, int] = {}
189
+ group_index = 0
190
+ row_index = 0
191
+ while row_index < len(rows) - 3:
192
+ start_left = rows[row_index][1][0]
193
+ first_entry = consume_entry(
194
+ row_index,
195
+ start_left,
196
+ None,
197
+ require_next_start=True,
198
+ )
199
+ if first_entry is None:
200
+ row_index += 1
201
+ continue
202
+
203
+ _next_start_index, continuation_left = first_entry
204
+ start_indices = [row_index]
205
+ current_start_index = row_index
206
+ end_index: int | None = None
207
+ while True:
208
+ next_entry = consume_entry(
209
+ current_start_index,
210
+ start_left,
211
+ continuation_left,
212
+ require_next_start=True,
213
+ )
214
+ if next_entry is None:
215
+ final_entry = consume_entry(
216
+ current_start_index,
217
+ start_left,
218
+ continuation_left,
219
+ require_next_start=False,
220
+ )
221
+ if final_entry is not None:
222
+ end_index = final_entry[0]
223
+ break
224
+ next_start_index, _continuation_left = next_entry
225
+ prospective_entry = consume_entry(
226
+ next_start_index,
227
+ start_left,
228
+ continuation_left,
229
+ require_next_start=False,
230
+ )
231
+ if prospective_entry is None:
232
+ # 当前条目已经完整确认;后面的普通左对齐段落只作为终止边界,
233
+ # 不能让它反向使此前所有悬挂缩进条目失效。
234
+ end_index = next_start_index
235
+ break
236
+ start_indices.append(next_start_index)
237
+ current_start_index = next_start_index
238
+ if len(start_indices) < 2 or end_index is None:
239
+ row_index += 1
240
+ continue
241
+
242
+ entry_ranges = [
243
+ (start, end)
244
+ for start, end in zip(
245
+ start_indices,
246
+ [*start_indices[1:], end_index],
247
+ strict=True,
248
+ )
249
+ ]
250
+ for start, end in entry_ranges:
251
+ for line, _bbox in rows[start:end]:
252
+ group_map[line.source_index] = group_index
253
+ group_index += 1
254
+ row_index = end_index
255
+
256
+ return group_map
257
+
258
+
259
+ def _infer_local_text_lane_map(lane: _TextLane) -> dict[int, _TextLane]:
260
+ """从连续同左缘正文推导局部栏宽,修正跨栏上文污染的全宽栏带。"""
261
+
262
+ if lane.is_span or len(lane.lines) < 3:
263
+ return {}
264
+ rows = sorted(
265
+ lane.lines,
266
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
267
+ )
268
+ median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in rows)
269
+ left_tolerance = max(3.0, 0.75 * median_height)
270
+ height_ratio_limit = 1.25
271
+ runs: list[list[tuple[_LineItem, BBox]]] = []
272
+ current_run: list[tuple[_LineItem, BBox]] = []
273
+
274
+ def submit_run() -> None:
275
+ """提交当前连续正文行,语义行和明显左缘变化都会结束该局部区段。"""
276
+
277
+ nonlocal current_run
278
+ if current_run:
279
+ runs.append(current_run)
280
+ current_run = []
281
+
282
+ for item in rows:
283
+ line, bbox = item
284
+ if line.semantic_type is not None:
285
+ submit_run()
286
+ continue
287
+ if not current_run:
288
+ current_run = [item]
289
+ continue
290
+ run_left = statistics.median(member[1][0] for member in current_run)
291
+ run_heights = [_line_effective_height(member, member_bbox) for member, member_bbox in current_run]
292
+ current_height = _line_effective_height(line, bbox)
293
+ if (
294
+ abs(bbox[0] - run_left) <= left_tolerance
295
+ and max([*run_heights, current_height]) / max(0.1, min([*run_heights, current_height])) <= height_ratio_limit
296
+ ):
297
+ current_run.append(item)
298
+ else:
299
+ submit_run()
300
+ current_run = [item]
301
+ submit_run()
302
+
303
+ global_width = max(0.1, lane.right - lane.left)
304
+ local_by_source: dict[int, _TextLane] = {}
305
+ for run in runs:
306
+ if len(run) < 3:
307
+ continue
308
+ local_left = statistics.median(bbox[0] for _line, bbox in run)
309
+ local_right = max(bbox[2] for _line, bbox in run)
310
+ local_width = max(0.1, local_right - local_left)
311
+ wide_support = sum(bbox[2] - bbox[0] >= 0.7 * local_width for _line, bbox in run)
312
+ if global_width < 1.4 * local_width or wide_support < 3:
313
+ continue
314
+ local_lane = _TextLane(
315
+ left=local_left,
316
+ right=local_right,
317
+ lines=run,
318
+ is_span=False,
319
+ )
320
+ for line, _bbox in run:
321
+ local_by_source[line.source_index] = local_lane
322
+ return local_by_source
323
+
324
+
325
+ def _structured_text_break_sources(
326
+ lane: _TextLane,
327
+ regular_gap: float,
328
+ gap_mad: float,
329
+ ) -> set[int]:
330
+ """用重复强调首行和前行右侧留白确认结构化正文的新段起点。"""
331
+
332
+ rows = sorted(
333
+ lane.lines,
334
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
335
+ )
336
+ lane_width = max(0.1, lane.right - lane.left)
337
+ break_sources: set[int] = set()
338
+ regions: list[list[tuple[_LineItem, BBox]]] = []
339
+ for row in rows:
340
+ if row[0].semantic_type is not None:
341
+ if regions and regions[-1]:
342
+ regions.append([])
343
+ continue
344
+ if not regions:
345
+ regions.append([])
346
+ regions[-1].append(row)
347
+
348
+ for region in regions:
349
+ candidates: list[int] = []
350
+ for index, (line, bbox) in enumerate(region):
351
+ height = _line_effective_height(line, bbox)
352
+ line_width = bbox[2] - bbox[0]
353
+ if (
354
+ line.leading_emphasis_width is not None
355
+ and line.leading_emphasis_width <= 0.2 * lane_width
356
+ and line_width >= 0.95 * lane_width
357
+ and abs(bbox[0] - lane.left) <= 0.75 * height
358
+ ):
359
+ candidates.append(index)
360
+ if len(candidates) < 3:
361
+ continue
362
+ for index in candidates:
363
+ if index == 0:
364
+ continue
365
+ previous = region[index - 1]
366
+ current = region[index]
367
+ pair_height = max(
368
+ _line_effective_height(*previous),
369
+ _line_effective_height(*current),
370
+ )
371
+ previous_fill = (previous[1][2] - lane.left) / lane_width
372
+ vertical_gap = _effective_text_row_gap(previous, current)
373
+ if previous_fill <= 0.8 and -0.25 * pair_height <= vertical_gap <= regular_gap + max(
374
+ 0.75 * pair_height, 3.0 * gap_mad
375
+ ):
376
+ break_sources.add(current[0].source_index)
377
+ return break_sources
378
+
379
+
380
+ def _isolated_indented_paragraph_break_sources(
381
+ lane: _TextLane,
382
+ regular_gap: float,
383
+ gap_mad: float,
384
+ ) -> set[int]:
385
+ """识别短终止尾行之后的缩进首行,并要求下一行回到稳定栏左缘。"""
386
+
387
+ rows = sorted(
388
+ (item for item in lane.lines if item[0].semantic_type is None),
389
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
390
+ )
391
+ lane_width = max(0.1, lane.right - lane.left)
392
+ output: set[int] = set()
393
+ terminal_re = re.compile(r"[.!?。!?::;;][\]\)})】》”’'\"]*$")
394
+ for previous, current, following in zip(
395
+ rows,
396
+ rows[1:],
397
+ rows[2:],
398
+ ):
399
+ previous_height = _line_effective_height(*previous)
400
+ current_height = _line_effective_height(*current)
401
+ following_height = _line_effective_height(*following)
402
+ pair_height = max(
403
+ previous_height,
404
+ current_height,
405
+ following_height,
406
+ )
407
+ current_indent = current[1][0] - lane.left
408
+ if (
409
+ previous[1][2] - previous[1][0] > 0.3 * lane_width
410
+ or terminal_re.search(previous[0].text.rstrip()) is None
411
+ or not max(5.0, 0.65 * pair_height) <= current_indent <= 3.0 * pair_height
412
+ or current[1][2] - current[1][0] < 0.75 * lane_width
413
+ or abs(following[1][0] - lane.left) > 0.75 * pair_height
414
+ or following[1][2] - following[1][0] < 0.65 * lane_width
415
+ or not _title_fonts_compatible(current[0], following[0])
416
+ ):
417
+ continue
418
+ first_gap = _effective_body_text_row_gap(previous, current)
419
+ second_gap = _effective_body_text_row_gap(current, following)
420
+ gap_limit = regular_gap + max(
421
+ 0.75 * pair_height,
422
+ 3.0 * gap_mad,
423
+ )
424
+ if -0.25 * pair_height <= first_gap <= gap_limit and -0.25 * pair_height <= second_gap <= gap_limit:
425
+ output.add(current[0].source_index)
426
+ return output
427
+
428
+
429
+ def _centered_visual_reset_break_sources(
430
+ lane: _TextLane,
431
+ visual_bboxes: Sequence[BBox],
432
+ local_page_height: float,
433
+ ) -> set[int]:
434
+ """识别视觉主体下方短居中行到更宽居中行的独立注释重启。"""
435
+
436
+ if not visual_bboxes:
437
+ return set()
438
+ rows = sorted(
439
+ (item for item in lane.lines if item[0].semantic_type is None),
440
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
441
+ )
442
+ output: set[int] = set()
443
+ for previous, current in zip(rows, rows[1:]):
444
+ previous_bbox = previous[1]
445
+ current_bbox = current[1]
446
+ previous_width = previous_bbox[2] - previous_bbox[0]
447
+ current_width = current_bbox[2] - current_bbox[0]
448
+ pair_height = max(
449
+ _line_effective_height(*previous),
450
+ _line_effective_height(*current),
451
+ )
452
+ if (
453
+ previous_width > 0.7 * current_width
454
+ or current_bbox[0] > previous_bbox[0] - 0.25 * pair_height
455
+ or current_bbox[2] < previous_bbox[2] + 0.25 * pair_height
456
+ or abs(_bbox_center_x(previous_bbox) - _bbox_center_x(current_bbox)) > 0.1 * current_width
457
+ ):
458
+ continue
459
+ vertical_gap = _effective_text_row_gap(previous, current)
460
+ if not -0.25 * pair_height <= vertical_gap <= 0.75 * pair_height:
461
+ continue
462
+ if any(
463
+ -0.25 * pair_height <= previous_bbox[1] - visual_bbox[3] <= max(2.0 * pair_height, 0.03 * local_page_height)
464
+ and _bbox_axis_overlap_ratio(current_bbox, visual_bbox, axis="x") >= 0.8
465
+ and abs(_bbox_center_x(current_bbox) - _bbox_center_x(visual_bbox))
466
+ <= 0.12 * max(current_width, visual_bbox[2] - visual_bbox[0])
467
+ for visual_bbox in visual_bboxes
468
+ ):
469
+ output.add(current[0].source_index)
470
+ return output
471
+
472
+
473
+ def _leading_typography_reset_break_sources(
474
+ lane: _TextLane,
475
+ regular_gap: float,
476
+ gap_mad: float,
477
+ ) -> set[int]:
478
+ """识别短尾之后以独立行首字体 run 开启的宽行结构段。"""
479
+
480
+ rows = sorted(
481
+ (item for item in lane.lines if item[0].semantic_type is None),
482
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
483
+ )
484
+ lane_width = max(0.1, lane.right - lane.left)
485
+ output: set[int] = set()
486
+ for previous, current in zip(rows, rows[1:]):
487
+ previous_width = previous[1][2] - previous[1][0]
488
+ current_width = current[1][2] - current[1][0]
489
+ pair_height = max(
490
+ _line_effective_height(*previous),
491
+ _line_effective_height(*current),
492
+ )
493
+ if (
494
+ current[0].leading_typography_width is None
495
+ or current[0].leading_typography_width > 0.2 * lane_width
496
+ or previous_width > 0.45 * lane_width
497
+ or current_width < 0.75 * lane_width
498
+ or abs(previous[1][0] - lane.left) > 0.75 * pair_height
499
+ or abs(current[1][0] - lane.left) > 0.75 * pair_height
500
+ or current[0].formula_candidate_only
501
+ or current[0].compact_formula_cluster
502
+ or current[0].inline_math_regions
503
+ ):
504
+ continue
505
+ vertical_gap = _effective_body_text_row_gap(previous, current)
506
+ if (
507
+ -0.25 * pair_height
508
+ <= vertical_gap
509
+ <= regular_gap
510
+ + max(
511
+ 0.75 * pair_height,
512
+ 3.0 * gap_mad,
513
+ )
514
+ ):
515
+ output.add(current[0].source_index)
516
+ return output
517
+
518
+
519
+ def _formula_style_text_row_break_sources(
520
+ lane: _TextLane,
521
+ ) -> set[int]:
522
+ """按相邻显示行几何拆分被公式检测回退为正文的独立文本行。"""
523
+
524
+ rows = sorted(
525
+ (item for item in lane.lines if item[0].semantic_type is None),
526
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
527
+ )
528
+ lane_width = max(0.1, lane.right - lane.left)
529
+ matching_edges: set[int] = set()
530
+ for index, (previous, current) in enumerate(zip(rows, rows[1:])):
531
+ previous_line, previous_bbox = previous
532
+ current_line, current_bbox = current
533
+ if not (previous_line.paragraph_formula_context and current_line.paragraph_formula_context):
534
+ continue
535
+ previous_height = _line_effective_height(*previous)
536
+ current_height = _line_effective_height(*current)
537
+ minimum_height = min(previous_height, current_height)
538
+ maximum_height = max(previous_height, current_height)
539
+ if minimum_height < 0.75 * maximum_height:
540
+ continue
541
+ previous_width = previous_bbox[2] - previous_bbox[0]
542
+ current_width = current_bbox[2] - current_bbox[0]
543
+ if min(previous_width, current_width) < 0.45 * lane_width or max(previous_width, current_width) > 0.95 * lane_width:
544
+ continue
545
+ lane_center = 0.5 * (lane.left + lane.right)
546
+ if (
547
+ abs(_bbox_center_x(previous_bbox) - lane_center) > 0.15 * lane_width
548
+ or abs(_bbox_center_x(current_bbox) - lane_center) > 0.15 * lane_width
549
+ ):
550
+ continue
551
+ vertical_overlap = max(
552
+ 0.0,
553
+ min(previous_bbox[3], current_bbox[3]) - max(previous_bbox[1], current_bbox[1]),
554
+ )
555
+ top_pitch = current_bbox[1] - previous_bbox[1]
556
+ pair_height = statistics.median((previous_height, current_height))
557
+ if vertical_overlap <= 0.2 * minimum_height and 0.9 * pair_height <= top_pitch <= 2.0 * pair_height:
558
+ matching_edges.add(index)
559
+
560
+ output: set[int] = set()
561
+ for index in matching_edges:
562
+ output.add(rows[index][0].source_index)
563
+ output.add(rows[index + 1][0].source_index)
564
+ if index + 2 < len(rows):
565
+ # 同时保护显示行组后的正文起点,避免上下文恢复阶段重新跨界合并。
566
+ output.add(rows[index + 2][0].source_index)
567
+ return output
568
+
569
+
570
+ def _front_matter_keyword_break_sources(
571
+ lane: _TextLane,
572
+ local_page_height: float,
573
+ page_index: int | None,
574
+ ) -> set[int]:
575
+ """把首页关键词和文献元数据行固定为独立文本块起点。"""
576
+
577
+ if page_index != 0:
578
+ return set()
579
+ return {
580
+ line.source_index
581
+ for line, bbox in lane.lines
582
+ if line.semantic_type is None
583
+ and bbox[1] <= 0.65 * local_page_height
584
+ and _FRONT_MATTER_FIELD_RE.match(line.text) is not None
585
+ }
586
+
587
+
588
+ def _component_starts_with_emphasized_row(
589
+ lines: list[_LineItem],
590
+ ) -> bool:
591
+ """识别行内强调或首行字重显著高于后续正文的组件起点。"""
592
+
593
+ if not lines:
594
+ return False
595
+ if lines[0].leading_emphasis_width is not None:
596
+ return True
597
+ first_weight = lines[0].dominant_font_weight
598
+ following_weights = [line.dominant_font_weight for line in lines[1:] if line.dominant_font_weight is not None]
599
+ if first_weight is None or not following_weights:
600
+ return False
601
+ body_weight = statistics.median(following_weights)
602
+ return first_weight - body_weight >= 100.0 and first_weight >= 1.15 * max(1.0, body_weight)
603
+
604
+
605
+ def _explicit_text_break_sources(
606
+ lane: _TextLane,
607
+ ) -> set[int]:
608
+ """用通用列表标记和 E-mail 元数据确认正文中的显式硬分段。"""
609
+
610
+ rows = sorted(
611
+ (item for item in lane.lines if item[0].semantic_type is None),
612
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
613
+ )
614
+ output = {line.source_index for line, _bbox in rows if _ABSTRACT_METADATA_RE.match(line.text) is not None}
615
+ lane_width = max(0.1, lane.right - lane.left)
616
+ output.update(
617
+ line.source_index
618
+ for line, bbox in rows
619
+ if _BULLET_ITEM_RE.match(line.text) is not None and bbox[2] - bbox[0] >= 0.8 * lane_width
620
+ )
621
+ for index, (line, bbox) in enumerate(rows):
622
+ if _EMAIL_METADATA_RE.match(line.text) is None or index == 0:
623
+ continue
624
+ previous_line, previous_bbox = rows[index - 1]
625
+ pair_height = max(
626
+ _line_effective_height(previous_line, previous_bbox),
627
+ _line_effective_height(line, bbox),
628
+ )
629
+ if abs(bbox[0] - previous_bbox[0]) <= 0.75 * pair_height:
630
+ output.add(line.source_index)
631
+ for row_index, (previous, current) in enumerate(
632
+ zip(rows, rows[1:]),
633
+ ):
634
+ previous_is_label = _LABELLED_METADATA_RE.match(
635
+ previous[0].text,
636
+ )
637
+ current_is_label = _LABELLED_METADATA_RE.match(
638
+ current[0].text,
639
+ )
640
+ label_pair_height = max(
641
+ _line_effective_height(*previous),
642
+ _line_effective_height(*current),
643
+ )
644
+ if (
645
+ previous_is_label is not None
646
+ and current_is_label is not None
647
+ and _URL_LINE_RE.match(current[0].text) is None
648
+ and len(previous_is_label.group("label")) >= 4
649
+ and len(current_is_label.group("label")) >= 4
650
+ and any("\u3400" <= char <= "\u9fff" for char in previous_is_label.group("label"))
651
+ and any("\u3400" <= char <= "\u9fff" for char in current_is_label.group("label"))
652
+ and previous_is_label.group("label").casefold() != current_is_label.group("label").casefold()
653
+ and current[1][1] - previous[1][1] <= 2.0 * label_pair_height
654
+ and previous[1][2] - previous[1][0] <= 0.75 * lane_width
655
+ and current[1][2] - current[1][0] <= 0.75 * lane_width
656
+ ):
657
+ output.add(current[0].source_index)
658
+ pair_height = max(
659
+ _line_effective_height(*previous),
660
+ _line_effective_height(*current),
661
+ )
662
+ next_row = rows[row_index + 2] if row_index + 2 < len(rows) else None
663
+ indented_item_continuation = (
664
+ current[1][0] - lane.left >= max(5.0, 0.65 * pair_height)
665
+ and next_row is not None
666
+ and next_row[1][0] - lane.left <= 0.5 * pair_height
667
+ and 0.5 * pair_height <= next_row[1][1] - current[1][1] <= 2.25 * pair_height
668
+ )
669
+ if (
670
+ _LIST_ITEM_RE.match(current[0].text) is not None
671
+ and previous[0].text.rstrip().endswith((":", ":"))
672
+ and previous[1][2] - previous[1][0] <= 0.8 * lane_width
673
+ and indented_item_continuation
674
+ ):
675
+ output.add(current[0].source_index)
676
+ return output
677
+
678
+
679
+ __all__ = [
680
+ "_local_tight_output_line_bboxes",
681
+ "_starts_structural_reference_entry",
682
+ "_build_hanging_indent_group_map",
683
+ "_infer_local_text_lane_map",
684
+ "_structured_text_break_sources",
685
+ "_isolated_indented_paragraph_break_sources",
686
+ "_centered_visual_reset_break_sources",
687
+ "_leading_typography_reset_break_sources",
688
+ "_formula_style_text_row_break_sources",
689
+ "_front_matter_keyword_break_sources",
690
+ "_component_starts_with_emphasized_row",
691
+ "_explicit_text_break_sources",
692
+ ]