docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1081 @@
1
+ """识别编号、排版重置及跨页一致的结构标题。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import statistics
7
+ import unicodedata
8
+ from collections import Counter
9
+ from dataclasses import replace
10
+
11
+ from .....schema import BBox
12
+ from ..geometry import _bbox_axis_overlap_ratio, _bbox_center_x, _bbox_center_y, _bbox_union_many, _rotate_bbox_to_upright
13
+ from ..line_layout import (
14
+ _effective_text_row_gap,
15
+ _estimate_lane_gap,
16
+ _font_signatures_share_family,
17
+ _infer_text_lanes,
18
+ _line_canonical_style_scale,
19
+ _line_effective_height,
20
+ _normalized_font_family,
21
+ _title_fonts_compatible,
22
+ )
23
+ from ..models import _DocumentBodyProfile, _DocumentTitleProfile, _LineItem, _PreparedPage, _TextLane
24
+ from .body_profile import _line_uses_document_regular_font
25
+ from .common import (
26
+ _NUMBERED_SECTION_TITLE_RE,
27
+ _SECTION_NUMBER_ONLY_RE,
28
+ _SECTION_TITLE_TERMINAL_RE,
29
+ _UNNUMBERED_SECTION_HEADING_RE,
30
+ _build_physical_title_gap_map,
31
+ _line_inside_visual_container,
32
+ )
33
+ from .page_titles import _classify_page_titles
34
+
35
+
36
+ def _normalized_section_title_text(text: str) -> str:
37
+ """规范全半角编号和空白,仅用于结构标题规则判断。"""
38
+
39
+ return re.sub(
40
+ r"\s+",
41
+ " ",
42
+ unicodedata.normalize("NFKC", text),
43
+ ).strip()
44
+
45
+
46
+ def _is_plausible_section_number(
47
+ number: str,
48
+ label: str = "",
49
+ ) -> bool:
50
+ """排除年代、小数值和整句正文冒充的章节编号。"""
51
+
52
+ parts = [int(part) for part in re.findall(r"\d+", number)]
53
+ if not parts or any(part > 99 for part in parts):
54
+ return False
55
+ if len(parts) > 1 and parts[0] == 0:
56
+ return False
57
+ if len(parts) == 1 and parts[0] > 12:
58
+ return False
59
+ stripped_label = label.lstrip()
60
+ if stripped_label and not stripped_label[0].isalpha():
61
+ return False
62
+ if stripped_label and stripped_label[0].isascii() and stripped_label[0].isalpha() and not stripped_label[0].isupper():
63
+ return False
64
+ if len(re.findall(r"\d+(?:\.\d+)?", label)) >= 2:
65
+ return False
66
+ return not any(char in label for char in ",,;;::。!?!?")
67
+
68
+
69
+ def _section_title_has_body_followers(
70
+ title_bbox: BBox,
71
+ geometry: list[tuple[_LineItem, BBox]],
72
+ body_height: float,
73
+ local_page_width: float,
74
+ *,
75
+ minimum_count: int,
76
+ ) -> bool:
77
+ """检查紧随标题的同栏常规正文行,避免把页码和数值标成标题。"""
78
+
79
+ followers = 0
80
+ for line, bbox in sorted(
81
+ geometry,
82
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
83
+ ):
84
+ if bbox[1] <= title_bbox[1] + 0.4 * body_height:
85
+ continue
86
+ if bbox[1] - title_bbox[3] > 8.0 * body_height:
87
+ break
88
+ line_height = _line_effective_height(line, bbox)
89
+ horizontally_related = (
90
+ _bbox_axis_overlap_ratio(title_bbox, bbox, axis="x") >= 0.15 or abs(bbox[0] - title_bbox[0]) <= 2.5 * body_height
91
+ )
92
+ if (
93
+ line.semantic_type is None
94
+ and horizontally_related
95
+ and bbox[2] - bbox[0] >= 0.25 * local_page_width
96
+ and 0.7 <= line_height / body_height <= 1.4
97
+ ):
98
+ followers += 1
99
+ if followers >= minimum_count:
100
+ return True
101
+ return False
102
+
103
+
104
+ def _classify_explicit_section_titles(
105
+ lines: list[_LineItem],
106
+ page_size: tuple[float, float],
107
+ *,
108
+ container_bboxes: list[BBox],
109
+ document_body_profile: _DocumentBodyProfile | None,
110
+ ) -> None:
111
+ """以通用编号行和紧凑结构转折补齐正文同字号章节标题。"""
112
+
113
+ if document_body_profile is None or document_body_profile.body_height <= 0:
114
+ return
115
+ body_height = document_body_profile.body_height
116
+ for angle in sorted(
117
+ {
118
+ line.angle
119
+ for line in lines
120
+ if (line.semantic_type is None or line.explicit_section_title) and not line.title_suppressed
121
+ }
122
+ ):
123
+ geometry = sorted(
124
+ [
125
+ (
126
+ line,
127
+ _rotate_bbox_to_upright(
128
+ line.bbox,
129
+ page_size,
130
+ angle,
131
+ ),
132
+ )
133
+ for line in lines
134
+ if line.angle == angle and line.semantic_type is None and not line.title_suppressed
135
+ ],
136
+ key=lambda item: (
137
+ item[1][1],
138
+ item[1][0],
139
+ item[0].source_index,
140
+ ),
141
+ )
142
+ if not geometry:
143
+ continue
144
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
145
+ local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
146
+ local_containers = [
147
+ _rotate_bbox_to_upright(
148
+ bbox,
149
+ page_size,
150
+ angle,
151
+ )
152
+ for bbox in container_bboxes
153
+ ]
154
+ numbered_groups: list[list[tuple[_LineItem, BBox]]] = []
155
+ grouped_sources: set[int] = set()
156
+ for line, bbox in geometry:
157
+ normalized = _normalized_section_title_text(line.text)
158
+ numbered_match = _NUMBERED_SECTION_TITLE_RE.match(
159
+ normalized,
160
+ )
161
+ if numbered_match is not None and _is_plausible_section_number(
162
+ numbered_match.group("number"),
163
+ numbered_match.group("label"),
164
+ ):
165
+ numbered_groups.append([(line, bbox)])
166
+ grouped_sources.add(line.source_index)
167
+ continue
168
+ if _SECTION_NUMBER_ONLY_RE.match(normalized) is None or not _is_plausible_section_number(normalized):
169
+ continue
170
+ marker_height = _line_effective_height(line, bbox)
171
+ companions = [
172
+ (candidate, candidate_bbox)
173
+ for candidate, candidate_bbox in geometry
174
+ if candidate is not line
175
+ and candidate.source_index not in grouped_sources
176
+ and candidate_bbox[0] >= bbox[2]
177
+ and candidate_bbox[0] - bbox[2] <= 4.0 * max(body_height, marker_height)
178
+ and _bbox_axis_overlap_ratio(
179
+ bbox,
180
+ candidate_bbox,
181
+ axis="y",
182
+ )
183
+ >= 0.5
184
+ and candidate_bbox[2] - candidate_bbox[0] <= 0.55 * local_page_width
185
+ and _SECTION_TITLE_TERMINAL_RE.search(
186
+ _normalized_section_title_text(candidate.text),
187
+ )
188
+ is None
189
+ ]
190
+ if not companions:
191
+ continue
192
+ companion = min(
193
+ companions,
194
+ key=lambda item: (
195
+ item[1][0] - bbox[2],
196
+ item[1][1],
197
+ ),
198
+ )
199
+ numbered_groups.append([(line, bbox), companion])
200
+ grouped_sources.update({line.source_index, companion[0].source_index})
201
+
202
+ for group in numbered_groups:
203
+ title_bbox = _bbox_union_many(
204
+ [bbox for _line, bbox in group],
205
+ )
206
+ group_line_ids = {id(line) for line, _bbox in group}
207
+ preceding = [
208
+ previous_bbox
209
+ for previous_line, previous_bbox in geometry
210
+ if id(previous_line) not in group_line_ids
211
+ and previous_bbox[3] <= title_bbox[1]
212
+ and (
213
+ _bbox_axis_overlap_ratio(
214
+ previous_bbox,
215
+ title_bbox,
216
+ axis="x",
217
+ )
218
+ >= 0.15
219
+ or abs(previous_bbox[0] - title_bbox[0]) <= 2.5 * body_height
220
+ )
221
+ ]
222
+ gap_above = title_bbox[1] - max(previous_bbox[3] for previous_bbox in preceding) if preceding else body_height
223
+ if (
224
+ title_bbox[2] - title_bbox[0] > 0.7 * local_page_width
225
+ or not 0.1 * local_page_height <= _bbox_center_y(title_bbox) <= 0.93 * local_page_height
226
+ or any(
227
+ _bbox_axis_overlap_ratio(
228
+ title_bbox,
229
+ container_bbox,
230
+ axis="x",
231
+ )
232
+ >= 0.8
233
+ and _bbox_axis_overlap_ratio(
234
+ title_bbox,
235
+ container_bbox,
236
+ axis="y",
237
+ )
238
+ >= 0.8
239
+ for container_bbox in local_containers
240
+ )
241
+ or not _section_title_has_body_followers(
242
+ title_bbox,
243
+ geometry,
244
+ body_height,
245
+ local_page_width,
246
+ minimum_count=1,
247
+ )
248
+ or gap_above < 0.4 * body_height
249
+ ):
250
+ continue
251
+ for line, _bbox in group:
252
+ line.semantic_type = "paragraph_title"
253
+ line.structural_title = True
254
+ line.explicit_section_title = True
255
+
256
+ for line, bbox in geometry:
257
+ if line.semantic_type is not None:
258
+ continue
259
+ normalized = _normalized_section_title_text(line.text)
260
+ canonical_heading = normalized.strip(
261
+ "[][]【】()()",
262
+ ).replace(" ", "")
263
+ if (
264
+ _UNNUMBERED_SECTION_HEADING_RE.fullmatch(
265
+ canonical_heading,
266
+ )
267
+ is None
268
+ or not 2 <= len(normalized) <= 24
269
+ or _SECTION_TITLE_TERMINAL_RE.search(normalized) is not None
270
+ or any(char in normalized for char in "[][]")
271
+ or bbox[2] - bbox[0] > 0.3 * local_page_width
272
+ or _bbox_center_y(bbox) < 0.35 * local_page_height
273
+ or not 0.75 <= _line_effective_height(line, bbox) / body_height <= 1.4
274
+ or any(
275
+ _bbox_axis_overlap_ratio(
276
+ bbox,
277
+ container_bbox,
278
+ axis="x",
279
+ )
280
+ >= 0.8
281
+ and _bbox_axis_overlap_ratio(
282
+ bbox,
283
+ container_bbox,
284
+ axis="y",
285
+ )
286
+ >= 0.8
287
+ for container_bbox in local_containers
288
+ )
289
+ or not _section_title_has_body_followers(
290
+ bbox,
291
+ geometry,
292
+ body_height,
293
+ local_page_width,
294
+ minimum_count=2,
295
+ )
296
+ ):
297
+ continue
298
+ preceding = [
299
+ previous_bbox
300
+ for previous_line, previous_bbox in geometry
301
+ if previous_line is not line
302
+ and previous_bbox[3] <= bbox[1]
303
+ and (
304
+ _bbox_axis_overlap_ratio(
305
+ previous_bbox,
306
+ bbox,
307
+ axis="x",
308
+ )
309
+ >= 0.15
310
+ or abs(previous_bbox[0] - bbox[0]) <= 2.5 * body_height
311
+ )
312
+ ]
313
+ gap_above = bbox[1] - max(previous_bbox[3] for previous_bbox in preceding) if preceding else body_height
314
+ if gap_above >= 0.5 * body_height:
315
+ line.semantic_type = "paragraph_title"
316
+ line.structural_title = True
317
+ line.explicit_section_title = True
318
+
319
+
320
+ def _classify_document_structural_titles(
321
+ prepared_pages: list[_PreparedPage],
322
+ document_body_profile: _DocumentBodyProfile | None,
323
+ *,
324
+ legacy_body_profile: _DocumentBodyProfile | None,
325
+ document_title_profile: _DocumentTitleProfile | None,
326
+ ) -> None:
327
+ """用跨页稳定栏带和段前后转折补齐正文同字号标题。"""
328
+
329
+ if document_body_profile is None or not document_body_profile.has_style_scale_repairs:
330
+ return
331
+ probe_pages = [
332
+ replace(
333
+ prepared,
334
+ remaining_lines=[
335
+ replace(
336
+ line,
337
+ style_scale_repaired=True,
338
+ structural_title=False,
339
+ )
340
+ for line in prepared.remaining_lines
341
+ ],
342
+ )
343
+ for prepared in prepared_pages
344
+ ]
345
+ _classify_document_structural_title_candidates(
346
+ probe_pages,
347
+ document_body_profile,
348
+ )
349
+ canonical_candidate_sources = {
350
+ (page_index, line.source_index)
351
+ for page_index, prepared in enumerate(probe_pages)
352
+ for line in prepared.remaining_lines
353
+ if line.structural_title
354
+ }
355
+ legacy_title_sources = _collect_legacy_paragraph_title_sources(
356
+ prepared_pages,
357
+ legacy_body_profile,
358
+ document_title_profile,
359
+ )
360
+ body_height = max(0.1, document_body_profile.body_height)
361
+ canonical_style_candidate_pages: dict[
362
+ tuple[str, int, float, int],
363
+ list[int],
364
+ ] = {}
365
+ for page_index, prepared in enumerate(prepared_pages):
366
+ for line in prepared.remaining_lines:
367
+ line_key = (page_index, line.source_index)
368
+ if line_key not in canonical_candidate_sources:
369
+ continue
370
+ local_bbox = _rotate_bbox_to_upright(
371
+ line.source_bbox or line.bbox,
372
+ prepared.page_size,
373
+ line.angle,
374
+ )
375
+ layout_ratio = (local_bbox[3] - local_bbox[1]) / body_height
376
+ if (
377
+ line_key in legacy_title_sources
378
+ or layout_ratio >= 1.8
379
+ or not _line_uses_document_regular_font(
380
+ line,
381
+ document_body_profile,
382
+ )
383
+ ):
384
+ line.semantic_type = "paragraph_title"
385
+ line.structural_title = True
386
+ if (
387
+ line_key not in legacy_title_sources
388
+ and layout_ratio >= 1.8
389
+ and (
390
+ style_key := _canonical_title_style_key(
391
+ line,
392
+ )
393
+ )
394
+ is not None
395
+ ):
396
+ canonical_style_candidate_pages.setdefault(
397
+ style_key,
398
+ [],
399
+ ).append(page_index)
400
+ canonical_style_prototypes = {
401
+ style_key
402
+ for style_key, page_indices in canonical_style_candidate_pages.items()
403
+ if len(set(page_indices)) >= 2 or max(Counter(page_indices).values(), default=0) >= 3
404
+ }
405
+ if canonical_style_prototypes:
406
+ for prepared in prepared_pages:
407
+ prepared.canonical_formula_geometry = True
408
+ for line in prepared.remaining_lines:
409
+ style_key = _canonical_title_style_key(line)
410
+ if style_key in canonical_style_prototypes:
411
+ line.style_scale_repaired = True
412
+
413
+
414
+ def _promote_noninitial_document_title_band(
415
+ lines: list[_LineItem],
416
+ page_size: tuple[float, float],
417
+ *,
418
+ page_index: int,
419
+ container_bboxes: list[BBox],
420
+ document_body_profile: _DocumentBodyProfile | None,
421
+ title_candidate_source_indices: set[int],
422
+ ) -> None:
423
+ """把非首页中已确认且显著大于正文的最强段落标题带升为文档标题。"""
424
+
425
+ if page_index == 0 or document_body_profile is None:
426
+ return
427
+ body_height = max(0.1, document_body_profile.body_height)
428
+ page_candidates: list[
429
+ tuple[
430
+ tuple[float, float, int, float],
431
+ list[tuple[_LineItem, BBox]],
432
+ ]
433
+ ] = []
434
+ for angle in sorted(
435
+ {
436
+ line.angle
437
+ for line in lines
438
+ if line.source_index in title_candidate_source_indices
439
+ and line.semantic_type in {None, "paragraph_title"}
440
+ and not line.title_suppressed
441
+ }
442
+ ):
443
+ geometry = sorted(
444
+ [
445
+ (
446
+ line,
447
+ _rotate_bbox_to_upright(
448
+ line.bbox,
449
+ page_size,
450
+ angle,
451
+ ),
452
+ )
453
+ for line in lines
454
+ if line.angle == angle
455
+ and line.source_index in title_candidate_source_indices
456
+ and line.semantic_type in {None, "paragraph_title"}
457
+ and not line.title_suppressed
458
+ ],
459
+ key=lambda item: (
460
+ item[1][1],
461
+ item[1][0],
462
+ item[0].source_index,
463
+ ),
464
+ )
465
+ if not geometry:
466
+ continue
467
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
468
+ local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
469
+ local_containers = [
470
+ _rotate_bbox_to_upright(
471
+ bbox,
472
+ page_size,
473
+ angle,
474
+ )
475
+ for bbox in container_bboxes
476
+ ]
477
+ for index, (line, bbox) in enumerate(geometry):
478
+ line_scale = _line_effective_height(line, bbox)
479
+ width_ratio = (bbox[2] - bbox[0]) / max(
480
+ 0.1,
481
+ local_page_width,
482
+ )
483
+ centered = abs(_bbox_center_x(bbox) - 0.5 * local_page_width) <= 0.08 * local_page_width
484
+ if (
485
+ line_scale < 1.25 * body_height
486
+ or not 0.45 <= width_ratio <= 0.85
487
+ or not centered
488
+ or not 0.12 * local_page_height <= _bbox_center_y(bbox) <= 0.75 * local_page_height
489
+ or _line_inside_visual_container(
490
+ bbox,
491
+ local_containers,
492
+ )
493
+ ):
494
+ continue
495
+
496
+ title_members = [(line, bbox)]
497
+ cursor = index + 1
498
+ while cursor < len(geometry):
499
+ candidate_line, candidate_bbox = geometry[cursor]
500
+ candidate_scale = _line_effective_height(
501
+ candidate_line,
502
+ candidate_bbox,
503
+ )
504
+ vertical_gap = max(
505
+ 0.0,
506
+ candidate_bbox[1] - title_members[-1][1][3],
507
+ )
508
+ if (
509
+ candidate_scale < 1.2 * body_height
510
+ or vertical_gap
511
+ > 1.5
512
+ * max(
513
+ line_scale,
514
+ candidate_scale,
515
+ )
516
+ or abs(_bbox_center_x(candidate_bbox) - 0.5 * local_page_width) > 0.1 * local_page_width
517
+ or candidate_bbox[2] - candidate_bbox[0] > 0.85 * local_page_width
518
+ or _line_inside_visual_container(
519
+ candidate_bbox,
520
+ local_containers,
521
+ )
522
+ or not _title_fonts_compatible(
523
+ title_members[-1][0],
524
+ candidate_line,
525
+ )
526
+ ):
527
+ break
528
+ title_members.append(
529
+ (candidate_line, candidate_bbox),
530
+ )
531
+ cursor += 1
532
+
533
+ title_scales = [_line_effective_height(title_line, title_bbox) for title_line, title_bbox in title_members]
534
+ title_bbox = _bbox_union_many(
535
+ [member_bbox for _member_line, member_bbox in title_members],
536
+ )
537
+ page_candidates.append(
538
+ (
539
+ (
540
+ statistics.median(title_scales) / body_height,
541
+ sum(member_bbox[2] - member_bbox[0] for _member_line, member_bbox in title_members) / local_page_width,
542
+ len(title_members),
543
+ -_bbox_center_y(title_bbox) / local_page_height,
544
+ ),
545
+ title_members,
546
+ )
547
+ )
548
+
549
+ if not page_candidates:
550
+ return
551
+ _score, title_members = max(
552
+ page_candidates,
553
+ key=lambda item: item[0],
554
+ )
555
+ for title_line, _title_bbox in title_members:
556
+ title_line.semantic_type = "doc_title"
557
+
558
+
559
+ def _canonical_title_style_key(
560
+ line: _LineItem,
561
+ ) -> tuple[str, int, float, int] | None:
562
+ """返回 canonical-only 标题向同样式正文传播时使用的稳定键。"""
563
+
564
+ if line.font_signature is None or line.em_height <= 0:
565
+ return None
566
+ font_family = _normalized_font_family(line.font_signature)
567
+ if font_family is None:
568
+ return None
569
+ return (
570
+ font_family,
571
+ line.font_signature[1],
572
+ round(line.em_height * 4.0) / 4.0,
573
+ line.angle,
574
+ )
575
+
576
+
577
+ def _classify_document_structural_title_candidates(
578
+ prepared_pages: list[_PreparedPage],
579
+ document_body_profile: _DocumentBodyProfile,
580
+ ) -> None:
581
+ """在 canonical 行副本上收集所有满足结构转折的标题候选。"""
582
+
583
+ body_height = max(0.1, document_body_profile.body_height)
584
+ strong_candidates: list[tuple[int, _LineItem, tuple[str, int] | None, float]] = []
585
+ start_candidates: list[tuple[int, _LineItem, tuple[str, int] | None, float]] = []
586
+ accepted_anchor_positions: list[tuple[int, int, float, float]] = []
587
+ for page_index, prepared in enumerate(prepared_pages):
588
+ container_bboxes = [
589
+ block["bbox"] for block in prepared.fixed_blocks if not isinstance(block.get("_inline_visual_row_id"), int)
590
+ ]
591
+ for angle in sorted({line.angle for line in prepared.remaining_lines if line.semantic_type is None}):
592
+ geometry = sorted(
593
+ [
594
+ (
595
+ line,
596
+ _rotate_bbox_to_upright(
597
+ line.source_bbox or line.bbox,
598
+ prepared.page_size,
599
+ angle,
600
+ ),
601
+ )
602
+ for line in prepared.remaining_lines
603
+ if line.angle == angle and line.semantic_type is None
604
+ ],
605
+ key=lambda item: (
606
+ item[1][1],
607
+ item[1][0],
608
+ item[0].source_index,
609
+ ),
610
+ )
611
+ if len(geometry) < 4:
612
+ continue
613
+ local_page_width = prepared.page_size[1] if angle in {90, 270} else prepared.page_size[0]
614
+ local_page_height = prepared.page_size[0] if angle in {90, 270} else prepared.page_size[1]
615
+ median_height = statistics.median(_line_canonical_style_scale(line, bbox) for line, bbox in geometry)
616
+ lanes = _infer_text_lanes(
617
+ geometry,
618
+ local_page_width,
619
+ median_height,
620
+ )
621
+ physical_gaps = _build_physical_title_gap_map(geometry)
622
+ local_containers = [
623
+ _rotate_bbox_to_upright(
624
+ bbox,
625
+ prepared.page_size,
626
+ angle,
627
+ )
628
+ for bbox in container_bboxes
629
+ ]
630
+ for line, bbox in geometry:
631
+ if page_index == 0 and _bbox_center_y(bbox) < 0.64 * local_page_height:
632
+ continue
633
+ related_lanes = [
634
+ lane
635
+ for lane in lanes
636
+ if not lane.is_span
637
+ and len(lane.lines) >= 3
638
+ and lane.left - body_height <= _bbox_center_x(bbox) <= lane.right + body_height
639
+ ]
640
+ if not related_lanes:
641
+ continue
642
+ lane = max(
643
+ related_lanes,
644
+ key=lambda item: (
645
+ len(item.lines),
646
+ item.right - item.left,
647
+ ),
648
+ )
649
+ lane_width = max(0.1, lane.right - lane.left)
650
+ width_ratio = (bbox[2] - bbox[0]) / lane_width
651
+ style_ratio = _line_canonical_style_scale(line, bbox) / body_height
652
+ left_offset = (bbox[0] - lane.left) / body_height
653
+ regular_font = _line_uses_document_regular_font(
654
+ line,
655
+ document_body_profile,
656
+ )
657
+ if (
658
+ not 0.75 <= style_ratio <= 1.35
659
+ or width_ratio > 0.8
660
+ or left_offset > 0.75
661
+ or left_offset < (-0.75 if regular_font else -3.0)
662
+ or _line_inside_visual_container(
663
+ bbox,
664
+ local_containers,
665
+ )
666
+ ):
667
+ continue
668
+ followers = [
669
+ (other_line, other_bbox)
670
+ for other_line, other_bbox in geometry
671
+ if other_line is not line
672
+ and bbox[1] < other_bbox[1]
673
+ and other_bbox[1] - bbox[3] <= 3.0 * body_height
674
+ and lane.left - body_height <= other_bbox[0] <= lane.left + 3.5 * body_height
675
+ and other_bbox[2] - other_bbox[0] >= 0.4 * lane_width
676
+ ]
677
+ if not followers:
678
+ continue
679
+ gap_above, gap_below = physical_gaps.get(
680
+ line.source_index,
681
+ (None, None),
682
+ )
683
+ layout_ratio = (bbox[3] - bbox[1]) / body_height
684
+ if regular_font and layout_ratio < 1.3 and line.font_coverage < 0.75:
685
+ continue
686
+ standard_transition = (
687
+ gap_above is not None
688
+ and gap_below is not None
689
+ and gap_above >= 0.65 * body_height
690
+ and gap_below >= 0.35 * body_height
691
+ and (not regular_font or gap_below <= 1.5 * body_height)
692
+ )
693
+ low_coverage_wide_transition = (
694
+ gap_above is not None
695
+ and gap_below is not None
696
+ and gap_above >= 0.5 * body_height
697
+ and gap_below >= 0.45 * body_height
698
+ and gap_below <= 0.7 * body_height
699
+ and width_ratio >= 0.75
700
+ and line.font_coverage <= 0.7
701
+ and layout_ratio >= 1.8
702
+ )
703
+ if low_coverage_wide_transition and any(
704
+ anchor_page_index == page_index
705
+ and anchor_angle == angle
706
+ and abs(anchor_left - lane.left) <= body_height
707
+ and 0 < _bbox_center_y(bbox) - anchor_center_y <= 12.0 * body_height
708
+ for (
709
+ anchor_page_index,
710
+ anchor_angle,
711
+ anchor_left,
712
+ anchor_center_y,
713
+ ) in accepted_anchor_positions
714
+ ):
715
+ low_coverage_wide_transition = False
716
+ compact_regular_transition = (
717
+ gap_above is not None
718
+ and gap_below is not None
719
+ and gap_above >= 0.65 * body_height
720
+ and gap_below >= 0.15 * body_height
721
+ and width_ratio <= 0.45
722
+ and line.font_coverage >= 0.75
723
+ and layout_ratio <= 1.3
724
+ )
725
+ family_key = (
726
+ (
727
+ _normalized_font_family(line.font_signature),
728
+ line.font_signature[1],
729
+ )
730
+ if line.font_signature is not None
731
+ else None
732
+ )
733
+ if (
734
+ standard_transition
735
+ or low_coverage_wide_transition
736
+ or compact_regular_transition
737
+ or (gap_above is None and gap_below is not None and gap_below >= 0.6 * body_height and not regular_font)
738
+ ):
739
+ strong_candidates.append(
740
+ (
741
+ page_index,
742
+ line,
743
+ family_key,
744
+ layout_ratio,
745
+ ),
746
+ )
747
+ accepted_anchor_positions.append(
748
+ (
749
+ page_index,
750
+ angle,
751
+ lane.left,
752
+ _bbox_center_y(bbox),
753
+ )
754
+ )
755
+ continue
756
+ if (
757
+ gap_above is None
758
+ and bbox[1] <= 0.18 * local_page_height
759
+ and width_ratio <= 0.6
760
+ and (line.font_coverage >= 0.75 or (not regular_font and line.font_coverage >= 0.5))
761
+ ):
762
+ start_candidates.append(
763
+ (
764
+ page_index,
765
+ line,
766
+ family_key,
767
+ layout_ratio,
768
+ ),
769
+ )
770
+
771
+ for _page_index, line, _family_key, _layout_ratio in strong_candidates:
772
+ line.semantic_type = "paragraph_title"
773
+ line.structural_title = True
774
+ strong_families_by_page = {
775
+ (page_index, family_key) for page_index, _line, family_key, _layout_ratio in strong_candidates if family_key is not None
776
+ }
777
+ for page_index, line, family_key, _layout_ratio in start_candidates:
778
+ if not _line_uses_document_regular_font(
779
+ line,
780
+ document_body_profile,
781
+ ) or (family_key is not None and (page_index, family_key) in strong_families_by_page):
782
+ line.semantic_type = "paragraph_title"
783
+ line.structural_title = True
784
+
785
+
786
+ def _collect_legacy_paragraph_title_sources(
787
+ prepared_pages: list[_PreparedPage],
788
+ document_body_profile: _DocumentBodyProfile | None,
789
+ document_title_profile: _DocumentTitleProfile | None,
790
+ ) -> set[tuple[int, int]]:
791
+ """在行副本上使用 legacy 尺度收集原本成立的段落标题身份。"""
792
+
793
+ if document_body_profile is None:
794
+ return set()
795
+ legacy_profile = replace(
796
+ document_body_profile,
797
+ has_style_scale_repairs=False,
798
+ )
799
+ output: set[tuple[int, int]] = set()
800
+ for page_index, prepared in enumerate(prepared_pages):
801
+ probe_lines = [
802
+ replace(
803
+ line,
804
+ style_scale_repaired=False,
805
+ structural_title=False,
806
+ )
807
+ for line in prepared.remaining_lines
808
+ ]
809
+ container_bboxes = [
810
+ block["bbox"] for block in prepared.fixed_blocks if not isinstance(block.get("_inline_visual_row_id"), int)
811
+ ]
812
+ caption_container_bboxes = [block["bbox"] for block in prepared.fixed_blocks if block.get("type") in {"image", "code"}]
813
+ _classify_page_titles(
814
+ probe_lines,
815
+ prepared.page_size,
816
+ page_index=page_index,
817
+ container_bboxes=container_bboxes,
818
+ caption_container_bboxes=caption_container_bboxes,
819
+ document_body_profile=legacy_profile,
820
+ document_title_profile=document_title_profile,
821
+ )
822
+ output.update((page_index, line.source_index) for line in probe_lines if line.semantic_type == "paragraph_title")
823
+ return output
824
+
825
+
826
+ def _classify_inline_typography_reset_titles(
827
+ lines: list[_LineItem],
828
+ page_size: tuple[float, float],
829
+ *,
830
+ container_bboxes: list[BBox],
831
+ document_body_profile: _DocumentBodyProfile | None,
832
+ ) -> None:
833
+ """用短段尾、字体切换和缩进正文识别行内结构标题。"""
834
+
835
+ if document_body_profile is None:
836
+ return
837
+ for angle in sorted({line.angle for line in lines if line.semantic_type is None and not line.title_suppressed}):
838
+ line_geometry = [
839
+ (line, _rotate_bbox_to_upright(line.bbox, page_size, angle))
840
+ for line in lines
841
+ if line.angle == angle and line.semantic_type is None and not line.title_suppressed
842
+ ]
843
+ if len(line_geometry) < 3:
844
+ continue
845
+ median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in line_geometry)
846
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
847
+ lanes = _infer_text_lanes(
848
+ line_geometry,
849
+ local_page_width,
850
+ median_height,
851
+ )
852
+ local_container_bboxes = [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in container_bboxes]
853
+ for lane in lanes:
854
+ rows = sorted(
855
+ lane.lines,
856
+ key=lambda item: (
857
+ item[1][1],
858
+ item[1][0],
859
+ item[0].source_index,
860
+ ),
861
+ )
862
+ lane_width = max(0.1, lane.right - lane.left)
863
+ for previous, current, following in zip(
864
+ rows,
865
+ rows[1:],
866
+ rows[2:],
867
+ ):
868
+ previous_line, previous_bbox = previous
869
+ current_line, current_bbox = current
870
+ following_line, following_bbox = following
871
+ if any(
872
+ line.semantic_type is not None
873
+ for line in (
874
+ previous_line,
875
+ current_line,
876
+ following_line,
877
+ )
878
+ ):
879
+ continue
880
+ if (
881
+ previous_line.font_signature is None
882
+ or current_line.font_signature is None
883
+ or following_line.font_signature is None
884
+ or previous_line.font_coverage < 0.65
885
+ or current_line.font_coverage < 0.65
886
+ or following_line.font_coverage < 0.65
887
+ ):
888
+ continue
889
+ if not _font_signatures_share_family(
890
+ previous_line.font_signature,
891
+ following_line.font_signature,
892
+ ) or _font_signatures_share_family(
893
+ current_line.font_signature,
894
+ previous_line.font_signature,
895
+ ):
896
+ continue
897
+ previous_height = _line_effective_height(*previous)
898
+ current_height = _line_effective_height(*current)
899
+ following_height = _line_effective_height(*following)
900
+ neighbor_height = statistics.median(
901
+ (previous_height, following_height),
902
+ )
903
+ pair_height = max(
904
+ previous_height,
905
+ current_height,
906
+ following_height,
907
+ )
908
+ previous_width = previous_bbox[2] - previous_bbox[0]
909
+ current_width = current_bbox[2] - current_bbox[0]
910
+ following_width = following_bbox[2] - following_bbox[0]
911
+ following_indent = following_bbox[0] - current_bbox[0]
912
+ if not (
913
+ previous_width <= 0.45 * lane_width
914
+ and 0.2 * lane_width <= current_width <= 0.65 * lane_width
915
+ and following_width >= 0.75 * lane_width
916
+ and abs(current_bbox[0] - lane.left) <= 0.75 * pair_height
917
+ and 0.75 * pair_height <= following_indent <= 3.0 * pair_height
918
+ and 0.85 <= current_height / max(0.1, neighbor_height) <= 1.15
919
+ and -0.25 * pair_height <= _effective_text_row_gap(previous, current) <= 0.5 * pair_height
920
+ and -0.25 * pair_height <= _effective_text_row_gap(current, following) <= 0.5 * pair_height
921
+ and not _line_inside_visual_container(
922
+ current_bbox,
923
+ local_container_bboxes,
924
+ )
925
+ ):
926
+ continue
927
+ current_line.semantic_type = "paragraph_title"
928
+ current_line.structural_title = True
929
+
930
+
931
+ def _classify_body_height_section_titles(
932
+ lines: list[_LineItem],
933
+ page_size: tuple[float, float],
934
+ *,
935
+ container_bboxes: list[BBox],
936
+ document_body_profile: _DocumentBodyProfile | None,
937
+ page_index: int = 1,
938
+ ) -> None:
939
+ """用重复的短行加正文组结构识别与正文同字号的独立章节标题。"""
940
+
941
+ if document_body_profile is None:
942
+ return
943
+ body_height = document_body_profile.body_height
944
+ if body_height <= 0:
945
+ return
946
+
947
+ for angle in sorted({line.angle for line in lines if line.semantic_type is None and not line.title_suppressed}):
948
+ line_geometry = sorted(
949
+ [
950
+ (line, _rotate_bbox_to_upright(line.bbox, page_size, angle))
951
+ for line in lines
952
+ if line.angle == angle and line.semantic_type is None and not line.title_suppressed
953
+ ],
954
+ key=lambda item: (item[1][1], item[1][0], item[0].source_index),
955
+ )
956
+ if len(line_geometry) < 8:
957
+ continue
958
+ median_height = statistics.median(_line_effective_height(line, bbox) for line, bbox in line_geometry)
959
+ local_page_width = page_size[1] if angle in {90, 270} else page_size[0]
960
+ local_page_height = page_size[0] if angle in {90, 270} else page_size[1]
961
+ lanes = _infer_text_lanes(
962
+ line_geometry,
963
+ local_page_width,
964
+ median_height,
965
+ )
966
+ lane_by_source: dict[int, _TextLane] = {}
967
+ regular_gaps: list[float] = []
968
+ for lane in lanes:
969
+ lane.lines.sort(key=lambda item: (item[1][1], item[1][0], item[0].source_index))
970
+ regular_gap, _gap_mad = _estimate_lane_gap(lane)
971
+ regular_gaps.append(regular_gap)
972
+ for line, _bbox in lane.lines:
973
+ lane_by_source[line.source_index] = lane
974
+ if not lane_by_source:
975
+ continue
976
+
977
+ page_regular_gap = statistics.median(regular_gaps) if regular_gaps else 0.2 * body_height
978
+ physical_gaps = _build_physical_title_gap_map(line_geometry)
979
+ local_container_bboxes = [_rotate_bbox_to_upright(bbox, page_size, angle) for bbox in container_bboxes]
980
+ candidates: list[tuple[_LineItem, BBox, _TextLane]] = []
981
+ for line, bbox in line_geometry:
982
+ lane = lane_by_source.get(line.source_index)
983
+ if lane is None:
984
+ continue
985
+ line_height = _line_effective_height(line, bbox)
986
+ lane_width = max(0.1, lane.right - lane.left)
987
+ if not 0.9 <= line_height / body_height <= 1.1:
988
+ continue
989
+ if bbox[2] - bbox[0] > 0.22 * lane_width:
990
+ continue
991
+ if abs(bbox[0] - lane.left) > 0.75 * body_height:
992
+ continue
993
+ if _line_inside_visual_container(bbox, local_container_bboxes):
994
+ continue
995
+
996
+ followers = _body_height_section_followers(
997
+ line,
998
+ bbox,
999
+ line_geometry,
1000
+ lane_by_source,
1001
+ body_height,
1002
+ )
1003
+ if len(followers) < 3:
1004
+ continue
1005
+ gap_above = physical_gaps.get(line.source_index, (None, None))[0]
1006
+ if gap_above is not None and gap_above <= 0.25 * body_height:
1007
+ continue
1008
+ starts_body = gap_above is None and bbox[1] <= 0.2 * local_page_height
1009
+ has_extra_gap = gap_above is not None and gap_above - page_regular_gap >= 0.75 * body_height
1010
+ has_full_width_follower = any(
1011
+ follower_bbox[2] - follower_bbox[0]
1012
+ >= 0.75
1013
+ * max(
1014
+ 0.1,
1015
+ lane_by_source[follower_line.source_index].right - lane_by_source[follower_line.source_index].left,
1016
+ )
1017
+ for follower_line, follower_bbox in followers
1018
+ if follower_line.source_index in lane_by_source
1019
+ )
1020
+ if starts_body or has_extra_gap or has_full_width_follower:
1021
+ candidates.append((line, bbox, lane))
1022
+
1023
+ for line, bbox, _lane in candidates:
1024
+ compatible_count = sum(
1025
+ 1
1026
+ for peer_line, peer_bbox, _peer_lane in candidates
1027
+ if abs(peer_bbox[0] - bbox[0]) <= body_height
1028
+ and 0.9 <= _line_effective_height(peer_line, peer_bbox) / _line_effective_height(line, bbox) <= 1.1
1029
+ and _title_fonts_compatible(line, peer_line)
1030
+ )
1031
+ if compatible_count >= 2:
1032
+ # 只标记结构锚点本身,避免普通正文被标题邻行扩展再次吞入。
1033
+ line.semantic_type = "paragraph_title"
1034
+
1035
+
1036
+ def _body_height_section_followers(
1037
+ candidate_line: _LineItem,
1038
+ candidate_bbox: BBox,
1039
+ line_geometry: list[tuple[_LineItem, BBox]],
1040
+ lane_by_source: dict[int, _TextLane],
1041
+ body_height: float,
1042
+ ) -> list[tuple[_LineItem, BBox]]:
1043
+ """返回短标题后方同锚点、同正文尺度且行距稳定的前三行。"""
1044
+
1045
+ followers: list[tuple[_LineItem, BBox]] = []
1046
+ previous_top = candidate_bbox[1]
1047
+ for line, bbox in line_geometry:
1048
+ if line is candidate_line or bbox[1] <= candidate_bbox[1] + 0.4 * body_height:
1049
+ continue
1050
+ if not (candidate_bbox[0] - 0.75 * body_height <= bbox[0] <= candidate_bbox[0] + 1.5 * body_height):
1051
+ continue
1052
+ top_pitch = bbox[1] - previous_top
1053
+ if top_pitch < 0.5 * body_height:
1054
+ continue
1055
+ if top_pitch > 1.8 * body_height:
1056
+ break
1057
+ if not 0.9 <= _line_effective_height(line, bbox) / body_height <= 1.1:
1058
+ break
1059
+ if line.source_index not in lane_by_source:
1060
+ break
1061
+ followers.append((line, bbox))
1062
+ previous_top = bbox[1]
1063
+ if len(followers) == 3:
1064
+ break
1065
+ return followers
1066
+
1067
+
1068
+ __all__ = [
1069
+ "_normalized_section_title_text",
1070
+ "_is_plausible_section_number",
1071
+ "_section_title_has_body_followers",
1072
+ "_classify_explicit_section_titles",
1073
+ "_classify_document_structural_titles",
1074
+ "_promote_noninitial_document_title_band",
1075
+ "_canonical_title_style_key",
1076
+ "_classify_document_structural_title_candidates",
1077
+ "_collect_legacy_paragraph_title_sources",
1078
+ "_classify_inline_typography_reset_titles",
1079
+ "_classify_body_height_section_titles",
1080
+ "_body_height_section_followers",
1081
+ ]