docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,851 @@
1
+ """把 WordDocument/Table/Data streams 解析为逐 section DOC 语义模型。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from loguru import logger
7
+
8
+ from ..errors import LegacyOfficeMalformedError
9
+ from ..legacy.binary import bounded_slice
10
+
11
+ from .bookmarks import parse_bookmarks
12
+ from .fib import (
13
+ FCLCB_BOOKMARK_END,
14
+ FCLCB_BOOKMARK_NAMES,
15
+ FCLCB_BOOKMARK_START,
16
+ FCLCB_BTE_CHPX,
17
+ FCLCB_BTE_PAPX,
18
+ FCLCB_CLX,
19
+ FCLCB_DGG_INFO,
20
+ FCLCB_ENDNOTE_REF,
21
+ FCLCB_ENDNOTE_TEXT,
22
+ FCLCB_FOOTNOTE_REF,
23
+ FCLCB_FOOTNOTE_TEXT,
24
+ FCLCB_HEADER,
25
+ FCLCB_LIST_OVERRIDES,
26
+ FCLCB_LISTS,
27
+ FCLCB_SECTION,
28
+ FCLCB_SHAPE_MAIN,
29
+ FCLCB_STSHF,
30
+ FileInformationBlock,
31
+ )
32
+ from .fields import apply_field_result, field_keyword, is_chart_embed_field, is_toc_field
33
+ from .formatting import FormattingRuns, parse_formatting_runs
34
+ from .images import ImageStore, floating_pictures, inline_picture
35
+ from .lists import ListTables, parse_list_tables
36
+ from .models import (
37
+ DocCharStyle,
38
+ DocChartPayload,
39
+ DocDocument,
40
+ DocElement,
41
+ DocImage,
42
+ DocParagraph,
43
+ DocSection,
44
+ DocTable,
45
+ DocTableCell,
46
+ DocTableCellFormat,
47
+ DocTableFormat,
48
+ DocTableRow,
49
+ DocTextRun,
50
+ DocVisualPayload,
51
+ )
52
+ from .pieces import Piece, TextStream, codec_for_lid, extract_text, legacy_single_piece, parse_clx
53
+ from .records import DocBudget, parse_plc
54
+ from .sprm import PapDelta, apply_character_sprms, apply_paragraph_sprms, chpx_picture_location, chpx_style_id
55
+ from .styles import Stylesheet, parse_stylesheet
56
+
57
+
58
+ @dataclass(slots=True)
59
+ class _FieldFrame:
60
+ """一个尚未闭合的 Word 字段。"""
61
+
62
+ instruction: str = ""
63
+ in_result: bool = False
64
+ transparent: bool = False
65
+ runs: list[DocTextRun] = field(default_factory=list)
66
+ native_formula: bool = False
67
+ native_chart: str | None = None
68
+ native_chart_emitted: bool = False
69
+
70
+
71
+ @dataclass(slots=True)
72
+ class _RawTableRow:
73
+ """应用全表网格前的一行单元格。"""
74
+
75
+ cells: list[DocTableCell]
76
+ table_format: DocTableFormat | None
77
+
78
+
79
+ class _Assembler:
80
+ """把全局文字流、格式 run 和辅助 PLC 组装为段落及表格。"""
81
+
82
+ def __init__(
83
+ self,
84
+ *,
85
+ text: TextStream,
86
+ pieces: list[Piece],
87
+ formatting: FormattingRuns,
88
+ stylesheet: Stylesheet,
89
+ lists: ListTables,
90
+ bookmarks: dict[int, list[str]],
91
+ data_stream: bytes,
92
+ image_store: ImageStore,
93
+ budget: DocBudget,
94
+ native_equations: dict[int, str] | None = None,
95
+ native_charts: dict[int, str] | None = None,
96
+ ) -> None:
97
+ """保存解析上下文;每个 story 开始时会重置字段栈。"""
98
+
99
+ self.text = text
100
+ self.pieces = pieces
101
+ self.formatting = formatting
102
+ self.stylesheet = stylesheet
103
+ self.lists = lists
104
+ self.bookmarks = bookmarks
105
+ self.data_stream = data_stream
106
+ self.image_store = image_store
107
+ self.budget = budget
108
+ self.native_equations = native_equations or {}
109
+ self.native_charts = native_charts or {}
110
+ self._fields: list[_FieldFrame] = []
111
+
112
+ def _piece_prm(self, char_index: int) -> bytes:
113
+ """返回字符所在 piece 的 Prm grpprl。"""
114
+
115
+ if char_index >= len(self.text.piece_indexes):
116
+ return b""
117
+ piece_index = self.text.piece_indexes[char_index]
118
+ return self.pieces[piece_index].prm if piece_index < len(self.pieces) else b""
119
+
120
+ def _character_style(self, fc: int, char_index: int) -> DocCharStyle:
121
+ """按样式链、CHPX、piece Prm 顺序解析字符样式。"""
122
+
123
+ paragraph_run = self.formatting.paragraph_at(fc)
124
+ paragraph_style_id = paragraph_run.style_id if paragraph_run is not None else 0
125
+ character_run = self.formatting.character_at(fc)
126
+ grpprl = character_run.grpprl if character_run is not None else b""
127
+ style_id = chpx_style_id(grpprl)
128
+ if style_id is None:
129
+ style_id = paragraph_style_id
130
+ base = self.stylesheet.get(style_id).character
131
+ result = apply_character_sprms(grpprl, base, base, budget=self.budget)
132
+ prm = self._piece_prm(char_index)
133
+ if prm:
134
+ result = apply_character_sprms(prm, result, base, budget=self.budget)
135
+ return result
136
+
137
+ def _paragraph_properties(self, fc: int, char_index: int) -> tuple[int, PapDelta]:
138
+ """按样式链、PAPX、piece Prm 顺序解析段落属性。"""
139
+
140
+ paragraph_run = self.formatting.paragraph_at(fc)
141
+ style_id = paragraph_run.style_id if paragraph_run is not None else 0
142
+ style = self.stylesheet.get(style_id)
143
+ delta = style.paragraph
144
+ if paragraph_run is not None:
145
+ delta = delta.merge(paragraph_run.delta)
146
+ prm = self._piece_prm(char_index)
147
+ if prm:
148
+ delta = apply_paragraph_sprms(prm, b"", delta, budget=self.budget)
149
+ return style_id, delta
150
+
151
+ @staticmethod
152
+ def _append_run(target: list[DocTextRun], run: DocTextRun) -> None:
153
+ """合并相邻同样式同链接 run。"""
154
+
155
+ if not run.text:
156
+ return
157
+ if (
158
+ target
159
+ and target[-1].style == run.style
160
+ and target[-1].hyperlink == run.hyperlink
161
+ and target[-1].formula == run.formula
162
+ ):
163
+ previous = target[-1]
164
+ target[-1] = DocTextRun(
165
+ previous.text + run.text,
166
+ previous.style,
167
+ previous.hyperlink,
168
+ previous.formula,
169
+ )
170
+ else:
171
+ target.append(run)
172
+
173
+ def _push_visible_run(self, visible: list[DocTextRun], run: DocTextRun) -> None:
174
+ """把一个结果 run 送入当前字段或段落。"""
175
+
176
+ if not self._fields:
177
+ self._append_run(visible, run)
178
+ return
179
+ frame = self._fields[-1]
180
+ if not frame.in_result:
181
+ frame.instruction += run.text
182
+ elif frame.transparent:
183
+ self._append_run(visible, run)
184
+ else:
185
+ self._append_run(frame.runs, run)
186
+
187
+ def _field_begin(self) -> None:
188
+ """打开一个嵌套字段。"""
189
+
190
+ self._fields.append(_FieldFrame())
191
+
192
+ def _field_separator(self, paragraph_keywords: set[str]) -> None:
193
+ """结束字段指令并识别跨段 TOC。"""
194
+
195
+ if not self._fields:
196
+ return
197
+ frame = self._fields[-1]
198
+ frame.in_result = True
199
+ keyword = field_keyword(frame.instruction)
200
+ if keyword:
201
+ paragraph_keywords.add(keyword)
202
+ frame.transparent = is_toc_field(frame.instruction)
203
+
204
+ def _field_end(
205
+ self,
206
+ visible: list[DocTextRun],
207
+ paragraph_keywords: set[str],
208
+ visuals: list[DocVisualPayload],
209
+ ) -> None:
210
+ """关闭字段,把缓存结果绑定链接后放回父上下文。"""
211
+
212
+ if not self._fields:
213
+ return
214
+ frame = self._fields.pop()
215
+ keyword = field_keyword(frame.instruction)
216
+ if keyword:
217
+ paragraph_keywords.add(keyword)
218
+ if frame.transparent:
219
+ return
220
+ if frame.native_chart is not None:
221
+ if not frame.native_chart_emitted:
222
+ visuals.append(DocChartPayload(content=frame.native_chart))
223
+ return
224
+ for run in apply_field_result(frame.instruction, frame.runs):
225
+ self._push_visible_run(visible, run)
226
+
227
+ def _paragraph_anchor(self, cp_start: int, cp_end: int) -> str | None:
228
+ """返回段落范围内优先级最高的书签起点。"""
229
+
230
+ candidates = [name for cp, names in self.bookmarks.items() if cp_start <= cp < cp_end for name in names]
231
+ if not candidates:
232
+ return None
233
+ return next((name for name in candidates if name.startswith("_Toc")), candidates[0])
234
+
235
+ def _finish_paragraph(
236
+ self,
237
+ *,
238
+ cp_start: int,
239
+ cp_end: int,
240
+ char_index: int,
241
+ fc: int,
242
+ terminator: str,
243
+ runs: list[DocTextRun],
244
+ images: list[DocVisualPayload],
245
+ keywords: set[str],
246
+ ) -> DocParagraph | None:
247
+ """把段落终止字符上的 PAPX 解析为完整语义段落。"""
248
+
249
+ style_id, pap = self._paragraph_properties(fc, char_index)
250
+ style = self.stylesheet.get(style_id)
251
+ cell_mark = terminator == "\x07" or bool(pap.inner_cell)
252
+ row_mark = bool(pap.row_mark) or bool(pap.inner_row)
253
+ in_table = bool(pap.in_table) or cell_mark or row_mark
254
+ visible = any(run.text for run in runs) or bool(images)
255
+ if not visible and not in_table:
256
+ return None
257
+ list_info = None
258
+ if pap.ilfo is not None:
259
+ list_info = self.lists.paragraph_info(pap.ilfo, pap.ilvl or 0)
260
+ outline = pap.outline_level if isinstance(pap.outline_level, int) and pap.outline_level > 0 else None
261
+ toc_level = style.toc_level
262
+ toc_active = toc_level is not None or any(frame.transparent for frame in self._fields)
263
+ return DocParagraph(
264
+ cp_start=cp_start,
265
+ cp_end=cp_end,
266
+ runs=runs,
267
+ images=images,
268
+ style_name=style.name,
269
+ heading_level=style.heading_level or outline,
270
+ is_title=style.is_title,
271
+ is_toc=toc_active,
272
+ toc_level=toc_level,
273
+ is_caption=any(keyword == "SEQ" for keyword in keywords),
274
+ is_code=style.is_code,
275
+ anchor=self._paragraph_anchor(cp_start, cp_end),
276
+ list_info=list_info,
277
+ in_table=in_table,
278
+ table_depth=max(pap.table_depth or (1 if in_table else 0), 0),
279
+ cell_mark=cell_mark,
280
+ row_mark=row_mark,
281
+ table_format=pap.table,
282
+ )
283
+
284
+ def paragraphs(
285
+ self,
286
+ cp_start: int,
287
+ cp_end: int,
288
+ *,
289
+ note_refs: dict[int, str] | None = None,
290
+ ) -> list[DocParagraph]:
291
+ """解析一个 story CP 范围内的全部段落。"""
292
+
293
+ self._fields = []
294
+ refs = note_refs or {}
295
+ start_index = self.text.index_of_cp(cp_start)
296
+ end_index = self.text.index_of_cp(cp_end)
297
+ paragraphs: list[DocParagraph] = []
298
+ visible: list[DocTextRun] = []
299
+ images: list[DocVisualPayload] = []
300
+ keywords: set[str] = set()
301
+ paragraph_start = cp_start
302
+ for char_index in range(start_index, min(end_index, len(self.text.chars))):
303
+ char = self.text.chars[char_index]
304
+ cp = self.text.cps[char_index]
305
+ fc = self.text.fcs[char_index]
306
+ if char == "\x13":
307
+ self._field_begin()
308
+ continue
309
+ if char == "\x14":
310
+ self._field_separator(keywords)
311
+ character_run = self.formatting.character_at(fc)
312
+ location = chpx_picture_location(character_run.grpprl) if character_run is not None else None
313
+ formula = self.native_equations.get(location) if location is not None else None
314
+ chart = self.native_charts.get(location) if location is not None else None
315
+ if chart is not None and self._fields and is_chart_embed_field(self._fields[-1].instruction):
316
+ self._fields[-1].native_chart = chart
317
+ if formula is not None and self._fields:
318
+ self._fields[-1].native_formula = True
319
+ self._push_visible_run(visible, DocTextRun(formula, formula=True))
320
+ continue
321
+ if char == "\x15":
322
+ self._field_end(visible, keywords, images)
323
+ continue
324
+ if cp in refs:
325
+ self._push_visible_run(
326
+ visible,
327
+ DocTextRun(f"[{refs[cp]}]", DocCharStyle(superscript=True)),
328
+ )
329
+ continue
330
+ if char in {"\r", "\x07", "\x0c", "\x0e"}:
331
+ paragraph = self._finish_paragraph(
332
+ cp_start=paragraph_start,
333
+ cp_end=cp + 1,
334
+ char_index=char_index,
335
+ fc=fc,
336
+ terminator=char,
337
+ runs=visible,
338
+ images=images,
339
+ keywords=keywords,
340
+ )
341
+ if paragraph is not None:
342
+ paragraphs.append(paragraph)
343
+ visible = []
344
+ images = []
345
+ keywords = set()
346
+ paragraph_start = cp + 1
347
+ continue
348
+ if char == "\x01":
349
+ if self._fields and self._fields[-1].native_formula:
350
+ continue
351
+ character_run = self.formatting.character_at(fc)
352
+ location = chpx_picture_location(character_run.grpprl) if character_run is not None else None
353
+ chart_frame = next(
354
+ (frame for frame in reversed(self._fields) if frame.native_chart is not None),
355
+ None,
356
+ )
357
+ if location is not None:
358
+ payload = inline_picture(
359
+ self.data_stream,
360
+ offset=location,
361
+ store=self.image_store,
362
+ budget=self.budget,
363
+ )
364
+ if chart_frame is not None:
365
+ images.append(
366
+ DocChartPayload(
367
+ content=chart_frame.native_chart or "",
368
+ preview=payload,
369
+ )
370
+ )
371
+ chart_frame.native_chart_emitted = True
372
+ elif payload is not None:
373
+ if all(existing is not payload for existing in images):
374
+ images.append(payload)
375
+ elif chart_frame is not None:
376
+ images.append(DocChartPayload(content=chart_frame.native_chart or ""))
377
+ chart_frame.native_chart_emitted = True
378
+ continue
379
+ if char == "\x0b":
380
+ char = "\n"
381
+ elif char == "\t":
382
+ char = "\t"
383
+ elif char == "\x1e":
384
+ char = "-"
385
+ elif ord(char) < 0x20:
386
+ continue
387
+ style = self._character_style(fc, char_index)
388
+ if self._fields and not self._fields[-1].in_result:
389
+ self._push_visible_run(visible, DocTextRun(char, style))
390
+ elif not style.hidden and not style.deleted:
391
+ self._push_visible_run(visible, DocTextRun(char, style))
392
+ if visible or images:
393
+ last_index = max(start_index, min(end_index, len(self.text.chars)) - 1)
394
+ if last_index < len(self.text.chars):
395
+ paragraph = self._finish_paragraph(
396
+ cp_start=paragraph_start,
397
+ cp_end=cp_end,
398
+ char_index=last_index,
399
+ fc=self.text.fcs[last_index],
400
+ terminator="",
401
+ runs=visible,
402
+ images=images,
403
+ keywords=keywords,
404
+ )
405
+ if paragraph is not None:
406
+ paragraphs.append(paragraph)
407
+ return paragraphs
408
+
409
+
410
+ def _paragraph_visible(paragraph: DocParagraph) -> bool:
411
+ """判断表格标记段落是否含有实际单元格内容。"""
412
+
413
+ return any(run.text for run in paragraph.runs) or bool(paragraph.images)
414
+
415
+
416
+ def _cell_formats(row: _RawTableRow) -> tuple[list[int], list[DocTableCellFormat]]:
417
+ """为缺失 TAP 的行构造稳定伪边界和默认 cell 属性。"""
418
+
419
+ count = len(row.cells)
420
+ table_format = row.table_format
421
+ if table_format is not None and len(table_format.boundaries) >= count + 1:
422
+ boundaries = list(table_format.boundaries[: count + 1])
423
+ formats = list(table_format.cells[:count])
424
+ while len(formats) < count:
425
+ formats.append(DocTableCellFormat(right=boundaries[len(formats) + 1]))
426
+ return boundaries, formats
427
+ boundaries = list(range(count + 1))
428
+ return boundaries, [DocTableCellFormat(right=index + 1) for index in range(count)]
429
+
430
+
431
+ def _materialize_table_rows(raw_rows: list[_RawTableRow], budget: DocBudget) -> list[DocTableRow]:
432
+ """把各行独立 twip 边界统一为 rowspan/colspan HTML 网格。"""
433
+
434
+ all_boundaries: set[int] = set()
435
+ row_formats: list[tuple[list[int], list[DocTableCellFormat]]] = []
436
+ for row in raw_rows:
437
+ boundaries, formats = _cell_formats(row)
438
+ row_formats.append((boundaries, formats))
439
+ all_boundaries.update(boundaries)
440
+ edges = sorted(all_boundaries)
441
+ edge_index = {value: index for index, value in enumerate(edges)}
442
+ vertical_origins: dict[tuple[int, int], DocTableCell] = {}
443
+ rows: list[DocTableRow] = []
444
+ for raw_row, (boundaries, formats) in zip(raw_rows, row_formats, strict=True):
445
+ cells: list[DocTableCell] = []
446
+ keys: list[tuple[int, int]] = []
447
+ index = 0
448
+ while index < len(raw_row.cells):
449
+ source = raw_row.cells[index]
450
+ cell_format = formats[index]
451
+ left = edge_index.get(boundaries[index], index)
452
+ right = edge_index.get(boundaries[index + 1], left + 1)
453
+ col_span = max(right - left, 1)
454
+ if cell_format.horizontal_continue and cells:
455
+ cells[-1].col_span += col_span
456
+ cells[-1].blocks.extend(source.blocks)
457
+ previous_left, _ = keys[-1]
458
+ keys[-1] = (previous_left, right)
459
+ index += 1
460
+ continue
461
+ cell = DocTableCell(blocks=source.blocks, col_span=col_span)
462
+ key = (left, right)
463
+ if cell_format.vertical_continue and key in vertical_origins:
464
+ vertical_origins[key].row_span += 1
465
+ else:
466
+ cells.append(cell)
467
+ keys.append(key)
468
+ if cell_format.vertical_first:
469
+ vertical_origins[key] = cell
470
+ index += 1
471
+ budget.charge(col_span)
472
+ rows.append(
473
+ DocTableRow(
474
+ cells=cells,
475
+ header=bool(raw_row.table_format and raw_row.table_format.header),
476
+ )
477
+ )
478
+ return rows
479
+
480
+
481
+ def _parse_table(
482
+ paragraphs: list[DocParagraph],
483
+ start: int,
484
+ depth: int,
485
+ budget: DocBudget,
486
+ ) -> tuple[DocTable, int]:
487
+ """递归解析指定 table depth 的表格和嵌套表格。"""
488
+
489
+ index = start
490
+ cp_start = paragraphs[start].cp_start
491
+ cp_end = cp_start
492
+ raw_rows: list[_RawTableRow] = []
493
+ row_cells: list[DocTableCell] = []
494
+ current_blocks: list[DocElement] = []
495
+ while index < len(paragraphs):
496
+ paragraph = paragraphs[index]
497
+ if not paragraph.in_table or paragraph.table_depth < depth:
498
+ break
499
+ if paragraph.table_depth > depth:
500
+ nested, index = _parse_table(paragraphs, index, paragraph.table_depth, budget)
501
+ current_blocks.append(nested)
502
+ cp_end = max(cp_end, nested.cp_end)
503
+ continue
504
+ cp_end = max(cp_end, paragraph.cp_end)
505
+ if paragraph.row_mark:
506
+ if _paragraph_visible(paragraph):
507
+ current_blocks.append(paragraph)
508
+ if current_blocks:
509
+ row_cells.append(DocTableCell(blocks=current_blocks))
510
+ current_blocks = []
511
+ if row_cells:
512
+ raw_rows.append(_RawTableRow(row_cells, paragraph.table_format))
513
+ row_cells = []
514
+ elif paragraph.cell_mark:
515
+ if _paragraph_visible(paragraph):
516
+ current_blocks.append(paragraph)
517
+ row_cells.append(DocTableCell(blocks=current_blocks))
518
+ current_blocks = []
519
+ else:
520
+ current_blocks.append(paragraph)
521
+ index += 1
522
+ if current_blocks:
523
+ row_cells.append(DocTableCell(blocks=current_blocks))
524
+ if row_cells:
525
+ raw_rows.append(_RawTableRow(row_cells, None))
526
+ rows = _materialize_table_rows(raw_rows, budget)
527
+ return DocTable(cp_start=cp_start, cp_end=cp_end, rows=rows), index
528
+
529
+
530
+ def _assemble_main_elements(
531
+ paragraphs: list[DocParagraph],
532
+ floating: list[DocImage],
533
+ budget: DocBudget,
534
+ ) -> list[DocElement]:
535
+ """把连续 table paragraph 收敛成表格,并按 CP 合入 floating 图片。"""
536
+
537
+ elements: list[DocElement] = []
538
+ index = 0
539
+ while index < len(paragraphs):
540
+ paragraph = paragraphs[index]
541
+ if paragraph.in_table and paragraph.table_depth > 0:
542
+ # DOC 可直接以 depth=2 的 inner-cell 开始;从 depth=1 建立隐式外层格。
543
+ table, index = _parse_table(paragraphs, index, 1, budget)
544
+ if table.rows:
545
+ elements.append(table)
546
+ continue
547
+ elements.append(paragraph)
548
+ index += 1
549
+ elements.extend(floating)
550
+ elements.sort(
551
+ key=lambda element: (
552
+ element.cp if isinstance(element, DocImage) else element.cp_start,
553
+ 1 if isinstance(element, DocImage) else 0,
554
+ )
555
+ )
556
+ return elements
557
+
558
+
559
+ def _section_ranges(
560
+ table_stream: bytes,
561
+ fib: FileInformationBlock,
562
+ budget: DocBudget,
563
+ ) -> list[tuple[int, int]]:
564
+ """从 PlcfSed 恢复 section CP 范围,损坏时回退单 section。"""
565
+
566
+ pair = fib.pair(FCLCB_SECTION)
567
+ payload = bounded_slice(table_stream, pair.fc, pair.lcb)
568
+ if payload is None:
569
+ return [(0, fib.ccp_text)]
570
+ cps, items = parse_plc(payload, item_size=12, budget=budget)
571
+ if not items or len(cps) != len(items) + 1:
572
+ return [(0, fib.ccp_text)]
573
+ ranges: list[tuple[int, int]] = []
574
+ for start, end in zip(cps, cps[1:]):
575
+ if start > end or start > fib.ccp_text:
576
+ logger.warning("DOC PlcfSed CP values are invalid; using one recovery section")
577
+ return [(0, fib.ccp_text)]
578
+ ranges.append((start, min(end, fib.ccp_text)))
579
+ return ranges or [(0, fib.ccp_text)]
580
+
581
+
582
+ def _distribute_sections(
583
+ ranges: list[tuple[int, int]],
584
+ elements: list[DocElement],
585
+ ) -> list[DocSection]:
586
+ """按元素起始 CP 将主文档内容绑定到 section。"""
587
+
588
+ sections = [DocSection(start, end) for start, end in ranges]
589
+ for element in elements:
590
+ cp = element.cp if isinstance(element, DocImage) else element.cp_start
591
+ target = sections[-1]
592
+ for section in sections:
593
+ if section.cp_start <= cp < section.cp_end or (section is sections[-1] and cp == section.cp_end):
594
+ target = section
595
+ break
596
+ target.elements.append(element)
597
+ return sections
598
+
599
+
600
+ def _header_story_ranges(
601
+ table_stream: bytes,
602
+ fib: FileInformationBlock,
603
+ budget: DocBudget,
604
+ ) -> list[tuple[int, int]]:
605
+ """返回 PlcfHdd 中相对于 header story 的范围。"""
606
+
607
+ pair = fib.pair(FCLCB_HEADER)
608
+ payload = bounded_slice(table_stream, pair.fc, pair.lcb)
609
+ if payload is None or len(payload) < 8 or len(payload) % 4:
610
+ return []
611
+ budget.charge(len(payload) // 4)
612
+ cps = [int.from_bytes(payload[index : index + 4], "little") for index in range(0, len(payload), 4)]
613
+ return [(start, end) for start, end in zip(cps, cps[1:]) if start <= end]
614
+
615
+
616
+ def _attach_headers(
617
+ sections: list[DocSection],
618
+ assembler: _Assembler,
619
+ table_stream: bytes,
620
+ fib: FileInformationBlock,
621
+ budget: DocBudget,
622
+ ) -> None:
623
+ """把每节六种 header/footer story 的非空段落绑定到 section。"""
624
+
625
+ ranges = _header_story_ranges(table_stream, fib, budget)
626
+ base = fib.story_bases["header"]
627
+ for section_index, section in enumerate(sections):
628
+ group = 6 + section_index * 6
629
+ if group + 5 >= len(ranges):
630
+ break
631
+ for story_index in (0, 1, 4):
632
+ start, end = ranges[group + story_index]
633
+ section.headers.extend(assembler.paragraphs(base + start, base + end))
634
+ for story_index in (2, 3, 5):
635
+ start, end = ranges[group + story_index]
636
+ section.footers.extend(assembler.paragraphs(base + start, base + end))
637
+ inherited_header: DocParagraph | None = None
638
+ for section in sections:
639
+ if section.headers:
640
+ inherited_header = section.headers[0]
641
+ elif inherited_header is not None:
642
+ section.headers.append(inherited_header)
643
+
644
+
645
+ def _note_ranges(
646
+ table_stream: bytes,
647
+ *,
648
+ ref_offset: int,
649
+ ref_size: int,
650
+ text_offset: int,
651
+ text_size: int,
652
+ story_base: int,
653
+ prefix: str,
654
+ budget: DocBudget,
655
+ ) -> tuple[dict[int, str], list[tuple[int, int, int, str]]]:
656
+ """解析脚注或尾注 reference CP 与正文 story 范围。"""
657
+
658
+ ref_payload = bounded_slice(table_stream, ref_offset, ref_size)
659
+ text_payload = bounded_slice(table_stream, text_offset, text_size)
660
+ if ref_payload is None or text_payload is None:
661
+ return {}, []
662
+ ref_cps, ref_items = parse_plc(ref_payload, item_size=2, budget=budget)
663
+ text_cps, _ = parse_plc(text_payload, item_size=0, budget=budget)
664
+ references: dict[int, str] = {}
665
+ ranges: list[tuple[int, int, int, str]] = []
666
+ for index, _item in enumerate(ref_items):
667
+ if index >= len(ref_cps) or index + 1 >= len(text_cps):
668
+ break
669
+ label = str(index + 1)
670
+ references[ref_cps[index]] = label
671
+ ranges.append(
672
+ (
673
+ ref_cps[index],
674
+ story_base + text_cps[index],
675
+ story_base + text_cps[index + 1],
676
+ f"{prefix}{label}",
677
+ )
678
+ )
679
+ return references, ranges
680
+
681
+
682
+ def _prepend_note_label(paragraphs: list[DocParagraph], label: str) -> None:
683
+ """把脚注或尾注编号写入首个可见段落。"""
684
+
685
+ if not paragraphs:
686
+ return
687
+ paragraphs[0].runs.insert(0, DocTextRun(f"[{label.removeprefix('fn').removeprefix('en')}] "))
688
+
689
+
690
+ def _attach_notes(
691
+ sections: list[DocSection],
692
+ assembler: _Assembler,
693
+ footnote_ranges: list[tuple[int, int, int, str]],
694
+ endnote_ranges: list[tuple[int, int, int, str]],
695
+ ) -> None:
696
+ """脚注按引用 section、尾注按最后 section 追加。"""
697
+
698
+ for reference_cp, start, end, label in footnote_ranges:
699
+ paragraphs = assembler.paragraphs(start, end)
700
+ _prepend_note_label(paragraphs, label)
701
+ target = sections[-1]
702
+ for section in sections:
703
+ if section.cp_start <= reference_cp < section.cp_end:
704
+ target = section
705
+ break
706
+ target.footnotes.extend(paragraphs)
707
+ for _reference_cp, start, end, label in endnote_ranges:
708
+ paragraphs = assembler.paragraphs(start, end)
709
+ _prepend_note_label(paragraphs, label)
710
+ sections[-1].footnotes.extend(paragraphs)
711
+
712
+
713
+ def parse_doc_document(
714
+ word_document: bytes,
715
+ table_stream: bytes,
716
+ data_stream: bytes,
717
+ fib: FileInformationBlock,
718
+ *,
719
+ native_equations: dict[int, str] | None = None,
720
+ native_charts: dict[int, str] | None = None,
721
+ ) -> DocDocument:
722
+ """解析三个核心 streams 并返回逐 section 语义文档。"""
723
+
724
+ budget = DocBudget()
725
+ clx = fib.pair(FCLCB_CLX)
726
+ if clx.lcb:
727
+ pieces = parse_clx(table_stream, offset=clx.fc, size=clx.lcb, budget=budget)
728
+ elif not fib.base.complex:
729
+ pieces = legacy_single_piece(
730
+ fc_min=fib.base.fc_min,
731
+ fc_mac=fib.base.fc_mac,
732
+ ccp_text=fib.ccp_text,
733
+ )
734
+ else:
735
+ raise LegacyOfficeMalformedError("complex DOC is missing its CLX piece table")
736
+ if not pieces and fib.total_story_cp:
737
+ raise LegacyOfficeMalformedError("DOC contains no recoverable text pieces")
738
+ lid = fib.base.lid
739
+ if fib.base.far_east:
740
+ lid_fe = int.from_bytes(word_document[0x3C:0x3E], "little") if len(word_document) >= 0x3E else 0
741
+ if lid_fe:
742
+ lid = lid_fe
743
+ elif len(fib.rgw) > 13 and fib.rgw[13]:
744
+ lid = fib.rgw[13]
745
+ text = extract_text(
746
+ word_document,
747
+ pieces,
748
+ total_cp=fib.total_story_cp,
749
+ codec=codec_for_lid(lid),
750
+ budget=budget,
751
+ )
752
+ stsh = fib.pair(FCLCB_STSHF)
753
+ stylesheet = parse_stylesheet(
754
+ table_stream,
755
+ offset=stsh.fc,
756
+ size=stsh.lcb,
757
+ budget=budget,
758
+ )
759
+ chpx = fib.pair(FCLCB_BTE_CHPX)
760
+ papx = fib.pair(FCLCB_BTE_PAPX)
761
+ formatting = parse_formatting_runs(
762
+ word_document,
763
+ table_stream,
764
+ data_stream,
765
+ chpx_offset=chpx.fc,
766
+ chpx_size=chpx.lcb,
767
+ papx_offset=papx.fc,
768
+ papx_size=papx.lcb,
769
+ budget=budget,
770
+ )
771
+ lists_pair = fib.pair(FCLCB_LISTS)
772
+ overrides_pair = fib.pair(FCLCB_LIST_OVERRIDES)
773
+ lists = parse_list_tables(
774
+ table_stream,
775
+ list_offset=lists_pair.fc,
776
+ list_size=lists_pair.lcb,
777
+ override_offset=overrides_pair.fc,
778
+ override_size=overrides_pair.lcb,
779
+ budget=budget,
780
+ )
781
+ names = fib.pair(FCLCB_BOOKMARK_NAMES)
782
+ starts = fib.pair(FCLCB_BOOKMARK_START)
783
+ ends = fib.pair(FCLCB_BOOKMARK_END)
784
+ bookmarks = parse_bookmarks(
785
+ table_stream,
786
+ names_offset=names.fc,
787
+ names_size=names.lcb,
788
+ starts_offset=starts.fc,
789
+ starts_size=starts.lcb,
790
+ ends_offset=ends.fc,
791
+ ends_size=ends.lcb,
792
+ budget=budget,
793
+ )
794
+ foot_ref = fib.pair(FCLCB_FOOTNOTE_REF)
795
+ foot_text = fib.pair(FCLCB_FOOTNOTE_TEXT)
796
+ foot_refs, foot_ranges = _note_ranges(
797
+ table_stream,
798
+ ref_offset=foot_ref.fc,
799
+ ref_size=foot_ref.lcb,
800
+ text_offset=foot_text.fc,
801
+ text_size=foot_text.lcb,
802
+ story_base=fib.story_bases["footnote"],
803
+ prefix="fn",
804
+ budget=budget,
805
+ )
806
+ end_ref = fib.pair(FCLCB_ENDNOTE_REF)
807
+ end_text = fib.pair(FCLCB_ENDNOTE_TEXT)
808
+ end_refs, end_ranges = _note_ranges(
809
+ table_stream,
810
+ ref_offset=end_ref.fc,
811
+ ref_size=end_ref.lcb,
812
+ text_offset=end_text.fc,
813
+ text_size=end_text.lcb,
814
+ story_base=fib.story_bases["endnote"],
815
+ prefix="en",
816
+ budget=budget,
817
+ )
818
+ note_refs = {**foot_refs, **end_refs}
819
+ store = ImageStore()
820
+ assembler = _Assembler(
821
+ text=text,
822
+ pieces=pieces,
823
+ formatting=formatting,
824
+ stylesheet=stylesheet,
825
+ lists=lists,
826
+ bookmarks=bookmarks,
827
+ data_stream=data_stream,
828
+ image_store=store,
829
+ budget=budget,
830
+ native_equations=native_equations,
831
+ native_charts=native_charts,
832
+ )
833
+ main_paragraphs = assembler.paragraphs(0, fib.ccp_text, note_refs=note_refs)
834
+ shape_pair = fib.pair(FCLCB_SHAPE_MAIN)
835
+ drawing_pair = fib.pair(FCLCB_DGG_INFO)
836
+ floating = floating_pictures(
837
+ table_stream,
838
+ word_document=word_document,
839
+ shape_plc_offset=shape_pair.fc,
840
+ shape_plc_size=shape_pair.lcb,
841
+ drawing_offset=drawing_pair.fc,
842
+ drawing_size=drawing_pair.lcb,
843
+ store=store,
844
+ budget=budget,
845
+ )
846
+ elements = _assemble_main_elements(main_paragraphs, floating, budget)
847
+ ranges = _section_ranges(table_stream, fib, budget)
848
+ sections = _distribute_sections(ranges, elements)
849
+ _attach_headers(sections, assembler, table_stream, fib, budget)
850
+ _attach_notes(sections, assembler, foot_ranges, end_ranges)
851
+ return DocDocument(sections=sections)