docvortex 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (364) hide show
  1. docvortex/__init__.py +21 -0
  2. docvortex/analyzers/__init__.py +3 -0
  3. docvortex/analyzers/native/__init__.py +37 -0
  4. docvortex/analyzers/native/_shared/__init__.py +3 -0
  5. docvortex/analyzers/native/_shared/hyperlink.py +16 -0
  6. docvortex/analyzers/native/_shared/image.py +8 -0
  7. docvortex/analyzers/native/_shared/markup/__init__.py +47 -0
  8. docvortex/analyzers/native/_shared/markup/anchors.py +30 -0
  9. docvortex/analyzers/native/_shared/markup/formula.py +37 -0
  10. docvortex/analyzers/native/_shared/markup/projector.py +57 -0
  11. docvortex/analyzers/native/_shared/markup/styles.py +19 -0
  12. docvortex/analyzers/native/_shared/mathml.py +17 -0
  13. docvortex/analyzers/native/_shared/names.py +7 -0
  14. docvortex/analyzers/native/_shared/xycut.py +414 -0
  15. docvortex/analyzers/native/contracts.py +44 -0
  16. docvortex/analyzers/native/csv.py +351 -0
  17. docvortex/analyzers/native/epub/__init__.py +18 -0
  18. docvortex/analyzers/native/epub/constants.py +45 -0
  19. docvortex/analyzers/native/epub/converter.py +80 -0
  20. docvortex/analyzers/native/epub/errors.py +20 -0
  21. docvortex/analyzers/native/epub/metadata.py +31 -0
  22. docvortex/analyzers/native/epub/package.py +649 -0
  23. docvortex/analyzers/native/epub/xhtml.py +306 -0
  24. docvortex/analyzers/native/html/__init__.py +6 -0
  25. docvortex/analyzers/native/html/anchors.py +271 -0
  26. docvortex/analyzers/native/html/constants.py +25 -0
  27. docvortex/analyzers/native/html/contracts.py +7 -0
  28. docvortex/analyzers/native/html/converter.py +117 -0
  29. docvortex/analyzers/native/html/document.py +389 -0
  30. docvortex/analyzers/native/html/errors.py +12 -0
  31. docvortex/analyzers/native/html/resources.py +365 -0
  32. docvortex/analyzers/native/html/selector.py +421 -0
  33. docvortex/analyzers/native/models.py +209 -0
  34. docvortex/analyzers/native/ofd/__init__.py +14 -0
  35. docvortex/analyzers/native/ofd/constants.py +59 -0
  36. docvortex/analyzers/native/ofd/converter.py +31 -0
  37. docvortex/analyzers/native/ofd/errors.py +16 -0
  38. docvortex/analyzers/native/ofd/geometry.py +215 -0
  39. docvortex/analyzers/native/ofd/images.py +124 -0
  40. docvortex/analyzers/native/ofd/metadata.py +53 -0
  41. docvortex/analyzers/native/ofd/models.py +170 -0
  42. docvortex/analyzers/native/ofd/package.py +346 -0
  43. docvortex/analyzers/native/ofd/path.py +201 -0
  44. docvortex/analyzers/native/ofd/reading_order.py +293 -0
  45. docvortex/analyzers/native/ofd/resources.py +122 -0
  46. docvortex/analyzers/native/ofd/scene.py +347 -0
  47. docvortex/analyzers/native/ofd/table.py +240 -0
  48. docvortex/analyzers/native/ofd/text.py +521 -0
  49. docvortex/analyzers/native/office/__init__.py +3 -0
  50. docvortex/analyzers/native/office/doc/__init__.py +3 -0
  51. docvortex/analyzers/native/office/doc/bookmarks.py +81 -0
  52. docvortex/analyzers/native/office/doc/doc_converter.py +573 -0
  53. docvortex/analyzers/native/office/doc/fib.py +261 -0
  54. docvortex/analyzers/native/office/doc/fields.py +101 -0
  55. docvortex/analyzers/native/office/doc/formatting.py +171 -0
  56. docvortex/analyzers/native/office/doc/images.py +162 -0
  57. docvortex/analyzers/native/office/doc/lists.py +356 -0
  58. docvortex/analyzers/native/office/doc/models.py +164 -0
  59. docvortex/analyzers/native/office/doc/parser.py +851 -0
  60. docvortex/analyzers/native/office/doc/pieces.py +256 -0
  61. docvortex/analyzers/native/office/doc/records.py +40 -0
  62. docvortex/analyzers/native/office/doc/sprm.py +269 -0
  63. docvortex/analyzers/native/office/doc/styles.py +215 -0
  64. docvortex/analyzers/native/office/docx/__init__.py +3 -0
  65. docvortex/analyzers/native/office/docx/context.py +71 -0
  66. docvortex/analyzers/native/office/docx/docx_converter.py +614 -0
  67. docvortex/analyzers/native/office/docx/equationxml.py +119 -0
  68. docvortex/analyzers/native/office/docx/fields.py +816 -0
  69. docvortex/analyzers/native/office/docx/formatting_types.py +23 -0
  70. docvortex/analyzers/native/office/docx/main.py +50 -0
  71. docvortex/analyzers/native/office/docx/numbering.py +491 -0
  72. docvortex/analyzers/native/office/docx/office_xml.py +57 -0
  73. docvortex/analyzers/native/office/docx/package_normalizer.py +248 -0
  74. docvortex/analyzers/native/office/docx/resources.py +526 -0
  75. docvortex/analyzers/native/office/docx/styles.py +693 -0
  76. docvortex/analyzers/native/office/docx/tables.py +515 -0
  77. docvortex/analyzers/native/office/equation/__init__.py +3 -0
  78. docvortex/analyzers/native/office/equation/image.py +470 -0
  79. docvortex/analyzers/native/office/equation/latex_dict.py +324 -0
  80. docvortex/analyzers/native/office/equation/mtef.py +885 -0
  81. docvortex/analyzers/native/office/equation/mtef_v5.py +941 -0
  82. docvortex/analyzers/native/office/equation/omml.py +561 -0
  83. docvortex/analyzers/native/office/equation/ooxml.py +62 -0
  84. docvortex/analyzers/native/office/errors.py +33 -0
  85. docvortex/analyzers/native/office/image.py +307 -0
  86. docvortex/analyzers/native/office/legacy/__init__.py +3 -0
  87. docvortex/analyzers/native/office/legacy/binary.py +43 -0
  88. docvortex/analyzers/native/office/legacy/officeart.py +362 -0
  89. docvortex/analyzers/native/office/legacy/ole.py +110 -0
  90. docvortex/analyzers/native/office/limits.py +12 -0
  91. docvortex/analyzers/native/office/odf/__init__.py +3 -0
  92. docvortex/analyzers/native/office/odf/chart.py +76 -0
  93. docvortex/analyzers/native/office/odf/constants.py +70 -0
  94. docvortex/analyzers/native/office/odf/converters.py +403 -0
  95. docvortex/analyzers/native/office/odf/errors.py +18 -0
  96. docvortex/analyzers/native/office/odf/metadata.py +91 -0
  97. docvortex/analyzers/native/office/odf/models.py +176 -0
  98. docvortex/analyzers/native/office/odf/package.py +270 -0
  99. docvortex/analyzers/native/office/odf/styles.py +329 -0
  100. docvortex/analyzers/native/office/odf/table.py +469 -0
  101. docvortex/analyzers/native/office/odf/text.py +1002 -0
  102. docvortex/analyzers/native/office/ooxml_chart.py +1016 -0
  103. docvortex/analyzers/native/office/opc.py +38 -0
  104. docvortex/analyzers/native/office/ppt/__init__.py +3 -0
  105. docvortex/analyzers/native/office/ppt/models.py +118 -0
  106. docvortex/analyzers/native/office/ppt/parser.py +1895 -0
  107. docvortex/analyzers/native/office/ppt/ppt_converter.py +292 -0
  108. docvortex/analyzers/native/office/ppt/records.py +131 -0
  109. docvortex/analyzers/native/office/ppt/style_text.py +247 -0
  110. docvortex/analyzers/native/office/pptx/__init__.py +3 -0
  111. docvortex/analyzers/native/office/pptx/context.py +113 -0
  112. docvortex/analyzers/native/office/pptx/lists.py +558 -0
  113. docvortex/analyzers/native/office/pptx/main.py +20 -0
  114. docvortex/analyzers/native/office/pptx/package_normalizer.py +321 -0
  115. docvortex/analyzers/native/office/pptx/pptx_converter.py +323 -0
  116. docvortex/analyzers/native/office/pptx/resources.py +329 -0
  117. docvortex/analyzers/native/office/pptx/shapes.py +393 -0
  118. docvortex/analyzers/native/office/pptx/text_styles.py +625 -0
  119. docvortex/analyzers/native/office/pptx/titles.py +178 -0
  120. docvortex/analyzers/native/office/rich_text.py +420 -0
  121. docvortex/analyzers/native/office/rtf/__init__.py +3 -0
  122. docvortex/analyzers/native/office/rtf/converter.py +708 -0
  123. docvortex/analyzers/native/office/rtf/lexer.py +213 -0
  124. docvortex/analyzers/native/office/rtf/math.py +339 -0
  125. docvortex/analyzers/native/office/rtf/models.py +185 -0
  126. docvortex/analyzers/native/office/rtf/parser.py +1553 -0
  127. docvortex/analyzers/native/office/spreadsheet/__init__.py +3 -0
  128. docvortex/analyzers/native/office/spreadsheet/html.py +81 -0
  129. docvortex/analyzers/native/office/spreadsheet/models.py +79 -0
  130. docvortex/analyzers/native/office/spreadsheet/projector.py +928 -0
  131. docvortex/analyzers/native/office/streams.py +18 -0
  132. docvortex/analyzers/native/office/xls/__init__.py +3 -0
  133. docvortex/analyzers/native/office/xls/chart.py +132 -0
  134. docvortex/analyzers/native/office/xls/embedded_chart.py +299 -0
  135. docvortex/analyzers/native/office/xls/models.py +109 -0
  136. docvortex/analyzers/native/office/xls/number_format.py +521 -0
  137. docvortex/analyzers/native/office/xls/parser.py +1145 -0
  138. docvortex/analyzers/native/office/xls/records.py +201 -0
  139. docvortex/analyzers/native/office/xls/strings.py +205 -0
  140. docvortex/analyzers/native/office/xls/xls_converter.py +354 -0
  141. docvortex/analyzers/native/office/xlsx/__init__.py +3 -0
  142. docvortex/analyzers/native/office/xlsx/main.py +20 -0
  143. docvortex/analyzers/native/office/xlsx/ooxml_ole.py +522 -0
  144. docvortex/analyzers/native/office/xlsx/package_normalizer.py +310 -0
  145. docvortex/analyzers/native/office/xlsx/xlsx_converter.py +716 -0
  146. docvortex/analyzers/native/pdf/__init__.py +3 -0
  147. docvortex/analyzers/native/pdf/auxiliary_text.py +1671 -0
  148. docvortex/analyzers/native/pdf/char_geometry.py +1662 -0
  149. docvortex/analyzers/native/pdf/code_blocks.py +535 -0
  150. docvortex/analyzers/native/pdf/formulas.py +1985 -0
  151. docvortex/analyzers/native/pdf/geometry.py +281 -0
  152. docvortex/analyzers/native/pdf/graphics.py +1501 -0
  153. docvortex/analyzers/native/pdf/index_blocks.py +268 -0
  154. docvortex/analyzers/native/pdf/inline/__init__.py +3 -0
  155. docvortex/analyzers/native/pdf/inline/common.py +112 -0
  156. docvortex/analyzers/native/pdf/inline/detection.py +590 -0
  157. docvortex/analyzers/native/pdf/inline/matching.py +1185 -0
  158. docvortex/analyzers/native/pdf/inline/materialize.py +457 -0
  159. docvortex/analyzers/native/pdf/inline/scripts.py +975 -0
  160. docvortex/analyzers/native/pdf/inline/types.py +385 -0
  161. docvortex/analyzers/native/pdf/line_layout.py +1106 -0
  162. docvortex/analyzers/native/pdf/line_merging.py +1223 -0
  163. docvortex/analyzers/native/pdf/models.py +246 -0
  164. docvortex/analyzers/native/pdf/native_text.py +1004 -0
  165. docvortex/analyzers/native/pdf/pipeline.py +1390 -0
  166. docvortex/analyzers/native/pdf/script_geometry.py +636 -0
  167. docvortex/analyzers/native/pdf/shared.py +30 -0
  168. docvortex/analyzers/native/pdf/spatial_text.py +383 -0
  169. docvortex/analyzers/native/pdf/table_annotations.py +446 -0
  170. docvortex/analyzers/native/pdf/table_constants.py +48 -0
  171. docvortex/analyzers/native/pdf/table_detection.py +227 -0
  172. docvortex/analyzers/native/pdf/table_filled_grid.py +218 -0
  173. docvortex/analyzers/native/pdf/table_geometry.py +40 -0
  174. docvortex/analyzers/native/pdf/table_materialization.py +444 -0
  175. docvortex/analyzers/native/pdf/table_recovery/__init__.py +15 -0
  176. docvortex/analyzers/native/pdf/table_recovery/candidate.py +356 -0
  177. docvortex/analyzers/native/pdf/table_recovery/contracts.py +154 -0
  178. docvortex/analyzers/native/pdf/table_recovery/engine.py +609 -0
  179. docvortex/analyzers/native/pdf/table_recovery/geometry.py +164 -0
  180. docvortex/analyzers/native/pdf/table_recovery/sparse_common.py +85 -0
  181. docvortex/analyzers/native/pdf/table_recovery/sparse_hybrid.py +804 -0
  182. docvortex/analyzers/native/pdf/table_recovery/sparse_multiline.py +1122 -0
  183. docvortex/analyzers/native/pdf/table_recovery/text.py +414 -0
  184. docvortex/analyzers/native/pdf/table_recovery/text_grid.py +618 -0
  185. docvortex/analyzers/native/pdf/table_recovery/vector.py +1931 -0
  186. docvortex/analyzers/native/pdf/table_rows.py +34 -0
  187. docvortex/analyzers/native/pdf/table_rules.py +1129 -0
  188. docvortex/analyzers/native/pdf/table_text_styles.py +283 -0
  189. docvortex/analyzers/native/pdf/tables.py +147 -0
  190. docvortex/analyzers/native/pdf/text_assembly/__init__.py +3 -0
  191. docvortex/analyzers/native/pdf/text_assembly/annotations.py +581 -0
  192. docvortex/analyzers/native/pdf/text_assembly/assembly.py +292 -0
  193. docvortex/analyzers/native/pdf/text_assembly/common.py +477 -0
  194. docvortex/analyzers/native/pdf/text_assembly/footnotes.py +394 -0
  195. docvortex/analyzers/native/pdf/text_assembly/merging.py +1274 -0
  196. docvortex/analyzers/native/pdf/text_assembly/rows.py +692 -0
  197. docvortex/analyzers/native/pdf/text_blocks.py +82 -0
  198. docvortex/analyzers/native/pdf/text_styles.py +55 -0
  199. docvortex/analyzers/native/pdf/title_analysis/__init__.py +3 -0
  200. docvortex/analyzers/native/pdf/title_analysis/body_profile.py +215 -0
  201. docvortex/analyzers/native/pdf/title_analysis/common.py +117 -0
  202. docvortex/analyzers/native/pdf/title_analysis/document_profile.py +164 -0
  203. docvortex/analyzers/native/pdf/title_analysis/lane_titles.py +758 -0
  204. docvortex/analyzers/native/pdf/title_analysis/page_titles.py +1024 -0
  205. docvortex/analyzers/native/pdf/title_analysis/prototype.py +194 -0
  206. docvortex/analyzers/native/pdf/title_analysis/structural.py +1081 -0
  207. docvortex/analyzers/native/pdf/titles.py +75 -0
  208. docvortex/analyzers/native/pdf/typography.py +19 -0
  209. docvortex/analyzers/native/pdf/visual_annotations.py +1262 -0
  210. docvortex/api.py +180 -0
  211. docvortex/assets/__init__.py +5 -0
  212. docvortex/assets/store.py +51 -0
  213. docvortex/cli.py +54 -0
  214. docvortex/codecs/__init__.py +3 -0
  215. docvortex/codecs/html/__init__.py +22 -0
  216. docvortex/codecs/html/contracts.py +236 -0
  217. docvortex/codecs/html/materializer.py +331 -0
  218. docvortex/codecs/html/parser.py +763 -0
  219. docvortex/codecs/html/resources.py +34 -0
  220. docvortex/codecs/json.py +17 -0
  221. docvortex/content/__init__.py +5 -0
  222. docvortex/content/inline.py +248 -0
  223. docvortex/content/markup/__init__.py +44 -0
  224. docvortex/content/markup/anchors.py +188 -0
  225. docvortex/content/markup/formula.py +280 -0
  226. docvortex/content/markup/projector.py +1237 -0
  227. docvortex/content/markup/styles.py +327 -0
  228. docvortex/content/mathml.py +167 -0
  229. docvortex/content/normalization.py +188 -0
  230. docvortex/content/spans.py +183 -0
  231. docvortex/content/table/__init__.py +18 -0
  232. docvortex/content/table/blocks.py +152 -0
  233. docvortex/content/table/content.py +425 -0
  234. docvortex/content/table/document.py +104 -0
  235. docvortex/content/table/html.py +399 -0
  236. docvortex/content/table/models.py +76 -0
  237. docvortex/content/table/rules.py +42 -0
  238. docvortex/content/table/structure.py +221 -0
  239. docvortex/content/tree.py +5 -0
  240. docvortex/document/__init__.py +3 -0
  241. docvortex/document/contracts.py +22 -0
  242. docvortex/document/detection.py +389 -0
  243. docvortex/document/filetypes.py +175 -0
  244. docvortex/document/page_range.py +167 -0
  245. docvortex/document/pdf/__init__.py +19 -0
  246. docvortex/document/pdf/classify.py +1138 -0
  247. docvortex/document/pdf/constants.py +132 -0
  248. docvortex/document/pdf/diagnostics.py +350 -0
  249. docvortex/document/pdf/document.py +582 -0
  250. docvortex/document/pdf/font_runtime.py +335 -0
  251. docvortex/document/pdf/geometry.py +31 -0
  252. docvortex/document/pdf/images.py +654 -0
  253. docvortex/document/pdf/native_annotations.py +367 -0
  254. docvortex/document/pdf/native_contracts.py +169 -0
  255. docvortex/document/pdf/native_coordinates.py +216 -0
  256. docvortex/document/pdf/native_lifecycle.py +16 -0
  257. docvortex/document/pdf/native_objects.py +902 -0
  258. docvortex/document/pdf/native_text_geometry.py +315 -0
  259. docvortex/document/pdf/pdfium.py +325 -0
  260. docvortex/document/pdf/raster.py +46 -0
  261. docvortex/document/pdf/text/__init__.py +62 -0
  262. docvortex/document/pdf/text/contracts.py +211 -0
  263. docvortex/document/pdf/text/extract.py +165 -0
  264. docvortex/document/pdf/text/geometry.py +16 -0
  265. docvortex/document/pdf/text/groups.py +162 -0
  266. docvortex/document/pdf/visual_geometry.py +201 -0
  267. docvortex/document/pdf/visuals.py +343 -0
  268. docvortex/document/source.py +92 -0
  269. docvortex/errors.py +21 -0
  270. docvortex/export/__init__.py +3 -0
  271. docvortex/export/bundle.py +98 -0
  272. docvortex/export/files.py +66 -0
  273. docvortex/export/middle.py +208 -0
  274. docvortex/foundation/__init__.py +3 -0
  275. docvortex/foundation/geometry.py +125 -0
  276. docvortex/foundation/hyperlink.py +65 -0
  277. docvortex/foundation/image.py +48 -0
  278. docvortex/foundation/image_encoding.py +30 -0
  279. docvortex/foundation/image_payload.py +280 -0
  280. docvortex/foundation/language.py +92 -0
  281. docvortex/foundation/platform.py +38 -0
  282. docvortex/foundation/text.py +153 -0
  283. docvortex/foundation/type_identity.py +20 -0
  284. docvortex/foundation/xml_names.py +20 -0
  285. docvortex/options.py +30 -0
  286. docvortex/postprocess/__init__.py +3 -0
  287. docvortex/postprocess/content.py +53 -0
  288. docvortex/postprocess/document.py +19 -0
  289. docvortex/postprocess/lists.py +236 -0
  290. docvortex/postprocess/page_blocks.py +214 -0
  291. docvortex/postprocess/pages.py +95 -0
  292. docvortex/postprocess/paragraphs.py +580 -0
  293. docvortex/postprocess/visual.py +715 -0
  294. docvortex/render/__init__.py +48 -0
  295. docvortex/render/_internal/__init__.py +3 -0
  296. docvortex/render/_internal/common/__init__.py +3 -0
  297. docvortex/render/_internal/common/context.py +43 -0
  298. docvortex/render/_internal/common/html_table.py +178 -0
  299. docvortex/render/_internal/common/index.py +33 -0
  300. docvortex/render/_internal/common/list_items.py +158 -0
  301. docvortex/render/_internal/common/planner.py +140 -0
  302. docvortex/render/_internal/docx/__init__.py +3 -0
  303. docvortex/render/_internal/docx/assets.py +202 -0
  304. docvortex/render/_internal/docx/inline.py +434 -0
  305. docvortex/render/_internal/docx/math.py +220 -0
  306. docvortex/render/_internal/docx/renderer.py +905 -0
  307. docvortex/render/_internal/docx/styles.py +195 -0
  308. docvortex/render/_internal/docx/table.py +442 -0
  309. docvortex/render/_internal/epub/__init__.py +5 -0
  310. docvortex/render/_internal/epub/assets.py +173 -0
  311. docvortex/render/_internal/epub/package.py +249 -0
  312. docvortex/render/_internal/epub/renderer.py +1156 -0
  313. docvortex/render/_internal/html/__init__.py +3 -0
  314. docvortex/render/_internal/html/inline.py +346 -0
  315. docvortex/render/_internal/html/renderer.py +1041 -0
  316. docvortex/render/_internal/html/sanitizer.py +478 -0
  317. docvortex/render/_internal/html/table.py +121 -0
  318. docvortex/render/_internal/latex/__init__.py +1 -0
  319. docvortex/render/_internal/latex/assets.py +85 -0
  320. docvortex/render/_internal/latex/inline.py +145 -0
  321. docvortex/render/_internal/latex/renderer.py +506 -0
  322. docvortex/render/_internal/latex/table.py +347 -0
  323. docvortex/render/_internal/markdown/__init__.py +3 -0
  324. docvortex/render/_internal/markdown/assets.py +78 -0
  325. docvortex/render/_internal/markdown/blocks.py +635 -0
  326. docvortex/render/_internal/markdown/escaping.py +51 -0
  327. docvortex/render/_internal/markdown/inline.py +260 -0
  328. docvortex/render/_internal/markdown/renderer.py +93 -0
  329. docvortex/render/_internal/markdown/table.py +281 -0
  330. docvortex/render/_internal/pdf/__init__.py +3 -0
  331. docvortex/render/_internal/pdf/assets.py +197 -0
  332. docvortex/render/_internal/pdf/formula.py +417 -0
  333. docvortex/render/_internal/pdf/inline.py +343 -0
  334. docvortex/render/_internal/pdf/renderer.py +734 -0
  335. docvortex/render/_internal/pdf/styles.py +206 -0
  336. docvortex/render/_internal/pdf/table.py +272 -0
  337. docvortex/render/_internal/structured_content/__init__.py +3 -0
  338. docvortex/render/_internal/structured_content/renderer.py +193 -0
  339. docvortex/render/api.py +199 -0
  340. docvortex/render/contracts.py +205 -0
  341. docvortex/render/docx.py +38 -0
  342. docvortex/render/epub.py +35 -0
  343. docvortex/render/fragments.py +70 -0
  344. docvortex/render/html.py +29 -0
  345. docvortex/render/latex.py +24 -0
  346. docvortex/render/markdown.py +50 -0
  347. docvortex/render/pdf.py +25 -0
  348. docvortex/render/structured_content.py +23 -0
  349. docvortex/resources/epub/docvortex.css +91 -0
  350. docvortex/resources/fasttext-langdetect/lid.176.ftz +0 -0
  351. docvortex/resources/fonts/DroidSansFallbackFull.ttf +0 -0
  352. docvortex/resources/fonts/NOTICE +190 -0
  353. docvortex/resources/fonts/manifest.json +9 -0
  354. docvortex/resources/html/docvortex.css +601 -0
  355. docvortex/resources/html/docvortex.min.css +1 -0
  356. docvortex/result.py +91 -0
  357. docvortex/schema.py +1141 -0
  358. docvortex/version.py +3 -0
  359. docvortex-0.2.1.dist-info/METADATA +193 -0
  360. docvortex-0.2.1.dist-info/RECORD +364 -0
  361. docvortex-0.2.1.dist-info/WHEEL +5 -0
  362. docvortex-0.2.1.dist-info/entry_points.txt +2 -0
  363. docvortex-0.2.1.dist-info/licenses/LICENSE.md +21 -0
  364. docvortex-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,521 @@
1
+ """从 TextObject/TextCode 恢复语义文字与页面几何。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import html
6
+ import math
7
+ import re
8
+ from collections import deque
9
+ from dataclasses import dataclass
10
+ from io import BytesIO, StringIO
11
+
12
+ from fontTools.pens.boundsPen import BoundsPen
13
+ from loguru import logger
14
+ from lxml import etree # type: ignore[reportMissingImports]
15
+
16
+ from ....content.spans import text_spans
17
+ from ....schema import BBox
18
+ from .constants import MAX_DELTA_TOKENS, MAX_EXPANDED_GLYPHS, MAX_EXPANDED_TEXT_BYTES, MAX_FONT_BYTES, MAX_GLYPH_TOKENS
19
+ from .errors import OfdResourceLimitError
20
+ from .geometry import (
21
+ Affine,
22
+ bbox_intersection,
23
+ bbox_union,
24
+ canonical_angle,
25
+ parse_affine,
26
+ parse_st_box,
27
+ quad_bbox,
28
+ rect_quad,
29
+ transform_angle,
30
+ transform_bbox,
31
+ transform_quad,
32
+ )
33
+ from .models import FontResource, GlyphItem, ResourceRegistry, TextLine
34
+ from .package import OfdPackage, element_text, local_name, parse_int
35
+
36
+ _HEX_ESCAPE_RE = re.compile(r"\\([0-9A-Fa-f]{4})")
37
+ _GLYPH_TOKEN_RE = re.compile(r"\S+")
38
+ _DELTA_TOKEN_RE = re.compile(r"[^,\s]+")
39
+ _HEX_DIGITS = frozenset("0123456789abcdefABCDEF")
40
+ _TEXT_CODE_DECODE_CHUNK_SIZE = 64 * 1024
41
+
42
+
43
+ @dataclass(slots=True)
44
+ class OfdTextBudget:
45
+ """累计限制 TextCode 文字与展开字形数量。"""
46
+
47
+ text_bytes: int = 0
48
+ glyph_count: int = 0
49
+ glyph_mapping_count: int = 0
50
+ glyph_token_count: int = 0
51
+ delta_token_count: int = 0
52
+
53
+ def charge(self, text: str) -> None:
54
+ """为一次 TextCode 展开计费。"""
55
+ glyph_count = self.glyph_count + len(text)
56
+ if glyph_count > MAX_EXPANDED_GLYPHS:
57
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_expanded_glyphs={MAX_EXPANDED_GLYPHS}")
58
+ text_bytes = self.text_bytes + len(text.encode("utf-8"))
59
+ if text_bytes > MAX_EXPANDED_TEXT_BYTES:
60
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_expanded_text_bytes={MAX_EXPANDED_TEXT_BYTES}")
61
+ self.glyph_count = glyph_count
62
+ self.text_bytes = text_bytes
63
+
64
+ def charge_glyph_mapping(self, count: int) -> None:
65
+ """累计 CGTransform 的有效字符映射数量并限制全文展开量。"""
66
+ self.glyph_mapping_count += count
67
+ if self.glyph_mapping_count > MAX_EXPANDED_GLYPHS:
68
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_expanded_glyphs={MAX_EXPANDED_GLYPHS}")
69
+
70
+ def charge_glyph_token(self) -> None:
71
+ """累计实际扫描的 Glyphs token 数量并限制全文解析量。"""
72
+ self.glyph_token_count += 1
73
+ if self.glyph_token_count > MAX_GLYPH_TOKENS:
74
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_glyph_tokens={MAX_GLYPH_TOKENS}")
75
+
76
+ def charge_delta_token(self) -> None:
77
+ """累计实际扫描的 Delta token 数量并限制全文解析量。"""
78
+ self.delta_token_count += 1
79
+ if self.delta_token_count > MAX_DELTA_TOKENS:
80
+ raise OfdResourceLimitError(f"OFD resource limit exceeded: max_delta_tokens={MAX_DELTA_TOKENS}")
81
+
82
+
83
+ @dataclass(slots=True)
84
+ class _LoadedFont:
85
+ """缓存 FontTools 中与字形几何相关的只读表。"""
86
+
87
+ font: object
88
+ glyph_order: list[str]
89
+ glyph_set: object
90
+ units_per_em: float
91
+ advances: dict[str, tuple[int, int]]
92
+ char_to_name: dict[int, str]
93
+ name_to_char: dict[str, str]
94
+
95
+
96
+ class FontMetricResolver:
97
+ """按 OFD 字体资源惰性读取内嵌 OpenType 指标。"""
98
+
99
+ def __init__(self, package: OfdPackage) -> None:
100
+ """绑定当前包并创建字体解析缓存。"""
101
+ self.package = package
102
+ self._cache: dict[str, _LoadedFont | None] = {}
103
+
104
+ def _load(self, resource: FontResource | None) -> _LoadedFont | None:
105
+ """读取一个受限字体成员,失败时缓存空结果。"""
106
+ if resource is None or resource.font_part is None:
107
+ return None
108
+ if resource.font_part in self._cache:
109
+ return self._cache[resource.font_part]
110
+ data = self.package.read_part(resource.font_part, asset=True)
111
+ if data is None or len(data) > MAX_FONT_BYTES:
112
+ logger.warning(
113
+ f"OFD_FONT_UNAVAILABLE: part={resource.font_part!r}, reason={'missing' if data is None else 'too_large'}"
114
+ )
115
+ self._cache[resource.font_part] = None
116
+ return None
117
+ try:
118
+ from fontTools.ttLib import TTFont
119
+
120
+ font = TTFont(BytesIO(data), lazy=True)
121
+ glyph_order = list(font.getGlyphOrder())
122
+ glyph_set = font.getGlyphSet()
123
+ units_per_em = float(font["head"].unitsPerEm) if "head" in font else 1000.0
124
+ advances = dict(font["hmtx"].metrics) if "hmtx" in font else {}
125
+ char_to_name: dict[int, str] = {}
126
+ if "cmap" in font:
127
+ for table in font["cmap"].tables:
128
+ char_to_name.update(table.cmap)
129
+ name_to_char: dict[str, str] = {}
130
+ for codepoint, glyph_name in char_to_name.items():
131
+ if glyph_name not in name_to_char and 0 <= codepoint <= 0x10FFFF:
132
+ name_to_char[glyph_name] = chr(codepoint)
133
+ loaded = _LoadedFont(
134
+ font=font,
135
+ glyph_order=glyph_order,
136
+ glyph_set=glyph_set,
137
+ units_per_em=max(units_per_em, 1.0),
138
+ advances=advances,
139
+ char_to_name=char_to_name,
140
+ name_to_char=name_to_char,
141
+ )
142
+ except Exception as exc:
143
+ logger.warning(f"OFD_FONT_INVALID: part={resource.font_part!r}, error={type(exc).__name__}")
144
+ loaded = None
145
+ self._cache[resource.font_part] = loaded
146
+ return loaded
147
+
148
+ def resolve_character(self, resource: FontResource | None, glyph_id: int | None, fallback: str) -> str:
149
+ """在 TextCode 使用占位符时尝试由 glyph cmap 恢复字符。"""
150
+ if fallback != "¤" or glyph_id is None:
151
+ return fallback
152
+ loaded = self._load(resource)
153
+ if loaded is None or not (0 <= glyph_id < len(loaded.glyph_order)):
154
+ return fallback
155
+ return loaded.name_to_char.get(loaded.glyph_order[glyph_id], fallback)
156
+
157
+ def glyph_bbox(
158
+ self,
159
+ resource: FontResource | None,
160
+ glyph_id: int | None,
161
+ character: str,
162
+ *,
163
+ size: float,
164
+ hscale: float,
165
+ advance_hint: float | None,
166
+ ) -> BBox:
167
+ """返回以 glyph origin 为基准的确定性局部字形框。"""
168
+ loaded = self._load(resource)
169
+ glyph_name: str | None = None
170
+ if loaded is not None:
171
+ if glyph_id is not None and 0 <= glyph_id < len(loaded.glyph_order):
172
+ glyph_name = loaded.glyph_order[glyph_id]
173
+ elif character:
174
+ glyph_name = loaded.char_to_name.get(ord(character[0]))
175
+ if loaded is not None and glyph_name is not None and glyph_name in loaded.glyph_set:
176
+ scale = size / loaded.units_per_em
177
+ try:
178
+ pen = BoundsPen(loaded.glyph_set)
179
+ loaded.glyph_set[glyph_name].draw(pen)
180
+ bounds = pen.bounds
181
+ except Exception:
182
+ bounds = None
183
+ advance = float(loaded.advances.get(glyph_name, (loaded.units_per_em, 0))[0]) * scale * hscale
184
+ if advance_hint is not None and advance_hint > 0:
185
+ advance = advance_hint
186
+ if bounds is not None:
187
+ x0, y0, x1, y1 = bounds
188
+ bbox = (x0 * scale * hscale, -y1 * scale, x1 * scale * hscale, -y0 * scale)
189
+ if bbox[2] > bbox[0] and bbox[3] > bbox[1]:
190
+ return bbox
191
+ return (0.0, -0.85 * size, max(advance, 0.2 * size), 0.2 * size)
192
+ fallback_advance = (
193
+ advance_hint if advance_hint is not None and advance_hint > 0 else max(0.5 * size * hscale, 0.2 * size)
194
+ )
195
+ return (0.0, -0.85 * size, fallback_advance, 0.2 * size)
196
+
197
+ def close(self) -> None:
198
+ """关闭已经打开的 FontTools 字体对象。"""
199
+ for loaded in self._cache.values():
200
+ if loaded is None:
201
+ continue
202
+ close = getattr(loaded.font, "close", None)
203
+ if callable(close):
204
+ close()
205
+
206
+
207
+ def decode_text_code(value: str) -> str:
208
+ """解码 OFD TextCode 中的反斜杠四位十六进制字符。"""
209
+ return _HEX_ESCAPE_RE.sub(lambda match: chr(int(match.group(1), 16)), value)
210
+
211
+
212
+ def _incomplete_hex_escape_length(value: str) -> int:
213
+ """返回末尾可能跨分片的反斜杠十六进制前缀长度。"""
214
+ for length in range(min(4, len(value)), 0, -1):
215
+ suffix = value[-length:]
216
+ if suffix[0] == "\\" and all(character in _HEX_DIGITS for character in suffix[1:]):
217
+ return length
218
+ return 0
219
+
220
+
221
+ def _decode_text_code_element(text_code: etree._Element, budget: OfdTextBudget) -> str:
222
+ """分片解码一个 TextCode,并在写入完整字符串前累计文字预算。"""
223
+ output = StringIO()
224
+ carry = ""
225
+ for part in text_code.itertext():
226
+ for offset in range(0, len(part), _TEXT_CODE_DECODE_CHUNK_SIZE):
227
+ chunk = part[offset : offset + _TEXT_CODE_DECODE_CHUNK_SIZE]
228
+ value = f"{carry}{chunk}" if carry else chunk
229
+ carry_length = _incomplete_hex_escape_length(value)
230
+ if carry_length:
231
+ complete = value[:-carry_length]
232
+ carry = value[-carry_length:]
233
+ else:
234
+ complete = value
235
+ carry = ""
236
+ if not complete:
237
+ continue
238
+ decoded = decode_text_code(complete)
239
+ budget.charge(decoded)
240
+ output.write(decoded)
241
+ if carry:
242
+ budget.charge(carry)
243
+ output.write(carry)
244
+ return output.getvalue()
245
+
246
+
247
+ def parse_delta(value: str | None, count: int, budget: OfdTextBudget) -> list[float]:
248
+ """流式展开普通与 g-count-value 压缩 Delta,并补齐不足项。"""
249
+ if count <= 0:
250
+ return []
251
+ output: list[float] = []
252
+ matches = iter(_DELTA_TOKEN_RE.finditer(value or ""))
253
+ pending: deque[str] = deque()
254
+
255
+ def next_token() -> str | None:
256
+ """返回下一个 Delta token,并只对首次扫描计费。"""
257
+ if pending:
258
+ return pending.popleft()
259
+ try:
260
+ match = next(matches)
261
+ except StopIteration:
262
+ return None
263
+ budget.charge_delta_token()
264
+ return match.group()
265
+
266
+ while len(output) < count:
267
+ token = next_token()
268
+ if token is None:
269
+ break
270
+ if token.casefold() == "g":
271
+ repeat_token = next_token()
272
+ value_token = next_token()
273
+ if repeat_token is None or value_token is None:
274
+ if repeat_token is not None:
275
+ pending.appendleft(repeat_token)
276
+ continue
277
+ try:
278
+ repeat = max(0, int(repeat_token))
279
+ repeated_value = float(value_token)
280
+ except ValueError:
281
+ pending.appendleft(value_token)
282
+ pending.appendleft(repeat_token)
283
+ continue
284
+ if math.isfinite(repeated_value):
285
+ output.extend([repeated_value] * min(repeat, count - len(output)))
286
+ continue
287
+ try:
288
+ parsed = float(token)
289
+ except ValueError:
290
+ continue
291
+ if math.isfinite(parsed):
292
+ output.append(parsed)
293
+ if len(output) < count:
294
+ output.extend([output[-1] if output else 0.0] * (count - len(output)))
295
+ return output[:count]
296
+
297
+
298
+ def _bounded_glyph_ids(glyphs_element: etree._Element, limit: int, budget: OfdTextBudget) -> list[int]:
299
+ """按需迭代 Glyphs token,只保留有效映射所需的有限 ID。"""
300
+ glyph_ids: list[int] = []
301
+ for match in _GLYPH_TOKEN_RE.finditer(element_text(glyphs_element)):
302
+ budget.charge_glyph_token()
303
+ glyph_id = parse_int(match.group())
304
+ if glyph_id is None:
305
+ continue
306
+ glyph_ids.append(glyph_id)
307
+ if len(glyph_ids) >= limit:
308
+ break
309
+ return glyph_ids
310
+
311
+
312
+ def _glyph_map(text_object: etree._Element, position_count: int, budget: OfdTextBudget) -> dict[int, int]:
313
+ """把实际 TextCode 字符位置映射到 glyph ID,并限制累计展开量。"""
314
+ result: dict[int, int] = {}
315
+ for element in text_object:
316
+ if local_name(element.tag) != "CGTransform":
317
+ continue
318
+ code_position = parse_int(element.get("CodePosition"))
319
+ code_count = parse_int(element.get("CodeCount"))
320
+ glyphs_element = next((child for child in element if local_name(child.tag) == "Glyphs"), None)
321
+ if code_position is None or code_count is None or glyphs_element is None:
322
+ continue
323
+ effective_count = min(code_count, max(0, position_count - code_position))
324
+ if effective_count <= 0:
325
+ continue
326
+ glyph_ids = _bounded_glyph_ids(glyphs_element, effective_count, budget)
327
+ if not glyph_ids:
328
+ continue
329
+ budget.charge_glyph_mapping(effective_count)
330
+ for offset in range(effective_count):
331
+ glyph_index = min(offset, len(glyph_ids) - 1)
332
+ result[code_position + offset] = glyph_ids[glyph_index]
333
+ return result
334
+
335
+
336
+ def _styles(
337
+ text_object: etree._Element,
338
+ font: FontResource | None,
339
+ resolved_style: dict[str, str],
340
+ ) -> tuple[str, ...]:
341
+ """从字体资源和 TextObject 属性恢复可投影行内样式。"""
342
+ styles: list[str] = []
343
+ weight = parse_int(resolved_style.get("Weight") or text_object.get("Weight"))
344
+ if (font is not None and font.bold) or (weight is not None and weight >= 600):
345
+ styles.append("bold")
346
+ if (font is not None and font.italic) or (resolved_style.get("Italic") or text_object.get("Italic") or "").casefold() in {
347
+ "true",
348
+ "1",
349
+ }:
350
+ styles.append("italic")
351
+ return tuple(styles)
352
+
353
+
354
+ def format_line_spans(text: str, styles: tuple[str, ...]) -> list[dict[str, object]]:
355
+ """把 OFD 原生文字和样式直接投影为结构化 Span。"""
356
+ return text_spans(text, styles)
357
+
358
+
359
+ def format_line_html(text: str, styles: tuple[str, ...]) -> str:
360
+ """把 OFD 表格单元格文字序列化为安全 HTML。"""
361
+ rendered = html.escape(text, quote=False)
362
+ if "superscript" in styles:
363
+ rendered = f"<sup>{rendered}</sup>"
364
+ elif "subscript" in styles:
365
+ rendered = f"<sub>{rendered}</sub>"
366
+ if "underline" in styles:
367
+ rendered = f"<u>{rendered}</u>"
368
+ if "bold" in styles:
369
+ rendered = f"<strong>{rendered}</strong>"
370
+ if "italic" in styles:
371
+ rendered = f"<em>{rendered}</em>"
372
+ if "strikethrough" in styles:
373
+ rendered = f"<s>{rendered}</s>"
374
+ return rendered
375
+
376
+
377
+ def build_text_lines(
378
+ text_object: etree._Element,
379
+ *,
380
+ parent_transform: Affine,
381
+ parent_clip: BBox,
382
+ resources: ResourceRegistry,
383
+ package: OfdPackage,
384
+ font_metrics: FontMetricResolver,
385
+ budget: OfdTextBudget,
386
+ paint_order: int,
387
+ layer_type: str,
388
+ template_id: int | None,
389
+ resolved_style: dict[str, str] | None = None,
390
+ ) -> list[TextLine]:
391
+ """把一个 TextObject 展开为按 TextCode 划分的页面文字行。"""
392
+ style = resolved_style or {}
393
+ if (style.get("Visible") or text_object.get("Visible") or "true").casefold() in {"false", "0"}:
394
+ return []
395
+ if (style.get("Alpha") or text_object.get("Alpha") or "255").strip() == "0":
396
+ return []
397
+ boundary = parse_st_box(text_object.get("Boundary"))
398
+ if boundary is None:
399
+ logger.warning(f"OFD_TEXT_INVALID_BOUNDARY: object_id={text_object.get('ID')!r}")
400
+ return []
401
+ boundary_page = transform_bbox(boundary, parent_transform)
402
+ if boundary_page is None:
403
+ return []
404
+ object_clip = bbox_intersection(parent_clip, boundary_page)
405
+ if object_clip is None:
406
+ return []
407
+ translation = Affine.translation(boundary[0], boundary[1])
408
+ object_transform = parent_transform.compose(translation).compose(parse_affine(text_object.get("CTM")))
409
+ read_direction = float(parse_int(text_object.get("ReadDirection")) or 0)
410
+ char_direction = float(parse_int(text_object.get("CharDirection")) or 0)
411
+ direction_transform = object_transform.compose(Affine.rotation(read_direction))
412
+ font_id = parse_int(text_object.get("Font"))
413
+ font = resources.fonts.get(font_id) if font_id is not None else None
414
+ try:
415
+ size = max(0.1, float(text_object.get("Size") or 1.0))
416
+ hscale = max(0.01, float(text_object.get("HScale") or 1.0))
417
+ except ValueError:
418
+ size, hscale = 1.0, 1.0
419
+ styles = _styles(text_object, font, style)
420
+ decoded_text_codes: list[tuple[etree._Element, str]] = []
421
+ position_count = 0
422
+ for text_code in (element for element in text_object if local_name(element.tag) == "TextCode"):
423
+ text = _decode_text_code_element(text_code, budget)
424
+ decoded_text_codes.append((text_code, text))
425
+ if text:
426
+ position_count += len(text)
427
+ glyph_by_position = _glyph_map(text_object, position_count, budget)
428
+ global_position = 0
429
+ inherited_x: float | None = None
430
+ inherited_y: float | None = None
431
+ lines: list[TextLine] = []
432
+ object_id = parse_int(text_object.get("ID"))
433
+ for code_index, (text_code, text) in enumerate(decoded_text_codes):
434
+ if not text:
435
+ continue
436
+ try:
437
+ if text_code.get("X") is not None:
438
+ inherited_x = float(text_code.get("X") or "")
439
+ if text_code.get("Y") is not None:
440
+ inherited_y = float(text_code.get("Y") or "")
441
+ except ValueError:
442
+ continue
443
+ if inherited_x is None or inherited_y is None:
444
+ logger.warning(f"OFD_TEXT_MISSING_ORIGIN: object_id={object_id}, text_code={code_index}")
445
+ global_position += len(text)
446
+ continue
447
+ delta_count = max(0, len(text) - 1)
448
+ delta_x = parse_delta(text_code.get("DeltaX"), delta_count, budget)
449
+ delta_y = parse_delta(text_code.get("DeltaY"), delta_count, budget)
450
+ origins: list[tuple[float, float]] = [(inherited_x, inherited_y)]
451
+ for index in range(delta_count):
452
+ previous = origins[-1]
453
+ origins.append((previous[0] + delta_x[index], previous[1] + delta_y[index]))
454
+ glyph_items: list[GlyphItem] = []
455
+ for index, (character, origin) in enumerate(zip(text, origins, strict=True)):
456
+ glyph_id = glyph_by_position.get(global_position + index)
457
+ resolved_character = font_metrics.resolve_character(font, glyph_id, character)
458
+ advance_hint = None
459
+ if index + 1 < len(origins):
460
+ advance_hint = math.dist(origin, origins[index + 1])
461
+ local_bbox = font_metrics.glyph_bbox(
462
+ font,
463
+ glyph_id,
464
+ resolved_character,
465
+ size=size,
466
+ hscale=hscale,
467
+ advance_hint=advance_hint,
468
+ )
469
+ char_transform = direction_transform.compose(Affine.translation(origin[0], origin[1])).compose(
470
+ Affine.rotation(char_direction)
471
+ )
472
+ quad = transform_quad(rect_quad(local_bbox), char_transform)
473
+ glyph_bbox = quad_bbox(quad)
474
+ if glyph_bbox is None:
475
+ continue
476
+ clipped_glyph_bbox = bbox_intersection(glyph_bbox, object_clip)
477
+ if clipped_glyph_bbox is None:
478
+ continue
479
+ glyph_items.append(
480
+ GlyphItem(
481
+ text=resolved_character,
482
+ bbox=clipped_glyph_bbox,
483
+ quad=quad,
484
+ origin=char_transform.apply((0.0, 0.0)),
485
+ glyph_id=glyph_id,
486
+ )
487
+ )
488
+ global_position += len(text)
489
+ if not glyph_items:
490
+ continue
491
+ line_bbox = bbox_union(item.bbox for item in glyph_items)
492
+ if line_bbox is None:
493
+ continue
494
+ if line_bbox[2] <= line_bbox[0] or line_bbox[3] <= line_bbox[1]:
495
+ continue
496
+ lines.append(
497
+ TextLine(
498
+ text="".join(item.text for item in glyph_items),
499
+ bbox=line_bbox,
500
+ glyphs=glyph_items,
501
+ angle=canonical_angle(transform_angle(direction_transform)),
502
+ font_size=size * math.hypot(direction_transform.a, direction_transform.b),
503
+ paint_order=paint_order + code_index,
504
+ object_id=object_id,
505
+ layer_type=layer_type,
506
+ template_id=template_id,
507
+ styles=styles,
508
+ )
509
+ )
510
+ return lines
511
+
512
+
513
+ __all__ = [
514
+ "FontMetricResolver",
515
+ "OfdTextBudget",
516
+ "build_text_lines",
517
+ "decode_text_code",
518
+ "format_line_html",
519
+ "format_line_spans",
520
+ "parse_delta",
521
+ ]
@@ -0,0 +1,3 @@
1
+ """Flash Office 模型共享实现。"""
2
+
3
+ __all__: list[str] = []
@@ -0,0 +1,3 @@
1
+ """Word 97–2003 二进制文档解析实现。"""
2
+
3
+ __all__: list[str] = []
@@ -0,0 +1,81 @@
1
+ """解析 DOC 标准书签名称及其主文档 CP 范围。"""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ..legacy.binary import bounded_slice, get_u16
6
+ from .records import DocBudget, parse_plc
7
+
8
+
9
+ def _parse_string_table(data: bytes, budget: DocBudget) -> list[str]:
10
+ """解析扩展或单字节 STTB 字符串表。"""
11
+
12
+ if len(data) < 2:
13
+ return []
14
+ extended = get_u16(data, 0) == 0xFFFF
15
+ if extended:
16
+ count = get_u16(data, 2)
17
+ extra = get_u16(data, 4)
18
+ cursor = 6
19
+ else:
20
+ count = get_u16(data, 0)
21
+ extra = get_u16(data, 2)
22
+ cursor = 4
23
+ if count is None or extra is None:
24
+ return []
25
+ strings: list[str] = []
26
+ for _ in range(count):
27
+ if extended:
28
+ length = get_u16(data, cursor)
29
+ cursor += 2
30
+ width = 2
31
+ else:
32
+ length = data[cursor] if cursor < len(data) else None
33
+ cursor += 1
34
+ width = 1
35
+ if length is None or length == 0xFFFF:
36
+ strings.append("")
37
+ continue
38
+ payload = bounded_slice(data, cursor, length * width)
39
+ if payload is None:
40
+ break
41
+ cursor += length * width
42
+ strings.append(payload.decode("utf-16le" if extended else "cp1252", errors="replace"))
43
+ cursor += extra
44
+ budget.charge()
45
+ return strings
46
+
47
+
48
+ def parse_bookmarks(
49
+ table_stream: bytes,
50
+ *,
51
+ names_offset: int,
52
+ names_size: int,
53
+ starts_offset: int,
54
+ starts_size: int,
55
+ ends_offset: int,
56
+ ends_size: int,
57
+ budget: DocBudget,
58
+ ) -> dict[int, list[str]]:
59
+ """返回主文档中书签起始 CP 到名称列表的映射。"""
60
+
61
+ names_payload = bounded_slice(table_stream, names_offset, names_size)
62
+ starts_payload = bounded_slice(table_stream, starts_offset, starts_size)
63
+ ends_payload = bounded_slice(table_stream, ends_offset, ends_size)
64
+ if names_payload is None or starts_payload is None or ends_payload is None:
65
+ return {}
66
+ names = _parse_string_table(names_payload, budget)
67
+ start_cps, start_items = parse_plc(starts_payload, item_size=4, budget=budget)
68
+ end_cps, _ = parse_plc(ends_payload, item_size=0, budget=budget)
69
+ result: dict[int, list[str]] = {}
70
+ for index, (name, item) in enumerate(zip(names, start_items, strict=False)):
71
+ if not name or name.startswith("_GoBack") or index >= len(start_cps):
72
+ continue
73
+ end_index = get_u16(item, 0)
74
+ if end_index is None or end_index >= max(len(end_cps) - 1, 0):
75
+ continue
76
+ start = start_cps[index]
77
+ end = end_cps[end_index]
78
+ if end < start:
79
+ continue
80
+ result.setdefault(start, []).append(name)
81
+ return result