@amaster.ai/pi-lark 0.1.7 → 0.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (215) hide show
  1. package/package.json +2 -2
  2. package/skills/lark-apps/SKILL.md +41 -6
  3. package/skills/lark-apps/references/lark-apps-cloud-dev.md +5 -4
  4. package/skills/lark-apps/references/lark-apps-create.md +6 -3
  5. package/skills/lark-apps/references/lark-apps-db.md +130 -2
  6. package/skills/lark-apps/references/lark-apps-get.md +1 -1
  7. package/skills/lark-apps/references/lark-apps-list.md +1 -1
  8. package/skills/lark-apps/references/lark-apps-local-dev.md +27 -1
  9. package/skills/lark-apps/references/lark-apps-release-create.md +1 -1
  10. package/skills/lark-apps/references/lark-apps-user-id-convert.md +63 -0
  11. package/skills/lark-base/SKILL.md +155 -159
  12. package/skills/lark-base/references/{lark-base-role-guide.md → lark-base-advanced-permission-and-role.md} +5 -5
  13. package/skills/lark-base/references/lark-base-app-block-data-config.md +122 -0
  14. package/skills/lark-base/references/lark-base-app.md +225 -0
  15. package/skills/lark-base/references/lark-base-cell-value.md +26 -19
  16. package/skills/lark-base/references/{dashboard-block-data-config.md → lark-base-dashboard-block-config.md} +37 -5
  17. package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +18 -2
  18. package/skills/lark-base/references/lark-base-dashboard.md +25 -12
  19. package/skills/lark-base/references/lark-base-data-analysis-pandas.md +93 -0
  20. package/skills/lark-base/references/lark-base-data-analysis-python-stdlib.md +120 -0
  21. package/skills/lark-base/references/lark-base-data-query.md +8 -11
  22. package/skills/lark-base/references/lark-base-field-create.md +13 -45
  23. package/skills/lark-base/references/{formula-field-guide.md → lark-base-field-formula.md} +1 -1
  24. package/skills/lark-base/references/{lookup-field-guide.md → lark-base-field-lookup.md} +1 -1
  25. package/skills/lark-base/references/{lark-base-field-json.md → lark-base-field-schema.md} +18 -100
  26. package/skills/lark-base/references/lark-base-field-update.md +13 -51
  27. package/skills/lark-base/references/lark-base-filter-condition.md +19 -31
  28. package/skills/lark-base/references/lark-base-record-batch-create.md +5 -1
  29. package/skills/lark-base/references/lark-base-record-batch-update.md +5 -2
  30. package/skills/lark-base/references/lark-base-record-query-and-analysis-cloud-sop.md +145 -0
  31. package/skills/lark-base/references/lark-base-record-query-and-analysis-sop.md +233 -0
  32. package/skills/lark-base/references/{role-config.md → lark-base-role-config.md} +2 -2
  33. package/skills/lark-base/references/lark-base-view-set-filter.md +1 -1
  34. package/skills/lark-base/references/lark-base-workflow-schema.md +2 -2
  35. package/skills/lark-base/references/{lark-base-workflow-guide.md → lark-base-workflow.md} +1 -1
  36. package/skills/lark-calendar/SKILL.md +3 -1
  37. package/skills/lark-calendar/references/lark-calendar-create.md +4 -3
  38. package/skills/lark-doc/SKILL.md +26 -61
  39. package/skills/lark-doc/references/genres/business-analysis.md +30 -0
  40. package/skills/lark-doc/references/genres/data-report.md +32 -0
  41. package/skills/lark-doc/references/genres/email.md +38 -0
  42. package/skills/lark-doc/references/genres/execution-plan.md +27 -0
  43. package/skills/lark-doc/references/genres/formal-doc.md +37 -0
  44. package/skills/lark-doc/references/genres/meeting-minutes.md +24 -0
  45. package/skills/lark-doc/references/genres/memo-brief.md +25 -0
  46. package/skills/lark-doc/references/genres/official-redhead.md +73 -0
  47. package/skills/lark-doc/references/genres/prd.md +26 -0
  48. package/skills/lark-doc/references/genres/proposal.md +24 -0
  49. package/skills/lark-doc/references/genres/research-report.md +32 -0
  50. package/skills/lark-doc/references/genres/retrospective.md +25 -0
  51. package/skills/lark-doc/references/genres/route-consumer.md +37 -0
  52. package/skills/lark-doc/references/genres/route-creative.md +36 -0
  53. package/skills/lark-doc/references/genres/route-knowledge.md +39 -0
  54. package/skills/lark-doc/references/genres/route-marketing.md +40 -0
  55. package/skills/lark-doc/references/genres/route-media.md +36 -0
  56. package/skills/lark-doc/references/genres/route-opinion.md +38 -0
  57. package/skills/lark-doc/references/genres/route-personal-brand.md +36 -0
  58. package/skills/lark-doc/references/genres/route-platform.md +9 -0
  59. package/skills/lark-doc/references/genres/route-report.md +10 -0
  60. package/skills/lark-doc/references/genres/route-workplace.md +17 -0
  61. package/skills/lark-doc/references/genres/sop-tutorial.md +41 -0
  62. package/skills/lark-doc/references/genres/technical-doc.md +39 -0
  63. package/skills/lark-doc/references/genres/wechat.md +39 -0
  64. package/skills/lark-doc/references/genres/weekly-report.md +24 -0
  65. package/skills/lark-doc/references/genres/white-paper.md +32 -0
  66. package/skills/lark-doc/references/genres/xiaohongshu.md +38 -0
  67. package/skills/lark-doc/references/lark-doc-create-workflow.md +121 -0
  68. package/skills/lark-doc/references/lark-doc-create.md +22 -48
  69. package/skills/lark-doc/references/lark-doc-fetch.md +80 -92
  70. package/skills/lark-doc/references/lark-doc-history.md +16 -15
  71. package/skills/lark-doc/references/lark-doc-md.md +5 -1
  72. package/skills/lark-doc/references/lark-doc-media-download.md +2 -1
  73. package/skills/lark-doc/references/lark-doc-script.md +76 -0
  74. package/skills/lark-doc/references/lark-doc-update.md +73 -221
  75. package/skills/lark-doc/references/lark-doc-whiteboard.md +5 -9
  76. package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +17 -12
  77. package/skills/lark-doc/references/lark-doc-xml.md +38 -167
  78. package/skills/lark-drive/SKILL.md +11 -7
  79. package/skills/lark-drive/references/lark-drive-apply-permission.md +1 -1
  80. package/skills/lark-drive/references/lark-drive-copy.md +87 -0
  81. package/skills/lark-drive/references/lark-drive-download.md +29 -2
  82. package/skills/lark-drive/references/lark-drive-export.md +4 -0
  83. package/skills/lark-drive/references/lark-drive-member-remove.md +59 -0
  84. package/skills/lark-drive/references/lark-drive-preview.md +21 -2
  85. package/skills/lark-drive/references/lark-drive-push.md +5 -1
  86. package/skills/lark-drive/references/lark-drive-search.md +2 -0
  87. package/skills/lark-drive/references/lark-drive-task-result.md +3 -0
  88. package/skills/lark-drive/references/lark-drive-update-title.md +78 -0
  89. package/skills/lark-event/SKILL.md +7 -4
  90. package/skills/lark-event/references/lark-event-vc.md +8 -2
  91. package/skills/lark-im/SKILL.md +14 -9
  92. package/skills/lark-im/references/lark-im-chat-list.md +9 -2
  93. package/skills/lark-im/references/lark-im-chat-members-list.md +7 -4
  94. package/skills/lark-im/references/lark-im-chat-messages-list.md +10 -3
  95. package/skills/lark-im/references/lark-im-chat-search.md +9 -2
  96. package/skills/lark-im/references/lark-im-feed-group-list-item.md +2 -2
  97. package/skills/lark-im/references/lark-im-feed-group-list.md +2 -2
  98. package/skills/lark-im/references/lark-im-feed-shortcut-list.md +1 -1
  99. package/skills/lark-im/references/lark-im-flag-list.md +2 -2
  100. package/skills/lark-im/references/lark-im-message-enrichment.md +1 -1
  101. package/skills/lark-im/references/lark-im-messages-resources-download.md +19 -25
  102. package/skills/lark-im/references/lark-im-messages-search.md +4 -5
  103. package/skills/lark-im/references/lark-im-threads-messages-list.md +8 -4
  104. package/skills/lark-mail/references/lark-mail-triage.md +19 -4
  105. package/skills/lark-minutes/SKILL.md +12 -6
  106. package/skills/lark-minutes/references/lark-minutes-apply-permission.md +95 -0
  107. package/skills/lark-minutes/references/lark-minutes-detail.md +7 -6
  108. package/skills/lark-minutes/references/lark-minutes-download.md +4 -2
  109. package/skills/lark-minutes/references/lark-minutes-search.md +6 -7
  110. package/skills/lark-note/SKILL.md +13 -9
  111. package/skills/lark-note/references/lark-note-detail.md +5 -2
  112. package/skills/lark-note/references/lark-note-transcript.md +2 -0
  113. package/skills/lark-shared/SKILL.md +39 -3
  114. package/skills/lark-sheets/SKILL.md +83 -82
  115. package/skills/lark-sheets/references/lark-sheets-batch-update.md +13 -58
  116. package/skills/lark-sheets/references/lark-sheets-chart.md +2 -1
  117. package/skills/lark-sheets/references/lark-sheets-conditional-format.md +1 -1
  118. package/skills/lark-sheets/references/lark-sheets-range-operations.md +5 -5
  119. package/skills/lark-sheets/references/lark-sheets-read-data.md +80 -6
  120. package/skills/lark-sheets/references/lark-sheets-sheet-structure.md +21 -10
  121. package/skills/lark-sheets/references/lark-sheets-styles-put.md +93 -0
  122. package/skills/lark-sheets/references/lark-sheets-visual-standards.md +2 -2
  123. package/skills/lark-sheets/references/lark-sheets-workbook.md +4 -3
  124. package/skills/lark-sheets/references/lark-sheets-write-cells.md +40 -12
  125. package/skills/lark-sheets/scripts/lark_detect_subtables.py +593 -0
  126. package/skills/lark-sheets/scripts/lark_inspect_workbook.py +188 -0
  127. package/skills/lark-sheets/scripts/lark_profile_table.py +614 -0
  128. package/skills/lark-sheets/scripts/lark_sheet_range.py +176 -0
  129. package/skills/lark-sheets/scripts/lark_sheet_read_cli.py +184 -0
  130. package/skills/lark-sheets/scripts/sheets_df.py +21 -3
  131. package/skills/lark-slides/SKILL.md +64 -81
  132. package/skills/lark-slides/references/cli/lark-slides-add-slide.md +92 -0
  133. package/skills/lark-slides/references/cli/lark-slides-create.md +176 -0
  134. package/skills/lark-slides/references/cli/lark-slides-delete-slide.md +65 -0
  135. package/skills/lark-slides/references/cli/lark-slides-history.md +132 -0
  136. package/skills/lark-slides/references/cli/lark-slides-media-upload.md +103 -0
  137. package/skills/lark-slides/references/cli/lark-slides-replace-slide.md +259 -0
  138. package/skills/lark-slides/references/cli/lark-slides-screenshot.md +115 -0
  139. package/skills/lark-slides/references/cli/lark-slides-update-slide.md +163 -0
  140. package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-get.md +110 -0
  141. package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-replace.md +188 -0
  142. package/skills/lark-slides/references/cli/lark-slides-xml-presentations-get.md +157 -0
  143. package/skills/lark-slides/references/iconpark-index.json +5 -41901
  144. package/skills/lark-slides/references/iconpark.md +3 -44
  145. package/skills/lark-slides/references/lark-slides-add-slide.md +5 -0
  146. package/skills/lark-slides/references/lark-slides-create.md +3 -162
  147. package/skills/lark-slides/references/lark-slides-delete-slide.md +5 -0
  148. package/skills/lark-slides/references/lark-slides-edit-workflows.md +3 -142
  149. package/skills/lark-slides/references/lark-slides-history.md +3 -130
  150. package/skills/lark-slides/references/lark-slides-media-upload.md +3 -124
  151. package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +3 -83
  152. package/skills/lark-slides/references/lark-slides-replace-slide.md +3 -235
  153. package/skills/lark-slides/references/lark-slides-screenshot.md +3 -95
  154. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +3 -108
  155. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +3 -186
  156. package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +3 -132
  157. package/skills/lark-slides/references/planning-layer.md +1 -1
  158. package/skills/lark-slides/references/slides_chart_demo.xml +5 -1416
  159. package/skills/lark-slides/references/slides_xml_schema_definition.xml +3 -3468
  160. package/skills/lark-slides/references/troubleshooting.md +3 -61
  161. package/skills/lark-slides/references/validation-checklist.md +3 -154
  162. package/skills/lark-slides/references/workflow/error-handling.md +62 -0
  163. package/skills/lark-slides/references/workflow/slides-editing.md +143 -0
  164. package/skills/lark-slides/references/workflow/template-editing.md +85 -0
  165. package/skills/lark-slides/references/workflow/validation-xml.md +156 -0
  166. package/skills/lark-slides/references/xml/iconpark-index.json +37458 -0
  167. package/skills/lark-slides/references/xml/iconpark.md +46 -0
  168. package/skills/lark-slides/references/xml/slides_chart_demo.xml +1415 -0
  169. package/skills/lark-slides/references/xml/slides_xml_schema_definition.xml +3514 -0
  170. package/skills/lark-slides/references/xml/xml-schema-quick-ref.md +497 -0
  171. package/skills/lark-slides/references/xml-schema-quick-ref.md +3 -483
  172. package/skills/lark-slides/scripts/iconpark_tool.py +1 -1
  173. package/skills/lark-slides/scripts/sxsd_validator.py +154 -10
  174. package/skills/lark-slides/scripts/xml_lint.py +2989 -0
  175. package/skills/lark-slides/scripts/xml_lint_test.py +4720 -0
  176. package/skills/lark-slides/scripts/xml_text_overlap_lint.py +3 -2691
  177. package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +5 -3788
  178. package/skills/lark-task/SKILL.md +12 -0
  179. package/skills/lark-task/references/lark-task-create.md +3 -1
  180. package/skills/lark-vc/SKILL.md +15 -5
  181. package/skills/lark-vc/references/lark-vc-detail.md +11 -6
  182. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-events.md → lark-vc/references/lark-vc-meeting-events.md} +121 -20
  183. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-list-active.md → lark-vc/references/lark-vc-meeting-list-active.md} +2 -2
  184. package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-message-send.md → lark-vc/references/lark-vc-meeting-message-send.md} +3 -3
  185. package/skills/lark-vc/references/lark-vc-recording.md +8 -6
  186. package/skills/lark-vc/references/vc-domain-boundaries.md +8 -1
  187. package/skills/lark-vc-agent/SKILL.md +24 -9
  188. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-join.md +2 -2
  189. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-leave.md +2 -2
  190. package/skills/lark-whiteboard/SKILL.md +15 -8
  191. package/skills/lark-whiteboard/references/lark-whiteboard-export.md +4 -3
  192. package/skills/lark-whiteboard/references/lark-whiteboard-update.md +4 -4
  193. package/skills/lark-whiteboard/references/lark-whiteboard-workflow.md +19 -17
  194. package/skills/lark-whiteboard/routes/dsl.md +8 -2
  195. package/skills/lark-whiteboard/routes/mermaid.md +1 -1
  196. package/skills/lark-whiteboard/routes/svg-edit.md +5 -2
  197. package/skills/lark-whiteboard/routes/svg.md +3 -1
  198. package/skills/lark-whiteboard/scenes/mention.md +71 -0
  199. package/skills/lark-wiki/SKILL.md +8 -4
  200. package/skills/lark-wiki/references/lark-wiki-delete-space.md +6 -3
  201. package/skills/lark-wiki/references/lark-wiki-node-copy.md +5 -19
  202. package/skills/lark-wiki/references/lark-wiki-node-create.md +19 -2
  203. package/skills/lark-wiki/references/lark-wiki-node-get.md +15 -0
  204. package/skills/lark-wiki/references/lark-wiki-node-list.md +1 -1
  205. package/skills/lark-base/references/lark-base-data-analysis-sop.md +0 -210
  206. package/skills/lark-base/references/lark-base-data-query-guide.md +0 -61
  207. package/skills/lark-base/references/lark-base-record-upsert.md +0 -63
  208. package/skills/lark-doc/references/lark-doc-word-stat.md +0 -93
  209. package/skills/lark-doc/references/style/lark-doc-create-workflow.md +0 -47
  210. package/skills/lark-doc/references/style/lark-doc-style.md +0 -68
  211. package/skills/lark-doc/references/style/lark-doc-update-workflow.md +0 -48
  212. package/skills/lark-doc/scripts/doc_word_stat.py +0 -1243
  213. package/skills/lark-slides/references/lark-slides-replace-pages.md +0 -95
  214. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-create.md +0 -219
  215. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +0 -126
@@ -1,1243 +0,0 @@
1
- #!/usr/bin/env python3
2
- # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
- # SPDX-License-Identifier: MIT
4
- """Standalone Lark Docs word and character counter for XML or Markdown input."""
5
-
6
- from __future__ import annotations
7
-
8
- import argparse
9
- import json
10
- import re
11
- import sys
12
- import unicodedata
13
- from dataclasses import dataclass, field
14
- from pathlib import Path
15
- from typing import Any, Literal, Protocol
16
- from xml.etree import ElementTree as ET
17
-
18
- # ---------------------------------------------------------------------------
19
- # Data model
20
- # ---------------------------------------------------------------------------
21
-
22
- @dataclass
23
- class TextRun:
24
- text: str
25
- attrs: dict[str, Any] = field(default_factory=dict)
26
-
27
-
28
- @dataclass
29
- class Block:
30
- type: str
31
- attrs: dict[str, Any] = field(default_factory=dict)
32
- children: list["Block"] = field(default_factory=list)
33
- text_runs: list[TextRun] = field(default_factory=list)
34
- raw: Any = None
35
-
36
-
37
- @dataclass
38
- class Segment:
39
- text: str
40
- block_type: str
41
- block_id: str | None = None
42
- kind: str = "text"
43
- boundary_before: bool = True
44
- boundary_after: bool = True
45
-
46
- def to_dict(self) -> dict[str, Any]:
47
- return {
48
- "text": self.text,
49
- "block_type": self.block_type,
50
- "block_id": self.block_id,
51
- "kind": self.kind,
52
- "boundary_before": self.boundary_before,
53
- "boundary_after": self.boundary_after,
54
- }
55
-
56
-
57
- @dataclass(frozen=True)
58
- class UnknownBlock:
59
- type: str
60
- block_id: str | None = None
61
- action: str = "recurse_children"
62
-
63
- def to_dict(self) -> dict[str, str | None]:
64
- return {
65
- "type": self.type,
66
- "block_id": self.block_id,
67
- "action": self.action,
68
- }
69
-
70
- # ---------------------------------------------------------------------------
71
- # Counting rules
72
- # ---------------------------------------------------------------------------
73
-
74
- CHINESE_PUNCTUATION = set(",。!?;:、()《》〈〉“”‘’【】「」『』〔〕…—~·¥")
75
- ENGLISH_PUNCTUATION = set(
76
- r"""!"#$%&'()*+,-./:;<=>?@[\]^_`{|}~"""
77
- )
78
-
79
-
80
- LexemeKind = Literal["english", "number"]
81
- URL_TOKEN_RE = re.compile(r"https?://[!-~]+")
82
- ASCII_COMPOUND_TOKEN_RE = re.compile(
83
- r"[A-Za-z0-9]+(?:[._/@:-][A-Za-z0-9]+)+"
84
- )
85
-
86
-
87
- @dataclass
88
- class Stats:
89
- word_count: int = 0
90
- char_count: int = 0
91
- han_chars: int = 0
92
- english_words: int = 0
93
- number_words: int = 0
94
- chinese_punctuations: int = 0
95
- english_letters: int = 0
96
- digits: int = 0
97
- english_punctuations: int = 0
98
- symbol_words: int = 0
99
- symbol_chars: int = 0
100
-
101
- def to_dict(self) -> dict[str, object]:
102
- return {
103
- "word_count": self.word_count,
104
- "char_count": self.char_count,
105
- "breakdown": {
106
- "han_chars": self.han_chars,
107
- "english_words": self.english_words,
108
- "number_words": self.number_words,
109
- "chinese_punctuations": self.chinese_punctuations,
110
- "english_letters": self.english_letters,
111
- "digits": self.digits,
112
- "english_punctuations": self.english_punctuations,
113
- "symbol_words": self.symbol_words,
114
- "symbol_chars": self.symbol_chars,
115
- },
116
- }
117
-
118
-
119
- def is_han(ch: str) -> bool:
120
- code = ord(ch)
121
- return (
122
- 0x3400 <= code <= 0x4DBF
123
- or 0x4E00 <= code <= 0x9FFF
124
- or 0xF900 <= code <= 0xFAFF
125
- or 0x20000 <= code <= 0x2A6DF
126
- or 0x2A700 <= code <= 0x2B73F
127
- or 0x2B740 <= code <= 0x2B81F
128
- or 0x2B820 <= code <= 0x2CEAF
129
- or 0x30000 <= code <= 0x3134F
130
- )
131
-
132
-
133
- def is_ascii_letter(ch: str) -> bool:
134
- return ("a" <= ch <= "z") or ("A" <= ch <= "Z")
135
-
136
-
137
- def is_digit(ch: str) -> bool:
138
- return "0" <= ch <= "9"
139
-
140
-
141
- def is_chinese_punctuation(ch: str) -> bool:
142
- if ch in CHINESE_PUNCTUATION:
143
- return True
144
- return unicodedata.category(ch).startswith("P") and unicodedata.east_asian_width(ch) in {
145
- "W",
146
- "F",
147
- }
148
-
149
-
150
- def is_english_punctuation(ch: str) -> bool:
151
- return ch in ENGLISH_PUNCTUATION
152
-
153
-
154
- def is_unicode_symbol(ch: str) -> bool:
155
- return unicodedata.category(ch).startswith("S")
156
-
157
-
158
- def utf16_units(ch: str) -> int:
159
- return len(ch.encode("utf-16-le")) // 2
160
-
161
-
162
- class Counter:
163
- def __init__(self) -> None:
164
- self.stats = Stats()
165
- self._lexeme_kind: LexemeKind | None = None
166
- self._lexeme_has_digit = False
167
- self._symbol_run_length = 0
168
- self._at_boundary = True
169
-
170
- def count_segments(self, segments: list[Segment]) -> Stats:
171
- for segment in segments:
172
- if segment.boundary_before:
173
- self._end_unit()
174
- self._at_boundary = True
175
- if segment.kind == "marker":
176
- self.write_marker(segment.text)
177
- elif segment.kind == "code":
178
- self.write_code(segment.text)
179
- else:
180
- self.write(segment.text)
181
- if segment.boundary_after:
182
- self._end_unit()
183
- self._at_boundary = True
184
- self._end_unit()
185
- return self.stats
186
-
187
- def write(self, text: str) -> None:
188
- i = 0
189
- while i < len(text):
190
- consumed = self._write_ascii_compound_token(text, i)
191
- if consumed:
192
- i += consumed
193
- continue
194
- if self._write_visible_ascii_separator(text, i):
195
- i += 1
196
- continue
197
- self._write_char(text[i])
198
- i += 1
199
-
200
- def write_marker(self, text: str) -> None:
201
- for ch in text:
202
- if ch.isspace():
203
- continue
204
- self._end_unit()
205
- self.stats.word_count += 1
206
- self.stats.char_count += 1
207
- self._at_boundary = False
208
-
209
- def write_code(self, text: str) -> None:
210
- for ch in text:
211
- self._write_code_char(ch)
212
-
213
- def _write_code_char(self, ch: str) -> None:
214
- if ch.isspace():
215
- self._end_unit()
216
- self._at_boundary = True
217
- return
218
-
219
- if is_han(ch):
220
- self._end_lexeme()
221
- self._end_symbol_run(count_word=False)
222
- self.stats.han_chars += 1
223
- self.stats.word_count += 1
224
- self.stats.char_count += 1
225
- self._at_boundary = False
226
- return
227
-
228
- if is_ascii_letter(ch):
229
- self._end_symbol_run(count_word=False)
230
- self.stats.english_letters += 1
231
- self.stats.char_count += 1
232
- if self._lexeme_kind is None:
233
- self._lexeme_kind = "english"
234
- elif self._lexeme_kind == "number":
235
- self._lexeme_kind = "english"
236
- self._at_boundary = False
237
- return
238
-
239
- if is_digit(ch):
240
- self._end_symbol_run(count_word=False)
241
- self.stats.digits += 1
242
- self.stats.char_count += 1
243
- self._at_boundary = False
244
- return
245
-
246
- if is_chinese_punctuation(ch):
247
- self._end_lexeme()
248
- self._end_symbol_run(count_word=False)
249
- self.stats.chinese_punctuations += 1
250
- self.stats.word_count += 1
251
- self.stats.char_count += 1
252
- self._at_boundary = False
253
- return
254
-
255
- if is_english_punctuation(ch):
256
- keeps_lexeme = self._lexeme_kind == "english" and ch in {"'", "-"}
257
- if not keeps_lexeme:
258
- had_lexeme = self._lexeme_kind is not None
259
- self._end_lexeme()
260
- if not had_lexeme and (self._symbol_run_length > 0 or self._at_boundary):
261
- self._symbol_run_length += 1
262
- self.stats.english_punctuations += 1
263
- self.stats.char_count += 1
264
- if keeps_lexeme:
265
- self._at_boundary = False
266
- return
267
-
268
- if is_unicode_symbol(ch):
269
- self._write_symbol_char(ch)
270
- return
271
-
272
- self._end_lexeme()
273
- self._end_symbol_run(count_word=False)
274
- self._at_boundary = False
275
-
276
- def _write_char(self, ch: str) -> None:
277
- if ch.isspace():
278
- self._end_unit()
279
- self._at_boundary = True
280
- return
281
-
282
- if is_han(ch):
283
- self._end_lexeme()
284
- self._end_symbol_run(count_word=False)
285
- self.stats.han_chars += 1
286
- self.stats.word_count += 1
287
- self.stats.char_count += 1
288
- self._at_boundary = False
289
- return
290
-
291
- if is_ascii_letter(ch):
292
- self._end_symbol_run(count_word=False)
293
- self.stats.english_letters += 1
294
- self.stats.char_count += 1
295
- if self._lexeme_kind is None:
296
- self._lexeme_kind = "english"
297
- elif self._lexeme_kind == "number":
298
- self._lexeme_kind = "english"
299
- self._at_boundary = False
300
- return
301
-
302
- if is_digit(ch):
303
- self._end_symbol_run(count_word=False)
304
- self.stats.digits += 1
305
- self.stats.char_count += 1
306
- self._lexeme_has_digit = True
307
- if self._lexeme_kind is None:
308
- self._lexeme_kind = "number"
309
- self._at_boundary = False
310
- return
311
-
312
- if is_chinese_punctuation(ch):
313
- self._end_lexeme()
314
- self._end_symbol_run(count_word=False)
315
- self.stats.chinese_punctuations += 1
316
- self.stats.word_count += 1
317
- self.stats.char_count += 1
318
- self._at_boundary = False
319
- return
320
-
321
- if is_english_punctuation(ch):
322
- # Apostrophes/hyphens can connect English runs. Dot/comma/hyphen
323
- # can format numeric runs such as 3.14, 1,000, 2026-06-30, or
324
- # 7-9. Alphanumeric versions like v1.2.3 should remain one semantic
325
- # run too. These punctuations still count as characters.
326
- keeps_lexeme = (
327
- self._lexeme_kind == "english"
328
- and (ch in {"'", "-"} or (self._lexeme_has_digit and ch == "."))
329
- ) or (
330
- self._lexeme_kind == "number"
331
- and ch in {".", ",", "-"}
332
- )
333
- if not keeps_lexeme:
334
- had_lexeme = self._lexeme_kind is not None
335
- self._end_lexeme()
336
- if not had_lexeme and (self._symbol_run_length > 0 or self._at_boundary):
337
- self._symbol_run_length += 1
338
- self.stats.english_punctuations += 1
339
- self.stats.char_count += 1
340
- if keeps_lexeme:
341
- self._at_boundary = False
342
- return
343
-
344
- if is_unicode_symbol(ch):
345
- self._write_symbol_char(ch)
346
- return
347
-
348
- self._end_lexeme()
349
- self._end_symbol_run(count_word=False)
350
- self._at_boundary = False
351
-
352
- def _write_visible_ascii_separator(self, text: str, index: int) -> bool:
353
- ch = text[index]
354
- if ch != "/" or index == 0 or index + 1 >= len(text):
355
- return False
356
- if not is_han(text[index - 1]) or not is_han(text[index + 1]):
357
- return False
358
-
359
- self._end_unit()
360
- self.stats.english_punctuations += 1
361
- self.stats.symbol_words += 1
362
- self.stats.word_count += 1
363
- self.stats.char_count += 1
364
- self._at_boundary = False
365
- return True
366
-
367
- def _write_ascii_compound_token(self, text: str, start: int) -> int:
368
- token = self._match_ascii_compound_token(text, start)
369
- if not token:
370
- return 0
371
-
372
- self._end_unit()
373
- self.stats.english_words += 1
374
- self.stats.word_count += 1
375
- for ch in token:
376
- if is_ascii_letter(ch):
377
- self.stats.english_letters += 1
378
- self.stats.char_count += 1
379
- elif is_digit(ch):
380
- self.stats.digits += 1
381
- self.stats.char_count += 1
382
- elif is_english_punctuation(ch):
383
- self.stats.english_punctuations += 1
384
- self.stats.char_count += 1
385
- elif is_unicode_symbol(ch):
386
- units = utf16_units(ch)
387
- self.stats.symbol_chars += units
388
- self.stats.char_count += units
389
- elif is_chinese_punctuation(ch):
390
- self.stats.chinese_punctuations += 1
391
- self.stats.char_count += 1
392
- elif is_han(ch):
393
- self.stats.han_chars += 1
394
- self.stats.char_count += 1
395
- self._at_boundary = False
396
- return len(token)
397
-
398
- def _match_ascii_compound_token(self, text: str, start: int) -> str | None:
399
- match = URL_TOKEN_RE.match(text, start)
400
- if match:
401
- return match.group(0)
402
-
403
- match = ASCII_COMPOUND_TOKEN_RE.match(text, start)
404
- if not match:
405
- return None
406
- token = match.group(0)
407
- if any(is_ascii_letter(ch) for ch in token):
408
- return token
409
- return None
410
-
411
- def _write_symbol_char(self, ch: str) -> None:
412
- self._end_lexeme()
413
- self._end_symbol_run(count_word=False)
414
- units = utf16_units(ch)
415
- self.stats.symbol_words += 1
416
- self.stats.symbol_chars += units
417
- self.stats.word_count += 1
418
- self.stats.char_count += units
419
- self._at_boundary = False
420
-
421
- def _end_unit(self) -> None:
422
- self._end_lexeme()
423
- self._end_symbol_run(count_word=True)
424
-
425
- def _end_lexeme(self) -> None:
426
- if self._lexeme_kind == "english":
427
- self.stats.english_words += 1
428
- self.stats.word_count += 1
429
- elif self._lexeme_kind == "number":
430
- self.stats.number_words += 1
431
- self.stats.word_count += 1
432
- self._lexeme_kind = None
433
- self._lexeme_has_digit = False
434
-
435
- def _end_symbol_run(self, *, count_word: bool) -> None:
436
- if self._symbol_run_length >= 1 and count_word:
437
- self.stats.symbol_words += 1
438
- self.stats.word_count += 1
439
- if self._symbol_run_length:
440
- self._at_boundary = False
441
- self._symbol_run_length = 0
442
-
443
- # ---------------------------------------------------------------------------
444
- # Markdown parser
445
- # ---------------------------------------------------------------------------
446
-
447
- HEADING_RE = re.compile(r"^(#{1,6})\s+(.*)$")
448
- LIST_RE = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+(.*)$")
449
- QUOTE_RE = re.compile(r"^\s*>\s?(.*)$")
450
- TABLE_SEP_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*$")
451
-
452
-
453
- def parse_markdown(source: str) -> list[Block]:
454
- lines = source.splitlines()
455
- blocks: list[Block] = []
456
- paragraph: list[str] = []
457
- i = 0
458
-
459
- def flush_paragraph() -> None:
460
- if paragraph:
461
- blocks.append(Block(type="paragraph", text_runs=[TextRun(clean_inline(" ".join(paragraph)))]))
462
- paragraph.clear()
463
-
464
- while i < len(lines):
465
- line = lines[i]
466
- stripped = line.strip()
467
- if not stripped:
468
- flush_paragraph()
469
- i += 1
470
- continue
471
-
472
- if stripped.startswith("```") or stripped.startswith("~~~"):
473
- flush_paragraph()
474
- fence = stripped[:3]
475
- code_lines: list[str] = []
476
- i += 1
477
- while i < len(lines) and not lines[i].strip().startswith(fence):
478
- code_lines.append(lines[i])
479
- i += 1
480
- if i < len(lines):
481
- i += 1
482
- blocks.append(Block(type="code", text_runs=[TextRun("\n".join(code_lines))]))
483
- continue
484
-
485
- heading = HEADING_RE.match(line)
486
- if heading:
487
- flush_paragraph()
488
- blocks.append(Block(type="heading", text_runs=[TextRun(clean_inline(heading.group(2)))]))
489
- i += 1
490
- continue
491
-
492
- if _looks_like_table(lines, i):
493
- flush_paragraph()
494
- table, consumed = _parse_table(lines, i)
495
- blocks.append(table)
496
- i += consumed
497
- continue
498
-
499
- item = LIST_RE.match(line)
500
- if item:
501
- flush_paragraph()
502
- items: list[Block] = []
503
- while i < len(lines):
504
- match = LIST_RE.match(lines[i])
505
- if not match:
506
- break
507
- items.append(Block(type="list_item", text_runs=[TextRun(clean_inline(match.group(1)))]))
508
- i += 1
509
- blocks.append(Block(type="list", children=items))
510
- continue
511
-
512
- quote = QUOTE_RE.match(line)
513
- if quote:
514
- flush_paragraph()
515
- quote_lines: list[str] = []
516
- while i < len(lines):
517
- match = QUOTE_RE.match(lines[i])
518
- if not match:
519
- break
520
- quote_lines.append(match.group(1))
521
- i += 1
522
- blocks.append(Block(type="quote", text_runs=[TextRun(clean_inline(" ".join(quote_lines)))]))
523
- continue
524
-
525
- paragraph.append(stripped)
526
- i += 1
527
-
528
- flush_paragraph()
529
- return blocks
530
-
531
-
532
- def clean_inline(text: str) -> str:
533
- text = re.sub(r"!\[([^\]]*)\]\([^)]+\)", r"\1", text)
534
- text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
535
- text = re.sub(r"([*_`~]{1,3})(.*?)\1", r"\2", text)
536
- text = text.replace("\\", "")
537
- return text
538
-
539
-
540
- def _looks_like_table(lines: list[str], i: int) -> bool:
541
- return i + 1 < len(lines) and "|" in lines[i] and TABLE_SEP_RE.match(lines[i + 1]) is not None
542
-
543
-
544
- def _parse_table(lines: list[str], i: int) -> tuple[Block, int]:
545
- rows: list[Block] = []
546
- consumed = 0
547
- while i + consumed < len(lines):
548
- line = lines[i + consumed]
549
- stripped = line.strip()
550
- if not stripped or "|" not in stripped:
551
- break
552
- if consumed == 1 and TABLE_SEP_RE.match(stripped):
553
- consumed += 1
554
- continue
555
- cells = [cell.strip() for cell in stripped.strip("|").split("|")]
556
- row = Block(
557
- type="tr",
558
- children=[
559
- Block(type="table_cell", text_runs=[TextRun(clean_inline(cell))])
560
- for cell in cells
561
- if cell
562
- ],
563
- )
564
- rows.append(row)
565
- consumed += 1
566
- return Block(type="table", children=rows), consumed
567
-
568
- # ---------------------------------------------------------------------------
569
- # XML parser
570
- # ---------------------------------------------------------------------------
571
-
572
- INLINE_TAGS = {
573
- "b",
574
- "strong",
575
- "i",
576
- "em",
577
- "u",
578
- "s",
579
- "del",
580
- "span",
581
- "text",
582
- "plain_text",
583
- "code",
584
- "a",
585
- "link",
586
- "mention",
587
- "mention-doc",
588
- "mention-user",
589
- }
590
-
591
- TYPE_ALIASES = {
592
- "doc": "document",
593
- "document": "document",
594
- "fragment": "fragment",
595
- "p": "paragraph",
596
- "paragraph": "paragraph",
597
- "heading": "heading",
598
- "h1": "heading",
599
- "h2": "heading",
600
- "h3": "heading",
601
- "h4": "heading",
602
- "h5": "heading",
603
- "h6": "heading",
604
- "h7": "heading",
605
- "h8": "heading",
606
- "h9": "heading",
607
- "ul": "list",
608
- "ol": "list",
609
- "li": "list_item",
610
- "task": "task",
611
- "todo": "list_item",
612
- "blockquote": "quote",
613
- "quote": "quote",
614
- "br": "br",
615
- "hr": "hr",
616
- "title": "title",
617
- "checkbox": "checkbox",
618
- "grid": "grid",
619
- "column": "column",
620
- "table": "table",
621
- "colgroup": "colgroup",
622
- "col": "col",
623
- "tr": "tr",
624
- "td": "table_cell",
625
- "th": "table_cell",
626
- "pre": "code",
627
- "code_block": "code",
628
- "callout": "callout",
629
- "figure": "figure",
630
- "toggle": "toggle",
631
- "img": "image",
632
- "source": "source",
633
- "file": "file",
634
- "media": "media",
635
- "latex": "latex",
636
- "cite": "cite",
637
- "bookmark": "bookmark",
638
- "button": "button",
639
- "whiteboard": "whiteboard",
640
- "mermaid": "mermaid",
641
- "plantuml": "plantuml",
642
- "poll": "poll",
643
- "isv": "isv",
644
- "mindnote": "mindnote",
645
- "diagram": "diagram",
646
- "sheet": "sheet",
647
- "bitable": "bitable",
648
- "base-ref": "base_ref",
649
- "base_ref": "base_ref",
650
- "base-refer": "base_ref",
651
- "base_refer": "base_ref",
652
- "synced-reference": "synced_reference",
653
- "synced_reference": "synced_reference",
654
- "synced-source": "synced_source",
655
- "synced_source": "synced_source",
656
- "okr": "okr",
657
- "chat-card": "chat_card",
658
- "chat_card": "chat_card",
659
- "sub_page_list": "sub-page-list",
660
- "sub-page-list": "sub-page-list",
661
- }
662
-
663
- SUBTYPE_ATTR_TAGS = {
664
- "a",
665
- "button",
666
- "cite",
667
- "img",
668
- "sheet",
669
- "source",
670
- "whiteboard",
671
- "base-ref",
672
- "base_ref",
673
- "base-refer",
674
- "base_refer",
675
- "synced-reference",
676
- "synced_reference",
677
- "synced-source",
678
- "synced_source",
679
- "okr",
680
- "chat-card",
681
- "chat_card",
682
- "sub_page_list",
683
- "sub-page-list",
684
- }
685
-
686
- MAX_XML_INPUT_CHARS = 20_000_000
687
- FORBIDDEN_XML_DECL_RE = re.compile(r"<!\s*(?:DOCTYPE|ENTITY)\b", re.IGNORECASE)
688
-
689
-
690
- class UserInputError(ValueError):
691
- pass
692
-
693
-
694
- def local_name(tag: str) -> str:
695
- if "}" in tag:
696
- return tag.rsplit("}", 1)[1]
697
- return tag
698
-
699
-
700
- def block_type_for(elem: ET.Element) -> str:
701
- tag = local_name(elem.tag)
702
- explicit = elem.attrib.get("block_type")
703
- if explicit is None and tag not in SUBTYPE_ATTR_TAGS:
704
- explicit = elem.attrib.get("type")
705
- if explicit:
706
- return TYPE_ALIASES.get(explicit, explicit)
707
- return TYPE_ALIASES.get(tag, tag)
708
-
709
-
710
- def ensure_safe_xml_source(source: str) -> None:
711
- if len(source) > MAX_XML_INPUT_CHARS:
712
- raise UserInputError(
713
- f"XML input is too large ({len(source)} chars, limit {MAX_XML_INPUT_CHARS})"
714
- )
715
- if FORBIDDEN_XML_DECL_RE.search(source):
716
- raise UserInputError("XML input must not contain DOCTYPE or ENTITY declarations")
717
-
718
-
719
- def parse_xml(source: str) -> list[Block]:
720
- source = source.strip()
721
- if not source:
722
- return []
723
- ensure_safe_xml_source(source)
724
- try:
725
- root = ET.fromstring(source)
726
- except ET.ParseError:
727
- # docs +fetch raw output can occasionally include adjacent top-level
728
- # blocks. Wrap them so standard ElementTree can parse the stream.
729
- root = ET.fromstring(f"<fragment>{source}</fragment>")
730
- return [_parse_block(root)]
731
-
732
-
733
- def _parse_block(elem: ET.Element) -> Block:
734
- block = Block(type=block_type_for(elem), attrs=dict(elem.attrib), raw=elem)
735
- _collect_content(elem, block)
736
- if not block.text_runs and not block.children:
737
- if block.type == "image":
738
- display = elem.attrib.get("caption")
739
- else:
740
- display = (
741
- elem.attrib.get("text")
742
- or elem.attrib.get("name")
743
- or elem.attrib.get("title")
744
- or elem.attrib.get("alt")
745
- or elem.attrib.get("caption")
746
- )
747
- if display:
748
- block.text_runs.append(TextRun(display, dict(elem.attrib)))
749
- return block
750
-
751
-
752
- def _collect_content(elem: ET.Element, block: Block) -> None:
753
- if elem.text:
754
- block.text_runs.append(TextRun(elem.text))
755
-
756
- for child in list(elem):
757
- tag = local_name(child.tag)
758
- if tag == "br":
759
- block.text_runs.append(TextRun("\n"))
760
- elif tag in INLINE_TAGS:
761
- _collect_inline(child, block)
762
- else:
763
- block.children.append(_parse_block(child))
764
- if child.tail:
765
- block.text_runs.append(TextRun(child.tail))
766
-
767
-
768
- def _collect_inline(elem: ET.Element, block: Block) -> None:
769
- if local_name(elem.tag) == "br":
770
- block.text_runs.append(TextRun("\n", dict(elem.attrib)))
771
- return
772
-
773
- display = (
774
- elem.attrib.get("text")
775
- or elem.attrib.get("name")
776
- or elem.attrib.get("title")
777
- or elem.attrib.get("alt")
778
- )
779
- if display:
780
- block.text_runs.append(TextRun(display, dict(elem.attrib)))
781
- return
782
-
783
- if elem.text:
784
- block.text_runs.append(TextRun(elem.text, dict(elem.attrib)))
785
- for child in list(elem):
786
- _collect_inline(child, block)
787
- if child.tail:
788
- block.text_runs.append(TextRun(child.tail))
789
-
790
- # ---------------------------------------------------------------------------
791
- # Block extraction registry
792
- # ---------------------------------------------------------------------------
793
-
794
- @dataclass
795
- class ExtractContext:
796
- unknown_blocks: list[UnknownBlock] = field(default_factory=list)
797
- resource_texts: dict[str, str] = field(default_factory=dict)
798
-
799
-
800
- class Handler(Protocol):
801
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
802
- raise NotImplementedError
803
-
804
-
805
- def block_id(block: Block) -> str | None:
806
- for key in ("id", "block_id", "block-id", "token"):
807
- value = block.attrs.get(key)
808
- if isinstance(value, str) and value:
809
- return value
810
- return None
811
-
812
-
813
- def runs_text(block: Block) -> str:
814
- return "".join(run.text for run in block.text_runs)
815
-
816
-
817
- def raw_tag(block: Block) -> str:
818
- tag = getattr(getattr(block, "raw", None), "tag", "") or ""
819
- if "}" in tag:
820
- return tag.rsplit("}", 1)[1]
821
- return tag
822
-
823
-
824
- class TextBlockHandler:
825
- def __init__(self, kind: str = "text") -> None:
826
- self.kind = kind
827
-
828
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
829
- segments: list[Segment] = []
830
- text = runs_text(block)
831
- if text.strip():
832
- segments.append(
833
- Segment(
834
- text=text,
835
- block_type=block.type,
836
- block_id=block_id(block),
837
- kind=self.kind,
838
- )
839
- )
840
- for child in block.children:
841
- segments.extend(registry.extract(child, ctx))
842
- return segments
843
-
844
-
845
- class ContainerHandler:
846
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
847
- segments: list[Segment] = []
848
- text = runs_text(block)
849
- if text.strip():
850
- segments.append(
851
- Segment(text=text, block_type=block.type, block_id=block_id(block), kind="text")
852
- )
853
- for child in block.children:
854
- segments.extend(registry.extract(child, ctx))
855
- return segments
856
-
857
-
858
- class ListHandler:
859
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
860
- tag = raw_tag(block)
861
- if tag not in {"ol", "ul"}:
862
- return ContainerHandler().extract(block, registry, ctx)
863
-
864
- segments: list[Segment] = []
865
- text = runs_text(block)
866
- if text.strip():
867
- segments.append(
868
- Segment(text=text, block_type=block.type, block_id=block_id(block), kind="text")
869
- )
870
-
871
- next_seq = 1
872
- for child in block.children:
873
- if child.type == "list_item":
874
- if tag == "ol":
875
- seq = child.attrs.get("seq")
876
- if isinstance(seq, str) and seq.isdigit():
877
- marker = seq
878
- next_seq = int(seq) + 1
879
- else:
880
- marker = str(next_seq)
881
- next_seq += 1
882
- segments.append(
883
- Segment(
884
- text=f"{marker}.",
885
- block_type="list_marker",
886
- block_id=block_id(child),
887
- kind="text",
888
- )
889
- )
890
- else:
891
- segments.append(
892
- Segment(
893
- text="•",
894
- block_type="list_marker",
895
- block_id=block_id(child),
896
- kind="marker",
897
- )
898
- )
899
- segments.extend(registry.extract(child, ctx))
900
- return segments
901
-
902
-
903
- class CheckboxHandler(TextBlockHandler):
904
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
905
- return [
906
- Segment(
907
- text="☑" if block.attrs.get("done") == "true" else "☐",
908
- block_type="checkbox_marker",
909
- block_id=block_id(block),
910
- kind="marker",
911
- ),
912
- *super().extract(block, registry, ctx),
913
- ]
914
-
915
-
916
- class UnknownHandler:
917
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
918
- ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block)))
919
- return ContainerHandler().extract(block, registry, ctx)
920
-
921
-
922
- class IgnoreHandler:
923
- def __init__(self, action: str = "ignored") -> None:
924
- self.action = action
925
-
926
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
927
- ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action=self.action))
928
- return []
929
-
930
-
931
- class TaskHandler:
932
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
933
- task_id = block.attrs.get("task-id") or block.attrs.get("task_id")
934
- if isinstance(task_id, str) and task_id:
935
- text = ctx.resource_texts.get(f"task:{task_id}")
936
- if text and text.strip():
937
- marker = "☑" if block.attrs.get("status") in {"done", "completed", "complete"} else "☐"
938
- return [
939
- Segment(text=marker, block_type="task_marker", block_id=block_id(block), kind="marker"),
940
- Segment(text=text, block_type="task", block_id=block_id(block), kind="resource_title"),
941
- ]
942
-
943
- ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action="ignored_resource"))
944
- return []
945
-
946
-
947
- class WhiteboardHandler:
948
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
949
- board_type = block.attrs.get("type")
950
- is_empty_resource_shell = not board_type and not block.children and not runs_text(block).strip()
951
- action = (
952
- "ignored_resource"
953
- if board_type in {"blank", "mermaid", "plantuml", "svg"} or is_empty_resource_shell
954
- else "unsupported_resource"
955
- )
956
- ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action=action))
957
- return []
958
-
959
-
960
- class SyncedSourceHandler:
961
- def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
962
- if block.children:
963
- segments: list[Segment] = []
964
- for child in block.children:
965
- segments.extend(registry.extract(child, ctx))
966
- return segments
967
-
968
- ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action="unsupported_resource"))
969
- return []
970
-
971
-
972
- class Registry:
973
- def __init__(self) -> None:
974
- self._handlers: dict[str, Handler] = {}
975
- self._unknown = UnknownHandler()
976
-
977
- def register(self, *types: str, handler: Handler) -> None:
978
- for typ in types:
979
- self._handlers[typ] = handler
980
-
981
- def extract(self, block: Block, ctx: ExtractContext) -> list[Segment]:
982
- handler = self._handlers.get(block.type, self._unknown)
983
- return handler.extract(block, self, ctx)
984
-
985
-
986
- def default_registry() -> Registry:
987
- registry = Registry()
988
- registry.register("document", "fragment", "root", handler=ContainerHandler())
989
- registry.register("title", handler=TextBlockHandler("title"))
990
- registry.register("paragraph", "p", handler=TextBlockHandler("text"))
991
- registry.register(
992
- "heading",
993
- "h",
994
- "h1",
995
- "h2",
996
- "h3",
997
- "h4",
998
- "h5",
999
- "h6",
1000
- "h7",
1001
- "h8",
1002
- "h9",
1003
- handler=TextBlockHandler("heading"),
1004
- )
1005
- registry.register("list", "ul", "ol", handler=ListHandler())
1006
- registry.register("list_item", "li", "todo", handler=TextBlockHandler("list_item"))
1007
- registry.register("checkbox", handler=CheckboxHandler("list_item"))
1008
- registry.register(
1009
- "quote",
1010
- "blockquote",
1011
- "callout",
1012
- "toggle",
1013
- "grid",
1014
- "column",
1015
- "figure",
1016
- handler=ContainerHandler(),
1017
- )
1018
- registry.register("table", "thead", "tbody", "tr", handler=ContainerHandler())
1019
- registry.register("table_cell", "td", "th", handler=TextBlockHandler("table_cell"))
1020
- registry.register("code", "code_block", "pre", handler=TextBlockHandler("code"))
1021
- registry.register("link", "a", "mention", "mention-doc", "mention-user", "time", handler=TextBlockHandler("inline"))
1022
- registry.register("image", "img", handler=TextBlockHandler("caption"))
1023
- registry.register("colgroup", "col", "br", "hr", handler=IgnoreHandler("ignored_structure"))
1024
- registry.register("button", "cite", "latex", "bookmark", handler=IgnoreHandler("ignored_inline"))
1025
- registry.register("task", handler=TaskHandler())
1026
- registry.register("whiteboard", handler=WhiteboardHandler())
1027
- registry.register("synced_source", handler=SyncedSourceHandler())
1028
- registry.register(
1029
- "mermaid",
1030
- "sheet",
1031
- "source",
1032
- "file",
1033
- "media",
1034
- "chat_card",
1035
- "base_ref",
1036
- "bitable",
1037
- "synced_reference",
1038
- "poll",
1039
- "isv",
1040
- "mindnote",
1041
- "diagram",
1042
- "sub-page-list",
1043
- handler=IgnoreHandler("ignored_resource"),
1044
- )
1045
- registry.register(
1046
- "okr",
1047
- "plantuml",
1048
- handler=IgnoreHandler("unsupported_resource"),
1049
- )
1050
- return registry
1051
-
1052
- # ---------------------------------------------------------------------------
1053
- # CLI
1054
- # ---------------------------------------------------------------------------
1055
-
1056
- VERSION = "0.1-alpha"
1057
-
1058
-
1059
- def build_diagnostics(items: list) -> dict[str, object]:
1060
- actions: dict[str, int] = {}
1061
- types: dict[str, int] = {}
1062
- unsupported_types: dict[str, int] = {}
1063
- unknown_types: dict[str, int] = {}
1064
- for item in items:
1065
- actions[item.action] = actions.get(item.action, 0) + 1
1066
- types[item.type] = types.get(item.type, 0) + 1
1067
- if item.action == "unsupported_resource":
1068
- unsupported_types[item.type] = unsupported_types.get(item.type, 0) + 1
1069
- if item.action == "recurse_children":
1070
- unknown_types[item.type] = unknown_types.get(item.type, 0) + 1
1071
- return {
1072
- "actions": actions,
1073
- "types": types,
1074
- "unsupported_types": unsupported_types,
1075
- "unknown_types": unknown_types,
1076
- "has_unsupported": bool(unsupported_types),
1077
- "has_unknown": bool(unknown_types),
1078
- }
1079
-
1080
-
1081
- def read_input(path: str) -> str:
1082
- if path == "-":
1083
- return sys.stdin.read()
1084
- return Path(path).read_text(encoding="utf-8")
1085
-
1086
-
1087
- def read_resource_texts(path: str | None) -> dict[str, str]:
1088
- if not path:
1089
- return {}
1090
- payload = json.loads(Path(path).read_text(encoding="utf-8"))
1091
- if not isinstance(payload, dict):
1092
- raise ValueError("--resource-texts must be a JSON object")
1093
- return {str(key): str(value) for key, value in payload.items()}
1094
-
1095
-
1096
- def extract_lark_json_content(source: str) -> str:
1097
- try:
1098
- envelope = json.loads(source)
1099
- except json.JSONDecodeError as exc:
1100
- raise UserInputError(f"could not parse lark-cli JSON envelope: {exc}") from exc
1101
-
1102
- if not isinstance(envelope, dict):
1103
- raise UserInputError("lark-cli JSON envelope must be an object")
1104
- data = envelope.get("data")
1105
- if not isinstance(data, dict):
1106
- raise UserInputError("lark-cli JSON envelope is missing object field data")
1107
- document = data.get("document")
1108
- if not isinstance(document, dict):
1109
- raise UserInputError("lark-cli JSON envelope is missing object field data.document")
1110
- content = document.get("content")
1111
- if not isinstance(content, str):
1112
- raise UserInputError("lark-cli JSON envelope is missing string field data.document.content")
1113
- return content
1114
-
1115
-
1116
- HELP_EPILOG = """
1117
- Examples:
1118
- Local XML file:
1119
- python3 doc_word_stat.py --protocol xml /absolute/path/doc.xml
1120
-
1121
- Local Markdown file:
1122
- python3 doc_word_stat.py --protocol md /absolute/path/doc.md
1123
-
1124
- Pipe an extracted local file:
1125
- cat /absolute/path/doc.xml | python3 doc_word_stat.py --protocol xml --pretty
1126
-
1127
- Lark CLI XML fetch, JSON envelope output:
1128
- lark-cli docs +fetch --doc "$URL" --doc-format xml --detail full --format json \\
1129
- | python3 doc_word_stat.py --protocol xml --lark-json --pretty
1130
-
1131
- Lark CLI Markdown fetch, raw content output:
1132
- lark-cli docs +fetch --doc "$URL" --doc-format markdown \\
1133
- | python3 doc_word_stat.py --protocol md
1134
-
1135
- Strict integration for agents or automation:
1136
- lark-cli docs +fetch --doc "$URL" --doc-format xml --detail full --format json \\
1137
- | python3 doc_word_stat.py --protocol xml --lark-json --fail-on-unsupported --fail-on-unknown
1138
- """
1139
-
1140
-
1141
- def parse_args() -> argparse.Namespace:
1142
- parser = argparse.ArgumentParser(
1143
- description="Count semantic words and visible characters in Lark Docs XML or Markdown.",
1144
- epilog=HELP_EPILOG,
1145
- formatter_class=argparse.RawDescriptionHelpFormatter,
1146
- )
1147
- parser.add_argument(
1148
- "--version",
1149
- action="version",
1150
- version=f"%(prog)s {VERSION}",
1151
- )
1152
- parser.add_argument(
1153
- "input",
1154
- nargs="?",
1155
- default="-",
1156
- help="input file path, or '-' / omitted for stdin",
1157
- )
1158
- parser.add_argument(
1159
- "--protocol",
1160
- choices=("xml", "md"),
1161
- required=True,
1162
- help="input protocol produced by docs +fetch",
1163
- )
1164
- parser.add_argument(
1165
- "--pretty",
1166
- action="store_true",
1167
- help="pretty-print JSON output",
1168
- )
1169
- parser.add_argument(
1170
- "--segments",
1171
- action="store_true",
1172
- help="include extracted text segments for debugging",
1173
- )
1174
- parser.add_argument(
1175
- "--lark-json",
1176
- action="store_true",
1177
- help="read lark-cli docs +fetch JSON and count data.document.content",
1178
- )
1179
- parser.add_argument(
1180
- "--resource-texts",
1181
- help='optional JSON object mapping resource keys to visible text, e.g. {"task:<task-id>": "title"}',
1182
- )
1183
- parser.add_argument(
1184
- "--fail-on-unsupported",
1185
- action="store_true",
1186
- help="exit with code 2 when unsupported_blocks is non-empty",
1187
- )
1188
- parser.add_argument(
1189
- "--fail-on-unknown",
1190
- action="store_true",
1191
- help="exit with code 3 when unknown XML/Markdown block types are encountered",
1192
- )
1193
- return parser.parse_args()
1194
-
1195
-
1196
- def main() -> int:
1197
- args = parse_args()
1198
- source = read_input(args.input)
1199
- if args.lark_json:
1200
- try:
1201
- source = extract_lark_json_content(source)
1202
- except UserInputError as exc:
1203
- print(f"error: {exc}", file=sys.stderr)
1204
- return 1
1205
-
1206
- if args.protocol == "xml":
1207
- try:
1208
- blocks = parse_xml(source)
1209
- except (ET.ParseError, UserInputError) as exc:
1210
- print(f"error: could not parse XML input: {exc}", file=sys.stderr)
1211
- return 1
1212
- else:
1213
- blocks = parse_markdown(source)
1214
-
1215
- ctx = ExtractContext(resource_texts=read_resource_texts(args.resource_texts))
1216
- registry = default_registry()
1217
- segments = []
1218
- for block in blocks:
1219
- segments.extend(registry.extract(block, ctx))
1220
-
1221
- stats = Counter().count_segments(segments)
1222
- payload = stats.to_dict()
1223
- payload["protocol"] = args.protocol
1224
- payload["unknown_blocks"] = [item.to_dict() for item in ctx.unknown_blocks]
1225
- payload["unsupported_blocks"] = [
1226
- item.to_dict() for item in ctx.unknown_blocks if item.action == "unsupported_resource"
1227
- ]
1228
- payload["diagnostics"] = build_diagnostics(ctx.unknown_blocks)
1229
- if args.segments:
1230
- payload["segments"] = [segment.to_dict() for segment in segments]
1231
-
1232
- indent = 2 if args.pretty else None
1233
- print(json.dumps(payload, ensure_ascii=False, indent=indent, sort_keys=args.pretty))
1234
- if args.fail_on_unsupported and payload["unsupported_blocks"]:
1235
- return 2
1236
- if args.fail_on_unknown and payload["diagnostics"]["has_unknown"]:
1237
- return 3
1238
- return 0
1239
-
1240
-
1241
- if __name__ == "__main__":
1242
- raise SystemExit(main())
1243
-