@amaster.ai/pi-lark 0.1.7 → 0.1.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/skills/lark-apps/SKILL.md +39 -6
- package/skills/lark-apps/references/lark-apps-cloud-dev.md +5 -4
- package/skills/lark-apps/references/lark-apps-create.md +6 -3
- package/skills/lark-apps/references/lark-apps-get.md +1 -1
- package/skills/lark-apps/references/lark-apps-list.md +1 -1
- package/skills/lark-apps/references/lark-apps-local-dev.md +27 -1
- package/skills/lark-apps/references/lark-apps-release-create.md +1 -1
- package/skills/lark-base/SKILL.md +14 -6
- package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +17 -1
- package/skills/lark-base/references/lark-base-dashboard.md +17 -4
- package/skills/lark-base/references/lark-base-data-query-guide.md +8 -0
- package/skills/lark-base/references/lark-base-field-create.md +19 -8
- package/skills/lark-base/references/lark-base-field-json.md +5 -2
- package/skills/lark-calendar/SKILL.md +1 -1
- package/skills/lark-doc/SKILL.md +26 -61
- package/skills/lark-doc/references/genres/business-analysis.md +30 -0
- package/skills/lark-doc/references/genres/data-report.md +32 -0
- package/skills/lark-doc/references/genres/email.md +38 -0
- package/skills/lark-doc/references/genres/execution-plan.md +27 -0
- package/skills/lark-doc/references/genres/formal-doc.md +37 -0
- package/skills/lark-doc/references/genres/meeting-minutes.md +24 -0
- package/skills/lark-doc/references/genres/memo-brief.md +25 -0
- package/skills/lark-doc/references/genres/official-redhead.md +73 -0
- package/skills/lark-doc/references/genres/prd.md +26 -0
- package/skills/lark-doc/references/genres/proposal.md +24 -0
- package/skills/lark-doc/references/genres/research-report.md +32 -0
- package/skills/lark-doc/references/genres/retrospective.md +25 -0
- package/skills/lark-doc/references/genres/route-consumer.md +37 -0
- package/skills/lark-doc/references/genres/route-creative.md +36 -0
- package/skills/lark-doc/references/genres/route-knowledge.md +39 -0
- package/skills/lark-doc/references/genres/route-marketing.md +40 -0
- package/skills/lark-doc/references/genres/route-media.md +36 -0
- package/skills/lark-doc/references/genres/route-opinion.md +38 -0
- package/skills/lark-doc/references/genres/route-personal-brand.md +36 -0
- package/skills/lark-doc/references/genres/route-platform.md +9 -0
- package/skills/lark-doc/references/genres/route-report.md +10 -0
- package/skills/lark-doc/references/genres/route-workplace.md +17 -0
- package/skills/lark-doc/references/genres/sop-tutorial.md +41 -0
- package/skills/lark-doc/references/genres/technical-doc.md +39 -0
- package/skills/lark-doc/references/genres/wechat.md +39 -0
- package/skills/lark-doc/references/genres/weekly-report.md +24 -0
- package/skills/lark-doc/references/genres/white-paper.md +32 -0
- package/skills/lark-doc/references/genres/xiaohongshu.md +38 -0
- package/skills/lark-doc/references/lark-doc-create-workflow.md +121 -0
- package/skills/lark-doc/references/lark-doc-create.md +22 -48
- package/skills/lark-doc/references/lark-doc-fetch.md +75 -92
- package/skills/lark-doc/references/lark-doc-history.md +16 -15
- package/skills/lark-doc/references/lark-doc-md.md +5 -1
- package/skills/lark-doc/references/lark-doc-media-download.md +2 -1
- package/skills/lark-doc/references/lark-doc-script.md +76 -0
- package/skills/lark-doc/references/lark-doc-update.md +70 -222
- package/skills/lark-doc/references/lark-doc-whiteboard.md +5 -9
- package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +17 -12
- package/skills/lark-doc/references/lark-doc-xml.md +38 -167
- package/skills/lark-drive/SKILL.md +7 -5
- package/skills/lark-drive/references/lark-drive-apply-permission.md +1 -1
- package/skills/lark-drive/references/lark-drive-copy.md +87 -0
- package/skills/lark-drive/references/lark-drive-download.md +2 -1
- package/skills/lark-drive/references/lark-drive-export.md +3 -0
- package/skills/lark-drive/references/lark-drive-task-result.md +3 -0
- package/skills/lark-drive/references/lark-drive-update-title.md +78 -0
- package/skills/lark-event/SKILL.md +7 -4
- package/skills/lark-event/references/lark-event-vc.md +8 -2
- package/skills/lark-im/SKILL.md +8 -8
- package/skills/lark-im/references/lark-im-chat-list.md +9 -2
- package/skills/lark-im/references/lark-im-chat-members-list.md +7 -4
- package/skills/lark-im/references/lark-im-chat-messages-list.md +10 -3
- package/skills/lark-im/references/lark-im-chat-search.md +9 -2
- package/skills/lark-im/references/lark-im-feed-group-list-item.md +2 -2
- package/skills/lark-im/references/lark-im-feed-group-list.md +2 -2
- package/skills/lark-im/references/lark-im-feed-shortcut-list.md +1 -1
- package/skills/lark-im/references/lark-im-flag-list.md +2 -2
- package/skills/lark-im/references/lark-im-message-enrichment.md +1 -1
- package/skills/lark-im/references/lark-im-messages-resources-download.md +19 -25
- package/skills/lark-im/references/lark-im-messages-search.md +4 -5
- package/skills/lark-im/references/lark-im-threads-messages-list.md +8 -4
- package/skills/lark-mail/references/lark-mail-triage.md +19 -4
- package/skills/lark-minutes/SKILL.md +1 -1
- package/skills/lark-minutes/references/lark-minutes-search.md +6 -7
- package/skills/lark-shared/SKILL.md +3 -3
- package/skills/lark-sheets/SKILL.md +83 -82
- package/skills/lark-sheets/references/lark-sheets-batch-update.md +13 -58
- package/skills/lark-sheets/references/lark-sheets-chart.md +2 -1
- package/skills/lark-sheets/references/lark-sheets-conditional-format.md +1 -1
- package/skills/lark-sheets/references/lark-sheets-range-operations.md +5 -5
- package/skills/lark-sheets/references/lark-sheets-read-data.md +80 -6
- package/skills/lark-sheets/references/lark-sheets-sheet-structure.md +21 -10
- package/skills/lark-sheets/references/lark-sheets-styles-put.md +93 -0
- package/skills/lark-sheets/references/lark-sheets-visual-standards.md +2 -2
- package/skills/lark-sheets/references/lark-sheets-workbook.md +4 -3
- package/skills/lark-sheets/references/lark-sheets-write-cells.md +40 -12
- package/skills/lark-sheets/scripts/lark_detect_subtables.py +593 -0
- package/skills/lark-sheets/scripts/lark_inspect_workbook.py +188 -0
- package/skills/lark-sheets/scripts/lark_profile_table.py +614 -0
- package/skills/lark-sheets/scripts/lark_sheet_range.py +176 -0
- package/skills/lark-sheets/scripts/lark_sheet_read_cli.py +184 -0
- package/skills/lark-sheets/scripts/sheets_df.py +21 -3
- package/skills/lark-slides/SKILL.md +27 -44
- package/skills/lark-slides/references/lark-slides-add-slide.md +92 -0
- package/skills/lark-slides/references/lark-slides-create.md +77 -65
- package/skills/lark-slides/references/lark-slides-delete-slide.md +65 -0
- package/skills/lark-slides/references/lark-slides-edit-workflows.md +6 -7
- package/skills/lark-slides/references/lark-slides-media-upload.md +3 -25
- package/skills/lark-slides/references/lark-slides-replace-slide.md +22 -1
- package/skills/lark-slides/references/lark-slides-screenshot.md +31 -13
- package/skills/lark-slides/references/lark-slides-update-slide.md +146 -0
- package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +31 -8
- package/skills/lark-slides/references/slides_chart_demo.xml +1 -2
- package/skills/lark-slides/references/slides_xml_schema_definition.xml +48 -4
- package/skills/lark-slides/references/troubleshooting.md +7 -8
- package/skills/lark-slides/references/validation-checklist.md +4 -4
- package/skills/lark-slides/references/xml-schema-quick-ref.md +23 -11
- package/skills/lark-slides/scripts/sxsd_validator.py +154 -10
- package/skills/lark-slides/scripts/xml_text_overlap_lint.py +360 -76
- package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +1138 -214
- package/skills/lark-whiteboard/SKILL.md +15 -8
- package/skills/lark-whiteboard/references/lark-whiteboard-export.md +4 -3
- package/skills/lark-whiteboard/references/lark-whiteboard-update.md +4 -4
- package/skills/lark-whiteboard/references/lark-whiteboard-workflow.md +19 -17
- package/skills/lark-whiteboard/routes/dsl.md +8 -2
- package/skills/lark-whiteboard/routes/mermaid.md +1 -1
- package/skills/lark-whiteboard/routes/svg-edit.md +5 -2
- package/skills/lark-whiteboard/routes/svg.md +3 -1
- package/skills/lark-whiteboard/scenes/mention.md +71 -0
- package/skills/lark-wiki/SKILL.md +5 -3
- package/skills/lark-wiki/references/lark-wiki-delete-space.md +6 -3
- package/skills/lark-doc/references/lark-doc-word-stat.md +0 -93
- package/skills/lark-doc/references/style/lark-doc-create-workflow.md +0 -47
- package/skills/lark-doc/references/style/lark-doc-style.md +0 -68
- package/skills/lark-doc/references/style/lark-doc-update-workflow.md +0 -48
- package/skills/lark-doc/scripts/doc_word_stat.py +0 -1243
- package/skills/lark-slides/references/lark-slides-replace-pages.md +0 -95
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-create.md +0 -219
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +0 -126
|
@@ -1,1243 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
# Copyright (c) 2026 Lark Technologies Pte. Ltd.
|
|
3
|
-
# SPDX-License-Identifier: MIT
|
|
4
|
-
"""Standalone Lark Docs word and character counter for XML or Markdown input."""
|
|
5
|
-
|
|
6
|
-
from __future__ import annotations
|
|
7
|
-
|
|
8
|
-
import argparse
|
|
9
|
-
import json
|
|
10
|
-
import re
|
|
11
|
-
import sys
|
|
12
|
-
import unicodedata
|
|
13
|
-
from dataclasses import dataclass, field
|
|
14
|
-
from pathlib import Path
|
|
15
|
-
from typing import Any, Literal, Protocol
|
|
16
|
-
from xml.etree import ElementTree as ET
|
|
17
|
-
|
|
18
|
-
# ---------------------------------------------------------------------------
|
|
19
|
-
# Data model
|
|
20
|
-
# ---------------------------------------------------------------------------
|
|
21
|
-
|
|
22
|
-
@dataclass
|
|
23
|
-
class TextRun:
|
|
24
|
-
text: str
|
|
25
|
-
attrs: dict[str, Any] = field(default_factory=dict)
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
@dataclass
|
|
29
|
-
class Block:
|
|
30
|
-
type: str
|
|
31
|
-
attrs: dict[str, Any] = field(default_factory=dict)
|
|
32
|
-
children: list["Block"] = field(default_factory=list)
|
|
33
|
-
text_runs: list[TextRun] = field(default_factory=list)
|
|
34
|
-
raw: Any = None
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
@dataclass
|
|
38
|
-
class Segment:
|
|
39
|
-
text: str
|
|
40
|
-
block_type: str
|
|
41
|
-
block_id: str | None = None
|
|
42
|
-
kind: str = "text"
|
|
43
|
-
boundary_before: bool = True
|
|
44
|
-
boundary_after: bool = True
|
|
45
|
-
|
|
46
|
-
def to_dict(self) -> dict[str, Any]:
|
|
47
|
-
return {
|
|
48
|
-
"text": self.text,
|
|
49
|
-
"block_type": self.block_type,
|
|
50
|
-
"block_id": self.block_id,
|
|
51
|
-
"kind": self.kind,
|
|
52
|
-
"boundary_before": self.boundary_before,
|
|
53
|
-
"boundary_after": self.boundary_after,
|
|
54
|
-
}
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
@dataclass(frozen=True)
|
|
58
|
-
class UnknownBlock:
|
|
59
|
-
type: str
|
|
60
|
-
block_id: str | None = None
|
|
61
|
-
action: str = "recurse_children"
|
|
62
|
-
|
|
63
|
-
def to_dict(self) -> dict[str, str | None]:
|
|
64
|
-
return {
|
|
65
|
-
"type": self.type,
|
|
66
|
-
"block_id": self.block_id,
|
|
67
|
-
"action": self.action,
|
|
68
|
-
}
|
|
69
|
-
|
|
70
|
-
# ---------------------------------------------------------------------------
|
|
71
|
-
# Counting rules
|
|
72
|
-
# ---------------------------------------------------------------------------
|
|
73
|
-
|
|
74
|
-
CHINESE_PUNCTUATION = set(",。!?;:、()《》〈〉“”‘’【】「」『』〔〕…—~·¥")
|
|
75
|
-
ENGLISH_PUNCTUATION = set(
|
|
76
|
-
r"""!"#$%&'()*+,-./:;<=>?@[\]^_`{|}~"""
|
|
77
|
-
)
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
LexemeKind = Literal["english", "number"]
|
|
81
|
-
URL_TOKEN_RE = re.compile(r"https?://[!-~]+")
|
|
82
|
-
ASCII_COMPOUND_TOKEN_RE = re.compile(
|
|
83
|
-
r"[A-Za-z0-9]+(?:[._/@:-][A-Za-z0-9]+)+"
|
|
84
|
-
)
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
@dataclass
|
|
88
|
-
class Stats:
|
|
89
|
-
word_count: int = 0
|
|
90
|
-
char_count: int = 0
|
|
91
|
-
han_chars: int = 0
|
|
92
|
-
english_words: int = 0
|
|
93
|
-
number_words: int = 0
|
|
94
|
-
chinese_punctuations: int = 0
|
|
95
|
-
english_letters: int = 0
|
|
96
|
-
digits: int = 0
|
|
97
|
-
english_punctuations: int = 0
|
|
98
|
-
symbol_words: int = 0
|
|
99
|
-
symbol_chars: int = 0
|
|
100
|
-
|
|
101
|
-
def to_dict(self) -> dict[str, object]:
|
|
102
|
-
return {
|
|
103
|
-
"word_count": self.word_count,
|
|
104
|
-
"char_count": self.char_count,
|
|
105
|
-
"breakdown": {
|
|
106
|
-
"han_chars": self.han_chars,
|
|
107
|
-
"english_words": self.english_words,
|
|
108
|
-
"number_words": self.number_words,
|
|
109
|
-
"chinese_punctuations": self.chinese_punctuations,
|
|
110
|
-
"english_letters": self.english_letters,
|
|
111
|
-
"digits": self.digits,
|
|
112
|
-
"english_punctuations": self.english_punctuations,
|
|
113
|
-
"symbol_words": self.symbol_words,
|
|
114
|
-
"symbol_chars": self.symbol_chars,
|
|
115
|
-
},
|
|
116
|
-
}
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
def is_han(ch: str) -> bool:
|
|
120
|
-
code = ord(ch)
|
|
121
|
-
return (
|
|
122
|
-
0x3400 <= code <= 0x4DBF
|
|
123
|
-
or 0x4E00 <= code <= 0x9FFF
|
|
124
|
-
or 0xF900 <= code <= 0xFAFF
|
|
125
|
-
or 0x20000 <= code <= 0x2A6DF
|
|
126
|
-
or 0x2A700 <= code <= 0x2B73F
|
|
127
|
-
or 0x2B740 <= code <= 0x2B81F
|
|
128
|
-
or 0x2B820 <= code <= 0x2CEAF
|
|
129
|
-
or 0x30000 <= code <= 0x3134F
|
|
130
|
-
)
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
def is_ascii_letter(ch: str) -> bool:
|
|
134
|
-
return ("a" <= ch <= "z") or ("A" <= ch <= "Z")
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
def is_digit(ch: str) -> bool:
|
|
138
|
-
return "0" <= ch <= "9"
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
def is_chinese_punctuation(ch: str) -> bool:
|
|
142
|
-
if ch in CHINESE_PUNCTUATION:
|
|
143
|
-
return True
|
|
144
|
-
return unicodedata.category(ch).startswith("P") and unicodedata.east_asian_width(ch) in {
|
|
145
|
-
"W",
|
|
146
|
-
"F",
|
|
147
|
-
}
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
def is_english_punctuation(ch: str) -> bool:
|
|
151
|
-
return ch in ENGLISH_PUNCTUATION
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
def is_unicode_symbol(ch: str) -> bool:
|
|
155
|
-
return unicodedata.category(ch).startswith("S")
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
def utf16_units(ch: str) -> int:
|
|
159
|
-
return len(ch.encode("utf-16-le")) // 2
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
class Counter:
|
|
163
|
-
def __init__(self) -> None:
|
|
164
|
-
self.stats = Stats()
|
|
165
|
-
self._lexeme_kind: LexemeKind | None = None
|
|
166
|
-
self._lexeme_has_digit = False
|
|
167
|
-
self._symbol_run_length = 0
|
|
168
|
-
self._at_boundary = True
|
|
169
|
-
|
|
170
|
-
def count_segments(self, segments: list[Segment]) -> Stats:
|
|
171
|
-
for segment in segments:
|
|
172
|
-
if segment.boundary_before:
|
|
173
|
-
self._end_unit()
|
|
174
|
-
self._at_boundary = True
|
|
175
|
-
if segment.kind == "marker":
|
|
176
|
-
self.write_marker(segment.text)
|
|
177
|
-
elif segment.kind == "code":
|
|
178
|
-
self.write_code(segment.text)
|
|
179
|
-
else:
|
|
180
|
-
self.write(segment.text)
|
|
181
|
-
if segment.boundary_after:
|
|
182
|
-
self._end_unit()
|
|
183
|
-
self._at_boundary = True
|
|
184
|
-
self._end_unit()
|
|
185
|
-
return self.stats
|
|
186
|
-
|
|
187
|
-
def write(self, text: str) -> None:
|
|
188
|
-
i = 0
|
|
189
|
-
while i < len(text):
|
|
190
|
-
consumed = self._write_ascii_compound_token(text, i)
|
|
191
|
-
if consumed:
|
|
192
|
-
i += consumed
|
|
193
|
-
continue
|
|
194
|
-
if self._write_visible_ascii_separator(text, i):
|
|
195
|
-
i += 1
|
|
196
|
-
continue
|
|
197
|
-
self._write_char(text[i])
|
|
198
|
-
i += 1
|
|
199
|
-
|
|
200
|
-
def write_marker(self, text: str) -> None:
|
|
201
|
-
for ch in text:
|
|
202
|
-
if ch.isspace():
|
|
203
|
-
continue
|
|
204
|
-
self._end_unit()
|
|
205
|
-
self.stats.word_count += 1
|
|
206
|
-
self.stats.char_count += 1
|
|
207
|
-
self._at_boundary = False
|
|
208
|
-
|
|
209
|
-
def write_code(self, text: str) -> None:
|
|
210
|
-
for ch in text:
|
|
211
|
-
self._write_code_char(ch)
|
|
212
|
-
|
|
213
|
-
def _write_code_char(self, ch: str) -> None:
|
|
214
|
-
if ch.isspace():
|
|
215
|
-
self._end_unit()
|
|
216
|
-
self._at_boundary = True
|
|
217
|
-
return
|
|
218
|
-
|
|
219
|
-
if is_han(ch):
|
|
220
|
-
self._end_lexeme()
|
|
221
|
-
self._end_symbol_run(count_word=False)
|
|
222
|
-
self.stats.han_chars += 1
|
|
223
|
-
self.stats.word_count += 1
|
|
224
|
-
self.stats.char_count += 1
|
|
225
|
-
self._at_boundary = False
|
|
226
|
-
return
|
|
227
|
-
|
|
228
|
-
if is_ascii_letter(ch):
|
|
229
|
-
self._end_symbol_run(count_word=False)
|
|
230
|
-
self.stats.english_letters += 1
|
|
231
|
-
self.stats.char_count += 1
|
|
232
|
-
if self._lexeme_kind is None:
|
|
233
|
-
self._lexeme_kind = "english"
|
|
234
|
-
elif self._lexeme_kind == "number":
|
|
235
|
-
self._lexeme_kind = "english"
|
|
236
|
-
self._at_boundary = False
|
|
237
|
-
return
|
|
238
|
-
|
|
239
|
-
if is_digit(ch):
|
|
240
|
-
self._end_symbol_run(count_word=False)
|
|
241
|
-
self.stats.digits += 1
|
|
242
|
-
self.stats.char_count += 1
|
|
243
|
-
self._at_boundary = False
|
|
244
|
-
return
|
|
245
|
-
|
|
246
|
-
if is_chinese_punctuation(ch):
|
|
247
|
-
self._end_lexeme()
|
|
248
|
-
self._end_symbol_run(count_word=False)
|
|
249
|
-
self.stats.chinese_punctuations += 1
|
|
250
|
-
self.stats.word_count += 1
|
|
251
|
-
self.stats.char_count += 1
|
|
252
|
-
self._at_boundary = False
|
|
253
|
-
return
|
|
254
|
-
|
|
255
|
-
if is_english_punctuation(ch):
|
|
256
|
-
keeps_lexeme = self._lexeme_kind == "english" and ch in {"'", "-"}
|
|
257
|
-
if not keeps_lexeme:
|
|
258
|
-
had_lexeme = self._lexeme_kind is not None
|
|
259
|
-
self._end_lexeme()
|
|
260
|
-
if not had_lexeme and (self._symbol_run_length > 0 or self._at_boundary):
|
|
261
|
-
self._symbol_run_length += 1
|
|
262
|
-
self.stats.english_punctuations += 1
|
|
263
|
-
self.stats.char_count += 1
|
|
264
|
-
if keeps_lexeme:
|
|
265
|
-
self._at_boundary = False
|
|
266
|
-
return
|
|
267
|
-
|
|
268
|
-
if is_unicode_symbol(ch):
|
|
269
|
-
self._write_symbol_char(ch)
|
|
270
|
-
return
|
|
271
|
-
|
|
272
|
-
self._end_lexeme()
|
|
273
|
-
self._end_symbol_run(count_word=False)
|
|
274
|
-
self._at_boundary = False
|
|
275
|
-
|
|
276
|
-
def _write_char(self, ch: str) -> None:
|
|
277
|
-
if ch.isspace():
|
|
278
|
-
self._end_unit()
|
|
279
|
-
self._at_boundary = True
|
|
280
|
-
return
|
|
281
|
-
|
|
282
|
-
if is_han(ch):
|
|
283
|
-
self._end_lexeme()
|
|
284
|
-
self._end_symbol_run(count_word=False)
|
|
285
|
-
self.stats.han_chars += 1
|
|
286
|
-
self.stats.word_count += 1
|
|
287
|
-
self.stats.char_count += 1
|
|
288
|
-
self._at_boundary = False
|
|
289
|
-
return
|
|
290
|
-
|
|
291
|
-
if is_ascii_letter(ch):
|
|
292
|
-
self._end_symbol_run(count_word=False)
|
|
293
|
-
self.stats.english_letters += 1
|
|
294
|
-
self.stats.char_count += 1
|
|
295
|
-
if self._lexeme_kind is None:
|
|
296
|
-
self._lexeme_kind = "english"
|
|
297
|
-
elif self._lexeme_kind == "number":
|
|
298
|
-
self._lexeme_kind = "english"
|
|
299
|
-
self._at_boundary = False
|
|
300
|
-
return
|
|
301
|
-
|
|
302
|
-
if is_digit(ch):
|
|
303
|
-
self._end_symbol_run(count_word=False)
|
|
304
|
-
self.stats.digits += 1
|
|
305
|
-
self.stats.char_count += 1
|
|
306
|
-
self._lexeme_has_digit = True
|
|
307
|
-
if self._lexeme_kind is None:
|
|
308
|
-
self._lexeme_kind = "number"
|
|
309
|
-
self._at_boundary = False
|
|
310
|
-
return
|
|
311
|
-
|
|
312
|
-
if is_chinese_punctuation(ch):
|
|
313
|
-
self._end_lexeme()
|
|
314
|
-
self._end_symbol_run(count_word=False)
|
|
315
|
-
self.stats.chinese_punctuations += 1
|
|
316
|
-
self.stats.word_count += 1
|
|
317
|
-
self.stats.char_count += 1
|
|
318
|
-
self._at_boundary = False
|
|
319
|
-
return
|
|
320
|
-
|
|
321
|
-
if is_english_punctuation(ch):
|
|
322
|
-
# Apostrophes/hyphens can connect English runs. Dot/comma/hyphen
|
|
323
|
-
# can format numeric runs such as 3.14, 1,000, 2026-06-30, or
|
|
324
|
-
# 7-9. Alphanumeric versions like v1.2.3 should remain one semantic
|
|
325
|
-
# run too. These punctuations still count as characters.
|
|
326
|
-
keeps_lexeme = (
|
|
327
|
-
self._lexeme_kind == "english"
|
|
328
|
-
and (ch in {"'", "-"} or (self._lexeme_has_digit and ch == "."))
|
|
329
|
-
) or (
|
|
330
|
-
self._lexeme_kind == "number"
|
|
331
|
-
and ch in {".", ",", "-"}
|
|
332
|
-
)
|
|
333
|
-
if not keeps_lexeme:
|
|
334
|
-
had_lexeme = self._lexeme_kind is not None
|
|
335
|
-
self._end_lexeme()
|
|
336
|
-
if not had_lexeme and (self._symbol_run_length > 0 or self._at_boundary):
|
|
337
|
-
self._symbol_run_length += 1
|
|
338
|
-
self.stats.english_punctuations += 1
|
|
339
|
-
self.stats.char_count += 1
|
|
340
|
-
if keeps_lexeme:
|
|
341
|
-
self._at_boundary = False
|
|
342
|
-
return
|
|
343
|
-
|
|
344
|
-
if is_unicode_symbol(ch):
|
|
345
|
-
self._write_symbol_char(ch)
|
|
346
|
-
return
|
|
347
|
-
|
|
348
|
-
self._end_lexeme()
|
|
349
|
-
self._end_symbol_run(count_word=False)
|
|
350
|
-
self._at_boundary = False
|
|
351
|
-
|
|
352
|
-
def _write_visible_ascii_separator(self, text: str, index: int) -> bool:
|
|
353
|
-
ch = text[index]
|
|
354
|
-
if ch != "/" or index == 0 or index + 1 >= len(text):
|
|
355
|
-
return False
|
|
356
|
-
if not is_han(text[index - 1]) or not is_han(text[index + 1]):
|
|
357
|
-
return False
|
|
358
|
-
|
|
359
|
-
self._end_unit()
|
|
360
|
-
self.stats.english_punctuations += 1
|
|
361
|
-
self.stats.symbol_words += 1
|
|
362
|
-
self.stats.word_count += 1
|
|
363
|
-
self.stats.char_count += 1
|
|
364
|
-
self._at_boundary = False
|
|
365
|
-
return True
|
|
366
|
-
|
|
367
|
-
def _write_ascii_compound_token(self, text: str, start: int) -> int:
|
|
368
|
-
token = self._match_ascii_compound_token(text, start)
|
|
369
|
-
if not token:
|
|
370
|
-
return 0
|
|
371
|
-
|
|
372
|
-
self._end_unit()
|
|
373
|
-
self.stats.english_words += 1
|
|
374
|
-
self.stats.word_count += 1
|
|
375
|
-
for ch in token:
|
|
376
|
-
if is_ascii_letter(ch):
|
|
377
|
-
self.stats.english_letters += 1
|
|
378
|
-
self.stats.char_count += 1
|
|
379
|
-
elif is_digit(ch):
|
|
380
|
-
self.stats.digits += 1
|
|
381
|
-
self.stats.char_count += 1
|
|
382
|
-
elif is_english_punctuation(ch):
|
|
383
|
-
self.stats.english_punctuations += 1
|
|
384
|
-
self.stats.char_count += 1
|
|
385
|
-
elif is_unicode_symbol(ch):
|
|
386
|
-
units = utf16_units(ch)
|
|
387
|
-
self.stats.symbol_chars += units
|
|
388
|
-
self.stats.char_count += units
|
|
389
|
-
elif is_chinese_punctuation(ch):
|
|
390
|
-
self.stats.chinese_punctuations += 1
|
|
391
|
-
self.stats.char_count += 1
|
|
392
|
-
elif is_han(ch):
|
|
393
|
-
self.stats.han_chars += 1
|
|
394
|
-
self.stats.char_count += 1
|
|
395
|
-
self._at_boundary = False
|
|
396
|
-
return len(token)
|
|
397
|
-
|
|
398
|
-
def _match_ascii_compound_token(self, text: str, start: int) -> str | None:
|
|
399
|
-
match = URL_TOKEN_RE.match(text, start)
|
|
400
|
-
if match:
|
|
401
|
-
return match.group(0)
|
|
402
|
-
|
|
403
|
-
match = ASCII_COMPOUND_TOKEN_RE.match(text, start)
|
|
404
|
-
if not match:
|
|
405
|
-
return None
|
|
406
|
-
token = match.group(0)
|
|
407
|
-
if any(is_ascii_letter(ch) for ch in token):
|
|
408
|
-
return token
|
|
409
|
-
return None
|
|
410
|
-
|
|
411
|
-
def _write_symbol_char(self, ch: str) -> None:
|
|
412
|
-
self._end_lexeme()
|
|
413
|
-
self._end_symbol_run(count_word=False)
|
|
414
|
-
units = utf16_units(ch)
|
|
415
|
-
self.stats.symbol_words += 1
|
|
416
|
-
self.stats.symbol_chars += units
|
|
417
|
-
self.stats.word_count += 1
|
|
418
|
-
self.stats.char_count += units
|
|
419
|
-
self._at_boundary = False
|
|
420
|
-
|
|
421
|
-
def _end_unit(self) -> None:
|
|
422
|
-
self._end_lexeme()
|
|
423
|
-
self._end_symbol_run(count_word=True)
|
|
424
|
-
|
|
425
|
-
def _end_lexeme(self) -> None:
|
|
426
|
-
if self._lexeme_kind == "english":
|
|
427
|
-
self.stats.english_words += 1
|
|
428
|
-
self.stats.word_count += 1
|
|
429
|
-
elif self._lexeme_kind == "number":
|
|
430
|
-
self.stats.number_words += 1
|
|
431
|
-
self.stats.word_count += 1
|
|
432
|
-
self._lexeme_kind = None
|
|
433
|
-
self._lexeme_has_digit = False
|
|
434
|
-
|
|
435
|
-
def _end_symbol_run(self, *, count_word: bool) -> None:
|
|
436
|
-
if self._symbol_run_length >= 1 and count_word:
|
|
437
|
-
self.stats.symbol_words += 1
|
|
438
|
-
self.stats.word_count += 1
|
|
439
|
-
if self._symbol_run_length:
|
|
440
|
-
self._at_boundary = False
|
|
441
|
-
self._symbol_run_length = 0
|
|
442
|
-
|
|
443
|
-
# ---------------------------------------------------------------------------
|
|
444
|
-
# Markdown parser
|
|
445
|
-
# ---------------------------------------------------------------------------
|
|
446
|
-
|
|
447
|
-
HEADING_RE = re.compile(r"^(#{1,6})\s+(.*)$")
|
|
448
|
-
LIST_RE = re.compile(r"^\s*(?:[-*+]|\d+[.)])\s+(.*)$")
|
|
449
|
-
QUOTE_RE = re.compile(r"^\s*>\s?(.*)$")
|
|
450
|
-
TABLE_SEP_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*$")
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
def parse_markdown(source: str) -> list[Block]:
|
|
454
|
-
lines = source.splitlines()
|
|
455
|
-
blocks: list[Block] = []
|
|
456
|
-
paragraph: list[str] = []
|
|
457
|
-
i = 0
|
|
458
|
-
|
|
459
|
-
def flush_paragraph() -> None:
|
|
460
|
-
if paragraph:
|
|
461
|
-
blocks.append(Block(type="paragraph", text_runs=[TextRun(clean_inline(" ".join(paragraph)))]))
|
|
462
|
-
paragraph.clear()
|
|
463
|
-
|
|
464
|
-
while i < len(lines):
|
|
465
|
-
line = lines[i]
|
|
466
|
-
stripped = line.strip()
|
|
467
|
-
if not stripped:
|
|
468
|
-
flush_paragraph()
|
|
469
|
-
i += 1
|
|
470
|
-
continue
|
|
471
|
-
|
|
472
|
-
if stripped.startswith("```") or stripped.startswith("~~~"):
|
|
473
|
-
flush_paragraph()
|
|
474
|
-
fence = stripped[:3]
|
|
475
|
-
code_lines: list[str] = []
|
|
476
|
-
i += 1
|
|
477
|
-
while i < len(lines) and not lines[i].strip().startswith(fence):
|
|
478
|
-
code_lines.append(lines[i])
|
|
479
|
-
i += 1
|
|
480
|
-
if i < len(lines):
|
|
481
|
-
i += 1
|
|
482
|
-
blocks.append(Block(type="code", text_runs=[TextRun("\n".join(code_lines))]))
|
|
483
|
-
continue
|
|
484
|
-
|
|
485
|
-
heading = HEADING_RE.match(line)
|
|
486
|
-
if heading:
|
|
487
|
-
flush_paragraph()
|
|
488
|
-
blocks.append(Block(type="heading", text_runs=[TextRun(clean_inline(heading.group(2)))]))
|
|
489
|
-
i += 1
|
|
490
|
-
continue
|
|
491
|
-
|
|
492
|
-
if _looks_like_table(lines, i):
|
|
493
|
-
flush_paragraph()
|
|
494
|
-
table, consumed = _parse_table(lines, i)
|
|
495
|
-
blocks.append(table)
|
|
496
|
-
i += consumed
|
|
497
|
-
continue
|
|
498
|
-
|
|
499
|
-
item = LIST_RE.match(line)
|
|
500
|
-
if item:
|
|
501
|
-
flush_paragraph()
|
|
502
|
-
items: list[Block] = []
|
|
503
|
-
while i < len(lines):
|
|
504
|
-
match = LIST_RE.match(lines[i])
|
|
505
|
-
if not match:
|
|
506
|
-
break
|
|
507
|
-
items.append(Block(type="list_item", text_runs=[TextRun(clean_inline(match.group(1)))]))
|
|
508
|
-
i += 1
|
|
509
|
-
blocks.append(Block(type="list", children=items))
|
|
510
|
-
continue
|
|
511
|
-
|
|
512
|
-
quote = QUOTE_RE.match(line)
|
|
513
|
-
if quote:
|
|
514
|
-
flush_paragraph()
|
|
515
|
-
quote_lines: list[str] = []
|
|
516
|
-
while i < len(lines):
|
|
517
|
-
match = QUOTE_RE.match(lines[i])
|
|
518
|
-
if not match:
|
|
519
|
-
break
|
|
520
|
-
quote_lines.append(match.group(1))
|
|
521
|
-
i += 1
|
|
522
|
-
blocks.append(Block(type="quote", text_runs=[TextRun(clean_inline(" ".join(quote_lines)))]))
|
|
523
|
-
continue
|
|
524
|
-
|
|
525
|
-
paragraph.append(stripped)
|
|
526
|
-
i += 1
|
|
527
|
-
|
|
528
|
-
flush_paragraph()
|
|
529
|
-
return blocks
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
def clean_inline(text: str) -> str:
|
|
533
|
-
text = re.sub(r"!\[([^\]]*)\]\([^)]+\)", r"\1", text)
|
|
534
|
-
text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
|
|
535
|
-
text = re.sub(r"([*_`~]{1,3})(.*?)\1", r"\2", text)
|
|
536
|
-
text = text.replace("\\", "")
|
|
537
|
-
return text
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
def _looks_like_table(lines: list[str], i: int) -> bool:
|
|
541
|
-
return i + 1 < len(lines) and "|" in lines[i] and TABLE_SEP_RE.match(lines[i + 1]) is not None
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
def _parse_table(lines: list[str], i: int) -> tuple[Block, int]:
|
|
545
|
-
rows: list[Block] = []
|
|
546
|
-
consumed = 0
|
|
547
|
-
while i + consumed < len(lines):
|
|
548
|
-
line = lines[i + consumed]
|
|
549
|
-
stripped = line.strip()
|
|
550
|
-
if not stripped or "|" not in stripped:
|
|
551
|
-
break
|
|
552
|
-
if consumed == 1 and TABLE_SEP_RE.match(stripped):
|
|
553
|
-
consumed += 1
|
|
554
|
-
continue
|
|
555
|
-
cells = [cell.strip() for cell in stripped.strip("|").split("|")]
|
|
556
|
-
row = Block(
|
|
557
|
-
type="tr",
|
|
558
|
-
children=[
|
|
559
|
-
Block(type="table_cell", text_runs=[TextRun(clean_inline(cell))])
|
|
560
|
-
for cell in cells
|
|
561
|
-
if cell
|
|
562
|
-
],
|
|
563
|
-
)
|
|
564
|
-
rows.append(row)
|
|
565
|
-
consumed += 1
|
|
566
|
-
return Block(type="table", children=rows), consumed
|
|
567
|
-
|
|
568
|
-
# ---------------------------------------------------------------------------
|
|
569
|
-
# XML parser
|
|
570
|
-
# ---------------------------------------------------------------------------
|
|
571
|
-
|
|
572
|
-
INLINE_TAGS = {
|
|
573
|
-
"b",
|
|
574
|
-
"strong",
|
|
575
|
-
"i",
|
|
576
|
-
"em",
|
|
577
|
-
"u",
|
|
578
|
-
"s",
|
|
579
|
-
"del",
|
|
580
|
-
"span",
|
|
581
|
-
"text",
|
|
582
|
-
"plain_text",
|
|
583
|
-
"code",
|
|
584
|
-
"a",
|
|
585
|
-
"link",
|
|
586
|
-
"mention",
|
|
587
|
-
"mention-doc",
|
|
588
|
-
"mention-user",
|
|
589
|
-
}
|
|
590
|
-
|
|
591
|
-
TYPE_ALIASES = {
|
|
592
|
-
"doc": "document",
|
|
593
|
-
"document": "document",
|
|
594
|
-
"fragment": "fragment",
|
|
595
|
-
"p": "paragraph",
|
|
596
|
-
"paragraph": "paragraph",
|
|
597
|
-
"heading": "heading",
|
|
598
|
-
"h1": "heading",
|
|
599
|
-
"h2": "heading",
|
|
600
|
-
"h3": "heading",
|
|
601
|
-
"h4": "heading",
|
|
602
|
-
"h5": "heading",
|
|
603
|
-
"h6": "heading",
|
|
604
|
-
"h7": "heading",
|
|
605
|
-
"h8": "heading",
|
|
606
|
-
"h9": "heading",
|
|
607
|
-
"ul": "list",
|
|
608
|
-
"ol": "list",
|
|
609
|
-
"li": "list_item",
|
|
610
|
-
"task": "task",
|
|
611
|
-
"todo": "list_item",
|
|
612
|
-
"blockquote": "quote",
|
|
613
|
-
"quote": "quote",
|
|
614
|
-
"br": "br",
|
|
615
|
-
"hr": "hr",
|
|
616
|
-
"title": "title",
|
|
617
|
-
"checkbox": "checkbox",
|
|
618
|
-
"grid": "grid",
|
|
619
|
-
"column": "column",
|
|
620
|
-
"table": "table",
|
|
621
|
-
"colgroup": "colgroup",
|
|
622
|
-
"col": "col",
|
|
623
|
-
"tr": "tr",
|
|
624
|
-
"td": "table_cell",
|
|
625
|
-
"th": "table_cell",
|
|
626
|
-
"pre": "code",
|
|
627
|
-
"code_block": "code",
|
|
628
|
-
"callout": "callout",
|
|
629
|
-
"figure": "figure",
|
|
630
|
-
"toggle": "toggle",
|
|
631
|
-
"img": "image",
|
|
632
|
-
"source": "source",
|
|
633
|
-
"file": "file",
|
|
634
|
-
"media": "media",
|
|
635
|
-
"latex": "latex",
|
|
636
|
-
"cite": "cite",
|
|
637
|
-
"bookmark": "bookmark",
|
|
638
|
-
"button": "button",
|
|
639
|
-
"whiteboard": "whiteboard",
|
|
640
|
-
"mermaid": "mermaid",
|
|
641
|
-
"plantuml": "plantuml",
|
|
642
|
-
"poll": "poll",
|
|
643
|
-
"isv": "isv",
|
|
644
|
-
"mindnote": "mindnote",
|
|
645
|
-
"diagram": "diagram",
|
|
646
|
-
"sheet": "sheet",
|
|
647
|
-
"bitable": "bitable",
|
|
648
|
-
"base-ref": "base_ref",
|
|
649
|
-
"base_ref": "base_ref",
|
|
650
|
-
"base-refer": "base_ref",
|
|
651
|
-
"base_refer": "base_ref",
|
|
652
|
-
"synced-reference": "synced_reference",
|
|
653
|
-
"synced_reference": "synced_reference",
|
|
654
|
-
"synced-source": "synced_source",
|
|
655
|
-
"synced_source": "synced_source",
|
|
656
|
-
"okr": "okr",
|
|
657
|
-
"chat-card": "chat_card",
|
|
658
|
-
"chat_card": "chat_card",
|
|
659
|
-
"sub_page_list": "sub-page-list",
|
|
660
|
-
"sub-page-list": "sub-page-list",
|
|
661
|
-
}
|
|
662
|
-
|
|
663
|
-
SUBTYPE_ATTR_TAGS = {
|
|
664
|
-
"a",
|
|
665
|
-
"button",
|
|
666
|
-
"cite",
|
|
667
|
-
"img",
|
|
668
|
-
"sheet",
|
|
669
|
-
"source",
|
|
670
|
-
"whiteboard",
|
|
671
|
-
"base-ref",
|
|
672
|
-
"base_ref",
|
|
673
|
-
"base-refer",
|
|
674
|
-
"base_refer",
|
|
675
|
-
"synced-reference",
|
|
676
|
-
"synced_reference",
|
|
677
|
-
"synced-source",
|
|
678
|
-
"synced_source",
|
|
679
|
-
"okr",
|
|
680
|
-
"chat-card",
|
|
681
|
-
"chat_card",
|
|
682
|
-
"sub_page_list",
|
|
683
|
-
"sub-page-list",
|
|
684
|
-
}
|
|
685
|
-
|
|
686
|
-
MAX_XML_INPUT_CHARS = 20_000_000
|
|
687
|
-
FORBIDDEN_XML_DECL_RE = re.compile(r"<!\s*(?:DOCTYPE|ENTITY)\b", re.IGNORECASE)
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
class UserInputError(ValueError):
|
|
691
|
-
pass
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
def local_name(tag: str) -> str:
|
|
695
|
-
if "}" in tag:
|
|
696
|
-
return tag.rsplit("}", 1)[1]
|
|
697
|
-
return tag
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
def block_type_for(elem: ET.Element) -> str:
|
|
701
|
-
tag = local_name(elem.tag)
|
|
702
|
-
explicit = elem.attrib.get("block_type")
|
|
703
|
-
if explicit is None and tag not in SUBTYPE_ATTR_TAGS:
|
|
704
|
-
explicit = elem.attrib.get("type")
|
|
705
|
-
if explicit:
|
|
706
|
-
return TYPE_ALIASES.get(explicit, explicit)
|
|
707
|
-
return TYPE_ALIASES.get(tag, tag)
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
def ensure_safe_xml_source(source: str) -> None:
|
|
711
|
-
if len(source) > MAX_XML_INPUT_CHARS:
|
|
712
|
-
raise UserInputError(
|
|
713
|
-
f"XML input is too large ({len(source)} chars, limit {MAX_XML_INPUT_CHARS})"
|
|
714
|
-
)
|
|
715
|
-
if FORBIDDEN_XML_DECL_RE.search(source):
|
|
716
|
-
raise UserInputError("XML input must not contain DOCTYPE or ENTITY declarations")
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
def parse_xml(source: str) -> list[Block]:
|
|
720
|
-
source = source.strip()
|
|
721
|
-
if not source:
|
|
722
|
-
return []
|
|
723
|
-
ensure_safe_xml_source(source)
|
|
724
|
-
try:
|
|
725
|
-
root = ET.fromstring(source)
|
|
726
|
-
except ET.ParseError:
|
|
727
|
-
# docs +fetch raw output can occasionally include adjacent top-level
|
|
728
|
-
# blocks. Wrap them so standard ElementTree can parse the stream.
|
|
729
|
-
root = ET.fromstring(f"<fragment>{source}</fragment>")
|
|
730
|
-
return [_parse_block(root)]
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
def _parse_block(elem: ET.Element) -> Block:
|
|
734
|
-
block = Block(type=block_type_for(elem), attrs=dict(elem.attrib), raw=elem)
|
|
735
|
-
_collect_content(elem, block)
|
|
736
|
-
if not block.text_runs and not block.children:
|
|
737
|
-
if block.type == "image":
|
|
738
|
-
display = elem.attrib.get("caption")
|
|
739
|
-
else:
|
|
740
|
-
display = (
|
|
741
|
-
elem.attrib.get("text")
|
|
742
|
-
or elem.attrib.get("name")
|
|
743
|
-
or elem.attrib.get("title")
|
|
744
|
-
or elem.attrib.get("alt")
|
|
745
|
-
or elem.attrib.get("caption")
|
|
746
|
-
)
|
|
747
|
-
if display:
|
|
748
|
-
block.text_runs.append(TextRun(display, dict(elem.attrib)))
|
|
749
|
-
return block
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
def _collect_content(elem: ET.Element, block: Block) -> None:
|
|
753
|
-
if elem.text:
|
|
754
|
-
block.text_runs.append(TextRun(elem.text))
|
|
755
|
-
|
|
756
|
-
for child in list(elem):
|
|
757
|
-
tag = local_name(child.tag)
|
|
758
|
-
if tag == "br":
|
|
759
|
-
block.text_runs.append(TextRun("\n"))
|
|
760
|
-
elif tag in INLINE_TAGS:
|
|
761
|
-
_collect_inline(child, block)
|
|
762
|
-
else:
|
|
763
|
-
block.children.append(_parse_block(child))
|
|
764
|
-
if child.tail:
|
|
765
|
-
block.text_runs.append(TextRun(child.tail))
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
def _collect_inline(elem: ET.Element, block: Block) -> None:
|
|
769
|
-
if local_name(elem.tag) == "br":
|
|
770
|
-
block.text_runs.append(TextRun("\n", dict(elem.attrib)))
|
|
771
|
-
return
|
|
772
|
-
|
|
773
|
-
display = (
|
|
774
|
-
elem.attrib.get("text")
|
|
775
|
-
or elem.attrib.get("name")
|
|
776
|
-
or elem.attrib.get("title")
|
|
777
|
-
or elem.attrib.get("alt")
|
|
778
|
-
)
|
|
779
|
-
if display:
|
|
780
|
-
block.text_runs.append(TextRun(display, dict(elem.attrib)))
|
|
781
|
-
return
|
|
782
|
-
|
|
783
|
-
if elem.text:
|
|
784
|
-
block.text_runs.append(TextRun(elem.text, dict(elem.attrib)))
|
|
785
|
-
for child in list(elem):
|
|
786
|
-
_collect_inline(child, block)
|
|
787
|
-
if child.tail:
|
|
788
|
-
block.text_runs.append(TextRun(child.tail))
|
|
789
|
-
|
|
790
|
-
# ---------------------------------------------------------------------------
|
|
791
|
-
# Block extraction registry
|
|
792
|
-
# ---------------------------------------------------------------------------
|
|
793
|
-
|
|
794
|
-
@dataclass
|
|
795
|
-
class ExtractContext:
|
|
796
|
-
unknown_blocks: list[UnknownBlock] = field(default_factory=list)
|
|
797
|
-
resource_texts: dict[str, str] = field(default_factory=dict)
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
class Handler(Protocol):
|
|
801
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
802
|
-
raise NotImplementedError
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
def block_id(block: Block) -> str | None:
|
|
806
|
-
for key in ("id", "block_id", "block-id", "token"):
|
|
807
|
-
value = block.attrs.get(key)
|
|
808
|
-
if isinstance(value, str) and value:
|
|
809
|
-
return value
|
|
810
|
-
return None
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
def runs_text(block: Block) -> str:
|
|
814
|
-
return "".join(run.text for run in block.text_runs)
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
def raw_tag(block: Block) -> str:
|
|
818
|
-
tag = getattr(getattr(block, "raw", None), "tag", "") or ""
|
|
819
|
-
if "}" in tag:
|
|
820
|
-
return tag.rsplit("}", 1)[1]
|
|
821
|
-
return tag
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
class TextBlockHandler:
|
|
825
|
-
def __init__(self, kind: str = "text") -> None:
|
|
826
|
-
self.kind = kind
|
|
827
|
-
|
|
828
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
829
|
-
segments: list[Segment] = []
|
|
830
|
-
text = runs_text(block)
|
|
831
|
-
if text.strip():
|
|
832
|
-
segments.append(
|
|
833
|
-
Segment(
|
|
834
|
-
text=text,
|
|
835
|
-
block_type=block.type,
|
|
836
|
-
block_id=block_id(block),
|
|
837
|
-
kind=self.kind,
|
|
838
|
-
)
|
|
839
|
-
)
|
|
840
|
-
for child in block.children:
|
|
841
|
-
segments.extend(registry.extract(child, ctx))
|
|
842
|
-
return segments
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
class ContainerHandler:
|
|
846
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
847
|
-
segments: list[Segment] = []
|
|
848
|
-
text = runs_text(block)
|
|
849
|
-
if text.strip():
|
|
850
|
-
segments.append(
|
|
851
|
-
Segment(text=text, block_type=block.type, block_id=block_id(block), kind="text")
|
|
852
|
-
)
|
|
853
|
-
for child in block.children:
|
|
854
|
-
segments.extend(registry.extract(child, ctx))
|
|
855
|
-
return segments
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
class ListHandler:
|
|
859
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
860
|
-
tag = raw_tag(block)
|
|
861
|
-
if tag not in {"ol", "ul"}:
|
|
862
|
-
return ContainerHandler().extract(block, registry, ctx)
|
|
863
|
-
|
|
864
|
-
segments: list[Segment] = []
|
|
865
|
-
text = runs_text(block)
|
|
866
|
-
if text.strip():
|
|
867
|
-
segments.append(
|
|
868
|
-
Segment(text=text, block_type=block.type, block_id=block_id(block), kind="text")
|
|
869
|
-
)
|
|
870
|
-
|
|
871
|
-
next_seq = 1
|
|
872
|
-
for child in block.children:
|
|
873
|
-
if child.type == "list_item":
|
|
874
|
-
if tag == "ol":
|
|
875
|
-
seq = child.attrs.get("seq")
|
|
876
|
-
if isinstance(seq, str) and seq.isdigit():
|
|
877
|
-
marker = seq
|
|
878
|
-
next_seq = int(seq) + 1
|
|
879
|
-
else:
|
|
880
|
-
marker = str(next_seq)
|
|
881
|
-
next_seq += 1
|
|
882
|
-
segments.append(
|
|
883
|
-
Segment(
|
|
884
|
-
text=f"{marker}.",
|
|
885
|
-
block_type="list_marker",
|
|
886
|
-
block_id=block_id(child),
|
|
887
|
-
kind="text",
|
|
888
|
-
)
|
|
889
|
-
)
|
|
890
|
-
else:
|
|
891
|
-
segments.append(
|
|
892
|
-
Segment(
|
|
893
|
-
text="•",
|
|
894
|
-
block_type="list_marker",
|
|
895
|
-
block_id=block_id(child),
|
|
896
|
-
kind="marker",
|
|
897
|
-
)
|
|
898
|
-
)
|
|
899
|
-
segments.extend(registry.extract(child, ctx))
|
|
900
|
-
return segments
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
class CheckboxHandler(TextBlockHandler):
|
|
904
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
905
|
-
return [
|
|
906
|
-
Segment(
|
|
907
|
-
text="☑" if block.attrs.get("done") == "true" else "☐",
|
|
908
|
-
block_type="checkbox_marker",
|
|
909
|
-
block_id=block_id(block),
|
|
910
|
-
kind="marker",
|
|
911
|
-
),
|
|
912
|
-
*super().extract(block, registry, ctx),
|
|
913
|
-
]
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
class UnknownHandler:
|
|
917
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
918
|
-
ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block)))
|
|
919
|
-
return ContainerHandler().extract(block, registry, ctx)
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
class IgnoreHandler:
|
|
923
|
-
def __init__(self, action: str = "ignored") -> None:
|
|
924
|
-
self.action = action
|
|
925
|
-
|
|
926
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
927
|
-
ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action=self.action))
|
|
928
|
-
return []
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
class TaskHandler:
|
|
932
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
933
|
-
task_id = block.attrs.get("task-id") or block.attrs.get("task_id")
|
|
934
|
-
if isinstance(task_id, str) and task_id:
|
|
935
|
-
text = ctx.resource_texts.get(f"task:{task_id}")
|
|
936
|
-
if text and text.strip():
|
|
937
|
-
marker = "☑" if block.attrs.get("status") in {"done", "completed", "complete"} else "☐"
|
|
938
|
-
return [
|
|
939
|
-
Segment(text=marker, block_type="task_marker", block_id=block_id(block), kind="marker"),
|
|
940
|
-
Segment(text=text, block_type="task", block_id=block_id(block), kind="resource_title"),
|
|
941
|
-
]
|
|
942
|
-
|
|
943
|
-
ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action="ignored_resource"))
|
|
944
|
-
return []
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
class WhiteboardHandler:
|
|
948
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
949
|
-
board_type = block.attrs.get("type")
|
|
950
|
-
is_empty_resource_shell = not board_type and not block.children and not runs_text(block).strip()
|
|
951
|
-
action = (
|
|
952
|
-
"ignored_resource"
|
|
953
|
-
if board_type in {"blank", "mermaid", "plantuml", "svg"} or is_empty_resource_shell
|
|
954
|
-
else "unsupported_resource"
|
|
955
|
-
)
|
|
956
|
-
ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action=action))
|
|
957
|
-
return []
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
class SyncedSourceHandler:
|
|
961
|
-
def extract(self, block: Block, registry: "Registry", ctx: ExtractContext) -> list[Segment]:
|
|
962
|
-
if block.children:
|
|
963
|
-
segments: list[Segment] = []
|
|
964
|
-
for child in block.children:
|
|
965
|
-
segments.extend(registry.extract(child, ctx))
|
|
966
|
-
return segments
|
|
967
|
-
|
|
968
|
-
ctx.unknown_blocks.append(UnknownBlock(type=block.type, block_id=block_id(block), action="unsupported_resource"))
|
|
969
|
-
return []
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
class Registry:
|
|
973
|
-
def __init__(self) -> None:
|
|
974
|
-
self._handlers: dict[str, Handler] = {}
|
|
975
|
-
self._unknown = UnknownHandler()
|
|
976
|
-
|
|
977
|
-
def register(self, *types: str, handler: Handler) -> None:
|
|
978
|
-
for typ in types:
|
|
979
|
-
self._handlers[typ] = handler
|
|
980
|
-
|
|
981
|
-
def extract(self, block: Block, ctx: ExtractContext) -> list[Segment]:
|
|
982
|
-
handler = self._handlers.get(block.type, self._unknown)
|
|
983
|
-
return handler.extract(block, self, ctx)
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
def default_registry() -> Registry:
|
|
987
|
-
registry = Registry()
|
|
988
|
-
registry.register("document", "fragment", "root", handler=ContainerHandler())
|
|
989
|
-
registry.register("title", handler=TextBlockHandler("title"))
|
|
990
|
-
registry.register("paragraph", "p", handler=TextBlockHandler("text"))
|
|
991
|
-
registry.register(
|
|
992
|
-
"heading",
|
|
993
|
-
"h",
|
|
994
|
-
"h1",
|
|
995
|
-
"h2",
|
|
996
|
-
"h3",
|
|
997
|
-
"h4",
|
|
998
|
-
"h5",
|
|
999
|
-
"h6",
|
|
1000
|
-
"h7",
|
|
1001
|
-
"h8",
|
|
1002
|
-
"h9",
|
|
1003
|
-
handler=TextBlockHandler("heading"),
|
|
1004
|
-
)
|
|
1005
|
-
registry.register("list", "ul", "ol", handler=ListHandler())
|
|
1006
|
-
registry.register("list_item", "li", "todo", handler=TextBlockHandler("list_item"))
|
|
1007
|
-
registry.register("checkbox", handler=CheckboxHandler("list_item"))
|
|
1008
|
-
registry.register(
|
|
1009
|
-
"quote",
|
|
1010
|
-
"blockquote",
|
|
1011
|
-
"callout",
|
|
1012
|
-
"toggle",
|
|
1013
|
-
"grid",
|
|
1014
|
-
"column",
|
|
1015
|
-
"figure",
|
|
1016
|
-
handler=ContainerHandler(),
|
|
1017
|
-
)
|
|
1018
|
-
registry.register("table", "thead", "tbody", "tr", handler=ContainerHandler())
|
|
1019
|
-
registry.register("table_cell", "td", "th", handler=TextBlockHandler("table_cell"))
|
|
1020
|
-
registry.register("code", "code_block", "pre", handler=TextBlockHandler("code"))
|
|
1021
|
-
registry.register("link", "a", "mention", "mention-doc", "mention-user", "time", handler=TextBlockHandler("inline"))
|
|
1022
|
-
registry.register("image", "img", handler=TextBlockHandler("caption"))
|
|
1023
|
-
registry.register("colgroup", "col", "br", "hr", handler=IgnoreHandler("ignored_structure"))
|
|
1024
|
-
registry.register("button", "cite", "latex", "bookmark", handler=IgnoreHandler("ignored_inline"))
|
|
1025
|
-
registry.register("task", handler=TaskHandler())
|
|
1026
|
-
registry.register("whiteboard", handler=WhiteboardHandler())
|
|
1027
|
-
registry.register("synced_source", handler=SyncedSourceHandler())
|
|
1028
|
-
registry.register(
|
|
1029
|
-
"mermaid",
|
|
1030
|
-
"sheet",
|
|
1031
|
-
"source",
|
|
1032
|
-
"file",
|
|
1033
|
-
"media",
|
|
1034
|
-
"chat_card",
|
|
1035
|
-
"base_ref",
|
|
1036
|
-
"bitable",
|
|
1037
|
-
"synced_reference",
|
|
1038
|
-
"poll",
|
|
1039
|
-
"isv",
|
|
1040
|
-
"mindnote",
|
|
1041
|
-
"diagram",
|
|
1042
|
-
"sub-page-list",
|
|
1043
|
-
handler=IgnoreHandler("ignored_resource"),
|
|
1044
|
-
)
|
|
1045
|
-
registry.register(
|
|
1046
|
-
"okr",
|
|
1047
|
-
"plantuml",
|
|
1048
|
-
handler=IgnoreHandler("unsupported_resource"),
|
|
1049
|
-
)
|
|
1050
|
-
return registry
|
|
1051
|
-
|
|
1052
|
-
# ---------------------------------------------------------------------------
|
|
1053
|
-
# CLI
|
|
1054
|
-
# ---------------------------------------------------------------------------
|
|
1055
|
-
|
|
1056
|
-
VERSION = "0.1-alpha"
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
def build_diagnostics(items: list) -> dict[str, object]:
|
|
1060
|
-
actions: dict[str, int] = {}
|
|
1061
|
-
types: dict[str, int] = {}
|
|
1062
|
-
unsupported_types: dict[str, int] = {}
|
|
1063
|
-
unknown_types: dict[str, int] = {}
|
|
1064
|
-
for item in items:
|
|
1065
|
-
actions[item.action] = actions.get(item.action, 0) + 1
|
|
1066
|
-
types[item.type] = types.get(item.type, 0) + 1
|
|
1067
|
-
if item.action == "unsupported_resource":
|
|
1068
|
-
unsupported_types[item.type] = unsupported_types.get(item.type, 0) + 1
|
|
1069
|
-
if item.action == "recurse_children":
|
|
1070
|
-
unknown_types[item.type] = unknown_types.get(item.type, 0) + 1
|
|
1071
|
-
return {
|
|
1072
|
-
"actions": actions,
|
|
1073
|
-
"types": types,
|
|
1074
|
-
"unsupported_types": unsupported_types,
|
|
1075
|
-
"unknown_types": unknown_types,
|
|
1076
|
-
"has_unsupported": bool(unsupported_types),
|
|
1077
|
-
"has_unknown": bool(unknown_types),
|
|
1078
|
-
}
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
def read_input(path: str) -> str:
|
|
1082
|
-
if path == "-":
|
|
1083
|
-
return sys.stdin.read()
|
|
1084
|
-
return Path(path).read_text(encoding="utf-8")
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
def read_resource_texts(path: str | None) -> dict[str, str]:
|
|
1088
|
-
if not path:
|
|
1089
|
-
return {}
|
|
1090
|
-
payload = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
1091
|
-
if not isinstance(payload, dict):
|
|
1092
|
-
raise ValueError("--resource-texts must be a JSON object")
|
|
1093
|
-
return {str(key): str(value) for key, value in payload.items()}
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
def extract_lark_json_content(source: str) -> str:
|
|
1097
|
-
try:
|
|
1098
|
-
envelope = json.loads(source)
|
|
1099
|
-
except json.JSONDecodeError as exc:
|
|
1100
|
-
raise UserInputError(f"could not parse lark-cli JSON envelope: {exc}") from exc
|
|
1101
|
-
|
|
1102
|
-
if not isinstance(envelope, dict):
|
|
1103
|
-
raise UserInputError("lark-cli JSON envelope must be an object")
|
|
1104
|
-
data = envelope.get("data")
|
|
1105
|
-
if not isinstance(data, dict):
|
|
1106
|
-
raise UserInputError("lark-cli JSON envelope is missing object field data")
|
|
1107
|
-
document = data.get("document")
|
|
1108
|
-
if not isinstance(document, dict):
|
|
1109
|
-
raise UserInputError("lark-cli JSON envelope is missing object field data.document")
|
|
1110
|
-
content = document.get("content")
|
|
1111
|
-
if not isinstance(content, str):
|
|
1112
|
-
raise UserInputError("lark-cli JSON envelope is missing string field data.document.content")
|
|
1113
|
-
return content
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
HELP_EPILOG = """
|
|
1117
|
-
Examples:
|
|
1118
|
-
Local XML file:
|
|
1119
|
-
python3 doc_word_stat.py --protocol xml /absolute/path/doc.xml
|
|
1120
|
-
|
|
1121
|
-
Local Markdown file:
|
|
1122
|
-
python3 doc_word_stat.py --protocol md /absolute/path/doc.md
|
|
1123
|
-
|
|
1124
|
-
Pipe an extracted local file:
|
|
1125
|
-
cat /absolute/path/doc.xml | python3 doc_word_stat.py --protocol xml --pretty
|
|
1126
|
-
|
|
1127
|
-
Lark CLI XML fetch, JSON envelope output:
|
|
1128
|
-
lark-cli docs +fetch --doc "$URL" --doc-format xml --detail full --format json \\
|
|
1129
|
-
| python3 doc_word_stat.py --protocol xml --lark-json --pretty
|
|
1130
|
-
|
|
1131
|
-
Lark CLI Markdown fetch, raw content output:
|
|
1132
|
-
lark-cli docs +fetch --doc "$URL" --doc-format markdown \\
|
|
1133
|
-
| python3 doc_word_stat.py --protocol md
|
|
1134
|
-
|
|
1135
|
-
Strict integration for agents or automation:
|
|
1136
|
-
lark-cli docs +fetch --doc "$URL" --doc-format xml --detail full --format json \\
|
|
1137
|
-
| python3 doc_word_stat.py --protocol xml --lark-json --fail-on-unsupported --fail-on-unknown
|
|
1138
|
-
"""
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
def parse_args() -> argparse.Namespace:
|
|
1142
|
-
parser = argparse.ArgumentParser(
|
|
1143
|
-
description="Count semantic words and visible characters in Lark Docs XML or Markdown.",
|
|
1144
|
-
epilog=HELP_EPILOG,
|
|
1145
|
-
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
1146
|
-
)
|
|
1147
|
-
parser.add_argument(
|
|
1148
|
-
"--version",
|
|
1149
|
-
action="version",
|
|
1150
|
-
version=f"%(prog)s {VERSION}",
|
|
1151
|
-
)
|
|
1152
|
-
parser.add_argument(
|
|
1153
|
-
"input",
|
|
1154
|
-
nargs="?",
|
|
1155
|
-
default="-",
|
|
1156
|
-
help="input file path, or '-' / omitted for stdin",
|
|
1157
|
-
)
|
|
1158
|
-
parser.add_argument(
|
|
1159
|
-
"--protocol",
|
|
1160
|
-
choices=("xml", "md"),
|
|
1161
|
-
required=True,
|
|
1162
|
-
help="input protocol produced by docs +fetch",
|
|
1163
|
-
)
|
|
1164
|
-
parser.add_argument(
|
|
1165
|
-
"--pretty",
|
|
1166
|
-
action="store_true",
|
|
1167
|
-
help="pretty-print JSON output",
|
|
1168
|
-
)
|
|
1169
|
-
parser.add_argument(
|
|
1170
|
-
"--segments",
|
|
1171
|
-
action="store_true",
|
|
1172
|
-
help="include extracted text segments for debugging",
|
|
1173
|
-
)
|
|
1174
|
-
parser.add_argument(
|
|
1175
|
-
"--lark-json",
|
|
1176
|
-
action="store_true",
|
|
1177
|
-
help="read lark-cli docs +fetch JSON and count data.document.content",
|
|
1178
|
-
)
|
|
1179
|
-
parser.add_argument(
|
|
1180
|
-
"--resource-texts",
|
|
1181
|
-
help='optional JSON object mapping resource keys to visible text, e.g. {"task:<task-id>": "title"}',
|
|
1182
|
-
)
|
|
1183
|
-
parser.add_argument(
|
|
1184
|
-
"--fail-on-unsupported",
|
|
1185
|
-
action="store_true",
|
|
1186
|
-
help="exit with code 2 when unsupported_blocks is non-empty",
|
|
1187
|
-
)
|
|
1188
|
-
parser.add_argument(
|
|
1189
|
-
"--fail-on-unknown",
|
|
1190
|
-
action="store_true",
|
|
1191
|
-
help="exit with code 3 when unknown XML/Markdown block types are encountered",
|
|
1192
|
-
)
|
|
1193
|
-
return parser.parse_args()
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
def main() -> int:
|
|
1197
|
-
args = parse_args()
|
|
1198
|
-
source = read_input(args.input)
|
|
1199
|
-
if args.lark_json:
|
|
1200
|
-
try:
|
|
1201
|
-
source = extract_lark_json_content(source)
|
|
1202
|
-
except UserInputError as exc:
|
|
1203
|
-
print(f"error: {exc}", file=sys.stderr)
|
|
1204
|
-
return 1
|
|
1205
|
-
|
|
1206
|
-
if args.protocol == "xml":
|
|
1207
|
-
try:
|
|
1208
|
-
blocks = parse_xml(source)
|
|
1209
|
-
except (ET.ParseError, UserInputError) as exc:
|
|
1210
|
-
print(f"error: could not parse XML input: {exc}", file=sys.stderr)
|
|
1211
|
-
return 1
|
|
1212
|
-
else:
|
|
1213
|
-
blocks = parse_markdown(source)
|
|
1214
|
-
|
|
1215
|
-
ctx = ExtractContext(resource_texts=read_resource_texts(args.resource_texts))
|
|
1216
|
-
registry = default_registry()
|
|
1217
|
-
segments = []
|
|
1218
|
-
for block in blocks:
|
|
1219
|
-
segments.extend(registry.extract(block, ctx))
|
|
1220
|
-
|
|
1221
|
-
stats = Counter().count_segments(segments)
|
|
1222
|
-
payload = stats.to_dict()
|
|
1223
|
-
payload["protocol"] = args.protocol
|
|
1224
|
-
payload["unknown_blocks"] = [item.to_dict() for item in ctx.unknown_blocks]
|
|
1225
|
-
payload["unsupported_blocks"] = [
|
|
1226
|
-
item.to_dict() for item in ctx.unknown_blocks if item.action == "unsupported_resource"
|
|
1227
|
-
]
|
|
1228
|
-
payload["diagnostics"] = build_diagnostics(ctx.unknown_blocks)
|
|
1229
|
-
if args.segments:
|
|
1230
|
-
payload["segments"] = [segment.to_dict() for segment in segments]
|
|
1231
|
-
|
|
1232
|
-
indent = 2 if args.pretty else None
|
|
1233
|
-
print(json.dumps(payload, ensure_ascii=False, indent=indent, sort_keys=args.pretty))
|
|
1234
|
-
if args.fail_on_unsupported and payload["unsupported_blocks"]:
|
|
1235
|
-
return 2
|
|
1236
|
-
if args.fail_on_unknown and payload["diagnostics"]["has_unknown"]:
|
|
1237
|
-
return 3
|
|
1238
|
-
return 0
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
if __name__ == "__main__":
|
|
1242
|
-
raise SystemExit(main())
|
|
1243
|
-
|