@xingwangzhe/stalux 1.25.9 → 1.25.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +3 -5
- package/src/plugins/feature-flags.ts +64 -21
- package/src/utils/count-words.ts +0 -20
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@xingwangzhe/stalux",
|
|
3
|
-
"version": "1.25.
|
|
3
|
+
"version": "1.25.10",
|
|
4
4
|
"description": "A powerful, modern Astro blog theme — use as template or install as plugin",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"astro",
|
|
@@ -78,8 +78,6 @@
|
|
|
78
78
|
"@astrojs/markdown-satteri": "0.3.7",
|
|
79
79
|
"@astrojs/rss": "^4.0.19",
|
|
80
80
|
"@astrojs/sitemap": "^3.7.3",
|
|
81
|
-
"@echogarden/icu-segmentation-wasm": "^0.4.0",
|
|
82
|
-
"@echogarden/text-segmentation": "^0.7.0",
|
|
83
81
|
"@expressive-code/plugin-line-numbers": "^0.44.1",
|
|
84
82
|
"@lucide/astro": "^1.33.0",
|
|
85
83
|
"@mcp-b/webmcp-polyfill": "^4.0.0",
|
|
@@ -93,7 +91,6 @@
|
|
|
93
91
|
"astro-seo": "^1.1.0",
|
|
94
92
|
"astro-typewriter": "^1.2.0",
|
|
95
93
|
"badge-maker": "^6.0.0",
|
|
96
|
-
"js-tokens": "^10.0.0",
|
|
97
94
|
"pagefind": "^1.5.2",
|
|
98
95
|
"photoswipe": "^5.4.4",
|
|
99
96
|
"satteri": "^0.10.5",
|
|
@@ -106,7 +103,8 @@
|
|
|
106
103
|
"knip": "^6.32.2",
|
|
107
104
|
"oxfmt": "^0.64.0",
|
|
108
105
|
"oxlint": "^1.79.0",
|
|
109
|
-
"typescript": "^7.0.2"
|
|
106
|
+
"typescript": "^7.0.2",
|
|
107
|
+
"vite": "8.2.2"
|
|
110
108
|
},
|
|
111
109
|
"peerDependencies": {
|
|
112
110
|
"astro": "7.2.4"
|
|
@@ -1,14 +1,12 @@
|
|
|
1
1
|
import { createSatteriMarkdownProcessor } from "@astrojs/markdown-satteri";
|
|
2
|
-
import { splitToWords } from "@echogarden/text-segmentation";
|
|
3
|
-
import jsTokens from "js-tokens";
|
|
4
2
|
/**
|
|
5
3
|
* Sätteri 插件:在构建时完成字数统计和特性标记,
|
|
6
4
|
* 只从 Sätteri 的 MDAST/HAST 节点读取信息,并通过 ctx 注入最终结果。
|
|
7
5
|
*
|
|
8
6
|
* 字数统计策略:
|
|
9
7
|
* - 统计 paragraph、heading 和 tableCell,覆盖列表、引用、表格等正文。
|
|
10
|
-
* -
|
|
11
|
-
* -
|
|
8
|
+
* - 行内 code 属于正文;独立 code、math/displayMath 由各自 lexer 统计。
|
|
9
|
+
* - 代码和数学公式都计入最终 wordCount,同时保留结构化阅读成本字段。
|
|
12
10
|
*
|
|
13
11
|
* 特性标记:
|
|
14
12
|
* - HAST 阶段 <img> 检测 → hasImage
|
|
@@ -22,10 +20,18 @@ const CODE_SECONDS_PER_NON_EMPTY_LINE = 2;
|
|
|
22
20
|
const INLINE_MATH_BASE_SECONDS = 1;
|
|
23
21
|
const DISPLAY_MATH_BASE_SECONDS = 4;
|
|
24
22
|
const MATH_SECONDS_PER_NON_WHITESPACE_CHARACTER = 0.03;
|
|
23
|
+
const CODE_SECONDS_PER_TOKEN = 0.12;
|
|
24
|
+
const MATH_SECONDS_PER_TOKEN = 0.2;
|
|
25
|
+
const proseSegmenter = new Intl.Segmenter("und", { granularity: "grapheme" });
|
|
26
|
+
const HAN_GRAPHEME = /^\p{Script=Han}$/u;
|
|
27
|
+
const LETTER_GRAPHEME = /^[\p{Letter}\p{Mark}]$/u;
|
|
28
|
+
const NUMBER_GRAPHEME = /^\p{Number}$/u;
|
|
29
|
+
const MATH_TOKEN = /\\[a-zA-Z]+|[a-zA-Z]+|\d+(?:\.\d+)?|[^\s{}]/gu;
|
|
25
30
|
|
|
26
31
|
type FeatureFlagsState = {
|
|
27
32
|
proseText: string;
|
|
28
33
|
codeTokens: number;
|
|
34
|
+
codeWordCount: number;
|
|
29
35
|
codeCharacters: number;
|
|
30
36
|
codeNonEmptyLines: number;
|
|
31
37
|
codeBlocks: number;
|
|
@@ -33,6 +39,7 @@ type FeatureFlagsState = {
|
|
|
33
39
|
displayMathBlocks: number;
|
|
34
40
|
mathCharacters: number;
|
|
35
41
|
mathNonWhitespaceCharacters: number;
|
|
42
|
+
mathWordCount: number;
|
|
36
43
|
hasImage: boolean;
|
|
37
44
|
};
|
|
38
45
|
|
|
@@ -41,6 +48,7 @@ function getState(ctx: any): FeatureFlagsState {
|
|
|
41
48
|
return (data.staluxFeatureFlags ??= {
|
|
42
49
|
proseText: "",
|
|
43
50
|
codeTokens: 0,
|
|
51
|
+
codeWordCount: 0,
|
|
44
52
|
codeCharacters: 0,
|
|
45
53
|
codeNonEmptyLines: 0,
|
|
46
54
|
codeBlocks: 0,
|
|
@@ -48,6 +56,7 @@ function getState(ctx: any): FeatureFlagsState {
|
|
|
48
56
|
displayMathBlocks: 0,
|
|
49
57
|
mathCharacters: 0,
|
|
50
58
|
mathNonWhitespaceCharacters: 0,
|
|
59
|
+
mathWordCount: 0,
|
|
51
60
|
hasImage: false,
|
|
52
61
|
}) as FeatureFlagsState;
|
|
53
62
|
}
|
|
@@ -57,6 +66,7 @@ function injectState(ctx: any) {
|
|
|
57
66
|
const injected = ctx.data.astro.frontmatter as Record<string, unknown>;
|
|
58
67
|
injected.proseText = state.proseText;
|
|
59
68
|
injected.codeTokens = state.codeTokens;
|
|
69
|
+
injected.codeWordCount = state.codeWordCount;
|
|
60
70
|
injected.codeCharacters = state.codeCharacters;
|
|
61
71
|
injected.codeNonEmptyLines = state.codeNonEmptyLines;
|
|
62
72
|
injected.codeBlocks = state.codeBlocks;
|
|
@@ -64,23 +74,51 @@ function injectState(ctx: any) {
|
|
|
64
74
|
injected.displayMathBlocks = state.displayMathBlocks;
|
|
65
75
|
injected.mathCharacters = state.mathCharacters;
|
|
66
76
|
injected.mathNonWhitespaceCharacters = state.mathNonWhitespaceCharacters;
|
|
77
|
+
injected.mathWordCount = state.mathWordCount;
|
|
67
78
|
injected.hasImage = state.hasImage;
|
|
68
79
|
}
|
|
69
80
|
|
|
70
81
|
function countCodeTokens(value: string, lang?: string): number {
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
+
// 统计阶段不依赖语法高亮器:移除注释后,按标识符、数字、字符串和符号计数。
|
|
83
|
+
// lang 保留在签名中,方便后续按语言增加规则,但当前规则对常见代码语言通用。
|
|
84
|
+
void lang;
|
|
85
|
+
const withoutComments = value.replace(/\/\/[^\r\n]*|\/\*[\s\S]*?\*\/|<!--[\s\S]*?-->/gu, " ");
|
|
86
|
+
return (
|
|
87
|
+
withoutComments.match(
|
|
88
|
+
/`(?:\\.|[^`])*`|"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'|\p{L}[\p{L}\p{N}_]*|\p{N}+(?:\.\p{N}+)?|[^\s]/gu,
|
|
89
|
+
)?.length ?? 0
|
|
90
|
+
);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function countProseWords(value: string): number {
|
|
94
|
+
let count = 0;
|
|
95
|
+
let inWord = false;
|
|
96
|
+
let inNumber = false;
|
|
97
|
+
|
|
98
|
+
for (const { segment } of proseSegmenter.segment(value)) {
|
|
99
|
+
if (HAN_GRAPHEME.test(segment)) {
|
|
100
|
+
count++;
|
|
101
|
+
inWord = false;
|
|
102
|
+
inNumber = false;
|
|
103
|
+
} else if (LETTER_GRAPHEME.test(segment)) {
|
|
104
|
+
if (!inWord) count++;
|
|
105
|
+
inWord = true;
|
|
106
|
+
inNumber = false;
|
|
107
|
+
} else if (NUMBER_GRAPHEME.test(segment)) {
|
|
108
|
+
if (!inNumber) count++;
|
|
109
|
+
inNumber = true;
|
|
110
|
+
inWord = false;
|
|
111
|
+
} else {
|
|
112
|
+
inWord = false;
|
|
113
|
+
inNumber = false;
|
|
114
|
+
}
|
|
82
115
|
}
|
|
83
|
-
|
|
116
|
+
|
|
117
|
+
return count;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function countMathTokens(value: string): number {
|
|
121
|
+
return value.match(MATH_TOKEN)?.length ?? 0;
|
|
84
122
|
}
|
|
85
123
|
|
|
86
124
|
function countMath(node: any, ctx: any, display: boolean) {
|
|
@@ -90,6 +128,7 @@ function countMath(node: any, ctx: any, display: boolean) {
|
|
|
90
128
|
else state.inlineMathBlocks++;
|
|
91
129
|
state.mathCharacters += [...value].length;
|
|
92
130
|
state.mathNonWhitespaceCharacters += (value.match(/\S/gu) ?? []).length;
|
|
131
|
+
state.mathWordCount += countMathTokens(value);
|
|
93
132
|
injectState(ctx);
|
|
94
133
|
}
|
|
95
134
|
|
|
@@ -113,7 +152,9 @@ export const featureFlagsMdast = defineMdastPlugin({
|
|
|
113
152
|
state.codeNonEmptyLines += node.value
|
|
114
153
|
.split(/\r?\n/u)
|
|
115
154
|
.filter((line) => /\S/u.test(line)).length;
|
|
116
|
-
|
|
155
|
+
const codeTokens = countCodeTokens(node.value, node.lang);
|
|
156
|
+
state.codeTokens += codeTokens;
|
|
157
|
+
state.codeWordCount += codeTokens;
|
|
117
158
|
injectState(ctx);
|
|
118
159
|
},
|
|
119
160
|
|
|
@@ -206,9 +247,9 @@ async function analyzeFeatureFlagsUncached(body: string): Promise<FeatureFlagsRe
|
|
|
206
247
|
const processor = await getFeatureFlagsProcessor();
|
|
207
248
|
const result = await processor.render(body ?? "");
|
|
208
249
|
const metadata = result.metadata.frontmatter as Record<string, unknown>;
|
|
209
|
-
const proseWords = (
|
|
210
|
-
|
|
211
|
-
|
|
250
|
+
const proseWords = countProseWords(String(metadata.proseText ?? ""));
|
|
251
|
+
const codeWordCount = Number(metadata.codeWordCount ?? 0);
|
|
252
|
+
const mathWordCount = Number(metadata.mathWordCount ?? 0);
|
|
212
253
|
const codeNonEmptyLines = Number(metadata.codeNonEmptyLines ?? 0);
|
|
213
254
|
const inlineMathBlocks = Number(metadata.inlineMathBlocks ?? 0);
|
|
214
255
|
const displayMathBlocks = Number(metadata.displayMathBlocks ?? 0);
|
|
@@ -218,11 +259,13 @@ async function analyzeFeatureFlagsUncached(body: string): Promise<FeatureFlagsRe
|
|
|
218
259
|
readingMinutes: Math.ceil(
|
|
219
260
|
proseWords / WORDS_PER_MINUTE +
|
|
220
261
|
(codeNonEmptyLines * CODE_SECONDS_PER_NON_EMPTY_LINE +
|
|
262
|
+
codeWordCount * CODE_SECONDS_PER_TOKEN +
|
|
221
263
|
inlineMathBlocks * INLINE_MATH_BASE_SECONDS +
|
|
222
264
|
displayMathBlocks * DISPLAY_MATH_BASE_SECONDS +
|
|
265
|
+
mathWordCount * MATH_SECONDS_PER_TOKEN +
|
|
223
266
|
mathNonWhitespaceCharacters * MATH_SECONDS_PER_NON_WHITESPACE_CHARACTER) /
|
|
224
267
|
60,
|
|
225
268
|
),
|
|
226
|
-
wordCount: proseWords,
|
|
269
|
+
wordCount: proseWords + codeWordCount + mathWordCount,
|
|
227
270
|
};
|
|
228
271
|
}
|
package/src/utils/count-words.ts
DELETED
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* CJK 友好的字数统计 — 基于 W3C Intl.Segmenter API(UAX #29 标准分词)。
|
|
3
|
-
*
|
|
4
|
-
* 替代手写 Unicode 正则的方案:
|
|
5
|
-
* - CJK 字符:`Intl.Segmenter` 逐字分割,每个字计 1
|
|
6
|
-
* - 拉丁/数字/其他字母文字:按词计 1
|
|
7
|
-
* - 标点/空白不计
|
|
8
|
-
*/
|
|
9
|
-
const segmenter = new Intl.Segmenter("zh", { granularity: "word" });
|
|
10
|
-
|
|
11
|
-
/**
|
|
12
|
-
* 统计文本字数(基于 Intl.Segmenter 分词)。
|
|
13
|
-
*/
|
|
14
|
-
export function countWords(text: string): number {
|
|
15
|
-
let count = 0;
|
|
16
|
-
for (const { isWordLike } of segmenter.segment(text)) {
|
|
17
|
-
if (isWordLike) count++;
|
|
18
|
-
}
|
|
19
|
-
return count;
|
|
20
|
-
}
|