@xingwangzhe/stalux 1.25.9 → 1.25.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@xingwangzhe/stalux",
3
- "version": "1.25.9",
3
+ "version": "1.25.10",
4
4
  "description": "A powerful, modern Astro blog theme — use as template or install as plugin",
5
5
  "keywords": [
6
6
  "astro",
@@ -78,8 +78,6 @@
78
78
  "@astrojs/markdown-satteri": "0.3.7",
79
79
  "@astrojs/rss": "^4.0.19",
80
80
  "@astrojs/sitemap": "^3.7.3",
81
- "@echogarden/icu-segmentation-wasm": "^0.4.0",
82
- "@echogarden/text-segmentation": "^0.7.0",
83
81
  "@expressive-code/plugin-line-numbers": "^0.44.1",
84
82
  "@lucide/astro": "^1.33.0",
85
83
  "@mcp-b/webmcp-polyfill": "^4.0.0",
@@ -93,7 +91,6 @@
93
91
  "astro-seo": "^1.1.0",
94
92
  "astro-typewriter": "^1.2.0",
95
93
  "badge-maker": "^6.0.0",
96
- "js-tokens": "^10.0.0",
97
94
  "pagefind": "^1.5.2",
98
95
  "photoswipe": "^5.4.4",
99
96
  "satteri": "^0.10.5",
@@ -106,7 +103,8 @@
106
103
  "knip": "^6.32.2",
107
104
  "oxfmt": "^0.64.0",
108
105
  "oxlint": "^1.79.0",
109
- "typescript": "^7.0.2"
106
+ "typescript": "^7.0.2",
107
+ "vite": "8.2.2"
110
108
  },
111
109
  "peerDependencies": {
112
110
  "astro": "7.2.4"
@@ -1,14 +1,12 @@
1
1
  import { createSatteriMarkdownProcessor } from "@astrojs/markdown-satteri";
2
- import { splitToWords } from "@echogarden/text-segmentation";
3
- import jsTokens from "js-tokens";
4
2
  /**
5
3
  * Sätteri 插件:在构建时完成字数统计和特性标记,
6
4
  * 只从 Sätteri 的 MDAST/HAST 节点读取信息,并通过 ctx 注入最终结果。
7
5
  *
8
6
  * 字数统计策略:
9
7
  * - 统计 paragraph、heading 和 tableCell,覆盖列表、引用、表格等正文。
10
- * - 独立的 code 块、math/displayMath 公式不参与正文词数统计。
11
- * - 行内 code 参与正文统计,inlineMath 单独计入公式阅读成本。
8
+ * - 行内 code 属于正文;独立 code、math/displayMath 由各自 lexer 统计。
9
+ * - 代码和数学公式都计入最终 wordCount,同时保留结构化阅读成本字段。
12
10
  *
13
11
  * 特性标记:
14
12
  * - HAST 阶段 <img> 检测 → hasImage
@@ -22,10 +20,18 @@ const CODE_SECONDS_PER_NON_EMPTY_LINE = 2;
22
20
  const INLINE_MATH_BASE_SECONDS = 1;
23
21
  const DISPLAY_MATH_BASE_SECONDS = 4;
24
22
  const MATH_SECONDS_PER_NON_WHITESPACE_CHARACTER = 0.03;
23
+ const CODE_SECONDS_PER_TOKEN = 0.12;
24
+ const MATH_SECONDS_PER_TOKEN = 0.2;
25
+ const proseSegmenter = new Intl.Segmenter("und", { granularity: "grapheme" });
26
+ const HAN_GRAPHEME = /^\p{Script=Han}$/u;
27
+ const LETTER_GRAPHEME = /^[\p{Letter}\p{Mark}]$/u;
28
+ const NUMBER_GRAPHEME = /^\p{Number}$/u;
29
+ const MATH_TOKEN = /\\[a-zA-Z]+|[a-zA-Z]+|\d+(?:\.\d+)?|[^\s{}]/gu;
25
30
 
26
31
  type FeatureFlagsState = {
27
32
  proseText: string;
28
33
  codeTokens: number;
34
+ codeWordCount: number;
29
35
  codeCharacters: number;
30
36
  codeNonEmptyLines: number;
31
37
  codeBlocks: number;
@@ -33,6 +39,7 @@ type FeatureFlagsState = {
33
39
  displayMathBlocks: number;
34
40
  mathCharacters: number;
35
41
  mathNonWhitespaceCharacters: number;
42
+ mathWordCount: number;
36
43
  hasImage: boolean;
37
44
  };
38
45
 
@@ -41,6 +48,7 @@ function getState(ctx: any): FeatureFlagsState {
41
48
  return (data.staluxFeatureFlags ??= {
42
49
  proseText: "",
43
50
  codeTokens: 0,
51
+ codeWordCount: 0,
44
52
  codeCharacters: 0,
45
53
  codeNonEmptyLines: 0,
46
54
  codeBlocks: 0,
@@ -48,6 +56,7 @@ function getState(ctx: any): FeatureFlagsState {
48
56
  displayMathBlocks: 0,
49
57
  mathCharacters: 0,
50
58
  mathNonWhitespaceCharacters: 0,
59
+ mathWordCount: 0,
51
60
  hasImage: false,
52
61
  }) as FeatureFlagsState;
53
62
  }
@@ -57,6 +66,7 @@ function injectState(ctx: any) {
57
66
  const injected = ctx.data.astro.frontmatter as Record<string, unknown>;
58
67
  injected.proseText = state.proseText;
59
68
  injected.codeTokens = state.codeTokens;
69
+ injected.codeWordCount = state.codeWordCount;
60
70
  injected.codeCharacters = state.codeCharacters;
61
71
  injected.codeNonEmptyLines = state.codeNonEmptyLines;
62
72
  injected.codeBlocks = state.codeBlocks;
@@ -64,23 +74,51 @@ function injectState(ctx: any) {
64
74
  injected.displayMathBlocks = state.displayMathBlocks;
65
75
  injected.mathCharacters = state.mathCharacters;
66
76
  injected.mathNonWhitespaceCharacters = state.mathNonWhitespaceCharacters;
77
+ injected.mathWordCount = state.mathWordCount;
67
78
  injected.hasImage = state.hasImage;
68
79
  }
69
80
 
70
81
  function countCodeTokens(value: string, lang?: string): number {
71
- if (/^(?:js|jsx|javascript|ts|tsx|typescript)$/.test(lang ?? "")) {
72
- return Array.from(jsTokens(value, { jsx: /jsx|tsx/.test(lang ?? "") })).filter(
73
- (token) =>
74
- ![
75
- "WhiteSpace",
76
- "LineTerminatorSequence",
77
- "MultiLineComment",
78
- "SingleLineComment",
79
- "HashbangComment",
80
- ].includes(token.type),
81
- ).length;
82
+ // 统计阶段不依赖语法高亮器:移除注释后,按标识符、数字、字符串和符号计数。
83
+ // lang 保留在签名中,方便后续按语言增加规则,但当前规则对常见代码语言通用。
84
+ void lang;
85
+ const withoutComments = value.replace(/\/\/[^\r\n]*|\/\*[\s\S]*?\*\/|<!--[\s\S]*?-->/gu, " ");
86
+ return (
87
+ withoutComments.match(
88
+ /`(?:\\.|[^`])*`|"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'|\p{L}[\p{L}\p{N}_]*|\p{N}+(?:\.\p{N}+)?|[^\s]/gu,
89
+ )?.length ?? 0
90
+ );
91
+ }
92
+
93
+ function countProseWords(value: string): number {
94
+ let count = 0;
95
+ let inWord = false;
96
+ let inNumber = false;
97
+
98
+ for (const { segment } of proseSegmenter.segment(value)) {
99
+ if (HAN_GRAPHEME.test(segment)) {
100
+ count++;
101
+ inWord = false;
102
+ inNumber = false;
103
+ } else if (LETTER_GRAPHEME.test(segment)) {
104
+ if (!inWord) count++;
105
+ inWord = true;
106
+ inNumber = false;
107
+ } else if (NUMBER_GRAPHEME.test(segment)) {
108
+ if (!inNumber) count++;
109
+ inNumber = true;
110
+ inWord = false;
111
+ } else {
112
+ inWord = false;
113
+ inNumber = false;
114
+ }
82
115
  }
83
- return value.match(/\p{L}[\p{L}\p{N}_]*|\p{N}+(?:\.\p{N}+)?|[^\s]/gu)?.length ?? 0;
116
+
117
+ return count;
118
+ }
119
+
120
+ function countMathTokens(value: string): number {
121
+ return value.match(MATH_TOKEN)?.length ?? 0;
84
122
  }
85
123
 
86
124
  function countMath(node: any, ctx: any, display: boolean) {
@@ -90,6 +128,7 @@ function countMath(node: any, ctx: any, display: boolean) {
90
128
  else state.inlineMathBlocks++;
91
129
  state.mathCharacters += [...value].length;
92
130
  state.mathNonWhitespaceCharacters += (value.match(/\S/gu) ?? []).length;
131
+ state.mathWordCount += countMathTokens(value);
93
132
  injectState(ctx);
94
133
  }
95
134
 
@@ -113,7 +152,9 @@ export const featureFlagsMdast = defineMdastPlugin({
113
152
  state.codeNonEmptyLines += node.value
114
153
  .split(/\r?\n/u)
115
154
  .filter((line) => /\S/u.test(line)).length;
116
- state.codeTokens += countCodeTokens(node.value, node.lang);
155
+ const codeTokens = countCodeTokens(node.value, node.lang);
156
+ state.codeTokens += codeTokens;
157
+ state.codeWordCount += codeTokens;
117
158
  injectState(ctx);
118
159
  },
119
160
 
@@ -206,9 +247,9 @@ async function analyzeFeatureFlagsUncached(body: string): Promise<FeatureFlagsRe
206
247
  const processor = await getFeatureFlagsProcessor();
207
248
  const result = await processor.render(body ?? "");
208
249
  const metadata = result.metadata.frontmatter as Record<string, unknown>;
209
- const proseWords = (
210
- await splitToWords(String(metadata.proseText ?? ""))
211
- ).nonPunctuationEntries.filter((entry) => /\S/u.test(entry.text)).length;
250
+ const proseWords = countProseWords(String(metadata.proseText ?? ""));
251
+ const codeWordCount = Number(metadata.codeWordCount ?? 0);
252
+ const mathWordCount = Number(metadata.mathWordCount ?? 0);
212
253
  const codeNonEmptyLines = Number(metadata.codeNonEmptyLines ?? 0);
213
254
  const inlineMathBlocks = Number(metadata.inlineMathBlocks ?? 0);
214
255
  const displayMathBlocks = Number(metadata.displayMathBlocks ?? 0);
@@ -218,11 +259,13 @@ async function analyzeFeatureFlagsUncached(body: string): Promise<FeatureFlagsRe
218
259
  readingMinutes: Math.ceil(
219
260
  proseWords / WORDS_PER_MINUTE +
220
261
  (codeNonEmptyLines * CODE_SECONDS_PER_NON_EMPTY_LINE +
262
+ codeWordCount * CODE_SECONDS_PER_TOKEN +
221
263
  inlineMathBlocks * INLINE_MATH_BASE_SECONDS +
222
264
  displayMathBlocks * DISPLAY_MATH_BASE_SECONDS +
265
+ mathWordCount * MATH_SECONDS_PER_TOKEN +
223
266
  mathNonWhitespaceCharacters * MATH_SECONDS_PER_NON_WHITESPACE_CHARACTER) /
224
267
  60,
225
268
  ),
226
- wordCount: proseWords,
269
+ wordCount: proseWords + codeWordCount + mathWordCount,
227
270
  };
228
271
  }
@@ -1,20 +0,0 @@
1
- /**
2
- * CJK 友好的字数统计 — 基于 W3C Intl.Segmenter API(UAX #29 标准分词)。
3
- *
4
- * 替代手写 Unicode 正则的方案:
5
- * - CJK 字符:`Intl.Segmenter` 逐字分割,每个字计 1
6
- * - 拉丁/数字/其他字母文字:按词计 1
7
- * - 标点/空白不计
8
- */
9
- const segmenter = new Intl.Segmenter("zh", { granularity: "word" });
10
-
11
- /**
12
- * 统计文本字数(基于 Intl.Segmenter 分词)。
13
- */
14
- export function countWords(text: string): number {
15
- let count = 0;
16
- for (const { isWordLike } of segmenter.segment(text)) {
17
- if (isWordLike) count++;
18
- }
19
- return count;
20
- }