@jerryliang122/openclaw-qqbot 1.0.9 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -203,8 +203,7 @@ interface QQBotAccountConfig {
203
203
  sendMode?: 'stream' | 'static';
204
204
  };
205
205
  /**
206
- * STT (语音转文字) 配置
207
- * 配置后,收到语音消息时会自动调用 STT 服务转录为文字
206
+ * STT (语音转文字) 历史遗留配置块(整块被忽略,详见 STTChannelConfig)
208
207
  */
209
208
  stt?: STTChannelConfig;
210
209
  /**
@@ -288,18 +287,21 @@ interface AudioFormatPolicy {
288
287
  transcodeEnabled?: boolean;
289
288
  }
290
289
  /**
291
- * STT (语音转文字) 配置
290
+ * STT (语音转文字) 配置块——历史遗留形状
291
+ *
292
+ * 2026-10-04 起整块被忽略(含历史行为开关 enabled/asrFallback 与旧凭证
293
+ * 键):STT 启停只由框架 `tools.media.models`(capabilities 含 "audio" 的
294
+ * 条目)+ `tools.media.audio.enabled` 控制。本接口仅为遗留检测
295
+ * (hasLegacySttConfig + 一次性迁移提示)保留键的类型形状。
292
296
  */
293
297
  interface STTChannelConfig {
294
- /** 是否启用 STT(默认 true,配置了 baseUrl+apiKey 即自动启用) */
295
- enabled?: boolean;
296
- /** STT 服务提供商 ID(对应 models.providers 中的 key,默认 "openai") */
298
+ /** @deprecated 2026-10 起忽略——STT 配置统一走框架 tools.media.models */
297
299
  provider?: string;
298
- /** STT API 地址(如 https://api.openai.com/v1) */
300
+ /** @deprecated 同上 */
299
301
  baseUrl?: string;
300
- /** STT API 密钥 */
302
+ /** @deprecated 同上 */
301
303
  apiKey?: string;
302
- /** STT 模型名称(默认 "whisper-1") */
304
+ /** @deprecated 同上 */
303
305
  model?: string;
304
306
  }
305
307
  /**
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@jerryliang122/openclaw-qqbot",
3
- "version": "1.0.9",
3
+ "version": "2.1.0",
4
4
  "description": "QQ Bot channel plugin for OpenClaw (independently maintained fork)",
5
5
  "publishConfig": {
6
6
  "access": "public",
@@ -164,10 +164,12 @@ function buildDynamicCtx(
164
164
  lines.push(`- Voice: ${voiceRefs.join(', ')}`);
165
165
  }
166
166
 
167
- // ASR:source==='asr' 的 text,或任意 transcript 上的 asrReferText
167
+ // ASR:仅平台转写来源(source==='asr')——严格信框架后 stt/fallback
168
+ // transcript 不再携带 asrReferText
168
169
  const asrTexts = unique(
169
170
  transcripts
170
- .map((t) => (t.source === 'asr' ? t.text : t.asrReferText))
171
+ .filter((t) => t.source === 'asr')
172
+ .map((t) => t.text)
171
173
  .filter(isNonEmpty),
172
174
  );
173
175
  if (asrTexts.length > 0) {
@@ -15,7 +15,7 @@ import {
15
15
  isVoiceAttachment,
16
16
  } from '@tencent-connect/qqbot-nodejs/protocol';
17
17
  import type { MessageAttachment } from '../types.js';
18
- import { transcribeAudio, resolveSTTConfig, shouldUsePlatformAsr } from '../utils/stt.js';
18
+ import { isFrameworkSttConfigured, hasLegacySttConfig, transcribeAudioViaFramework } from '../utils/stt.js';
19
19
  import { formatVoiceText, formatDuration, type VoiceTranscript, type TranscriptSource } from '../utils/voice-text.js';
20
20
  import { downloadRemoteMedia } from '../adapter/media.js';
21
21
  import { getAdapters } from '../adapter/resolve.js';
@@ -96,8 +96,8 @@ export async function processAttachments(
96
96
  cfg: Record<string, unknown>,
97
97
  log?: Log,
98
98
  ): Promise<ProcessedAttachments> {
99
- const sttCfg = resolveSTTConfig(cfg);
100
- const usePlatformAsr = shouldUsePlatformAsr(cfg);
99
+ const sttConfigured = isFrameworkSttConfigured(cfg);
100
+ warnLegacySttConfig(cfg, log);
101
101
  const audioPolicy = resolveAudioPolicy(cfg);
102
102
 
103
103
  const imageUrls: string[] = [];
@@ -120,7 +120,7 @@ export async function processAttachments(
120
120
  }
121
121
 
122
122
  if (isVoice) {
123
- const transcript = await processVoiceAttachment(att, sttCfg, usePlatformAsr, audioPolicy, log);
123
+ const transcript = await processVoiceAttachment(att, cfg, sttConfigured, audioPolicy, log);
124
124
  return { type: 'voice' as const, transcript };
125
125
  }
126
126
 
@@ -208,39 +208,46 @@ function kindFromContentType(contentType: string | undefined): InboundMediaEntry
208
208
 
209
209
  // ── 语音处理 ──
210
210
 
211
+ /** 废弃 channels.qqbot.stt 配置块的一次性迁移提示(每进程一条,避免每条语音刷屏) */
212
+ let legacySttWarned = false;
213
+
214
+ function warnLegacySttConfig(cfg: Record<string, unknown>, log?: Log): void {
215
+ if (!legacySttWarned && hasLegacySttConfig(cfg)) {
216
+ legacySttWarned = true;
217
+ log?.info(
218
+ 'Voice: channels.qqbot.stt is deprecated and ignored entirely (credentials + enabled/asrFallback); ' +
219
+ 'configure an audio-capable tools.media.models entry for STT — platform asr_refer_text is used only when framework STT is absent',
220
+ );
221
+ }
222
+ }
223
+
211
224
  async function processVoiceAttachment(
212
225
  att: MessageAttachment,
213
- sttCfg: ReturnType<typeof resolveSTTConfig>,
214
- usePlatformAsr: boolean,
226
+ cfg: Record<string, unknown>,
227
+ sttConfigured: boolean,
215
228
  audioPolicy: AudioPolicyResolved,
216
229
  log?: Log,
217
230
  ): Promise<VoiceTranscript> {
218
- // 平台转写(asr_refer_text)仅在显式 asrFallback: true 时参与;
219
- // 缺省/false 时在所有场景下丢弃——包括 STT 未配置(语音落占位文本)
220
- // 与 STT 失败(不当兜底),三条泄漏路径(转写成功携带 / 转写失败回退 /
221
- // 下载失败回退)一并堵死。
222
- const rawAsrText = att.asr_refer_text?.trim() || undefined;
223
- const asrReferText = usePlatformAsr ? rawAsrText : undefined;
224
231
  // 远端 URL 兜底:优先 wav_url,其次原始 url
225
232
  const remoteUrl = normalizeUrl(att.voice_wav_url) || normalizeUrl(att.url) || undefined;
226
233
 
227
- // STT 未配置:占位文本;显式 asrFallback: true 时退回平台转写
228
- if (!sttCfg) {
229
- if (!usePlatformAsr && rawAsrText) {
230
- log?.info(`Voice: STT not configured; platform asr_refer_text discarded (asrFallback not enabled)`);
231
- }
234
+ // 框架 STT 未配置 → 平台转写(asr_refer_text,QQ 平台自动 STT 随事件
235
+ // JSON 下发)直接作为唯一来源(无下载、零外部调用);无平台转写 → 占位。
236
+ if (!sttConfigured) {
237
+ const asrReferText = att.asr_refer_text?.trim() || undefined;
232
238
  if (asrReferText) {
233
- log?.debug?.(`Voice: using asr_refer_text (STT not configured, asrFallback enabled)`);
239
+ log?.debug?.(`Voice: using platform asr_refer_text (framework STT not configured)`);
234
240
  return { text: asrReferText, source: 'asr', asrReferText, remoteUrl };
235
241
  }
236
242
  return {
237
243
  text: '[Voice message - transcription unavailable]',
238
244
  source: 'fallback',
239
- asrReferText,
240
245
  remoteUrl,
241
246
  };
242
247
  }
243
248
 
249
+ // 框架 STT 已配置 → 严格信框架:下载/转码后提交框架转录,
250
+ // 失败/为空/下载失败一律占位文本,不回退平台转写。
244
251
  let localPath: string | undefined;
245
252
  let duration: number | undefined;
246
253
 
@@ -281,27 +288,22 @@ async function processVoiceAttachment(
281
288
 
282
289
  if (localPath) {
283
290
  try {
284
- const transcript = await transcribeAudio(localPath, cfg2stt(sttCfg));
291
+ const transcript = await transcribeAudioViaFramework(localPath, cfg);
285
292
  if (transcript) {
286
- log?.debug?.(`Voice STT: ${transcript.slice(0, 80)}...`);
287
- return { text: transcript, source: 'stt', duration, localPath, remoteUrl, asrReferText };
293
+ log?.debug?.(`Voice STT (framework): ${transcript.slice(0, 80)}...`);
294
+ return { text: transcript, source: 'stt', duration, localPath, remoteUrl };
288
295
  }
289
296
  } catch (err) {
290
- log?.error(`Voice STT failed: ${err instanceof Error ? err.message : String(err)}`);
297
+ log?.error(`Voice STT (framework) failed: ${err instanceof Error ? err.message : String(err)}`);
291
298
  }
292
299
  }
293
300
 
294
- if (asrReferText) {
295
- return { text: asrReferText, source: 'asr', duration, localPath, remoteUrl, asrReferText };
296
- }
297
-
298
301
  return {
299
302
  text: '[Voice message - transcription failed]',
300
303
  source: 'fallback',
301
304
  duration,
302
305
  localPath,
303
306
  remoteUrl,
304
- asrReferText,
305
307
  };
306
308
  }
307
309
 
@@ -334,10 +336,6 @@ function normalizeFormats(formats: string[]): string[] {
334
336
  });
335
337
  }
336
338
 
337
- function cfg2stt(sttCfg: NonNullable<ReturnType<typeof resolveSTTConfig>>): Record<string, unknown> {
338
- return { channels: { qqbot: { stt: sttCfg } } };
339
- }
340
-
341
339
  // ── 文件工具 ──
342
340
 
343
341
  function normalizeUrl(url: string | undefined): string {
package/src/types.ts CHANGED
@@ -211,8 +211,7 @@ export interface QQBotAccountConfig {
211
211
  sendMode?: 'stream' | 'static';
212
212
  };
213
213
  /**
214
- * STT (语音转文字) 配置
215
- * 配置后,收到语音消息时会自动调用 STT 服务转录为文字
214
+ * STT (语音转文字) 历史遗留配置块(整块被忽略,详见 STTChannelConfig)
216
215
  */
217
216
  stt?: STTChannelConfig;
218
217
  /**
@@ -300,18 +299,21 @@ export interface AudioFormatPolicy {
300
299
  }
301
300
 
302
301
  /**
303
- * STT (语音转文字) 配置
302
+ * STT (语音转文字) 配置块——历史遗留形状
303
+ *
304
+ * 2026-10-04 起整块被忽略(含历史行为开关 enabled/asrFallback 与旧凭证
305
+ * 键):STT 启停只由框架 `tools.media.models`(capabilities 含 "audio" 的
306
+ * 条目)+ `tools.media.audio.enabled` 控制。本接口仅为遗留检测
307
+ * (hasLegacySttConfig + 一次性迁移提示)保留键的类型形状。
304
308
  */
305
309
  export interface STTChannelConfig {
306
- /** 是否启用 STT(默认 true,配置了 baseUrl+apiKey 即自动启用) */
307
- enabled?: boolean;
308
- /** STT 服务提供商 ID(对应 models.providers 中的 key,默认 "openai") */
310
+ /** @deprecated 2026-10 起忽略——STT 配置统一走框架 tools.media.models */
309
311
  provider?: string;
310
- /** STT API 地址(如 https://api.openai.com/v1) */
312
+ /** @deprecated 同上 */
311
313
  baseUrl?: string;
312
- /** STT API 密钥 */
314
+ /** @deprecated 同上 */
313
315
  apiKey?: string;
314
- /** STT 模型名称(默认 "whisper-1") */
316
+ /** @deprecated 同上 */
315
317
  model?: string;
316
318
  }
317
319
 
package/src/utils/stt.ts CHANGED
@@ -1,114 +1,88 @@
1
1
  /**
2
- * STT (Speech-to-Text) 语音转文字服务
2
+ * STT (Speech-to-Text) 语音转文字 — 框架音频理解管线
3
3
  *
4
- * 支持 OpenAI 兼容的 /audio/transcriptions 接口。
5
- * 配置优先级:
6
- * 1. channels.qqbot.stt(插件级)
7
- * 2. 框架级 audio model 配置
4
+ * 转录统一委托给 openclaw/plugin-sdk/media-understanding-runtime 的
5
+ * `transcribeAudioFile`(provider 注册表、附件缓存、SSRF 策略与错误语义
6
+ * 均由框架维护),插件不自带 OpenAI 兼容 HTTP 调用;STT 配置只认框架级
7
+ * `tools.media.models` 的 audio 能力条目(与内置 telegram 通道一致)。
8
+ *
9
+ * 硬编码两分支策略(2026-10-04 起,无插件级开关):
10
+ * - 框架 STT 未配置 → QQ 平台转写(asr_refer_text,随事件 JSON 下发)直
11
+ * 接作为唯一来源(零下载、零外部调用);无平台转写 → 占位文本。
12
+ * - 框架 STT 已配置 → 下载语音提交框架转录;**严格信框架**——失败/为空/
13
+ * 下载失败一律占位文本,不回退平台转写。
8
14
  */
9
- import * as fs from 'node:fs';
10
15
  import * as path from 'node:path';
16
+ import { transcribeAudioFile } from 'openclaw/plugin-sdk/media-understanding-runtime';
11
17
 
12
- export interface STTConfig {
13
- enabled: boolean;
14
- baseUrl: string;
15
- apiKey: string;
16
- model: string;
17
- }
18
+ type TranscribeParams = Parameters<typeof transcribeAudioFile>[0];
18
19
 
19
20
  /**
20
- * 平台转写(asr_refer_text)参与判定。
21
- * 仅当显式配置 channels.qqbot.stt.asrFallback: true 时保留平台转写;
22
- * 缺省、false 或 stt 块整体不存在时一律丢弃——包括 STT 未配置的场景
23
- * (此时语音消息落占位文本,而不是退回平台转写)。
24
- * 读取独立于 STT 凭证解析成败:stt 块无凭证但 asrFallback: true 仍生效。
21
+ * 框架 STT(语音转录)是否可用:
22
+ * - `tools.media.audio.enabled === false` → 框架级 per-capability 关闭
23
+ * - `tools.media.models` 无显式 `capabilities` 含 `"audio"` 的条目 → 未配置
24
+ *
25
+ * 规范路径是 `tools.media.models`——openclaw 2026.9.1 schema 中模型列表只
26
+ * 存在于此(`tools.media.audio` 块的类型为 `Omit<…, "models">`,不含
27
+ * models 键)。**只认显式 `capabilities` 标签**(有意保守):无标签 CLI
28
+ * 条目按框架语义本就不参与共享列表的 audio 匹配;无标签 provider 条目框
29
+ * 架会从 provider 注册表推断能力,但插件侧无法廉价复刻注册表——宁可漏判
30
+ * (降级走平台转写,功能仍可用)也不误判(严格模式下误判会变成彻底无转
31
+ * 写)。
32
+ *
33
+ * 仅做存在性探测控制流程;provider 解析与实际调用由 transcribeAudioFile
34
+ * 完成(错误在调用点捕获处理)。
25
35
  */
26
- export function shouldUsePlatformAsr(cfg: Record<string, unknown>): boolean {
27
- const channels = asRecord(cfg.channels);
28
- const qqbot = asRecord(channels?.qqbot);
29
- return asRecord(qqbot?.stt)?.asrFallback === true;
36
+ export function isFrameworkSttConfigured(cfg: Record<string, unknown>): boolean {
37
+ const tools = asRecord(cfg.tools);
38
+ const media = asRecord(tools?.media);
39
+ if (asRecord(media?.audio)?.enabled === false) {
40
+ return false;
41
+ }
42
+ const models = media?.models;
43
+ if (!Array.isArray(models)) {
44
+ return false;
45
+ }
46
+ return models.some((entry) => {
47
+ const capabilities = asRecord(entry)?.capabilities;
48
+ return Array.isArray(capabilities) && capabilities.includes('audio');
49
+ });
30
50
  }
31
51
 
32
52
  /**
33
- * 从 OpenClaw 配置中解析 STT 设置
53
+ * 检测已废弃的 `channels.qqbot.stt` 配置块:旧凭证键
54
+ * (provider/baseUrl/apiKey/model)与历史行为开关(enabled/asrFallback)。
55
+ * 2026-10-04 起整块被忽略——STT 启停只由框架 `tools.media.models`(audio
56
+ * 能力条目)+ `tools.media.audio.enabled` 控制;返回 true 时调用方打一次
57
+ * 性迁移提示。
34
58
  */
35
- export function resolveSTTConfig(cfg: Record<string, unknown>): STTConfig | null {
59
+ export function hasLegacySttConfig(cfg: Record<string, unknown>): boolean {
36
60
  const channels = asRecord(cfg.channels);
37
61
  const qqbot = asRecord(channels?.qqbot);
38
- const sttCfg = asRecord(qqbot?.stt);
39
-
40
- // 显式禁用
41
- if (sttCfg?.enabled === false) {
42
- return null;
43
- }
44
-
45
- const models = asRecord(cfg.models);
46
- const providers = asRecord(models?.providers);
47
-
48
- // 1. 插件级 STT 配置
49
- if (sttCfg) {
50
- const providerId = readString(sttCfg, 'provider') ?? 'openai';
51
- const providerCfg = asRecord(providers?.[providerId]);
52
- const baseUrl = readString(sttCfg, 'baseUrl') ?? readString(providerCfg, 'baseUrl');
53
- const apiKey = readString(sttCfg, 'apiKey') ?? readString(providerCfg, 'apiKey');
54
- const model = readString(sttCfg, 'model') ?? 'whisper-1';
55
- if (baseUrl && apiKey) {
56
- return { enabled: true, baseUrl: baseUrl.replace(/\/+$/, ''), apiKey, model };
57
- }
58
- }
59
-
60
- // 2. 框架级 audio model fallback
61
- const tools = asRecord(cfg.tools);
62
- const media = asRecord(tools?.media);
63
- const audio = asRecord(media?.audio);
64
- const audioModels = audio?.models;
65
- const audioModelEntry = Array.isArray(audioModels) ? asRecord(audioModels[0]) : undefined;
66
- if (audioModelEntry) {
67
- const providerId = readString(audioModelEntry, 'provider') ?? 'openai';
68
- const providerCfg = asRecord(providers?.[providerId]);
69
- const baseUrl = readString(audioModelEntry, 'baseUrl') ?? readString(providerCfg, 'baseUrl');
70
- const apiKey = readString(audioModelEntry, 'apiKey') ?? readString(providerCfg, 'apiKey');
71
- const model = readString(audioModelEntry, 'model') ?? 'whisper-1';
72
- if (baseUrl && apiKey) {
73
- return { enabled: true, baseUrl: baseUrl.replace(/\/+$/, ''), apiKey, model };
74
- }
75
- }
76
-
77
- return null;
62
+ const stt = asRecord(qqbot?.stt);
63
+ if (!stt) return false;
64
+ const legacyKeys = ['provider', 'baseUrl', 'apiKey', 'model', 'enabled', 'asrFallback'] as const;
65
+ return legacyKeys.some((key) => {
66
+ const value = stt[key];
67
+ if (typeof value === 'string') return value.trim().length > 0;
68
+ return value != null;
69
+ });
78
70
  }
79
71
 
80
72
  /**
81
- * 调用 STT 服务转录音频文件
73
+ * 经框架音频理解管线转录本地音频文件。
74
+ * 返回修剪后的转录文本;无文本返回 null。
75
+ * provider 缺失/调用失败会抛错,由调用方捕获后输出失败占位文本。
82
76
  */
83
- export async function transcribeAudio(
77
+ export async function transcribeAudioViaFramework(
84
78
  audioPath: string,
85
79
  cfg: Record<string, unknown>,
86
80
  ): Promise<string | null> {
87
- const sttCfg = resolveSTTConfig(cfg);
88
- if (!sttCfg) {
89
- return null;
90
- }
91
-
92
- const fileBuffer = fs.readFileSync(audioPath);
93
- const fileName = sanitizeFileName(path.basename(audioPath));
94
- const mime = guessMimeType(fileName);
95
-
96
- const form = new FormData();
97
- form.append('file', new Blob([fileBuffer], { type: mime }), fileName);
98
- form.append('model', sttCfg.model);
99
-
100
- const resp = await fetch(`${sttCfg.baseUrl}/audio/transcriptions`, {
101
- method: 'POST',
102
- headers: { Authorization: `Bearer ${sttCfg.apiKey}` },
103
- body: form,
81
+ const result = await transcribeAudioFile({
82
+ filePath: audioPath,
83
+ cfg: cfg as unknown as TranscribeParams['cfg'],
84
+ mime: guessMimeType(audioPath),
104
85
  });
105
-
106
- if (!resp.ok) {
107
- const detail = await resp.text().catch(() => '');
108
- throw new Error(`STT failed (HTTP ${resp.status}): ${detail.slice(0, 300)}`);
109
- }
110
-
111
- const result = (await resp.json()) as { text?: string };
112
86
  return result.text?.trim() || null;
113
87
  }
114
88
 
@@ -121,18 +95,6 @@ function asRecord(value: unknown): Record<string, unknown> | undefined {
121
95
  return undefined;
122
96
  }
123
97
 
124
- function readString(obj: Record<string, unknown> | undefined, key: string): string | undefined {
125
- const val = obj?.[key];
126
- if (typeof val === 'string' && val.trim()) {
127
- return val.trim();
128
- }
129
- return undefined;
130
- }
131
-
132
- function sanitizeFileName(name: string): string {
133
- return name.replace(/[^a-zA-Z0-9._-]/g, '_');
134
- }
135
-
136
98
  function guessMimeType(fileName: string): string {
137
99
  const ext = path.extname(fileName).toLowerCase();
138
100
  const mimeMap: Record<string, string> = {