@jerryliang122/openclaw-qqbot 1.0.9 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -203,8 +203,9 @@ interface QQBotAccountConfig {
203
203
  sendMode?: 'stream' | 'static';
204
204
  };
205
205
  /**
206
- * STT (语音转文字) 配置
207
- * 配置后,收到语音消息时会自动调用 STT 服务转录为文字
206
+ * STT (语音转文字) 行为开关
207
+ * 转录凭证统一走框架 tools.media.audio.models;
208
+ * 框架未配置时直接使用 QQ 平台转写(asr_refer_text)
208
209
  */
209
210
  stt?: STTChannelConfig;
210
211
  /**
@@ -289,17 +290,30 @@ interface AudioFormatPolicy {
289
290
  }
290
291
  /**
291
292
  * STT (语音转文字) 配置
293
+ *
294
+ * 2026-10 起转录统一走框架音频理解管线(tools.media.audio.models 凭证),
295
+ * 本块只保留行为开关;旧凭证键已废弃(检测到会打迁移提示日志)。
296
+ * 框架 STT 未配置时,QQ 平台转写(asr_refer_text)直接作为唯一来源。
292
297
  */
293
298
  interface STTChannelConfig {
294
- /** 是否启用 STT(默认 true,配置了 baseUrl+apiKey 即自动启用) */
299
+ /**
300
+ * 是否启用框架 STT 转录(默认 true)。
301
+ * false = 不调用外部 STT,语音只用平台转写(或无转写时占位文本)。
302
+ */
295
303
  enabled?: boolean;
296
- /** STT 服务提供商 ID(对应 models.providers 中的 key,默认 "openai") */
304
+ /**
305
+ * 平台转写(asr_refer_text)参与开关。默认参与:
306
+ * 框架 STT 未配置时直接作为唯一来源,STT 失败/为空时兜底。
307
+ * 设为 false 恢复严格模式——所有场景丢弃平台转写。
308
+ */
309
+ asrFallback?: boolean;
310
+ /** @deprecated 2026-10 起忽略——STT 凭证统一配置在框架 tools.media.audio.models */
297
311
  provider?: string;
298
- /** STT API 地址(如 https://api.openai.com/v1) */
312
+ /** @deprecated 同上 */
299
313
  baseUrl?: string;
300
- /** STT API 密钥 */
314
+ /** @deprecated 同上 */
301
315
  apiKey?: string;
302
- /** STT 模型名称(默认 "whisper-1") */
316
+ /** @deprecated 同上 */
303
317
  model?: string;
304
318
  }
305
319
  /**
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@jerryliang122/openclaw-qqbot",
3
- "version": "1.0.9",
3
+ "version": "2.0.0",
4
4
  "description": "QQ Bot channel plugin for OpenClaw (independently maintained fork)",
5
5
  "publishConfig": {
6
6
  "access": "public",
@@ -15,7 +15,7 @@ import {
15
15
  isVoiceAttachment,
16
16
  } from '@tencent-connect/qqbot-nodejs/protocol';
17
17
  import type { MessageAttachment } from '../types.js';
18
- import { transcribeAudio, resolveSTTConfig, shouldUsePlatformAsr } from '../utils/stt.js';
18
+ import { shouldUsePlatformAsr, isFrameworkSttConfigured, hasLegacySttCredentials, transcribeAudioViaFramework } from '../utils/stt.js';
19
19
  import { formatVoiceText, formatDuration, type VoiceTranscript, type TranscriptSource } from '../utils/voice-text.js';
20
20
  import { downloadRemoteMedia } from '../adapter/media.js';
21
21
  import { getAdapters } from '../adapter/resolve.js';
@@ -96,8 +96,9 @@ export async function processAttachments(
96
96
  cfg: Record<string, unknown>,
97
97
  log?: Log,
98
98
  ): Promise<ProcessedAttachments> {
99
- const sttCfg = resolveSTTConfig(cfg);
100
99
  const usePlatformAsr = shouldUsePlatformAsr(cfg);
100
+ const sttConfigured = isFrameworkSttConfigured(cfg);
101
+ warnLegacySttCredentials(cfg, log);
101
102
  const audioPolicy = resolveAudioPolicy(cfg);
102
103
 
103
104
  const imageUrls: string[] = [];
@@ -120,7 +121,7 @@ export async function processAttachments(
120
121
  }
121
122
 
122
123
  if (isVoice) {
123
- const transcript = await processVoiceAttachment(att, sttCfg, usePlatformAsr, audioPolicy, log);
124
+ const transcript = await processVoiceAttachment(att, cfg, usePlatformAsr, sttConfigured, audioPolicy, log);
124
125
  return { type: 'voice' as const, transcript };
125
126
  }
126
127
 
@@ -208,29 +209,42 @@ function kindFromContentType(contentType: string | undefined): InboundMediaEntry
208
209
 
209
210
  // ── 语音处理 ──
210
211
 
212
+ /** 废弃插件级 STT 凭证的一次性迁移提示(每进程一条,避免每条语音刷屏) */
213
+ let legacySttWarned = false;
214
+
215
+ function warnLegacySttCredentials(cfg: Record<string, unknown>, log?: Log): void {
216
+ if (!legacySttWarned && hasLegacySttCredentials(cfg)) {
217
+ legacySttWarned = true;
218
+ log?.info(
219
+ 'Voice: channels.qqbot.stt credentials (provider/baseUrl/apiKey/model) are deprecated and ignored; ' +
220
+ 'configure tools.media.audio.models instead — platform asr_refer_text is used when framework STT is absent',
221
+ );
222
+ }
223
+ }
224
+
211
225
  async function processVoiceAttachment(
212
226
  att: MessageAttachment,
213
- sttCfg: ReturnType<typeof resolveSTTConfig>,
227
+ cfg: Record<string, unknown>,
214
228
  usePlatformAsr: boolean,
229
+ sttConfigured: boolean,
215
230
  audioPolicy: AudioPolicyResolved,
216
231
  log?: Log,
217
232
  ): Promise<VoiceTranscript> {
218
- // 平台转写(asr_refer_text)仅在显式 asrFallback: true 时参与;
219
- // 缺省/false 时在所有场景下丢弃——包括 STT 未配置(语音落占位文本)
220
- // 与 STT 失败(不当兜底),三条泄漏路径(转写成功携带 / 转写失败回退 /
221
- // 下载失败回退)一并堵死。
233
+ // 平台转写(asr_refer_text,QQ 平台自动 STT 随事件 JSON 下发):
234
+ // 默认参与——框架 STT 未配置时直接作为唯一来源,STT 失败时兜底;
235
+ // 仅 asrFallback: false(严格模式)时在所有场景丢弃。
222
236
  const rawAsrText = att.asr_refer_text?.trim() || undefined;
223
237
  const asrReferText = usePlatformAsr ? rawAsrText : undefined;
224
238
  // 远端 URL 兜底:优先 wav_url,其次原始 url
225
239
  const remoteUrl = normalizeUrl(att.voice_wav_url) || normalizeUrl(att.url) || undefined;
226
240
 
227
- // STT 未配置:占位文本;显式 asrFallback: true 时退回平台转写
228
- if (!sttCfg) {
241
+ // 框架 STT 未配置 → 平台转写直接作为 transcript(无下载、无外部调用)
242
+ if (!sttConfigured) {
229
243
  if (!usePlatformAsr && rawAsrText) {
230
- log?.info(`Voice: STT not configured; platform asr_refer_text discarded (asrFallback not enabled)`);
244
+ log?.info(`Voice: framework STT not configured; platform asr_refer_text discarded (asrFallback: false)`);
231
245
  }
232
246
  if (asrReferText) {
233
- log?.debug?.(`Voice: using asr_refer_text (STT not configured, asrFallback enabled)`);
247
+ log?.debug?.(`Voice: using platform asr_refer_text (framework STT not configured)`);
234
248
  return { text: asrReferText, source: 'asr', asrReferText, remoteUrl };
235
249
  }
236
250
  return {
@@ -281,17 +295,18 @@ async function processVoiceAttachment(
281
295
 
282
296
  if (localPath) {
283
297
  try {
284
- const transcript = await transcribeAudio(localPath, cfg2stt(sttCfg));
298
+ const transcript = await transcribeAudioViaFramework(localPath, cfg);
285
299
  if (transcript) {
286
- log?.debug?.(`Voice STT: ${transcript.slice(0, 80)}...`);
300
+ log?.debug?.(`Voice STT (framework): ${transcript.slice(0, 80)}...`);
287
301
  return { text: transcript, source: 'stt', duration, localPath, remoteUrl, asrReferText };
288
302
  }
289
303
  } catch (err) {
290
- log?.error(`Voice STT failed: ${err instanceof Error ? err.message : String(err)}`);
304
+ log?.error(`Voice STT (framework) failed: ${err instanceof Error ? err.message : String(err)}`);
291
305
  }
292
306
  }
293
307
 
294
308
  if (asrReferText) {
309
+ log?.debug?.(`Voice: falling back to platform asr_refer_text after framework STT failure`);
295
310
  return { text: asrReferText, source: 'asr', duration, localPath, remoteUrl, asrReferText };
296
311
  }
297
312
 
@@ -334,10 +349,6 @@ function normalizeFormats(formats: string[]): string[] {
334
349
  });
335
350
  }
336
351
 
337
- function cfg2stt(sttCfg: NonNullable<ReturnType<typeof resolveSTTConfig>>): Record<string, unknown> {
338
- return { channels: { qqbot: { stt: sttCfg } } };
339
- }
340
-
341
352
  // ── 文件工具 ──
342
353
 
343
354
  function normalizeUrl(url: string | undefined): string {
package/src/types.ts CHANGED
@@ -211,8 +211,9 @@ export interface QQBotAccountConfig {
211
211
  sendMode?: 'stream' | 'static';
212
212
  };
213
213
  /**
214
- * STT (语音转文字) 配置
215
- * 配置后,收到语音消息时会自动调用 STT 服务转录为文字
214
+ * STT (语音转文字) 行为开关
215
+ * 转录凭证统一走框架 tools.media.audio.models;
216
+ * 框架未配置时直接使用 QQ 平台转写(asr_refer_text)
216
217
  */
217
218
  stt?: STTChannelConfig;
218
219
  /**
@@ -301,17 +302,30 @@ export interface AudioFormatPolicy {
301
302
 
302
303
  /**
303
304
  * STT (语音转文字) 配置
305
+ *
306
+ * 2026-10 起转录统一走框架音频理解管线(tools.media.audio.models 凭证),
307
+ * 本块只保留行为开关;旧凭证键已废弃(检测到会打迁移提示日志)。
308
+ * 框架 STT 未配置时,QQ 平台转写(asr_refer_text)直接作为唯一来源。
304
309
  */
305
310
  export interface STTChannelConfig {
306
- /** 是否启用 STT(默认 true,配置了 baseUrl+apiKey 即自动启用) */
311
+ /**
312
+ * 是否启用框架 STT 转录(默认 true)。
313
+ * false = 不调用外部 STT,语音只用平台转写(或无转写时占位文本)。
314
+ */
307
315
  enabled?: boolean;
308
- /** STT 服务提供商 ID(对应 models.providers 中的 key,默认 "openai") */
316
+ /**
317
+ * 平台转写(asr_refer_text)参与开关。默认参与:
318
+ * 框架 STT 未配置时直接作为唯一来源,STT 失败/为空时兜底。
319
+ * 设为 false 恢复严格模式——所有场景丢弃平台转写。
320
+ */
321
+ asrFallback?: boolean;
322
+ /** @deprecated 2026-10 起忽略——STT 凭证统一配置在框架 tools.media.audio.models */
309
323
  provider?: string;
310
- /** STT API 地址(如 https://api.openai.com/v1) */
324
+ /** @deprecated 同上 */
311
325
  baseUrl?: string;
312
- /** STT API 密钥 */
326
+ /** @deprecated 同上 */
313
327
  apiKey?: string;
314
- /** STT 模型名称(默认 "whisper-1") */
328
+ /** @deprecated 同上 */
315
329
  model?: string;
316
330
  }
317
331
 
package/src/utils/stt.ts CHANGED
@@ -1,114 +1,83 @@
1
1
  /**
2
- * STT (Speech-to-Text) 语音转文字服务
2
+ * STT (Speech-to-Text) 语音转文字 — 框架音频理解管线
3
3
  *
4
- * 支持 OpenAI 兼容的 /audio/transcriptions 接口。
5
- * 配置优先级:
6
- * 1. channels.qqbot.stt(插件级)
7
- * 2. 框架级 audio model 配置
4
+ * 转录统一委托给 openclaw/plugin-sdk/media-understanding-runtime 的
5
+ * `transcribeAudioFile`(provider 注册表、附件缓存、SSRF 策略与错误语义
6
+ * 均由框架维护),插件不再自带 OpenAI 兼容 HTTP 调用;STT 凭证只认
7
+ * 框架级 `tools.media.audio.models` 配置(与内置 telegram 通道一致)。
8
+ *
9
+ * 平台转写(asr_refer_text):QQ 平台对语音消息自动 STT 并随事件 JSON 下发。
10
+ * 默认策略——框架 STT 未配置时**直接采用平台转写**;已配置时平台转写作为
11
+ * 自有转录失败/为空的兜底。`channels.qqbot.stt.asrFallback: false` 可整体
12
+ * 禁用平台转写(严格模式,恢复 2026-08-17 的丢弃行为)。
8
13
  */
9
- import * as fs from 'node:fs';
10
14
  import * as path from 'node:path';
15
+ import { transcribeAudioFile } from 'openclaw/plugin-sdk/media-understanding-runtime';
11
16
 
12
- export interface STTConfig {
13
- enabled: boolean;
14
- baseUrl: string;
15
- apiKey: string;
16
- model: string;
17
- }
17
+ type TranscribeParams = Parameters<typeof transcribeAudioFile>[0];
18
18
 
19
19
  /**
20
- * 平台转写(asr_refer_text)参与判定。
21
- * 仅当显式配置 channels.qqbot.stt.asrFallback: true 时保留平台转写;
22
- * 缺省、false 或 stt 块整体不存在时一律丢弃——包括 STT 未配置的场景
23
- * (此时语音消息落占位文本,而不是退回平台转写)。
24
- * 读取独立于 STT 凭证解析成败:stt 块无凭证但 asrFallback: true 仍生效。
20
+ * 平台转写(asr_refer_text)是否参与(独立于框架 STT 配置读取)。
21
+ * 默认 true;显式 `channels.qqbot.stt.asrFallback: false` 时关闭(严格模式)。
25
22
  */
26
23
  export function shouldUsePlatformAsr(cfg: Record<string, unknown>): boolean {
27
24
  const channels = asRecord(cfg.channels);
28
25
  const qqbot = asRecord(channels?.qqbot);
29
- return asRecord(qqbot?.stt)?.asrFallback === true;
26
+ return asRecord(qqbot?.stt)?.asrFallback !== false;
30
27
  }
31
28
 
32
29
  /**
33
- * 从 OpenClaw 配置中解析 STT 设置
30
+ * 框架 STT(tools.media.audio)是否可用:
31
+ * - `channels.qqbot.stt.enabled === false` → 插件级显式关闭(只用平台转写)
32
+ * - `tools.media.audio.enabled === false` → 框架级关闭
33
+ * - `models` 为空 → 未配置
34
+ *
35
+ * 仅做存在性探测控制流程;provider 解析与实际调用由 transcribeAudioFile 完成。
34
36
  */
35
- export function resolveSTTConfig(cfg: Record<string, unknown>): STTConfig | null {
37
+ export function isFrameworkSttConfigured(cfg: Record<string, unknown>): boolean {
36
38
  const channels = asRecord(cfg.channels);
37
39
  const qqbot = asRecord(channels?.qqbot);
38
- const sttCfg = asRecord(qqbot?.stt);
39
-
40
- // 显式禁用
41
- if (sttCfg?.enabled === false) {
42
- return null;
43
- }
44
-
45
- const models = asRecord(cfg.models);
46
- const providers = asRecord(models?.providers);
47
-
48
- // 1. 插件级 STT 配置
49
- if (sttCfg) {
50
- const providerId = readString(sttCfg, 'provider') ?? 'openai';
51
- const providerCfg = asRecord(providers?.[providerId]);
52
- const baseUrl = readString(sttCfg, 'baseUrl') ?? readString(providerCfg, 'baseUrl');
53
- const apiKey = readString(sttCfg, 'apiKey') ?? readString(providerCfg, 'apiKey');
54
- const model = readString(sttCfg, 'model') ?? 'whisper-1';
55
- if (baseUrl && apiKey) {
56
- return { enabled: true, baseUrl: baseUrl.replace(/\/+$/, ''), apiKey, model };
57
- }
40
+ if (asRecord(qqbot?.stt)?.enabled === false) {
41
+ return false;
58
42
  }
59
-
60
- // 2. 框架级 audio model fallback
61
43
  const tools = asRecord(cfg.tools);
62
44
  const media = asRecord(tools?.media);
63
45
  const audio = asRecord(media?.audio);
64
- const audioModels = audio?.models;
65
- const audioModelEntry = Array.isArray(audioModels) ? asRecord(audioModels[0]) : undefined;
66
- if (audioModelEntry) {
67
- const providerId = readString(audioModelEntry, 'provider') ?? 'openai';
68
- const providerCfg = asRecord(providers?.[providerId]);
69
- const baseUrl = readString(audioModelEntry, 'baseUrl') ?? readString(providerCfg, 'baseUrl');
70
- const apiKey = readString(audioModelEntry, 'apiKey') ?? readString(providerCfg, 'apiKey');
71
- const model = readString(audioModelEntry, 'model') ?? 'whisper-1';
72
- if (baseUrl && apiKey) {
73
- return { enabled: true, baseUrl: baseUrl.replace(/\/+$/, ''), apiKey, model };
74
- }
46
+ if (!audio || audio.enabled === false) {
47
+ return false;
75
48
  }
49
+ return Array.isArray(audio.models) && audio.models.length > 0;
50
+ }
76
51
 
77
- return null;
52
+ /**
53
+ * 检测已废弃的插件级 STT 凭证(channels.qqbot.stt.provider/baseUrl/apiKey/model)。
54
+ * 2026-10 起凭证统一走框架 `tools.media.audio.models`,旧键被忽略;
55
+ * 返回 true 时调用方打一次性迁移提示。
56
+ */
57
+ export function hasLegacySttCredentials(cfg: Record<string, unknown>): boolean {
58
+ const channels = asRecord(cfg.channels);
59
+ const qqbot = asRecord(channels?.qqbot);
60
+ const stt = asRecord(qqbot?.stt);
61
+ if (!stt) return false;
62
+ return ['provider', 'baseUrl', 'apiKey', 'model'].some(
63
+ (key) => typeof stt[key] === 'string' && (stt[key] as string).trim().length > 0,
64
+ );
78
65
  }
79
66
 
80
67
  /**
81
- * 调用 STT 服务转录音频文件
68
+ * 经框架音频理解管线转录本地音频文件。
69
+ * 返回修剪后的转录文本;无文本返回 null。
70
+ * provider 缺失/调用失败会抛错,由调用方捕获后走平台转写兜底。
82
71
  */
83
- export async function transcribeAudio(
72
+ export async function transcribeAudioViaFramework(
84
73
  audioPath: string,
85
74
  cfg: Record<string, unknown>,
86
75
  ): Promise<string | null> {
87
- const sttCfg = resolveSTTConfig(cfg);
88
- if (!sttCfg) {
89
- return null;
90
- }
91
-
92
- const fileBuffer = fs.readFileSync(audioPath);
93
- const fileName = sanitizeFileName(path.basename(audioPath));
94
- const mime = guessMimeType(fileName);
95
-
96
- const form = new FormData();
97
- form.append('file', new Blob([fileBuffer], { type: mime }), fileName);
98
- form.append('model', sttCfg.model);
99
-
100
- const resp = await fetch(`${sttCfg.baseUrl}/audio/transcriptions`, {
101
- method: 'POST',
102
- headers: { Authorization: `Bearer ${sttCfg.apiKey}` },
103
- body: form,
76
+ const result = await transcribeAudioFile({
77
+ filePath: audioPath,
78
+ cfg: cfg as unknown as TranscribeParams['cfg'],
79
+ mime: guessMimeType(audioPath),
104
80
  });
105
-
106
- if (!resp.ok) {
107
- const detail = await resp.text().catch(() => '');
108
- throw new Error(`STT failed (HTTP ${resp.status}): ${detail.slice(0, 300)}`);
109
- }
110
-
111
- const result = (await resp.json()) as { text?: string };
112
81
  return result.text?.trim() || null;
113
82
  }
114
83
 
@@ -121,18 +90,6 @@ function asRecord(value: unknown): Record<string, unknown> | undefined {
121
90
  return undefined;
122
91
  }
123
92
 
124
- function readString(obj: Record<string, unknown> | undefined, key: string): string | undefined {
125
- const val = obj?.[key];
126
- if (typeof val === 'string' && val.trim()) {
127
- return val.trim();
128
- }
129
- return undefined;
130
- }
131
-
132
- function sanitizeFileName(name: string): string {
133
- return name.replace(/[^a-zA-Z0-9._-]/g, '_');
134
- }
135
-
136
93
  function guessMimeType(fileName: string): string {
137
94
  const ext = path.extname(fileName).toLowerCase();
138
95
  const mimeMap: Record<string, string> = {