mioku-plugin-chat 2.8.0 → 2.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/configs/personalization.ts +1 -1
- package/core/chat-engine.ts +80 -17
- package/core/prompt.ts +25 -45
- package/core/tools/info.ts +9 -9
- package/core/tools/web.ts +2 -2
- package/handlers/message.ts +5 -5
- package/package.json +1 -1
|
@@ -85,7 +85,7 @@ export const PERSONALIZATION_CONFIG: {
|
|
|
85
85
|
|
|
86
86
|
replyStyle: {
|
|
87
87
|
baseStyle:
|
|
88
|
-
"Casual and cute,
|
|
88
|
+
"Casual and cute,can occasionally mix in a small amount of natural everyday Japanese words, but should not heavily rely on Japanese. Do not end sentences with commas or periods.",
|
|
89
89
|
multipleStyles: [
|
|
90
90
|
"Playing cute, likes to add 'w' at the end of cute phrases, commonly used to replace sentence-ending particles such as '呀'.",
|
|
91
91
|
"Hometown dialect mode, can occasionally use a small amount of natural everyday Japanese expressions in replies, and starts replies with '呐'. Avoid ending sentences with commas or periods.",
|
package/core/chat-engine.ts
CHANGED
|
@@ -1,7 +1,4 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
AIInstance,
|
|
3
|
-
SessionToolDefinition,
|
|
4
|
-
} from "mioku";
|
|
1
|
+
import type { AIInstance, SessionToolDefinition } from "mioku";
|
|
5
2
|
import { logger } from "mioki";
|
|
6
3
|
import type { AITool } from "mioku";
|
|
7
4
|
import type {
|
|
@@ -11,16 +8,10 @@ import type {
|
|
|
11
8
|
ChatResult,
|
|
12
9
|
} from "../types";
|
|
13
10
|
import type { HumanizeEngine } from "../humanize";
|
|
14
|
-
import type {
|
|
15
|
-
StaticPromptContext,
|
|
16
|
-
DynamicPromptContext,
|
|
17
|
-
} from "./prompt";
|
|
11
|
+
import type { StaticPromptContext, DynamicPromptContext } from "./prompt";
|
|
18
12
|
import type { SkillSessionManager } from "../manage/skill-session";
|
|
19
13
|
import { createTools } from "./tools";
|
|
20
|
-
import {
|
|
21
|
-
buildStaticSystemPrompt,
|
|
22
|
-
buildDynamicUserContext,
|
|
23
|
-
} from "./prompt";
|
|
14
|
+
import { buildStaticSystemPrompt, buildDynamicUserContext } from "./prompt";
|
|
24
15
|
import type { PromptCtxForRunChat } from "../manage/types";
|
|
25
16
|
import {
|
|
26
17
|
isExternalSkillAllowed,
|
|
@@ -86,7 +77,6 @@ export async function runChat(
|
|
|
86
77
|
|
|
87
78
|
const staticCtx: StaticPromptContext = {
|
|
88
79
|
config: promptCtx.config,
|
|
89
|
-
botNickname: promptCtx.botNickname,
|
|
90
80
|
aiService: promptCtx.aiService,
|
|
91
81
|
enableExternalSkills: promptCtx.config.enableExternalSkills,
|
|
92
82
|
triggerSkillRole: toolCtx.triggerSkillRole,
|
|
@@ -420,7 +410,9 @@ function estimateChatHistoryTokens(history: ChatMessage[]): number {
|
|
|
420
410
|
);
|
|
421
411
|
}
|
|
422
412
|
|
|
423
|
-
function estimateMessageContentTokens(
|
|
413
|
+
function estimateMessageContentTokens(
|
|
414
|
+
messages: Array<{ content?: unknown }>,
|
|
415
|
+
): number {
|
|
424
416
|
return messages.reduce(
|
|
425
417
|
(sum, message) => sum + estimateContentTokens(message.content),
|
|
426
418
|
0,
|
|
@@ -474,7 +466,10 @@ function buildCurrentMessages(
|
|
|
474
466
|
if (currentUserMessages.length > 0) {
|
|
475
467
|
messages.push(
|
|
476
468
|
...prependDynamicContextToFirstUserMessage(
|
|
477
|
-
attachImagesToCurrentUserMessages(
|
|
469
|
+
attachImagesToCurrentUserMessages(
|
|
470
|
+
currentUserMessages,
|
|
471
|
+
pendingImageUrls,
|
|
472
|
+
),
|
|
478
473
|
dynamicUserContext,
|
|
479
474
|
pendingImageUrls,
|
|
480
475
|
),
|
|
@@ -651,6 +646,7 @@ function createExternalSkillRuntimeContext(toolCtx: ToolContext): any {
|
|
|
651
646
|
rawEvent,
|
|
652
647
|
session_id: toolCtx.sessionId,
|
|
653
648
|
trigger_role: toolCtx.triggerSkillRole,
|
|
649
|
+
isMultimodal: Boolean(toolCtx.config?.isMultimodal),
|
|
654
650
|
};
|
|
655
651
|
}
|
|
656
652
|
|
|
@@ -722,9 +718,76 @@ function cleanMarkers(text: string): string {
|
|
|
722
718
|
.replace(/<Ai>\s*<think>[\s\S]*?<\/Ai>/gi, "")
|
|
723
719
|
.replace(/<||DSML||tool_calls>[\s\S]*?<\/||DSML||tool_calls>/gi, "")
|
|
724
720
|
.replace(/<||DSML||invoke[^>]*>[\s\S]*?<\/||DSML||invoke>/gi, "")
|
|
725
|
-
.replace(
|
|
721
|
+
.replace(
|
|
722
|
+
/<||DSML||parameter[^>]*>[\s\S]*?<\/||DSML||parameter>/gi,
|
|
723
|
+
"",
|
|
724
|
+
);
|
|
725
|
+
|
|
726
|
+
return sanitizeBrackets(cleaned);
|
|
727
|
+
}
|
|
728
|
+
|
|
729
|
+
const FUNCTIONAL_BRACKET_PREFIX =
|
|
730
|
+
/\[(at|reply|poke|audio|emotion):[^\]\n]*\]?/gi;
|
|
726
731
|
|
|
727
|
-
|
|
732
|
+
function sanitizeBrackets(text: string): string {
|
|
733
|
+
if (!text) return text;
|
|
734
|
+
|
|
735
|
+
const placeholders: string[] = [];
|
|
736
|
+
let working = text.replace(FUNCTIONAL_BRACKET_PREFIX, (match) => {
|
|
737
|
+
const idx = placeholders.length;
|
|
738
|
+
placeholders.push(match);
|
|
739
|
+
return `${idx}`;
|
|
740
|
+
});
|
|
741
|
+
|
|
742
|
+
let result = "";
|
|
743
|
+
let orphanLeftCount = 0;
|
|
744
|
+
let i = 0;
|
|
745
|
+
|
|
746
|
+
while (i < working.length) {
|
|
747
|
+
const ch = working[i];
|
|
748
|
+
|
|
749
|
+
if (ch === "") {
|
|
750
|
+
const endIdx = working.indexOf("", i + 1);
|
|
751
|
+
if (endIdx > i) {
|
|
752
|
+
const idx = parseInt(working.slice(i + 1, endIdx), 10);
|
|
753
|
+
if (!Number.isNaN(idx) && placeholders[idx] !== undefined) {
|
|
754
|
+
result += placeholders[idx];
|
|
755
|
+
i = endIdx + 1;
|
|
756
|
+
continue;
|
|
757
|
+
}
|
|
758
|
+
}
|
|
759
|
+
i++;
|
|
760
|
+
continue;
|
|
761
|
+
}
|
|
762
|
+
|
|
763
|
+
if (ch === "[") {
|
|
764
|
+
let j = i + 1;
|
|
765
|
+
while (j < working.length && working[j] !== "]") {
|
|
766
|
+
j++;
|
|
767
|
+
}
|
|
768
|
+
if (j < working.length) {
|
|
769
|
+
i = j + 1;
|
|
770
|
+
} else {
|
|
771
|
+
orphanLeftCount++;
|
|
772
|
+
i++;
|
|
773
|
+
}
|
|
774
|
+
continue;
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
if (ch === "]") {
|
|
778
|
+
i++;
|
|
779
|
+
continue;
|
|
780
|
+
}
|
|
781
|
+
|
|
782
|
+
result += ch;
|
|
783
|
+
i++;
|
|
784
|
+
}
|
|
785
|
+
|
|
786
|
+
if (orphanLeftCount > 0) {
|
|
787
|
+
result += "]".repeat(orphanLeftCount);
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
return result;
|
|
728
791
|
}
|
|
729
792
|
|
|
730
793
|
function removeStickerIntentLines(text: string): string {
|
package/core/prompt.ts
CHANGED
|
@@ -15,13 +15,12 @@ import type { SkillSessionManager } from "../manage/skill-session";
|
|
|
15
15
|
* Everything here is required to be identical across consecutive requests for the same bot
|
|
16
16
|
* so OpenAI's auto prompt caching can hit the system block.
|
|
17
17
|
*
|
|
18
|
-
* Concretely: persona, config flags, enabled features, allowed external skills
|
|
19
|
-
* Anything that changes turn-to-turn (time, group, history, target, emotion, replies context, etc.)
|
|
18
|
+
* Concretely: persona, config flags, enabled features, allowed external skills.
|
|
19
|
+
* Anything that changes turn-to-turn (time, group, history, target, emotion, reply style, replies context, etc.)
|
|
20
20
|
* belongs in DynamicPromptContext instead.
|
|
21
21
|
*/
|
|
22
22
|
export interface StaticPromptContext {
|
|
23
23
|
config: ChatConfig;
|
|
24
|
-
botNickname: string;
|
|
25
24
|
aiService: AIService;
|
|
26
25
|
enableExternalSkills: boolean;
|
|
27
26
|
triggerSkillRole?: SkillPermissionRole;
|
|
@@ -144,17 +143,16 @@ const REPLY_STYLE_LENGTH: Record<Strength, string> = {
|
|
|
144
143
|
};
|
|
145
144
|
|
|
146
145
|
const AUDIO_MODE_LINE: Record<Strength, string> = {
|
|
147
|
-
high:
|
|
148
|
-
"- Use voice sparingly. Only use it when spoken delivery is clearly better than text, such as a greeting, a sharp emotional reaction, or a daily phrase.",
|
|
146
|
+
high: "- Use voice sparingly. Only use it when spoken delivery is clearly better than text, such as a greeting, a sharp emotional reaction, or a daily phrase.",
|
|
149
147
|
medium:
|
|
150
148
|
"- You may use voice for greetings, reactions, calls, confirmations, or comforting words, but stay selective.",
|
|
151
149
|
low: "- When a short spoken reaction would make the conversation feel more natural or vivid, you can use voice more freely.",
|
|
152
150
|
};
|
|
153
151
|
|
|
154
152
|
const MARKDOWN_MODE_LINE: Record<Strength, string> = {
|
|
155
|
-
high:
|
|
156
|
-
|
|
157
|
-
|
|
153
|
+
high: "- Prefer normal chat text. Use Markdown only when the reply truly needs structured presentation, such as a tutorial, comparison, detailed explanation, code sample or processing large amounts of data, such as after a web search or viewing a webpage.",
|
|
154
|
+
medium:
|
|
155
|
+
"- Use Markdown when your responses require a structured presentation.",
|
|
158
156
|
low: "- Use Markdown freely where it can make your responses clearer.",
|
|
159
157
|
};
|
|
160
158
|
|
|
@@ -175,15 +173,13 @@ const TOOL_INTENSITY_BLOCK: Record<Strength, string> = {
|
|
|
175
173
|
};
|
|
176
174
|
|
|
177
175
|
const EMOJI_MODE_LINE: Record<Strength, string> = {
|
|
178
|
-
high:
|
|
179
|
-
"- Do not use stickers in consecutive chat turns. Use a sticker only when you are in an emotionally intense state.",
|
|
176
|
+
high: "- Do not use stickers in consecutive chat turns. Use a sticker only when you are in an emotionally intense state.",
|
|
180
177
|
medium: "- Do not use stickers in consecutive chat turns.",
|
|
181
178
|
low: "- Feel free to use a sticker whenever you want.",
|
|
182
179
|
};
|
|
183
180
|
|
|
184
181
|
const WEB_SEARCH_LINE: Record<Strength, string> = {
|
|
185
|
-
high:
|
|
186
|
-
"- When facts may be outdated or uncertain, proactively call web_search instead of guessing.",
|
|
182
|
+
high: "- When facts may be outdated or uncertain, proactively call web_search instead of guessing.",
|
|
187
183
|
medium: "- Use web_search when current or external info is needed.",
|
|
188
184
|
low: "- Use web_search only when the user explicitly needs external/current information.",
|
|
189
185
|
};
|
|
@@ -210,7 +206,6 @@ export function buildStaticSystemPrompt(ctx: StaticPromptContext): string {
|
|
|
210
206
|
const sections: string[] = [];
|
|
211
207
|
|
|
212
208
|
sections.push(buildPersonaSection(ctx.config.persona));
|
|
213
|
-
sections.push(buildReplyStyleSection(ctx.config, ctx.botNickname, lengthStrength));
|
|
214
209
|
sections.push(
|
|
215
210
|
buildResponseFormatSection(
|
|
216
211
|
ctx,
|
|
@@ -244,7 +239,7 @@ function buildReplyStyleSection(
|
|
|
244
239
|
### Behavior Guidelines
|
|
245
240
|
- You are a participant in this chat, not an assistant. Chat naturally like a real person.
|
|
246
241
|
- ${REPLY_STYLE_LENGTH[lengthStrength]}
|
|
247
|
-
- Match the language used by others in the chat
|
|
242
|
+
- Match the language used by others in the chat.
|
|
248
243
|
- Don't repeat yourself or echo what others just said.
|
|
249
244
|
- **NEVER use action descriptions like *xxx* or (xxx) — just speak as a normal person would**
|
|
250
245
|
- **${markdownBehaviorLine(config)}**
|
|
@@ -272,31 +267,24 @@ function buildResponseFormatSection(
|
|
|
272
267
|
|
|
273
268
|
lines.push(`Your text response IS your reply to the chat. It will be sent directly as a message.
|
|
274
269
|
- **IMPORTANT: Output ONLY your final reply text. Do NOT include your thinking process, reasoning, analysis, or internal thoughts.**
|
|
275
|
-
- Do NOT prefix your response with phrases like "Let me think", "I should", "I need to", "Based on", "Looking at", etc.
|
|
276
270
|
- Do NOT explain what you're doing or why. Just say what you want to say directly.
|
|
277
|
-
- **MULTIPLE MESSAGES
|
|
271
|
+
- **MULTIPLE MESSAGES: Each line (separated by Enter/Return) will be sent as a SEPARATE message.**
|
|
278
272
|
- If you want to send multiple messages, just press Enter and write the next line
|
|
279
|
-
- Each line = one message sent to the chat
|
|
280
273
|
- **If your reply has multiple sentences or different points, ALWAYS use real line breaks to separate them**
|
|
281
|
-
|
|
282
|
-
- **MESSAGE ORDER MATTERS**: messages are sent top-to-bottom, one line at a time.
|
|
283
|
-
- For action markers like [] or [audio:...], put them on their own line when they are meant to be a separate action.
|
|
274
|
+
- For action markers like [] , put them on their own line when they are meant to be a separate action.
|
|
284
275
|
|
|
285
276
|
- **SPECIAL ACTIONS in your text (auto-parsed and removed from message):**
|
|
286
277
|
- Use [at:123456] in your text to @ someone (123456 is the QQ number)
|
|
287
|
-
- Use [poke:123456] in your text to poke someone. IMPORTANT: when you plan to poke a user, DON't
|
|
278
|
+
- Use [poke:123456] in your text to poke someone. IMPORTANT: when you plan to poke a user, DON't describe your poke actions.
|
|
288
279
|
- Use [reply:123456] at the START of a line to quote-reply that message (123456 is message_id)
|
|
289
|
-
- **You can use MULTIPLE [reply:xxx] markers in different lines to quote multiple messages
|
|
290
|
-
- These markers will be automatically parsed and removed from your sent message`);
|
|
280
|
+
- **You can use MULTIPLE [reply:xxx] markers in different lines to quote multiple messages!**`);
|
|
291
281
|
|
|
292
282
|
if (ctx.config.audio?.enabled && ctx.config.audio.baseUrl?.trim()) {
|
|
293
283
|
lines.push(`
|
|
294
284
|
### Optional Voice Message Format
|
|
295
285
|
- You MAY optionally send one voice message by writing [audio:content]
|
|
296
286
|
- Audio is OPTIONAL. Do NOT use it in every reply
|
|
297
|
-
The voice message function sends plain text and cannot be used for singing
|
|
298
|
-
- Put [audio:...] on its own line when you want it sent as a separate message in sequence
|
|
299
|
-
- Example: "[audio:おはようー]"
|
|
287
|
+
The voice message function sends plain text and cannot be used for singing。
|
|
300
288
|
${AUDIO_MODE_LINE[audioStrength]}`);
|
|
301
289
|
}
|
|
302
290
|
|
|
@@ -316,9 +304,6 @@ ${MARKDOWN_MODE_LINE[markdownStrength]}
|
|
|
316
304
|
lines.push(`
|
|
317
305
|
### Tool Calling Format
|
|
318
306
|
- When you decide to use a tool, you MUST use the structured tool_calls mechanism provided by the API
|
|
319
|
-
- Do NOT output tool calls, tool names, or tool arguments in your reply text under any circumstances
|
|
320
|
-
- Do NOT use XML, JSON, or any text format to describe tool calls — only use the API's tool_calls field
|
|
321
|
-
- Each tool's description contains its own usage guidance; read those before calling a tool. If a tool's description says "use only when X" or "do not call for every question", respect that.
|
|
322
307
|
- web_search and web_read_page are limited per conversation; do not retry excessively`);
|
|
323
308
|
|
|
324
309
|
appendEmojiSection(lines, ctx, emojiStrength);
|
|
@@ -343,9 +328,7 @@ function appendEmojiSection(
|
|
|
343
328
|
### Optional Sticker / Emoji Format
|
|
344
329
|
- You MAY optionally request one matching sticker by writing exactly [] on its own line
|
|
345
330
|
${EMOJI_MODE_LINE[emojiStrength]}
|
|
346
|
-
- Never put an emotion, label, character, or any other text inside the brackets
|
|
347
|
-
- Output at most one [] block in a turn
|
|
348
|
-
- A separate sticker selection agent will read the current conversation and choose the actual sticker`);
|
|
331
|
+
- Never put an emotion, label, character, or any other text inside the brackets`);
|
|
349
332
|
}
|
|
350
333
|
|
|
351
334
|
function appendExternalSkillsSection(
|
|
@@ -455,6 +438,9 @@ export function buildDynamicUserContext(ctx: DynamicPromptContext): string {
|
|
|
455
438
|
sections.push(`## Planner's Analysis\n${ctx.plannerThoughts}`);
|
|
456
439
|
}
|
|
457
440
|
|
|
441
|
+
sections.push(
|
|
442
|
+
buildReplyStyleSection(ctx.config, ctx.botNickname, lengthStrength),
|
|
443
|
+
);
|
|
458
444
|
sections.push(buildEmotionSection(ctx));
|
|
459
445
|
|
|
460
446
|
return sections.join("\n\n");
|
|
@@ -540,7 +526,6 @@ function buildPokedGuidance(length: Strength): string[] {
|
|
|
540
526
|
return [
|
|
541
527
|
"Someone pokes you in a group, probably out of non-malicious play or to draw your attention to what happened in the group chat.",
|
|
542
528
|
"Don't make a fuss about replying, just observe whether the chat history in the group has noteworthy content, and if not, simply say hello or express concern to the user.",
|
|
543
|
-
"Reply naturally in combination with the context, don't say something like \"怎么又来戳我了\"",
|
|
544
529
|
POKED_LENGTH[length],
|
|
545
530
|
].filter(Boolean);
|
|
546
531
|
}
|
|
@@ -622,13 +607,8 @@ function buildChatHistorySection(ctx: DynamicPromptContext): string {
|
|
|
622
607
|
}
|
|
623
608
|
flushAssistant();
|
|
624
609
|
|
|
625
|
-
return `## Recent Context
|
|
626
|
-
Just the last few messages - don't overthink it or dig into old conversations:
|
|
627
|
-
|
|
610
|
+
return `## Recent Context
|
|
628
611
|
${mergedLines.join("\n")}
|
|
629
|
-
|
|
630
|
-
Note: Messages may contain media tags like [image:描述], [video:描述], [forward:摘要], [card:摘要], or [group_notice:摘要]. If you need detailed information about an image or video, use the view_media tool with the message ID.
|
|
631
|
-
|
|
632
612
|
-- DON'T repeat yourself or bring up old topics - focus on what's being said right now. --`;
|
|
633
613
|
}
|
|
634
614
|
|
|
@@ -657,7 +637,7 @@ ${messageBlocks.join("\n")}
|
|
|
657
637
|
IMPORTANT: You do NOT need to reply to each person or each message above. Give ONE casual response to the group as a whole.`;
|
|
658
638
|
}
|
|
659
639
|
|
|
660
|
-
return `## >>> Target Message
|
|
640
|
+
return `## >>> Target Message <<<
|
|
661
641
|
[${timeStr}] ${target.userName}(${target.userId}, ${target.userRole}${target.userTitle ? `, ${target.userTitle}` : ""})${msgIdStr}: ${target.content}`;
|
|
662
642
|
}
|
|
663
643
|
|
|
@@ -687,7 +667,7 @@ function buildEmotionSection(ctx: DynamicPromptContext): string {
|
|
|
687
667
|
lines.push(`Available emotions: ${availableEmotions.join(", ")}`);
|
|
688
668
|
}
|
|
689
669
|
lines.push(
|
|
690
|
-
"You may switch your emotion state by writing [emotion:emotion_name].
|
|
670
|
+
"You may switch your emotion state by writing [emotion:emotion_name].",
|
|
691
671
|
);
|
|
692
672
|
|
|
693
673
|
if (examples.length > 0) {
|
|
@@ -702,7 +682,9 @@ function buildEmotionSection(ctx: DynamicPromptContext): string {
|
|
|
702
682
|
}
|
|
703
683
|
|
|
704
684
|
function normalizeEmotionName(value: unknown): string {
|
|
705
|
-
return String(value || "")
|
|
685
|
+
return String(value || "")
|
|
686
|
+
.trim()
|
|
687
|
+
.toLowerCase();
|
|
706
688
|
}
|
|
707
689
|
|
|
708
690
|
function normalizeEmotionExamples(value: unknown): string[] {
|
|
@@ -744,7 +726,5 @@ export function buildRecallMemoryFeatureSection(config: ChatConfig): string {
|
|
|
744
726
|
return `
|
|
745
727
|
### Memory Recall Tool
|
|
746
728
|
- recall_memory: Delegate recall to a memory worker model. Pass a clear recall question and let the worker search historical logs.
|
|
747
|
-
- Use recall_memory ONLY when there is explicit need to recall past content and required information is clearly missing from current context
|
|
748
|
-
- Do NOT call recall_memory for every question.
|
|
749
|
-
- The worker returns historical logs with timestamps; treat them as past records, not newly sent messages.`;
|
|
729
|
+
- Use recall_memory ONLY when there is explicit need to recall past content and required information is clearly missing from current context.`;
|
|
750
730
|
}
|
package/core/tools/info.ts
CHANGED
|
@@ -161,11 +161,11 @@ export function createInfoTools(toolCtx: ToolContext): AITool[] {
|
|
|
161
161
|
);
|
|
162
162
|
}
|
|
163
163
|
|
|
164
|
-
const
|
|
165
|
-
if (!
|
|
164
|
+
const visionAI = toolCtx.aiService.getInstanceByRole?.("vision") ?? toolCtx.aiService.getDefault();
|
|
165
|
+
if (!visionAI) return { error: "AI instance not available" };
|
|
166
166
|
|
|
167
167
|
const result = await describeImage(
|
|
168
|
-
|
|
168
|
+
visionAI,
|
|
169
169
|
media.url,
|
|
170
170
|
toolCtx.config.multimodalWorkingModel,
|
|
171
171
|
toolCtx.event?.raw_message || undefined,
|
|
@@ -214,11 +214,11 @@ export function createInfoTools(toolCtx: ToolContext): AITool[] {
|
|
|
214
214
|
);
|
|
215
215
|
}
|
|
216
216
|
|
|
217
|
-
const
|
|
218
|
-
if (!
|
|
217
|
+
const visionAI = toolCtx.aiService.getInstanceByRole?.("vision") ?? toolCtx.aiService.getDefault();
|
|
218
|
+
if (!visionAI) return { error: "AI instance not available" };
|
|
219
219
|
|
|
220
220
|
const summary = await summarizeVideoContent(videoFile.path, videoFile.byteSize, {
|
|
221
|
-
ai,
|
|
221
|
+
ai: visionAI,
|
|
222
222
|
multimodalWorkingModel: toolCtx.config.multimodalWorkingModel,
|
|
223
223
|
logger: {
|
|
224
224
|
info: (m) => logger.info(m),
|
|
@@ -269,11 +269,11 @@ export function createInfoTools(toolCtx: ToolContext): AITool[] {
|
|
|
269
269
|
}
|
|
270
270
|
|
|
271
271
|
const { describeImage } = await import("../multimodal");
|
|
272
|
-
const
|
|
273
|
-
if (!
|
|
272
|
+
const visionAI = toolCtx.aiService.getInstanceByRole?.("vision") ?? toolCtx.aiService.getDefault();
|
|
273
|
+
if (!visionAI) return { error: "AI instance not available" };
|
|
274
274
|
|
|
275
275
|
const result = await describeImage(
|
|
276
|
-
|
|
276
|
+
visionAI,
|
|
277
277
|
avatarUrl,
|
|
278
278
|
toolCtx.config.multimodalWorkingModel,
|
|
279
279
|
`User ${args.user_id}'s QQ avatar`,
|
package/core/tools/web.ts
CHANGED
|
@@ -234,7 +234,7 @@ export function createWebReadPageTool(toolCtx: ToolContext): AITool {
|
|
|
234
234
|
handler: async (args) => {
|
|
235
235
|
try {
|
|
236
236
|
const ai = toolCtx.config.webReader.useWorkingModel
|
|
237
|
-
? toolCtx.aiService.getDefault()
|
|
237
|
+
? (toolCtx.aiService.getInstanceByRole?.("working") ?? toolCtx.aiService.getDefault())
|
|
238
238
|
: undefined;
|
|
239
239
|
if (toolCtx.config.webReader.useWorkingModel && !ai) {
|
|
240
240
|
return { success: false, error: "AI instance not available" };
|
|
@@ -275,7 +275,7 @@ export function createRecallMemoryTool(toolCtx: ToolContext): AITool {
|
|
|
275
275
|
const question = String(args?.question || "").trim();
|
|
276
276
|
if (!question) return { success: false, error: "question is required" };
|
|
277
277
|
|
|
278
|
-
const ai = toolCtx.aiService.getDefault();
|
|
278
|
+
const ai = toolCtx.aiService.getInstanceByRole?.("working") ?? toolCtx.aiService.getDefault();
|
|
279
279
|
if (!ai) return { success: false, error: "AI instance not available" };
|
|
280
280
|
|
|
281
281
|
const groupHistoryLimit = resolveGroupRecallLimit(toolCtx.config.memory?.groupHistoryLimit);
|
package/handlers/message.ts
CHANGED
|
@@ -68,11 +68,11 @@ export function createMessageHandler(
|
|
|
68
68
|
|
|
69
69
|
// 媒体分析
|
|
70
70
|
if (isGroup && groupId && e.message && !isMediaAnalysisBlocked(cfg, userId)) {
|
|
71
|
-
const
|
|
71
|
+
const visionAI = pluginCtx.visionAIInstance ?? pluginCtx.aiService.getDefault();
|
|
72
72
|
const bot = ctx.pickBot(e.self_id) as any;
|
|
73
|
-
const mediaOptions =
|
|
73
|
+
const mediaOptions = visionAI
|
|
74
74
|
? buildHistoryMediaProcessingOptions(
|
|
75
|
-
|
|
75
|
+
visionAI,
|
|
76
76
|
cfg,
|
|
77
77
|
pluginCtx.db,
|
|
78
78
|
bot,
|
|
@@ -91,13 +91,13 @@ export function createMessageHandler(
|
|
|
91
91
|
)
|
|
92
92
|
: undefined;
|
|
93
93
|
|
|
94
|
-
if (
|
|
94
|
+
if (visionAI && cfg.enableMediaRecognition) {
|
|
95
95
|
const { processImage } = await import("../core/media/image-analyzer");
|
|
96
96
|
for (const seg of e.message) {
|
|
97
97
|
if (seg.type === "image") {
|
|
98
98
|
const imageUrl = getSegmentUrl(seg);
|
|
99
99
|
if (imageUrl) {
|
|
100
|
-
processImage(
|
|
100
|
+
processImage(visionAI, imageUrl, cfg.multimodalWorkingModel, pluginCtx.db, {
|
|
101
101
|
runAIRequest: (request) =>
|
|
102
102
|
pluginCtx.runWithRateLimitGuard(request, {
|
|
103
103
|
userId,
|