@vellumai/assistant 0.9.0-dev.202606191938.289c607 → 0.9.0-dev.202606191950.a2e5513
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/conversation-skill-tools.test.ts +3 -1
- package/src/__tests__/first-class-skill-defs.test.ts +99 -0
- package/src/__tests__/skill-tool-factory.test.ts +18 -0
- package/src/daemon/conversation-skill-tools.ts +4 -1
- package/src/daemon/conversation-tool-setup.ts +55 -1
- package/src/tools/skills/skill-tool-factory.ts +6 -1
package/package.json
CHANGED
|
@@ -372,7 +372,9 @@ describe("projectSkillTools", () => {
|
|
|
372
372
|
previouslyActiveSkillIds: sessionState,
|
|
373
373
|
});
|
|
374
374
|
|
|
375
|
-
// Tool definitions are
|
|
375
|
+
// Tool definitions are not sent to the LLM here — tools are invoked via
|
|
376
|
+
// skill_execute dispatch; weak-open-model first-class exposure resolves
|
|
377
|
+
// defs from the registry by name in createResolveToolsCallback.
|
|
376
378
|
expect(result.toolDefinitions).toEqual([]);
|
|
377
379
|
expect(result.allowedToolNames).toEqual(
|
|
378
380
|
new Set(["deploy_run", "deploy_status"]),
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for resolveFirstClassSkillDefs: weak open models get loaded skill tools
|
|
3
|
+
* exposed first-class (resolved from the registry by name), capable models do
|
|
4
|
+
* not, and the turn allowlist still gates which defs are included.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
|
|
8
|
+
|
|
9
|
+
import { resolveFirstClassSkillDefs } from "../daemon/conversation-tool-setup.js";
|
|
10
|
+
import { RiskLevel } from "../permissions/types.js";
|
|
11
|
+
import { registerSkillTools, unregisterSkillTools } from "../tools/registry.js";
|
|
12
|
+
import type { Tool } from "../tools/types.js";
|
|
13
|
+
|
|
14
|
+
const MINIMAX = "accounts/fireworks/models/minimax-m3";
|
|
15
|
+
const CLAUDE = "claude-opus-4-8";
|
|
16
|
+
|
|
17
|
+
const SKILL_ID = "first-class-test-skill";
|
|
18
|
+
|
|
19
|
+
function makeSkillTool(name: string): Tool {
|
|
20
|
+
return {
|
|
21
|
+
name,
|
|
22
|
+
description: `Test tool ${name}`,
|
|
23
|
+
category: "testing",
|
|
24
|
+
defaultRiskLevel: RiskLevel.Low,
|
|
25
|
+
executionTarget: "host",
|
|
26
|
+
input_schema: {
|
|
27
|
+
type: "object",
|
|
28
|
+
properties: { content: { type: "string" } },
|
|
29
|
+
required: ["content"],
|
|
30
|
+
},
|
|
31
|
+
execute: async () => ({ content: "", isError: false }),
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
describe("resolveFirstClassSkillDefs", () => {
|
|
36
|
+
beforeEach(() => {
|
|
37
|
+
registerSkillTools(SKILL_ID, [
|
|
38
|
+
makeSkillTool("document_update"),
|
|
39
|
+
makeSkillTool("document_create"),
|
|
40
|
+
]);
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
afterEach(() => {
|
|
44
|
+
unregisterSkillTools(SKILL_ID);
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
const allowed = new Set(["document_update", "document_create"]);
|
|
48
|
+
const turnAllowed = new Set([
|
|
49
|
+
"document_update",
|
|
50
|
+
"document_create",
|
|
51
|
+
"skill_execute",
|
|
52
|
+
]);
|
|
53
|
+
|
|
54
|
+
test("exposes loaded skill tools for a weak open model", () => {
|
|
55
|
+
const defs = resolveFirstClassSkillDefs(allowed, turnAllowed, MINIMAX);
|
|
56
|
+
expect(defs.map((d) => d.name).sort()).toEqual([
|
|
57
|
+
"document_create",
|
|
58
|
+
"document_update",
|
|
59
|
+
]);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
test("returns nothing for a capable model", () => {
|
|
63
|
+
const defs = resolveFirstClassSkillDefs(allowed, turnAllowed, CLAUDE);
|
|
64
|
+
expect(defs).toEqual([]);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test("returns nothing when the model is unknown/absent", () => {
|
|
68
|
+
expect(resolveFirstClassSkillDefs(allowed, turnAllowed, null)).toEqual([]);
|
|
69
|
+
expect(resolveFirstClassSkillDefs(allowed, turnAllowed, undefined)).toEqual(
|
|
70
|
+
[],
|
|
71
|
+
);
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
test("respects the turn allowlist (subagent / exclude gating)", () => {
|
|
75
|
+
// Only document_update is allowed this turn — document_create is filtered.
|
|
76
|
+
const restricted = new Set(["document_update", "skill_execute"]);
|
|
77
|
+
const defs = resolveFirstClassSkillDefs(allowed, restricted, MINIMAX);
|
|
78
|
+
expect(defs.map((d) => d.name)).toEqual(["document_update"]);
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
test("skips names with no registered tool", () => {
|
|
82
|
+
const withGhost = new Set([...allowed, "not_registered"]);
|
|
83
|
+
const turn = new Set([...turnAllowed, "not_registered"]);
|
|
84
|
+
const defs = resolveFirstClassSkillDefs(withGhost, turn, MINIMAX);
|
|
85
|
+
expect(defs.map((d) => d.name).sort()).toEqual([
|
|
86
|
+
"document_create",
|
|
87
|
+
"document_update",
|
|
88
|
+
]);
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
test("each exposed def carries its real scalar schema (single-escape)", () => {
|
|
92
|
+
const defs = resolveFirstClassSkillDefs(allowed, turnAllowed, MINIMAX);
|
|
93
|
+
const update = defs.find((d) => d.name === "document_update");
|
|
94
|
+
expect(update?.input_schema).toMatchObject({
|
|
95
|
+
properties: { content: { type: "string" } },
|
|
96
|
+
required: ["content"],
|
|
97
|
+
});
|
|
98
|
+
});
|
|
99
|
+
});
|
|
@@ -255,6 +255,24 @@ describe("createSkillTool — unknown parameter validation", () => {
|
|
|
255
255
|
expect(result.isError).toBe(false);
|
|
256
256
|
});
|
|
257
257
|
|
|
258
|
+
test("strips the harness `activity` field before validation", async () => {
|
|
259
|
+
const hash = computeSkillVersionHash(tempDir);
|
|
260
|
+
const tool = createSkillTool(
|
|
261
|
+
makeEntry({ executor: "echo.ts" }),
|
|
262
|
+
tempDir,
|
|
263
|
+
hash,
|
|
264
|
+
);
|
|
265
|
+
|
|
266
|
+
// A first-class skill-tool call may carry `activity` (the harness progress
|
|
267
|
+
// field). It must not be rejected as an unknown parameter.
|
|
268
|
+
const result = await tool.execute(
|
|
269
|
+
{ query: "hello", activity: "Doing the thing" },
|
|
270
|
+
makeContext(),
|
|
271
|
+
);
|
|
272
|
+
|
|
273
|
+
expect(result.isError).toBe(false);
|
|
274
|
+
});
|
|
275
|
+
|
|
258
276
|
test("allows empty input when schema has no required fields", async () => {
|
|
259
277
|
const hash = computeSkillVersionHash(tempDir);
|
|
260
278
|
const tool = createSkillTool(
|
|
@@ -360,7 +360,10 @@ export function projectSkillTools(
|
|
|
360
360
|
return { toolDefinitions: [], allowedToolNames: new Set() };
|
|
361
361
|
}
|
|
362
362
|
|
|
363
|
-
// Tool definitions are
|
|
363
|
+
// Tool definitions are not sent to the LLM here — tools are invoked via
|
|
364
|
+
// skill_execute dispatch (capable models), and the weak-open-model
|
|
365
|
+
// first-class exposure resolves the registered defs from the registry in
|
|
366
|
+
// createResolveToolsCallback by name.
|
|
364
367
|
const allToolNames = new Set<string>();
|
|
365
368
|
const successfulEntries = new Map<string, string>();
|
|
366
369
|
// Track skills already unregistered in the version-change branch so the
|
|
@@ -19,6 +19,7 @@ import type { PermissionPrompter } from "../permissions/prompter.js";
|
|
|
19
19
|
import type { SecretPrompter } from "../permissions/secret-prompter.js";
|
|
20
20
|
import { advisorEnabledForProfile } from "../plugins/defaults/advisor/advisor-gate.js";
|
|
21
21
|
import type { Message, ToolDefinition } from "../providers/types.js";
|
|
22
|
+
import { isWeakOpenModel } from "../providers/weak-open-model.js";
|
|
22
23
|
import { assistantEventHub } from "../runtime/assistant-event-hub.js";
|
|
23
24
|
import { registerConversationSender } from "../tools/browser/browser-screencast.js";
|
|
24
25
|
import type { ToolExecutor } from "../tools/executor.js";
|
|
@@ -681,6 +682,36 @@ export function isToolActiveForContext(
|
|
|
681
682
|
return true;
|
|
682
683
|
}
|
|
683
684
|
|
|
685
|
+
/**
|
|
686
|
+
* Resolve the loaded skill tools to expose first-class to weak open models.
|
|
687
|
+
*
|
|
688
|
+
* Weak open models (MiniMax, Kimi, DeepSeek, GLM) fail to serialize the nested
|
|
689
|
+
* `skill_execute` envelope — a large value double-escaped as a JSON string in
|
|
690
|
+
* `input` — and either drop it or emit it bare. Exposing each loaded skill
|
|
691
|
+
* tool first-class lets the model fill the tool's own scalar params directly
|
|
692
|
+
* (single-escape). Capable models keep the envelope-only contract: this returns
|
|
693
|
+
* `[]` for them, so their wire surface is unchanged.
|
|
694
|
+
*
|
|
695
|
+
* Defs are resolved from the registry by name; `skillToolNames` come from the
|
|
696
|
+
* projection (skill tools only), filtered by `turnAllowed` so subagent
|
|
697
|
+
* allowlists and the tool exclude list still apply. The generic `skill_execute`
|
|
698
|
+
* tool remains available as a fallback for both model tiers.
|
|
699
|
+
*/
|
|
700
|
+
export function resolveFirstClassSkillDefs(
|
|
701
|
+
skillToolNames: Iterable<string>,
|
|
702
|
+
turnAllowed: ReadonlySet<string>,
|
|
703
|
+
resolvedModel: string | null | undefined,
|
|
704
|
+
): ToolDefinition[] {
|
|
705
|
+
if (!isWeakOpenModel(resolvedModel)) return [];
|
|
706
|
+
const defs: ToolDefinition[] = [];
|
|
707
|
+
for (const name of skillToolNames) {
|
|
708
|
+
if (!turnAllowed.has(name)) continue;
|
|
709
|
+
const tool = getTool(name);
|
|
710
|
+
if (tool) defs.push(tool);
|
|
711
|
+
}
|
|
712
|
+
return defs;
|
|
713
|
+
}
|
|
714
|
+
|
|
684
715
|
/**
|
|
685
716
|
* Build a resolveTools callback that merges base tool definitions with
|
|
686
717
|
* dynamically projected skill tools on each agent turn. Also updates
|
|
@@ -810,7 +841,30 @@ export function createResolveToolsCallback(
|
|
|
810
841
|
}
|
|
811
842
|
|
|
812
843
|
ctx.allowedToolNames = turnAllowed;
|
|
813
|
-
|
|
844
|
+
|
|
845
|
+
// Weak open models fail to serialize the nested `skill_execute` envelope
|
|
846
|
+
// (a large value double-escaped as a JSON string in `input`), so expose the
|
|
847
|
+
// loaded skill tools first-class — their scalar params are emitted directly
|
|
848
|
+
// (single-escape). The defs are resolved from the registry by name; the
|
|
849
|
+
// names are already in `turnAllowed`. The generic `skill_execute` stays
|
|
850
|
+
// available as a fallback, and capable models keep the envelope-only
|
|
851
|
+
// contract (this branch is skipped for them).
|
|
852
|
+
const resolvedModel = resolveConversationAttribution({
|
|
853
|
+
conversationId: ctx.conversationId ?? "",
|
|
854
|
+
currentCallSite: ctx.currentCallSite,
|
|
855
|
+
currentTurnOverrideProfile: ctx.currentTurnOverrideProfile,
|
|
856
|
+
})?.resolvedModel;
|
|
857
|
+
const baseDefs = [
|
|
858
|
+
...injectActivityField(allBaseDefs, ACTIVITY_SKIP_SET),
|
|
859
|
+
...injectActivityField(
|
|
860
|
+
resolveFirstClassSkillDefs(
|
|
861
|
+
projection.allowedToolNames,
|
|
862
|
+
turnAllowed,
|
|
863
|
+
resolvedModel,
|
|
864
|
+
),
|
|
865
|
+
ACTIVITY_SKIP_SET,
|
|
866
|
+
),
|
|
867
|
+
];
|
|
814
868
|
|
|
815
869
|
const config = getConfig();
|
|
816
870
|
if (
|
|
@@ -46,7 +46,12 @@ export function createSkillTool(
|
|
|
46
46
|
context: ToolContext,
|
|
47
47
|
): Promise<ToolExecutionResult> {
|
|
48
48
|
const schema = entry.input_schema as Record<string, unknown> | undefined;
|
|
49
|
-
|
|
49
|
+
// `activity` is a harness field (the skill_execute envelope and
|
|
50
|
+
// first-class skill-tool exposure both surface it for progress display),
|
|
51
|
+
// never an inner tool parameter. Strip it before validation so a model
|
|
52
|
+
// that includes it on a direct call isn't rejected for an unknown field.
|
|
53
|
+
const { activity: _activity, ...rest } = input;
|
|
54
|
+
const coercedInput = coerceStringBooleans(rest, schema);
|
|
50
55
|
const validation = validateInputAgainstSchema(
|
|
51
56
|
entry.name,
|
|
52
57
|
coercedInput,
|