@vellumai/assistant 0.9.0-dev.202606191938.289c607 → 0.9.0-dev.202606191950.a2e5513

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.9.0-dev.202606191938.289c607",
3
+ "version": "0.9.0-dev.202606191950.a2e5513",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -372,7 +372,9 @@ describe("projectSkillTools", () => {
372
372
  previouslyActiveSkillIds: sessionState,
373
373
  });
374
374
 
375
- // Tool definitions are no longer sent to the LLM — tools are invoked via skill_execute dispatch.
375
+ // Tool definitions are not sent to the LLM here — tools are invoked via
376
+ // skill_execute dispatch; weak-open-model first-class exposure resolves
377
+ // defs from the registry by name in createResolveToolsCallback.
376
378
  expect(result.toolDefinitions).toEqual([]);
377
379
  expect(result.allowedToolNames).toEqual(
378
380
  new Set(["deploy_run", "deploy_status"]),
@@ -0,0 +1,99 @@
1
+ /**
2
+ * Tests for resolveFirstClassSkillDefs: weak open models get loaded skill tools
3
+ * exposed first-class (resolved from the registry by name), capable models do
4
+ * not, and the turn allowlist still gates which defs are included.
5
+ */
6
+
7
+ import { afterEach, beforeEach, describe, expect, test } from "bun:test";
8
+
9
+ import { resolveFirstClassSkillDefs } from "../daemon/conversation-tool-setup.js";
10
+ import { RiskLevel } from "../permissions/types.js";
11
+ import { registerSkillTools, unregisterSkillTools } from "../tools/registry.js";
12
+ import type { Tool } from "../tools/types.js";
13
+
14
+ const MINIMAX = "accounts/fireworks/models/minimax-m3";
15
+ const CLAUDE = "claude-opus-4-8";
16
+
17
+ const SKILL_ID = "first-class-test-skill";
18
+
19
+ function makeSkillTool(name: string): Tool {
20
+ return {
21
+ name,
22
+ description: `Test tool ${name}`,
23
+ category: "testing",
24
+ defaultRiskLevel: RiskLevel.Low,
25
+ executionTarget: "host",
26
+ input_schema: {
27
+ type: "object",
28
+ properties: { content: { type: "string" } },
29
+ required: ["content"],
30
+ },
31
+ execute: async () => ({ content: "", isError: false }),
32
+ };
33
+ }
34
+
35
+ describe("resolveFirstClassSkillDefs", () => {
36
+ beforeEach(() => {
37
+ registerSkillTools(SKILL_ID, [
38
+ makeSkillTool("document_update"),
39
+ makeSkillTool("document_create"),
40
+ ]);
41
+ });
42
+
43
+ afterEach(() => {
44
+ unregisterSkillTools(SKILL_ID);
45
+ });
46
+
47
+ const allowed = new Set(["document_update", "document_create"]);
48
+ const turnAllowed = new Set([
49
+ "document_update",
50
+ "document_create",
51
+ "skill_execute",
52
+ ]);
53
+
54
+ test("exposes loaded skill tools for a weak open model", () => {
55
+ const defs = resolveFirstClassSkillDefs(allowed, turnAllowed, MINIMAX);
56
+ expect(defs.map((d) => d.name).sort()).toEqual([
57
+ "document_create",
58
+ "document_update",
59
+ ]);
60
+ });
61
+
62
+ test("returns nothing for a capable model", () => {
63
+ const defs = resolveFirstClassSkillDefs(allowed, turnAllowed, CLAUDE);
64
+ expect(defs).toEqual([]);
65
+ });
66
+
67
+ test("returns nothing when the model is unknown/absent", () => {
68
+ expect(resolveFirstClassSkillDefs(allowed, turnAllowed, null)).toEqual([]);
69
+ expect(resolveFirstClassSkillDefs(allowed, turnAllowed, undefined)).toEqual(
70
+ [],
71
+ );
72
+ });
73
+
74
+ test("respects the turn allowlist (subagent / exclude gating)", () => {
75
+ // Only document_update is allowed this turn — document_create is filtered.
76
+ const restricted = new Set(["document_update", "skill_execute"]);
77
+ const defs = resolveFirstClassSkillDefs(allowed, restricted, MINIMAX);
78
+ expect(defs.map((d) => d.name)).toEqual(["document_update"]);
79
+ });
80
+
81
+ test("skips names with no registered tool", () => {
82
+ const withGhost = new Set([...allowed, "not_registered"]);
83
+ const turn = new Set([...turnAllowed, "not_registered"]);
84
+ const defs = resolveFirstClassSkillDefs(withGhost, turn, MINIMAX);
85
+ expect(defs.map((d) => d.name).sort()).toEqual([
86
+ "document_create",
87
+ "document_update",
88
+ ]);
89
+ });
90
+
91
+ test("each exposed def carries its real scalar schema (single-escape)", () => {
92
+ const defs = resolveFirstClassSkillDefs(allowed, turnAllowed, MINIMAX);
93
+ const update = defs.find((d) => d.name === "document_update");
94
+ expect(update?.input_schema).toMatchObject({
95
+ properties: { content: { type: "string" } },
96
+ required: ["content"],
97
+ });
98
+ });
99
+ });
@@ -255,6 +255,24 @@ describe("createSkillTool — unknown parameter validation", () => {
255
255
  expect(result.isError).toBe(false);
256
256
  });
257
257
 
258
+ test("strips the harness `activity` field before validation", async () => {
259
+ const hash = computeSkillVersionHash(tempDir);
260
+ const tool = createSkillTool(
261
+ makeEntry({ executor: "echo.ts" }),
262
+ tempDir,
263
+ hash,
264
+ );
265
+
266
+ // A first-class skill-tool call may carry `activity` (the harness progress
267
+ // field). It must not be rejected as an unknown parameter.
268
+ const result = await tool.execute(
269
+ { query: "hello", activity: "Doing the thing" },
270
+ makeContext(),
271
+ );
272
+
273
+ expect(result.isError).toBe(false);
274
+ });
275
+
258
276
  test("allows empty input when schema has no required fields", async () => {
259
277
  const hash = computeSkillVersionHash(tempDir);
260
278
  const tool = createSkillTool(
@@ -360,7 +360,10 @@ export function projectSkillTools(
360
360
  return { toolDefinitions: [], allowedToolNames: new Set() };
361
361
  }
362
362
 
363
- // Tool definitions are no longer sent to the LLM — tools are invoked via skill_execute dispatch.
363
+ // Tool definitions are not sent to the LLM here — tools are invoked via
364
+ // skill_execute dispatch (capable models), and the weak-open-model
365
+ // first-class exposure resolves the registered defs from the registry in
366
+ // createResolveToolsCallback by name.
364
367
  const allToolNames = new Set<string>();
365
368
  const successfulEntries = new Map<string, string>();
366
369
  // Track skills already unregistered in the version-change branch so the
@@ -19,6 +19,7 @@ import type { PermissionPrompter } from "../permissions/prompter.js";
19
19
  import type { SecretPrompter } from "../permissions/secret-prompter.js";
20
20
  import { advisorEnabledForProfile } from "../plugins/defaults/advisor/advisor-gate.js";
21
21
  import type { Message, ToolDefinition } from "../providers/types.js";
22
+ import { isWeakOpenModel } from "../providers/weak-open-model.js";
22
23
  import { assistantEventHub } from "../runtime/assistant-event-hub.js";
23
24
  import { registerConversationSender } from "../tools/browser/browser-screencast.js";
24
25
  import type { ToolExecutor } from "../tools/executor.js";
@@ -681,6 +682,36 @@ export function isToolActiveForContext(
681
682
  return true;
682
683
  }
683
684
 
685
+ /**
686
+ * Resolve the loaded skill tools to expose first-class to weak open models.
687
+ *
688
+ * Weak open models (MiniMax, Kimi, DeepSeek, GLM) fail to serialize the nested
689
+ * `skill_execute` envelope — a large value double-escaped as a JSON string in
690
+ * `input` — and either drop it or emit it bare. Exposing each loaded skill
691
+ * tool first-class lets the model fill the tool's own scalar params directly
692
+ * (single-escape). Capable models keep the envelope-only contract: this returns
693
+ * `[]` for them, so their wire surface is unchanged.
694
+ *
695
+ * Defs are resolved from the registry by name; `skillToolNames` come from the
696
+ * projection (skill tools only), filtered by `turnAllowed` so subagent
697
+ * allowlists and the tool exclude list still apply. The generic `skill_execute`
698
+ * tool remains available as a fallback for both model tiers.
699
+ */
700
+ export function resolveFirstClassSkillDefs(
701
+ skillToolNames: Iterable<string>,
702
+ turnAllowed: ReadonlySet<string>,
703
+ resolvedModel: string | null | undefined,
704
+ ): ToolDefinition[] {
705
+ if (!isWeakOpenModel(resolvedModel)) return [];
706
+ const defs: ToolDefinition[] = [];
707
+ for (const name of skillToolNames) {
708
+ if (!turnAllowed.has(name)) continue;
709
+ const tool = getTool(name);
710
+ if (tool) defs.push(tool);
711
+ }
712
+ return defs;
713
+ }
714
+
684
715
  /**
685
716
  * Build a resolveTools callback that merges base tool definitions with
686
717
  * dynamically projected skill tools on each agent turn. Also updates
@@ -810,7 +841,30 @@ export function createResolveToolsCallback(
810
841
  }
811
842
 
812
843
  ctx.allowedToolNames = turnAllowed;
813
- const baseDefs = injectActivityField(allBaseDefs, ACTIVITY_SKIP_SET);
844
+
845
+ // Weak open models fail to serialize the nested `skill_execute` envelope
846
+ // (a large value double-escaped as a JSON string in `input`), so expose the
847
+ // loaded skill tools first-class — their scalar params are emitted directly
848
+ // (single-escape). The defs are resolved from the registry by name; the
849
+ // names are already in `turnAllowed`. The generic `skill_execute` stays
850
+ // available as a fallback, and capable models keep the envelope-only
851
+ // contract (this branch is skipped for them).
852
+ const resolvedModel = resolveConversationAttribution({
853
+ conversationId: ctx.conversationId ?? "",
854
+ currentCallSite: ctx.currentCallSite,
855
+ currentTurnOverrideProfile: ctx.currentTurnOverrideProfile,
856
+ })?.resolvedModel;
857
+ const baseDefs = [
858
+ ...injectActivityField(allBaseDefs, ACTIVITY_SKIP_SET),
859
+ ...injectActivityField(
860
+ resolveFirstClassSkillDefs(
861
+ projection.allowedToolNames,
862
+ turnAllowed,
863
+ resolvedModel,
864
+ ),
865
+ ACTIVITY_SKIP_SET,
866
+ ),
867
+ ];
814
868
 
815
869
  const config = getConfig();
816
870
  if (
@@ -46,7 +46,12 @@ export function createSkillTool(
46
46
  context: ToolContext,
47
47
  ): Promise<ToolExecutionResult> {
48
48
  const schema = entry.input_schema as Record<string, unknown> | undefined;
49
- const coercedInput = coerceStringBooleans(input, schema);
49
+ // `activity` is a harness field (the skill_execute envelope and
50
+ // first-class skill-tool exposure both surface it for progress display),
51
+ // never an inner tool parameter. Strip it before validation so a model
52
+ // that includes it on a direct call isn't rejected for an unknown field.
53
+ const { activity: _activity, ...rest } = input;
54
+ const coercedInput = coerceStringBooleans(rest, schema);
50
55
  const validation = validateInputAgainstSchema(
51
56
  entry.name,
52
57
  coercedInput,