@vellumai/assistant 0.12.2-dev.202609171913.b5e95d3 → 0.12.2-dev.202609172115.0b9be8f

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/node_modules/@vellumai/slack-text/src/index.ts +4 -8
  2. package/openapi.yaml +3 -3
  3. package/package.json +1 -1
  4. package/src/__tests__/btw-routes.test.ts +4 -1
  5. package/src/__tests__/conversation-runtime-assembly.test.ts +29 -0
  6. package/src/__tests__/inference-profile-session-handler.test.ts +1 -1
  7. package/src/__tests__/llm-catalog-parity.test.ts +2 -2
  8. package/src/config/__tests__/default-provider.test.ts +3 -1
  9. package/src/config/__tests__/memory-retrospective-schema.test.ts +14 -1
  10. package/src/config/bundled-skills/sequences/TOOLS.json +1 -5
  11. package/src/config/profile-text-generation.test.ts +2 -2
  12. package/src/config/schemas/llm.ts +1 -1
  13. package/src/config/schemas/memory-retrospective.ts +11 -2
  14. package/src/daemon/conversation-runtime-assembly.ts +7 -18
  15. package/src/mcp/__tests__/manager-tool-caps.test.ts +32 -20
  16. package/src/persistence/conversation-crud.ts +6 -0
  17. package/src/persistence/conversation-queries.ts +3 -16
  18. package/src/plugins/defaults/memory/v3/__tests__/pool-select.test.ts +119 -1
  19. package/src/plugins/defaults/memory/v3/orchestrate.ts +3 -1
  20. package/src/plugins/defaults/memory/v3/pool-select.ts +263 -6
  21. package/src/providers/inference/adapter-factory.ts +1 -1
  22. package/src/providers/jev/client.test.ts +10 -0
  23. package/src/providers/jev/client.ts +16 -1
  24. package/src/providers/model-catalog.ts +4 -3
  25. package/src/runtime/guardian-reply-router.ts +5 -11
  26. package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +1 -1
  27. package/src/runtime/routes/btw-routes.ts +9 -4
  28. package/src/runtime/routes/identity-routes.ts +1 -5
  29. package/src/runtime/routes/secret-routes.ts +1 -1
  30. package/src/tools/__tests__/tool-schema-root-combinator-guard.test.ts +81 -0
  31. package/src/tools/ask-question/ask-question-tool.ts +6 -4
  32. package/src/tools/document/document-tool.ts +2 -6
@@ -127,7 +127,7 @@ export async function buildSlackUserLabelMap(
127
127
  ids.map(async (id): Promise<[string, string] | undefined> => {
128
128
  try {
129
129
  const label = await resolveLabel(id);
130
- const sanitized = sanitizeOptionalLabel(label ?? undefined);
130
+ const sanitized = sanitizeSlackLabel(label ?? undefined);
131
131
  if (!sanitized || sanitized === id) return undefined;
132
132
  return [id, sanitized];
133
133
  } catch {
@@ -166,7 +166,7 @@ export async function buildSlackChannelLabelMap(
166
166
  ids.map(async (id): Promise<[string, string] | undefined> => {
167
167
  try {
168
168
  const label = await resolveLabel(id);
169
- const sanitized = sanitizeOptionalLabel(label ?? undefined);
169
+ const sanitized = sanitizeSlackLabel(label ?? undefined);
170
170
  if (!sanitized || sanitized === id) return undefined;
171
171
  return [id, sanitized];
172
172
  } catch {
@@ -222,7 +222,7 @@ function renderChannelReference(
222
222
  return `#${embeddedLabel}`;
223
223
  }
224
224
 
225
- const resolvedLabel = sanitizeOptionalLabel(
225
+ const resolvedLabel = sanitizeSlackLabel(
226
226
  options.channelLabels?.[channelId],
227
227
  );
228
228
  if (resolvedLabel && resolvedLabel !== channelId) {
@@ -320,15 +320,11 @@ export function sanitizeSlackLabel(
320
320
  function sanitizeEmbeddedSlackLabel(
321
321
  label: string | undefined,
322
322
  ): string | undefined {
323
- return sanitizeOptionalLabel(
323
+ return sanitizeSlackLabel(
324
324
  label === undefined ? undefined : decodeSlackHtmlEntities(label),
325
325
  );
326
326
  }
327
327
 
328
- function sanitizeOptionalLabel(label: string | undefined): string | undefined {
329
- return sanitizeSlackLabel(label);
330
- }
331
-
332
328
  function isSlackUserId(value: string): boolean {
333
329
  return /^[UW][A-Z0-9]+$/.test(value);
334
330
  }
package/openapi.yaml CHANGED
@@ -16556,7 +16556,7 @@ paths:
16556
16556
  type: string
16557
16557
  description:
16558
16558
  "Filter by provider id. One of: anthropic, openai, gemini, ollama, fireworks, together, openrouter,
16559
- vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside, jev,
16559
+ vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside, typesafe,
16560
16560
  vellum"
16561
16561
  responses:
16562
16562
  "200":
@@ -16829,7 +16829,7 @@ paths:
16829
16829
  type: string
16830
16830
  description:
16831
16831
  "Filter by provider. One of: anthropic, openai, gemini, ollama, fireworks, together, openrouter,
16832
- vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside, jev,
16832
+ vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside, typesafe,
16833
16833
  vellum, chatgpt"
16834
16834
  responses:
16835
16835
  "200":
@@ -38400,7 +38400,7 @@ components:
38400
38400
  - atlascloud
38401
38401
  - baseten
38402
38402
  - poolside
38403
- - jev
38403
+ - typesafe
38404
38404
  - vellum
38405
38405
  - chatgpt
38406
38406
  Auth:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.12.2-dev.202609171913.b5e95d3",
3
+ "version": "0.12.2-dev.202609172115.0b9be8f",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -338,7 +338,7 @@ describe("POST /v1/btw", () => {
338
338
  expect(options!.config!.modelIntent).toBeUndefined();
339
339
  });
340
340
 
341
- test("greeting requests pass callSite: 'emptyStateGreeting'", async () => {
341
+ test("greeting requests pass callSite: 'emptyStateGreeting' and send no tools", async () => {
342
342
  const provider = makeMockProvider();
343
343
  const session = makeMockSession(provider);
344
344
  mockGetOrCreateConversation.mockImplementationOnce(async () => session);
@@ -352,6 +352,9 @@ describe("POST /v1/btw", () => {
352
352
  expect(provider.sendMessage).toHaveBeenCalledTimes(1);
353
353
  const [, options] = provider.sendMessage.mock.calls[0];
354
354
  expect(options!.config!.callSite).toBe("emptyStateGreeting");
355
+ // The greeting targets no real conversation, so there is no cache prefix
356
+ // for tool definitions to share; they would only cost tokens.
357
+ expect(options!.tools).toEqual([]);
355
358
  });
356
359
 
357
360
  test("greeting requests include fresh turn context using the client timezone", async () => {
@@ -673,6 +673,35 @@ describe("injectChannelCapabilityContext", () => {
673
673
  const text = (result.content[0] as { type: "text"; text: string }).text;
674
674
  expect(text).not.toContain("Do NOT use markdown tables");
675
675
  });
676
+
677
+ test("injects email send CLI constraint for email channel", () => {
678
+ const caps: ChannelCapabilities = {
679
+ channel: "email",
680
+ dashboardCapable: false,
681
+ supportsDynamicUi: false,
682
+ supportsVoiceInput: false,
683
+ };
684
+
685
+ const result = injectChannelCapabilityContext(baseUserMessage, caps);
686
+ const text = (result.content[0] as { type: "text"; text: string }).text;
687
+ expect(text).toContain("Conversation text is not emailed");
688
+ expect(text).toContain("assistant email send");
689
+ expect(text).toContain("--reply-to");
690
+ });
691
+
692
+ test("does NOT inject email send CLI constraint for non-email channels", () => {
693
+ const caps: ChannelCapabilities = {
694
+ channel: "telegram",
695
+ dashboardCapable: false,
696
+ supportsDynamicUi: false,
697
+ supportsVoiceInput: false,
698
+ };
699
+
700
+ const result = injectChannelCapabilityContext(baseUserMessage, caps);
701
+ const text = (result.content[0] as { type: "text"; text: string }).text;
702
+ expect(text).not.toContain("Conversation text is not emailed");
703
+ expect(text).not.toContain("assistant email send");
704
+ });
676
705
  });
677
706
 
678
707
  // ---------------------------------------------------------------------------
@@ -389,7 +389,7 @@ describe("setInferenceProfileSession", () => {
389
389
  profiles: {
390
390
  jev: {
391
391
  source: "user",
392
- provider: "jev",
392
+ provider: "typesafe",
393
393
  model: "jev-latest",
394
394
  status: "active",
395
395
  },
@@ -234,12 +234,12 @@ describe("LLM catalog parity: daemon vs client", () => {
234
234
  });
235
235
 
236
236
  test("jev-latest opts out of chat text generation", () => {
237
- expect(catalogModelSupportsText("jev", "jev-latest")).toBe(false);
237
+ expect(catalogModelSupportsText("typesafe", "jev-latest")).toBe(false);
238
238
  expect(catalogModelSupportsText("anthropic", "claude-opus-4-8")).toBe(true);
239
239
  expect(catalogModelSupportsText("openai-compatible", "local-model")).toBe(
240
240
  true,
241
241
  );
242
- expect(DEFAULT_PROVIDER_CHOICES).not.toContain("jev");
242
+ expect(DEFAULT_PROVIDER_CHOICES).not.toContain("typesafe");
243
243
  expect(DEFAULT_PROVIDER_CHOICES).toContain("poolside");
244
244
  });
245
245
 
@@ -73,7 +73,9 @@ describe("LLMSchema.defaultProvider", () => {
73
73
  });
74
74
 
75
75
  test("rejects a structured-decision catalog provider", () => {
76
- expect(() => DefaultProviderSchema.parse({ provider: "jev" })).toThrow();
76
+ expect(() =>
77
+ DefaultProviderSchema.parse({ provider: "typesafe" }),
78
+ ).toThrow();
77
79
  });
78
80
 
79
81
  test("rejects an empty connectionName", () => {
@@ -12,10 +12,11 @@ import { describe, expect, test } from "bun:test";
12
12
  import { MemoryRetrospectiveConfigSchema } from "../schemas/memory-retrospective.js";
13
13
 
14
14
  describe("memory.retrospective config schema", () => {
15
- test("an empty block leaves retrospectives and skill improvement on", () => {
15
+ test("an empty block leaves skill improvement on and monitoring off", () => {
16
16
  const parsed = MemoryRetrospectiveConfigSchema.parse({});
17
17
  expect(parsed.enabled).toBe(true);
18
18
  expect(parsed.skillImprovement).toBe(true);
19
+ expect(parsed.skillImprovementMonitoring).toBe(false);
19
20
  expect(parsed).not.toHaveProperty("forkStrategy");
20
21
  });
21
22
 
@@ -41,6 +42,18 @@ describe("memory.retrospective config schema", () => {
41
42
  ).toBe(false);
42
43
  });
43
44
 
45
+ test("skillImprovementMonitoring is a boolean-only opt-in", () => {
46
+ const parsed = MemoryRetrospectiveConfigSchema.parse({
47
+ skillImprovementMonitoring: true,
48
+ });
49
+ expect(parsed.skillImprovementMonitoring).toBe(true);
50
+ expect(
51
+ MemoryRetrospectiveConfigSchema.safeParse({
52
+ skillImprovementMonitoring: "true",
53
+ }).success,
54
+ ).toBe(false);
55
+ });
56
+
44
57
  test("a leftover forkStrategy key is ignored", () => {
45
58
  const parsed = MemoryRetrospectiveConfigSchema.parse({
46
59
  forkStrategy: "cloning",
@@ -165,11 +165,7 @@
165
165
  },
166
166
  "description": "Replacement steps (replaces all existing steps)"
167
167
  }
168
- },
169
- "oneOf": [
170
- { "required": ["id"] },
171
- { "required": ["enrollment_id", "enrollment_action"] }
172
- ]
168
+ }
173
169
  },
174
170
  "executor": "tools/sequence-update.ts",
175
171
  "execution_target": "host"
@@ -9,7 +9,7 @@ describe("profileSupportsTextGeneration", () => {
9
9
  test("false for a structured-decision catalog model", () => {
10
10
  expect(
11
11
  profileSupportsTextGeneration(
12
- { provider: "jev", model: "jev-latest" },
12
+ { provider: "typesafe", model: "jev-latest" },
13
13
  {},
14
14
  ),
15
15
  ).toBe(false);
@@ -35,7 +35,7 @@ describe("profileSupportsTextGeneration", () => {
35
35
  profileSupportsTextGeneration(
36
36
  { mix: [{ profile: "jev" }, { profile: "balanced" }] },
37
37
  {
38
- jev: { provider: "jev", model: "jev-latest" },
38
+ jev: { provider: "typesafe", model: "jev-latest" },
39
39
  balanced: { provider: "anthropic", model: "claude-opus-4-8" },
40
40
  },
41
41
  ),
@@ -62,7 +62,7 @@ export const KNOWN_LLM_PROVIDERS = [
62
62
  "opencode",
63
63
  "baseten",
64
64
  "poolside",
65
- "jev",
65
+ "typesafe",
66
66
  // Routing identities: "vellum" = the platform-managed route (upstream
67
67
  // derived from the model at dispatch) and the catalog owner of
68
68
  // Vellum-hosted GPU models; "chatgpt" = the subscription route to OpenAI.
@@ -11,14 +11,23 @@ export const MemoryRetrospectiveConfigSchema = z
11
11
 
12
12
  skillImprovement: z
13
13
  .boolean({
14
- error:
15
- "memory.retrospective.skillImprovement must be a boolean",
14
+ error: "memory.retrospective.skillImprovement must be a boolean",
16
15
  })
17
16
  .default(true)
18
17
  .describe(
19
18
  "Whether retrospectives may discover, refine, and create managed skills from observed procedures. When false, retrospectives still capture ordinary memories through `remember`, but cannot load skill management, search for similar skills, or scaffold managed skills.",
20
19
  ),
21
20
 
21
+ skillImprovementMonitoring: z
22
+ .boolean({
23
+ error:
24
+ "memory.retrospective.skillImprovementMonitoring must be a boolean",
25
+ })
26
+ .default(false)
27
+ .describe(
28
+ "Reserved opt-in for monitoring retrospective skill-improvement decisions. This setting currently has no effect.",
29
+ ),
30
+
22
31
  timeThresholdMs: z
23
32
  .number({
24
33
  error: "memory.retrospective.timeThresholdMs must be a number",
@@ -903,6 +903,11 @@ export function buildChannelCapabilityBlock(
903
903
  "- Do NOT use markdown tables — use bullet lists instead. No markdown headers — use **bold** or CAPS for emphasis.",
904
904
  );
905
905
  }
906
+ if (caps.channel === "email") {
907
+ lines.push(
908
+ "- Conversation text is not emailed. To reply, run `assistant email send` (see `assistant email send --help`). Use `--reply-to` to keep the thread. Skip a reply only when none is needed.",
909
+ );
910
+ }
906
911
  }
907
912
 
908
913
  // Inject group chat etiquette only when the chat type indicates a multi-party
@@ -2035,20 +2040,6 @@ export async function composeInjectorChain(ctx: TurnContext): Promise<string> {
2035
2040
  */
2036
2041
  const DEFAULT_PLACEMENT: InjectionPlacement = "append-user-tail";
2037
2042
 
2038
- /**
2039
- * Count leading memory-prefix blocks on a user message's `content`.
2040
- *
2041
- * Delegates to {@link countMemoryPrefixBlocks} from
2042
- * `memory/graph/conversation-graph-memory.js` — the canonical state-machine
2043
- * for locating the memory-prefix boundary. Reusing it here keeps the
2044
- * PKB-context / PKB-reminder / NOW splice rules aligned on a single source
2045
- * of truth so their ordering relative to any memory prefix is stable and
2046
- * testable.
2047
- */
2048
- function countMemoryPrefixBlocksOnContent(content: ContentBlock[]): number {
2049
- return countMemoryPrefixBlocks(content);
2050
- }
2051
-
2052
2043
  /**
2053
2044
  * Apply one injector block to a `runMessages` array according to its
2054
2045
  * declared {@link InjectionPlacement}:
@@ -2101,9 +2092,7 @@ function applyInjectionBlock(
2101
2092
  { ...userTail, content: [...userTail.content, textBlock] },
2102
2093
  ];
2103
2094
  case "after-memory-prefix": {
2104
- const memoryPrefixCount = countMemoryPrefixBlocksOnContent(
2105
- userTail.content,
2106
- );
2095
+ const memoryPrefixCount = countMemoryPrefixBlocks(userTail.content);
2107
2096
  return [
2108
2097
  ...runMessages.slice(0, -1),
2109
2098
  {
@@ -2161,7 +2150,7 @@ function stripTailV2DynamicMemoryPrefix(
2161
2150
  if (!last || last.role !== "user") {
2162
2151
  return messages;
2163
2152
  }
2164
- const prefixCount = countMemoryPrefixBlocksOnContent(last.content);
2153
+ const prefixCount = countMemoryPrefixBlocks(last.content);
2165
2154
  if (prefixCount === 0) {
2166
2155
  return messages;
2167
2156
  }
@@ -10,7 +10,8 @@ import { beforeEach, describe, expect, mock, test } from "bun:test";
10
10
  import type { ResolvedMcpConfig } from "../../config/schemas/mcp.js";
11
11
 
12
12
  const toolsByServer = new Map<string, Array<{ name: string }>>();
13
- const connectDelays = new Map<string, number>();
13
+ const connectGates = new Map<string, Promise<void>>();
14
+ const listedServers = new Set<string>();
14
15
  let mcpGlobalMaxTools: number | undefined;
15
16
 
16
17
  mock.module("../../config/loader.js", () => ({
@@ -29,12 +30,13 @@ mock.module("../client.js", () => ({
29
30
  return null;
30
31
  }
31
32
  async connect() {
32
- const delay = connectDelays.get(this.serverId) ?? 0;
33
- if (delay > 0) {
34
- await new Promise((resolve) => setTimeout(resolve, delay));
33
+ const gate = connectGates.get(this.serverId);
34
+ if (gate) {
35
+ await gate;
35
36
  }
36
37
  }
37
38
  async listTools() {
39
+ listedServers.add(this.serverId);
38
40
  return (toolsByServer.get(this.serverId) ?? []).map((tool) => ({
39
41
  name: tool.name,
40
42
  description: `${this.serverId} ${tool.name}`,
@@ -66,7 +68,8 @@ function configWith(ids: string[]): ResolvedMcpConfig {
66
68
  describe("McpServerManager tool selection", () => {
67
69
  beforeEach(() => {
68
70
  toolsByServer.clear();
69
- connectDelays.clear();
71
+ connectGates.clear();
72
+ listedServers.clear();
70
73
  mcpGlobalMaxTools = undefined;
71
74
  });
72
75
 
@@ -98,23 +101,32 @@ describe("McpServerManager tool selection", () => {
98
101
  test("a slow earlier server does not prevent later servers from connecting", async () => {
99
102
  toolsByServer.set("slow", [{ name: "slow_tool" }]);
100
103
  toolsByServer.set("fast", [{ name: "fast_tool" }]);
101
- connectDelays.set("slow", 40);
104
+ let releaseSlow!: () => void;
105
+ connectGates.set(
106
+ "slow",
107
+ new Promise<void>((resolve) => {
108
+ releaseSlow = resolve;
109
+ }),
110
+ );
102
111
 
103
112
  const manager = new McpServerManager();
104
- const startedAt = Date.now();
105
- const started = await manager.start(configWith(["slow", "fast"]));
106
- const elapsed = Date.now() - startedAt;
107
-
108
- expect(started.connectedServerCount).toBe(2);
109
- expect(started.servers.map((server) => server.serverId).sort()).toEqual([
110
- "fast",
111
- "slow",
112
- ]);
113
- expect(
114
- elapsed < 80,
115
- "Servers connect in parallel, so one 40ms delay should not serialize both.",
116
- ).toBe(true);
117
- await manager.stop();
113
+ const starting = manager.start(configWith(["slow", "fast"]));
114
+ try {
115
+ await new Promise<void>((resolve) => setImmediate(resolve));
116
+ expect([...listedServers]).toEqual(["fast"]);
117
+
118
+ releaseSlow();
119
+ const started = await starting;
120
+ expect(started.connectedServerCount).toBe(2);
121
+ expect(started.servers.map((server) => server.serverId).sort()).toEqual([
122
+ "fast",
123
+ "slow",
124
+ ]);
125
+ } finally {
126
+ releaseSlow();
127
+ await starting;
128
+ await manager.stop();
129
+ }
118
130
  });
119
131
 
120
132
  test("a workspace global-max override raises how many tools are kept", async () => {
@@ -577,6 +577,12 @@ export function isProviderErrorMetadata(
577
577
  * assistant rows, and turn grouping closes on them, so display merging and
578
578
  * the turn resolver agree on boundaries. Takes the raw persisted `metadata`
579
579
  * JSON string; malformed JSON and non-assistant roles are never standalone.
580
+ *
581
+ * The web folds adjacent assistant rows again after pagination and reads the
582
+ * same rule off the wire projection in its own `isStandaloneAssistantMessage`
583
+ * (clients/web/src/domains/chat/utils/is-standalone-assistant-message.ts). A
584
+ * kind added here without a matching flag and check there merges on the
585
+ * client anyway.
580
586
  */
581
587
  export function isStandaloneAssistantMessage(
582
588
  role: string,
@@ -955,19 +955,6 @@ function likeContainsPattern(query: string): string {
955
955
  .replace(/_/g, "\\_")}%`;
956
956
  }
957
957
 
958
- /**
959
- * Whether the sparse Qdrant `messages_lexical` index — the only source of
960
- * message-content matches — is a safe read source. Content matching is
961
- * unavailable (title matches only) until the one-time upgrade backfill has
962
- * fully drained: a partially populated collection would silently miss older
963
- * content (an empty result — not a throw). Indexing itself is unconditional
964
- * host infrastructure, so completion is the only gate; the recall read site
965
- * applies the same one via the shared {@link isLexicalBackfillComplete}.
966
- */
967
- function isMessageContentSearchAvailable(): boolean {
968
- return isLexicalBackfillComplete();
969
- }
970
-
971
958
  /**
972
959
  * Full-text search across message content.
973
960
  *
@@ -976,9 +963,9 @@ function isMessageContentSearchAvailable(): boolean {
976
963
  * merged with a `LIKE` match on conversation titles; matching conversations
977
964
  * return with their relevant messages, ordered by most recently updated.
978
965
  *
979
- * Content matching is index-only — there is no `messages.content` scan
966
+ * Content matching is index-only: there is no `messages.content` scan
980
967
  * fallback and no other content source. Only the title arm can match while
981
- * the index is not a safe read source ({@link isMessageContentSearchAvailable}),
968
+ * the index is not a safe read source ({@link isLexicalBackfillComplete}),
982
969
  * for a query that tokenizes to nothing under the shared tokenizer (non-ASCII
983
970
  * or single-char input like "你", "é", "C++"), or when the Qdrant lexical
984
971
  * lookup fails (logged). An unindexed or unreachable index yields fewer
@@ -1017,7 +1004,7 @@ export async function searchConversations(
1017
1004
  const maxMsgsPerConv = opts?.maxMessagesPerConversation ?? 3;
1018
1005
 
1019
1006
  const hasTokens = hasLexicalTokens(trimmed);
1020
- const contentSearchAvailable = isMessageContentSearchAvailable();
1007
+ const contentSearchAvailable = isLexicalBackfillComplete();
1021
1008
 
1022
1009
  // LIKE pattern for title matching (message-content indexes don't cover titles).
1023
1010
  const titlePattern = likeContainsPattern(query);
@@ -85,7 +85,7 @@ mock.module("../../../../../util/logger.js", () => ({
85
85
  }),
86
86
  }));
87
87
 
88
- const { selectPool, MemoryV3RetrievalUnavailableError } =
88
+ const { selectPool, MemoryV3RetrievalUnavailableError, TYPE_SAFE_POOL_KEEP_NOUL } =
89
89
  await import("../pool-select.js");
90
90
  type SelectorPool = Parameters<typeof selectPool>[0];
91
91
 
@@ -968,3 +968,121 @@ describe("selectPool: cataloged thinking and forced-tool compatibility", () => {
968
968
  expect(selection.pages).toEqual([{ slug: "topic-x", sections: [] }]);
969
969
  });
970
970
  });
971
+
972
+ // ---------------------------------------------------------------------------
973
+ // selectPool: TypeSafe System One noul-per-candidate path.
974
+ // ---------------------------------------------------------------------------
975
+
976
+ function typesafeResponse(answers: Record<string, unknown>): ProviderResponse {
977
+ return {
978
+ model: "jev-latest",
979
+ stopReason: "end_turn",
980
+ usage: { inputTokens: 0, outputTokens: 0 },
981
+ content: [{ type: "text", text: JSON.stringify(answers, null, 2) }],
982
+ rawResponse: { answers },
983
+ };
984
+ }
985
+
986
+ function makeTypesafeProvider(response: ProviderResponse): Provider {
987
+ return {
988
+ name: "typesafe",
989
+ sendMessage: async (messages, options) => {
990
+ providerCalls.push({ messages, options });
991
+ return response;
992
+ },
993
+ };
994
+ }
995
+
996
+ function noulAnswer(noul: number): { type: "noul"; noul: number } {
997
+ return { type: "noul", noul };
998
+ }
999
+
1000
+ describe("selectPool: TypeSafe System One", () => {
1001
+ test("sends one noul per candidate and no select_pages tool", async () => {
1002
+ providerStub = makeTypesafeProvider(
1003
+ typesafeResponse({
1004
+ "1": noulAnswer(0.9),
1005
+ "2": noulAnswer(0.1),
1006
+ "3": noulAnswer(0.8),
1007
+ "4": noulAnswer(0.2),
1008
+ }),
1009
+ );
1010
+
1011
+ await selectPool(makePool(), makeTurn("rollout?"));
1012
+
1013
+ expect(providerCalls).toHaveLength(1);
1014
+ const [call] = providerCalls;
1015
+ expect(call.options?.tools).toBeUndefined();
1016
+ expect(
1017
+ (call.options?.config as Record<string, unknown> | undefined)?.tool_choice,
1018
+ ).toBeUndefined();
1019
+ expect(
1020
+ (call.options?.config as Record<string, unknown> | undefined)?.callSite,
1021
+ ).toBe("memoryV3SelectL2");
1022
+
1023
+ const payload = JSON.parse(
1024
+ (call.messages[0]!.content[0] as { text: string }).text,
1025
+ ) as {
1026
+ state: {
1027
+ candidates: Record<string, { slug: string; text: string }>;
1028
+ current_message: string;
1029
+ selector_instructions: string;
1030
+ };
1031
+ questions: Record<string, { type: string; instructions: string }>;
1032
+ };
1033
+ expect(Object.keys(payload.questions)).toEqual(["1", "2", "3", "4"]);
1034
+ expect(payload.questions["1"]?.type).toBe("noul");
1035
+ expect(payload.questions["1"]?.instructions).toContain("`candidates.1`");
1036
+ expect(payload.state.candidates["1"]?.slug).toBe("page-a");
1037
+ expect(payload.state.candidates["1"]?.text).toBe(CARD_A);
1038
+ expect(payload.state.candidates["3"]?.slug).toBe("topic-x");
1039
+ expect(payload.state.current_message).toBe("rollout?");
1040
+ expect(payload.state.selector_instructions.length).toBeGreaterThan(0);
1041
+ });
1042
+
1043
+ test("keeps candidates at or above the inclusive noul threshold", async () => {
1044
+ providerStub = makeTypesafeProvider(
1045
+ typesafeResponse({
1046
+ "1": noulAnswer(TYPE_SAFE_POOL_KEEP_NOUL),
1047
+ "2": noulAnswer(TYPE_SAFE_POOL_KEEP_NOUL - 0.01),
1048
+ "3": noulAnswer(0.91),
1049
+ "4": noulAnswer(0.12),
1050
+ }),
1051
+ );
1052
+
1053
+ const result = await selectPool(makePool(), makeTurn("rollout?"));
1054
+ expect(result.keptAll).toBe(false);
1055
+ expect(result.pages).toEqual([
1056
+ { slug: "page-a", sections: [] },
1057
+ { slug: "topic-x", sections: [] },
1058
+ ]);
1059
+ });
1060
+
1061
+ test("an all-below-threshold pool is a deliberate empty selection", async () => {
1062
+ providerStub = makeTypesafeProvider(
1063
+ typesafeResponse({
1064
+ "1": noulAnswer(0.1),
1065
+ "2": noulAnswer(0.2),
1066
+ "3": noulAnswer(0.05),
1067
+ "4": noulAnswer(0.3),
1068
+ }),
1069
+ );
1070
+
1071
+ const result = await selectPool(makePool(), makeTurn("nothing relevant"));
1072
+ expect(result).toEqual({ pages: [], keptAll: false });
1073
+ });
1074
+
1075
+ test("unusable answers throw after the re-prompt retry", async () => {
1076
+ providerStub = makeTypesafeProvider({
1077
+ model: "jev-latest",
1078
+ stopReason: "end_turn",
1079
+ usage: { inputTokens: 0, outputTokens: 0 },
1080
+ content: [{ type: "text", text: "not-json" }],
1081
+ });
1082
+
1083
+ await expect(selectPool(makePool(), makeTurn("x"))).rejects.toThrow(
1084
+ MemoryV3RetrievalUnavailableError,
1085
+ );
1086
+ expect(providerCalls).toHaveLength(3);
1087
+ });
1088
+ });
@@ -67,7 +67,9 @@
67
67
  * 3. A SINGLE forced-tool select (`selectPool`) over the whole pool. The
68
68
  * result is this turn's selections — current turn only. Cross-turn
69
69
  * persistence is the injector's job (net-new blocks frozen into history),
70
- * not a per-turn re-rendered carry set.
70
+ * not a per-turn re-rendered carry set. When the call site resolves to
71
+ * TypeSafe, `selectPool` asks one System One noul per numbered candidate
72
+ * instead of forcing `select_pages`.
71
73
  */
72
74
 
73
75
  import type { AssistantConfig } from "../../../../config/schema.js";
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Memory v3 — single pool selector.
3
3
  *
4
- * Runs a SINGLE forced-tool call over one unified candidate pool rendered in
4
+ * Runs a SINGLE selector call over one unified candidate pool rendered in
5
5
  * two segments that share one numbering:
6
6
  *
7
7
  * 1. STABLE PREFIX — the core+hot lane pages as FULL CARDS (head section +
@@ -52,6 +52,8 @@
52
52
  import type {
53
53
  ContentBlock,
54
54
  Message,
55
+ Provider,
56
+ ProviderResponse,
55
57
  ToolUseContent,
56
58
  } from "@vellumai/plugin-api";
57
59
  import { getConfiguredProvider, safeStringSlice } from "@vellumai/plugin-api";
@@ -179,7 +181,8 @@ type PoolSelectorAttemptFailureReason =
179
181
  | "provider_error"
180
182
  | "missing_tool_use"
181
183
  | "unexpected_tool_name"
182
- | "schema_mismatch";
184
+ | "schema_mismatch"
185
+ | "unusable_answers";
183
186
 
184
187
  interface PoolSelectorAttemptFailure {
185
188
  attempt: number;
@@ -493,6 +496,248 @@ export function selectAllPoolCandidates(pool: SelectorPool): SelectedPage[] {
493
496
  );
494
497
  }
495
498
 
499
+ /**
500
+ * Inclusive noul threshold for TypeSafe pool selection. 0.5 is calibrated
501
+ * equal yes/no. This sits slightly below that so a plausible candidate is
502
+ * kept, matching the selector's recall-heavy rule.
503
+ */
504
+ export const TYPE_SAFE_POOL_KEEP_NOUL = 0.4;
505
+
506
+ /** Catalog id of the TypeSafe System One provider. */
507
+ const TYPE_SAFE_PROVIDER_ID = "typesafe";
508
+
509
+ type TypesafeNoulQuestion = {
510
+ type: "noul";
511
+ instructions: string;
512
+ criteria: { yes: string; no: string };
513
+ };
514
+
515
+ function isRecord(value: unknown): value is Record<string, unknown> {
516
+ return typeof value === "object" && value !== null && !Array.isArray(value);
517
+ }
518
+
519
+ function noulFromAnswer(answer: unknown): number | undefined {
520
+ if (typeof answer === "number" && Number.isFinite(answer)) {
521
+ return answer;
522
+ }
523
+ if (
524
+ isRecord(answer) &&
525
+ typeof answer.noul === "number" &&
526
+ Number.isFinite(answer.noul)
527
+ ) {
528
+ return answer.noul;
529
+ }
530
+ return undefined;
531
+ }
532
+
533
+ function typesafeCandidateEntries(
534
+ pool: SelectorPool,
535
+ ): Record<string, { slug: Slug; text: string }> {
536
+ const entries: Record<string, { slug: Slug; text: string }> = {};
537
+ pool.stable.forEach((candidate, index) => {
538
+ entries[String(index + 1)] = {
539
+ slug: candidate.slug,
540
+ text: candidate.card,
541
+ };
542
+ });
543
+ pool.finder.forEach((candidate, index) => {
544
+ entries[String(pool.stable.length + index + 1)] = {
545
+ slug: candidate.slug,
546
+ text: renderFinderLine(candidate),
547
+ };
548
+ });
549
+ return entries;
550
+ }
551
+
552
+ function typesafeKeepQuestion(id: string): TypesafeNoulQuestion {
553
+ return {
554
+ type: "noul",
555
+ instructions:
556
+ `Would the upcoming assistant reply draw on \`candidates.${id}\`? ` +
557
+ "Lean inclusive. Facts, current task and event state, register, " +
558
+ "framing, calibration, and relationship texture all count. True when " +
559
+ "the candidate could plausibly inform the reply.",
560
+ criteria: {
561
+ yes: "The reply would draw on this candidate.",
562
+ no: "The reply would not draw on this candidate.",
563
+ },
564
+ };
565
+ }
566
+
567
+ function answersFromSelectorResponse(
568
+ response: ProviderResponse,
569
+ ): Record<string, unknown> | null {
570
+ const raw = response.rawResponse;
571
+ if (isRecord(raw) && isRecord(raw.answers)) {
572
+ return raw.answers;
573
+ }
574
+ const textBlock = response.content.find((block) => block.type === "text");
575
+ if (!textBlock || textBlock.type !== "text") {
576
+ return null;
577
+ }
578
+ try {
579
+ const parsed: unknown = JSON.parse(textBlock.text);
580
+ return isRecord(parsed) ? parsed : null;
581
+ } catch {
582
+ return null;
583
+ }
584
+ }
585
+
586
+ async function selectPoolWithTypesafe(
587
+ pool: SelectorPool,
588
+ turn: MemoryRoutingTurn,
589
+ ordered: PoolLine[],
590
+ systemPrompt: string,
591
+ provider: Provider,
592
+ ): Promise<PoolSelection> {
593
+ const candidates = typesafeCandidateEntries(pool);
594
+ const questions: Record<string, TypesafeNoulQuestion> = {};
595
+ for (const id of Object.keys(candidates)) {
596
+ questions[id] = typesafeKeepQuestion(id);
597
+ }
598
+ const state = {
599
+ selector_instructions: systemPrompt,
600
+ candidates,
601
+ ...(turn.situationalContext
602
+ ? { situation: turn.situationalContext }
603
+ : {}),
604
+ recent_context: turn.recentContext,
605
+ current_message: turn.currentMessage,
606
+ };
607
+ const userMsg: Message = {
608
+ role: "user",
609
+ content: [
610
+ {
611
+ type: "text",
612
+ text: JSON.stringify({ state, questions }),
613
+ },
614
+ ],
615
+ };
616
+
617
+ const failures: PoolSelectorAttemptFailure[] = [];
618
+ let attempt = 0;
619
+ const recordFailure = (
620
+ failure: Omit<
621
+ PoolSelectorAttemptFailure,
622
+ | "callSite"
623
+ | "providerName"
624
+ | "candidateCount"
625
+ | "stableCount"
626
+ | "finderCount"
627
+ >,
628
+ ): void => {
629
+ const diagnostic: PoolSelectorAttemptFailure = {
630
+ ...failure,
631
+ callSite: MEMORY_V3_SELECT_CALL_SITE,
632
+ providerName: provider.name,
633
+ candidateCount: ordered.length,
634
+ stableCount: pool.stable.length,
635
+ finderCount: pool.finder.length,
636
+ };
637
+ failures.push(diagnostic);
638
+ log.warn(diagnostic, "pool selector attempt failed");
639
+ };
640
+
641
+ let lastError: unknown = null;
642
+ const parsed = await retryForResult(async () => {
643
+ attempt += 1;
644
+ let response: Awaited<ReturnType<typeof provider.sendMessage>>;
645
+ try {
646
+ response = await provider.sendMessage([userMsg], {
647
+ config: {
648
+ callSite: MEMORY_V3_SELECT_CALL_SITE,
649
+ conversationId: turn.conversationId,
650
+ disableTurnStartCache: true,
651
+ },
652
+ });
653
+ lastError = null;
654
+ } catch (error) {
655
+ lastError = error;
656
+ recordFailure({
657
+ attempt,
658
+ reason: "provider_error",
659
+ error: summarizeError(error),
660
+ });
661
+ throw error;
662
+ }
663
+ const answers = answersFromSelectorResponse(response);
664
+ if (!answers) {
665
+ recordFailure({
666
+ attempt,
667
+ reason: "unusable_answers",
668
+ response: summarizeResponse(response),
669
+ });
670
+ return null;
671
+ }
672
+ const picked: number[] = [];
673
+ let parsedCount = 0;
674
+ for (let index = 0; index < ordered.length; index++) {
675
+ const noul = noulFromAnswer(answers[String(index + 1)]);
676
+ if (noul === undefined) {
677
+ continue;
678
+ }
679
+ parsedCount += 1;
680
+ if (noul >= TYPE_SAFE_POOL_KEEP_NOUL) {
681
+ picked.push(index);
682
+ }
683
+ }
684
+ if (parsedCount === 0) {
685
+ recordFailure({
686
+ attempt,
687
+ reason: "unusable_answers",
688
+ response: summarizeResponse(response),
689
+ });
690
+ return null;
691
+ }
692
+ return { pages: mergeSelectedLines(ordered, picked), keptAll: false };
693
+ });
694
+
695
+ if (parsed === null) {
696
+ if (lastError !== null) {
697
+ const detail =
698
+ lastError instanceof Error ? lastError.message : String(lastError);
699
+ const redactedDetail = truncate(
700
+ redactLogString(detail),
701
+ ERROR_MESSAGE_MAX_CHARS,
702
+ );
703
+ log.warn(
704
+ {
705
+ candidateCount: ordered.length,
706
+ stableCount: pool.stable.length,
707
+ finderCount: pool.finder.length,
708
+ callSite: MEMORY_V3_SELECT_CALL_SITE,
709
+ providerName: provider.name,
710
+ failures,
711
+ },
712
+ "pool selector provider call failed after retries",
713
+ );
714
+ throw new MemoryV3RetrievalUnavailableError(
715
+ `memory-v3 pool selector provider call failed after retries: ${redactedDetail}`,
716
+ {
717
+ cause: lastError,
718
+ conversationNotice: providerBillingNoticeFromError(lastError),
719
+ },
720
+ );
721
+ }
722
+ log.warn(
723
+ {
724
+ candidateCount: ordered.length,
725
+ stableCount: pool.stable.length,
726
+ finderCount: pool.finder.length,
727
+ callSite: MEMORY_V3_SELECT_CALL_SITE,
728
+ providerName: provider.name,
729
+ failures,
730
+ },
731
+ "pool selector returned no usable TypeSafe answers after retries",
732
+ );
733
+ throw new MemoryV3RetrievalUnavailableError(
734
+ "memory-v3 pool selector returned no usable selection after retries",
735
+ );
736
+ }
737
+
738
+ return parsed;
739
+ }
740
+
496
741
  /** A selection plus whether it came from the recall-safe keep-all fallback. */
497
742
  export interface PoolSelection {
498
743
  pages: SelectedPage[];
@@ -504,14 +749,16 @@ export interface PoolSelection {
504
749
  }
505
750
 
506
751
  /**
507
- * Run the single forced-tool selector over the unified candidate pool. Returns
752
+ * Run the single selector over the unified candidate pool. Returns
508
753
  * the pages to inject, merged per slug (a page selected as a card and on
509
754
  * finder lines yields one entry carrying every selected section), plus a
510
755
  * `keptAll` flag marking the recall-safe fallback.
511
756
  *
512
- * An omitted `ids` keeps ALL candidates (the recall-safe "all of these are
513
- * relevant" signal, `keptAll: true`); an explicit `[]` keeps none; an
514
- * infrastructure failure (after a short re-prompt retry) throws
757
+ * On a chat model, an omitted `ids` keeps ALL candidates (the recall-safe
758
+ * "all of these are relevant" signal, `keptAll: true`); an explicit `[]`
759
+ * keeps none. TypeSafe answers one noul per candidate and never omits ids,
760
+ * so `keptAll` is always false on that path. An infrastructure failure
761
+ * (after a short re-prompt retry) throws
515
762
  * {@link MemoryV3RetrievalUnavailableError}, and the orchestrator keeps the
516
763
  * stable prefix unjudged in its place.
517
764
  *
@@ -545,6 +792,16 @@ export async function selectPool(
545
792
  );
546
793
  }
547
794
 
795
+ if (provider.name === TYPE_SAFE_PROVIDER_ID) {
796
+ return selectPoolWithTypesafe(
797
+ pool,
798
+ turn,
799
+ ordered,
800
+ systemPrompt,
801
+ provider,
802
+ );
803
+ }
804
+
548
805
  // Two content blocks: the stable prefix (cards) carries the cache
549
806
  // breakpoint; the dynamic tail (finder lines + per-turn context) does not.
550
807
  // See the module doc for the cache contract.
@@ -206,7 +206,7 @@ const ADAPTER_FACTORIES: Record<string, AdapterFactory> = {
206
206
  streamTimeoutMs,
207
207
  ...(baseURL ? { baseURL } : {}),
208
208
  }),
209
- jev: ({ apiKey, model, streamTimeoutMs, baseURL }) =>
209
+ typesafe: ({ apiKey, model, streamTimeoutMs, baseURL }) =>
210
210
  new JevProvider(apiKey, model, {
211
211
  streamTimeoutMs,
212
212
  ...(baseURL ? { baseURL } : {}),
@@ -5,6 +5,7 @@ import {
5
5
  conversationToState,
6
6
  DEFAULT_JEV_MODEL,
7
7
  JevProvider,
8
+ noulFromAnswer,
8
9
  parseSystemOneOverride,
9
10
  validateJevApiKey,
10
11
  } from "./client.js";
@@ -66,6 +67,15 @@ describe("parseSystemOneOverride", () => {
66
67
  });
67
68
  });
68
69
 
70
+ describe("noulFromAnswer", () => {
71
+ test("reads a typed noul object or a bare number", () => {
72
+ expect(noulFromAnswer({ type: "noul", noul: 0.91 })).toBe(0.91);
73
+ expect(noulFromAnswer(0.4)).toBe(0.4);
74
+ expect(noulFromAnswer({ type: "choice", choice: "keep" })).toBeUndefined();
75
+ expect(noulFromAnswer(null)).toBeUndefined();
76
+ });
77
+ });
78
+
69
79
  describe("conversationToState", () => {
70
80
  test("flattens system prompt and turns into a string", () => {
71
81
  const state = conversationToState(
@@ -15,7 +15,7 @@ import type {
15
15
 
16
16
  const log = getLogger("jev-client");
17
17
 
18
- export const JEV_PROVIDER_ID = "jev";
18
+ export const JEV_PROVIDER_ID = "typesafe";
19
19
  export const DEFAULT_JEV_BASE_URL = "https://api.typesafe.ai";
20
20
  export const DEFAULT_JEV_MODEL = "jev-latest";
21
21
 
@@ -115,6 +115,21 @@ function isJevQuestions(value: unknown): value is JevQuestions {
115
115
  return entries.every(([, question]) => isJevQuestion(question));
116
116
  }
117
117
 
118
+ /** Calibrated P(yes) from a System One noul answer, or undefined when absent. */
119
+ export function noulFromAnswer(answer: unknown): number | undefined {
120
+ if (typeof answer === "number" && Number.isFinite(answer)) {
121
+ return answer;
122
+ }
123
+ if (
124
+ isRecord(answer) &&
125
+ typeof answer.noul === "number" &&
126
+ Number.isFinite(answer.noul)
127
+ ) {
128
+ return answer.noul;
129
+ }
130
+ return undefined;
131
+ }
132
+
118
133
  /**
119
134
  * If the last user message is a TypeSafe System One payload, use it as the
120
135
  * evaluation request. `state` is optional; callers fall back to the
@@ -77,7 +77,8 @@ export interface CatalogModel {
77
77
  * Whether the model produces free-form chat text. Omit (or true) for
78
78
  * ordinary chat models. False for structured-decision models that return
79
79
  * answers rather than generated text; those stay out of conversation
80
- * pickers and cannot be the conversation model.
80
+ * pickers and cannot be the conversation model. They can still back a
81
+ * saved profile and a call-site pin.
81
82
  */
82
83
  supportsText?: boolean;
83
84
  supportsEffort?: boolean;
@@ -2532,8 +2533,8 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
2532
2533
  apiKeyPlaceholder: "Your Poolside API key",
2533
2534
  },
2534
2535
  {
2535
- id: "jev",
2536
- displayName: "Jev",
2536
+ id: "typesafe",
2537
+ displayName: "TypeSafe",
2537
2538
  subtitle:
2538
2539
  "TypeSafe System One decision model. Returns structured answers, not generated text. Requires a TypeSafe API key.",
2539
2540
  setupMode: "api-key",
@@ -388,7 +388,7 @@ export async function routeGuardianReply(
388
388
  const request = await getGuardianRequestOrNull(answerTap.requestId);
389
389
  if (
390
390
  request &&
391
- resolveRequestInstructionMode(request) === "answer" &&
391
+ resolveGuardianInstructionModeForRequest(request) === "answer" &&
392
392
  parseQuestionAnswerActionId(answerTap.token) &&
393
393
  !request.callSessionId &&
394
394
  hasLiveQuestionInteraction(request.id)
@@ -581,7 +581,7 @@ export async function routeGuardianReply(
581
581
  if (messageText.length > 0 && pendingRequests.length === 1) {
582
582
  const soleRequest = pendingRequests[0];
583
583
  if (
584
- resolveRequestInstructionMode(soleRequest) === "answer" &&
584
+ resolveGuardianInstructionModeForRequest(soleRequest) === "answer" &&
585
585
  !soleRequest.callSessionId &&
586
586
  soleRequest.sourceConversationId === conversationId &&
587
587
  hasLiveQuestionInteraction(soleRequest.id)
@@ -1005,12 +1005,6 @@ function inferActionFromText(
1005
1005
  return "approve_once";
1006
1006
  }
1007
1007
 
1008
- function resolveRequestInstructionMode(
1009
- request?: Pick<GuardianRequestWire, "kind" | "toolName"> | null,
1010
- ): "approval" | "answer" {
1011
- return resolveGuardianInstructionModeForRequest(request);
1012
- }
1013
-
1014
1008
  // ---------------------------------------------------------------------------
1015
1009
  // Failure reason reply text
1016
1010
  // ---------------------------------------------------------------------------
@@ -1044,7 +1038,7 @@ function failureReplyText(
1044
1038
  return "Something went wrong with this request on our end, so I couldn't apply your decision.";
1045
1039
  case "invalid_action":
1046
1040
  return buildGuardianInvalidActionReply(
1047
- resolveRequestInstructionMode(request),
1041
+ resolveGuardianInstructionModeForRequest(request),
1048
1042
  requestCode ?? undefined,
1049
1043
  );
1050
1044
  default:
@@ -1063,7 +1057,7 @@ function failureReplyText(
1063
1057
  */
1064
1058
  function composeCodeOnlyClarification(request: GuardianRequestWire): string {
1065
1059
  const code = request.requestCode ?? "unknown";
1066
- const mode = resolveRequestInstructionMode(request);
1060
+ const mode = resolveGuardianInstructionModeForRequest(request);
1067
1061
  return buildGuardianCodeOnlyClarification(mode, {
1068
1062
  requestCode: code,
1069
1063
  questionText: request.questionText,
@@ -1087,7 +1081,7 @@ function composeDisambiguationReply(
1087
1081
  const lines: string[] = [];
1088
1082
  const requestsWithMode = pendingRequests.map((request) => ({
1089
1083
  request,
1090
- mode: resolveRequestInstructionMode(request),
1084
+ mode: resolveGuardianInstructionModeForRequest(request),
1091
1085
  }));
1092
1086
 
1093
1087
  if (engineReplyText) {
@@ -927,7 +927,7 @@ describe("PUT inference/active-profile validation", () => {
927
927
  profiles: {
928
928
  jev: {
929
929
  source: "user",
930
- provider: "jev",
930
+ provider: "typesafe",
931
931
  model: "jev-latest",
932
932
  status: "active",
933
933
  },
@@ -2,9 +2,10 @@
2
2
  * Route handler for the POST /v1/btw SSE-streaming side-chain endpoint.
3
3
  *
4
4
  * Runs an ephemeral LLM call that reuses the conversation's provider, tool
5
- * definitions, and message history for prompt-cache efficiency. Uses the
6
- * conversation's system prompt when a conversation-specific override is active;
7
- * otherwise builds a fresh prompt excluding BOOTSTRAP.md so first-run
5
+ * definitions, and message history for prompt-cache efficiency; the
6
+ * empty-state greeting targets no real conversation and sends no tools. Uses
7
+ * the conversation's system prompt when a conversation-specific override is
8
+ * active; otherwise builds a fresh prompt excluding BOOTSTRAP.md so first-run
8
9
  * onboarding instructions don't leak into cosmetic UI calls like identity
9
10
  * intro generation. The response is streamed as SSE events (`btw_text_delta`,
10
11
  * `btw_complete`, `btw_error`).
@@ -145,7 +146,11 @@ async function handleBtw({
145
146
  const result = await runBtwSidechain({
146
147
  content: effectiveContent,
147
148
  conversation,
148
- tools: getAllToolDefinitions(),
149
+ // The side-chain forces `tool_choice: none`, so tool definitions
150
+ // only earn their tokens as a shared cache prefix with a real
151
+ // conversation. The greeting runs against an ephemeral one with its
152
+ // own system prompt, so it shares nothing and sends no tools.
153
+ tools: isGreeting ? [] : getAllToolDefinitions(),
149
154
  signal: abortSignal,
150
155
  ...(isGreeting ? { callSite: "emptyStateGreeting" as const } : {}),
151
156
  onEvent: (event) => {
@@ -228,7 +228,7 @@ function getIdentity() {
228
228
 
229
229
  const version = APP_VERSION;
230
230
 
231
- const createdAt = resolveIdentityCreatedAt(identityPath);
231
+ const createdAt = resolveHatchedAtReadOnly(identityPath);
232
232
 
233
233
  return {
234
234
  name: fields.name ?? "",
@@ -241,10 +241,6 @@ function getIdentity() {
241
241
  };
242
242
  }
243
243
 
244
- function resolveIdentityCreatedAt(identityPath: string): string | undefined {
245
- return resolveHatchedAtReadOnly(identityPath);
246
- }
247
-
248
244
  // ---------------------------------------------------------------------------
249
245
  // Zod schemas for profiler health metadata
250
246
  // ---------------------------------------------------------------------------
@@ -303,7 +303,7 @@ async function handleAddSecret({ body }: RouteHandlerArgs) {
303
303
  );
304
304
  return { success: false, error: validation.reason };
305
305
  }
306
- } else if (name === "jev") {
306
+ } else if (name === "typesafe") {
307
307
  const validation = await validateJevApiKey(value);
308
308
  if (!validation.valid) {
309
309
  log.warn(
@@ -0,0 +1,81 @@
1
+ import { dirname, join, resolve } from "node:path";
2
+ import { fileURLToPath } from "node:url";
3
+ import { describe, expect, test } from "bun:test";
4
+
5
+ import { parseToolManifestFile } from "../../skills/tool-manifest.js";
6
+ import { explicitTools } from "../tool-manifest.js";
7
+
8
+ /**
9
+ * Tool input schemas must be plain objects at the root.
10
+ *
11
+ * Anthropic's Messages API rejects a tool whose `input_schema` carries
12
+ * `oneOf`, `anyOf`, or `allOf` at the top level ("input_schema does not
13
+ * support oneOf, allOf, or anyOf at the top level"). The rejection is a 400
14
+ * for the whole request, so one offending definition takes down every call
15
+ * that advertises it. Other hosts accept the same schema, which lets the
16
+ * mistake hide until a request routes to Anthropic directly. Either/or rules
17
+ * between fields belong in the tool description and the tool's own input
18
+ * validation instead.
19
+ *
20
+ * Covers the core manifest and every bundled skill's `TOOLS.json`.
21
+ * Combinators nested under `properties` are accepted by Anthropic and stay
22
+ * out of scope.
23
+ */
24
+
25
+ const ROOT_COMBINATORS = ["oneOf", "anyOf", "allOf"] as const;
26
+
27
+ const BUNDLED_SKILLS_DIR = resolve(
28
+ dirname(fileURLToPath(import.meta.url)),
29
+ "../../config/bundled-skills",
30
+ );
31
+
32
+ interface SchemaCase {
33
+ /** Tool name, prefixed with the skill directory for bundled skill tools. */
34
+ label: string;
35
+ schema: Record<string, unknown>;
36
+ }
37
+
38
+ function bundledSkillSchemas(): SchemaCase[] {
39
+ const cases: SchemaCase[] = [];
40
+ for (const relative of new Bun.Glob("*/TOOLS.json").scanSync({
41
+ cwd: BUNDLED_SKILLS_DIR,
42
+ })) {
43
+ const manifest = parseToolManifestFile(join(BUNDLED_SKILLS_DIR, relative));
44
+ for (const tool of manifest.tools) {
45
+ cases.push({
46
+ label: `${dirname(relative)}/${tool.name}`,
47
+ schema: tool.input_schema,
48
+ });
49
+ }
50
+ }
51
+ return cases;
52
+ }
53
+
54
+ function coreSchemas(): SchemaCase[] {
55
+ return explicitTools.map((tool) => {
56
+ if (!tool.name) {
57
+ throw new Error("core manifest entries carry explicit names");
58
+ }
59
+ return {
60
+ label: tool.name,
61
+ schema: tool.input_schema as Record<string, unknown>,
62
+ };
63
+ });
64
+ }
65
+
66
+ const CASES: SchemaCase[] = [...coreSchemas(), ...bundledSkillSchemas()];
67
+
68
+ describe("tool input schema root", () => {
69
+ test("covers the core manifest and the bundled skills", () => {
70
+ expect(CASES.length).toBeGreaterThan(explicitTools.length);
71
+ });
72
+
73
+ for (const { label, schema } of CASES) {
74
+ test(`${label} keeps combinators out of the schema root`, () => {
75
+ for (const keyword of ROOT_COMBINATORS) {
76
+ expect(schema).not.toHaveProperty(keyword);
77
+ }
78
+ expect(schema.type).toBe("object");
79
+ });
80
+ }
81
+ });
@@ -124,6 +124,8 @@ const DESCRIPTION = [
124
124
  "For logins, use saved credentials first. Securely collect missing credentials",
125
125
  "with assistant credentials prompt, then fill the login form yourself.",
126
126
  "",
127
+ "Every call passes exactly one of `questions` or `desktopHelp`.",
128
+ "",
127
129
  "Use this tool whenever a request is ambiguous and can be resolved",
128
130
  "by 2–4 plausible interpretations or discrete choices. Prefer it over",
129
131
  "plain-text clarification — structured options are faster to answer and",
@@ -255,10 +257,10 @@ export const askQuestionTool = {
255
257
  category: "interaction",
256
258
  executionTarget: "sandbox",
257
259
  defaultRiskLevel: RiskLevel.Low,
258
- input_schema: {
259
- ...toToolInputSchema(askQuestionInputSchema),
260
- oneOf: [{ required: ["questions"] }, { required: ["desktopHelp"] }],
261
- },
260
+ // Anthropic rejects `oneOf` / `anyOf` / `allOf` at the root of a tool
261
+ // schema, so the either/or rule between `questions` and `desktopHelp` lives
262
+ // in the description and the Zod refine, not in the wire schema.
263
+ input_schema: toToolInputSchema(askQuestionInputSchema),
262
264
 
263
265
  async execute(
264
266
  input: Record<string, unknown>,
@@ -23,10 +23,6 @@ import {
23
23
  } from "../shared/zod-tool-schema.js";
24
24
  import type { ToolContext, ToolExecutionResult } from "../types.js";
25
25
 
26
- function isPrivilegedDocumentActor(context: ToolContext): boolean {
27
- return canActOnPrivilegedDocuments(context);
28
- }
29
-
30
26
  export function documentNotFound(surfaceId: string): ToolExecutionResult {
31
27
  return {
32
28
  content: JSON.stringify({
@@ -43,7 +39,7 @@ export function canAccessDocument(
43
39
  context: ToolContext,
44
40
  ): boolean {
45
41
  return (
46
- isPrivilegedDocumentActor(context) ||
42
+ canActOnPrivilegedDocuments(context) ||
47
43
  isDocumentAssociatedWithConversation(surfaceId, context.conversationId)
48
44
  );
49
45
  }
@@ -522,7 +518,7 @@ export function executeDocumentList(
522
518
  const docs = query
523
519
  ? searchDocumentsByTitle(
524
520
  query,
525
- isPrivilegedDocumentActor(context)
521
+ canActOnPrivilegedDocuments(context)
526
522
  ? {}
527
523
  : { conversationId: context.conversationId },
528
524
  )