@vellumai/assistant 0.12.2-dev.202609171913.b5e95d3 → 0.12.2-dev.202609172115.0b9be8f
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/node_modules/@vellumai/slack-text/src/index.ts +4 -8
- package/openapi.yaml +3 -3
- package/package.json +1 -1
- package/src/__tests__/btw-routes.test.ts +4 -1
- package/src/__tests__/conversation-runtime-assembly.test.ts +29 -0
- package/src/__tests__/inference-profile-session-handler.test.ts +1 -1
- package/src/__tests__/llm-catalog-parity.test.ts +2 -2
- package/src/config/__tests__/default-provider.test.ts +3 -1
- package/src/config/__tests__/memory-retrospective-schema.test.ts +14 -1
- package/src/config/bundled-skills/sequences/TOOLS.json +1 -5
- package/src/config/profile-text-generation.test.ts +2 -2
- package/src/config/schemas/llm.ts +1 -1
- package/src/config/schemas/memory-retrospective.ts +11 -2
- package/src/daemon/conversation-runtime-assembly.ts +7 -18
- package/src/mcp/__tests__/manager-tool-caps.test.ts +32 -20
- package/src/persistence/conversation-crud.ts +6 -0
- package/src/persistence/conversation-queries.ts +3 -16
- package/src/plugins/defaults/memory/v3/__tests__/pool-select.test.ts +119 -1
- package/src/plugins/defaults/memory/v3/orchestrate.ts +3 -1
- package/src/plugins/defaults/memory/v3/pool-select.ts +263 -6
- package/src/providers/inference/adapter-factory.ts +1 -1
- package/src/providers/jev/client.test.ts +10 -0
- package/src/providers/jev/client.ts +16 -1
- package/src/providers/model-catalog.ts +4 -3
- package/src/runtime/guardian-reply-router.ts +5 -11
- package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +1 -1
- package/src/runtime/routes/btw-routes.ts +9 -4
- package/src/runtime/routes/identity-routes.ts +1 -5
- package/src/runtime/routes/secret-routes.ts +1 -1
- package/src/tools/__tests__/tool-schema-root-combinator-guard.test.ts +81 -0
- package/src/tools/ask-question/ask-question-tool.ts +6 -4
- package/src/tools/document/document-tool.ts +2 -6
|
@@ -127,7 +127,7 @@ export async function buildSlackUserLabelMap(
|
|
|
127
127
|
ids.map(async (id): Promise<[string, string] | undefined> => {
|
|
128
128
|
try {
|
|
129
129
|
const label = await resolveLabel(id);
|
|
130
|
-
const sanitized =
|
|
130
|
+
const sanitized = sanitizeSlackLabel(label ?? undefined);
|
|
131
131
|
if (!sanitized || sanitized === id) return undefined;
|
|
132
132
|
return [id, sanitized];
|
|
133
133
|
} catch {
|
|
@@ -166,7 +166,7 @@ export async function buildSlackChannelLabelMap(
|
|
|
166
166
|
ids.map(async (id): Promise<[string, string] | undefined> => {
|
|
167
167
|
try {
|
|
168
168
|
const label = await resolveLabel(id);
|
|
169
|
-
const sanitized =
|
|
169
|
+
const sanitized = sanitizeSlackLabel(label ?? undefined);
|
|
170
170
|
if (!sanitized || sanitized === id) return undefined;
|
|
171
171
|
return [id, sanitized];
|
|
172
172
|
} catch {
|
|
@@ -222,7 +222,7 @@ function renderChannelReference(
|
|
|
222
222
|
return `#${embeddedLabel}`;
|
|
223
223
|
}
|
|
224
224
|
|
|
225
|
-
const resolvedLabel =
|
|
225
|
+
const resolvedLabel = sanitizeSlackLabel(
|
|
226
226
|
options.channelLabels?.[channelId],
|
|
227
227
|
);
|
|
228
228
|
if (resolvedLabel && resolvedLabel !== channelId) {
|
|
@@ -320,15 +320,11 @@ export function sanitizeSlackLabel(
|
|
|
320
320
|
function sanitizeEmbeddedSlackLabel(
|
|
321
321
|
label: string | undefined,
|
|
322
322
|
): string | undefined {
|
|
323
|
-
return
|
|
323
|
+
return sanitizeSlackLabel(
|
|
324
324
|
label === undefined ? undefined : decodeSlackHtmlEntities(label),
|
|
325
325
|
);
|
|
326
326
|
}
|
|
327
327
|
|
|
328
|
-
function sanitizeOptionalLabel(label: string | undefined): string | undefined {
|
|
329
|
-
return sanitizeSlackLabel(label);
|
|
330
|
-
}
|
|
331
|
-
|
|
332
328
|
function isSlackUserId(value: string): boolean {
|
|
333
329
|
return /^[UW][A-Z0-9]+$/.test(value);
|
|
334
330
|
}
|
package/openapi.yaml
CHANGED
|
@@ -16556,7 +16556,7 @@ paths:
|
|
|
16556
16556
|
type: string
|
|
16557
16557
|
description:
|
|
16558
16558
|
"Filter by provider id. One of: anthropic, openai, gemini, ollama, fireworks, together, openrouter,
|
|
16559
|
-
vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside,
|
|
16559
|
+
vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside, typesafe,
|
|
16560
16560
|
vellum"
|
|
16561
16561
|
responses:
|
|
16562
16562
|
"200":
|
|
@@ -16829,7 +16829,7 @@ paths:
|
|
|
16829
16829
|
type: string
|
|
16830
16830
|
description:
|
|
16831
16831
|
"Filter by provider. One of: anthropic, openai, gemini, ollama, fireworks, together, openrouter,
|
|
16832
|
-
vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside,
|
|
16832
|
+
vercel-ai-gateway, litellm, opencode, openai-compatible, minimax, atlascloud, baseten, poolside, typesafe,
|
|
16833
16833
|
vellum, chatgpt"
|
|
16834
16834
|
responses:
|
|
16835
16835
|
"200":
|
|
@@ -38400,7 +38400,7 @@ components:
|
|
|
38400
38400
|
- atlascloud
|
|
38401
38401
|
- baseten
|
|
38402
38402
|
- poolside
|
|
38403
|
-
-
|
|
38403
|
+
- typesafe
|
|
38404
38404
|
- vellum
|
|
38405
38405
|
- chatgpt
|
|
38406
38406
|
Auth:
|
package/package.json
CHANGED
|
@@ -338,7 +338,7 @@ describe("POST /v1/btw", () => {
|
|
|
338
338
|
expect(options!.config!.modelIntent).toBeUndefined();
|
|
339
339
|
});
|
|
340
340
|
|
|
341
|
-
test("greeting requests pass callSite: 'emptyStateGreeting'", async () => {
|
|
341
|
+
test("greeting requests pass callSite: 'emptyStateGreeting' and send no tools", async () => {
|
|
342
342
|
const provider = makeMockProvider();
|
|
343
343
|
const session = makeMockSession(provider);
|
|
344
344
|
mockGetOrCreateConversation.mockImplementationOnce(async () => session);
|
|
@@ -352,6 +352,9 @@ describe("POST /v1/btw", () => {
|
|
|
352
352
|
expect(provider.sendMessage).toHaveBeenCalledTimes(1);
|
|
353
353
|
const [, options] = provider.sendMessage.mock.calls[0];
|
|
354
354
|
expect(options!.config!.callSite).toBe("emptyStateGreeting");
|
|
355
|
+
// The greeting targets no real conversation, so there is no cache prefix
|
|
356
|
+
// for tool definitions to share; they would only cost tokens.
|
|
357
|
+
expect(options!.tools).toEqual([]);
|
|
355
358
|
});
|
|
356
359
|
|
|
357
360
|
test("greeting requests include fresh turn context using the client timezone", async () => {
|
|
@@ -673,6 +673,35 @@ describe("injectChannelCapabilityContext", () => {
|
|
|
673
673
|
const text = (result.content[0] as { type: "text"; text: string }).text;
|
|
674
674
|
expect(text).not.toContain("Do NOT use markdown tables");
|
|
675
675
|
});
|
|
676
|
+
|
|
677
|
+
test("injects email send CLI constraint for email channel", () => {
|
|
678
|
+
const caps: ChannelCapabilities = {
|
|
679
|
+
channel: "email",
|
|
680
|
+
dashboardCapable: false,
|
|
681
|
+
supportsDynamicUi: false,
|
|
682
|
+
supportsVoiceInput: false,
|
|
683
|
+
};
|
|
684
|
+
|
|
685
|
+
const result = injectChannelCapabilityContext(baseUserMessage, caps);
|
|
686
|
+
const text = (result.content[0] as { type: "text"; text: string }).text;
|
|
687
|
+
expect(text).toContain("Conversation text is not emailed");
|
|
688
|
+
expect(text).toContain("assistant email send");
|
|
689
|
+
expect(text).toContain("--reply-to");
|
|
690
|
+
});
|
|
691
|
+
|
|
692
|
+
test("does NOT inject email send CLI constraint for non-email channels", () => {
|
|
693
|
+
const caps: ChannelCapabilities = {
|
|
694
|
+
channel: "telegram",
|
|
695
|
+
dashboardCapable: false,
|
|
696
|
+
supportsDynamicUi: false,
|
|
697
|
+
supportsVoiceInput: false,
|
|
698
|
+
};
|
|
699
|
+
|
|
700
|
+
const result = injectChannelCapabilityContext(baseUserMessage, caps);
|
|
701
|
+
const text = (result.content[0] as { type: "text"; text: string }).text;
|
|
702
|
+
expect(text).not.toContain("Conversation text is not emailed");
|
|
703
|
+
expect(text).not.toContain("assistant email send");
|
|
704
|
+
});
|
|
676
705
|
});
|
|
677
706
|
|
|
678
707
|
// ---------------------------------------------------------------------------
|
|
@@ -234,12 +234,12 @@ describe("LLM catalog parity: daemon vs client", () => {
|
|
|
234
234
|
});
|
|
235
235
|
|
|
236
236
|
test("jev-latest opts out of chat text generation", () => {
|
|
237
|
-
expect(catalogModelSupportsText("
|
|
237
|
+
expect(catalogModelSupportsText("typesafe", "jev-latest")).toBe(false);
|
|
238
238
|
expect(catalogModelSupportsText("anthropic", "claude-opus-4-8")).toBe(true);
|
|
239
239
|
expect(catalogModelSupportsText("openai-compatible", "local-model")).toBe(
|
|
240
240
|
true,
|
|
241
241
|
);
|
|
242
|
-
expect(DEFAULT_PROVIDER_CHOICES).not.toContain("
|
|
242
|
+
expect(DEFAULT_PROVIDER_CHOICES).not.toContain("typesafe");
|
|
243
243
|
expect(DEFAULT_PROVIDER_CHOICES).toContain("poolside");
|
|
244
244
|
});
|
|
245
245
|
|
|
@@ -73,7 +73,9 @@ describe("LLMSchema.defaultProvider", () => {
|
|
|
73
73
|
});
|
|
74
74
|
|
|
75
75
|
test("rejects a structured-decision catalog provider", () => {
|
|
76
|
-
expect(() =>
|
|
76
|
+
expect(() =>
|
|
77
|
+
DefaultProviderSchema.parse({ provider: "typesafe" }),
|
|
78
|
+
).toThrow();
|
|
77
79
|
});
|
|
78
80
|
|
|
79
81
|
test("rejects an empty connectionName", () => {
|
|
@@ -12,10 +12,11 @@ import { describe, expect, test } from "bun:test";
|
|
|
12
12
|
import { MemoryRetrospectiveConfigSchema } from "../schemas/memory-retrospective.js";
|
|
13
13
|
|
|
14
14
|
describe("memory.retrospective config schema", () => {
|
|
15
|
-
test("an empty block leaves
|
|
15
|
+
test("an empty block leaves skill improvement on and monitoring off", () => {
|
|
16
16
|
const parsed = MemoryRetrospectiveConfigSchema.parse({});
|
|
17
17
|
expect(parsed.enabled).toBe(true);
|
|
18
18
|
expect(parsed.skillImprovement).toBe(true);
|
|
19
|
+
expect(parsed.skillImprovementMonitoring).toBe(false);
|
|
19
20
|
expect(parsed).not.toHaveProperty("forkStrategy");
|
|
20
21
|
});
|
|
21
22
|
|
|
@@ -41,6 +42,18 @@ describe("memory.retrospective config schema", () => {
|
|
|
41
42
|
).toBe(false);
|
|
42
43
|
});
|
|
43
44
|
|
|
45
|
+
test("skillImprovementMonitoring is a boolean-only opt-in", () => {
|
|
46
|
+
const parsed = MemoryRetrospectiveConfigSchema.parse({
|
|
47
|
+
skillImprovementMonitoring: true,
|
|
48
|
+
});
|
|
49
|
+
expect(parsed.skillImprovementMonitoring).toBe(true);
|
|
50
|
+
expect(
|
|
51
|
+
MemoryRetrospectiveConfigSchema.safeParse({
|
|
52
|
+
skillImprovementMonitoring: "true",
|
|
53
|
+
}).success,
|
|
54
|
+
).toBe(false);
|
|
55
|
+
});
|
|
56
|
+
|
|
44
57
|
test("a leftover forkStrategy key is ignored", () => {
|
|
45
58
|
const parsed = MemoryRetrospectiveConfigSchema.parse({
|
|
46
59
|
forkStrategy: "cloning",
|
|
@@ -165,11 +165,7 @@
|
|
|
165
165
|
},
|
|
166
166
|
"description": "Replacement steps (replaces all existing steps)"
|
|
167
167
|
}
|
|
168
|
-
}
|
|
169
|
-
"oneOf": [
|
|
170
|
-
{ "required": ["id"] },
|
|
171
|
-
{ "required": ["enrollment_id", "enrollment_action"] }
|
|
172
|
-
]
|
|
168
|
+
}
|
|
173
169
|
},
|
|
174
170
|
"executor": "tools/sequence-update.ts",
|
|
175
171
|
"execution_target": "host"
|
|
@@ -9,7 +9,7 @@ describe("profileSupportsTextGeneration", () => {
|
|
|
9
9
|
test("false for a structured-decision catalog model", () => {
|
|
10
10
|
expect(
|
|
11
11
|
profileSupportsTextGeneration(
|
|
12
|
-
{ provider: "
|
|
12
|
+
{ provider: "typesafe", model: "jev-latest" },
|
|
13
13
|
{},
|
|
14
14
|
),
|
|
15
15
|
).toBe(false);
|
|
@@ -35,7 +35,7 @@ describe("profileSupportsTextGeneration", () => {
|
|
|
35
35
|
profileSupportsTextGeneration(
|
|
36
36
|
{ mix: [{ profile: "jev" }, { profile: "balanced" }] },
|
|
37
37
|
{
|
|
38
|
-
jev: { provider: "
|
|
38
|
+
jev: { provider: "typesafe", model: "jev-latest" },
|
|
39
39
|
balanced: { provider: "anthropic", model: "claude-opus-4-8" },
|
|
40
40
|
},
|
|
41
41
|
),
|
|
@@ -62,7 +62,7 @@ export const KNOWN_LLM_PROVIDERS = [
|
|
|
62
62
|
"opencode",
|
|
63
63
|
"baseten",
|
|
64
64
|
"poolside",
|
|
65
|
-
"
|
|
65
|
+
"typesafe",
|
|
66
66
|
// Routing identities: "vellum" = the platform-managed route (upstream
|
|
67
67
|
// derived from the model at dispatch) and the catalog owner of
|
|
68
68
|
// Vellum-hosted GPU models; "chatgpt" = the subscription route to OpenAI.
|
|
@@ -11,14 +11,23 @@ export const MemoryRetrospectiveConfigSchema = z
|
|
|
11
11
|
|
|
12
12
|
skillImprovement: z
|
|
13
13
|
.boolean({
|
|
14
|
-
error:
|
|
15
|
-
"memory.retrospective.skillImprovement must be a boolean",
|
|
14
|
+
error: "memory.retrospective.skillImprovement must be a boolean",
|
|
16
15
|
})
|
|
17
16
|
.default(true)
|
|
18
17
|
.describe(
|
|
19
18
|
"Whether retrospectives may discover, refine, and create managed skills from observed procedures. When false, retrospectives still capture ordinary memories through `remember`, but cannot load skill management, search for similar skills, or scaffold managed skills.",
|
|
20
19
|
),
|
|
21
20
|
|
|
21
|
+
skillImprovementMonitoring: z
|
|
22
|
+
.boolean({
|
|
23
|
+
error:
|
|
24
|
+
"memory.retrospective.skillImprovementMonitoring must be a boolean",
|
|
25
|
+
})
|
|
26
|
+
.default(false)
|
|
27
|
+
.describe(
|
|
28
|
+
"Reserved opt-in for monitoring retrospective skill-improvement decisions. This setting currently has no effect.",
|
|
29
|
+
),
|
|
30
|
+
|
|
22
31
|
timeThresholdMs: z
|
|
23
32
|
.number({
|
|
24
33
|
error: "memory.retrospective.timeThresholdMs must be a number",
|
|
@@ -903,6 +903,11 @@ export function buildChannelCapabilityBlock(
|
|
|
903
903
|
"- Do NOT use markdown tables — use bullet lists instead. No markdown headers — use **bold** or CAPS for emphasis.",
|
|
904
904
|
);
|
|
905
905
|
}
|
|
906
|
+
if (caps.channel === "email") {
|
|
907
|
+
lines.push(
|
|
908
|
+
"- Conversation text is not emailed. To reply, run `assistant email send` (see `assistant email send --help`). Use `--reply-to` to keep the thread. Skip a reply only when none is needed.",
|
|
909
|
+
);
|
|
910
|
+
}
|
|
906
911
|
}
|
|
907
912
|
|
|
908
913
|
// Inject group chat etiquette only when the chat type indicates a multi-party
|
|
@@ -2035,20 +2040,6 @@ export async function composeInjectorChain(ctx: TurnContext): Promise<string> {
|
|
|
2035
2040
|
*/
|
|
2036
2041
|
const DEFAULT_PLACEMENT: InjectionPlacement = "append-user-tail";
|
|
2037
2042
|
|
|
2038
|
-
/**
|
|
2039
|
-
* Count leading memory-prefix blocks on a user message's `content`.
|
|
2040
|
-
*
|
|
2041
|
-
* Delegates to {@link countMemoryPrefixBlocks} from
|
|
2042
|
-
* `memory/graph/conversation-graph-memory.js` — the canonical state-machine
|
|
2043
|
-
* for locating the memory-prefix boundary. Reusing it here keeps the
|
|
2044
|
-
* PKB-context / PKB-reminder / NOW splice rules aligned on a single source
|
|
2045
|
-
* of truth so their ordering relative to any memory prefix is stable and
|
|
2046
|
-
* testable.
|
|
2047
|
-
*/
|
|
2048
|
-
function countMemoryPrefixBlocksOnContent(content: ContentBlock[]): number {
|
|
2049
|
-
return countMemoryPrefixBlocks(content);
|
|
2050
|
-
}
|
|
2051
|
-
|
|
2052
2043
|
/**
|
|
2053
2044
|
* Apply one injector block to a `runMessages` array according to its
|
|
2054
2045
|
* declared {@link InjectionPlacement}:
|
|
@@ -2101,9 +2092,7 @@ function applyInjectionBlock(
|
|
|
2101
2092
|
{ ...userTail, content: [...userTail.content, textBlock] },
|
|
2102
2093
|
];
|
|
2103
2094
|
case "after-memory-prefix": {
|
|
2104
|
-
const memoryPrefixCount =
|
|
2105
|
-
userTail.content,
|
|
2106
|
-
);
|
|
2095
|
+
const memoryPrefixCount = countMemoryPrefixBlocks(userTail.content);
|
|
2107
2096
|
return [
|
|
2108
2097
|
...runMessages.slice(0, -1),
|
|
2109
2098
|
{
|
|
@@ -2161,7 +2150,7 @@ function stripTailV2DynamicMemoryPrefix(
|
|
|
2161
2150
|
if (!last || last.role !== "user") {
|
|
2162
2151
|
return messages;
|
|
2163
2152
|
}
|
|
2164
|
-
const prefixCount =
|
|
2153
|
+
const prefixCount = countMemoryPrefixBlocks(last.content);
|
|
2165
2154
|
if (prefixCount === 0) {
|
|
2166
2155
|
return messages;
|
|
2167
2156
|
}
|
|
@@ -10,7 +10,8 @@ import { beforeEach, describe, expect, mock, test } from "bun:test";
|
|
|
10
10
|
import type { ResolvedMcpConfig } from "../../config/schemas/mcp.js";
|
|
11
11
|
|
|
12
12
|
const toolsByServer = new Map<string, Array<{ name: string }>>();
|
|
13
|
-
const
|
|
13
|
+
const connectGates = new Map<string, Promise<void>>();
|
|
14
|
+
const listedServers = new Set<string>();
|
|
14
15
|
let mcpGlobalMaxTools: number | undefined;
|
|
15
16
|
|
|
16
17
|
mock.module("../../config/loader.js", () => ({
|
|
@@ -29,12 +30,13 @@ mock.module("../client.js", () => ({
|
|
|
29
30
|
return null;
|
|
30
31
|
}
|
|
31
32
|
async connect() {
|
|
32
|
-
const
|
|
33
|
-
if (
|
|
34
|
-
await
|
|
33
|
+
const gate = connectGates.get(this.serverId);
|
|
34
|
+
if (gate) {
|
|
35
|
+
await gate;
|
|
35
36
|
}
|
|
36
37
|
}
|
|
37
38
|
async listTools() {
|
|
39
|
+
listedServers.add(this.serverId);
|
|
38
40
|
return (toolsByServer.get(this.serverId) ?? []).map((tool) => ({
|
|
39
41
|
name: tool.name,
|
|
40
42
|
description: `${this.serverId} ${tool.name}`,
|
|
@@ -66,7 +68,8 @@ function configWith(ids: string[]): ResolvedMcpConfig {
|
|
|
66
68
|
describe("McpServerManager tool selection", () => {
|
|
67
69
|
beforeEach(() => {
|
|
68
70
|
toolsByServer.clear();
|
|
69
|
-
|
|
71
|
+
connectGates.clear();
|
|
72
|
+
listedServers.clear();
|
|
70
73
|
mcpGlobalMaxTools = undefined;
|
|
71
74
|
});
|
|
72
75
|
|
|
@@ -98,23 +101,32 @@ describe("McpServerManager tool selection", () => {
|
|
|
98
101
|
test("a slow earlier server does not prevent later servers from connecting", async () => {
|
|
99
102
|
toolsByServer.set("slow", [{ name: "slow_tool" }]);
|
|
100
103
|
toolsByServer.set("fast", [{ name: "fast_tool" }]);
|
|
101
|
-
|
|
104
|
+
let releaseSlow!: () => void;
|
|
105
|
+
connectGates.set(
|
|
106
|
+
"slow",
|
|
107
|
+
new Promise<void>((resolve) => {
|
|
108
|
+
releaseSlow = resolve;
|
|
109
|
+
}),
|
|
110
|
+
);
|
|
102
111
|
|
|
103
112
|
const manager = new McpServerManager();
|
|
104
|
-
const
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
113
|
+
const starting = manager.start(configWith(["slow", "fast"]));
|
|
114
|
+
try {
|
|
115
|
+
await new Promise<void>((resolve) => setImmediate(resolve));
|
|
116
|
+
expect([...listedServers]).toEqual(["fast"]);
|
|
117
|
+
|
|
118
|
+
releaseSlow();
|
|
119
|
+
const started = await starting;
|
|
120
|
+
expect(started.connectedServerCount).toBe(2);
|
|
121
|
+
expect(started.servers.map((server) => server.serverId).sort()).toEqual([
|
|
122
|
+
"fast",
|
|
123
|
+
"slow",
|
|
124
|
+
]);
|
|
125
|
+
} finally {
|
|
126
|
+
releaseSlow();
|
|
127
|
+
await starting;
|
|
128
|
+
await manager.stop();
|
|
129
|
+
}
|
|
118
130
|
});
|
|
119
131
|
|
|
120
132
|
test("a workspace global-max override raises how many tools are kept", async () => {
|
|
@@ -577,6 +577,12 @@ export function isProviderErrorMetadata(
|
|
|
577
577
|
* assistant rows, and turn grouping closes on them, so display merging and
|
|
578
578
|
* the turn resolver agree on boundaries. Takes the raw persisted `metadata`
|
|
579
579
|
* JSON string; malformed JSON and non-assistant roles are never standalone.
|
|
580
|
+
*
|
|
581
|
+
* The web folds adjacent assistant rows again after pagination and reads the
|
|
582
|
+
* same rule off the wire projection in its own `isStandaloneAssistantMessage`
|
|
583
|
+
* (clients/web/src/domains/chat/utils/is-standalone-assistant-message.ts). A
|
|
584
|
+
* kind added here without a matching flag and check there merges on the
|
|
585
|
+
* client anyway.
|
|
580
586
|
*/
|
|
581
587
|
export function isStandaloneAssistantMessage(
|
|
582
588
|
role: string,
|
|
@@ -955,19 +955,6 @@ function likeContainsPattern(query: string): string {
|
|
|
955
955
|
.replace(/_/g, "\\_")}%`;
|
|
956
956
|
}
|
|
957
957
|
|
|
958
|
-
/**
|
|
959
|
-
* Whether the sparse Qdrant `messages_lexical` index — the only source of
|
|
960
|
-
* message-content matches — is a safe read source. Content matching is
|
|
961
|
-
* unavailable (title matches only) until the one-time upgrade backfill has
|
|
962
|
-
* fully drained: a partially populated collection would silently miss older
|
|
963
|
-
* content (an empty result — not a throw). Indexing itself is unconditional
|
|
964
|
-
* host infrastructure, so completion is the only gate; the recall read site
|
|
965
|
-
* applies the same one via the shared {@link isLexicalBackfillComplete}.
|
|
966
|
-
*/
|
|
967
|
-
function isMessageContentSearchAvailable(): boolean {
|
|
968
|
-
return isLexicalBackfillComplete();
|
|
969
|
-
}
|
|
970
|
-
|
|
971
958
|
/**
|
|
972
959
|
* Full-text search across message content.
|
|
973
960
|
*
|
|
@@ -976,9 +963,9 @@ function isMessageContentSearchAvailable(): boolean {
|
|
|
976
963
|
* merged with a `LIKE` match on conversation titles; matching conversations
|
|
977
964
|
* return with their relevant messages, ordered by most recently updated.
|
|
978
965
|
*
|
|
979
|
-
* Content matching is index-only
|
|
966
|
+
* Content matching is index-only: there is no `messages.content` scan
|
|
980
967
|
* fallback and no other content source. Only the title arm can match while
|
|
981
|
-
* the index is not a safe read source ({@link
|
|
968
|
+
* the index is not a safe read source ({@link isLexicalBackfillComplete}),
|
|
982
969
|
* for a query that tokenizes to nothing under the shared tokenizer (non-ASCII
|
|
983
970
|
* or single-char input like "你", "é", "C++"), or when the Qdrant lexical
|
|
984
971
|
* lookup fails (logged). An unindexed or unreachable index yields fewer
|
|
@@ -1017,7 +1004,7 @@ export async function searchConversations(
|
|
|
1017
1004
|
const maxMsgsPerConv = opts?.maxMessagesPerConversation ?? 3;
|
|
1018
1005
|
|
|
1019
1006
|
const hasTokens = hasLexicalTokens(trimmed);
|
|
1020
|
-
const contentSearchAvailable =
|
|
1007
|
+
const contentSearchAvailable = isLexicalBackfillComplete();
|
|
1021
1008
|
|
|
1022
1009
|
// LIKE pattern for title matching (message-content indexes don't cover titles).
|
|
1023
1010
|
const titlePattern = likeContainsPattern(query);
|
|
@@ -85,7 +85,7 @@ mock.module("../../../../../util/logger.js", () => ({
|
|
|
85
85
|
}),
|
|
86
86
|
}));
|
|
87
87
|
|
|
88
|
-
const { selectPool, MemoryV3RetrievalUnavailableError } =
|
|
88
|
+
const { selectPool, MemoryV3RetrievalUnavailableError, TYPE_SAFE_POOL_KEEP_NOUL } =
|
|
89
89
|
await import("../pool-select.js");
|
|
90
90
|
type SelectorPool = Parameters<typeof selectPool>[0];
|
|
91
91
|
|
|
@@ -968,3 +968,121 @@ describe("selectPool: cataloged thinking and forced-tool compatibility", () => {
|
|
|
968
968
|
expect(selection.pages).toEqual([{ slug: "topic-x", sections: [] }]);
|
|
969
969
|
});
|
|
970
970
|
});
|
|
971
|
+
|
|
972
|
+
// ---------------------------------------------------------------------------
|
|
973
|
+
// selectPool: TypeSafe System One noul-per-candidate path.
|
|
974
|
+
// ---------------------------------------------------------------------------
|
|
975
|
+
|
|
976
|
+
function typesafeResponse(answers: Record<string, unknown>): ProviderResponse {
|
|
977
|
+
return {
|
|
978
|
+
model: "jev-latest",
|
|
979
|
+
stopReason: "end_turn",
|
|
980
|
+
usage: { inputTokens: 0, outputTokens: 0 },
|
|
981
|
+
content: [{ type: "text", text: JSON.stringify(answers, null, 2) }],
|
|
982
|
+
rawResponse: { answers },
|
|
983
|
+
};
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
function makeTypesafeProvider(response: ProviderResponse): Provider {
|
|
987
|
+
return {
|
|
988
|
+
name: "typesafe",
|
|
989
|
+
sendMessage: async (messages, options) => {
|
|
990
|
+
providerCalls.push({ messages, options });
|
|
991
|
+
return response;
|
|
992
|
+
},
|
|
993
|
+
};
|
|
994
|
+
}
|
|
995
|
+
|
|
996
|
+
function noulAnswer(noul: number): { type: "noul"; noul: number } {
|
|
997
|
+
return { type: "noul", noul };
|
|
998
|
+
}
|
|
999
|
+
|
|
1000
|
+
describe("selectPool: TypeSafe System One", () => {
|
|
1001
|
+
test("sends one noul per candidate and no select_pages tool", async () => {
|
|
1002
|
+
providerStub = makeTypesafeProvider(
|
|
1003
|
+
typesafeResponse({
|
|
1004
|
+
"1": noulAnswer(0.9),
|
|
1005
|
+
"2": noulAnswer(0.1),
|
|
1006
|
+
"3": noulAnswer(0.8),
|
|
1007
|
+
"4": noulAnswer(0.2),
|
|
1008
|
+
}),
|
|
1009
|
+
);
|
|
1010
|
+
|
|
1011
|
+
await selectPool(makePool(), makeTurn("rollout?"));
|
|
1012
|
+
|
|
1013
|
+
expect(providerCalls).toHaveLength(1);
|
|
1014
|
+
const [call] = providerCalls;
|
|
1015
|
+
expect(call.options?.tools).toBeUndefined();
|
|
1016
|
+
expect(
|
|
1017
|
+
(call.options?.config as Record<string, unknown> | undefined)?.tool_choice,
|
|
1018
|
+
).toBeUndefined();
|
|
1019
|
+
expect(
|
|
1020
|
+
(call.options?.config as Record<string, unknown> | undefined)?.callSite,
|
|
1021
|
+
).toBe("memoryV3SelectL2");
|
|
1022
|
+
|
|
1023
|
+
const payload = JSON.parse(
|
|
1024
|
+
(call.messages[0]!.content[0] as { text: string }).text,
|
|
1025
|
+
) as {
|
|
1026
|
+
state: {
|
|
1027
|
+
candidates: Record<string, { slug: string; text: string }>;
|
|
1028
|
+
current_message: string;
|
|
1029
|
+
selector_instructions: string;
|
|
1030
|
+
};
|
|
1031
|
+
questions: Record<string, { type: string; instructions: string }>;
|
|
1032
|
+
};
|
|
1033
|
+
expect(Object.keys(payload.questions)).toEqual(["1", "2", "3", "4"]);
|
|
1034
|
+
expect(payload.questions["1"]?.type).toBe("noul");
|
|
1035
|
+
expect(payload.questions["1"]?.instructions).toContain("`candidates.1`");
|
|
1036
|
+
expect(payload.state.candidates["1"]?.slug).toBe("page-a");
|
|
1037
|
+
expect(payload.state.candidates["1"]?.text).toBe(CARD_A);
|
|
1038
|
+
expect(payload.state.candidates["3"]?.slug).toBe("topic-x");
|
|
1039
|
+
expect(payload.state.current_message).toBe("rollout?");
|
|
1040
|
+
expect(payload.state.selector_instructions.length).toBeGreaterThan(0);
|
|
1041
|
+
});
|
|
1042
|
+
|
|
1043
|
+
test("keeps candidates at or above the inclusive noul threshold", async () => {
|
|
1044
|
+
providerStub = makeTypesafeProvider(
|
|
1045
|
+
typesafeResponse({
|
|
1046
|
+
"1": noulAnswer(TYPE_SAFE_POOL_KEEP_NOUL),
|
|
1047
|
+
"2": noulAnswer(TYPE_SAFE_POOL_KEEP_NOUL - 0.01),
|
|
1048
|
+
"3": noulAnswer(0.91),
|
|
1049
|
+
"4": noulAnswer(0.12),
|
|
1050
|
+
}),
|
|
1051
|
+
);
|
|
1052
|
+
|
|
1053
|
+
const result = await selectPool(makePool(), makeTurn("rollout?"));
|
|
1054
|
+
expect(result.keptAll).toBe(false);
|
|
1055
|
+
expect(result.pages).toEqual([
|
|
1056
|
+
{ slug: "page-a", sections: [] },
|
|
1057
|
+
{ slug: "topic-x", sections: [] },
|
|
1058
|
+
]);
|
|
1059
|
+
});
|
|
1060
|
+
|
|
1061
|
+
test("an all-below-threshold pool is a deliberate empty selection", async () => {
|
|
1062
|
+
providerStub = makeTypesafeProvider(
|
|
1063
|
+
typesafeResponse({
|
|
1064
|
+
"1": noulAnswer(0.1),
|
|
1065
|
+
"2": noulAnswer(0.2),
|
|
1066
|
+
"3": noulAnswer(0.05),
|
|
1067
|
+
"4": noulAnswer(0.3),
|
|
1068
|
+
}),
|
|
1069
|
+
);
|
|
1070
|
+
|
|
1071
|
+
const result = await selectPool(makePool(), makeTurn("nothing relevant"));
|
|
1072
|
+
expect(result).toEqual({ pages: [], keptAll: false });
|
|
1073
|
+
});
|
|
1074
|
+
|
|
1075
|
+
test("unusable answers throw after the re-prompt retry", async () => {
|
|
1076
|
+
providerStub = makeTypesafeProvider({
|
|
1077
|
+
model: "jev-latest",
|
|
1078
|
+
stopReason: "end_turn",
|
|
1079
|
+
usage: { inputTokens: 0, outputTokens: 0 },
|
|
1080
|
+
content: [{ type: "text", text: "not-json" }],
|
|
1081
|
+
});
|
|
1082
|
+
|
|
1083
|
+
await expect(selectPool(makePool(), makeTurn("x"))).rejects.toThrow(
|
|
1084
|
+
MemoryV3RetrievalUnavailableError,
|
|
1085
|
+
);
|
|
1086
|
+
expect(providerCalls).toHaveLength(3);
|
|
1087
|
+
});
|
|
1088
|
+
});
|
|
@@ -67,7 +67,9 @@
|
|
|
67
67
|
* 3. A SINGLE forced-tool select (`selectPool`) over the whole pool. The
|
|
68
68
|
* result is this turn's selections — current turn only. Cross-turn
|
|
69
69
|
* persistence is the injector's job (net-new blocks frozen into history),
|
|
70
|
-
* not a per-turn re-rendered carry set.
|
|
70
|
+
* not a per-turn re-rendered carry set. When the call site resolves to
|
|
71
|
+
* TypeSafe, `selectPool` asks one System One noul per numbered candidate
|
|
72
|
+
* instead of forcing `select_pages`.
|
|
71
73
|
*/
|
|
72
74
|
|
|
73
75
|
import type { AssistantConfig } from "../../../../config/schema.js";
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Memory v3 — single pool selector.
|
|
3
3
|
*
|
|
4
|
-
* Runs a SINGLE
|
|
4
|
+
* Runs a SINGLE selector call over one unified candidate pool rendered in
|
|
5
5
|
* two segments that share one numbering:
|
|
6
6
|
*
|
|
7
7
|
* 1. STABLE PREFIX — the core+hot lane pages as FULL CARDS (head section +
|
|
@@ -52,6 +52,8 @@
|
|
|
52
52
|
import type {
|
|
53
53
|
ContentBlock,
|
|
54
54
|
Message,
|
|
55
|
+
Provider,
|
|
56
|
+
ProviderResponse,
|
|
55
57
|
ToolUseContent,
|
|
56
58
|
} from "@vellumai/plugin-api";
|
|
57
59
|
import { getConfiguredProvider, safeStringSlice } from "@vellumai/plugin-api";
|
|
@@ -179,7 +181,8 @@ type PoolSelectorAttemptFailureReason =
|
|
|
179
181
|
| "provider_error"
|
|
180
182
|
| "missing_tool_use"
|
|
181
183
|
| "unexpected_tool_name"
|
|
182
|
-
| "schema_mismatch"
|
|
184
|
+
| "schema_mismatch"
|
|
185
|
+
| "unusable_answers";
|
|
183
186
|
|
|
184
187
|
interface PoolSelectorAttemptFailure {
|
|
185
188
|
attempt: number;
|
|
@@ -493,6 +496,248 @@ export function selectAllPoolCandidates(pool: SelectorPool): SelectedPage[] {
|
|
|
493
496
|
);
|
|
494
497
|
}
|
|
495
498
|
|
|
499
|
+
/**
|
|
500
|
+
* Inclusive noul threshold for TypeSafe pool selection. 0.5 is calibrated
|
|
501
|
+
* equal yes/no. This sits slightly below that so a plausible candidate is
|
|
502
|
+
* kept, matching the selector's recall-heavy rule.
|
|
503
|
+
*/
|
|
504
|
+
export const TYPE_SAFE_POOL_KEEP_NOUL = 0.4;
|
|
505
|
+
|
|
506
|
+
/** Catalog id of the TypeSafe System One provider. */
|
|
507
|
+
const TYPE_SAFE_PROVIDER_ID = "typesafe";
|
|
508
|
+
|
|
509
|
+
type TypesafeNoulQuestion = {
|
|
510
|
+
type: "noul";
|
|
511
|
+
instructions: string;
|
|
512
|
+
criteria: { yes: string; no: string };
|
|
513
|
+
};
|
|
514
|
+
|
|
515
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
516
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
517
|
+
}
|
|
518
|
+
|
|
519
|
+
function noulFromAnswer(answer: unknown): number | undefined {
|
|
520
|
+
if (typeof answer === "number" && Number.isFinite(answer)) {
|
|
521
|
+
return answer;
|
|
522
|
+
}
|
|
523
|
+
if (
|
|
524
|
+
isRecord(answer) &&
|
|
525
|
+
typeof answer.noul === "number" &&
|
|
526
|
+
Number.isFinite(answer.noul)
|
|
527
|
+
) {
|
|
528
|
+
return answer.noul;
|
|
529
|
+
}
|
|
530
|
+
return undefined;
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
function typesafeCandidateEntries(
|
|
534
|
+
pool: SelectorPool,
|
|
535
|
+
): Record<string, { slug: Slug; text: string }> {
|
|
536
|
+
const entries: Record<string, { slug: Slug; text: string }> = {};
|
|
537
|
+
pool.stable.forEach((candidate, index) => {
|
|
538
|
+
entries[String(index + 1)] = {
|
|
539
|
+
slug: candidate.slug,
|
|
540
|
+
text: candidate.card,
|
|
541
|
+
};
|
|
542
|
+
});
|
|
543
|
+
pool.finder.forEach((candidate, index) => {
|
|
544
|
+
entries[String(pool.stable.length + index + 1)] = {
|
|
545
|
+
slug: candidate.slug,
|
|
546
|
+
text: renderFinderLine(candidate),
|
|
547
|
+
};
|
|
548
|
+
});
|
|
549
|
+
return entries;
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
function typesafeKeepQuestion(id: string): TypesafeNoulQuestion {
|
|
553
|
+
return {
|
|
554
|
+
type: "noul",
|
|
555
|
+
instructions:
|
|
556
|
+
`Would the upcoming assistant reply draw on \`candidates.${id}\`? ` +
|
|
557
|
+
"Lean inclusive. Facts, current task and event state, register, " +
|
|
558
|
+
"framing, calibration, and relationship texture all count. True when " +
|
|
559
|
+
"the candidate could plausibly inform the reply.",
|
|
560
|
+
criteria: {
|
|
561
|
+
yes: "The reply would draw on this candidate.",
|
|
562
|
+
no: "The reply would not draw on this candidate.",
|
|
563
|
+
},
|
|
564
|
+
};
|
|
565
|
+
}
|
|
566
|
+
|
|
567
|
+
function answersFromSelectorResponse(
|
|
568
|
+
response: ProviderResponse,
|
|
569
|
+
): Record<string, unknown> | null {
|
|
570
|
+
const raw = response.rawResponse;
|
|
571
|
+
if (isRecord(raw) && isRecord(raw.answers)) {
|
|
572
|
+
return raw.answers;
|
|
573
|
+
}
|
|
574
|
+
const textBlock = response.content.find((block) => block.type === "text");
|
|
575
|
+
if (!textBlock || textBlock.type !== "text") {
|
|
576
|
+
return null;
|
|
577
|
+
}
|
|
578
|
+
try {
|
|
579
|
+
const parsed: unknown = JSON.parse(textBlock.text);
|
|
580
|
+
return isRecord(parsed) ? parsed : null;
|
|
581
|
+
} catch {
|
|
582
|
+
return null;
|
|
583
|
+
}
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
async function selectPoolWithTypesafe(
|
|
587
|
+
pool: SelectorPool,
|
|
588
|
+
turn: MemoryRoutingTurn,
|
|
589
|
+
ordered: PoolLine[],
|
|
590
|
+
systemPrompt: string,
|
|
591
|
+
provider: Provider,
|
|
592
|
+
): Promise<PoolSelection> {
|
|
593
|
+
const candidates = typesafeCandidateEntries(pool);
|
|
594
|
+
const questions: Record<string, TypesafeNoulQuestion> = {};
|
|
595
|
+
for (const id of Object.keys(candidates)) {
|
|
596
|
+
questions[id] = typesafeKeepQuestion(id);
|
|
597
|
+
}
|
|
598
|
+
const state = {
|
|
599
|
+
selector_instructions: systemPrompt,
|
|
600
|
+
candidates,
|
|
601
|
+
...(turn.situationalContext
|
|
602
|
+
? { situation: turn.situationalContext }
|
|
603
|
+
: {}),
|
|
604
|
+
recent_context: turn.recentContext,
|
|
605
|
+
current_message: turn.currentMessage,
|
|
606
|
+
};
|
|
607
|
+
const userMsg: Message = {
|
|
608
|
+
role: "user",
|
|
609
|
+
content: [
|
|
610
|
+
{
|
|
611
|
+
type: "text",
|
|
612
|
+
text: JSON.stringify({ state, questions }),
|
|
613
|
+
},
|
|
614
|
+
],
|
|
615
|
+
};
|
|
616
|
+
|
|
617
|
+
const failures: PoolSelectorAttemptFailure[] = [];
|
|
618
|
+
let attempt = 0;
|
|
619
|
+
const recordFailure = (
|
|
620
|
+
failure: Omit<
|
|
621
|
+
PoolSelectorAttemptFailure,
|
|
622
|
+
| "callSite"
|
|
623
|
+
| "providerName"
|
|
624
|
+
| "candidateCount"
|
|
625
|
+
| "stableCount"
|
|
626
|
+
| "finderCount"
|
|
627
|
+
>,
|
|
628
|
+
): void => {
|
|
629
|
+
const diagnostic: PoolSelectorAttemptFailure = {
|
|
630
|
+
...failure,
|
|
631
|
+
callSite: MEMORY_V3_SELECT_CALL_SITE,
|
|
632
|
+
providerName: provider.name,
|
|
633
|
+
candidateCount: ordered.length,
|
|
634
|
+
stableCount: pool.stable.length,
|
|
635
|
+
finderCount: pool.finder.length,
|
|
636
|
+
};
|
|
637
|
+
failures.push(diagnostic);
|
|
638
|
+
log.warn(diagnostic, "pool selector attempt failed");
|
|
639
|
+
};
|
|
640
|
+
|
|
641
|
+
let lastError: unknown = null;
|
|
642
|
+
const parsed = await retryForResult(async () => {
|
|
643
|
+
attempt += 1;
|
|
644
|
+
let response: Awaited<ReturnType<typeof provider.sendMessage>>;
|
|
645
|
+
try {
|
|
646
|
+
response = await provider.sendMessage([userMsg], {
|
|
647
|
+
config: {
|
|
648
|
+
callSite: MEMORY_V3_SELECT_CALL_SITE,
|
|
649
|
+
conversationId: turn.conversationId,
|
|
650
|
+
disableTurnStartCache: true,
|
|
651
|
+
},
|
|
652
|
+
});
|
|
653
|
+
lastError = null;
|
|
654
|
+
} catch (error) {
|
|
655
|
+
lastError = error;
|
|
656
|
+
recordFailure({
|
|
657
|
+
attempt,
|
|
658
|
+
reason: "provider_error",
|
|
659
|
+
error: summarizeError(error),
|
|
660
|
+
});
|
|
661
|
+
throw error;
|
|
662
|
+
}
|
|
663
|
+
const answers = answersFromSelectorResponse(response);
|
|
664
|
+
if (!answers) {
|
|
665
|
+
recordFailure({
|
|
666
|
+
attempt,
|
|
667
|
+
reason: "unusable_answers",
|
|
668
|
+
response: summarizeResponse(response),
|
|
669
|
+
});
|
|
670
|
+
return null;
|
|
671
|
+
}
|
|
672
|
+
const picked: number[] = [];
|
|
673
|
+
let parsedCount = 0;
|
|
674
|
+
for (let index = 0; index < ordered.length; index++) {
|
|
675
|
+
const noul = noulFromAnswer(answers[String(index + 1)]);
|
|
676
|
+
if (noul === undefined) {
|
|
677
|
+
continue;
|
|
678
|
+
}
|
|
679
|
+
parsedCount += 1;
|
|
680
|
+
if (noul >= TYPE_SAFE_POOL_KEEP_NOUL) {
|
|
681
|
+
picked.push(index);
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
if (parsedCount === 0) {
|
|
685
|
+
recordFailure({
|
|
686
|
+
attempt,
|
|
687
|
+
reason: "unusable_answers",
|
|
688
|
+
response: summarizeResponse(response),
|
|
689
|
+
});
|
|
690
|
+
return null;
|
|
691
|
+
}
|
|
692
|
+
return { pages: mergeSelectedLines(ordered, picked), keptAll: false };
|
|
693
|
+
});
|
|
694
|
+
|
|
695
|
+
if (parsed === null) {
|
|
696
|
+
if (lastError !== null) {
|
|
697
|
+
const detail =
|
|
698
|
+
lastError instanceof Error ? lastError.message : String(lastError);
|
|
699
|
+
const redactedDetail = truncate(
|
|
700
|
+
redactLogString(detail),
|
|
701
|
+
ERROR_MESSAGE_MAX_CHARS,
|
|
702
|
+
);
|
|
703
|
+
log.warn(
|
|
704
|
+
{
|
|
705
|
+
candidateCount: ordered.length,
|
|
706
|
+
stableCount: pool.stable.length,
|
|
707
|
+
finderCount: pool.finder.length,
|
|
708
|
+
callSite: MEMORY_V3_SELECT_CALL_SITE,
|
|
709
|
+
providerName: provider.name,
|
|
710
|
+
failures,
|
|
711
|
+
},
|
|
712
|
+
"pool selector provider call failed after retries",
|
|
713
|
+
);
|
|
714
|
+
throw new MemoryV3RetrievalUnavailableError(
|
|
715
|
+
`memory-v3 pool selector provider call failed after retries: ${redactedDetail}`,
|
|
716
|
+
{
|
|
717
|
+
cause: lastError,
|
|
718
|
+
conversationNotice: providerBillingNoticeFromError(lastError),
|
|
719
|
+
},
|
|
720
|
+
);
|
|
721
|
+
}
|
|
722
|
+
log.warn(
|
|
723
|
+
{
|
|
724
|
+
candidateCount: ordered.length,
|
|
725
|
+
stableCount: pool.stable.length,
|
|
726
|
+
finderCount: pool.finder.length,
|
|
727
|
+
callSite: MEMORY_V3_SELECT_CALL_SITE,
|
|
728
|
+
providerName: provider.name,
|
|
729
|
+
failures,
|
|
730
|
+
},
|
|
731
|
+
"pool selector returned no usable TypeSafe answers after retries",
|
|
732
|
+
);
|
|
733
|
+
throw new MemoryV3RetrievalUnavailableError(
|
|
734
|
+
"memory-v3 pool selector returned no usable selection after retries",
|
|
735
|
+
);
|
|
736
|
+
}
|
|
737
|
+
|
|
738
|
+
return parsed;
|
|
739
|
+
}
|
|
740
|
+
|
|
496
741
|
/** A selection plus whether it came from the recall-safe keep-all fallback. */
|
|
497
742
|
export interface PoolSelection {
|
|
498
743
|
pages: SelectedPage[];
|
|
@@ -504,14 +749,16 @@ export interface PoolSelection {
|
|
|
504
749
|
}
|
|
505
750
|
|
|
506
751
|
/**
|
|
507
|
-
* Run the single
|
|
752
|
+
* Run the single selector over the unified candidate pool. Returns
|
|
508
753
|
* the pages to inject, merged per slug (a page selected as a card and on
|
|
509
754
|
* finder lines yields one entry carrying every selected section), plus a
|
|
510
755
|
* `keptAll` flag marking the recall-safe fallback.
|
|
511
756
|
*
|
|
512
|
-
*
|
|
513
|
-
* relevant" signal, `keptAll: true`); an explicit `[]`
|
|
514
|
-
*
|
|
757
|
+
* On a chat model, an omitted `ids` keeps ALL candidates (the recall-safe
|
|
758
|
+
* "all of these are relevant" signal, `keptAll: true`); an explicit `[]`
|
|
759
|
+
* keeps none. TypeSafe answers one noul per candidate and never omits ids,
|
|
760
|
+
* so `keptAll` is always false on that path. An infrastructure failure
|
|
761
|
+
* (after a short re-prompt retry) throws
|
|
515
762
|
* {@link MemoryV3RetrievalUnavailableError}, and the orchestrator keeps the
|
|
516
763
|
* stable prefix unjudged in its place.
|
|
517
764
|
*
|
|
@@ -545,6 +792,16 @@ export async function selectPool(
|
|
|
545
792
|
);
|
|
546
793
|
}
|
|
547
794
|
|
|
795
|
+
if (provider.name === TYPE_SAFE_PROVIDER_ID) {
|
|
796
|
+
return selectPoolWithTypesafe(
|
|
797
|
+
pool,
|
|
798
|
+
turn,
|
|
799
|
+
ordered,
|
|
800
|
+
systemPrompt,
|
|
801
|
+
provider,
|
|
802
|
+
);
|
|
803
|
+
}
|
|
804
|
+
|
|
548
805
|
// Two content blocks: the stable prefix (cards) carries the cache
|
|
549
806
|
// breakpoint; the dynamic tail (finder lines + per-turn context) does not.
|
|
550
807
|
// See the module doc for the cache contract.
|
|
@@ -206,7 +206,7 @@ const ADAPTER_FACTORIES: Record<string, AdapterFactory> = {
|
|
|
206
206
|
streamTimeoutMs,
|
|
207
207
|
...(baseURL ? { baseURL } : {}),
|
|
208
208
|
}),
|
|
209
|
-
|
|
209
|
+
typesafe: ({ apiKey, model, streamTimeoutMs, baseURL }) =>
|
|
210
210
|
new JevProvider(apiKey, model, {
|
|
211
211
|
streamTimeoutMs,
|
|
212
212
|
...(baseURL ? { baseURL } : {}),
|
|
@@ -5,6 +5,7 @@ import {
|
|
|
5
5
|
conversationToState,
|
|
6
6
|
DEFAULT_JEV_MODEL,
|
|
7
7
|
JevProvider,
|
|
8
|
+
noulFromAnswer,
|
|
8
9
|
parseSystemOneOverride,
|
|
9
10
|
validateJevApiKey,
|
|
10
11
|
} from "./client.js";
|
|
@@ -66,6 +67,15 @@ describe("parseSystemOneOverride", () => {
|
|
|
66
67
|
});
|
|
67
68
|
});
|
|
68
69
|
|
|
70
|
+
describe("noulFromAnswer", () => {
|
|
71
|
+
test("reads a typed noul object or a bare number", () => {
|
|
72
|
+
expect(noulFromAnswer({ type: "noul", noul: 0.91 })).toBe(0.91);
|
|
73
|
+
expect(noulFromAnswer(0.4)).toBe(0.4);
|
|
74
|
+
expect(noulFromAnswer({ type: "choice", choice: "keep" })).toBeUndefined();
|
|
75
|
+
expect(noulFromAnswer(null)).toBeUndefined();
|
|
76
|
+
});
|
|
77
|
+
});
|
|
78
|
+
|
|
69
79
|
describe("conversationToState", () => {
|
|
70
80
|
test("flattens system prompt and turns into a string", () => {
|
|
71
81
|
const state = conversationToState(
|
|
@@ -15,7 +15,7 @@ import type {
|
|
|
15
15
|
|
|
16
16
|
const log = getLogger("jev-client");
|
|
17
17
|
|
|
18
|
-
export const JEV_PROVIDER_ID = "
|
|
18
|
+
export const JEV_PROVIDER_ID = "typesafe";
|
|
19
19
|
export const DEFAULT_JEV_BASE_URL = "https://api.typesafe.ai";
|
|
20
20
|
export const DEFAULT_JEV_MODEL = "jev-latest";
|
|
21
21
|
|
|
@@ -115,6 +115,21 @@ function isJevQuestions(value: unknown): value is JevQuestions {
|
|
|
115
115
|
return entries.every(([, question]) => isJevQuestion(question));
|
|
116
116
|
}
|
|
117
117
|
|
|
118
|
+
/** Calibrated P(yes) from a System One noul answer, or undefined when absent. */
|
|
119
|
+
export function noulFromAnswer(answer: unknown): number | undefined {
|
|
120
|
+
if (typeof answer === "number" && Number.isFinite(answer)) {
|
|
121
|
+
return answer;
|
|
122
|
+
}
|
|
123
|
+
if (
|
|
124
|
+
isRecord(answer) &&
|
|
125
|
+
typeof answer.noul === "number" &&
|
|
126
|
+
Number.isFinite(answer.noul)
|
|
127
|
+
) {
|
|
128
|
+
return answer.noul;
|
|
129
|
+
}
|
|
130
|
+
return undefined;
|
|
131
|
+
}
|
|
132
|
+
|
|
118
133
|
/**
|
|
119
134
|
* If the last user message is a TypeSafe System One payload, use it as the
|
|
120
135
|
* evaluation request. `state` is optional; callers fall back to the
|
|
@@ -77,7 +77,8 @@ export interface CatalogModel {
|
|
|
77
77
|
* Whether the model produces free-form chat text. Omit (or true) for
|
|
78
78
|
* ordinary chat models. False for structured-decision models that return
|
|
79
79
|
* answers rather than generated text; those stay out of conversation
|
|
80
|
-
* pickers and cannot be the conversation model.
|
|
80
|
+
* pickers and cannot be the conversation model. They can still back a
|
|
81
|
+
* saved profile and a call-site pin.
|
|
81
82
|
*/
|
|
82
83
|
supportsText?: boolean;
|
|
83
84
|
supportsEffort?: boolean;
|
|
@@ -2532,8 +2533,8 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
|
|
|
2532
2533
|
apiKeyPlaceholder: "Your Poolside API key",
|
|
2533
2534
|
},
|
|
2534
2535
|
{
|
|
2535
|
-
id: "
|
|
2536
|
-
displayName: "
|
|
2536
|
+
id: "typesafe",
|
|
2537
|
+
displayName: "TypeSafe",
|
|
2537
2538
|
subtitle:
|
|
2538
2539
|
"TypeSafe System One decision model. Returns structured answers, not generated text. Requires a TypeSafe API key.",
|
|
2539
2540
|
setupMode: "api-key",
|
|
@@ -388,7 +388,7 @@ export async function routeGuardianReply(
|
|
|
388
388
|
const request = await getGuardianRequestOrNull(answerTap.requestId);
|
|
389
389
|
if (
|
|
390
390
|
request &&
|
|
391
|
-
|
|
391
|
+
resolveGuardianInstructionModeForRequest(request) === "answer" &&
|
|
392
392
|
parseQuestionAnswerActionId(answerTap.token) &&
|
|
393
393
|
!request.callSessionId &&
|
|
394
394
|
hasLiveQuestionInteraction(request.id)
|
|
@@ -581,7 +581,7 @@ export async function routeGuardianReply(
|
|
|
581
581
|
if (messageText.length > 0 && pendingRequests.length === 1) {
|
|
582
582
|
const soleRequest = pendingRequests[0];
|
|
583
583
|
if (
|
|
584
|
-
|
|
584
|
+
resolveGuardianInstructionModeForRequest(soleRequest) === "answer" &&
|
|
585
585
|
!soleRequest.callSessionId &&
|
|
586
586
|
soleRequest.sourceConversationId === conversationId &&
|
|
587
587
|
hasLiveQuestionInteraction(soleRequest.id)
|
|
@@ -1005,12 +1005,6 @@ function inferActionFromText(
|
|
|
1005
1005
|
return "approve_once";
|
|
1006
1006
|
}
|
|
1007
1007
|
|
|
1008
|
-
function resolveRequestInstructionMode(
|
|
1009
|
-
request?: Pick<GuardianRequestWire, "kind" | "toolName"> | null,
|
|
1010
|
-
): "approval" | "answer" {
|
|
1011
|
-
return resolveGuardianInstructionModeForRequest(request);
|
|
1012
|
-
}
|
|
1013
|
-
|
|
1014
1008
|
// ---------------------------------------------------------------------------
|
|
1015
1009
|
// Failure reason reply text
|
|
1016
1010
|
// ---------------------------------------------------------------------------
|
|
@@ -1044,7 +1038,7 @@ function failureReplyText(
|
|
|
1044
1038
|
return "Something went wrong with this request on our end, so I couldn't apply your decision.";
|
|
1045
1039
|
case "invalid_action":
|
|
1046
1040
|
return buildGuardianInvalidActionReply(
|
|
1047
|
-
|
|
1041
|
+
resolveGuardianInstructionModeForRequest(request),
|
|
1048
1042
|
requestCode ?? undefined,
|
|
1049
1043
|
);
|
|
1050
1044
|
default:
|
|
@@ -1063,7 +1057,7 @@ function failureReplyText(
|
|
|
1063
1057
|
*/
|
|
1064
1058
|
function composeCodeOnlyClarification(request: GuardianRequestWire): string {
|
|
1065
1059
|
const code = request.requestCode ?? "unknown";
|
|
1066
|
-
const mode =
|
|
1060
|
+
const mode = resolveGuardianInstructionModeForRequest(request);
|
|
1067
1061
|
return buildGuardianCodeOnlyClarification(mode, {
|
|
1068
1062
|
requestCode: code,
|
|
1069
1063
|
questionText: request.questionText,
|
|
@@ -1087,7 +1081,7 @@ function composeDisambiguationReply(
|
|
|
1087
1081
|
const lines: string[] = [];
|
|
1088
1082
|
const requestsWithMode = pendingRequests.map((request) => ({
|
|
1089
1083
|
request,
|
|
1090
|
-
mode:
|
|
1084
|
+
mode: resolveGuardianInstructionModeForRequest(request),
|
|
1091
1085
|
}));
|
|
1092
1086
|
|
|
1093
1087
|
if (engineReplyText) {
|
|
@@ -2,9 +2,10 @@
|
|
|
2
2
|
* Route handler for the POST /v1/btw SSE-streaming side-chain endpoint.
|
|
3
3
|
*
|
|
4
4
|
* Runs an ephemeral LLM call that reuses the conversation's provider, tool
|
|
5
|
-
* definitions, and message history for prompt-cache efficiency
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* definitions, and message history for prompt-cache efficiency; the
|
|
6
|
+
* empty-state greeting targets no real conversation and sends no tools. Uses
|
|
7
|
+
* the conversation's system prompt when a conversation-specific override is
|
|
8
|
+
* active; otherwise builds a fresh prompt excluding BOOTSTRAP.md so first-run
|
|
8
9
|
* onboarding instructions don't leak into cosmetic UI calls like identity
|
|
9
10
|
* intro generation. The response is streamed as SSE events (`btw_text_delta`,
|
|
10
11
|
* `btw_complete`, `btw_error`).
|
|
@@ -145,7 +146,11 @@ async function handleBtw({
|
|
|
145
146
|
const result = await runBtwSidechain({
|
|
146
147
|
content: effectiveContent,
|
|
147
148
|
conversation,
|
|
148
|
-
|
|
149
|
+
// The side-chain forces `tool_choice: none`, so tool definitions
|
|
150
|
+
// only earn their tokens as a shared cache prefix with a real
|
|
151
|
+
// conversation. The greeting runs against an ephemeral one with its
|
|
152
|
+
// own system prompt, so it shares nothing and sends no tools.
|
|
153
|
+
tools: isGreeting ? [] : getAllToolDefinitions(),
|
|
149
154
|
signal: abortSignal,
|
|
150
155
|
...(isGreeting ? { callSite: "emptyStateGreeting" as const } : {}),
|
|
151
156
|
onEvent: (event) => {
|
|
@@ -228,7 +228,7 @@ function getIdentity() {
|
|
|
228
228
|
|
|
229
229
|
const version = APP_VERSION;
|
|
230
230
|
|
|
231
|
-
const createdAt =
|
|
231
|
+
const createdAt = resolveHatchedAtReadOnly(identityPath);
|
|
232
232
|
|
|
233
233
|
return {
|
|
234
234
|
name: fields.name ?? "",
|
|
@@ -241,10 +241,6 @@ function getIdentity() {
|
|
|
241
241
|
};
|
|
242
242
|
}
|
|
243
243
|
|
|
244
|
-
function resolveIdentityCreatedAt(identityPath: string): string | undefined {
|
|
245
|
-
return resolveHatchedAtReadOnly(identityPath);
|
|
246
|
-
}
|
|
247
|
-
|
|
248
244
|
// ---------------------------------------------------------------------------
|
|
249
245
|
// Zod schemas for profiler health metadata
|
|
250
246
|
// ---------------------------------------------------------------------------
|
|
@@ -303,7 +303,7 @@ async function handleAddSecret({ body }: RouteHandlerArgs) {
|
|
|
303
303
|
);
|
|
304
304
|
return { success: false, error: validation.reason };
|
|
305
305
|
}
|
|
306
|
-
} else if (name === "
|
|
306
|
+
} else if (name === "typesafe") {
|
|
307
307
|
const validation = await validateJevApiKey(value);
|
|
308
308
|
if (!validation.valid) {
|
|
309
309
|
log.warn(
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { dirname, join, resolve } from "node:path";
|
|
2
|
+
import { fileURLToPath } from "node:url";
|
|
3
|
+
import { describe, expect, test } from "bun:test";
|
|
4
|
+
|
|
5
|
+
import { parseToolManifestFile } from "../../skills/tool-manifest.js";
|
|
6
|
+
import { explicitTools } from "../tool-manifest.js";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Tool input schemas must be plain objects at the root.
|
|
10
|
+
*
|
|
11
|
+
* Anthropic's Messages API rejects a tool whose `input_schema` carries
|
|
12
|
+
* `oneOf`, `anyOf`, or `allOf` at the top level ("input_schema does not
|
|
13
|
+
* support oneOf, allOf, or anyOf at the top level"). The rejection is a 400
|
|
14
|
+
* for the whole request, so one offending definition takes down every call
|
|
15
|
+
* that advertises it. Other hosts accept the same schema, which lets the
|
|
16
|
+
* mistake hide until a request routes to Anthropic directly. Either/or rules
|
|
17
|
+
* between fields belong in the tool description and the tool's own input
|
|
18
|
+
* validation instead.
|
|
19
|
+
*
|
|
20
|
+
* Covers the core manifest and every bundled skill's `TOOLS.json`.
|
|
21
|
+
* Combinators nested under `properties` are accepted by Anthropic and stay
|
|
22
|
+
* out of scope.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
const ROOT_COMBINATORS = ["oneOf", "anyOf", "allOf"] as const;
|
|
26
|
+
|
|
27
|
+
const BUNDLED_SKILLS_DIR = resolve(
|
|
28
|
+
dirname(fileURLToPath(import.meta.url)),
|
|
29
|
+
"../../config/bundled-skills",
|
|
30
|
+
);
|
|
31
|
+
|
|
32
|
+
interface SchemaCase {
|
|
33
|
+
/** Tool name, prefixed with the skill directory for bundled skill tools. */
|
|
34
|
+
label: string;
|
|
35
|
+
schema: Record<string, unknown>;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function bundledSkillSchemas(): SchemaCase[] {
|
|
39
|
+
const cases: SchemaCase[] = [];
|
|
40
|
+
for (const relative of new Bun.Glob("*/TOOLS.json").scanSync({
|
|
41
|
+
cwd: BUNDLED_SKILLS_DIR,
|
|
42
|
+
})) {
|
|
43
|
+
const manifest = parseToolManifestFile(join(BUNDLED_SKILLS_DIR, relative));
|
|
44
|
+
for (const tool of manifest.tools) {
|
|
45
|
+
cases.push({
|
|
46
|
+
label: `${dirname(relative)}/${tool.name}`,
|
|
47
|
+
schema: tool.input_schema,
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
return cases;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
function coreSchemas(): SchemaCase[] {
|
|
55
|
+
return explicitTools.map((tool) => {
|
|
56
|
+
if (!tool.name) {
|
|
57
|
+
throw new Error("core manifest entries carry explicit names");
|
|
58
|
+
}
|
|
59
|
+
return {
|
|
60
|
+
label: tool.name,
|
|
61
|
+
schema: tool.input_schema as Record<string, unknown>,
|
|
62
|
+
};
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const CASES: SchemaCase[] = [...coreSchemas(), ...bundledSkillSchemas()];
|
|
67
|
+
|
|
68
|
+
describe("tool input schema root", () => {
|
|
69
|
+
test("covers the core manifest and the bundled skills", () => {
|
|
70
|
+
expect(CASES.length).toBeGreaterThan(explicitTools.length);
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
for (const { label, schema } of CASES) {
|
|
74
|
+
test(`${label} keeps combinators out of the schema root`, () => {
|
|
75
|
+
for (const keyword of ROOT_COMBINATORS) {
|
|
76
|
+
expect(schema).not.toHaveProperty(keyword);
|
|
77
|
+
}
|
|
78
|
+
expect(schema.type).toBe("object");
|
|
79
|
+
});
|
|
80
|
+
}
|
|
81
|
+
});
|
|
@@ -124,6 +124,8 @@ const DESCRIPTION = [
|
|
|
124
124
|
"For logins, use saved credentials first. Securely collect missing credentials",
|
|
125
125
|
"with assistant credentials prompt, then fill the login form yourself.",
|
|
126
126
|
"",
|
|
127
|
+
"Every call passes exactly one of `questions` or `desktopHelp`.",
|
|
128
|
+
"",
|
|
127
129
|
"Use this tool whenever a request is ambiguous and can be resolved",
|
|
128
130
|
"by 2–4 plausible interpretations or discrete choices. Prefer it over",
|
|
129
131
|
"plain-text clarification — structured options are faster to answer and",
|
|
@@ -255,10 +257,10 @@ export const askQuestionTool = {
|
|
|
255
257
|
category: "interaction",
|
|
256
258
|
executionTarget: "sandbox",
|
|
257
259
|
defaultRiskLevel: RiskLevel.Low,
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
260
|
+
// Anthropic rejects `oneOf` / `anyOf` / `allOf` at the root of a tool
|
|
261
|
+
// schema, so the either/or rule between `questions` and `desktopHelp` lives
|
|
262
|
+
// in the description and the Zod refine, not in the wire schema.
|
|
263
|
+
input_schema: toToolInputSchema(askQuestionInputSchema),
|
|
262
264
|
|
|
263
265
|
async execute(
|
|
264
266
|
input: Record<string, unknown>,
|
|
@@ -23,10 +23,6 @@ import {
|
|
|
23
23
|
} from "../shared/zod-tool-schema.js";
|
|
24
24
|
import type { ToolContext, ToolExecutionResult } from "../types.js";
|
|
25
25
|
|
|
26
|
-
function isPrivilegedDocumentActor(context: ToolContext): boolean {
|
|
27
|
-
return canActOnPrivilegedDocuments(context);
|
|
28
|
-
}
|
|
29
|
-
|
|
30
26
|
export function documentNotFound(surfaceId: string): ToolExecutionResult {
|
|
31
27
|
return {
|
|
32
28
|
content: JSON.stringify({
|
|
@@ -43,7 +39,7 @@ export function canAccessDocument(
|
|
|
43
39
|
context: ToolContext,
|
|
44
40
|
): boolean {
|
|
45
41
|
return (
|
|
46
|
-
|
|
42
|
+
canActOnPrivilegedDocuments(context) ||
|
|
47
43
|
isDocumentAssociatedWithConversation(surfaceId, context.conversationId)
|
|
48
44
|
);
|
|
49
45
|
}
|
|
@@ -522,7 +518,7 @@ export function executeDocumentList(
|
|
|
522
518
|
const docs = query
|
|
523
519
|
? searchDocumentsByTitle(
|
|
524
520
|
query,
|
|
525
|
-
|
|
521
|
+
canActOnPrivilegedDocuments(context)
|
|
526
522
|
? {}
|
|
527
523
|
: { conversationId: context.conversationId },
|
|
528
524
|
)
|