@selesai/code 0.13.33 → 0.13.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/extensions/capability-gateway/catalog.ts +4 -1
- package/dist/extensions/capability-gateway/index.test.ts +93 -4
- package/dist/extensions/capability-gateway/index.ts +294 -59
- package/dist/extensions/capability-gateway/integration.test.ts +35 -0
- package/dist/extensions/capability-gateway/routing.test.ts +154 -1
- package/dist/extensions/capability-gateway/routing.ts +227 -45
- package/dist/extensions/jev/decisions.test.ts +37 -0
- package/dist/extensions/jev/decisions.ts +148 -43
- package/dist/extensions/jev-ask-tool.test.ts +436 -0
- package/dist/extensions/jev-ask-tool.ts +587 -0
- package/dist/extensions/package.json +1 -0
- package/dist/extensions/pi-hermes-memory/README.md +11 -36
- package/dist/extensions/pi-hermes-memory/src/config.ts +36 -8
- package/dist/extensions/pi-hermes-memory/src/constants.ts +5 -5
- package/dist/extensions/pi-hermes-memory/src/handlers/auto-consolidate.ts +60 -46
- package/dist/extensions/pi-hermes-memory/tests/config.test.ts +26 -2
- package/dist/extensions/pi-hermes-memory/tests/handlers/auto-consolidate.test.ts +7 -1
- package/dist/extensions/pi-intercom/index.ts +5 -1
- package/dist/extensions/rtk.test.ts +21 -13
- package/dist/extensions/tps.test.ts +32 -1
- package/dist/extensions/tps.ts +3 -1
- package/docs/settings.md +46 -7
- package/package.json +3 -3
|
@@ -208,6 +208,7 @@ describe("capability gateway integration", () => {
|
|
|
208
208
|
const prompt = h.session.systemPrompt;
|
|
209
209
|
expect(prompt).toContain("capability_catalog");
|
|
210
210
|
expect(prompt).toContain("Never invent optional tool names");
|
|
211
|
+
expect(prompt).toContain("Never tell the user a tool is unavailable");
|
|
211
212
|
// The full skill list is gone from the default prompt: the research
|
|
212
213
|
// skill name and body are absent, only the compact instruction remains.
|
|
213
214
|
expect(prompt).not.toContain("Research instructions body.");
|
|
@@ -604,6 +605,40 @@ describe("capability gateway Jev routing", () => {
|
|
|
604
605
|
);
|
|
605
606
|
});
|
|
606
607
|
|
|
608
|
+
it("leaves the job route out of the skill index while Jev has no credential", async () => {
|
|
609
|
+
const h = await createGatewaySession({ enabled: true, extensions: [GATEWAY_DIR, GREP_APP_DIR] });
|
|
610
|
+
harnesses.push(h);
|
|
611
|
+
|
|
612
|
+
await route(h, NO_SIGNAL_PROMPT);
|
|
613
|
+
|
|
614
|
+
expect(h.session.systemPrompt).not.toContain("capability_discover with `job`");
|
|
615
|
+
expect(h.session.systemPrompt).toContain("then call capability_discover with the exact `name`");
|
|
616
|
+
});
|
|
617
|
+
|
|
618
|
+
it("offers the job route in the skill index while Jev is reachable", async () => {
|
|
619
|
+
allowNetwork();
|
|
620
|
+
stubJev("grep_app_search");
|
|
621
|
+
const h = await createGatewaySession({ enabled: true, extensions: [GATEWAY_DIR, GREP_APP_DIR], jev: jevSettings() });
|
|
622
|
+
harnesses.push(h);
|
|
623
|
+
|
|
624
|
+
await route(h, NO_SIGNAL_PROMPT);
|
|
625
|
+
|
|
626
|
+
expect(h.session.systemPrompt).toContain("capability_discover with `job`");
|
|
627
|
+
});
|
|
628
|
+
|
|
629
|
+
it("reports an already-active tool instead of activating it again", async () => {
|
|
630
|
+
const h = await createGatewaySession({ enabled: true, extensions: [GATEWAY_DIR, GREP_APP_DIR] });
|
|
631
|
+
harnesses.push(h);
|
|
632
|
+
await route(h, "search github code with grep_app_search");
|
|
633
|
+
expect(h.session.getActiveToolNames()).toContain("grep_app_search");
|
|
634
|
+
|
|
635
|
+
const discover = h.session.getToolDefinition("capability_discover");
|
|
636
|
+
const result = await discover!.execute("call-again", { name: "grep_app_search" }, undefined, undefined, {} as never);
|
|
637
|
+
expect(String(result.content[0]!.text)).toContain("already active");
|
|
638
|
+
expect(String(result.content[0]!.text)).not.toContain("Activated");
|
|
639
|
+
expect(result.details).toMatchObject({ alreadyActive: true });
|
|
640
|
+
});
|
|
641
|
+
|
|
607
642
|
it("does not attempt Jev when explicitly disabled", async () => {
|
|
608
643
|
const disabled = stubJev("grep_app_search");
|
|
609
644
|
const h = await createGatewaySession({
|
|
@@ -8,14 +8,21 @@ import type { CatalogEntry } from "./catalog.ts";
|
|
|
8
8
|
import {
|
|
9
9
|
candidateCriteria,
|
|
10
10
|
DEFAULT_GATEWAY_JEV_CONFIG,
|
|
11
|
+
DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS,
|
|
11
12
|
DEFAULT_GATEWAY_JEV_TIMEOUT_MS,
|
|
13
|
+
GATEWAY_JEV_DISCOVER_MAX_TIMEOUT_MS,
|
|
12
14
|
GATEWAY_JEV_MAX_TIMEOUT_MS,
|
|
13
15
|
GATEWAY_JEV_PROMPT_CHARS,
|
|
14
16
|
gatewayJevConnection,
|
|
15
17
|
hintedToolCandidates,
|
|
18
|
+
fitCatalogCandidates,
|
|
16
19
|
JEV_CAPABILITY_QUESTION,
|
|
20
|
+
JEV_SKILL_QUESTION,
|
|
21
|
+
JEV_TOOL_QUESTION,
|
|
22
|
+
MAX_CANDIDATE_CRITERION_CHARS,
|
|
17
23
|
MAX_GATEWAY_JEV_CANDIDATES,
|
|
18
24
|
readGatewayJevConfig,
|
|
25
|
+
routeJobToJev,
|
|
19
26
|
routeToJevTool,
|
|
20
27
|
type GatewayJevConfig,
|
|
21
28
|
type JevAsk,
|
|
@@ -122,8 +129,9 @@ describe("gateway Jev configuration", () => {
|
|
|
122
129
|
provider: "tokenin",
|
|
123
130
|
model: "jev-1.13",
|
|
124
131
|
timeoutMs: DEFAULT_GATEWAY_JEV_TIMEOUT_MS,
|
|
132
|
+
discoverTimeoutMs: DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS,
|
|
125
133
|
minConfidence: 0.6,
|
|
126
|
-
payloadBytes:
|
|
134
|
+
payloadBytes: 32 * 1024,
|
|
127
135
|
});
|
|
128
136
|
expect(DEFAULT_GATEWAY_JEV_TIMEOUT_MS).toBeLessThan(8_000);
|
|
129
137
|
|
|
@@ -153,6 +161,7 @@ describe("gateway Jev configuration", () => {
|
|
|
153
161
|
model: "jev-1.13",
|
|
154
162
|
baseUrl: "https://jev.example/v1",
|
|
155
163
|
timeoutMs: 500,
|
|
164
|
+
discoverTimeoutMs: DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS,
|
|
156
165
|
minConfidence: 0.8,
|
|
157
166
|
payloadBytes: 4096,
|
|
158
167
|
});
|
|
@@ -209,6 +218,17 @@ describe("gateway Jev configuration", () => {
|
|
|
209
218
|
});
|
|
210
219
|
expect(gatewayJevConnection({ ...config, timeoutMs: 60_000 }).timeoutMs).toBe(GATEWAY_JEV_MAX_TIMEOUT_MS);
|
|
211
220
|
});
|
|
221
|
+
|
|
222
|
+
it("gives the agent's on-demand route its own longer deadline, not the pre-turn cap", () => {
|
|
223
|
+
expect(gatewayJevConnection(config, true).timeoutMs).toBe(DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS);
|
|
224
|
+
expect(DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS).toBeGreaterThan(GATEWAY_JEV_MAX_TIMEOUT_MS);
|
|
225
|
+
expect(gatewayJevConnection({ ...config, discoverTimeoutMs: 8_000 }, true).timeoutMs).toBe(8_000);
|
|
226
|
+
expect(gatewayJevConnection({ ...config, discoverTimeoutMs: 60_000 }, true).timeoutMs).toBe(
|
|
227
|
+
GATEWAY_JEV_DISCOVER_MAX_TIMEOUT_MS,
|
|
228
|
+
);
|
|
229
|
+
// The pre-turn tie-break is untouched by the on-demand setting.
|
|
230
|
+
expect(gatewayJevConnection({ ...config, discoverTimeoutMs: 8_000 }).timeoutMs).toBe(DEFAULT_GATEWAY_JEV_TIMEOUT_MS);
|
|
231
|
+
});
|
|
212
232
|
});
|
|
213
233
|
|
|
214
234
|
describe("gateway Jev candidates", () => {
|
|
@@ -419,3 +439,136 @@ describe("routeToJevTool through the shared Jev client", () => {
|
|
|
419
439
|
expect(Object.keys(body.questions[JEV_CAPABILITY_QUESTION].criteria)).toEqual(["none", "grep_app_search"]);
|
|
420
440
|
});
|
|
421
441
|
});
|
|
442
|
+
|
|
443
|
+
describe("catalog fitting for the on-demand route", () => {
|
|
444
|
+
const entry = (name: string, summary = "Do a thing"): CatalogEntry => ({
|
|
445
|
+
name,
|
|
446
|
+
kind: "tool",
|
|
447
|
+
summary,
|
|
448
|
+
aliases: [],
|
|
449
|
+
eligible: true,
|
|
450
|
+
});
|
|
451
|
+
|
|
452
|
+
it("offers every eligible entry that fits and reports the ones it dropped", () => {
|
|
453
|
+
const entries = Array.from({ length: 20 }, (_, index) => entry(`tool_${index}`));
|
|
454
|
+
const { candidates, dropped } = fitCatalogCandidates(entries, 400);
|
|
455
|
+
expect(candidates.length).toBeGreaterThan(0);
|
|
456
|
+
expect(candidates.length + dropped).toBe(20);
|
|
457
|
+
expect(candidates.map((c) => c.name)).toEqual(entries.slice(0, candidates.length).map((c) => c.name));
|
|
458
|
+
});
|
|
459
|
+
|
|
460
|
+
it("never offers an ineligible entry", () => {
|
|
461
|
+
const { candidates, dropped } = fitCatalogCandidates([{ ...entry("hidden"), eligible: false }, entry("shown")], 10_000);
|
|
462
|
+
expect(candidates.map((c) => c.name)).toEqual(["shown"]);
|
|
463
|
+
expect(dropped).toBe(0);
|
|
464
|
+
});
|
|
465
|
+
|
|
466
|
+
it("clips a long criterion to its character ceiling", () => {
|
|
467
|
+
const { candidates } = fitCatalogCandidates([entry("verbose", "x".repeat(1_000))], 10_000);
|
|
468
|
+
const criterion = candidateCriteria(candidates)["verbose"]!;
|
|
469
|
+
expect(criterion).toHaveLength(MAX_CANDIDATE_CRITERION_CHARS + 1);
|
|
470
|
+
expect(criterion.endsWith("…")).toBe(true);
|
|
471
|
+
});
|
|
472
|
+
});
|
|
473
|
+
|
|
474
|
+
describe("routeJobToJev through an injected decision call", () => {
|
|
475
|
+
const ask = (decision: JevDecision): JevAsk => vi.fn(async () => decision) as unknown as JevAsk;
|
|
476
|
+
const sentPayload = (injected: JevAsk) => (injected as unknown as ReturnType<typeof vi.fn>).mock.calls[0]![2] as any;
|
|
477
|
+
|
|
478
|
+
it("asks the tool and skill questions separately in one request, without any schema", async () => {
|
|
479
|
+
const injected = ask({
|
|
480
|
+
choices: {
|
|
481
|
+
[JEV_TOOL_QUESTION]: { choice: "grep_app_search", confidence: 0.9 },
|
|
482
|
+
[JEV_SKILL_QUESTION]: { choice: "research", confidence: 0.8 },
|
|
483
|
+
},
|
|
484
|
+
rejected: {},
|
|
485
|
+
elapsedMs: 5,
|
|
486
|
+
});
|
|
487
|
+
const route = await routeJobToJev(
|
|
488
|
+
[searchTool, fetchTool, otherTool, fourthTool],
|
|
489
|
+
[skillEntry],
|
|
490
|
+
"research how others use this API",
|
|
491
|
+
ctxWith(),
|
|
492
|
+
config,
|
|
493
|
+
injected,
|
|
494
|
+
);
|
|
495
|
+
expect(route).toEqual({
|
|
496
|
+
tool: { pick: { selected: true, name: "grep_app_search", confidence: 0.9 }, offered: 4, dropped: 0 },
|
|
497
|
+
skill: { pick: { selected: true, name: "research", confidence: 0.8 }, offered: 1, dropped: 0 },
|
|
498
|
+
elapsedMs: 5,
|
|
499
|
+
});
|
|
500
|
+
|
|
501
|
+
const request = sentPayload(injected);
|
|
502
|
+
expect(Object.keys(request.payload.questions[JEV_TOOL_QUESTION].criteria)).toEqual([
|
|
503
|
+
"none",
|
|
504
|
+
"grep_app_search",
|
|
505
|
+
"grep_app_fetch",
|
|
506
|
+
"other_tool",
|
|
507
|
+
"fourth_tool",
|
|
508
|
+
]);
|
|
509
|
+
expect(Object.keys(request.payload.questions[JEV_SKILL_QUESTION].criteria)).toEqual(["none", "research"]);
|
|
510
|
+
expect(request.allowed[JEV_SKILL_QUESTION]).not.toContain("grep_app_search");
|
|
511
|
+
expect(request.payload.state.conversation).toEqual([{ role: "user", text: "research how others use this API" }]);
|
|
512
|
+
expect(JSON.stringify(request.payload)).not.toContain("parameters");
|
|
513
|
+
expect((injected as unknown as ReturnType<typeof vi.fn>).mock.calls[0]![1].timeoutMs).toBe(
|
|
514
|
+
DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS,
|
|
515
|
+
);
|
|
516
|
+
});
|
|
517
|
+
|
|
518
|
+
it("answers each kind on its own: `none` for one does not hide the other", async () => {
|
|
519
|
+
const injected = ask({
|
|
520
|
+
choices: {
|
|
521
|
+
[JEV_TOOL_QUESTION]: { choice: "none", confidence: 0.95 },
|
|
522
|
+
[JEV_SKILL_QUESTION]: { choice: "research", confidence: 0.7 },
|
|
523
|
+
},
|
|
524
|
+
rejected: {},
|
|
525
|
+
elapsedMs: 3,
|
|
526
|
+
});
|
|
527
|
+
const route = await routeJobToJev([searchTool], [skillEntry], "write up findings", ctxWith(), config, injected);
|
|
528
|
+
expect(route.tool.pick).toEqual({ selected: false, reason: "none" });
|
|
529
|
+
expect(route.skill.pick).toEqual({ selected: true, name: "research", confidence: 0.7 });
|
|
530
|
+
});
|
|
531
|
+
|
|
532
|
+
it("fits tools first, so a long skill tail can only drop skills", async () => {
|
|
533
|
+
const skills = Array.from({ length: 200 }, (_, index) => ({
|
|
534
|
+
...skillEntry,
|
|
535
|
+
name: `skill_${index}`,
|
|
536
|
+
summary: "A long procedure description ".repeat(5),
|
|
537
|
+
}));
|
|
538
|
+
const injected = ask({ choices: {}, rejected: {}, failure: "missing", elapsedMs: 1 });
|
|
539
|
+
const route = await routeJobToJev([searchTool, fetchTool], skills, "anything", ctxWith(), { ...config, payloadBytes: 8_192 }, injected);
|
|
540
|
+
expect(route.tool).toMatchObject({ offered: 2, dropped: 0 });
|
|
541
|
+
expect(route.skill.offered).toBeGreaterThan(0);
|
|
542
|
+
expect(route.skill.dropped).toBeGreaterThan(0);
|
|
543
|
+
expect(route.skill.offered + route.skill.dropped).toBe(200);
|
|
544
|
+
});
|
|
545
|
+
|
|
546
|
+
it("leaves an empty kind out of the request, and skips Jev when both are empty", async () => {
|
|
547
|
+
const injected = ask({ choices: { [JEV_TOOL_QUESTION]: { choice: "none", confidence: 0.9 } }, rejected: {}, elapsedMs: 2 });
|
|
548
|
+
const route = await routeJobToJev([searchTool], [], "search code", ctxWith(), config, injected);
|
|
549
|
+
expect(Object.keys(sentPayload(injected).payload.questions)).toEqual([JEV_TOOL_QUESTION]);
|
|
550
|
+
expect(route.skill.pick).toEqual({ selected: false, reason: "no-candidates" });
|
|
551
|
+
|
|
552
|
+
const unused = ask({ choices: {}, rejected: {}, elapsedMs: 1 });
|
|
553
|
+
expect(await routeJobToJev([], [], "anything", ctxWith(), config, unused)).toMatchObject({ elapsedMs: 0 });
|
|
554
|
+
expect(unused).not.toHaveBeenCalled();
|
|
555
|
+
});
|
|
556
|
+
|
|
557
|
+
it("reports an unavailable Jev on every asked question, and unquantified answers as abstentions", async () => {
|
|
558
|
+
const timeout = ask({
|
|
559
|
+
choices: {},
|
|
560
|
+
rejected: { [JEV_TOOL_QUESTION]: "timeout", [JEV_SKILL_QUESTION]: "timeout" },
|
|
561
|
+
failure: "timeout",
|
|
562
|
+
elapsedMs: 5_000,
|
|
563
|
+
});
|
|
564
|
+
const route = await routeJobToJev([searchTool], [skillEntry], "x", ctxWith(), config, timeout);
|
|
565
|
+
expect(route.tool.pick).toEqual({ selected: false, reason: "timeout" });
|
|
566
|
+
expect(route.skill.pick).toEqual({ selected: false, reason: "timeout" });
|
|
567
|
+
|
|
568
|
+
const unquantified = ask({ choices: { [JEV_TOOL_QUESTION]: { choice: "grep_app_search" } }, rejected: {}, elapsedMs: 3 });
|
|
569
|
+
expect((await routeJobToJev([searchTool], [], "find code", ctxWith(), config, unquantified)).tool.pick).toEqual({
|
|
570
|
+
selected: false,
|
|
571
|
+
reason: "unquantified",
|
|
572
|
+
});
|
|
573
|
+
});
|
|
574
|
+
});
|
|
@@ -1,23 +1,31 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Capability-gateway Jev-assisted tool
|
|
2
|
+
* Capability-gateway Jev-assisted tool routing.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* ambiguous lexical hint among optional tools, the gateway offers just those hinted tools to Jev
|
|
6
|
-
* as one bounded choice question. Jev may answer `none` or one hinted canonical tool name; the
|
|
7
|
-
* host revalidates eligibility against the live catalog before activating anything for this run.
|
|
4
|
+
* Two callers, one decision body:
|
|
8
5
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
6
|
+
* - The **pre-turn tie-break** (host-driven): when the deterministic router returns an ambiguous
|
|
7
|
+
* lexical hint among optional tools, the gateway offers just those hinted tools to Jev as one
|
|
8
|
+
* bounded choice question. Jev may answer `none` or one hinted canonical tool name; the host
|
|
9
|
+
* revalidates eligibility against the live catalog before activating anything for this run.
|
|
10
|
+
* - The **on-demand route** (agent-driven): the agent calls `capability_discover` with a job
|
|
11
|
+
* description instead of a name. Tools and skills are different decisions — a tool is code the
|
|
12
|
+
* agent may call many times, a skill is a procedure it reads once — so one request asks two
|
|
13
|
+
* choice questions, each over its own catalog slice with its own `none`. Tools are fitted into
|
|
14
|
+
* the budget first, so skills can never crowd a tool out.
|
|
15
|
+
*
|
|
16
|
+
* Either way Jev never sees a tool schema, prior conversation, or system prompt: the request carries
|
|
17
|
+
* a bounded material text and compact discovery lines (name, kind, category, one-line purpose). It
|
|
18
|
+
* is never consulted for a unique deterministic activation, a skill-only match, or a prompt with no
|
|
19
|
+
* lexical tool signal; every failure, timeout, low-confidence answer, or `none` is an abstention.
|
|
20
|
+
* The transport and validation half lives in the shared `../jev/decisions.ts` client.
|
|
14
21
|
*
|
|
15
22
|
* Configuration (enabled by default, but requires Token-In credentials). The gateway reads its
|
|
16
23
|
* own settings area — it does not inherit `jevAdvisory` route policy; only the Jev provider/model
|
|
17
24
|
* identity is shared so every Jev consumer defaults to the same deployment:
|
|
18
25
|
*
|
|
19
26
|
* "capabilityGateway": {
|
|
20
|
-
* "routing": { "jev": { "enabled": true, "timeoutMs": 1000, "
|
|
27
|
+
* "routing": { "jev": { "enabled": true, "timeoutMs": 1000, "discoverTimeoutMs": 5000,
|
|
28
|
+
* "minConfidence": 0.6, "payloadBytes": 32768 } }
|
|
21
29
|
* }
|
|
22
30
|
*/
|
|
23
31
|
import { readFileSync } from "node:fs";
|
|
@@ -55,6 +63,13 @@ export const GATEWAY_JEV_PROMPT_CHARS = 2_000;
|
|
|
55
63
|
export const DEFAULT_GATEWAY_JEV_TIMEOUT_MS = 1_000;
|
|
56
64
|
export const GATEWAY_JEV_MAX_TIMEOUT_MS = 2_000;
|
|
57
65
|
|
|
66
|
+
/**
|
|
67
|
+
* The agent's on-demand route asks over the whole catalog and the agent chose to wait for it, so it
|
|
68
|
+
* gets its own, longer deadline — capped like the `ask_jev` route rather than like a pre-turn hop.
|
|
69
|
+
*/
|
|
70
|
+
export const DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS = 5_000;
|
|
71
|
+
export const GATEWAY_JEV_DISCOVER_MAX_TIMEOUT_MS = 15_000;
|
|
72
|
+
|
|
58
73
|
/** One gateway route's Jev settings: the shared Jev endpoint plus this route's timing. */
|
|
59
74
|
export interface GatewayJevConfig {
|
|
60
75
|
enabled: boolean;
|
|
@@ -62,6 +77,8 @@ export interface GatewayJevConfig {
|
|
|
62
77
|
model: string;
|
|
63
78
|
baseUrl?: string;
|
|
64
79
|
timeoutMs: number;
|
|
80
|
+
/** Deadline for the agent's `capability_discover({ job })` decision. */
|
|
81
|
+
discoverTimeoutMs: number;
|
|
65
82
|
minConfidence: number;
|
|
66
83
|
/** Hard cap on the serialized decision request; oversized requests abstain. */
|
|
67
84
|
payloadBytes: number;
|
|
@@ -73,8 +90,10 @@ export const DEFAULT_GATEWAY_JEV_CONFIG: GatewayJevConfig = {
|
|
|
73
90
|
provider: DEFAULT_JEV_ADVISORY_CONFIG.provider,
|
|
74
91
|
model: DEFAULT_JEV_ADVISORY_CONFIG.model,
|
|
75
92
|
timeoutMs: DEFAULT_GATEWAY_JEV_TIMEOUT_MS,
|
|
93
|
+
discoverTimeoutMs: DEFAULT_GATEWAY_JEV_DISCOVER_TIMEOUT_MS,
|
|
76
94
|
minConfidence: 0.6,
|
|
77
|
-
|
|
95
|
+
// Room for a whole catalog on the on-demand route; the tie-break sends a few hundred bytes.
|
|
96
|
+
payloadBytes: 32 * 1024,
|
|
78
97
|
};
|
|
79
98
|
|
|
80
99
|
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
@@ -110,18 +129,21 @@ export function readGatewayJevConfig(settingsPath: string = getSettingsPath()):
|
|
|
110
129
|
model: stringOr(raw.model, DEFAULT_GATEWAY_JEV_CONFIG.model),
|
|
111
130
|
baseUrl: typeof baseUrl === "string" && baseUrl.trim() !== "" ? baseUrl : undefined,
|
|
112
131
|
timeoutMs: numberOr(raw.timeoutMs, DEFAULT_GATEWAY_JEV_CONFIG.timeoutMs),
|
|
132
|
+
discoverTimeoutMs: numberOr(raw.discoverTimeoutMs, DEFAULT_GATEWAY_JEV_CONFIG.discoverTimeoutMs),
|
|
113
133
|
minConfidence: numberOr(raw.minConfidence, DEFAULT_GATEWAY_JEV_CONFIG.minConfidence),
|
|
114
134
|
payloadBytes: numberOr(raw.payloadBytes, DEFAULT_GATEWAY_JEV_CONFIG.payloadBytes),
|
|
115
135
|
};
|
|
116
136
|
}
|
|
117
137
|
|
|
118
|
-
/** The transport settings
|
|
119
|
-
export function gatewayJevConnection(config: GatewayJevConfig): JevConnection {
|
|
138
|
+
/** The transport settings one gateway route uses, each with its own hard-capped deadline. */
|
|
139
|
+
export function gatewayJevConnection(config: GatewayJevConfig, onDemand = false): JevConnection {
|
|
120
140
|
return {
|
|
121
141
|
provider: config.provider,
|
|
122
142
|
model: config.model,
|
|
123
143
|
baseUrl: config.baseUrl,
|
|
124
|
-
timeoutMs:
|
|
144
|
+
timeoutMs: onDemand
|
|
145
|
+
? Math.min(config.discoverTimeoutMs, GATEWAY_JEV_DISCOVER_MAX_TIMEOUT_MS)
|
|
146
|
+
: Math.min(config.timeoutMs, GATEWAY_JEV_MAX_TIMEOUT_MS),
|
|
125
147
|
minConfidence: config.minConfidence,
|
|
126
148
|
};
|
|
127
149
|
}
|
|
@@ -135,19 +157,56 @@ export function hintedToolCandidates(hints: readonly CatalogEntry[] = []): Catal
|
|
|
135
157
|
return hints.filter((entry) => entry.kind === "tool" && entry.eligible).slice(0, MAX_GATEWAY_JEV_CANDIDATES);
|
|
136
158
|
}
|
|
137
159
|
|
|
160
|
+
/** Longest candidate summary sent as a criterion; a whole catalog has to fit one request. */
|
|
161
|
+
export const MAX_CANDIDATE_CRITERION_CHARS = 160;
|
|
162
|
+
|
|
163
|
+
/** JSON scaffolding one candidate costs in the decision request: key, quoting, separator. */
|
|
164
|
+
const CANDIDATE_ENVELOPE_BYTES = 32;
|
|
165
|
+
|
|
166
|
+
function clipText(text: string, maxChars: number): string {
|
|
167
|
+
return text.length > maxChars ? `${text.slice(0, maxChars)}…` : text;
|
|
168
|
+
}
|
|
169
|
+
|
|
138
170
|
/** Allowlisted choices: `none` plus one compact discovery-metadata line per candidate. */
|
|
139
171
|
export function candidateCriteria(candidates: CatalogEntry[]): Record<string, string> {
|
|
140
172
|
const criteria: Record<string, string> = {
|
|
141
|
-
[NO_TOOL]: "No optional
|
|
173
|
+
[NO_TOOL]: "No optional capability is needed for the job.",
|
|
142
174
|
};
|
|
143
175
|
for (const candidate of candidates) {
|
|
144
176
|
const category = candidate.category ? ` Category: ${candidate.category}.` : "";
|
|
145
177
|
const aliases = candidate.aliases.length > 0 ? ` Aliases: ${candidate.aliases.join(", ")}.` : "";
|
|
146
|
-
criteria[candidate.name] = `${candidate.summary}${category}${aliases}
|
|
178
|
+
criteria[candidate.name] = clipText(`${candidate.summary}${category}${aliases}`, MAX_CANDIDATE_CRITERION_CHARS);
|
|
147
179
|
}
|
|
148
180
|
return criteria;
|
|
149
181
|
}
|
|
150
182
|
|
|
183
|
+
/**
|
|
184
|
+
* The offered entries that fit one bounded decision request, in catalog order.
|
|
185
|
+
*
|
|
186
|
+
* An entry that does not fit is skipped rather than truncated: a candidate offered without a usable
|
|
187
|
+
* criterion would be judged on its name alone. `dropped` is reported so the caller can tell the
|
|
188
|
+
* agent the search was not exhaustive.
|
|
189
|
+
*
|
|
190
|
+
* ponytail: greedy order-independent fit; rank by relevance first if a real install drops enough
|
|
191
|
+
* candidates to misroute.
|
|
192
|
+
*/
|
|
193
|
+
export function fitCatalogCandidates(
|
|
194
|
+
entries: readonly CatalogEntry[],
|
|
195
|
+
budgetBytes: number,
|
|
196
|
+
): { candidates: CatalogEntry[]; dropped: number } {
|
|
197
|
+
const offered = entries.filter((entry) => entry.eligible);
|
|
198
|
+
const candidates: CatalogEntry[] = [];
|
|
199
|
+
let used = 0;
|
|
200
|
+
for (const entry of offered) {
|
|
201
|
+
const criterion = candidateCriteria([entry])[entry.name] ?? "";
|
|
202
|
+
const cost = CANDIDATE_ENVELOPE_BYTES + Buffer.byteLength(`${entry.name}${criterion}`, "utf-8");
|
|
203
|
+
if (used + cost > budgetBytes) continue;
|
|
204
|
+
candidates.push(entry);
|
|
205
|
+
used += cost;
|
|
206
|
+
}
|
|
207
|
+
return { candidates, dropped: offered.length - candidates.length };
|
|
208
|
+
}
|
|
209
|
+
|
|
151
210
|
/** Why no tool was routed: a clean `none`, an empty candidate set, or an absent Jev decision. */
|
|
152
211
|
export type JevToolAbstainReason = "none" | "no-candidates" | "unquantified" | JevClientAbstainReason;
|
|
153
212
|
|
|
@@ -166,30 +225,66 @@ export const JEV_UNAVAILABLE_REASONS: ReadonlySet<JevToolAbstainReason> = new Se
|
|
|
166
225
|
/** The decision call, injectable so catalog routing is testable without a live transport. */
|
|
167
226
|
export type JevAsk = typeof askJev;
|
|
168
227
|
|
|
228
|
+
/** One route's question wording; the criteria and allowlist are built from the candidates. */
|
|
229
|
+
export interface CapabilityQuestion {
|
|
230
|
+
question: string;
|
|
231
|
+
focus: string;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
/** The host's pre-turn tie-break: only the hinted tools, and only a weak lexical signal. */
|
|
235
|
+
export const TIE_BREAK_QUESTION: CapabilityQuestion = {
|
|
236
|
+
question:
|
|
237
|
+
"Which single hinted optional tool should be activated for the latest request in `conversation`, " +
|
|
238
|
+
"or `none` when none of them is clearly needed?",
|
|
239
|
+
focus: "Choose `none` unless exactly one listed tool is clearly needed; the prompt only weakly suggests these tools.",
|
|
240
|
+
};
|
|
241
|
+
|
|
242
|
+
/** Question keys of the on-demand route: one decision per capability kind. */
|
|
243
|
+
export const JEV_TOOL_QUESTION = "tool";
|
|
244
|
+
export const JEV_SKILL_QUESTION = "skill";
|
|
245
|
+
|
|
246
|
+
/** On demand, tools: callable code the agent may use many times during the job. */
|
|
247
|
+
export const JOB_TOOL_QUESTION: CapabilityQuestion = {
|
|
248
|
+
question:
|
|
249
|
+
"Which single listed tool should the agent activate for the job in `conversation`, or `none` when the " +
|
|
250
|
+
"built-in tools (read, bash, edit, write, grep, find, ls) or a direct answer are enough?",
|
|
251
|
+
focus:
|
|
252
|
+
"A tool is callable code the agent may call many times during the job. Most jobs need no optional tool: " +
|
|
253
|
+
"choose one name only when it clearly fits the job.",
|
|
254
|
+
};
|
|
255
|
+
|
|
256
|
+
/** On demand, skills: a written procedure the agent reads once before doing the job. */
|
|
257
|
+
export const JOB_SKILL_QUESTION: CapabilityQuestion = {
|
|
258
|
+
question:
|
|
259
|
+
"Which single listed skill's instructions should the agent load for the job in `conversation`, or `none` " +
|
|
260
|
+
"when no written procedure is needed?",
|
|
261
|
+
focus:
|
|
262
|
+
"A skill is a written procedure the agent reads once before doing the job. Choose one name only when the " +
|
|
263
|
+
"job clearly matches what that skill describes.",
|
|
264
|
+
};
|
|
265
|
+
|
|
169
266
|
/**
|
|
170
|
-
* Offer
|
|
267
|
+
* Offer one candidate set to Jev and return the capability it selected. Anything outside one
|
|
171
268
|
* confident, canonical, allowlisted choice is an abstention.
|
|
172
269
|
*/
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
270
|
+
async function offerCandidates(
|
|
271
|
+
candidates: readonly CatalogEntry[],
|
|
272
|
+
material: string,
|
|
273
|
+
prompt: CapabilityQuestion,
|
|
176
274
|
ctx: Pick<ExtensionContext, "modelRegistry">,
|
|
177
275
|
config: GatewayJevConfig,
|
|
178
|
-
ask: JevAsk
|
|
276
|
+
ask: JevAsk,
|
|
179
277
|
): Promise<JevToolRoute> {
|
|
180
|
-
const candidates = hintedToolCandidates(hints);
|
|
181
278
|
if (candidates.length === 0) return { selected: false, reason: "no-candidates", candidates: 0, elapsedMs: 0 };
|
|
182
279
|
|
|
183
280
|
const question: JevQuestion = {
|
|
184
|
-
question:
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
focus: "Choose `none` unless exactly one listed tool is clearly needed; the prompt only weakly suggests these tools.",
|
|
188
|
-
criteria: candidateCriteria(candidates),
|
|
281
|
+
question: prompt.question,
|
|
282
|
+
focus: prompt.focus,
|
|
283
|
+
criteria: candidateCriteria([...candidates]),
|
|
189
284
|
};
|
|
190
|
-
// Only the current
|
|
285
|
+
// Only the current material, bounded: no prior conversation, no tool schema, no history.
|
|
191
286
|
const payload = buildJevPayload(
|
|
192
|
-
buildConversation(
|
|
287
|
+
buildConversation(material, [], { contextTurns: 1, contextChars: GATEWAY_JEV_PROMPT_CHARS }),
|
|
193
288
|
{ [JEV_CAPABILITY_QUESTION]: question },
|
|
194
289
|
);
|
|
195
290
|
const decision = await ask(ctx, gatewayJevConnection(config), {
|
|
@@ -198,24 +293,111 @@ export async function routeToJevTool(
|
|
|
198
293
|
allowed: { [JEV_CAPABILITY_QUESTION]: [NO_TOOL, ...candidates.map((candidate) => candidate.name)] },
|
|
199
294
|
});
|
|
200
295
|
|
|
201
|
-
const
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
candidates: candidates.length,
|
|
205
|
-
|
|
296
|
+
const picked = pickFrom(decision, JEV_CAPABILITY_QUESTION);
|
|
297
|
+
return picked.selected
|
|
298
|
+
? { selected: true, tool: picked.name, confidence: picked.confidence, candidates: candidates.length, elapsedMs: decision.elapsedMs }
|
|
299
|
+
: { selected: false, reason: picked.reason, candidates: candidates.length, elapsedMs: decision.elapsedMs };
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/** One question's outcome: a confident allowlisted name, or why there is none. */
|
|
303
|
+
export type JevPick =
|
|
304
|
+
| { selected: true; name: string; confidence: number }
|
|
305
|
+
| { selected: false; reason: JevToolAbstainReason };
|
|
306
|
+
|
|
307
|
+
/** Read one question out of a decision. Anything outside one confident, canonical, non-`none` choice abstains. */
|
|
308
|
+
function pickFrom(decision: JevDecision, question: string): JevPick {
|
|
309
|
+
const answer = decision.choices[question];
|
|
310
|
+
if (!answer) return { selected: false, reason: decision.rejected[question] ?? decision.failure ?? "missing" };
|
|
311
|
+
if (answer.choice === NO_TOOL) return { selected: false, reason: "none" };
|
|
312
|
+
// A choice without a numeric confidence is not a confident enough answer to act on.
|
|
313
|
+
if (answer.confidence === undefined) return { selected: false, reason: "unquantified" };
|
|
314
|
+
return { selected: true, name: answer.choice, confidence: answer.confidence };
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
/**
|
|
318
|
+
* Offer the hinted tools to Jev and return the tool it selected for this run.
|
|
319
|
+
*/
|
|
320
|
+
export async function routeToJevTool(
|
|
321
|
+
hints: readonly CatalogEntry[],
|
|
322
|
+
prompt: string,
|
|
323
|
+
ctx: Pick<ExtensionContext, "modelRegistry">,
|
|
324
|
+
config: GatewayJevConfig,
|
|
325
|
+
ask: JevAsk = askJev,
|
|
326
|
+
): Promise<JevToolRoute> {
|
|
327
|
+
return offerCandidates(hintedToolCandidates(hints), prompt, TIE_BREAK_QUESTION, ctx, config, ask);
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
/** Room kept per question for its wording and scaffolding around the candidate criteria. */
|
|
331
|
+
const QUESTION_SCAFFOLDING_BYTES = 512;
|
|
332
|
+
|
|
333
|
+
/** One capability kind's side of an on-demand decision. */
|
|
334
|
+
export interface JevJobPick {
|
|
335
|
+
pick: JevPick;
|
|
336
|
+
/** Candidates Jev was shown. */
|
|
337
|
+
offered: number;
|
|
338
|
+
/** Eligible candidates left out to fit the request budget. */
|
|
339
|
+
dropped: number;
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
/** The on-demand decision: a tool and a skill, each chosen or not on its own. */
|
|
343
|
+
export interface JevJobRoute {
|
|
344
|
+
tool: JevJobPick;
|
|
345
|
+
skill: JevJobPick;
|
|
346
|
+
elapsedMs: number;
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
/**
|
|
350
|
+
* The agent's on-demand route: ask which tool and which skill fit the job, as two choice questions
|
|
351
|
+
* in one request. Pass an empty list to leave a kind out; it is then not asked at all.
|
|
352
|
+
*
|
|
353
|
+
* The budget goes to tools first. Skills are the long tail users, agents, and teammates keep adding,
|
|
354
|
+
* so if anything is dropped it is a skill, and the caller reports it.
|
|
355
|
+
*/
|
|
356
|
+
export async function routeJobToJev(
|
|
357
|
+
tools: readonly CatalogEntry[],
|
|
358
|
+
skills: readonly CatalogEntry[],
|
|
359
|
+
job: string,
|
|
360
|
+
ctx: Pick<ExtensionContext, "modelRegistry">,
|
|
361
|
+
config: GatewayJevConfig,
|
|
362
|
+
ask: JevAsk = askJev,
|
|
363
|
+
): Promise<JevJobRoute> {
|
|
364
|
+
const maxBytes = Math.min(config.payloadBytes, JEV_REQUEST_MAX_BYTES);
|
|
365
|
+
const material = job.slice(0, GATEWAY_JEV_PROMPT_CHARS);
|
|
366
|
+
let budget = maxBytes - Buffer.byteLength(material, "utf-8") - 2 * QUESTION_SCAFFOLDING_BYTES;
|
|
367
|
+
const fittedTools = fitCatalogCandidates(tools, Math.max(0, budget));
|
|
368
|
+
budget -= Buffer.byteLength(JSON.stringify(candidateCriteria(fittedTools.candidates)), "utf-8");
|
|
369
|
+
const fittedSkills = fitCatalogCandidates(skills, Math.max(0, budget));
|
|
370
|
+
|
|
371
|
+
const side = (fitted: { candidates: CatalogEntry[]; dropped: number }, pick: JevPick): JevJobPick => ({
|
|
372
|
+
pick,
|
|
373
|
+
offered: fitted.candidates.length,
|
|
374
|
+
dropped: fitted.dropped,
|
|
206
375
|
});
|
|
207
|
-
const
|
|
208
|
-
|
|
209
|
-
|
|
376
|
+
const notAsked: JevPick = { selected: false, reason: "no-candidates" };
|
|
377
|
+
|
|
378
|
+
const questions: Record<string, JevQuestion> = {};
|
|
379
|
+
const allowed: Record<string, string[]> = {};
|
|
380
|
+
for (const [key, fitted, wording] of [
|
|
381
|
+
[JEV_TOOL_QUESTION, fittedTools, JOB_TOOL_QUESTION],
|
|
382
|
+
[JEV_SKILL_QUESTION, fittedSkills, JOB_SKILL_QUESTION],
|
|
383
|
+
] as const) {
|
|
384
|
+
if (fitted.candidates.length === 0) continue;
|
|
385
|
+
questions[key] = { ...wording, criteria: candidateCriteria(fitted.candidates) };
|
|
386
|
+
allowed[key] = [NO_TOOL, ...fitted.candidates.map((candidate) => candidate.name)];
|
|
387
|
+
}
|
|
388
|
+
if (Object.keys(questions).length === 0) {
|
|
389
|
+
return { tool: side(fittedTools, notAsked), skill: side(fittedSkills, notAsked), elapsedMs: 0 };
|
|
210
390
|
}
|
|
211
|
-
|
|
212
|
-
//
|
|
213
|
-
|
|
391
|
+
|
|
392
|
+
// Only the job, bounded: no prior conversation, no tool schema, no skill body.
|
|
393
|
+
const payload = buildJevPayload(
|
|
394
|
+
buildConversation(material, [], { contextTurns: 1, contextChars: GATEWAY_JEV_PROMPT_CHARS }),
|
|
395
|
+
questions,
|
|
396
|
+
);
|
|
397
|
+
const decision = await ask(ctx, gatewayJevConnection(config, true), { payload, maxBytes, allowed });
|
|
214
398
|
return {
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
confidence: answer.confidence,
|
|
218
|
-
candidates: candidates.length,
|
|
399
|
+
tool: side(fittedTools, allowed[JEV_TOOL_QUESTION] ? pickFrom(decision, JEV_TOOL_QUESTION) : notAsked),
|
|
400
|
+
skill: side(fittedSkills, allowed[JEV_SKILL_QUESTION] ? pickFrom(decision, JEV_SKILL_QUESTION) : notAsked),
|
|
219
401
|
elapsedMs: decision.elapsedMs,
|
|
220
402
|
};
|
|
221
403
|
}
|
|
@@ -25,9 +25,11 @@ import {
|
|
|
25
25
|
JEV_ROUTING_EVENT,
|
|
26
26
|
jevConnection,
|
|
27
27
|
jevModel,
|
|
28
|
+
jevUnavailable,
|
|
28
29
|
readJevAdvisoryConfig,
|
|
29
30
|
readJevChoice,
|
|
30
31
|
serializeJevRequest,
|
|
32
|
+
warnJevUnavailableOnce,
|
|
31
33
|
type JevConnection,
|
|
32
34
|
} from "./decisions.ts";
|
|
33
35
|
import { jevAnswers, jevResponse, providerTemplate } from "./test-support.ts";
|
|
@@ -81,6 +83,19 @@ describe("readJevAdvisoryConfig", () => {
|
|
|
81
83
|
expect(readJevAdvisoryConfig(state.settingsPath).routes.memory.enabled).toBe(false);
|
|
82
84
|
});
|
|
83
85
|
|
|
86
|
+
it("keeps the agent's ask route on unless it is explicitly turned off", () => {
|
|
87
|
+
state.settingsPath = join(tmpdir(), "missing-settings.json");
|
|
88
|
+
expect(readJevAdvisoryConfig(state.settingsPath).routes.ask.enabled).toBe(true);
|
|
89
|
+
|
|
90
|
+
write({ jevAdvisory: { routes: { ask: { timeoutMs: 2_000 }, memory: { timeoutMs: 2_000 } } } });
|
|
91
|
+
const config = readJevAdvisoryConfig(state.settingsPath);
|
|
92
|
+
expect(config.routes.ask).toMatchObject({ enabled: true, timeoutMs: 2_000, payloadBytes: 32 * 1024 });
|
|
93
|
+
expect(config.routes.memory.enabled).toBe(false);
|
|
94
|
+
|
|
95
|
+
write({ jevAdvisory: { routes: { ask: { enabled: false } } } });
|
|
96
|
+
expect(readJevAdvisoryConfig(state.settingsPath).routes.ask.enabled).toBe(false);
|
|
97
|
+
});
|
|
98
|
+
|
|
84
99
|
it("merges per-route settings over the shared endpoint defaults", () => {
|
|
85
100
|
write({
|
|
86
101
|
jevAdvisory: {
|
|
@@ -314,3 +329,25 @@ describe("telemetry helpers", () => {
|
|
|
314
329
|
expect(seen).toEqual([{ channel: JEV_ROUTING_EVENT, data: { event: "decision", route: "memory" } }]);
|
|
315
330
|
});
|
|
316
331
|
});
|
|
332
|
+
|
|
333
|
+
describe("reachability before any work", () => {
|
|
334
|
+
it("names why Jev is out of reach without sending anything", async () => {
|
|
335
|
+
expect(await jevUnavailable(runtime([]) as never, CONNECTION)).toBe("no-template");
|
|
336
|
+
expect(await jevUnavailable(runtime([providerTemplate()], false) as never, CONNECTION)).toBe("no-credential");
|
|
337
|
+
expect(await jevUnavailable(runtime() as never, CONNECTION)).toBeUndefined();
|
|
338
|
+
expect(completeMock).not.toHaveBeenCalled();
|
|
339
|
+
});
|
|
340
|
+
|
|
341
|
+
it("warns each session once, whichever feature notices first", () => {
|
|
342
|
+
const notices: string[] = [];
|
|
343
|
+
const ui = { notify: (message: string) => notices.push(message) };
|
|
344
|
+
warnJevUnavailableOnce(ui, "tokenin");
|
|
345
|
+
warnJevUnavailableOnce(ui, "tokenin");
|
|
346
|
+
expect(notices).toHaveLength(1);
|
|
347
|
+
expect(notices[0]).toContain("/tokenin add");
|
|
348
|
+
|
|
349
|
+
const other: string[] = [];
|
|
350
|
+
warnJevUnavailableOnce({ notify: (message: string) => other.push(message) }, "custom");
|
|
351
|
+
expect(other[0]).toContain('"custom" provider');
|
|
352
|
+
});
|
|
353
|
+
});
|