@selesai/code 0.14.0 → 0.14.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.d.ts +10 -1
- package/dist/config.js +39 -1
- package/dist/core/resource-loader.js +2 -0
- package/dist/extensions/codemode/execute.js +18 -6
- package/dist/extensions/jev/classifier.test.ts +85 -20
- package/dist/extensions/jev/classifier.ts +147 -61
- package/dist/extensions/jev/decisions.test.ts +21 -0
- package/dist/extensions/jev/decisions.ts +36 -5
- package/dist/extensions/jev-ask-tool.ts +1 -1
- package/dist/extensions/jev-tool-ranker.test.ts +185 -0
- package/dist/extensions/jev-tool-ranker.ts +169 -0
- package/dist/extensions/package.json +1 -0
- package/dist/extensions/pi-graft/index.ts +4 -19
- package/dist/extensions/pi-graft/lifecycle.test.ts +32 -25
- package/dist/extensions/pi-graft/tools.ts +2 -2
- package/dist/extensions/pi-hermes-memory/src/tools/skill-tool.ts +1 -1
- package/dist/extensions/pi-subagents/src/extension/index.ts +2 -2
- package/dist/extensions/pi-subagents/test/unit/index-child-registration.test.ts +1 -1
- package/dist/extensions/question/index.ts +1 -1
- package/dist/extensions/tokenin-onboarding.ts +3 -3
- package/dist/extensions/tool-search/tool.d.ts +23 -4
- package/dist/extensions/tool-search/tool.js +35 -9
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/docs/codemode.md +1 -1
- package/package.json +3 -3
|
@@ -7,12 +7,16 @@ import { fitsClassifierRequest, toClassifierContext } from "./classifier-context
|
|
|
7
7
|
// Advisory configuration (one opt-in area, per-route enablement)
|
|
8
8
|
// ---------------------------------------------------------------------------
|
|
9
9
|
|
|
10
|
-
export const JEV_ROUTE_NAMES = ["memory", "recommendations", "ask", "subagent"] as const;
|
|
10
|
+
export const JEV_ROUTE_NAMES = ["memory", "recommendations", "ask", "subagent", "toolSearch"] as const;
|
|
11
11
|
export type JevRouteName = (typeof JEV_ROUTE_NAMES)[number];
|
|
12
12
|
|
|
13
13
|
export interface JevRouteConfig {
|
|
14
14
|
/** Host-side routes are off unless the user turns them on; the agent's `ask` route is on. */
|
|
15
15
|
enabled: boolean;
|
|
16
|
+
/** Classifier model this route asks instead of the shared `jevAdvisory.model`. */
|
|
17
|
+
model?: string;
|
|
18
|
+
/** Provider of that model; defaults to the shared `jevAdvisory.provider`. */
|
|
19
|
+
provider?: string;
|
|
16
20
|
timeoutMs: number;
|
|
17
21
|
/** Below this confidence the answer is an abstention, not a decision. */
|
|
18
22
|
minConfidence: number;
|
|
@@ -73,6 +77,23 @@ export const DEFAULT_JEV_SUBAGENT_ROUTE_CONFIG: JevRouteConfig = {
|
|
|
73
77
|
payloadBytes: 16 * 1024,
|
|
74
78
|
};
|
|
75
79
|
|
|
80
|
+
/**
|
|
81
|
+
* The `toolSearch` route ranks tools for `tool_search` and codemode's `searchTools()`: one yes/no
|
|
82
|
+
* question per candidate tool, batched. It asks the shared Jev model unless the route sets its own
|
|
83
|
+
* `model` (any Token-In decisions model, e.g. `cloudflare/clef-flash`). On by default like `ask`: it
|
|
84
|
+
* runs only when the agent searches for tools, and a missing Token-In credential falls back to the
|
|
85
|
+
* local BM25 ranking.
|
|
86
|
+
*
|
|
87
|
+
* `payloadBytes` bounds one request; `timeoutMs` is per request (batches run in parallel).
|
|
88
|
+
* `minConfidence`, `contextTurns`, and `contextChars` are unused.
|
|
89
|
+
*/
|
|
90
|
+
export const DEFAULT_JEV_TOOL_SEARCH_ROUTE_CONFIG: JevRouteConfig = {
|
|
91
|
+
...DEFAULT_JEV_ROUTE_CONFIG,
|
|
92
|
+
enabled: true,
|
|
93
|
+
timeoutMs: 6_000,
|
|
94
|
+
payloadBytes: 32 * 1024,
|
|
95
|
+
};
|
|
96
|
+
|
|
76
97
|
export const DEFAULT_JEV_ADVISORY_CONFIG: JevAdvisoryConfig = {
|
|
77
98
|
provider: "tokenin",
|
|
78
99
|
model: "jev-1.13",
|
|
@@ -81,6 +102,7 @@ export const DEFAULT_JEV_ADVISORY_CONFIG: JevAdvisoryConfig = {
|
|
|
81
102
|
recommendations: { ...DEFAULT_JEV_ROUTE_CONFIG },
|
|
82
103
|
ask: { ...DEFAULT_JEV_ASK_ROUTE_CONFIG },
|
|
83
104
|
subagent: { ...DEFAULT_JEV_SUBAGENT_ROUTE_CONFIG },
|
|
105
|
+
toolSearch: { ...DEFAULT_JEV_TOOL_SEARCH_ROUTE_CONFIG },
|
|
84
106
|
},
|
|
85
107
|
};
|
|
86
108
|
|
|
@@ -112,7 +134,11 @@ function numberOr(value: unknown, fallback: number): number {
|
|
|
112
134
|
|
|
113
135
|
function routeOr(value: unknown, fallback: JevRouteConfig = DEFAULT_JEV_ROUTE_CONFIG): JevRouteConfig {
|
|
114
136
|
if (!isRecord(value)) return { ...fallback };
|
|
137
|
+
const model = stringOr(value.model, fallback.model ?? "");
|
|
138
|
+
const provider = stringOr(value.provider, fallback.provider ?? "");
|
|
115
139
|
return {
|
|
140
|
+
...(model ? { model } : {}),
|
|
141
|
+
...(provider ? { provider } : {}),
|
|
116
142
|
enabled: value.enabled === undefined ? fallback.enabled : value.enabled === true,
|
|
117
143
|
timeoutMs: numberOr(value.timeoutMs, fallback.timeoutMs),
|
|
118
144
|
minConfidence: numberOr(value.minConfidence, fallback.minConfidence),
|
|
@@ -145,19 +171,24 @@ export function readJevAdvisoryConfig(settingsPath: string): JevAdvisoryConfig {
|
|
|
145
171
|
recommendations: routeOr(routes.recommendations),
|
|
146
172
|
ask: routeOr(routes.ask, DEFAULT_JEV_ASK_ROUTE_CONFIG),
|
|
147
173
|
subagent: routeOr(routes.subagent, DEFAULT_JEV_SUBAGENT_ROUTE_CONFIG),
|
|
174
|
+
toolSearch: routeOr(routes.toolSearch, DEFAULT_JEV_TOOL_SEARCH_ROUTE_CONFIG),
|
|
148
175
|
},
|
|
149
176
|
};
|
|
150
177
|
}
|
|
151
178
|
|
|
152
|
-
/**
|
|
179
|
+
/**
|
|
180
|
+
* The transport settings one route uses: the shared endpoint plus route timing. A route that names
|
|
181
|
+
* its own model keeps that model's registered endpoint: `jevAdvisory.baseUrl` overrides the shared
|
|
182
|
+
* Jev deployment's URL, not another model's.
|
|
183
|
+
*/
|
|
153
184
|
export function jevConnection(
|
|
154
185
|
config: JevAdvisoryConfig,
|
|
155
186
|
route: JevRouteConfig,
|
|
156
187
|
): JevConnection {
|
|
157
188
|
return {
|
|
158
|
-
provider: config.provider,
|
|
159
|
-
model: config.model,
|
|
160
|
-
baseUrl: config.baseUrl,
|
|
189
|
+
provider: route.provider ?? config.provider,
|
|
190
|
+
model: route.model ?? config.model,
|
|
191
|
+
baseUrl: route.model === undefined ? config.baseUrl : undefined,
|
|
161
192
|
timeoutMs: route.timeoutMs,
|
|
162
193
|
minConfidence: route.minConfidence,
|
|
163
194
|
};
|
|
@@ -673,7 +673,7 @@ interface JudgeOutcome {
|
|
|
673
673
|
}
|
|
674
674
|
|
|
675
675
|
/** Ask one noul per item, batched and fitted, within the stage's request caps. */
|
|
676
|
-
async function judgeNouls(
|
|
676
|
+
export async function judgeNouls(
|
|
677
677
|
requests: FindRequests,
|
|
678
678
|
stage: JudgeStage,
|
|
679
679
|
items: readonly JudgeItem[],
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Jev tool ranker: what tool_search and searchTools() get back, what Jev is sent, and that
|
|
3
|
+
* discovery keeps working (BM25) whenever Jev cannot answer.
|
|
4
|
+
*/
|
|
5
|
+
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
|
6
|
+
import { tmpdir } from "node:os";
|
|
7
|
+
import { join } from "node:path";
|
|
8
|
+
import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from "vitest";
|
|
9
|
+
|
|
10
|
+
const state = vi.hoisted(() => ({ settingsPath: "" }));
|
|
11
|
+
|
|
12
|
+
vi.mock("@selesai/code", async () => ({
|
|
13
|
+
...(await import("./tool-search/tool.ts")),
|
|
14
|
+
getSettingsPath: () => state.settingsPath,
|
|
15
|
+
}));
|
|
16
|
+
|
|
17
|
+
import { TOKENIN_JEV_CLASSIFIER } from "./jev/classifier.ts";
|
|
18
|
+
import { JEV_ROUTING_EVENT } from "./jev/decisions.ts";
|
|
19
|
+
import { classifierResponse, providerTemplate } from "./jev/test-support.ts";
|
|
20
|
+
import { createJevToolRanker, TOOL_RANK_MAX_REQUESTS } from "./jev-tool-ranker.ts";
|
|
21
|
+
import { FIND_BATCH } from "./jev-ask-tool.ts";
|
|
22
|
+
import { Bm25Ranker, createToolSearchDocument, type ToolSearchDocument } from "./tool-search/tool.ts";
|
|
23
|
+
|
|
24
|
+
let root: string;
|
|
25
|
+
beforeAll(() => {
|
|
26
|
+
root = mkdtempSync(join(tmpdir(), "jev-tool-ranker-"));
|
|
27
|
+
mkdirSync(join(root, "agent"), { recursive: true });
|
|
28
|
+
state.settingsPath = join(root, "agent", "settings.json");
|
|
29
|
+
});
|
|
30
|
+
afterAll(() => rmSync(root, { recursive: true, force: true }));
|
|
31
|
+
|
|
32
|
+
const tool = (name: string, description: string): ToolSearchDocument =>
|
|
33
|
+
createToolSearchDocument({ name, description, parameters: {} as never });
|
|
34
|
+
|
|
35
|
+
const TOOLS = [
|
|
36
|
+
tool("mcp__gh__create_issue", "Open a new issue in a repository to track a defect or task."),
|
|
37
|
+
tool("mcp__db__run_query", "Run a read-only SQL query against the analytics database."),
|
|
38
|
+
tool("mcp__cal__add_event", "Add an event to the calendar."),
|
|
39
|
+
];
|
|
40
|
+
|
|
41
|
+
interface Sent {
|
|
42
|
+
state: { request: string; tools: Record<string, string> };
|
|
43
|
+
questions: Record<string, { instructions: string }>;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function setup(options: { enabled?: boolean; credential?: boolean; relevant?: string[]; score?: number } = {}) {
|
|
47
|
+
writeFileSync(
|
|
48
|
+
state.settingsPath,
|
|
49
|
+
JSON.stringify({ jevAdvisory: { routes: { toolSearch: { enabled: options.enabled ?? true } } } }),
|
|
50
|
+
"utf-8",
|
|
51
|
+
);
|
|
52
|
+
const relevant = new Set(options.relevant ?? []);
|
|
53
|
+
const sent: Sent[] = [];
|
|
54
|
+
const classify = vi.fn(async (_model: unknown, context: Sent) => {
|
|
55
|
+
sent.push(context);
|
|
56
|
+
const names = Object.keys(context.state.tools);
|
|
57
|
+
const answers = Object.fromEntries(
|
|
58
|
+
Object.keys(context.questions).map((id, index) => [
|
|
59
|
+
id,
|
|
60
|
+
{ noul: relevant.has(names[Number(id.slice(1))]) ? (options.score ?? 0.92) : 0.08 },
|
|
61
|
+
]),
|
|
62
|
+
);
|
|
63
|
+
return classifierResponse(JSON.stringify({ answers }));
|
|
64
|
+
});
|
|
65
|
+
const events: Record<string, unknown>[] = [];
|
|
66
|
+
const ranker = createJevToolRanker({
|
|
67
|
+
events: { emit: (channel, data) => channel === JEV_ROUTING_EVENT && events.push(data as Record<string, unknown>) },
|
|
68
|
+
});
|
|
69
|
+
const ctx = {
|
|
70
|
+
modelRegistry: {
|
|
71
|
+
getAll: () => [providerTemplate()],
|
|
72
|
+
// Only Jev is registered: asking for any other model would find none and fall back to BM25.
|
|
73
|
+
findOfType: (type: string, provider: string, id: string) =>
|
|
74
|
+
type === "classifier" && provider === "tokenin" && id === "jev-1.13"
|
|
75
|
+
? { ...TOKENIN_JEV_CLASSIFIER }
|
|
76
|
+
: undefined,
|
|
77
|
+
getApiKeyAndHeaders: async () => (options.credential === false ? { ok: false } : { ok: true, apiKey: "key", headers: {} }),
|
|
78
|
+
classify,
|
|
79
|
+
},
|
|
80
|
+
};
|
|
81
|
+
return { ranker, ctx: ctx as never, classify, sent, events };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
beforeEach(() => vi.clearAllMocks());
|
|
85
|
+
|
|
86
|
+
describe("Jev tool ranker", () => {
|
|
87
|
+
it("finds a tool whose words the query never uses", async () => {
|
|
88
|
+
const query = "file a bug report for the crash";
|
|
89
|
+
expect(new Bm25Ranker().rank(query, TOOLS, 5)).toEqual([]);
|
|
90
|
+
const { ranker, ctx, classify, sent, events } = setup({ relevant: ["mcp__gh__create_issue"] });
|
|
91
|
+
|
|
92
|
+
const matches = await ranker.rank(query, TOOLS, 5, { ctx });
|
|
93
|
+
|
|
94
|
+
expect(matches.map((match) => match.name)).toEqual(["mcp__gh__create_issue"]);
|
|
95
|
+
expect(classify.mock.calls[0][0]).toMatchObject({ id: "jev-1.13", api: "typesafe-system-one" });
|
|
96
|
+
expect(sent).toHaveLength(1);
|
|
97
|
+
expect(sent[0].state.request).toBe(query);
|
|
98
|
+
expect(Object.keys(sent[0].state.tools)).toEqual(TOOLS.map((document) => document.name));
|
|
99
|
+
expect(sent[0].state.tools["mcp__cal__add_event"]).toContain("Add an event to the calendar.");
|
|
100
|
+
expect(events).toEqual([
|
|
101
|
+
expect.objectContaining({ route: "toolSearch", outcome: "jev", candidates: 3, judged: 3, matched: 1, confidence: "high" }),
|
|
102
|
+
]);
|
|
103
|
+
// Telemetry is shape only.
|
|
104
|
+
expect(JSON.stringify(events)).not.toContain("bug report");
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
it("orders matches by Jev's probability and honors the limit", async () => {
|
|
108
|
+
const { ranker, ctx } = setup({ relevant: ["mcp__db__run_query", "mcp__cal__add_event"] });
|
|
109
|
+
const matches = await ranker.rank("anything", TOOLS, 1, { ctx });
|
|
110
|
+
expect(matches).toHaveLength(1);
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
it("treats Jev as the judge: low probabilities mean no match", async () => {
|
|
114
|
+
const { ranker, ctx } = setup({ relevant: [] });
|
|
115
|
+
expect(await ranker.rank("calendar event", TOOLS, 5, { ctx })).toEqual([]);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
it("ranks with BM25 and sends nothing without a session", async () => {
|
|
119
|
+
const { ranker, classify } = setup({ relevant: ["mcp__gh__create_issue"] });
|
|
120
|
+
const matches = await ranker.rank("calendar event", TOOLS, 5);
|
|
121
|
+
expect(matches.map((match) => match.name)).toEqual(["mcp__cal__add_event"]);
|
|
122
|
+
expect(classify).not.toHaveBeenCalled();
|
|
123
|
+
});
|
|
124
|
+
|
|
125
|
+
it("ranks with BM25 when the route is off", async () => {
|
|
126
|
+
const { ranker, ctx, classify } = setup({ enabled: false });
|
|
127
|
+
expect((await ranker.rank("calendar event", TOOLS, 5, { ctx })).map((match) => match.name)).toEqual(["mcp__cal__add_event"]);
|
|
128
|
+
expect(classify).not.toHaveBeenCalled();
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
it("ranks with BM25 without a Token-In credential", async () => {
|
|
132
|
+
const { ranker, ctx, classify, events } = setup({ credential: false });
|
|
133
|
+
expect((await ranker.rank("calendar event", TOOLS, 5, { ctx })).map((match) => match.name)).toEqual(["mcp__cal__add_event"]);
|
|
134
|
+
expect(classify).not.toHaveBeenCalled();
|
|
135
|
+
expect(events).toEqual([expect.objectContaining({ route: "toolSearch", outcome: "fallback", reason: "no-credential" })]);
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
it("ranks with BM25 when every request fails", async () => {
|
|
139
|
+
const { ranker, ctx, classify, events } = setup();
|
|
140
|
+
classify.mockRejectedValue(new Error("down"));
|
|
141
|
+
expect((await ranker.rank("calendar event", TOOLS, 5, { ctx })).map((match) => match.name)).toEqual(["mcp__cal__add_event"]);
|
|
142
|
+
expect(events).toEqual([expect.objectContaining({ outcome: "fallback" })]);
|
|
143
|
+
});
|
|
144
|
+
|
|
145
|
+
it("ranks with BM25 when the caller already aborted", async () => {
|
|
146
|
+
const { ranker, ctx } = setup({ relevant: ["mcp__gh__create_issue"] });
|
|
147
|
+
const controller = new AbortController();
|
|
148
|
+
controller.abort();
|
|
149
|
+
const matches = await ranker.rank("calendar event", TOOLS, 5, { ctx, signal: controller.signal });
|
|
150
|
+
expect(matches.map((match) => match.name)).toEqual(["mcp__cal__add_event"]);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
it("keeps BM25 hits for tools whose batch failed", async () => {
|
|
154
|
+
// 18 BM25 hits fill one batch of 16 and spill two into a second batch.
|
|
155
|
+
const documents = Array.from({ length: FIND_BATCH + 2 }, (_, index) => tool(`mcp__x__gadget_${index}`, `A gadget, number ${index}.`));
|
|
156
|
+
const { ranker, ctx, classify } = setup({ relevant: ["mcp__x__gadget_3"] });
|
|
157
|
+
const answer = classify.getMockImplementation()!;
|
|
158
|
+
classify.mockImplementation(async (model, context) => {
|
|
159
|
+
if ("mcp__x__gadget_16" in context.state.tools) throw new Error("down");
|
|
160
|
+
return answer(model, context);
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
const matches = await ranker.rank("gadget", documents, 30, { ctx });
|
|
164
|
+
|
|
165
|
+
// Jev's verdicts come first, then the tools Jev never scored in BM25 order; scored-and-rejected tools are gone.
|
|
166
|
+
expect(matches.map((match) => match.name)).toEqual(["mcp__x__gadget_3", "mcp__x__gadget_16", "mcp__x__gadget_17"]);
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
it("sends at most TOOL_RANK_MAX_REQUESTS requests, choosing by BM25 past the cap", async () => {
|
|
170
|
+
const cap = TOOL_RANK_MAX_REQUESTS * FIND_BATCH;
|
|
171
|
+
const documents = [
|
|
172
|
+
...Array.from({ length: cap + 50 }, (_, index) => tool(`mcp__bulk__tool_${index}`, `Generic helper ${index}.`)),
|
|
173
|
+
tool("mcp__late__quokka", "Pet the quokka."),
|
|
174
|
+
];
|
|
175
|
+
const { ranker, ctx, classify, sent } = setup({ relevant: ["mcp__late__quokka"] });
|
|
176
|
+
|
|
177
|
+
const matches = await ranker.rank("quokka", documents, 3, { ctx });
|
|
178
|
+
|
|
179
|
+
expect(classify.mock.calls.length).toBeLessThanOrEqual(TOOL_RANK_MAX_REQUESTS);
|
|
180
|
+
const judged = sent.flatMap((request) => Object.keys(request.state.tools));
|
|
181
|
+
expect(judged).toHaveLength(cap);
|
|
182
|
+
expect(judged).toContain("mcp__late__quokka");
|
|
183
|
+
expect(matches.map((match) => match.name)).toEqual(["mcp__late__quokka"]);
|
|
184
|
+
});
|
|
185
|
+
});
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* jev-tool-ranker — ranks tools for `tool_search` and codemode's `searchTools()` with Jev.
|
|
3
|
+
*
|
|
4
|
+
* Each candidate tool is one yes/no question ("would this tool help with the request?"), batched
|
|
5
|
+
* into parallel requests; the model's probability is the score. This is the `toolSearch` route of
|
|
6
|
+
* `jevAdvisory` (on by default). It asks the shared Jev model; set
|
|
7
|
+
* `jevAdvisory.routes.toolSearch.model` to use another decisions model from `jev/classifier.ts`.
|
|
8
|
+
* The ranker never throws and never blocks discovery: with no
|
|
9
|
+
* session, the route off, no Token-In credential, or any request failing, it returns the local
|
|
10
|
+
* BM25 ranking, and tools the model did not score are appended in BM25 order.
|
|
11
|
+
*
|
|
12
|
+
* Which tools are sent: every candidate, up to TOOL_RANK_MAX_REQUESTS * FIND_BATCH. Past that cap
|
|
13
|
+
* the BM25 order decides which candidates are sent; the rest keep their BM25 place.
|
|
14
|
+
*/
|
|
15
|
+
import { Bm25Ranker, getSettingsPath, registerToolRanker } from "@selesai/code";
|
|
16
|
+
import type {
|
|
17
|
+
ExtensionAPI,
|
|
18
|
+
ToolRankOptions,
|
|
19
|
+
ToolRanker,
|
|
20
|
+
ToolSearchDocument,
|
|
21
|
+
ToolSearchMatch,
|
|
22
|
+
} from "@selesai/code";
|
|
23
|
+
import { FIND_BATCH, FindRequests, type JudgeItem, type JudgeStage, judgeNouls } from "./jev-ask-tool.ts";
|
|
24
|
+
import {
|
|
25
|
+
askJevAnswers,
|
|
26
|
+
confidenceBucket,
|
|
27
|
+
emitJevTelemetry,
|
|
28
|
+
type JevAdvisoryConfig,
|
|
29
|
+
jevConnection,
|
|
30
|
+
type JevRuntime,
|
|
31
|
+
jevUnavailable,
|
|
32
|
+
readJevAdvisoryConfig,
|
|
33
|
+
} from "./jev/decisions.ts";
|
|
34
|
+
|
|
35
|
+
/** Parallel Jev requests one ranking may send (16 tools each). */
|
|
36
|
+
export const TOOL_RANK_MAX_REQUESTS = 12;
|
|
37
|
+
/** Spare requests for retrying transport failures. */
|
|
38
|
+
export const TOOL_RANK_RETRIES = 2;
|
|
39
|
+
/** A tool is a match at this probability or above: matches become part of the model's context. */
|
|
40
|
+
export const TOOL_RANK_KEEP = 0.5;
|
|
41
|
+
/** Characters of a tool's summary shown to Jev. */
|
|
42
|
+
export const TOOL_RANK_SUMMARY_CHARS = 480;
|
|
43
|
+
|
|
44
|
+
const STAGE: Pick<JudgeStage, "stateKey" | "criteria"> = {
|
|
45
|
+
stateKey: "tools",
|
|
46
|
+
criteria: {
|
|
47
|
+
true: "The tool's name and description show it can do what the request asks for, or is a direct step toward it.",
|
|
48
|
+
false: "The tool does something unrelated to the request.",
|
|
49
|
+
},
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
export interface JevToolRankerOptions {
|
|
53
|
+
/** Receives route telemetry (shape and outcome only; never the query or tool text). */
|
|
54
|
+
events?: { emit(channel: string, data: unknown): void };
|
|
55
|
+
/** Ranks when Jev cannot, and orders the candidates sent past the cap. Default: BM25. */
|
|
56
|
+
fallback?: ToolRanker;
|
|
57
|
+
/** Reads the `jevAdvisory` settings. Default: the agent settings file. */
|
|
58
|
+
readConfig?: () => JevAdvisoryConfig;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Resolves with `work`, or rejects if `signal` aborts first. The request itself is not cancelled. */
|
|
62
|
+
function abortable<T>(work: Promise<T>, signal: AbortSignal | undefined): Promise<T> {
|
|
63
|
+
if (!signal) return work;
|
|
64
|
+
return new Promise<T>((resolve, reject) => {
|
|
65
|
+
const onAbort = () => reject(new Error("aborted"));
|
|
66
|
+
if (signal.aborted) return onAbort();
|
|
67
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
68
|
+
work.then(resolve, reject).finally(() => signal.removeEventListener("abort", onAbort));
|
|
69
|
+
});
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export function createJevToolRanker(options: JevToolRankerOptions = {}): ToolRanker {
|
|
73
|
+
const fallback = options.fallback ?? new Bm25Ranker();
|
|
74
|
+
const readConfig = options.readConfig ?? (() => readJevAdvisoryConfig(getSettingsPath()));
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
async rank(
|
|
78
|
+
query: string,
|
|
79
|
+
documents: readonly ToolSearchDocument[],
|
|
80
|
+
limit: number,
|
|
81
|
+
rankOptions?: ToolRankOptions,
|
|
82
|
+
): Promise<ToolSearchMatch[]> {
|
|
83
|
+
const local = await fallback.rank(query, documents, documents.length, rankOptions);
|
|
84
|
+
const ctx = rankOptions?.ctx;
|
|
85
|
+
if (!ctx || documents.length === 0 || limit <= 0 || query.trim() === "") return local.slice(0, limit);
|
|
86
|
+
|
|
87
|
+
const started = Date.now();
|
|
88
|
+
const telemetry = (outcome: "jev" | "fallback", extra: Record<string, unknown> = {}) =>
|
|
89
|
+
emitJevTelemetry(options.events, "decision", {
|
|
90
|
+
route: "toolSearch",
|
|
91
|
+
outcome,
|
|
92
|
+
candidates: documents.length,
|
|
93
|
+
elapsedMs: Date.now() - started,
|
|
94
|
+
...extra,
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
try {
|
|
98
|
+
// Same cast as ask_jev: the session registry is wider than JevRuntime's header type.
|
|
99
|
+
const runtime = ctx as unknown as JevRuntime;
|
|
100
|
+
const config = readConfig();
|
|
101
|
+
const route = config.routes.toolSearch;
|
|
102
|
+
if (!route.enabled) return local.slice(0, limit);
|
|
103
|
+
const connection = jevConnection(config, route);
|
|
104
|
+
const unreachable = await jevUnavailable(runtime, connection);
|
|
105
|
+
if (unreachable) {
|
|
106
|
+
telemetry("fallback", { confidence: confidenceBucket(undefined), reason: unreachable });
|
|
107
|
+
return local.slice(0, limit);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
// BM25 hits first, then the rest in document order: the cap drops the least likely tools.
|
|
111
|
+
const byName = new Map(documents.map((document) => [document.name, document]));
|
|
112
|
+
const hits = new Set(local.map((match) => match.name));
|
|
113
|
+
const order = [...local.map((match) => match.name), ...documents.map((d) => d.name).filter((n) => !hits.has(n))];
|
|
114
|
+
const judged = order.slice(0, TOOL_RANK_MAX_REQUESTS * FIND_BATCH).map((name) => byName.get(name)!);
|
|
115
|
+
const items: JudgeItem[] = judged.map((document) => ({
|
|
116
|
+
key: document.name,
|
|
117
|
+
text: (document.summary ?? document.text).slice(0, TOOL_RANK_SUMMARY_CHARS),
|
|
118
|
+
question: `Would calling the tool "${document.name}" help carry out the request?`,
|
|
119
|
+
}));
|
|
120
|
+
|
|
121
|
+
const requests = new FindRequests(
|
|
122
|
+
(payload) => askJevAnswers(runtime, connection, { payload, maxBytes: route.payloadBytes }),
|
|
123
|
+
TOOL_RANK_MAX_REQUESTS + TOOL_RANK_RETRIES,
|
|
124
|
+
);
|
|
125
|
+
const outcome = await abortable(
|
|
126
|
+
judgeNouls(requests, { request: query, ...STAGE }, items, route.payloadBytes, {
|
|
127
|
+
send: TOOL_RANK_MAX_REQUESTS,
|
|
128
|
+
retry: TOOL_RANK_MAX_REQUESTS + TOOL_RANK_RETRIES,
|
|
129
|
+
}),
|
|
130
|
+
rankOptions?.signal,
|
|
131
|
+
);
|
|
132
|
+
|
|
133
|
+
const unscored = new Set(documents.map((document) => document.name));
|
|
134
|
+
const position = new Map(order.map((name, index) => [name, index]));
|
|
135
|
+
const scored: ToolSearchMatch[] = [];
|
|
136
|
+
items.forEach((item, index) => {
|
|
137
|
+
const score = outcome.scores[index];
|
|
138
|
+
if (typeof score !== "number" || !Number.isFinite(score)) return;
|
|
139
|
+
unscored.delete(item.key);
|
|
140
|
+
if (score >= TOOL_RANK_KEEP) scored.push({ name: item.key, score });
|
|
141
|
+
});
|
|
142
|
+
if (unscored.size === documents.length) {
|
|
143
|
+
telemetry("fallback", { confidence: confidenceBucket(undefined), reason: outcome.reasons[0] ?? "missing" });
|
|
144
|
+
return local.slice(0, limit);
|
|
145
|
+
}
|
|
146
|
+
scored.sort((a, b) => b.score - a.score || (position.get(a.name) ?? 0) - (position.get(b.name) ?? 0));
|
|
147
|
+
const extras = local.filter((match) => unscored.has(match.name)).map((match) => ({ name: match.name, score: 0 }));
|
|
148
|
+
telemetry("jev", {
|
|
149
|
+
judged: items.length,
|
|
150
|
+
matched: scored.length,
|
|
151
|
+
confidence: confidenceBucket(scored[0]?.score),
|
|
152
|
+
...(unscored.size > 0 ? { unscored: unscored.size } : {}),
|
|
153
|
+
});
|
|
154
|
+
return [...scored, ...extras].slice(0, limit);
|
|
155
|
+
} catch {
|
|
156
|
+
// Discovery must work without Jev: an abort, a transport error, or a bad answer ranks locally.
|
|
157
|
+
telemetry("fallback", { confidence: confidenceBucket(undefined), reason: "transport" });
|
|
158
|
+
return local.slice(0, limit);
|
|
159
|
+
}
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
export default function jevToolRankerExtension(pi: ExtensionAPI): void {
|
|
165
|
+
const dispose = registerToolRanker(createJevToolRanker({ events: pi.events }));
|
|
166
|
+
pi.on("session_shutdown", () => {
|
|
167
|
+
dispose();
|
|
168
|
+
});
|
|
169
|
+
}
|
|
@@ -64,8 +64,6 @@ import {
|
|
|
64
64
|
recoveryFor,
|
|
65
65
|
reduceGraftState,
|
|
66
66
|
type RetrievalMode,
|
|
67
|
-
STATUS_KEY,
|
|
68
|
-
statusText,
|
|
69
67
|
} from "./state.ts";
|
|
70
68
|
import { GRAFT_TOOL_NAMES, registerGraftTools } from "./tools.ts";
|
|
71
69
|
|
|
@@ -280,14 +278,12 @@ export default function graftExtension(pi: ExtensionAPI): void {
|
|
|
280
278
|
let structuralBuildSession: GraftSession | undefined;
|
|
281
279
|
const guard = createInjectionGuard();
|
|
282
280
|
|
|
283
|
-
const applyEvent = (
|
|
281
|
+
const applyEvent = (_ctx: ExtensionContext | undefined, event: GraftEvent): void => {
|
|
284
282
|
session.state = reduceGraftState(session.state, event);
|
|
285
|
-
if (ctx?.hasUI) ctx.ui.setStatus(STATUS_KEY, statusText(session.state));
|
|
286
283
|
};
|
|
287
284
|
|
|
288
|
-
const setMode = (mode: RetrievalMode,
|
|
285
|
+
const setMode = (mode: RetrievalMode, _ctx?: ExtensionContext): void => {
|
|
289
286
|
session.mode = mode;
|
|
290
|
-
if (ctx?.hasUI) ctx.ui.setStatus(`${STATUS_KEY}-mode`, `graft mode: ${mode}`);
|
|
291
287
|
};
|
|
292
288
|
|
|
293
289
|
/** Share one global npm install between startup and a concurrent `/graft` command. */
|
|
@@ -506,10 +502,7 @@ export default function graftExtension(pi: ExtensionAPI): void {
|
|
|
506
502
|
session.disposed = false;
|
|
507
503
|
session.settings = readGraftSettings({ cwd: ctx.cwd, trusted: ctx.isProjectTrusted() });
|
|
508
504
|
session.enabled = session.settings.enabled !== false;
|
|
509
|
-
if (!session.enabled)
|
|
510
|
-
if (ctx.hasUI) ctx.ui.setStatus(STATUS_KEY, undefined);
|
|
511
|
-
return;
|
|
512
|
-
}
|
|
505
|
+
if (!session.enabled) return;
|
|
513
506
|
if (session.settings.telemetry !== "inherit") session.telemetry = applyTelemetryDefault();
|
|
514
507
|
setMode(readBranchMode(ctx.sessionManager.getBranch()) ?? session.settings.mode ?? DEFAULT_RETRIEVAL_MODE, ctx);
|
|
515
508
|
|
|
@@ -518,13 +511,9 @@ export default function graftExtension(pi: ExtensionAPI): void {
|
|
|
518
511
|
else autoBuild(ctx, session);
|
|
519
512
|
});
|
|
520
513
|
|
|
521
|
-
pi.on("session_shutdown", (_event: SessionShutdownEvent
|
|
514
|
+
pi.on("session_shutdown", (_event: SessionShutdownEvent) => {
|
|
522
515
|
session.disposed = true;
|
|
523
516
|
session.telemetry?.restore();
|
|
524
|
-
if (ctx.hasUI) {
|
|
525
|
-
ctx.ui.setStatus(STATUS_KEY, undefined);
|
|
526
|
-
ctx.ui.setStatus(`${STATUS_KEY}-mode`, undefined);
|
|
527
|
-
}
|
|
528
517
|
guard.beginTurn();
|
|
529
518
|
session = freshSession();
|
|
530
519
|
});
|
|
@@ -638,10 +627,6 @@ export default function graftExtension(pi: ExtensionAPI): void {
|
|
|
638
627
|
const debounceMs = (session.settings.refreshDebounceSeconds ?? DEFAULT_REFRESH_DEBOUNCE_SECONDS) * 1000;
|
|
639
628
|
if (Date.now() - session.lastRefreshAt < debounceMs) return;
|
|
640
629
|
session.lastRefreshAt = Date.now();
|
|
641
|
-
if (ctx.hasUI) {
|
|
642
|
-
ctx.ui.setStatus(STATUS_KEY, statusText(reduceGraftState(session.state, { type: "refresh-started" })));
|
|
643
|
-
}
|
|
644
|
-
|
|
645
630
|
void (async () => {
|
|
646
631
|
try {
|
|
647
632
|
const run = await buildGraph(ctx, session, "refresh");
|