@selesai/code 0.14.0 → 0.14.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.d.ts +10 -1
- package/dist/config.js +39 -1
- package/dist/core/resource-loader.js +2 -0
- package/dist/extensions/codemode/execute.js +18 -6
- package/dist/extensions/jev/classifier.test.ts +85 -20
- package/dist/extensions/jev/classifier.ts +147 -61
- package/dist/extensions/jev/decisions.test.ts +21 -0
- package/dist/extensions/jev/decisions.ts +36 -5
- package/dist/extensions/jev-ask-tool.ts +1 -1
- package/dist/extensions/jev-tool-ranker.test.ts +185 -0
- package/dist/extensions/jev-tool-ranker.ts +169 -0
- package/dist/extensions/package.json +1 -0
- package/dist/extensions/pi-graft/index.ts +4 -19
- package/dist/extensions/pi-graft/lifecycle.test.ts +32 -25
- package/dist/extensions/pi-graft/tools.ts +2 -2
- package/dist/extensions/pi-hermes-memory/src/tools/skill-tool.ts +1 -1
- package/dist/extensions/pi-subagents/src/extension/index.ts +2 -2
- package/dist/extensions/pi-subagents/test/unit/index-child-registration.test.ts +1 -1
- package/dist/extensions/question/index.ts +1 -1
- package/dist/extensions/tokenin-onboarding.ts +3 -3
- package/dist/extensions/tool-search/tool.d.ts +23 -4
- package/dist/extensions/tool-search/tool.js +35 -9
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/docs/codemode.md +1 -1
- package/package.json +3 -3
package/dist/config.d.ts
CHANGED
|
@@ -187,6 +187,15 @@ export declare function seedMissingSubagentSettings(destPath: string, bundledDef
|
|
|
187
187
|
* rewritten.
|
|
188
188
|
*/
|
|
189
189
|
export declare function seedMissingGraftSettings(destPath: string, bundledDefaultsDir?: string): boolean;
|
|
190
|
+
/**
|
|
191
|
+
* Add the bundled `defaultTools` selection (which enables `codemode`) to an existing user
|
|
192
|
+
* settings file that does not set `defaultTools` at all. Settings files created before the
|
|
193
|
+
* bundled default existed would otherwise never get it. A user-provided `defaultTools`, even an
|
|
194
|
+
* empty one or one with only `-name` entries, is never touched. The file is modified with a
|
|
195
|
+
* minimal textual insertion, so other settings keep their formatting. Returns true when the file
|
|
196
|
+
* was rewritten.
|
|
197
|
+
*/
|
|
198
|
+
export declare function seedMissingDefaultToolsSettings(destPath: string, bundledDefaultsDir?: string): boolean;
|
|
190
199
|
/**
|
|
191
200
|
* Bundled extensions load directly from the installed package. Do not copy them
|
|
192
201
|
* into the user's agent dir, otherwise startup discovers both copies and tool
|
|
@@ -213,7 +222,7 @@ export declare function seedDefaultSkills(agentDir: string, bundledSkillsDir?: s
|
|
|
213
222
|
export declare function seedDefaultThemes(agentDir: string, bundledThemesDir?: string): string[];
|
|
214
223
|
/**
|
|
215
224
|
* Run bootstrap for the agent dir: ensure directories, seed a missing
|
|
216
|
-
* settings.json, and add missing bundled subagent
|
|
225
|
+
* settings.json, and add missing bundled subagent, graft, and default-tools settings. Idempotent — safe
|
|
217
226
|
* on every startup.
|
|
218
227
|
* Bundled models load package-locally; copying them would expose internal
|
|
219
228
|
* providers as user config. Bundled extensions also stay package-local.
|
package/dist/config.js
CHANGED
|
@@ -801,6 +801,43 @@ export function seedMissingGraftSettings(destPath, bundledDefaultsDir = getBundl
|
|
|
801
801
|
return false;
|
|
802
802
|
}
|
|
803
803
|
}
|
|
804
|
+
/**
|
|
805
|
+
* Add the bundled `defaultTools` selection (which enables `codemode`) to an existing user
|
|
806
|
+
* settings file that does not set `defaultTools` at all. Settings files created before the
|
|
807
|
+
* bundled default existed would otherwise never get it. A user-provided `defaultTools`, even an
|
|
808
|
+
* empty one or one with only `-name` entries, is never touched. The file is modified with a
|
|
809
|
+
* minimal textual insertion, so other settings keep their formatting. Returns true when the file
|
|
810
|
+
* was rewritten.
|
|
811
|
+
*/
|
|
812
|
+
export function seedMissingDefaultToolsSettings(destPath, bundledDefaultsDir = getBundledDefaultsDir()) {
|
|
813
|
+
if (!existsSync(destPath))
|
|
814
|
+
return false;
|
|
815
|
+
try {
|
|
816
|
+
const raw = readFileSync(destPath, "utf-8");
|
|
817
|
+
const settings = JSON.parse(raw);
|
|
818
|
+
const defaults = JSON.parse(readFileSync(join(bundledDefaultsDir, "settings.json"), "utf-8"));
|
|
819
|
+
if (typeof settings !== "object" ||
|
|
820
|
+
settings === null ||
|
|
821
|
+
Array.isArray(settings) ||
|
|
822
|
+
Object.hasOwn(settings, "defaultTools") ||
|
|
823
|
+
typeof defaults !== "object" ||
|
|
824
|
+
defaults === null ||
|
|
825
|
+
Array.isArray(defaults)) {
|
|
826
|
+
return false;
|
|
827
|
+
}
|
|
828
|
+
const defaultTools = defaults.defaultTools;
|
|
829
|
+
if (!Array.isArray(defaultTools) || !defaultTools.every((entry) => typeof entry === "string"))
|
|
830
|
+
return false;
|
|
831
|
+
const injected = injectRootObjectKey(raw, "defaultTools", defaultTools);
|
|
832
|
+
if (injected === undefined)
|
|
833
|
+
return false;
|
|
834
|
+
writeFileSync(destPath, injected);
|
|
835
|
+
return true;
|
|
836
|
+
}
|
|
837
|
+
catch {
|
|
838
|
+
return false;
|
|
839
|
+
}
|
|
840
|
+
}
|
|
804
841
|
/**
|
|
805
842
|
* Recursively copy a bundled directory tree into the user's agent dir,
|
|
806
843
|
* overwriting any existing file so the bundled copy stays authoritative.
|
|
@@ -902,7 +939,7 @@ export function seedDefaultThemes(agentDir, bundledThemesDir = getBundledThemesD
|
|
|
902
939
|
}
|
|
903
940
|
/**
|
|
904
941
|
* Run bootstrap for the agent dir: ensure directories, seed a missing
|
|
905
|
-
* settings.json, and add missing bundled subagent
|
|
942
|
+
* settings.json, and add missing bundled subagent, graft, and default-tools settings. Idempotent — safe
|
|
906
943
|
* on every startup.
|
|
907
944
|
* Bundled models load package-locally; copying them would expose internal
|
|
908
945
|
* providers as user config. Bundled extensions also stay package-local.
|
|
@@ -913,6 +950,7 @@ export function bootstrapAgentDir(agentDir = getAgentDir()) {
|
|
|
913
950
|
seedDefaultConfigFile(settingsPath, "settings.json");
|
|
914
951
|
seedMissingSubagentSettings(settingsPath);
|
|
915
952
|
seedMissingGraftSettings(settingsPath);
|
|
953
|
+
seedMissingDefaultToolsSettings(settingsPath);
|
|
916
954
|
seedDefaultExtensions(agentDir);
|
|
917
955
|
seedDefaultSkills(agentDir);
|
|
918
956
|
seedDefaultThemes(agentDir);
|
|
@@ -389,6 +389,8 @@ export class DefaultResourceLoader {
|
|
|
389
389
|
if (this.loaded) {
|
|
390
390
|
clearExtensionCache();
|
|
391
391
|
}
|
|
392
|
+
// The filter closes over the previous extension runtime (now stale); the reloaded extensions re-register theirs.
|
|
393
|
+
this.skillsIndexFilter = undefined;
|
|
392
394
|
let preTrustExtensions;
|
|
393
395
|
if (options?.resolveProjectTrust) {
|
|
394
396
|
preTrustExtensions = await this.loadProjectTrustExtensions();
|
|
@@ -7,12 +7,14 @@ import { getCodemodeWorkerSpecifier, getQuickJSWasmPath } from "../../config.js"
|
|
|
7
7
|
import { formatSize } from "../../core/tools/truncate.js";
|
|
8
8
|
import { combineUsage } from "../../core/usage-totals.js";
|
|
9
9
|
import { writeOutputFile } from "../../utils/output-files.js";
|
|
10
|
-
import {
|
|
10
|
+
import { createToolSearchDocument, DEFAULT_TOOL_SEARCH_LIMIT, getToolRanker } from "../tool-search/tool.js";
|
|
11
11
|
import { CODEMODE_DOCS_PATH, CODEMODE_STORE_ENTRY_TYPE, getCodemodeCallableTools, toCodemodeDeclaration, } from "./tool.js";
|
|
12
12
|
const ARGS_PREVIEW_CHARS = 200;
|
|
13
13
|
const ERROR_PREVIEW_CHARS = 500;
|
|
14
14
|
/** `models.classify()` and `models.generateImages()` calls one script may have in flight; `Promise.all` over many items queues the rest. */
|
|
15
15
|
const MAX_CONCURRENT_MODEL_CALLS = 4;
|
|
16
|
+
/** `searchTools()` calls per script that may use a model-backed ranker; later calls rank locally. */
|
|
17
|
+
const MAX_SESSION_SEARCHES_PER_SCRIPT = 8;
|
|
16
18
|
/**
|
|
17
19
|
* Heap limit for the QuickJS VM. The worker shares pi's process, so without a limit a runaway
|
|
18
20
|
* script can grow to wasm32's 4 GiB and take the session down. Overruns throw
|
|
@@ -413,7 +415,7 @@ export async function executeCodemode(toolCallId, input, signal, onUpdate, ctx,
|
|
|
413
415
|
const sandbox = new CodemodeSandbox({
|
|
414
416
|
tools: sandboxTools,
|
|
415
417
|
globals: [
|
|
416
|
-
...createDiscoveryGlobals(callable, samples, options),
|
|
418
|
+
...createDiscoveryGlobals(callable, samples, options, ctx),
|
|
417
419
|
...(options.models && ctx
|
|
418
420
|
? createModelGlobals(ctx.modelRegistry, toolCallId, calls, publish, addModelUsage, addGeneratedImages)
|
|
419
421
|
: []),
|
|
@@ -485,14 +487,16 @@ function isNamespaceName(namespace, query) {
|
|
|
485
487
|
* `searchTools()`, `describeTool()`, and `describeNamespace()`: ranked search and lookup over the
|
|
486
488
|
* script's nested tools and their namespaces.
|
|
487
489
|
*/
|
|
488
|
-
function createDiscoveryGlobals(tools, samples, options) {
|
|
489
|
-
|
|
490
|
+
function createDiscoveryGlobals(tools, samples, options, ctx) {
|
|
491
|
+
// A script may loop over searchTools(). Past this many searches the session is withheld from the
|
|
492
|
+
// ranker, so a model-backed ranker falls back to its local ranking instead of billing per iteration.
|
|
493
|
+
let sessionSearches = 0;
|
|
490
494
|
const entry = (name) => ({ name: toCodemodeIdentifier(name), description: samples.get(name) ?? "" });
|
|
491
495
|
return [
|
|
492
496
|
{
|
|
493
497
|
name: "searchTools",
|
|
494
498
|
spread: true,
|
|
495
|
-
execute: (args) => {
|
|
499
|
+
execute: async (args, { signal }) => {
|
|
496
500
|
const [query, searchOptions] = args;
|
|
497
501
|
if (typeof query !== "string")
|
|
498
502
|
throw new Error("searchTools() expects a query string");
|
|
@@ -510,7 +514,15 @@ function createDiscoveryGlobals(tools, samples, options) {
|
|
|
510
514
|
return [];
|
|
511
515
|
return [createToolSearchDocument(tool, toolNamespace)];
|
|
512
516
|
});
|
|
513
|
-
|
|
517
|
+
const session = ctx && sessionSearches < MAX_SESSION_SEARCHES_PER_SCRIPT ? ctx : undefined;
|
|
518
|
+
if (session)
|
|
519
|
+
sessionSearches += 1;
|
|
520
|
+
const ranked = await getToolRanker().rank(query, documents, limit, { signal, ctx: session });
|
|
521
|
+
const known = new Set(documents.map((document) => document.name));
|
|
522
|
+
return ranked
|
|
523
|
+
.filter((match) => known.has(match.name))
|
|
524
|
+
.slice(0, limit)
|
|
525
|
+
.map((match) => entry(match.name));
|
|
514
526
|
},
|
|
515
527
|
},
|
|
516
528
|
{
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { expect, it, vi } from "vitest";
|
|
2
2
|
import type { ClassifierContext } from "@earendil-works/pi-ai";
|
|
3
|
-
import { classifyTokenIn, TOKENIN_JEV_CLASSIFIER } from "./classifier.ts";
|
|
3
|
+
import { classifyTokenIn, MAX_IMAGE_BYTES, TOKENIN_CLASSIFIERS, TOKENIN_JEV_CLASSIFIER } from "./classifier.ts";
|
|
4
4
|
import { fitsClassifierRequest, toClassifierContext } from "./classifier-context.ts";
|
|
5
5
|
import { askJevAnswers } from "./decisions.ts";
|
|
6
6
|
|
|
@@ -9,38 +9,57 @@ const context: ClassifierContext = { state: { public: "protocol fixture" }, ques
|
|
|
9
9
|
risk: { type: "score", instructions: "Rate", criteria: ["low", "high"] },
|
|
10
10
|
ready: { type: "bool", instructions: "Ready?", criteria: { true: "Ready", false: "Not ready" } },
|
|
11
11
|
} };
|
|
12
|
-
const
|
|
12
|
+
const billed = .0000158;
|
|
13
13
|
const body = () => ({ answers: {
|
|
14
14
|
pick: { type: "choice", choice: "yes", probabilities: { yes: .9, no: .1 }, confidence: .9 },
|
|
15
15
|
risk: { type: "score", score: 1.4, confidence: .8, legend: { 0: "low", 1: "high" } },
|
|
16
16
|
ready: { type: "noul", noul: .87, confidence: .93 },
|
|
17
|
-
}, usage: { input_tokens: 10, output_tokens: 7, cost } });
|
|
18
|
-
|
|
19
|
-
|
|
17
|
+
}, usage: { input_tokens: 10, output_tokens: 7, cost: .0000122 } });
|
|
18
|
+
/** The gateway's reply: the System One body is a chat completion's message content, the billed cost a header. */
|
|
19
|
+
function transport(response: unknown, headers: Record<string, string> = { "x-litellm-response-cost": String(billed) }) {
|
|
20
|
+
return vi.fn(async (_input: string | URL | Request, _init?: RequestInit) =>
|
|
21
|
+
new Response(JSON.stringify({ choices: [{ message: { role: "assistant", content: JSON.stringify(response) } }] }), { headers: { "content-type": "application/json", ...headers } }));
|
|
20
22
|
}
|
|
21
|
-
it("
|
|
23
|
+
it("posts the System One request as a chat message to the gateway, with bool -> noul, and reports the billed cost", async () => {
|
|
22
24
|
const fetch = transport(body());
|
|
23
25
|
const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch });
|
|
24
26
|
expect(result.stopReason).toBe("stop");
|
|
25
|
-
expect(String(fetch.mock.calls[0][0])).toBe("https://lite.andlet.me/v1/
|
|
27
|
+
expect(String(fetch.mock.calls[0][0])).toBe("https://lite.andlet.me/v1/chat/completions");
|
|
26
28
|
const init = fetch.mock.calls[0][1]!;
|
|
27
29
|
expect(new Headers(init.headers).get("authorization")).toBe("Bearer fixture");
|
|
28
|
-
|
|
30
|
+
const sent = JSON.parse(String(init.body));
|
|
31
|
+
expect(sent).toEqual({ model: "jev-1.13", messages: [{ role: "user", content: expect.any(String) }] });
|
|
32
|
+
expect(JSON.parse(sent.messages[0].content)).toEqual({ state: JSON.stringify(context.state), questions: { ...context.questions, ready: { ...context.questions.ready, type: "noul" } } });
|
|
29
33
|
expect(result.answers.ready).toEqual({ type: "bool", probability: .87, confidence: .93 });
|
|
30
34
|
expect(result.answers.risk).toMatchObject({ legend: { 0: "low", 1: "high" } });
|
|
31
|
-
|
|
35
|
+
// The header is the only cost; the body's own usage.cost is the provider's raw charge.
|
|
36
|
+
expect(result.reportedCost).toBe(billed);
|
|
37
|
+
expect(result.usage).toBeUndefined();
|
|
38
|
+
expect(result.unpricedUsage).toMatchObject({ input: 10, output: 7, totalTokens: 17 });
|
|
32
39
|
});
|
|
33
|
-
it("
|
|
34
|
-
for (const
|
|
35
|
-
const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport(
|
|
40
|
+
it("reports no cost when the gateway sends none or an invalid one, and 0 for a cache hit", async () => {
|
|
41
|
+
for (const headers of [{}, { "x-litellm-response-cost": "-1" }, { "x-litellm-response-cost": "free" }, { "x-litellm-response-cost": "" }]) {
|
|
42
|
+
const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport(body(), headers) });
|
|
43
|
+
expect(result.stopReason).toBe("stop");
|
|
44
|
+
expect(result.reportedCost).toBeUndefined();
|
|
36
45
|
expect(result.usage).toBeUndefined();
|
|
37
46
|
expect(result).toHaveProperty("unpricedUsage.totalTokens", 17);
|
|
38
47
|
}
|
|
48
|
+
const cached = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport(body(), { "x-litellm-response-cost": "0.0001", "x-litellm-cache-key": "abc" }) });
|
|
49
|
+
expect(cached.reportedCost).toBe(0);
|
|
50
|
+
});
|
|
51
|
+
it("fails once, without a retry, on a reply that is not a chat completion carrying System One answers", async () => {
|
|
52
|
+
for (const reply of ["not json", JSON.stringify({ choices: [] }), JSON.stringify({ choices: [{ message: { content: "no json" } }] })]) {
|
|
53
|
+
const fetch = vi.fn(async () => new Response(reply, { headers: { "x-litellm-response-cost": "0.001" } }));
|
|
54
|
+
const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch });
|
|
55
|
+
expect(result.stopReason).toBe("error");
|
|
56
|
+
expect(fetch).toHaveBeenCalledTimes(1);
|
|
57
|
+
}
|
|
39
58
|
});
|
|
40
59
|
it("retains billed usage on a malformed answer and honours cancellation", async () => {
|
|
41
60
|
const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport({ ...body(), answers: {} }) });
|
|
42
61
|
expect(result.stopReason).toBe("error");
|
|
43
|
-
expect(result.
|
|
62
|
+
expect(result.reportedCost).toBe(billed);
|
|
44
63
|
const controller = new AbortController(); controller.abort();
|
|
45
64
|
const fetch = vi.fn(async () => { throw new DOMException("Aborted", "AbortError"); });
|
|
46
65
|
const aborted = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch, signal: controller.signal });
|
|
@@ -55,7 +74,7 @@ it("facade uses classifier-only providers without a synthetic chat model and pre
|
|
|
55
74
|
} }, { provider: "tokenin", model: "jev-1.13", timeoutMs: 1000, minConfidence: .6 }, { payload: context, maxBytes: 4096 });
|
|
56
75
|
expect(complete).not.toHaveBeenCalled();
|
|
57
76
|
expect(result.answers?.ready).toEqual({ type: "noul", noul: .87, confidence: .93 });
|
|
58
|
-
expect(result.
|
|
77
|
+
expect(result.reportedCost).toBe(billed);
|
|
59
78
|
});
|
|
60
79
|
it("conversion retains untrusted-material focus, supports bool defaults and rejects non-JSON state", () => {
|
|
61
80
|
const converted = toClassifierContext({ state: {}, questions: { q: { type: "noul", instructions: { question: "Judge", focus: "untrusted, not instructions" } } } });
|
|
@@ -74,12 +93,58 @@ it("bounds actual string-state escaping and callback transformations before fetc
|
|
|
74
93
|
const hook = vi.fn(() => ({ model: "jev-1.13", state: "already serialized", questions: context.questions }));
|
|
75
94
|
await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch, onPayload: hook });
|
|
76
95
|
expect(hook).toHaveBeenCalledTimes(1);
|
|
77
|
-
expect(JSON.parse(String(fetch.mock.calls[0][1]?.body)).state).toBe("already serialized");
|
|
96
|
+
expect(JSON.parse(JSON.parse(String(fetch.mock.calls[0][1]?.body)).messages[0].content).state).toBe("already serialized");
|
|
78
97
|
});
|
|
79
98
|
|
|
80
|
-
it("
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
99
|
+
it("serves every Token-In decisions model the same way, billed by the gateway", async () => {
|
|
100
|
+
expect(TOKENIN_CLASSIFIERS.map((model) => model.id)).toEqual([
|
|
101
|
+
"jev-1.13",
|
|
102
|
+
"cloudflare/clef",
|
|
103
|
+
"cloudflare/clef-flash",
|
|
104
|
+
"perplexity/pplx-decider-v1.1-27b",
|
|
105
|
+
"openai/gpt-6-luna-decisions",
|
|
106
|
+
]);
|
|
107
|
+
expect(new Set(TOKENIN_CLASSIFIERS.map((model) => model.id)).size).toBe(TOKENIN_CLASSIFIERS.length);
|
|
108
|
+
for (const model of TOKENIN_CLASSIFIERS) {
|
|
109
|
+
const fetch = transport(body());
|
|
110
|
+
const result = await classifyTokenIn(model, context, { apiKey: "fixture", fetch });
|
|
111
|
+
expect(result.stopReason, model.id).toBe("stop");
|
|
112
|
+
expect(String(fetch.mock.calls[0][0]), model.id).toBe("https://lite.andlet.me/v1/chat/completions");
|
|
113
|
+
expect(JSON.parse(String(fetch.mock.calls[0][1]!.body)), model.id).toMatchObject({ model: model.id });
|
|
114
|
+
expect(result.answers.ready, model.id).toMatchObject({ type: "bool", probability: .87 });
|
|
115
|
+
expect(result.reportedCost, model.id).toBe(billed);
|
|
116
|
+
expect(model.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
|
|
117
|
+
}
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
const image = { type: "image" as const, data: "QUJD", mimeType: "image/png" };
|
|
121
|
+
it("sends images inside the state array, only to the models that read them", async () => {
|
|
122
|
+
expect(TOKENIN_CLASSIFIERS.filter((model) => model.input.includes("image")).map((model) => model.id)).toEqual([
|
|
123
|
+
"perplexity/pplx-decider-v1.1-27b",
|
|
124
|
+
"openai/gpt-6-luna-decisions",
|
|
125
|
+
]);
|
|
126
|
+
for (const model of TOKENIN_CLASSIFIERS.filter((m) => m.input.includes("image"))) {
|
|
127
|
+
const fetch = transport(body());
|
|
128
|
+
const result = await classifyTokenIn(model, { ...context, images: [image, image] }, { apiKey: "fixture", fetch });
|
|
129
|
+
expect(result.stopReason, model.id).toBe("stop");
|
|
130
|
+
const request = JSON.parse(JSON.parse(String(fetch.mock.calls[0][1]!.body)).messages[0].content);
|
|
131
|
+
expect(request.images, model.id).toBeUndefined();
|
|
132
|
+
expect(request.state, model.id).toEqual([
|
|
133
|
+
JSON.stringify(context.state),
|
|
134
|
+
{ type: "image_url", image_url: { url: "data:image/png;base64,QUJD" } },
|
|
135
|
+
{ type: "image_url", image_url: { url: "data:image/png;base64,QUJD" } },
|
|
136
|
+
]);
|
|
137
|
+
}
|
|
138
|
+
// A text-only model is not sent the image: the call fails before any request.
|
|
139
|
+
const fetch = transport(body());
|
|
140
|
+
const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, { ...context, images: [image] }, { apiKey: "fixture", fetch });
|
|
141
|
+
expect(result.stopReason).toBe("error");
|
|
142
|
+
expect(fetch).not.toHaveBeenCalled();
|
|
143
|
+
});
|
|
144
|
+
it("refuses images over the size cap before sending anything", async () => {
|
|
145
|
+
const fetch = transport(body());
|
|
146
|
+
const lune = TOKENIN_CLASSIFIERS[4];
|
|
147
|
+
const result = await classifyTokenIn(lune, { ...context, images: [{ ...image, data: "A".repeat(MAX_IMAGE_BYTES + 1) }] }, { apiKey: "fixture", fetch });
|
|
148
|
+
expect(result.stopReason).toBe("error");
|
|
149
|
+
expect(fetch).not.toHaveBeenCalled();
|
|
85
150
|
});
|
|
@@ -1,86 +1,176 @@
|
|
|
1
|
-
/**
|
|
2
|
-
|
|
1
|
+
/**
|
|
2
|
+
* TokenIn's decisions classifiers: Jev and the gateway's other decisions models (Clef, Perplexity
|
|
3
|
+
* Decider, GPT-6 Luna). They all take the same System One request and answer in the same format, so
|
|
4
|
+
* one implementation serves every model; the model id in the request body picks the deployment.
|
|
5
|
+
*
|
|
6
|
+
* Transport: the gateway serves them as chat deployments, so a request is `POST {baseUrl}/chat/completions`
|
|
7
|
+
* with one user message whose content is the JSON System One request `{state, questions}`, and the
|
|
8
|
+
* answer is the System One body (`{answers, usage}`) as the reply's message content. (Its
|
|
9
|
+
* `/v1/systemone` route is not deployed.) Protocol parsing and retries stay upstream-owned: this runs
|
|
10
|
+
* upstream's System One classifier and only changes the envelope through its `onPayload` and `fetch` hooks.
|
|
11
|
+
*
|
|
12
|
+
* Images travel inside the request's `state` array as `{ type: "image_url", image_url: { url: "data:..." } }`
|
|
13
|
+
* parts; the gateway rejects a top-level `images`. Only models that really read them declare image input.
|
|
14
|
+
*
|
|
15
|
+
* Billing: the gateway bills more than the provider's raw charge and puts what the account was charged
|
|
16
|
+
* in the `x-litellm-response-cost` header (0 for a response-cache hit). That is the only cost reported.
|
|
17
|
+
* The `usage.cost` in the body is the provider's raw charge (and 0 for some models), and the catalog
|
|
18
|
+
* price below is a placeholder, so neither is ever presented as cost.
|
|
19
|
+
*/
|
|
20
|
+
import type {
|
|
21
|
+
ClassifierApi,
|
|
22
|
+
ClassifierContext,
|
|
23
|
+
ClassifierModel,
|
|
24
|
+
ClassifierOptions,
|
|
25
|
+
ClassifierResult,
|
|
26
|
+
ImageContent,
|
|
27
|
+
Usage,
|
|
28
|
+
} from "@earendil-works/pi-ai";
|
|
3
29
|
import { classify as classifySystemOne } from "@earendil-works/pi-ai/api/typesafe-system-one";
|
|
4
30
|
import { fitsClassifierRequest, tokenInWirePayload } from "./classifier-context.ts";
|
|
5
31
|
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
32
|
+
/**
|
|
33
|
+
* `cost` is a required catalog field, NOT a price. Context windows are the ones pi-ai's OpenRouter
|
|
34
|
+
* catalog lists for the same models. `vision` models were checked against the live gateway with
|
|
35
|
+
* solid-colour images: Perplexity Decider and GPT-6 Luna answered every colour correctly, while Clef and
|
|
36
|
+
* Clef Flash accept an image but answer as if they had not seen it (their token counts follow the
|
|
37
|
+
* base64 text), so they are text-only here even though the OpenRouter catalog lists image input.
|
|
38
|
+
*/
|
|
39
|
+
function tokenInDecisionsModel(
|
|
40
|
+
id: string,
|
|
41
|
+
name: string,
|
|
42
|
+
contextWindow: number,
|
|
43
|
+
vision = false,
|
|
44
|
+
): ClassifierModel<"typesafe-system-one"> {
|
|
45
|
+
return {
|
|
46
|
+
type: "classifier", provider: "tokenin", id, name: `${name} (service-priced)`,
|
|
47
|
+
api: "typesafe-system-one", baseUrl: "https://lite.andlet.me/v1", input: vision ? ["text", "image"] : ["text"], contextWindow,
|
|
48
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export const TOKENIN_JEV_CLASSIFIER = tokenInDecisionsModel("jev-1.13", "Jev 1.13", 64000);
|
|
53
|
+
|
|
54
|
+
/** Cloudflare's Clef Flash: 65,536 tokens of context. */
|
|
55
|
+
export const TOKENIN_CLEF_FLASH_CLASSIFIER = tokenInDecisionsModel("cloudflare/clef-flash", "Clef Flash", 65536);
|
|
56
|
+
|
|
57
|
+
/** Every decisions model Token-In serves, Jev first. */
|
|
58
|
+
export const TOKENIN_CLASSIFIERS: readonly ClassifierModel<"typesafe-system-one">[] = [
|
|
59
|
+
TOKENIN_JEV_CLASSIFIER,
|
|
60
|
+
tokenInDecisionsModel("cloudflare/clef", "Clef", 65536),
|
|
61
|
+
TOKENIN_CLEF_FLASH_CLASSIFIER,
|
|
62
|
+
tokenInDecisionsModel("perplexity/pplx-decider-v1.1-27b", "Perplexity Decider 1.1 27B", 262144, true),
|
|
63
|
+
tokenInDecisionsModel("openai/gpt-6-luna-decisions", "GPT-6 Luna Decisions", 1050000, true),
|
|
64
|
+
];
|
|
65
|
+
|
|
66
|
+
const COST_HEADER = "x-litellm-response-cost";
|
|
67
|
+
const CACHE_HIT_HEADER = "x-litellm-cache-key";
|
|
68
|
+
/** The request without its images: state and questions, as before. */
|
|
69
|
+
const MAX_TEXT_BYTES = 64 * 1024;
|
|
70
|
+
/** ponytail: 2 MiB of base64 in total, the most checked against the live gateway; raise once larger ones are measured. */
|
|
71
|
+
export const MAX_IMAGE_BYTES = 2 * 1024 * 1024;
|
|
13
72
|
|
|
14
73
|
function record(value: unknown): value is Record<string, unknown> {
|
|
15
74
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
16
75
|
}
|
|
17
|
-
function reportedCost(value: unknown): Usage["cost"] | undefined {
|
|
18
|
-
if (!record(value)) return undefined;
|
|
19
|
-
const { input, output, cacheRead, cacheWrite, total } = value;
|
|
20
|
-
const money = (n: unknown): n is number => typeof n === "number" && Number.isFinite(n) && n >= 0;
|
|
21
|
-
return money(input) && money(output) && money(cacheRead) && money(cacheWrite) && money(total)
|
|
22
|
-
? { input, output, cacheRead, cacheWrite, total } : undefined;
|
|
23
|
-
}
|
|
24
76
|
|
|
25
77
|
export interface ServicePricedClassifierResult extends ClassifierResult {
|
|
26
|
-
/**
|
|
78
|
+
/** Token counts without a monetary claim: the catalog has no price for these calls. */
|
|
27
79
|
unpricedUsage?: Omit<Usage, "cost">;
|
|
28
|
-
/**
|
|
80
|
+
/** What the gateway billed for the call; absent when it did not say. */
|
|
29
81
|
reportedCost?: number;
|
|
30
82
|
}
|
|
31
83
|
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
84
|
+
function imagePart(image: ImageContent): unknown {
|
|
85
|
+
return { type: "image_url", image_url: { url: `data:${image.mimeType};base64,${image.data}` } };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** The System One request upstream built, as the gateway's chat body, with any images added to its `state`. */
|
|
89
|
+
function chatBody(payload: unknown, images: readonly ImageContent[]): unknown {
|
|
90
|
+
const wire = tokenInWirePayload(payload);
|
|
91
|
+
if (!record(wire) || typeof wire.model !== "string") throw new Error("TokenIn classifier requires a model");
|
|
92
|
+
if (!fitsClassifierRequest(wire, MAX_TEXT_BYTES)) throw new Error("TokenIn classifier payload exceeds 64 KiB");
|
|
93
|
+
if (images.reduce((total, image) => total + image.data.length, 0) > MAX_IMAGE_BYTES) {
|
|
94
|
+
throw new Error(`TokenIn classifier images exceed ${MAX_IMAGE_BYTES / 1024 / 1024} MiB`);
|
|
95
|
+
}
|
|
96
|
+
const { model, ...request } = wire;
|
|
97
|
+
const state = images.length > 0 ? [wire.state, ...images.map(imagePart)] : wire.state;
|
|
98
|
+
return { model, messages: [{ role: "user", content: JSON.stringify({ ...request, state }) }] };
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/** The System One body a chat completion carries as its message content, or undefined. */
|
|
102
|
+
function systemOneBody(completion: unknown): Record<string, unknown> | undefined {
|
|
103
|
+
const choices = record(completion) ? completion.choices : undefined;
|
|
104
|
+
const message = Array.isArray(choices) && record(choices[0]) ? choices[0].message : undefined;
|
|
105
|
+
const content = record(message) ? message.content : undefined;
|
|
106
|
+
if (typeof content !== "string") return undefined;
|
|
107
|
+
try {
|
|
108
|
+
const body: unknown = JSON.parse(content);
|
|
109
|
+
return record(body) ? body : undefined;
|
|
110
|
+
} catch {
|
|
111
|
+
return undefined;
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** The amount the account was charged for this response, or undefined when the gateway did not say. */
|
|
116
|
+
function billedCost(headers: Headers): number | undefined {
|
|
117
|
+
if ((headers.get(CACHE_HIT_HEADER) ?? "").trim() !== "") return 0;
|
|
118
|
+
const reported = headers.get(COST_HEADER);
|
|
119
|
+
if (reported === null || reported.trim() === "") return undefined;
|
|
120
|
+
const cost = Number(reported);
|
|
121
|
+
return Number.isFinite(cost) && cost >= 0 ? cost : undefined;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
export const classifyTokenIn = async (
|
|
125
|
+
model: ClassifierModel<ClassifierApi>,
|
|
126
|
+
context: ClassifierContext,
|
|
127
|
+
options?: ClassifierOptions,
|
|
128
|
+
): Promise<ServicePricedClassifierResult> => {
|
|
129
|
+
let billed: number | undefined;
|
|
35
130
|
let booleanConfidence: Record<string, number> = {};
|
|
36
131
|
let scoreLegend: Record<string, Record<string, string> | string[]> = {};
|
|
37
132
|
const requestFetch = options?.fetch ?? globalThis.fetch;
|
|
38
|
-
|
|
133
|
+
// Upstream's System One client rejects images, so a model that reads them gets them through `state`
|
|
134
|
+
// instead. For a text-only model they stay in the context and upstream reports them as unsupported.
|
|
135
|
+
const images = model.input.includes("image") ? (context.images ?? []) : [];
|
|
136
|
+
const { images: _inState, ...textContext } = context;
|
|
137
|
+
const result: ServicePricedClassifierResult = await classifySystemOne(model, images.length > 0 ? textContext : context, {
|
|
39
138
|
...options,
|
|
40
139
|
onPayload: async (payload, requestModel) => {
|
|
41
140
|
const transformed = await options?.onPayload?.(payload, requestModel);
|
|
42
|
-
|
|
43
|
-
if (!fitsClassifierRequest(wire, 64 * 1024)) throw new Error("TokenIn classifier payload exceeds 64 KiB");
|
|
44
|
-
return wire;
|
|
141
|
+
return chatBody(transformed === undefined ? payload : transformed, images);
|
|
45
142
|
},
|
|
46
143
|
fetch: async (input, init) => {
|
|
47
144
|
// Reset per attempt: a retry must not inherit pricing from another response.
|
|
48
|
-
|
|
49
|
-
totalCost = undefined;
|
|
145
|
+
billed = undefined;
|
|
50
146
|
booleanConfidence = {};
|
|
51
147
|
scoreLegend = {};
|
|
52
|
-
const
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
if (Object.keys(labels).length === Object.keys(legend).length) scoreLegend[id] = labels;
|
|
71
|
-
}
|
|
72
|
-
}
|
|
73
|
-
}
|
|
74
|
-
}
|
|
75
|
-
} catch { /* Upstream owns malformed-response diagnostics. */ }
|
|
148
|
+
const url = new URL(input instanceof Request ? input.url : String(input));
|
|
149
|
+
url.pathname = url.pathname.replace(/\/systemone$/u, "/chat/completions");
|
|
150
|
+
const response = await requestFetch(url, init);
|
|
151
|
+
if (!response.ok) return response;
|
|
152
|
+
billed = billedCost(response.headers);
|
|
153
|
+
// An unreadable envelope becomes an empty body: upstream reports the missing answers once,
|
|
154
|
+
// instead of this layer failing the attempt and having it retried (and billed) again.
|
|
155
|
+
const body = systemOneBody(await response.json().catch(() => undefined)) ?? {};
|
|
156
|
+
if (record(body.answers)) for (const [id, answer] of Object.entries(body.answers)) {
|
|
157
|
+
if (!record(answer)) continue;
|
|
158
|
+
if (typeof answer.confidence === "number" && Number.isFinite(answer.confidence) && answer.confidence >= 0 && answer.confidence <= 1) booleanConfidence[id] = answer.confidence;
|
|
159
|
+
const legend = answer.legend;
|
|
160
|
+
if (Array.isArray(legend) && legend.every((value) => typeof value === "string")) scoreLegend[id] = legend;
|
|
161
|
+
else if (record(legend)) {
|
|
162
|
+
const labels: Record<string, string> = {};
|
|
163
|
+
for (const [key, label] of Object.entries(legend)) if (/^\d+$/.test(key) && typeof label === "string") labels[key] = label;
|
|
164
|
+
if (Object.keys(labels).length === Object.keys(legend).length) scoreLegend[id] = labels;
|
|
165
|
+
}
|
|
76
166
|
}
|
|
77
|
-
return
|
|
167
|
+
return new Response(JSON.stringify(body), { status: 200, headers: { "content-type": "application/json" } });
|
|
78
168
|
},
|
|
79
169
|
});
|
|
80
170
|
for (const [id, confidence] of Object.entries(booleanConfidence)) {
|
|
81
171
|
const answer = result.answers[id];
|
|
82
172
|
if (answer?.type === "bool") {
|
|
83
|
-
const reported = { ...answer, confidence };
|
|
173
|
+
const reported = { ...answer, confidence }; // not a fresh literal: the public type has no `confidence`
|
|
84
174
|
result.answers[id] = reported;
|
|
85
175
|
}
|
|
86
176
|
}
|
|
@@ -92,14 +182,10 @@ export const classifyTokenIn: typeof classifySystemOne = async (model, context,
|
|
|
92
182
|
}
|
|
93
183
|
}
|
|
94
184
|
if (result.usage) {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
result.unpricedUsage = counts;
|
|
99
|
-
delete result.usage;
|
|
100
|
-
}
|
|
185
|
+
const { cost: _catalogEstimate, ...counts } = result.usage;
|
|
186
|
+
result.unpricedUsage = counts;
|
|
187
|
+
delete result.usage;
|
|
101
188
|
}
|
|
102
|
-
if (
|
|
103
|
-
else if (!result.usage && cost) result.reportedCost = cost.total;
|
|
189
|
+
if (billed !== undefined) result.reportedCost = billed;
|
|
104
190
|
return result;
|
|
105
191
|
};
|
|
@@ -130,6 +130,27 @@ describe("readJevAdvisoryConfig", () => {
|
|
|
130
130
|
minConfidence: 0.8,
|
|
131
131
|
});
|
|
132
132
|
});
|
|
133
|
+
|
|
134
|
+
it("asks the shared Jev model for tool search, on by default, unless the route names its own model", () => {
|
|
135
|
+
state.settingsPath = join(tmpdir(), "missing-settings.json");
|
|
136
|
+
const defaults = readJevAdvisoryConfig(state.settingsPath);
|
|
137
|
+
expect(defaults.routes.toolSearch).toMatchObject({ enabled: true });
|
|
138
|
+
expect(defaults.routes.toolSearch.model).toBeUndefined();
|
|
139
|
+
expect(jevConnection(defaults, defaults.routes.toolSearch)).toMatchObject({ provider: "tokenin", model: "jev-1.13" });
|
|
140
|
+
|
|
141
|
+
write({ jevAdvisory: { model: "jev-9", baseUrl: "https://custom/v1", routes: { toolSearch: { timeoutMs: 2_000 } } } });
|
|
142
|
+
const config = readJevAdvisoryConfig(state.settingsPath);
|
|
143
|
+
expect(jevConnection(config, config.routes.toolSearch)).toMatchObject({ model: "jev-9", baseUrl: "https://custom/v1", timeoutMs: 2_000 });
|
|
144
|
+
|
|
145
|
+
// A route with its own model never inherits the shared endpoint override.
|
|
146
|
+
write({ jevAdvisory: { baseUrl: "https://custom/v1", routes: { toolSearch: { model: "cloudflare/clef-flash" } } } });
|
|
147
|
+
const own = readJevAdvisoryConfig(state.settingsPath);
|
|
148
|
+
expect(jevConnection(own, own.routes.toolSearch)).toMatchObject({ provider: "tokenin", model: "cloudflare/clef-flash", baseUrl: undefined });
|
|
149
|
+
|
|
150
|
+
write({ jevAdvisory: { routes: { toolSearch: { model: "cloudflare/clef", provider: "other", enabled: false } } } });
|
|
151
|
+
const custom = readJevAdvisoryConfig(state.settingsPath).routes.toolSearch;
|
|
152
|
+
expect(custom).toMatchObject({ enabled: false, model: "cloudflare/clef", provider: "other" });
|
|
153
|
+
});
|
|
133
154
|
});
|
|
134
155
|
|
|
135
156
|
describe("buildJevPayload", () => {
|