@selesai/code 0.14.0 → 0.14.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/config.d.ts CHANGED
@@ -187,6 +187,15 @@ export declare function seedMissingSubagentSettings(destPath: string, bundledDef
187
187
  * rewritten.
188
188
  */
189
189
  export declare function seedMissingGraftSettings(destPath: string, bundledDefaultsDir?: string): boolean;
190
+ /**
191
+ * Add the bundled `defaultTools` selection (which enables `codemode`) to an existing user
192
+ * settings file that does not set `defaultTools` at all. Settings files created before the
193
+ * bundled default existed would otherwise never get it. A user-provided `defaultTools`, even an
194
+ * empty one or one with only `-name` entries, is never touched. The file is modified with a
195
+ * minimal textual insertion, so other settings keep their formatting. Returns true when the file
196
+ * was rewritten.
197
+ */
198
+ export declare function seedMissingDefaultToolsSettings(destPath: string, bundledDefaultsDir?: string): boolean;
190
199
  /**
191
200
  * Bundled extensions load directly from the installed package. Do not copy them
192
201
  * into the user's agent dir, otherwise startup discovers both copies and tool
@@ -213,7 +222,7 @@ export declare function seedDefaultSkills(agentDir: string, bundledSkillsDir?: s
213
222
  export declare function seedDefaultThemes(agentDir: string, bundledThemesDir?: string): string[];
214
223
  /**
215
224
  * Run bootstrap for the agent dir: ensure directories, seed a missing
216
- * settings.json, and add missing bundled subagent defaults. Idempotent — safe
225
+ * settings.json, and add missing bundled subagent, graft, and default-tools settings. Idempotent — safe
217
226
  * on every startup.
218
227
  * Bundled models load package-locally; copying them would expose internal
219
228
  * providers as user config. Bundled extensions also stay package-local.
package/dist/config.js CHANGED
@@ -801,6 +801,43 @@ export function seedMissingGraftSettings(destPath, bundledDefaultsDir = getBundl
801
801
  return false;
802
802
  }
803
803
  }
804
+ /**
805
+ * Add the bundled `defaultTools` selection (which enables `codemode`) to an existing user
806
+ * settings file that does not set `defaultTools` at all. Settings files created before the
807
+ * bundled default existed would otherwise never get it. A user-provided `defaultTools`, even an
808
+ * empty one or one with only `-name` entries, is never touched. The file is modified with a
809
+ * minimal textual insertion, so other settings keep their formatting. Returns true when the file
810
+ * was rewritten.
811
+ */
812
+ export function seedMissingDefaultToolsSettings(destPath, bundledDefaultsDir = getBundledDefaultsDir()) {
813
+ if (!existsSync(destPath))
814
+ return false;
815
+ try {
816
+ const raw = readFileSync(destPath, "utf-8");
817
+ const settings = JSON.parse(raw);
818
+ const defaults = JSON.parse(readFileSync(join(bundledDefaultsDir, "settings.json"), "utf-8"));
819
+ if (typeof settings !== "object" ||
820
+ settings === null ||
821
+ Array.isArray(settings) ||
822
+ Object.hasOwn(settings, "defaultTools") ||
823
+ typeof defaults !== "object" ||
824
+ defaults === null ||
825
+ Array.isArray(defaults)) {
826
+ return false;
827
+ }
828
+ const defaultTools = defaults.defaultTools;
829
+ if (!Array.isArray(defaultTools) || !defaultTools.every((entry) => typeof entry === "string"))
830
+ return false;
831
+ const injected = injectRootObjectKey(raw, "defaultTools", defaultTools);
832
+ if (injected === undefined)
833
+ return false;
834
+ writeFileSync(destPath, injected);
835
+ return true;
836
+ }
837
+ catch {
838
+ return false;
839
+ }
840
+ }
804
841
  /**
805
842
  * Recursively copy a bundled directory tree into the user's agent dir,
806
843
  * overwriting any existing file so the bundled copy stays authoritative.
@@ -902,7 +939,7 @@ export function seedDefaultThemes(agentDir, bundledThemesDir = getBundledThemesD
902
939
  }
903
940
  /**
904
941
  * Run bootstrap for the agent dir: ensure directories, seed a missing
905
- * settings.json, and add missing bundled subagent defaults. Idempotent — safe
942
+ * settings.json, and add missing bundled subagent, graft, and default-tools settings. Idempotent — safe
906
943
  * on every startup.
907
944
  * Bundled models load package-locally; copying them would expose internal
908
945
  * providers as user config. Bundled extensions also stay package-local.
@@ -913,6 +950,7 @@ export function bootstrapAgentDir(agentDir = getAgentDir()) {
913
950
  seedDefaultConfigFile(settingsPath, "settings.json");
914
951
  seedMissingSubagentSettings(settingsPath);
915
952
  seedMissingGraftSettings(settingsPath);
953
+ seedMissingDefaultToolsSettings(settingsPath);
916
954
  seedDefaultExtensions(agentDir);
917
955
  seedDefaultSkills(agentDir);
918
956
  seedDefaultThemes(agentDir);
@@ -389,6 +389,8 @@ export class DefaultResourceLoader {
389
389
  if (this.loaded) {
390
390
  clearExtensionCache();
391
391
  }
392
+ // The filter closes over the previous extension runtime (now stale); the reloaded extensions re-register theirs.
393
+ this.skillsIndexFilter = undefined;
392
394
  let preTrustExtensions;
393
395
  if (options?.resolveProjectTrust) {
394
396
  preTrustExtensions = await this.loadProjectTrustExtensions();
@@ -7,12 +7,14 @@ import { getCodemodeWorkerSpecifier, getQuickJSWasmPath } from "../../config.js"
7
7
  import { formatSize } from "../../core/tools/truncate.js";
8
8
  import { combineUsage } from "../../core/usage-totals.js";
9
9
  import { writeOutputFile } from "../../utils/output-files.js";
10
- import { Bm25Ranker, createToolSearchDocument, DEFAULT_TOOL_SEARCH_LIMIT } from "../tool-search/tool.js";
10
+ import { createToolSearchDocument, DEFAULT_TOOL_SEARCH_LIMIT, getToolRanker } from "../tool-search/tool.js";
11
11
  import { CODEMODE_DOCS_PATH, CODEMODE_STORE_ENTRY_TYPE, getCodemodeCallableTools, toCodemodeDeclaration, } from "./tool.js";
12
12
  const ARGS_PREVIEW_CHARS = 200;
13
13
  const ERROR_PREVIEW_CHARS = 500;
14
14
  /** `models.classify()` and `models.generateImages()` calls one script may have in flight; `Promise.all` over many items queues the rest. */
15
15
  const MAX_CONCURRENT_MODEL_CALLS = 4;
16
+ /** `searchTools()` calls per script that may use a model-backed ranker; later calls rank locally. */
17
+ const MAX_SESSION_SEARCHES_PER_SCRIPT = 8;
16
18
  /**
17
19
  * Heap limit for the QuickJS VM. The worker shares pi's process, so without a limit a runaway
18
20
  * script can grow to wasm32's 4 GiB and take the session down. Overruns throw
@@ -413,7 +415,7 @@ export async function executeCodemode(toolCallId, input, signal, onUpdate, ctx,
413
415
  const sandbox = new CodemodeSandbox({
414
416
  tools: sandboxTools,
415
417
  globals: [
416
- ...createDiscoveryGlobals(callable, samples, options),
418
+ ...createDiscoveryGlobals(callable, samples, options, ctx),
417
419
  ...(options.models && ctx
418
420
  ? createModelGlobals(ctx.modelRegistry, toolCallId, calls, publish, addModelUsage, addGeneratedImages)
419
421
  : []),
@@ -485,14 +487,16 @@ function isNamespaceName(namespace, query) {
485
487
  * `searchTools()`, `describeTool()`, and `describeNamespace()`: ranked search and lookup over the
486
488
  * script's nested tools and their namespaces.
487
489
  */
488
- function createDiscoveryGlobals(tools, samples, options) {
489
- const ranker = new Bm25Ranker();
490
+ function createDiscoveryGlobals(tools, samples, options, ctx) {
491
+ // A script may loop over searchTools(). Past this many searches the session is withheld from the
492
+ // ranker, so a model-backed ranker falls back to its local ranking instead of billing per iteration.
493
+ let sessionSearches = 0;
490
494
  const entry = (name) => ({ name: toCodemodeIdentifier(name), description: samples.get(name) ?? "" });
491
495
  return [
492
496
  {
493
497
  name: "searchTools",
494
498
  spread: true,
495
- execute: (args) => {
499
+ execute: async (args, { signal }) => {
496
500
  const [query, searchOptions] = args;
497
501
  if (typeof query !== "string")
498
502
  throw new Error("searchTools() expects a query string");
@@ -510,7 +514,15 @@ function createDiscoveryGlobals(tools, samples, options) {
510
514
  return [];
511
515
  return [createToolSearchDocument(tool, toolNamespace)];
512
516
  });
513
- return ranker.rank(query, documents, limit).map((match) => entry(match.name));
517
+ const session = ctx && sessionSearches < MAX_SESSION_SEARCHES_PER_SCRIPT ? ctx : undefined;
518
+ if (session)
519
+ sessionSearches += 1;
520
+ const ranked = await getToolRanker().rank(query, documents, limit, { signal, ctx: session });
521
+ const known = new Set(documents.map((document) => document.name));
522
+ return ranked
523
+ .filter((match) => known.has(match.name))
524
+ .slice(0, limit)
525
+ .map((match) => entry(match.name));
514
526
  },
515
527
  },
516
528
  {
@@ -1,6 +1,6 @@
1
1
  import { expect, it, vi } from "vitest";
2
2
  import type { ClassifierContext } from "@earendil-works/pi-ai";
3
- import { classifyTokenIn, TOKENIN_JEV_CLASSIFIER } from "./classifier.ts";
3
+ import { classifyTokenIn, MAX_IMAGE_BYTES, TOKENIN_CLASSIFIERS, TOKENIN_JEV_CLASSIFIER } from "./classifier.ts";
4
4
  import { fitsClassifierRequest, toClassifierContext } from "./classifier-context.ts";
5
5
  import { askJevAnswers } from "./decisions.ts";
6
6
 
@@ -9,38 +9,57 @@ const context: ClassifierContext = { state: { public: "protocol fixture" }, ques
9
9
  risk: { type: "score", instructions: "Rate", criteria: ["low", "high"] },
10
10
  ready: { type: "bool", instructions: "Ready?", criteria: { true: "Ready", false: "Not ready" } },
11
11
  } };
12
- const cost = { input: .001, output: .002, cacheRead: 0, cacheWrite: 0, total: .003 };
12
+ const billed = .0000158;
13
13
  const body = () => ({ answers: {
14
14
  pick: { type: "choice", choice: "yes", probabilities: { yes: .9, no: .1 }, confidence: .9 },
15
15
  risk: { type: "score", score: 1.4, confidence: .8, legend: { 0: "low", 1: "high" } },
16
16
  ready: { type: "noul", noul: .87, confidence: .93 },
17
- }, usage: { input_tokens: 10, output_tokens: 7, cost } });
18
- function transport(response: unknown) {
19
- return vi.fn(async (_input: string | URL | Request, _init?: RequestInit) => new Response(JSON.stringify(response), { headers: { "content-type": "application/json" } }));
17
+ }, usage: { input_tokens: 10, output_tokens: 7, cost: .0000122 } });
18
+ /** The gateway's reply: the System One body is a chat completion's message content, the billed cost a header. */
19
+ function transport(response: unknown, headers: Record<string, string> = { "x-litellm-response-cost": String(billed) }) {
20
+ return vi.fn(async (_input: string | URL | Request, _init?: RequestInit) =>
21
+ new Response(JSON.stringify({ choices: [{ message: { role: "assistant", content: JSON.stringify(response) } }] }), { headers: { "content-type": "application/json", ...headers } }));
20
22
  }
21
- it("uses the exact native endpoint, bearer key and TypeSafe wire shape, including bool -> noul", async () => {
23
+ it("posts the System One request as a chat message to the gateway, with bool -> noul, and reports the billed cost", async () => {
22
24
  const fetch = transport(body());
23
25
  const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch });
24
26
  expect(result.stopReason).toBe("stop");
25
- expect(String(fetch.mock.calls[0][0])).toBe("https://lite.andlet.me/v1/systemone");
27
+ expect(String(fetch.mock.calls[0][0])).toBe("https://lite.andlet.me/v1/chat/completions");
26
28
  const init = fetch.mock.calls[0][1]!;
27
29
  expect(new Headers(init.headers).get("authorization")).toBe("Bearer fixture");
28
- expect(JSON.parse(String(init.body))).toEqual({ model: "jev-1.13", state: JSON.stringify(context.state), questions: { ...context.questions, ready: { ...context.questions.ready, type: "noul" } } });
30
+ const sent = JSON.parse(String(init.body));
31
+ expect(sent).toEqual({ model: "jev-1.13", messages: [{ role: "user", content: expect.any(String) }] });
32
+ expect(JSON.parse(sent.messages[0].content)).toEqual({ state: JSON.stringify(context.state), questions: { ...context.questions, ready: { ...context.questions.ready, type: "noul" } } });
29
33
  expect(result.answers.ready).toEqual({ type: "bool", probability: .87, confidence: .93 });
30
34
  expect(result.answers.risk).toMatchObject({ legend: { 0: "low", 1: "high" } });
31
- expect(result.usage).toMatchObject({ input: 10, output: 7, totalTokens: 17, cost });
35
+ // The header is the only cost; the body's own usage.cost is the provider's raw charge.
36
+ expect(result.reportedCost).toBe(billed);
37
+ expect(result.usage).toBeUndefined();
38
+ expect(result.unpricedUsage).toMatchObject({ input: 10, output: 7, totalTokens: 17 });
32
39
  });
33
- it("does not represent missing or invalid service pricing as a free catalog estimate", async () => {
34
- for (const usage of [{ input_tokens: 10, output_tokens: 7 }, { input_tokens: 10, output_tokens: 7, cost: { ...cost, total: -1 } }]) {
35
- const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport({ ...body(), usage }) });
40
+ it("reports no cost when the gateway sends none or an invalid one, and 0 for a cache hit", async () => {
41
+ for (const headers of [{}, { "x-litellm-response-cost": "-1" }, { "x-litellm-response-cost": "free" }, { "x-litellm-response-cost": "" }]) {
42
+ const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport(body(), headers) });
43
+ expect(result.stopReason).toBe("stop");
44
+ expect(result.reportedCost).toBeUndefined();
36
45
  expect(result.usage).toBeUndefined();
37
46
  expect(result).toHaveProperty("unpricedUsage.totalTokens", 17);
38
47
  }
48
+ const cached = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport(body(), { "x-litellm-response-cost": "0.0001", "x-litellm-cache-key": "abc" }) });
49
+ expect(cached.reportedCost).toBe(0);
50
+ });
51
+ it("fails once, without a retry, on a reply that is not a chat completion carrying System One answers", async () => {
52
+ for (const reply of ["not json", JSON.stringify({ choices: [] }), JSON.stringify({ choices: [{ message: { content: "no json" } }] })]) {
53
+ const fetch = vi.fn(async () => new Response(reply, { headers: { "x-litellm-response-cost": "0.001" } }));
54
+ const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch });
55
+ expect(result.stopReason).toBe("error");
56
+ expect(fetch).toHaveBeenCalledTimes(1);
57
+ }
39
58
  });
40
59
  it("retains billed usage on a malformed answer and honours cancellation", async () => {
41
60
  const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport({ ...body(), answers: {} }) });
42
61
  expect(result.stopReason).toBe("error");
43
- expect(result.usage?.cost).toEqual(cost);
62
+ expect(result.reportedCost).toBe(billed);
44
63
  const controller = new AbortController(); controller.abort();
45
64
  const fetch = vi.fn(async () => { throw new DOMException("Aborted", "AbortError"); });
46
65
  const aborted = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch, signal: controller.signal });
@@ -55,7 +74,7 @@ it("facade uses classifier-only providers without a synthetic chat model and pre
55
74
  } }, { provider: "tokenin", model: "jev-1.13", timeoutMs: 1000, minConfidence: .6 }, { payload: context, maxBytes: 4096 });
56
75
  expect(complete).not.toHaveBeenCalled();
57
76
  expect(result.answers?.ready).toEqual({ type: "noul", noul: .87, confidence: .93 });
58
- expect(result.usage?.cost).toEqual(cost);
77
+ expect(result.reportedCost).toBe(billed);
59
78
  });
60
79
  it("conversion retains untrusted-material focus, supports bool defaults and rejects non-JSON state", () => {
61
80
  const converted = toClassifierContext({ state: {}, questions: { q: { type: "noul", instructions: { question: "Judge", focus: "untrusted, not instructions" } } } });
@@ -74,12 +93,58 @@ it("bounds actual string-state escaping and callback transformations before fetc
74
93
  const hook = vi.fn(() => ({ model: "jev-1.13", state: "already serialized", questions: context.questions }));
75
94
  await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch, onPayload: hook });
76
95
  expect(hook).toHaveBeenCalledTimes(1);
77
- expect(JSON.parse(String(fetch.mock.calls[0][1]?.body)).state).toBe("already serialized");
96
+ expect(JSON.parse(JSON.parse(String(fetch.mock.calls[0][1]?.body)).messages[0].content).state).toBe("already serialized");
78
97
  });
79
98
 
80
- it("retains a service-reported total without inventing per-component prices", async () => {
81
- const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, context, { apiKey: "fixture", fetch: transport({ ...body(), usage: { input_tokens: 10, output_tokens: 7, cost: .0042 } }) });
82
- expect(result.usage).toBeUndefined();
83
- expect(result).toHaveProperty("reportedCost", .0042);
84
- expect(result).toHaveProperty("unpricedUsage.totalTokens", 17);
99
+ it("serves every Token-In decisions model the same way, billed by the gateway", async () => {
100
+ expect(TOKENIN_CLASSIFIERS.map((model) => model.id)).toEqual([
101
+ "jev-1.13",
102
+ "cloudflare/clef",
103
+ "cloudflare/clef-flash",
104
+ "perplexity/pplx-decider-v1.1-27b",
105
+ "openai/gpt-6-luna-decisions",
106
+ ]);
107
+ expect(new Set(TOKENIN_CLASSIFIERS.map((model) => model.id)).size).toBe(TOKENIN_CLASSIFIERS.length);
108
+ for (const model of TOKENIN_CLASSIFIERS) {
109
+ const fetch = transport(body());
110
+ const result = await classifyTokenIn(model, context, { apiKey: "fixture", fetch });
111
+ expect(result.stopReason, model.id).toBe("stop");
112
+ expect(String(fetch.mock.calls[0][0]), model.id).toBe("https://lite.andlet.me/v1/chat/completions");
113
+ expect(JSON.parse(String(fetch.mock.calls[0][1]!.body)), model.id).toMatchObject({ model: model.id });
114
+ expect(result.answers.ready, model.id).toMatchObject({ type: "bool", probability: .87 });
115
+ expect(result.reportedCost, model.id).toBe(billed);
116
+ expect(model.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
117
+ }
118
+ });
119
+
120
+ const image = { type: "image" as const, data: "QUJD", mimeType: "image/png" };
121
+ it("sends images inside the state array, only to the models that read them", async () => {
122
+ expect(TOKENIN_CLASSIFIERS.filter((model) => model.input.includes("image")).map((model) => model.id)).toEqual([
123
+ "perplexity/pplx-decider-v1.1-27b",
124
+ "openai/gpt-6-luna-decisions",
125
+ ]);
126
+ for (const model of TOKENIN_CLASSIFIERS.filter((m) => m.input.includes("image"))) {
127
+ const fetch = transport(body());
128
+ const result = await classifyTokenIn(model, { ...context, images: [image, image] }, { apiKey: "fixture", fetch });
129
+ expect(result.stopReason, model.id).toBe("stop");
130
+ const request = JSON.parse(JSON.parse(String(fetch.mock.calls[0][1]!.body)).messages[0].content);
131
+ expect(request.images, model.id).toBeUndefined();
132
+ expect(request.state, model.id).toEqual([
133
+ JSON.stringify(context.state),
134
+ { type: "image_url", image_url: { url: "data:image/png;base64,QUJD" } },
135
+ { type: "image_url", image_url: { url: "data:image/png;base64,QUJD" } },
136
+ ]);
137
+ }
138
+ // A text-only model is not sent the image: the call fails before any request.
139
+ const fetch = transport(body());
140
+ const result = await classifyTokenIn(TOKENIN_JEV_CLASSIFIER, { ...context, images: [image] }, { apiKey: "fixture", fetch });
141
+ expect(result.stopReason).toBe("error");
142
+ expect(fetch).not.toHaveBeenCalled();
143
+ });
144
+ it("refuses images over the size cap before sending anything", async () => {
145
+ const fetch = transport(body());
146
+ const lune = TOKENIN_CLASSIFIERS[4];
147
+ const result = await classifyTokenIn(lune, { ...context, images: [{ ...image, data: "A".repeat(MAX_IMAGE_BYTES + 1) }] }, { apiKey: "fixture", fetch });
148
+ expect(result.stopReason).toBe("error");
149
+ expect(fetch).not.toHaveBeenCalled();
85
150
  });
@@ -1,86 +1,176 @@
1
- /** TokenIn's native System One classifier. Protocol parsing/retries remain upstream-owned. */
2
- import type { ClassifierModel, ClassifierResult, Usage } from "@earendil-works/pi-ai";
1
+ /**
2
+ * TokenIn's decisions classifiers: Jev and the gateway's other decisions models (Clef, Perplexity
3
+ * Decider, GPT-6 Luna). They all take the same System One request and answer in the same format, so
4
+ * one implementation serves every model; the model id in the request body picks the deployment.
5
+ *
6
+ * Transport: the gateway serves them as chat deployments, so a request is `POST {baseUrl}/chat/completions`
7
+ * with one user message whose content is the JSON System One request `{state, questions}`, and the
8
+ * answer is the System One body (`{answers, usage}`) as the reply's message content. (Its
9
+ * `/v1/systemone` route is not deployed.) Protocol parsing and retries stay upstream-owned: this runs
10
+ * upstream's System One classifier and only changes the envelope through its `onPayload` and `fetch` hooks.
11
+ *
12
+ * Images travel inside the request's `state` array as `{ type: "image_url", image_url: { url: "data:..." } }`
13
+ * parts; the gateway rejects a top-level `images`. Only models that really read them declare image input.
14
+ *
15
+ * Billing: the gateway bills more than the provider's raw charge and puts what the account was charged
16
+ * in the `x-litellm-response-cost` header (0 for a response-cache hit). That is the only cost reported.
17
+ * The `usage.cost` in the body is the provider's raw charge (and 0 for some models), and the catalog
18
+ * price below is a placeholder, so neither is ever presented as cost.
19
+ */
20
+ import type {
21
+ ClassifierApi,
22
+ ClassifierContext,
23
+ ClassifierModel,
24
+ ClassifierOptions,
25
+ ClassifierResult,
26
+ ImageContent,
27
+ Usage,
28
+ } from "@earendil-works/pi-ai";
3
29
  import { classify as classifySystemOne } from "@earendil-works/pi-ai/api/typesafe-system-one";
4
30
  import { fitsClassifierRequest, tokenInWirePayload } from "./classifier-context.ts";
5
31
 
6
- export const TOKENIN_JEV_CLASSIFIER: ClassifierModel<"typesafe-system-one"> = {
7
- type: "classifier", provider: "tokenin", id: "jev-1.13", name: "Jev 1.13 (service-priced)",
8
- api: "typesafe-system-one", baseUrl: "https://lite.andlet.me/v1", input: ["text"], contextWindow: 64000,
9
- // Required catalog fields, NOT a free-price assertion. Never return catalog-computed cost:
10
- // expose usage.cost only when the service explicitly reports its monetary breakdown.
11
- cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
12
- };
32
+ /**
33
+ * `cost` is a required catalog field, NOT a price. Context windows are the ones pi-ai's OpenRouter
34
+ * catalog lists for the same models. `vision` models were checked against the live gateway with
35
+ * solid-colour images: Perplexity Decider and GPT-6 Luna answered every colour correctly, while Clef and
36
+ * Clef Flash accept an image but answer as if they had not seen it (their token counts follow the
37
+ * base64 text), so they are text-only here even though the OpenRouter catalog lists image input.
38
+ */
39
+ function tokenInDecisionsModel(
40
+ id: string,
41
+ name: string,
42
+ contextWindow: number,
43
+ vision = false,
44
+ ): ClassifierModel<"typesafe-system-one"> {
45
+ return {
46
+ type: "classifier", provider: "tokenin", id, name: `${name} (service-priced)`,
47
+ api: "typesafe-system-one", baseUrl: "https://lite.andlet.me/v1", input: vision ? ["text", "image"] : ["text"], contextWindow,
48
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
49
+ };
50
+ }
51
+
52
+ export const TOKENIN_JEV_CLASSIFIER = tokenInDecisionsModel("jev-1.13", "Jev 1.13", 64000);
53
+
54
+ /** Cloudflare's Clef Flash: 65,536 tokens of context. */
55
+ export const TOKENIN_CLEF_FLASH_CLASSIFIER = tokenInDecisionsModel("cloudflare/clef-flash", "Clef Flash", 65536);
56
+
57
+ /** Every decisions model Token-In serves, Jev first. */
58
+ export const TOKENIN_CLASSIFIERS: readonly ClassifierModel<"typesafe-system-one">[] = [
59
+ TOKENIN_JEV_CLASSIFIER,
60
+ tokenInDecisionsModel("cloudflare/clef", "Clef", 65536),
61
+ TOKENIN_CLEF_FLASH_CLASSIFIER,
62
+ tokenInDecisionsModel("perplexity/pplx-decider-v1.1-27b", "Perplexity Decider 1.1 27B", 262144, true),
63
+ tokenInDecisionsModel("openai/gpt-6-luna-decisions", "GPT-6 Luna Decisions", 1050000, true),
64
+ ];
65
+
66
+ const COST_HEADER = "x-litellm-response-cost";
67
+ const CACHE_HIT_HEADER = "x-litellm-cache-key";
68
+ /** The request without its images: state and questions, as before. */
69
+ const MAX_TEXT_BYTES = 64 * 1024;
70
+ /** ponytail: 2 MiB of base64 in total, the most checked against the live gateway; raise once larger ones are measured. */
71
+ export const MAX_IMAGE_BYTES = 2 * 1024 * 1024;
13
72
 
14
73
  function record(value: unknown): value is Record<string, unknown> {
15
74
  return typeof value === "object" && value !== null && !Array.isArray(value);
16
75
  }
17
- function reportedCost(value: unknown): Usage["cost"] | undefined {
18
- if (!record(value)) return undefined;
19
- const { input, output, cacheRead, cacheWrite, total } = value;
20
- const money = (n: unknown): n is number => typeof n === "number" && Number.isFinite(n) && n >= 0;
21
- return money(input) && money(output) && money(cacheRead) && money(cacheWrite) && money(total)
22
- ? { input, output, cacheRead, cacheWrite, total } : undefined;
23
- }
24
76
 
25
77
  export interface ServicePricedClassifierResult extends ClassifierResult {
26
- /** Counts without a monetary claim when billing data is absent/unrecognised. */
78
+ /** Token counts without a monetary claim: the catalog has no price for these calls. */
27
79
  unpricedUsage?: Omit<Usage, "cost">;
28
- /** Service-reported total when no per-component breakdown is supplied. */
80
+ /** What the gateway billed for the call; absent when it did not say. */
29
81
  reportedCost?: number;
30
82
  }
31
83
 
32
- export const classifyTokenIn: typeof classifySystemOne = async (model, context, options) => {
33
- let cost: Usage["cost"] | undefined;
34
- let totalCost: number | undefined;
84
+ function imagePart(image: ImageContent): unknown {
85
+ return { type: "image_url", image_url: { url: `data:${image.mimeType};base64,${image.data}` } };
86
+ }
87
+
88
+ /** The System One request upstream built, as the gateway's chat body, with any images added to its `state`. */
89
+ function chatBody(payload: unknown, images: readonly ImageContent[]): unknown {
90
+ const wire = tokenInWirePayload(payload);
91
+ if (!record(wire) || typeof wire.model !== "string") throw new Error("TokenIn classifier requires a model");
92
+ if (!fitsClassifierRequest(wire, MAX_TEXT_BYTES)) throw new Error("TokenIn classifier payload exceeds 64 KiB");
93
+ if (images.reduce((total, image) => total + image.data.length, 0) > MAX_IMAGE_BYTES) {
94
+ throw new Error(`TokenIn classifier images exceed ${MAX_IMAGE_BYTES / 1024 / 1024} MiB`);
95
+ }
96
+ const { model, ...request } = wire;
97
+ const state = images.length > 0 ? [wire.state, ...images.map(imagePart)] : wire.state;
98
+ return { model, messages: [{ role: "user", content: JSON.stringify({ ...request, state }) }] };
99
+ }
100
+
101
+ /** The System One body a chat completion carries as its message content, or undefined. */
102
+ function systemOneBody(completion: unknown): Record<string, unknown> | undefined {
103
+ const choices = record(completion) ? completion.choices : undefined;
104
+ const message = Array.isArray(choices) && record(choices[0]) ? choices[0].message : undefined;
105
+ const content = record(message) ? message.content : undefined;
106
+ if (typeof content !== "string") return undefined;
107
+ try {
108
+ const body: unknown = JSON.parse(content);
109
+ return record(body) ? body : undefined;
110
+ } catch {
111
+ return undefined;
112
+ }
113
+ }
114
+
115
+ /** The amount the account was charged for this response, or undefined when the gateway did not say. */
116
+ function billedCost(headers: Headers): number | undefined {
117
+ if ((headers.get(CACHE_HIT_HEADER) ?? "").trim() !== "") return 0;
118
+ const reported = headers.get(COST_HEADER);
119
+ if (reported === null || reported.trim() === "") return undefined;
120
+ const cost = Number(reported);
121
+ return Number.isFinite(cost) && cost >= 0 ? cost : undefined;
122
+ }
123
+
124
+ export const classifyTokenIn = async (
125
+ model: ClassifierModel<ClassifierApi>,
126
+ context: ClassifierContext,
127
+ options?: ClassifierOptions,
128
+ ): Promise<ServicePricedClassifierResult> => {
129
+ let billed: number | undefined;
35
130
  let booleanConfidence: Record<string, number> = {};
36
131
  let scoreLegend: Record<string, Record<string, string> | string[]> = {};
37
132
  const requestFetch = options?.fetch ?? globalThis.fetch;
38
- const result: ServicePricedClassifierResult = await classifySystemOne(model, context, {
133
+ // Upstream's System One client rejects images, so a model that reads them gets them through `state`
134
+ // instead. For a text-only model they stay in the context and upstream reports them as unsupported.
135
+ const images = model.input.includes("image") ? (context.images ?? []) : [];
136
+ const { images: _inState, ...textContext } = context;
137
+ const result: ServicePricedClassifierResult = await classifySystemOne(model, images.length > 0 ? textContext : context, {
39
138
  ...options,
40
139
  onPayload: async (payload, requestModel) => {
41
140
  const transformed = await options?.onPayload?.(payload, requestModel);
42
- const wire = tokenInWirePayload(transformed === undefined ? payload : transformed);
43
- if (!fitsClassifierRequest(wire, 64 * 1024)) throw new Error("TokenIn classifier payload exceeds 64 KiB");
44
- return wire;
141
+ return chatBody(transformed === undefined ? payload : transformed, images);
45
142
  },
46
143
  fetch: async (input, init) => {
47
144
  // Reset per attempt: a retry must not inherit pricing from another response.
48
- cost = undefined;
49
- totalCost = undefined;
145
+ billed = undefined;
50
146
  booleanConfidence = {};
51
147
  scoreLegend = {};
52
- const response = await requestFetch(input, init);
53
- if (response.ok) {
54
- try {
55
- const body: unknown = await response.clone().json();
56
- if (record(body)) {
57
- if (record(body.usage)) {
58
- cost = reportedCost(body.usage.cost);
59
- const total = body.usage.cost;
60
- if (typeof total === "number" && Number.isFinite(total) && total >= 0) totalCost = total;
61
- }
62
- if (record(body.answers)) for (const [id, answer] of Object.entries(body.answers)) {
63
- if (record(answer)) {
64
- if (typeof answer.confidence === "number" && Number.isFinite(answer.confidence) && answer.confidence >= 0 && answer.confidence <= 1) booleanConfidence[id] = answer.confidence;
65
- const legend = answer.legend;
66
- if (Array.isArray(legend) && legend.every((value) => typeof value === "string")) scoreLegend[id] = legend;
67
- else if (record(legend)) {
68
- const labels: Record<string, string> = {};
69
- for (const [key, label] of Object.entries(legend)) if (/^\d+$/.test(key) && typeof label === "string") labels[key] = label;
70
- if (Object.keys(labels).length === Object.keys(legend).length) scoreLegend[id] = labels;
71
- }
72
- }
73
- }
74
- }
75
- } catch { /* Upstream owns malformed-response diagnostics. */ }
148
+ const url = new URL(input instanceof Request ? input.url : String(input));
149
+ url.pathname = url.pathname.replace(/\/systemone$/u, "/chat/completions");
150
+ const response = await requestFetch(url, init);
151
+ if (!response.ok) return response;
152
+ billed = billedCost(response.headers);
153
+ // An unreadable envelope becomes an empty body: upstream reports the missing answers once,
154
+ // instead of this layer failing the attempt and having it retried (and billed) again.
155
+ const body = systemOneBody(await response.json().catch(() => undefined)) ?? {};
156
+ if (record(body.answers)) for (const [id, answer] of Object.entries(body.answers)) {
157
+ if (!record(answer)) continue;
158
+ if (typeof answer.confidence === "number" && Number.isFinite(answer.confidence) && answer.confidence >= 0 && answer.confidence <= 1) booleanConfidence[id] = answer.confidence;
159
+ const legend = answer.legend;
160
+ if (Array.isArray(legend) && legend.every((value) => typeof value === "string")) scoreLegend[id] = legend;
161
+ else if (record(legend)) {
162
+ const labels: Record<string, string> = {};
163
+ for (const [key, label] of Object.entries(legend)) if (/^\d+$/.test(key) && typeof label === "string") labels[key] = label;
164
+ if (Object.keys(labels).length === Object.keys(legend).length) scoreLegend[id] = labels;
165
+ }
76
166
  }
77
- return response;
167
+ return new Response(JSON.stringify(body), { status: 200, headers: { "content-type": "application/json" } });
78
168
  },
79
169
  });
80
170
  for (const [id, confidence] of Object.entries(booleanConfidence)) {
81
171
  const answer = result.answers[id];
82
172
  if (answer?.type === "bool") {
83
- const reported = { ...answer, confidence };
173
+ const reported = { ...answer, confidence }; // not a fresh literal: the public type has no `confidence`
84
174
  result.answers[id] = reported;
85
175
  }
86
176
  }
@@ -92,14 +182,10 @@ export const classifyTokenIn: typeof classifySystemOne = async (model, context,
92
182
  }
93
183
  }
94
184
  if (result.usage) {
95
- if (cost) result.usage.cost = cost;
96
- else {
97
- const { cost: _catalogEstimate, ...counts } = result.usage;
98
- result.unpricedUsage = counts;
99
- delete result.usage;
100
- }
185
+ const { cost: _catalogEstimate, ...counts } = result.usage;
186
+ result.unpricedUsage = counts;
187
+ delete result.usage;
101
188
  }
102
- if (totalCost !== undefined) result.reportedCost = totalCost;
103
- else if (!result.usage && cost) result.reportedCost = cost.total;
189
+ if (billed !== undefined) result.reportedCost = billed;
104
190
  return result;
105
191
  };
@@ -130,6 +130,27 @@ describe("readJevAdvisoryConfig", () => {
130
130
  minConfidence: 0.8,
131
131
  });
132
132
  });
133
+
134
+ it("asks the shared Jev model for tool search, on by default, unless the route names its own model", () => {
135
+ state.settingsPath = join(tmpdir(), "missing-settings.json");
136
+ const defaults = readJevAdvisoryConfig(state.settingsPath);
137
+ expect(defaults.routes.toolSearch).toMatchObject({ enabled: true });
138
+ expect(defaults.routes.toolSearch.model).toBeUndefined();
139
+ expect(jevConnection(defaults, defaults.routes.toolSearch)).toMatchObject({ provider: "tokenin", model: "jev-1.13" });
140
+
141
+ write({ jevAdvisory: { model: "jev-9", baseUrl: "https://custom/v1", routes: { toolSearch: { timeoutMs: 2_000 } } } });
142
+ const config = readJevAdvisoryConfig(state.settingsPath);
143
+ expect(jevConnection(config, config.routes.toolSearch)).toMatchObject({ model: "jev-9", baseUrl: "https://custom/v1", timeoutMs: 2_000 });
144
+
145
+ // A route with its own model never inherits the shared endpoint override.
146
+ write({ jevAdvisory: { baseUrl: "https://custom/v1", routes: { toolSearch: { model: "cloudflare/clef-flash" } } } });
147
+ const own = readJevAdvisoryConfig(state.settingsPath);
148
+ expect(jevConnection(own, own.routes.toolSearch)).toMatchObject({ provider: "tokenin", model: "cloudflare/clef-flash", baseUrl: undefined });
149
+
150
+ write({ jevAdvisory: { routes: { toolSearch: { model: "cloudflare/clef", provider: "other", enabled: false } } } });
151
+ const custom = readJevAdvisoryConfig(state.settingsPath).routes.toolSearch;
152
+ expect(custom).toMatchObject({ enabled: false, model: "cloudflare/clef", provider: "other" });
153
+ });
133
154
  });
134
155
 
135
156
  describe("buildJevPayload", () => {