@selesai/code 0.13.29 → 0.13.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/README.md +8 -2
  3. package/dist/core/model-registry.d.ts +13 -1
  4. package/dist/core/model-registry.js +16 -0
  5. package/dist/defaults/settings.json +8 -13
  6. package/dist/extensions/capability-gateway/catalog.ts +2 -2
  7. package/dist/extensions/capability-gateway/index.ts +103 -18
  8. package/dist/extensions/capability-gateway/integration.test.ts +394 -8
  9. package/dist/extensions/capability-gateway/routing.test.ts +415 -0
  10. package/dist/extensions/capability-gateway/routing.ts +221 -0
  11. package/dist/extensions/grep-app/index.ts +10 -0
  12. package/dist/extensions/jev/decisions.test.ts +316 -0
  13. package/dist/extensions/jev/decisions.ts +527 -0
  14. package/dist/extensions/jev/test-support.ts +233 -0
  15. package/dist/extensions/jev-advisory-lifecycle.test.ts +206 -0
  16. package/dist/extensions/jev-advisory-memory.test.ts +191 -0
  17. package/dist/extensions/jev-advisory-recommendations.test.ts +240 -0
  18. package/dist/extensions/jev-advisory-routing.ts +539 -0
  19. package/dist/extensions/package.json +2 -2
  20. package/dist/extensions/pi-hermes-memory/src/memory-search-bridge.ts +40 -0
  21. package/dist/extensions/pi-hermes-memory/src/tools/memory-search-tool.ts +57 -46
  22. package/dist/extensions/pi-hermes-memory/src/tools/memory-tool.ts +17 -0
  23. package/dist/extensions/pi-hermes-memory/src/tools/session-search-tool.ts +10 -0
  24. package/dist/extensions/pi-hermes-memory/src/tools/skill-tool.ts +5 -0
  25. package/dist/extensions/pi-hermes-memory/tests/tools/memory-search-tool.test.ts +25 -0
  26. package/dist/extensions/pi-intercom/index.ts +10 -0
  27. package/dist/extensions/pi-subagents/src/extension/fanout-child.ts +5 -0
  28. package/dist/extensions/pi-subagents/src/extension/index.ts +5 -0
  29. package/dist/extensions/pi-subagents/src/intercom/native-supervisor-channel.ts +10 -0
  30. package/dist/extensions/pi-subagents/src/runs/background/wait-tool.ts +10 -0
  31. package/dist/extensions/pi-web-agent/src/extension.ts +5 -0
  32. package/dist/extensions/question/index.ts +5 -0
  33. package/dist/extensions/tokenin-onboarding.ts +185 -0
  34. package/dist/skills/code-review-and-quality/SKILL.md +396 -0
  35. package/dist/skills/code-simplification/SKILL.md +331 -0
  36. package/dist/skills/incremental-implementation/SKILL.md +249 -0
  37. package/dist/skills/planning-and-task-breakdown/SKILL.md +257 -0
  38. package/dist/skills/references/agent-skills-LICENSE +21 -0
  39. package/dist/skills/references/definition-of-done.md +67 -0
  40. package/dist/skills/references/performance-checklist.md +236 -0
  41. package/dist/skills/references/security-checklist.md +248 -0
  42. package/docs/settings.md +68 -31
  43. package/package.json +3 -3
  44. package/dist/extensions/auto-model.test.ts +0 -438
  45. package/dist/extensions/auto-model.ts +0 -357
@@ -6,13 +6,12 @@ import { searchMemories, getMemoryStats } from '../store/sqlite-memory-store.js'
6
6
  import type { MemoryCategory } from '../types.js';
7
7
  import { createSharedToolResultRenderer } from './shared-output-view.js';
8
8
  import { searchResultView } from './tool-result-views.js';
9
-
10
- interface SearchResult {
11
- success: boolean;
12
- count?: number;
13
- message?: string;
14
- output?: string;
15
- }
9
+ import {
10
+ isMemorySearchRequest,
11
+ MEMORY_SEARCH_REQUEST_EVENT,
12
+ type MemorySearchRequestInput,
13
+ type MemorySearchResponse,
14
+ } from '../memory-search-bridge.js';
16
15
 
17
16
  function mutationTarget(entry: { target: "memory" | "user" | "failure"; project: string | null }): "memory" | "user" | "failure" | "project" {
18
17
  // A project name scopes ordinary memory entries, but project-attributed
@@ -24,7 +23,50 @@ function scopeLabel(project: string | null): string {
24
23
  return project ? `project:${encodeURIComponent(project)}` : "global";
25
24
  }
26
25
 
26
+ /** The read-only implementation shared by the model tool and local extensions. */
27
+ export function searchMemory(dbManager: DatabaseManager, args: MemorySearchRequestInput): MemorySearchResponse {
28
+ const query = args.query;
29
+ const project = args.project;
30
+ const target = args.target;
31
+ const category = args.category as MemoryCategory | undefined;
32
+ const limit = Math.min(args.limit || 10, 20);
33
+
34
+ if (!query || query.trim().length === 0) {
35
+ return { success: false, message: 'query is required' };
36
+ }
37
+
38
+ const stats = getMemoryStats(dbManager);
39
+ if (stats.total === 0) {
40
+ return { success: false, message: 'No memories in extended store yet. Use memory_add to store memories.' };
41
+ }
42
+
43
+ const results = searchMemories(dbManager, query, { project, target, category, limit });
44
+
45
+ if (results.length === 0) {
46
+ return { success: true, count: 0, message: `No memories found matching "${query}". Try a different search term or broader query.` };
47
+ }
48
+
49
+ let output = `Found ${results.length} memories matching "${query}":\n\n`;
50
+
51
+ for (const entry of results) {
52
+ const resultTarget = mutationTarget(entry);
53
+ const projectLabel = `scope=${scopeLabel(entry.project)}`;
54
+ const mutationTargetLabel = `[target=${resultTarget}]`;
55
+ const targetLabel = entry.target === 'user' ? '👤' : entry.target === 'failure' ? '⚠️' : '🧠';
56
+ const categoryLabel = entry.category ? ` [${entry.category}]` : '';
57
+ output += `${targetLabel} ${projectLabel} ${mutationTargetLabel}${categoryLabel} ${entry.content}\n`;
58
+ output += ` Created: ${entry.created} | Last used: ${entry.lastReferenced}\n\n`;
59
+ }
60
+
61
+ return { success: true, count: results.length, output: output.trim() };
62
+ }
63
+
27
64
  export function registerMemorySearchTool(pi: ExtensionAPI, dbManager: DatabaseManager): void {
65
+ pi.events?.on(MEMORY_SEARCH_REQUEST_EVENT, (request: unknown) => {
66
+ if (!isMemorySearchRequest(request)) return;
67
+ request.respond(searchMemory(dbManager, request.input));
68
+ });
69
+
28
70
  pi.registerTool({
29
71
  name: 'memory_search',
30
72
  label: 'Memory Search',
@@ -40,6 +82,11 @@ target="project" returns only project-attributed memory entries (the ones labele
40
82
 
41
83
  Returns matching memory entries with their mutation target, scope, and dates. The displayed target is the value required by memory_replace and memory_remove.`,
42
84
  promptSnippet: 'Search extended memory store (unlimited capacity)',
85
+ ...{ discovery: {
86
+ summary: 'Search the extended durable memory store for relevant entries',
87
+ aliases: ['memory lookup', 'search memories', 'recall memory'],
88
+ category: 'memory',
89
+ } },
43
90
  promptGuidelines: [
44
91
  'Use memory_search when you need context beyond what is in the system prompt.',
45
92
  'Use memory_search to find project-specific memories or user preferences.',
@@ -53,45 +100,9 @@ Returns matching memory entries with their mutation target, scope, and dates. Th
53
100
  category: Type.Optional(StringEnum(['failure', 'correction', 'insight', 'preference', 'convention', 'tool-quirk'] as const, { description: 'Filter by memory category.' })),
54
101
  limit: Type.Optional(Type.Number({ description: 'Maximum results to return (default: 10, max: 20).' })),
55
102
  }),
56
- execute: async (_id: string, args: { query: string; project?: string; target?: 'memory' | 'user' | 'failure' | 'project'; category?: string; limit?: number }) => {
57
- const query = args.query;
58
- const project = args.project;
59
- const target = args.target;
60
- const category = args.category as MemoryCategory | undefined;
61
- const limit = Math.min(args.limit || 10, 20);
62
-
63
- if (!query || query.trim().length === 0) {
64
- const result: SearchResult = { success: false, message: 'query is required' };
65
- return { content: [{ type: 'text' as const, text: result.message! }], details: result };
66
- }
67
-
68
- const stats = getMemoryStats(dbManager);
69
- if (stats.total === 0) {
70
- const result: SearchResult = { success: false, message: 'No memories in extended store yet. Use memory_add to store memories.' };
71
- return { content: [{ type: 'text' as const, text: result.message! }], details: result };
72
- }
73
-
74
- const results = searchMemories(dbManager, query, { project, target, category, limit });
75
-
76
- if (results.length === 0) {
77
- const result: SearchResult = { success: true, count: 0, message: `No memories found matching "${query}". Try a different search term or broader query.` };
78
- return { content: [{ type: 'text' as const, text: result.message! }], details: result };
79
- }
80
-
81
- let output = `Found ${results.length} memories matching "${query}":\n\n`;
82
-
83
- for (const entry of results) {
84
- const target = mutationTarget(entry);
85
- const projectLabel = `scope=${scopeLabel(entry.project)}`;
86
- const mutationTargetLabel = `[target=${target}]`;
87
- const targetLabel = entry.target === 'user' ? '👤' : entry.target === 'failure' ? '⚠️' : '🧠';
88
- const categoryLabel = entry.category ? ` [${entry.category}]` : '';
89
- output += `${targetLabel} ${projectLabel} ${mutationTargetLabel}${categoryLabel} ${entry.content}\n`;
90
- output += ` Created: ${entry.created} | Last used: ${entry.lastReferenced}\n\n`;
91
- }
92
-
93
- const finalResult: SearchResult = { success: true, count: results.length, output: output.trim() };
94
- return { content: [{ type: 'text' as const, text: output.trim() }], details: finalResult };
103
+ execute: async (_id: string, args: { query: string; project?: string; target?: 'memory' | 'user' | 'failure' | 'project'; category?: MemorySearchRequestInput['category']; limit?: number }) => {
104
+ const result = searchMemory(dbManager, args);
105
+ return { content: [{ type: 'text' as const, text: result.output ?? result.message ?? '' }], details: result };
95
106
  },
96
107
  });
97
108
  }
@@ -413,12 +413,14 @@ This action-specific tool accepts only the parameters listed in its schema.`;
413
413
  label: string,
414
414
  description: string,
415
415
  parameters: TSchema,
416
+ discovery: { summary: string; aliases: string[]; category: string },
416
417
  ) => {
417
418
  pi.registerTool({
418
419
  name,
419
420
  label,
420
421
  description,
421
422
  promptSnippet: `${label}: persistent memory that survives across sessions`,
423
+ ...{ discovery },
422
424
  promptGuidelines: [
423
425
  "Use this tool proactively when the user corrects you, shares a preference, or reveals durable environment or project facts.",
424
426
  "Do not use memory tools for temporary task state, TODO items, or session progress.",
@@ -454,6 +456,11 @@ Add one durable entry. The target and content fields are required.`,
454
456
  category: Type.Optional(category),
455
457
  failure_reason: Type.Optional(Type.String({ description: "Why a failure occurred." })),
456
458
  }),
459
+ {
460
+ summary: "Save durable information to persistent memory",
461
+ aliases: ["remember this", "save memory", "store memory"],
462
+ category: "memory",
463
+ },
457
464
  );
458
465
  registerActionTool(
459
466
  "replace",
@@ -467,6 +474,11 @@ Replace one existing entry. The target, old_text, and content fields are require
467
474
  old_text: Type.String({ description: "Substring identifying the entry to replace." }),
468
475
  content: Type.String({ description: "Replacement entry content." }),
469
476
  }),
477
+ {
478
+ summary: "Replace an existing durable memory entry in place",
479
+ aliases: ["update memory", "edit memory"],
480
+ category: "memory",
481
+ },
470
482
  );
471
483
  registerActionTool(
472
484
  "remove",
@@ -479,6 +491,11 @@ Remove one existing entry. The target and old_text fields are required.`,
479
491
  target,
480
492
  old_text: Type.String({ description: "Substring identifying the entry to remove." }),
481
493
  }),
494
+ {
495
+ summary: "Remove a durable memory entry",
496
+ aliases: ["forget memory", "delete memory", "remove memory"],
497
+ category: "memory",
498
+ },
482
499
  );
483
500
  return configureProjectStore;
484
501
  }
@@ -87,6 +87,11 @@ any:
87
87
  exclude:
88
88
  - delta`,
89
89
  promptSnippet: 'Search past session JSONL files for compact source anchors',
90
+ ...{ discovery: {
91
+ summary: 'Search past Selesai sessions for compact source anchors',
92
+ aliases: ['past sessions', 'session history anchors'],
93
+ category: 'memory',
94
+ } },
90
95
  promptGuidelines: [
91
96
  'Use session_search with markdown only when the session search anchor mode is configured.',
92
97
  'Request source anchors, not summaries or previews.',
@@ -156,6 +161,11 @@ Examples:
156
161
 
157
162
  Returns bounded conversation snippets with session dates and project context. Large messages are truncated with their original character count.`,
158
163
  promptSnippet: 'Search past conversations for relevant context',
164
+ ...{ discovery: {
165
+ summary: 'Search past Selesai sessions for relevant conversation context',
166
+ aliases: ['past sessions', 'conversation history', 'session history'],
167
+ category: 'memory',
168
+ } },
159
169
  promptGuidelines: [
160
170
  'Use session_search when the user asks about previous discussions or past work.',
161
171
  'Use session_search when you need context from earlier sessions.',
@@ -95,6 +95,11 @@ export function registerSkillTool(pi: ExtensionAPI, store: SkillStore): void {
95
95
  label: "Skill Manager",
96
96
  description: SKILL_TOOL_DESCRIPTION,
97
97
  promptSnippet: "Create, inspect, and update reusable procedures and patterns",
98
+ ...{ discovery: {
99
+ summary: "Create, inspect, or update reusable procedure skills",
100
+ aliases: ["manage skills", "save procedure", "create skill"],
101
+ category: "capability",
102
+ } },
98
103
  promptGuidelines: [
99
104
  "Use the skill_manage tool after completing complex tasks that required trial and error or multiple tool calls.",
100
105
  "Use 'create' to save a new reusable procedure, 'patch' to update a section of an existing skill by skill_id, and 'update' for a full rewrite.",
@@ -7,6 +7,7 @@ import { DatabaseManager } from '../../src/store/db.js';
7
7
  import { addMemory } from '../../src/store/sqlite-memory-store.js';
8
8
  import { normalizeMemoryLookupText } from '../../src/store/memory-lookup.js';
9
9
  import { registerMemorySearchTool } from '../../src/tools/memory-search-tool.js';
10
+ import { MEMORY_SEARCH_REQUEST_EVENT } from '../../src/memory-search-bridge.js';
10
11
 
11
12
  let ROOT_DIR = '';
12
13
 
@@ -103,4 +104,28 @@ describe('registerMemorySearchTool', () => {
103
104
 
104
105
  dbManager.close();
105
106
  });
107
+
108
+ it('serves one local read-only lookup through the extension bridge', () => {
109
+ const dbManager = makeDbManager();
110
+ addMemory(dbManager, 'project deployment convention', 'memory', 'project-a');
111
+
112
+ let handler: ((request: unknown) => void) | undefined;
113
+ registerMemorySearchTool({
114
+ registerTool: () => {},
115
+ events: { on: (channel: string, listener: (request: unknown) => void) => {
116
+ if (channel === MEMORY_SEARCH_REQUEST_EVENT) handler = listener;
117
+ } },
118
+ } as any, dbManager);
119
+
120
+ let response: any;
121
+ handler!({
122
+ input: { query: 'deployment', target: 'project', project: 'project-a', limit: 5 },
123
+ respond: (result: unknown) => { response = result; },
124
+ });
125
+
126
+ assert.equal(response.success, true);
127
+ assert.equal(response.count, 1);
128
+ assert.match(response.output, /project deployment convention/);
129
+ dbManager.close();
130
+ });
106
131
  });
@@ -1819,6 +1819,11 @@ export default function piIntercomExtension(pi: ExtensionAPI) {
1819
1819
  label: "Contact Supervisor",
1820
1820
  description: "Subagent-only tool for contacting the supervisor agent that delegated this task. Use need_decision when blocked, uncertain, needing approval, or facing a product/API/scope decision before continuing; this waits for the supervisor's reply. Use interview_request when multiple structured questions need supervisor answers; this also waits for a reply. Use progress_update only for meaningful progress or unexpected discoveries that change the plan; this does not wait for a reply. Do not use for routine completion handoffs.",
1821
1821
  promptSnippet: "Subagent-only: contact the supervisor for decisions, structured interviews, or meaningful plan-changing updates. Do not use for routine completion handoffs.",
1822
+ discovery: {
1823
+ summary: "Subagent-only channel to ask the supervisor for decisions or send plan-changing updates",
1824
+ aliases: ["contact supervisor", "ask supervisor"],
1825
+ category: "coordination",
1826
+ },
1822
1827
  promptGuidelines: [
1823
1828
  "Use contact_supervisor with reason='need_decision' when a subagent is blocked, uncertain, needs approval, or faces a product/API/scope decision before continuing.",
1824
1829
  "Use contact_supervisor with reason='interview_request' when the child needs multiple structured answers from the supervisor in one blocking exchange.",
@@ -2088,6 +2093,11 @@ export default function piIntercomExtension(pi: ExtensionAPI) {
2088
2093
  pi.registerTool(defineTool({
2089
2094
  name: "intercom",
2090
2095
  label: "Intercom",
2096
+ discovery: {
2097
+ summary: "Send messages and coordinate with other local Selesai sessions",
2098
+ aliases: ["contact session", "session message", "peer session"],
2099
+ category: "coordination",
2100
+ },
2091
2101
  description: `Send a message to another Selesai session running on this machine.
2092
2102
  Use this to communicate findings, request help, or coordinate work with other sessions.
2093
2103
 
@@ -181,6 +181,11 @@ export default function registerFanoutChildSubagentExtension(pi: ExtensionAPI, c
181
181
  "Allowed management/control actions: list, get, status, lane.status, interrupt, resume, steer, doctor.",
182
182
  "Mutating management actions (create, update, delete, eject, disable, enable, reset, grant-spawn-budget, lane.recordMerge, lane.recordSupersession) are blocked in this mode.",
183
183
  ].join("\n"),
184
+ ...{ discovery: {
185
+ summary: "Delegate to subagents from child-safe fanout mode",
186
+ aliases: ["delegate task", "child agent", "parallel agents"],
187
+ category: "agents",
188
+ } },
184
189
  parameters: params,
185
190
  async execute(id, params, signal, onUpdate, ctx) {
186
191
  return finalizeToolResult(await executor.executePublic(id, params as SubagentParamsLike, signal ?? new AbortController().signal, onUpdate, ctx));
@@ -757,6 +757,11 @@ export default function registerSubagentExtension(pi: ExtensionAPI): void {
757
757
  label: "Subagent",
758
758
  description: buildSubagentToolDescription(config),
759
759
  ...buildSubagentToolPromptMetadata(config),
760
+ ...{ discovery: {
761
+ summary: "Delegate a task to one or more child agents, or run a workflow script",
762
+ aliases: ["delegate task", "child agent", "parallel agents"],
763
+ category: "agents",
764
+ } },
760
765
  parameters,
761
766
 
762
767
  async execute(id, params, signal, onUpdate, ctx) {
@@ -259,6 +259,11 @@ export function registerNativeSupervisorClient(pi: ExtensionAPI, metadata: Child
259
259
  name: "contact_supervisor",
260
260
  label: "Contact Supervisor",
261
261
  description: "Contact the parent/supervisor session for a blocking decision, structured interview, or progress update.",
262
+ ...{ discovery: {
263
+ summary: "Contact the parent/supervisor session for decisions, interviews, or progress updates",
264
+ aliases: ["contact supervisor", "ask supervisor"],
265
+ category: "coordination",
266
+ } },
262
267
  parameters: ContactSupervisorParamsSchema,
263
268
  execute(id, params, signal) {
264
269
  return sendSupervisorRequest(params as ContactSupervisorParams, metadata, signal, id);
@@ -578,6 +583,11 @@ function buildParentSupervisorTool(pi: ExtensionAPI, pending: Map<string, Pendin
578
583
  name: NATIVE_SUPERVISOR_TOOL_NAME,
579
584
  label: "Subagent Supervisor",
580
585
  description: "Native pi-subagents supervisor channel. Use reply/pending/status to answer child subagent requests without overriding pi-intercom.",
586
+ ...{ discovery: {
587
+ summary: "Answer child subagent requests: reply, pending, or status",
588
+ aliases: ["supervisor channel", "reply to subagent"],
589
+ category: "coordination",
590
+ } },
581
591
  parameters: IntercomParamsSchema,
582
592
  async execute(_id, params) {
583
593
  // Discover new request files even when demand-gated polling is idle.
@@ -37,6 +37,11 @@ Non-blocking subscriptions are visible in subagent status and differ from disabl
37
37
  name: "bg_wait",
38
38
  label: "Background Wait",
39
39
  description,
40
+ ...{ discovery: {
41
+ summary: "Wait for background or detached subagent work to finish",
42
+ aliases: ["wait for background work", "background wait"],
43
+ category: "agents",
44
+ } },
40
45
  parameters: SubagentWaitParams,
41
46
  execute,
42
47
  };
@@ -46,6 +51,11 @@ Non-blocking subscriptions are visible in subagent status and differ from disabl
46
51
  name: "subagent_wait",
47
52
  label: "Subagent Wait (deprecated)",
48
53
  description: "Deprecated compatibility alias for `bg_wait`. Use `bg_wait` for background, provider, or detached work without a native completion notification. It has the same parameters and behavior.",
54
+ ...{ discovery: {
55
+ summary: "Deprecated compatibility alias for bg_wait",
56
+ aliases: ["wait for subagent"],
57
+ category: "agents",
58
+ } },
49
59
  });
50
60
  }
51
61
 
@@ -114,6 +114,11 @@ export default function extension(pi: ExtensionAPI) {
114
114
  label: 'Web Explore',
115
115
  description:
116
116
  'Research a web question using bounded search/fetch passes, source ranking, and targeted headless escalation. Use this for web research, current docs/discussion lookups, and recommendation summaries.',
117
+ discovery: {
118
+ summary: 'Research a web question with bounded search, fetch, and source ranking',
119
+ aliases: ['web research', 'web search', 'current docs lookup'],
120
+ category: 'web'
121
+ },
117
122
  parameters: Type.Object({
118
123
  query: Type.String({ description: 'Web research question to explore.' })
119
124
  }),
@@ -616,6 +616,11 @@ export default function questionExtension(pi: ExtensionAPI) {
616
616
  description:
617
617
  "Ask one or more typed questions in a terminal wizard. Supports select, multiselect, text, stable option values, Other answers, and partial atomic submission.",
618
618
  promptSnippet: "Ask the user one or more typed questions when a decision is required.",
619
+ discovery: {
620
+ summary: "Ask the user one or more typed questions in a terminal wizard",
621
+ aliases: ["ask user", "user decision", "clarifying questions"],
622
+ category: "interaction",
623
+ },
619
624
  promptGuidelines: [
620
625
  "Use question only for decisions the user must make; inspect facts yourself.",
621
626
  "Send one questions array containing independent decisions that can be answered together.",
@@ -14,13 +14,18 @@ import { dirname, join } from "node:path";
14
14
  import { getAgentDir, getModelsPath } from "@selesai/code";
15
15
  import type { AuthStorage, ExtensionAPI, ExtensionCommandContext, ExtensionContext, SessionStartEvent } from "@selesai/code";
16
16
  import {
17
+ calculateCost,
18
+ contentText,
17
19
  createAssistantMessageEventStream,
20
+ getSystemMessageText,
18
21
  lazyStream,
22
+ type AssistantMessage,
19
23
  type AssistantMessageEvent,
20
24
  type AssistantMessageEventStream,
21
25
  type Context,
22
26
  type Model,
23
27
  type SimpleStreamOptions,
28
+ type TextContent,
24
29
  } from "@earendil-works/pi-ai";
25
30
  import { getApiProvider } from "@earendil-works/pi-ai/compat";
26
31
 
@@ -335,6 +340,171 @@ function persistActiveTokenInAccount(account: TokenInAccount, authPath: string =
335
340
  }
336
341
  }
337
342
 
343
+ /**
344
+ * The decisions deployments Token-In serves (Jev).
345
+ *
346
+ * These answer exactly one non-streaming JSON completion, and the gateway's
347
+ * Jev deployment rejects the SSE path outright, so the normal OpenAI-compatible
348
+ * streaming adapter can only turn a working decision into an HTTP 500. Every
349
+ * other tokenin model keeps the untouched streaming path below.
350
+ */
351
+ export const TOKEN_IN_DECISIONS_MODEL_PATTERN = /^jev-/;
352
+
353
+ export function isTokenInDecisionsModel(model: { id: string }): boolean {
354
+ return TOKEN_IN_DECISIONS_MODEL_PATTERN.test(model.id);
355
+ }
356
+
357
+ function emptyUsage(): AssistantMessage["usage"] {
358
+ return {
359
+ input: 0,
360
+ output: 0,
361
+ cacheRead: 0,
362
+ cacheWrite: 0,
363
+ totalTokens: 0,
364
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
365
+ };
366
+ }
367
+
368
+ /**
369
+ * The plain-text chat messages a decisions request carries: the transcript's
370
+ * text, and nothing else. Decisions deployments take no tools, no images, and
371
+ * no thinking blocks, so anything richer is dropped rather than mis-sent.
372
+ */
373
+ export function tokenInDecisionMessages(context: Context): Array<{ role: "system" | "user" | "assistant"; content: string }> {
374
+ const messages: Array<{ role: "system" | "user" | "assistant"; content: string }> = [];
375
+ for (const message of context.messages) {
376
+ if (message.role === "system") {
377
+ const content = getSystemMessageText(message).trim();
378
+ if (content) messages.push({ role: "system", content });
379
+ continue;
380
+ }
381
+ if (message.role === "assistant") {
382
+ const content = contentText(message.content).trim();
383
+ if (content) messages.push({ role: "assistant", content });
384
+ continue;
385
+ }
386
+ // User and tool-result text both travel as user turns: a decisions model has
387
+ // no tool protocol to answer through.
388
+ const content = contentText(message.content).trim();
389
+ if (content) messages.push({ role: "user", content });
390
+ }
391
+ return messages;
392
+ }
393
+
394
+ /** The decisions completion's text, when the body is OpenAI-shaped. */
395
+ function decisionCompletion(body: unknown): { text: string; finishReason: unknown; usage: Record<string, unknown> } | undefined {
396
+ if (typeof body !== "object" || body === null) return undefined;
397
+ const choices = (body as { choices?: unknown }).choices;
398
+ if (!Array.isArray(choices) || choices.length === 0) return undefined;
399
+ const choice = choices[0] as { message?: { content?: unknown }; finish_reason?: unknown };
400
+ const content = choice?.message?.content;
401
+ const text =
402
+ typeof content === "string"
403
+ ? content
404
+ : Array.isArray(content)
405
+ ? contentText(content as readonly { type: "text"; text: string }[])
406
+ : undefined;
407
+ if (text === undefined) return undefined;
408
+ const usage = (body as { usage?: unknown }).usage;
409
+ return {
410
+ text,
411
+ finishReason: choice?.finish_reason,
412
+ usage: typeof usage === "object" && usage !== null ? (usage as Record<string, unknown>) : {},
413
+ };
414
+ }
415
+
416
+ function usageNumber(value: unknown): number {
417
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
418
+ }
419
+
420
+ /**
421
+ * One non-streaming decisions call, rendered into the normal event protocol so
422
+ * every pi consumer (and cost accounting) sees an ordinary assistant message.
423
+ * The returned `errorMessage` feeds the caller's account-rotation decision.
424
+ */
425
+ async function requestTokenInDecision(
426
+ model: Model<any>,
427
+ context: Context,
428
+ options: SimpleStreamOptions,
429
+ ): Promise<{ stream: AssistantMessageEventStream; errorMessage?: string }> {
430
+ const stream = createAssistantMessageEventStream();
431
+ const message: AssistantMessage = {
432
+ role: "assistant",
433
+ content: [],
434
+ api: model.api,
435
+ provider: model.provider,
436
+ model: model.id,
437
+ usage: emptyUsage(),
438
+ stopReason: "pending",
439
+ timestamp: Date.now(),
440
+ };
441
+ const fail = (text: string): { stream: AssistantMessageEventStream; errorMessage: string } => {
442
+ const aborted = options.signal?.aborted === true;
443
+ message.stopReason = aborted ? "aborted" : "error";
444
+ message.errorMessage = text;
445
+ stream.push({ type: "error", reason: message.stopReason, error: message });
446
+ stream.end(message);
447
+ return { stream, errorMessage: text };
448
+ };
449
+
450
+ const requested: Record<string, unknown> = {
451
+ model: model.id,
452
+ messages: tokenInDecisionMessages(context),
453
+ };
454
+ if (options.maxTokens !== undefined) requested.max_tokens = options.maxTokens;
455
+ const payload = (await options.onPayload?.(requested, model)) ?? requested;
456
+
457
+ let response: Response;
458
+ try {
459
+ response = await (options.fetch ?? fetch)(`${model.baseUrl.replace(/\/+$/, "")}/chat/completions`, {
460
+ method: "POST",
461
+ headers: { "Content-Type": "application/json", ...(options.headers ?? {}) },
462
+ body: JSON.stringify(payload),
463
+ ...(options.signal ? { signal: options.signal } : {}),
464
+ });
465
+ } catch (error) {
466
+ return fail(error instanceof Error ? error.message : String(error));
467
+ }
468
+ await options.onResponse?.(
469
+ { status: response.status, headers: Object.fromEntries(response.headers.entries()) },
470
+ model,
471
+ );
472
+
473
+ const raw = await response.text();
474
+ if (!response.ok) return fail(`HTTP ${response.status}: ${raw.slice(0, 400)}`);
475
+
476
+ let body: unknown;
477
+ try {
478
+ body = JSON.parse(raw);
479
+ } catch {
480
+ return fail(`Malformed decisions response: ${raw.slice(0, 200)}`);
481
+ }
482
+ const completion = decisionCompletion(body);
483
+ if (!completion) return fail(`Unexpected decisions response: ${raw.slice(0, 200)}`);
484
+
485
+ const input = usageNumber(completion.usage.prompt_tokens ?? completion.usage.input_tokens);
486
+ const output = usageNumber(completion.usage.completion_tokens ?? completion.usage.output_tokens);
487
+ message.usage = {
488
+ ...emptyUsage(),
489
+ input,
490
+ output,
491
+ totalTokens: usageNumber(completion.usage.total_tokens) || input + output,
492
+ };
493
+ calculateCost(model, message.usage);
494
+ message.stopReason = completion.finishReason === "length" ? "length" : "stop";
495
+
496
+ stream.push({ type: "start", partial: { ...message } });
497
+ const block: TextContent = { type: "text", text: "" };
498
+ message.content = [block];
499
+ stream.push({ type: "text_start", contentIndex: 0, partial: { ...message } });
500
+ block.text = completion.text;
501
+ stream.push({ type: "text_delta", contentIndex: 0, delta: completion.text, partial: { ...message } });
502
+ stream.push({ type: "text_end", contentIndex: 0, content: completion.text, partial: { ...message } });
503
+ stream.push({ type: "done", reason: message.stopReason === "length" ? "length" : "stop", message });
504
+ stream.end(message);
505
+ return { stream };
506
+ }
507
+
338
508
  /**
339
509
  * Custom streamSimple implementation for tokenin provider that wraps OpenAI completions
340
510
  * and automatically rotates to alternative saved keys on 401/429/budget exceeded errors.
@@ -407,6 +577,21 @@ export function createTokenInStreamSimple(options?: {
407
577
 
408
578
  const requestOptions = withAccountAuthorization(streamOptions, currentAccount);
409
579
 
580
+ // Decisions deployments never stream: one POST, one text block, and only
581
+ // a rotateable failure moves on to the next account.
582
+ if (isTokenInDecisionsModel(model)) {
583
+ const decided = await requestTokenInDecision(model, context, requestOptions);
584
+ if (decided.errorMessage === undefined) return decided.stream;
585
+ if (isRotateableTokenInError(decided.errorMessage) && i + 1 < candidateAccounts.length) {
586
+ setTokenInCooldown(currentAccount.id);
587
+ const nextAccount = candidateAccounts[i + 1]!;
588
+ activate(nextAccount);
589
+ options?.onRotate?.(currentAccount, nextAccount, decided.errorMessage);
590
+ continue;
591
+ }
592
+ return decided.stream;
593
+ }
594
+
410
595
  let underlyingStream: AssistantMessageEventStream;
411
596
  try {
412
597
  underlyingStream = streamSimple(model, context, requestOptions);