@punica/editor 1.0.5 → 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.bundle.esm.js +1 -1
- package/dist/index.bundle.esm.js.map +1 -1
- package/dist/index.bundle.umd.js +1 -1
- package/dist/index.bundle.umd.js.map +1 -1
- package/package.json +28 -3
- package/types/index.d.ts +120 -11
- package/types/punica.module.bootstrap.d.ts +45 -0
- package/types/punica.module.capability.d.ts +359 -0
- package/types/punica.module.extensions.api.d.ts +740 -0
- package/types/punica.module.extensions.settings.d.ts +106 -0
- package/types/punica.module.flow.agent.d.ts +75 -0
- package/types/punica.module.flow.api.d.ts +128 -0
- package/types/punica.module.flow.d.ts +490 -0
- package/types/punica.module.flow.engine.d.ts +228 -0
- package/types/punica.module.flow.mcp.d.ts +26 -0
- package/types/punica.module.flow.notebook.d.ts +210 -0
- package/types/punica.module.flow.primitives.d.ts +700 -0
- package/types/punica.module.flow.shell.d.ts +374 -0
- package/types/punica.module.kernel.ai.d.ts +462 -0
- package/types/punica.module.kernel.commands.d.ts +49 -0
- package/types/punica.module.kernel.events.d.ts +274 -0
- package/types/punica.module.kernel.history.d.ts +20 -0
- package/types/punica.module.kernel.llm.d.ts +343 -0
- package/types/punica.module.kernel.notifications.d.ts +64 -0
- package/types/punica.module.kernel.policy.d.ts +273 -0
- package/types/punica.module.kernel.tasks.d.ts +107 -0
- package/types/punica.module.kernel.timeServer.d.ts +16 -0
- package/types/punica.module.runtime.api.d.ts +214 -0
- package/types/punica.module.runtime.capabilities.d.ts +175 -0
- package/types/punica.module.runtime.compute.d.ts +339 -0
- package/types/punica.module.runtime.datasets.d.ts +234 -0
- package/types/punica.module.runtime.fs.d.ts +385 -0
- package/types/punica.module.runtime.harness.d.ts +246 -0
- package/types/punica.module.runtime.host.d.ts +272 -0
- package/types/punica.module.runtime.inference.d.ts +164 -0
- package/types/punica.module.runtime.lifecycle.d.ts +15 -0
- package/types/punica.module.runtime.llm.d.ts +470 -0
- package/types/punica.module.runtime.mcp.d.ts +139 -0
- package/types/punica.module.runtime.modelRuntimes.d.ts +90 -0
- package/types/punica.module.runtime.models.d.ts +254 -0
- package/types/punica.module.runtime.search.d.ts +59 -0
- package/types/punica.module.runtime.secrets.d.ts +26 -0
- package/types/punica.module.runtime.tasks.d.ts +27 -0
- package/types/punica.module.runtime.vcs.d.ts +67 -0
- package/types/punica.module.runtime.vectors.d.ts +74 -0
- package/types/punica.module.runtime.workspace.d.ts +134 -0
- package/types/punica.module.shell.activityBar.d.ts +42 -0
- package/types/punica.module.shell.components.d.ts +87 -0
- package/types/punica.module.shell.contentTabs.d.ts +33 -0
- package/types/punica.module.shell.dragDrop.d.ts +25 -0
- package/types/punica.module.shell.keyboardShortcuts.d.ts +38 -0
- package/types/punica.module.shell.layout.d.ts +106 -0
- package/types/punica.module.shell.markdown.d.ts +36 -0
- package/types/punica.module.shell.panelTabs.d.ts +48 -0
- package/types/punica.module.shell.profile.d.ts +278 -0
- package/types/punica.module.shell.statusbar.d.ts +26 -0
- package/types/punica.module.shell.view.d.ts +455 -0
- package/types/punica.module.shell.views.d.ts +150 -0
- package/types/punica.module.test.d.ts +562 -0
- package/types/punica.module.activityBar.d.ts +0 -21
- package/types/punica.module.commands.d.ts +0 -21
- package/types/punica.module.dragDrop.d.ts +0 -23
- package/types/punica.module.extensions.d.ts +0 -157
- package/types/punica.module.history.d.ts +0 -18
- package/types/punica.module.keyboardShortcuts.d.ts +0 -29
- package/types/punica.module.layout.d.ts +0 -23
- package/types/punica.module.statusbar.d.ts +0 -21
- package/types/punica.module.timeServer.d.ts +0 -14
- package/types/punica.module.view.d.ts +0 -8
|
@@ -0,0 +1,470 @@
|
|
|
1
|
+
declare module 'punica' {
|
|
2
|
+
export namespace runtime {
|
|
3
|
+
/**
|
|
4
|
+
* User-facing quality profile. Providers are free to map these to concrete
|
|
5
|
+
* models / quantizations / routing policies.
|
|
6
|
+
*/
|
|
7
|
+
export type LlmProfile = 'fast' | 'balanced' | 'accurate';
|
|
8
|
+
|
|
9
|
+
export type LlmRole = 'system' | 'user' | 'assistant' | 'tool';
|
|
10
|
+
|
|
11
|
+
export type LlmRoutePreference = 'auto' | 'local' | 'remote';
|
|
12
|
+
|
|
13
|
+
export interface LlmMessage {
|
|
14
|
+
role: LlmRole;
|
|
15
|
+
content: string;
|
|
16
|
+
name?: string;
|
|
17
|
+
/**
|
|
18
|
+
* Tool calls requested by an assistant turn (substrate shape). Set when
|
|
19
|
+
* replaying an assistant message that invoked tools back to the model on
|
|
20
|
+
* the next agent-loop iteration. The host transport encodes these to the
|
|
21
|
+
* vendor wire format (OpenAI `tool_calls`, Anthropic `tool_use`, …).
|
|
22
|
+
* Strict remote APIs require this so a following tool-role message can be
|
|
23
|
+
* correlated via `toolCallId`.
|
|
24
|
+
*/
|
|
25
|
+
toolCalls?: ToolCall[];
|
|
26
|
+
/**
|
|
27
|
+
* For a `role: 'tool'` result message — the id of the assistant tool call
|
|
28
|
+
* this result answers. Encoded to the vendor `tool_call_id` field. Lenient
|
|
29
|
+
* local servers ignore it; strict remote APIs (OpenAI) require it.
|
|
30
|
+
*/
|
|
31
|
+
toolCallId?: string;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Substrate-neutral prompt-cache hints. The substrate states WHAT is
|
|
36
|
+
* stable about the request; hosts translate to vendor wire syntax
|
|
37
|
+
* (Anthropic `cache_control`, OpenAI `prompt_cache_key`, llama.cpp
|
|
38
|
+
* `cache_prompt`). Vendors without prompt caching ignore the hints —
|
|
39
|
+
* the deterministic ordering the hints assume still helps implicit
|
|
40
|
+
* KV-cache reuse on local servers.
|
|
41
|
+
*/
|
|
42
|
+
export interface LlmCacheHints {
|
|
43
|
+
/**
|
|
44
|
+
* The `tools` array and system prompt are byte-stable for the
|
|
45
|
+
* lifetime of `cacheKey` — safe to cache as a prefix.
|
|
46
|
+
*/
|
|
47
|
+
stableToolsAndSystem?: boolean;
|
|
48
|
+
/**
|
|
49
|
+
* Number of leading messages guaranteed identical to the previous
|
|
50
|
+
* request with the same `cacheKey` (the cached conversation
|
|
51
|
+
* prefix). 0 on the first request of a run; grows monotonically as
|
|
52
|
+
* the caller appends — and resets after history compaction rewrites
|
|
53
|
+
* the prefix.
|
|
54
|
+
*/
|
|
55
|
+
stableMessagePrefix?: number;
|
|
56
|
+
/**
|
|
57
|
+
* Cache routing key — requests sharing it target the same cached
|
|
58
|
+
* prefix (typically the run/conversation correlation id).
|
|
59
|
+
*/
|
|
60
|
+
cacheKey?: string;
|
|
61
|
+
/** TTL preference; vendors ignore when unsupported. */
|
|
62
|
+
ttl?: '5m' | '1h';
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export interface LlmChatRequest {
|
|
66
|
+
messages: LlmMessage[];
|
|
67
|
+
profile?: LlmProfile;
|
|
68
|
+
temperature?: number;
|
|
69
|
+
maxTokens?: number;
|
|
70
|
+
/**
|
|
71
|
+
* Optional correlation id used for auditing and tracing.
|
|
72
|
+
*/
|
|
73
|
+
correlationId?: string;
|
|
74
|
+
/**
|
|
75
|
+
* Optional routing preference. When set to 'remote', the runtime may need
|
|
76
|
+
* explicit user approval (Policy kind = "llm.remote").
|
|
77
|
+
*/
|
|
78
|
+
route?: LlmRoutePreference;
|
|
79
|
+
/**
|
|
80
|
+
* Optional provider identifier (e.g. "remote:openai", "local:llama.cpp").
|
|
81
|
+
* Used for policy scoping and audit.
|
|
82
|
+
*/
|
|
83
|
+
providerId?: string;
|
|
84
|
+
/**
|
|
85
|
+
* Optional response format. When set to 'json_object', the LLM is instructed
|
|
86
|
+
* to return valid JSON. Used for structured output.
|
|
87
|
+
*/
|
|
88
|
+
responseFormat?: 'json_object' | 'text';
|
|
89
|
+
/**
|
|
90
|
+
* Optional concrete model identifier hint.
|
|
91
|
+
* When set, providers may use this to select a specific expert model
|
|
92
|
+
* (e.g. a particular local GGUF or remote model family/size).
|
|
93
|
+
*/
|
|
94
|
+
modelId?: string;
|
|
95
|
+
/**
|
|
96
|
+
* Optional list of concrete model identifiers for multi-expert scenarios.
|
|
97
|
+
* The runtime/capability layer is responsible for orchestrating multiple
|
|
98
|
+
* calls; providers typically receive one modelId at a time.
|
|
99
|
+
*/
|
|
100
|
+
modelIds?: string[];
|
|
101
|
+
/**
|
|
102
|
+
* Optional substrate-side tool catalog for AI tool calling.
|
|
103
|
+
* Providers translate this to vendor-specific tool/function-call format
|
|
104
|
+
* (OpenAI functions, Anthropic tool_use, native JSON schema, etc.).
|
|
105
|
+
* The set of tools an AI can invoke during this chat turn.
|
|
106
|
+
*/
|
|
107
|
+
tools?: ToolDef[];
|
|
108
|
+
/**
|
|
109
|
+
* Optional tool-call control:
|
|
110
|
+
* - 'auto' : provider decides when to call tools (default)
|
|
111
|
+
* - 'none' : never call tools (text-only response)
|
|
112
|
+
* - 'required' : must call at least one tool
|
|
113
|
+
* - { type: 'tool'; name: string } : force a specific tool
|
|
114
|
+
*/
|
|
115
|
+
toolChoice?: ToolChoice;
|
|
116
|
+
/**
|
|
117
|
+
* Optional cancellation signal (Faz 0b Stop/iptal). When the caller aborts
|
|
118
|
+
* the signal, the runtime stops consuming the provider stream and host
|
|
119
|
+
* adapters should abort the in-flight request (fetch/SDK). Carried on the
|
|
120
|
+
* request because the host `LlmApi` only receives the request object;
|
|
121
|
+
* inside the gateway the same signal travels on `InvocationContext.signal`.
|
|
122
|
+
*/
|
|
123
|
+
signal?: AbortSignal;
|
|
124
|
+
/**
|
|
125
|
+
* Optional prompt-cache hints (see LlmCacheHints). Set by callers
|
|
126
|
+
* whose request prefix is deterministic across turns (the agent
|
|
127
|
+
* loop); translated to vendor syntax host-side.
|
|
128
|
+
*/
|
|
129
|
+
cacheHints?: LlmCacheHints;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
export interface LlmUsage {
|
|
133
|
+
inputTokens?: number;
|
|
134
|
+
outputTokens?: number;
|
|
135
|
+
totalTokens?: number;
|
|
136
|
+
/**
|
|
137
|
+
* Input tokens served from the vendor's prompt cache (Anthropic
|
|
138
|
+
* `cache_read_input_tokens`, OpenAI `prompt_tokens_details.cached_tokens`).
|
|
139
|
+
* A subset of `inputTokens`; billed at the vendor's reduced cache-read
|
|
140
|
+
* rate. Absent when the vendor reports no cache activity.
|
|
141
|
+
*/
|
|
142
|
+
cacheReadInputTokens?: number;
|
|
143
|
+
/**
|
|
144
|
+
* Input tokens written to the vendor's prompt cache this call
|
|
145
|
+
* (Anthropic `cache_creation_input_tokens`). Absent when the vendor
|
|
146
|
+
* reports no cache activity.
|
|
147
|
+
*/
|
|
148
|
+
cacheCreationInputTokens?: number;
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
export interface LlmChatResponse {
|
|
152
|
+
text: string;
|
|
153
|
+
modelId?: string;
|
|
154
|
+
provider?: string; // e.g. "local", "remote:openai", etc.
|
|
155
|
+
usage?: LlmUsage;
|
|
156
|
+
raw?: unknown;
|
|
157
|
+
/**
|
|
158
|
+
* Optional accountability metadata describing this concrete call.
|
|
159
|
+
* Populated by the substrate's LLM perimeter; useful for cost
|
|
160
|
+
* reporting, fallback chain tracing, and audit events.
|
|
161
|
+
*/
|
|
162
|
+
meta?: LlmCallMeta;
|
|
163
|
+
/**
|
|
164
|
+
* Optional tool calls emitted by the assistant in this response.
|
|
165
|
+
* Each entry is the unified substrate ToolCall shape (vendor
|
|
166
|
+
* formats are normalized by the provider). Empty / undefined
|
|
167
|
+
* means a plain text response.
|
|
168
|
+
*/
|
|
169
|
+
toolCalls?: ToolCall[];
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Substrate-side tool / capability declaration shape passed to an LLM
|
|
174
|
+
* provider. Providers translate to vendor-specific formats internally.
|
|
175
|
+
*/
|
|
176
|
+
export interface ToolDef {
|
|
177
|
+
/**
|
|
178
|
+
* Stable capability identifier (matches the entry in the Trinity
|
|
179
|
+
* capability registry, e.g. "fs.readFile", "llm.plan",
|
|
180
|
+
* "ivy.node.echo").
|
|
181
|
+
*/
|
|
182
|
+
name: string;
|
|
183
|
+
/**
|
|
184
|
+
* Human-readable description of what the tool does. Surfaced to
|
|
185
|
+
* the LLM as part of the tool catalog prompt.
|
|
186
|
+
*/
|
|
187
|
+
description: string;
|
|
188
|
+
/**
|
|
189
|
+
* Input JSON Schema. Providers convert this to their native
|
|
190
|
+
* tool-parameter schema (OpenAI function parameters, Anthropic
|
|
191
|
+
* tool input_schema, etc.).
|
|
192
|
+
*/
|
|
193
|
+
parameters: JSONSchema;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* Tool-call control directive sent to the provider. Mirrors common
|
|
198
|
+
* vendor semantics (OpenAI tool_choice, Anthropic tool_choice).
|
|
199
|
+
*/
|
|
200
|
+
export type ToolChoice =
|
|
201
|
+
| 'auto'
|
|
202
|
+
| 'none'
|
|
203
|
+
| 'required'
|
|
204
|
+
| { type: 'tool'; name: string };
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Unified tool call emitted by the LLM. Substrate providers normalize
|
|
208
|
+
* vendor formats (OpenAI function_call / tool_calls, Anthropic
|
|
209
|
+
* tool_use, native JSON) to this single shape.
|
|
210
|
+
*/
|
|
211
|
+
export interface ToolCall {
|
|
212
|
+
/**
|
|
213
|
+
* Provider-supplied call id (used to correlate tool result messages
|
|
214
|
+
* back to the originating call when streamed).
|
|
215
|
+
*/
|
|
216
|
+
id: string;
|
|
217
|
+
/**
|
|
218
|
+
* Capability id the LLM wants to invoke. Matches ToolDef.name.
|
|
219
|
+
*/
|
|
220
|
+
name: string;
|
|
221
|
+
/**
|
|
222
|
+
* Arguments object the LLM produced. Substrate-side validation
|
|
223
|
+
* (matching ToolDef.parameters) happens in the runtime layer
|
|
224
|
+
* before dispatch through the gateway.
|
|
225
|
+
*/
|
|
226
|
+
arguments: Record<string, unknown>;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Accountability metadata for a concrete LLM call. Populated by the
|
|
231
|
+
* substrate's perimeter; surfaced via response.meta, the `llm.call`
|
|
232
|
+
* kernel event, and the cost.report capability.
|
|
233
|
+
*/
|
|
234
|
+
export interface LlmCallMeta {
|
|
235
|
+
/**
|
|
236
|
+
* Provider id that actually served the request (e.g. "anthropic",
|
|
237
|
+
* "openai", "ollama", "local:llama.cpp", "litellm-proxy"). May
|
|
238
|
+
* differ from the requested providerId when fallback advances.
|
|
239
|
+
*/
|
|
240
|
+
provider: string;
|
|
241
|
+
/**
|
|
242
|
+
* Concrete model id the provider used (e.g. "claude-sonnet-4-6",
|
|
243
|
+
* "gpt-4o", "llama3:70b-q4_K_M").
|
|
244
|
+
*/
|
|
245
|
+
model: string;
|
|
246
|
+
promptTokens: number;
|
|
247
|
+
completionTokens: number;
|
|
248
|
+
/**
|
|
249
|
+
* Total cost of this call in USD. Computed from the host-injected
|
|
250
|
+
* pricing table (`setLlmPricingTable`). Zero for substrate-only
|
|
251
|
+
* paths (MockLlmProvider) and when no pricing is configured.
|
|
252
|
+
*/
|
|
253
|
+
totalCost: number;
|
|
254
|
+
latencyMs: number;
|
|
255
|
+
/**
|
|
256
|
+
* True when the response was served from a substrate-side cache
|
|
257
|
+
* (semantic or deterministic) without hitting the provider.
|
|
258
|
+
*/
|
|
259
|
+
cacheHit?: boolean;
|
|
260
|
+
/**
|
|
261
|
+
* Zero-indexed position in the fallback chain that served this
|
|
262
|
+
* call. 0 = primary provider served; >0 = fallback advanced.
|
|
263
|
+
* Undefined when the call did not go through a fallback chain.
|
|
264
|
+
*/
|
|
265
|
+
fallbackChainPosition?: number;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
/**
|
|
269
|
+
* Streaming chunk emitted by `LlmProvider.chatStream`. The variants
|
|
270
|
+
* form a discriminated union on `type`.
|
|
271
|
+
*
|
|
272
|
+
* `thinking` carries the model's reasoning trace (e.g. Anthropic
|
|
273
|
+
* extended thinking `thinking_delta`), kept on a separate channel
|
|
274
|
+
* from `content` so consumers can render it in a distinct,
|
|
275
|
+
* collapsible surface. Providers that do not expose a reasoning
|
|
276
|
+
* channel simply never emit it — consumers must feature-detect and
|
|
277
|
+
* hide the surface rather than fabricate one.
|
|
278
|
+
*/
|
|
279
|
+
export type LlmChunk =
|
|
280
|
+
| { type: 'content'; text: string }
|
|
281
|
+
| { type: 'thinking'; text: string }
|
|
282
|
+
| { type: 'tool_call'; toolCall: ToolCall }
|
|
283
|
+
| { type: 'done'; response: LlmChatResponse }
|
|
284
|
+
| { type: 'error'; error: { code: string; message: string } };
|
|
285
|
+
|
|
286
|
+
export interface EmbedRequest {
|
|
287
|
+
/**
|
|
288
|
+
* Texts to embed. Providers may batch internally; substrate does
|
|
289
|
+
* not impose a per-request size limit but hosts may.
|
|
290
|
+
*/
|
|
291
|
+
texts: string[];
|
|
292
|
+
/**
|
|
293
|
+
* Optional concrete embedder model identifier (e.g.
|
|
294
|
+
* "text-embedding-3-small"). Host-specific.
|
|
295
|
+
*/
|
|
296
|
+
modelId?: string;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
export interface EmbedResponse {
|
|
300
|
+
/**
|
|
301
|
+
* One vector per input text, in the same order as `texts`.
|
|
302
|
+
*/
|
|
303
|
+
vectors: number[][];
|
|
304
|
+
modelId?: string;
|
|
305
|
+
provider?: string;
|
|
306
|
+
usage?: LlmUsage;
|
|
307
|
+
meta?: LlmCallMeta;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/**
|
|
311
|
+
* Provider capability surface flags. A provider advertises which
|
|
312
|
+
* features it supports; the substrate uses this set to decide what
|
|
313
|
+
* it can dispatch through a given provider (and to pick fallbacks
|
|
314
|
+
* when a request requires a feature the primary does not have).
|
|
315
|
+
*/
|
|
316
|
+
export type LlmProviderCapability =
|
|
317
|
+
| 'chat'
|
|
318
|
+
| 'complete'
|
|
319
|
+
| 'embed'
|
|
320
|
+
| 'image'
|
|
321
|
+
| 'tool-use'
|
|
322
|
+
| 'streaming'
|
|
323
|
+
| 'thinking';
|
|
324
|
+
|
|
325
|
+
/**
|
|
326
|
+
* Substrate-pure LLM provider abstraction.
|
|
327
|
+
*
|
|
328
|
+
* Hosts (Electron, browser) implement this interface and inject
|
|
329
|
+
* concrete providers via the runtime registry. The substrate itself
|
|
330
|
+
* ships only `MockLlmProvider` (fixture-driven, no network); real
|
|
331
|
+
* HTTP/IPC adapters (Anthropic, OpenAI, Ollama, LiteLLM, etc.) live
|
|
332
|
+
* in host territory. Neutrality is proven by multi-shape fixture
|
|
333
|
+
* conformance, not by a single reference implementation.
|
|
334
|
+
*/
|
|
335
|
+
export interface LlmProvider {
|
|
336
|
+
/**
|
|
337
|
+
* Stable provider identifier (e.g. "anthropic", "openai",
|
|
338
|
+
* "ollama-local", "litellm-proxy", "mock"). Used in policy
|
|
339
|
+
* scoping, audit events, and provider-invariant tracking.
|
|
340
|
+
*/
|
|
341
|
+
id: string;
|
|
342
|
+
/**
|
|
343
|
+
* Set of capabilities this provider exposes. Substrate consults
|
|
344
|
+
* this to gate which dispatches the provider can serve.
|
|
345
|
+
*/
|
|
346
|
+
capabilities: Set<LlmProviderCapability>;
|
|
347
|
+
chat(req: LlmChatRequest): Promise<LlmChatResponse>;
|
|
348
|
+
/**
|
|
349
|
+
* Streaming chat. Optional — providers that don't support
|
|
350
|
+
* streaming should omit this and not advertise the 'streaming'
|
|
351
|
+
* capability flag.
|
|
352
|
+
*/
|
|
353
|
+
chatStream?(req: LlmChatRequest): AsyncIterable<LlmChunk>;
|
|
354
|
+
/**
|
|
355
|
+
* Embedding generation. Optional — providers that don't support
|
|
356
|
+
* embeddings should omit this and not advertise the 'embed'
|
|
357
|
+
* capability flag.
|
|
358
|
+
*/
|
|
359
|
+
embed?(req: EmbedRequest): Promise<EmbedResponse>;
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
/**
|
|
363
|
+
* Provider specification entry used inside a fallback chain. Each
|
|
364
|
+
* entry names a provider id (must be registered) plus optional
|
|
365
|
+
* per-spec model override; the substrate's fallback state machine
|
|
366
|
+
* advances through `attempts` of these.
|
|
367
|
+
*/
|
|
368
|
+
export interface ProviderSpec {
|
|
369
|
+
providerId: string;
|
|
370
|
+
modelId?: string;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
/**
|
|
374
|
+
* Fallback policy: how the substrate's fallback state machine reacts
|
|
375
|
+
* to each terminal signal coming from a provider. The substrate
|
|
376
|
+
* recognizes three signals; hosts choose how each maps to behavior.
|
|
377
|
+
*
|
|
378
|
+
* - 'next' : advance to the next ProviderSpec in the chain
|
|
379
|
+
* - 'fail' : terminate the call with the last error
|
|
380
|
+
* - 'queue' : flag the call for later retry (substrate sets the
|
|
381
|
+
* `pendingQueue` state; an actual queue mechanism lives
|
|
382
|
+
* in the host)
|
|
383
|
+
*/
|
|
384
|
+
export interface FallbackPolicy {
|
|
385
|
+
onError?: 'next' | 'fail' | 'queue';
|
|
386
|
+
onTimeout?: 'next' | 'fail' | 'queue';
|
|
387
|
+
onRateLimit?: 'next' | 'fail' | 'queue';
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* LLM model metadata (catalog record).
|
|
392
|
+
* Mirrors host-side model catalogs (Electron).
|
|
393
|
+
*/
|
|
394
|
+
export interface LlmModel {
|
|
395
|
+
id: string;
|
|
396
|
+
family: string;
|
|
397
|
+
parameters: string;
|
|
398
|
+
variant?: string;
|
|
399
|
+
quant?: string;
|
|
400
|
+
sizeBytes: number;
|
|
401
|
+
minRamGb?: number;
|
|
402
|
+
minVramGb?: number | null;
|
|
403
|
+
profiles?: LlmProfile[];
|
|
404
|
+
license?: string;
|
|
405
|
+
downloadUrls?: string[];
|
|
406
|
+
sha256?: string;
|
|
407
|
+
fileName?: string;
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
export interface InstalledLlmModel {
|
|
411
|
+
id: string;
|
|
412
|
+
filePath: string;
|
|
413
|
+
installedAtMs: number;
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
export interface LlmModelManagerState {
|
|
417
|
+
activeModelId: string | null;
|
|
418
|
+
installed: InstalledLlmModel[];
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
export interface LocalLlmServerStatus {
|
|
422
|
+
running: boolean;
|
|
423
|
+
host: string;
|
|
424
|
+
port: number;
|
|
425
|
+
url: string | null;
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
/**
|
|
429
|
+
* Optional host-provided LLM manager API.
|
|
430
|
+
* Electron typically provides this via IPC (window.api.llm.*) and the runtime
|
|
431
|
+
* exposes it as a facade under runtime.llm.manager.
|
|
432
|
+
*/
|
|
433
|
+
export interface LlmManagerApi {
|
|
434
|
+
listModels(opts?: { profile?: LlmProfile }): Promise<LlmModel[]>;
|
|
435
|
+
listInstalled(): Promise<LlmModelManagerState>;
|
|
436
|
+
setActiveModel(modelId: string | null): Promise<void>;
|
|
437
|
+
removeModel(modelId: string): Promise<void>;
|
|
438
|
+
downloadModel(modelId: string): Promise<void>;
|
|
439
|
+
pickModelFile(): Promise<{ canceled: boolean; path?: string }>;
|
|
440
|
+
importModelFile(opts: {
|
|
441
|
+
modelId: string;
|
|
442
|
+
sourcePath: string;
|
|
443
|
+
fileName?: string;
|
|
444
|
+
}): Promise<void>;
|
|
445
|
+
localServerStatus(): Promise<LocalLlmServerStatus>;
|
|
446
|
+
startLocalServer(opts?: {
|
|
447
|
+
host?: string;
|
|
448
|
+
port?: number;
|
|
449
|
+
}): Promise<LocalLlmServerStatus>;
|
|
450
|
+
stopLocalServer(): Promise<void>;
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
export interface LlmApi {
|
|
454
|
+
chat(req: LlmChatRequest): Promise<LlmChatResponse>;
|
|
455
|
+
/**
|
|
456
|
+
* Optional streaming chat. Hosts that back a streaming-capable
|
|
457
|
+
* provider expose it here; the substrate runtime facade wraps it
|
|
458
|
+
* with the same remote-policy gating as `chat`. When absent, the
|
|
459
|
+
* substrate falls back to a single-shot `chat` surfaced as one
|
|
460
|
+
* `content` + `done` chunk, so streaming consumers keep working
|
|
461
|
+
* (without live tokens) against non-streaming hosts.
|
|
462
|
+
*/
|
|
463
|
+
chatStream?(req: LlmChatRequest): AsyncIterable<LlmChunk>;
|
|
464
|
+
/**
|
|
465
|
+
* Optional manager API (Electron).
|
|
466
|
+
*/
|
|
467
|
+
manager?: LlmManagerApi;
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
}
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
declare module 'punica' {
|
|
2
|
+
namespace runtime {
|
|
3
|
+
namespace mcp {
|
|
4
|
+
type McpTransportKind = 'stdio' | 'streamable-http' | 'sse';
|
|
5
|
+
|
|
6
|
+
interface McpStdioConfig {
|
|
7
|
+
kind: 'stdio';
|
|
8
|
+
/** CLI command to spawn (e.g. "npx"). */
|
|
9
|
+
command: string;
|
|
10
|
+
args?: string[];
|
|
11
|
+
/** Extra environment variables injected into the child process. */
|
|
12
|
+
env?: Record<string, string>;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
interface McpHttpConfig {
|
|
16
|
+
kind: 'streamable-http' | 'sse';
|
|
17
|
+
url: string;
|
|
18
|
+
headers?: Record<string, string>;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
type McpTransportConfig = McpStdioConfig | McpHttpConfig;
|
|
22
|
+
|
|
23
|
+
interface McpServerConfig extends McpTransportConfig {
|
|
24
|
+
/** Unique server identifier — used as the `mcp.<id>.<tool>` capability namespace. */
|
|
25
|
+
id: string;
|
|
26
|
+
/** Human-readable display name (optional; falls back to id). */
|
|
27
|
+
name?: string;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
interface McpToolInfo {
|
|
31
|
+
name: string;
|
|
32
|
+
description?: string;
|
|
33
|
+
/** JSON Schema for the tool's input object. */
|
|
34
|
+
inputSchema: JSONSchema;
|
|
35
|
+
/** Punica-specific overrides embedded in MCP `_meta.punica`. */
|
|
36
|
+
_meta?: JSONObject;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
interface McpServerHandle {
|
|
40
|
+
readonly serverId: string;
|
|
41
|
+
readonly transportKind: McpTransportKind;
|
|
42
|
+
/** Tool list at the time of connection (or after last tools/list refresh). */
|
|
43
|
+
readonly tools: McpToolInfo[];
|
|
44
|
+
/** Disconnect this server and unregister all its capabilities. */
|
|
45
|
+
disconnect(): Promise<void>;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
type McpTrafficDirection = 'outbound' | 'inbound';
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Payload of the `mcp.request` / `mcp.response` kernel events.
|
|
52
|
+
*
|
|
53
|
+
* Every JSON-RPC exchange with an MCP server is published as a
|
|
54
|
+
* request/response pair matched by `callId` (also stamped as
|
|
55
|
+
* `envelope.correlationId`). Emitters:
|
|
56
|
+
* - `runtime:mcp:client` — outbound traffic to connected servers
|
|
57
|
+
* (all methods, including the initialize/tools-list handshake);
|
|
58
|
+
* - `extension:ivy-mcp-server` — inbound traffic from external
|
|
59
|
+
* MCP clients served by the editor (`transport: 'host'`).
|
|
60
|
+
*
|
|
61
|
+
* Bodies are pretty-printed JSON truncated to 16 384 chars
|
|
62
|
+
* (`truncated: true`, `*Bytes` = untruncated size). Token counts
|
|
63
|
+
* are estimates from the kernel token estimator. Notifications
|
|
64
|
+
* emit a single `mcp.request` with `kind: 'notification'` and no
|
|
65
|
+
* matching response. The event names deliberately avoid `.error`
|
|
66
|
+
* / `.failed` suffixes so the event store never persists bodies
|
|
67
|
+
* to disk; errors surface as `isError` + `envelope.kind: 'error'`.
|
|
68
|
+
*/
|
|
69
|
+
interface McpTrafficEventPayload {
|
|
70
|
+
direction: McpTrafficDirection;
|
|
71
|
+
serverId: string;
|
|
72
|
+
serverName?: string;
|
|
73
|
+
transport?: McpTransportKind | 'host';
|
|
74
|
+
method: string;
|
|
75
|
+
/** Only set for `tools/call`. */
|
|
76
|
+
toolName?: string;
|
|
77
|
+
/** Pairing key between the request and response events. */
|
|
78
|
+
callId: string;
|
|
79
|
+
kind: 'request' | 'notification';
|
|
80
|
+
requestJson?: string;
|
|
81
|
+
requestBytes?: number;
|
|
82
|
+
requestTokens?: number;
|
|
83
|
+
responseJson?: string;
|
|
84
|
+
responseBytes?: number;
|
|
85
|
+
responseTokens?: number;
|
|
86
|
+
isError?: boolean;
|
|
87
|
+
errorMessage?: string;
|
|
88
|
+
durationMs?: number;
|
|
89
|
+
/** True when the body was cut at the 16 384-char cap. */
|
|
90
|
+
truncated?: boolean;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
interface McpApi {
|
|
94
|
+
/**
|
|
95
|
+
* Connect to an MCP server, discover its tools, and register them
|
|
96
|
+
* as capabilities in the unified registry under the `mcp` origin.
|
|
97
|
+
*
|
|
98
|
+
* Fails with a descriptive error if transport cannot be established.
|
|
99
|
+
* Emits `capability.registered` for each discovered tool.
|
|
100
|
+
*/
|
|
101
|
+
connect(config: McpServerConfig): Promise<McpServerHandle>;
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Disconnect a previously connected server by its id.
|
|
105
|
+
* Unregisters all its capabilities and emits `capability.unregistered`.
|
|
106
|
+
* No-op if the server is not currently connected.
|
|
107
|
+
*/
|
|
108
|
+
disconnect(serverId: string): Promise<void>;
|
|
109
|
+
|
|
110
|
+
/** List all currently connected MCP server handles. */
|
|
111
|
+
listServers(): McpServerHandle[];
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* List tools for a specific server (or all servers when omitted).
|
|
115
|
+
* Returns an empty array if the server is not connected.
|
|
116
|
+
*/
|
|
117
|
+
listTools(serverId?: string): McpToolInfo[];
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Directly invoke an MCP tool (bypasses the capability gateway).
|
|
121
|
+
* Prefer routing through the gateway via `capability.call()` for
|
|
122
|
+
* governance. This method is exposed for host-level diagnostics.
|
|
123
|
+
*/
|
|
124
|
+
callTool(
|
|
125
|
+
serverId: string,
|
|
126
|
+
toolName: string,
|
|
127
|
+
args: JSONObject
|
|
128
|
+
): Promise<JSONObject>;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
namespace runtime {
|
|
134
|
+
interface RuntimeApi {
|
|
135
|
+
/** MCP server connection + tool-as-capability bridge. */
|
|
136
|
+
readonly mcp: runtime.mcp.McpApi;
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/// <reference path="punica.module.runtime.models.d.ts" />
|
|
2
|
+
/// <reference path="punica.module.runtime.inference.d.ts" />
|
|
3
|
+
|
|
4
|
+
declare module 'punica' {
|
|
5
|
+
export namespace runtime {
|
|
6
|
+
/**
|
|
7
|
+
* Local model execution seam. A `ModelRuntime` is a pluggable engine
|
|
8
|
+
* (GGUF / Transformers / ONNX / custom) that turns a `LocalModelRecord`
|
|
9
|
+
* into a running local API. The substrate owns only the abstraction +
|
|
10
|
+
* dispatch; the actual engine launch lives in the host/extension layer.
|
|
11
|
+
*
|
|
12
|
+
* See docs/model-runtimes.md (§2.1, §2.4).
|
|
13
|
+
*/
|
|
14
|
+
export namespace modelRuntimes {
|
|
15
|
+
/**
|
|
16
|
+
* Handle to a started local runtime. `endpoint` is OpenAI-compatible
|
|
17
|
+
* (e.g. http://127.0.0.1:8080) and is what `runtime.llm.chat({ route:
|
|
18
|
+
* 'local' })` ultimately talks to. `runtimeId` identifies the owning
|
|
19
|
+
* runtime so the dispatcher can route `stop` back to it.
|
|
20
|
+
*/
|
|
21
|
+
export interface LocalRuntimeHandle {
|
|
22
|
+
endpoint: string;
|
|
23
|
+
runtimeId: string;
|
|
24
|
+
/**
|
|
25
|
+
* Tasks this started runtime can serve (e.g. ['text-generation',
|
|
26
|
+
* 'embeddings']). Lets `runtime.inference` consumers know what a handle
|
|
27
|
+
* supports before calling `infer`. See docs/model-runtimes.md §9.2.
|
|
28
|
+
*/
|
|
29
|
+
tasks?: models.ModelTask[];
|
|
30
|
+
/**
|
|
31
|
+
* Host-assigned managed-process id, when the runtime spawned its server
|
|
32
|
+
* via `runtime.host.servers` (so `stop` can target the exact process).
|
|
33
|
+
*/
|
|
34
|
+
serverId?: string;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* A pluggable local model execution engine. Constructed eagerly;
|
|
39
|
+
* `punica.*` must only be dereferenced lazily inside method bodies
|
|
40
|
+
* (mirrors the facade lazy-init invariant).
|
|
41
|
+
*/
|
|
42
|
+
export interface ModelRuntime {
|
|
43
|
+
/** Stable id, e.g. "gguf", "transformers", "onnx". */
|
|
44
|
+
id: string;
|
|
45
|
+
/** Format-based dispatch test against a model record. */
|
|
46
|
+
canRun(record: models.LocalModelRecord): boolean;
|
|
47
|
+
/** Start the engine for this record and return a live endpoint. */
|
|
48
|
+
start(record: models.LocalModelRecord): Promise<LocalRuntimeHandle>;
|
|
49
|
+
/**
|
|
50
|
+
* Stop a handle previously returned by `start`. Handle-scoped (not
|
|
51
|
+
* global) so multiple live models can be stopped independently.
|
|
52
|
+
*/
|
|
53
|
+
stop(handle: LocalRuntimeHandle): Promise<void>;
|
|
54
|
+
/**
|
|
55
|
+
* Optional task-general inference for non-chat tasks (embeddings /
|
|
56
|
+
* vision / ASR / rerank). Engines that only serve chat omit this; the
|
|
57
|
+
* `runtime.inference` dispatcher errors clearly when a runtime lacks it
|
|
58
|
+
* or does not support the requested task. Chat/text-generation stays on
|
|
59
|
+
* `runtime.llm.chat({ route: 'local' })`. See docs/model-runtimes.md §9.2.
|
|
60
|
+
*/
|
|
61
|
+
infer?(req: inference.InferRequest): Promise<inference.InferResponse>;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Registry of available `ModelRuntime` engines. The dispatcher selects
|
|
66
|
+
* the first runtime whose `canRun(record)` is true.
|
|
67
|
+
*/
|
|
68
|
+
export interface ModelRuntimeRegistry {
|
|
69
|
+
register(runtime: ModelRuntime): void;
|
|
70
|
+
get(id: string): ModelRuntime | undefined;
|
|
71
|
+
list(): ModelRuntime[];
|
|
72
|
+
/** First registered runtime whose `canRun(record)` is true. */
|
|
73
|
+
select(record: models.LocalModelRecord): ModelRuntime | undefined;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Dispatcher facade surfaced on the runtime namespace. Picks a runtime
|
|
78
|
+
* for a record and starts it; routes `stop` back to the owning runtime
|
|
79
|
+
* via `handle.runtimeId`.
|
|
80
|
+
*/
|
|
81
|
+
export interface ModelRuntimesApi {
|
|
82
|
+
registry: ModelRuntimeRegistry;
|
|
83
|
+
/** Select a runtime for `record` and start it. Throws if none match. */
|
|
84
|
+
start(record: models.LocalModelRecord): Promise<LocalRuntimeHandle>;
|
|
85
|
+
/** Stop via the runtime that owns `handle`. Throws if unregistered. */
|
|
86
|
+
stop(handle: LocalRuntimeHandle): Promise<void>;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
}
|