@punica/editor 1.26.1 → 1.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.bundle.esm.js +2 -2
- package/dist/index.bundle.esm.js.map +1 -1
- package/dist/index.bundle.umd.js +2 -2
- package/dist/index.bundle.umd.js.map +1 -1
- package/package.json +2 -1
- package/types/punica.module.kernel.llm.d.ts +31 -3
- package/types/punica.module.runtime.api.d.ts +43 -0
- package/types/punica.module.runtime.llm.d.ts +111 -0
- package/types/punica.module.runtime.modelRuntimes.d.ts +24 -3
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@punica/editor",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.28.0",
|
|
4
4
|
"description": "Punica Editor",
|
|
5
5
|
"private": false,
|
|
6
6
|
"type": "module",
|
|
@@ -26,6 +26,7 @@
|
|
|
26
26
|
"check:permissions": "node scripts/check-permissions.mjs",
|
|
27
27
|
"check:empty-states": "node scripts/check-empty-states.mjs",
|
|
28
28
|
"check:capability-handlers": "node scripts/check-capability-handlers.mjs",
|
|
29
|
+
"check:instruction-classes": "node scripts/check-instruction-classes.mjs",
|
|
29
30
|
"report:capability-actions": "node scripts/report-capability-actions.mjs",
|
|
30
31
|
"test": "vitest run",
|
|
31
32
|
"test:watch": "vitest",
|
|
@@ -23,16 +23,42 @@ declare module 'punica' {
|
|
|
23
23
|
maxTokens?: number;
|
|
24
24
|
/** Response format: 'json_object' for structured output, 'text' for free-form */
|
|
25
25
|
responseFormat?: 'json_object' | 'text';
|
|
26
|
+
/**
|
|
27
|
+
* How much of `maxTokens` this class may spend on reasoning.
|
|
28
|
+
*
|
|
29
|
+
* Absent means `none`, which is what twelve of the fleet's fifteen
|
|
30
|
+
* classes want: `maxTokens` reaches a local server as `num_predict`
|
|
31
|
+
* and caps thinking and answer together, so a formatting job that
|
|
32
|
+
* declares 400 tokens against a reasoning core returns nothing.
|
|
33
|
+
* Declare `auto` to keep the pre-1.27.0 wire and let the model think.
|
|
34
|
+
*/
|
|
35
|
+
reasoning?: runtime.LlmReasoningMode;
|
|
26
36
|
/**
|
|
27
37
|
* Preferred model identifier for this instruction class (e.g. 'llama-3.2-8b-instruct-q4_k_m').
|
|
28
38
|
* Used by expert routing to select a single expert model.
|
|
39
|
+
*
|
|
40
|
+
* **Reaches a model only when NO provider chain applies.** The chain —
|
|
41
|
+
* the class's own, or the default one the active AI core registers —
|
|
42
|
+
* is resolved first and replaces the single-model path entirely, and a
|
|
43
|
+
* host may additionally distrust a `modelId` on a request that named
|
|
44
|
+
* no target (the Electron host does, so a local GGUF name is never
|
|
45
|
+
* sent to a remote vendor). To say where a class runs, declare
|
|
46
|
+
* `providerChain`. Four shipped classes named a model installed
|
|
47
|
+
* nowhere here for months without anything noticing, which is why
|
|
48
|
+
* `scripts/check-instruction-classes.mjs` now refuses the field in an
|
|
49
|
+
* `.ic.yaml`.
|
|
29
50
|
*/
|
|
30
51
|
expertModelId?: string;
|
|
31
52
|
/**
|
|
32
|
-
*
|
|
33
|
-
*
|
|
53
|
+
* This class's risk level, which decides single- vs multi-expert
|
|
54
|
+
* routing.
|
|
55
|
+
*
|
|
56
|
+
* Declare one. `assess` — and, for classes written before it existed,
|
|
57
|
+
* omitting the field — asks the kernel to work it out, and that is a
|
|
58
|
+
* full extra `llm.chat` round trip on every invocation whose answer
|
|
59
|
+
* the chain path then discards unread.
|
|
34
60
|
*/
|
|
35
|
-
riskLevel?: 'low' | 'medium' | 'high';
|
|
61
|
+
riskLevel?: 'low' | 'medium' | 'high' | 'assess';
|
|
36
62
|
/**
|
|
37
63
|
* Optional explicit list of expert model identifiers for multi-expert scenarios.
|
|
38
64
|
* When provided, auto-selection can use or extend this list.
|
|
@@ -106,6 +132,8 @@ declare module 'punica' {
|
|
|
106
132
|
options?: {
|
|
107
133
|
temperature?: number;
|
|
108
134
|
maxTokens?: number;
|
|
135
|
+
/** Per-call override of the class's reasoning mode. */
|
|
136
|
+
reasoning?: runtime.LlmReasoningMode;
|
|
109
137
|
timeoutMs?: number;
|
|
110
138
|
/** Skip cache lookup (force fresh response) */
|
|
111
139
|
skipCache?: boolean;
|
|
@@ -313,6 +313,49 @@ declare module 'punica' {
|
|
|
313
313
|
|
|
314
314
|
/** The installed event exporter — never undefined; a no-op by default. */
|
|
315
315
|
getEventTelemetryExporter: () => kernel.EventTelemetryExporter;
|
|
316
|
+
|
|
317
|
+
/**
|
|
318
|
+
* Install the development response cache, so a repeated LLM request is
|
|
319
|
+
* answered from the last answer instead of from the model.
|
|
320
|
+
*
|
|
321
|
+
* **There is no substrate default and the absent state is the shipped
|
|
322
|
+
* one**: with nothing installed every call reaches a model, which is
|
|
323
|
+
* how the product has behaved since the previous cache was deleted for
|
|
324
|
+
* leaking answers across tenants. That is the opposite default from the
|
|
325
|
+
* cost aggregator on purpose — an aggregator nobody enabled records
|
|
326
|
+
* numbers nobody reads, while a cache nobody enabled changes what the
|
|
327
|
+
* product answers.
|
|
328
|
+
*
|
|
329
|
+
* A hit skips the provider call and nothing else: the remote-policy
|
|
330
|
+
* gate runs first and is evaluated against the provider the stored
|
|
331
|
+
* answer names, so a hit is governed exactly as the call it replaces.
|
|
332
|
+
* It reports `usage` as zeros and `LlmCallMeta.cacheHit: true`.
|
|
333
|
+
*
|
|
334
|
+
* Pass `undefined` to turn it off; the store it replaces keeps its
|
|
335
|
+
* entries until someone calls `clear()`.
|
|
336
|
+
*/
|
|
337
|
+
setLlmResponseCache: (cache: LlmResponseCache | undefined) => void;
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* The installed cache, or `undefined` when none is — which is also how
|
|
341
|
+
* a host asks it for `stats()`.
|
|
342
|
+
*
|
|
343
|
+
* On the facade because the setter does not survive without it: a
|
|
344
|
+
* written value that no published code reads is what the build
|
|
345
|
+
* eliminates, and `setApprovalTokenSigner` once shipped compiled down
|
|
346
|
+
* to an empty function for exactly that reason.
|
|
347
|
+
*/
|
|
348
|
+
getLlmResponseCache: () => LlmResponseCache | undefined;
|
|
349
|
+
|
|
350
|
+
/**
|
|
351
|
+
* The substrate's store: bounded, expiring, process-local. Constructing
|
|
352
|
+
* one is on the facade beside the setter so that turning the cache on
|
|
353
|
+
* is one line — a host that had to write its own store first would be
|
|
354
|
+
* looking at a seam with no callers.
|
|
355
|
+
*/
|
|
356
|
+
InMemoryLlmResponseCache: new (
|
|
357
|
+
options?: InMemoryLlmResponseCacheOptions
|
|
358
|
+
) => LlmResponseCache;
|
|
316
359
|
}
|
|
317
360
|
|
|
318
361
|
/**
|
|
@@ -10,6 +10,31 @@ declare module 'punica' {
|
|
|
10
10
|
|
|
11
11
|
export type LlmRoutePreference = 'auto' | 'local' | 'remote';
|
|
12
12
|
|
|
13
|
+
/**
|
|
14
|
+
* How much of the output budget a reasoning model may spend on thinking.
|
|
15
|
+
*
|
|
16
|
+
* `maxTokens` reaches a local server as `num_predict`, which caps thinking
|
|
17
|
+
* and answer TOGETHER — so a class that declares a small budget against a
|
|
18
|
+
* reasoning core spends all of it thinking and returns nothing. Measured
|
|
19
|
+
* 2026-09-01: a 125-character tutor hint costs 2,138 output tokens on
|
|
20
|
+
* `qwen3.5:9b`, against a class that declares 400.
|
|
21
|
+
*
|
|
22
|
+
* - `none` — ask the provider not to think. This is the DEFAULT for an
|
|
23
|
+
* instruction class that declares nothing, because twelve of the fleet's
|
|
24
|
+
* fifteen classes are formatting or extraction jobs.
|
|
25
|
+
* - `auto` — send nothing and let the provider decide, which is the wire
|
|
26
|
+
* exactly as it was before this field existed.
|
|
27
|
+
*
|
|
28
|
+
* There is deliberately no `extended`: no class asks for it today, and a
|
|
29
|
+
* value that cannot be measured is a claim rather than a capability. It
|
|
30
|
+
* arrives with the measurement that needs it.
|
|
31
|
+
*
|
|
32
|
+
* Only the local route and the `ollama` vendor put anything on the wire
|
|
33
|
+
* for `none`; for the others "off" is already the default and an
|
|
34
|
+
* unrecognised argument is an error, so nothing is sent.
|
|
35
|
+
*/
|
|
36
|
+
export type LlmReasoningMode = 'none' | 'auto';
|
|
37
|
+
|
|
13
38
|
export interface LlmMessage {
|
|
14
39
|
role: LlmRole;
|
|
15
40
|
content: string;
|
|
@@ -136,6 +161,28 @@ declare module 'punica' {
|
|
|
136
161
|
* loop); translated to vendor syntax host-side.
|
|
137
162
|
*/
|
|
138
163
|
cacheHints?: LlmCacheHints;
|
|
164
|
+
/**
|
|
165
|
+
* How much of `maxTokens` the provider may spend on reasoning. Absent is
|
|
166
|
+
* read as `none` by the LLM capability perimeter; a provider adapter
|
|
167
|
+
* that receives `undefined` sends nothing.
|
|
168
|
+
*/
|
|
169
|
+
reasoning?: LlmReasoningMode;
|
|
170
|
+
/**
|
|
171
|
+
* Bypass the development response cache for this call — go to the
|
|
172
|
+
* model, and store what comes back.
|
|
173
|
+
*
|
|
174
|
+
* It skips the READ, not the write. A flag that skipped both would
|
|
175
|
+
* leave the stale entry in place for the next caller, so an entry could
|
|
176
|
+
* never be refreshed — only waited out.
|
|
177
|
+
*
|
|
178
|
+
* Only meaningful while a host has installed a cache
|
|
179
|
+
* (`setLlmResponseCache`); with none installed every call already
|
|
180
|
+
* reaches the model and this changes nothing. `kernel.llm`'s
|
|
181
|
+
* `options.skipCache` is forwarded here, which is what finally gave
|
|
182
|
+
* that flag a reader: it was accepted and ignored from the day the old
|
|
183
|
+
* cache was deleted until 2026-09-02.
|
|
184
|
+
*/
|
|
185
|
+
skipCache?: boolean;
|
|
139
186
|
}
|
|
140
187
|
|
|
141
188
|
export interface LlmUsage {
|
|
@@ -495,6 +542,70 @@ declare module 'punica' {
|
|
|
495
542
|
stopLocalServer(): Promise<void>;
|
|
496
543
|
}
|
|
497
544
|
|
|
545
|
+
/**
|
|
546
|
+
* One stored answer.
|
|
547
|
+
*/
|
|
548
|
+
export interface LlmResponseCacheEntry {
|
|
549
|
+
/** The response as the provider returned it, `usage` included. */
|
|
550
|
+
response: LlmChatResponse;
|
|
551
|
+
/** Epoch ms — the store's own expiry clock. */
|
|
552
|
+
storedAtMs: number;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
export interface LlmResponseCacheStats {
|
|
556
|
+
entries: number;
|
|
557
|
+
hits: number;
|
|
558
|
+
misses: number;
|
|
559
|
+
stores: number;
|
|
560
|
+
/**
|
|
561
|
+
* Tokens the hits did not spend, summed from each entry's own
|
|
562
|
+
* `usage.totalTokens` as it was served. Absent usage counts as 0, so
|
|
563
|
+
* this is a floor.
|
|
564
|
+
*/
|
|
565
|
+
tokensSaved: number;
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
/**
|
|
569
|
+
* The development response cache (LLMOps 2.6).
|
|
570
|
+
*
|
|
571
|
+
* **The substrate computes the key; a store only holds what it is
|
|
572
|
+
* handed.** That division is the whole point of the type: the cache this
|
|
573
|
+
* replaces was keyed on `instructionClass:userPrompt`, which named
|
|
574
|
+
* neither the workspace, the user, the model nor the temperature, and two
|
|
575
|
+
* tenants asking one question shared an answer. A host store keyed by the
|
|
576
|
+
* host would put that decision back where no substrate test can see it.
|
|
577
|
+
*
|
|
578
|
+
* `get` and `set` may be async so a host can back this with disk or a
|
|
579
|
+
* shared service. A host store that is SHARED between people must be
|
|
580
|
+
* paired with `setHostIdentity`, because an unset identity keys every one
|
|
581
|
+
* of them as the same anonymous subject.
|
|
582
|
+
*
|
|
583
|
+
* See `docs/llm-response-cache.md`.
|
|
584
|
+
*/
|
|
585
|
+
export interface LlmResponseCache {
|
|
586
|
+
get(
|
|
587
|
+
key: string
|
|
588
|
+
):
|
|
589
|
+
| LlmResponseCacheEntry
|
|
590
|
+
| undefined
|
|
591
|
+
| Promise<LlmResponseCacheEntry | undefined>;
|
|
592
|
+
set(key: string, entry: LlmResponseCacheEntry): void | Promise<void>;
|
|
593
|
+
clear(): void | Promise<void>;
|
|
594
|
+
stats(): LlmResponseCacheStats;
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
export interface InMemoryLlmResponseCacheOptions {
|
|
598
|
+
/** LRU cap. Defaults to 200 — a development session, not a corpus. */
|
|
599
|
+
maxEntries?: number;
|
|
600
|
+
/**
|
|
601
|
+
* Entry lifetime in ms. Defaults to one hour. The prompt is in the key,
|
|
602
|
+
* so this is not about edited prompts; it is about the case the key
|
|
603
|
+
* cannot see — the same `modelId` pointing at different weights after a
|
|
604
|
+
* pull.
|
|
605
|
+
*/
|
|
606
|
+
ttlMs?: number;
|
|
607
|
+
}
|
|
608
|
+
|
|
498
609
|
export interface LlmApi {
|
|
499
610
|
chat(req: LlmChatRequest): Promise<LlmChatResponse>;
|
|
500
611
|
/**
|
|
@@ -22,9 +22,17 @@ declare module 'punica' {
|
|
|
22
22
|
endpoint: string;
|
|
23
23
|
runtimeId: string;
|
|
24
24
|
/**
|
|
25
|
-
* Tasks
|
|
26
|
-
* 'embeddings'])
|
|
27
|
-
*
|
|
25
|
+
* Tasks THIS started runtime can serve (e.g. ['text-generation',
|
|
26
|
+
* 'embeddings']), when they differ from what the engine can serve in
|
|
27
|
+
* general — a llama.cpp server started without `--embedding` is the
|
|
28
|
+
* case this exists for. `runtime.inference.infer` refuses a task that
|
|
29
|
+
* is not in the list.
|
|
30
|
+
*
|
|
31
|
+
* Absent means "not stated", not "none": the engine's static
|
|
32
|
+
* `ModelRuntime.tasks` is consulted next, and if that is absent too the
|
|
33
|
+
* request proceeds and the engine answers for itself. That ordering is
|
|
34
|
+
* what keeps runtimes written before either field kept working.
|
|
35
|
+
* See docs/model-runtimes.md §9.2.
|
|
28
36
|
*/
|
|
29
37
|
tasks?: models.ModelTask[];
|
|
30
38
|
/**
|
|
@@ -42,6 +50,19 @@ declare module 'punica' {
|
|
|
42
50
|
export interface ModelRuntime {
|
|
43
51
|
/** Stable id, e.g. "gguf", "transformers", "onnx". */
|
|
44
52
|
id: string;
|
|
53
|
+
/**
|
|
54
|
+
* What this engine can serve, readable WITHOUT starting it.
|
|
55
|
+
*
|
|
56
|
+
* The per-handle `tasks` could only ever be read after a model was
|
|
57
|
+
* loaded, so "can this engine embed?" was a question with no cheap
|
|
58
|
+
* answer — a picker had to start a runtime to find out. Declare the
|
|
59
|
+
* engine's general capability here and narrow it per handle only when
|
|
60
|
+
* a particular start is more limited.
|
|
61
|
+
*
|
|
62
|
+
* An engine with no `infer` serves chat through
|
|
63
|
+
* `runtime.llm.chat({ route: 'local' })` and should still say so.
|
|
64
|
+
*/
|
|
65
|
+
tasks?: models.ModelTask[];
|
|
45
66
|
/** Format-based dispatch test against a model record. */
|
|
46
67
|
canRun(record: models.LocalModelRecord): boolean;
|
|
47
68
|
/** Start the engine for this record and return a live endpoint. */
|