@punica/editor 1.27.0 → 1.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@punica/editor",
3
- "version": "1.27.0",
3
+ "version": "1.28.0",
4
4
  "description": "Punica Editor",
5
5
  "private": false,
6
6
  "type": "module",
@@ -26,6 +26,7 @@
26
26
  "check:permissions": "node scripts/check-permissions.mjs",
27
27
  "check:empty-states": "node scripts/check-empty-states.mjs",
28
28
  "check:capability-handlers": "node scripts/check-capability-handlers.mjs",
29
+ "check:instruction-classes": "node scripts/check-instruction-classes.mjs",
29
30
  "report:capability-actions": "node scripts/report-capability-actions.mjs",
30
31
  "test": "vitest run",
31
32
  "test:watch": "vitest",
@@ -36,13 +36,29 @@ declare module 'punica' {
36
36
  /**
37
37
  * Preferred model identifier for this instruction class (e.g. 'llama-3.2-8b-instruct-q4_k_m').
38
38
  * Used by expert routing to select a single expert model.
39
+ *
40
+ * **Reaches a model only when NO provider chain applies.** The chain —
41
+ * the class's own, or the default one the active AI core registers —
42
+ * is resolved first and replaces the single-model path entirely, and a
43
+ * host may additionally distrust a `modelId` on a request that named
44
+ * no target (the Electron host does, so a local GGUF name is never
45
+ * sent to a remote vendor). To say where a class runs, declare
46
+ * `providerChain`. Four shipped classes named a model installed
47
+ * nowhere here for months without anything noticing, which is why
48
+ * `scripts/check-instruction-classes.mjs` now refuses the field in an
49
+ * `.ic.yaml`.
39
50
  */
40
51
  expertModelId?: string;
41
52
  /**
42
- * Optional static risk level override for this instruction class.
43
- * When not provided, risk is determined dynamically via LLM-based assessment.
53
+ * This class's risk level, which decides single- vs multi-expert
54
+ * routing.
55
+ *
56
+ * Declare one. `assess` — and, for classes written before it existed,
57
+ * omitting the field — asks the kernel to work it out, and that is a
58
+ * full extra `llm.chat` round trip on every invocation whose answer
59
+ * the chain path then discards unread.
44
60
  */
45
- riskLevel?: 'low' | 'medium' | 'high';
61
+ riskLevel?: 'low' | 'medium' | 'high' | 'assess';
46
62
  /**
47
63
  * Optional explicit list of expert model identifiers for multi-expert scenarios.
48
64
  * When provided, auto-selection can use or extend this list.
@@ -313,6 +313,49 @@ declare module 'punica' {
313
313
 
314
314
  /** The installed event exporter — never undefined; a no-op by default. */
315
315
  getEventTelemetryExporter: () => kernel.EventTelemetryExporter;
316
+
317
+ /**
318
+ * Install the development response cache, so a repeated LLM request is
319
+ * answered from the last answer instead of from the model.
320
+ *
321
+ * **There is no substrate default and the absent state is the shipped
322
+ * one**: with nothing installed every call reaches a model, which is
323
+ * how the product has behaved since the previous cache was deleted for
324
+ * leaking answers across tenants. That is the opposite default from the
325
+ * cost aggregator on purpose — an aggregator nobody enabled records
326
+ * numbers nobody reads, while a cache nobody enabled changes what the
327
+ * product answers.
328
+ *
329
+ * A hit skips the provider call and nothing else: the remote-policy
330
+ * gate runs first and is evaluated against the provider the stored
331
+ * answer names, so a hit is governed exactly as the call it replaces.
332
+ * It reports `usage` as zeros and `LlmCallMeta.cacheHit: true`.
333
+ *
334
+ * Pass `undefined` to turn it off; the store it replaces keeps its
335
+ * entries until someone calls `clear()`.
336
+ */
337
+ setLlmResponseCache: (cache: LlmResponseCache | undefined) => void;
338
+
339
+ /**
340
+ * The installed cache, or `undefined` when none is — which is also how
341
+ * a host asks it for `stats()`.
342
+ *
343
+ * On the facade because the setter does not survive without it: a
344
+ * written value that no published code reads is what the build
345
+ * eliminates, and `setApprovalTokenSigner` once shipped compiled down
346
+ * to an empty function for exactly that reason.
347
+ */
348
+ getLlmResponseCache: () => LlmResponseCache | undefined;
349
+
350
+ /**
351
+ * The substrate's store: bounded, expiring, process-local. Constructing
352
+ * one is on the facade beside the setter so that turning the cache on
353
+ * is one line — a host that had to write its own store first would be
354
+ * looking at a seam with no callers.
355
+ */
356
+ InMemoryLlmResponseCache: new (
357
+ options?: InMemoryLlmResponseCacheOptions
358
+ ) => LlmResponseCache;
316
359
  }
317
360
 
318
361
  /**
@@ -167,6 +167,22 @@ declare module 'punica' {
167
167
  * that receives `undefined` sends nothing.
168
168
  */
169
169
  reasoning?: LlmReasoningMode;
170
+ /**
171
+ * Bypass the development response cache for this call — go to the
172
+ * model, and store what comes back.
173
+ *
174
+ * It skips the READ, not the write. A flag that skipped both would
175
+ * leave the stale entry in place for the next caller, so an entry could
176
+ * never be refreshed — only waited out.
177
+ *
178
+ * Only meaningful while a host has installed a cache
179
+ * (`setLlmResponseCache`); with none installed every call already
180
+ * reaches the model and this changes nothing. `kernel.llm`'s
181
+ * `options.skipCache` is forwarded here, which is what finally gave
182
+ * that flag a reader: it was accepted and ignored from the day the old
183
+ * cache was deleted until 2026-09-02.
184
+ */
185
+ skipCache?: boolean;
170
186
  }
171
187
 
172
188
  export interface LlmUsage {
@@ -526,6 +542,70 @@ declare module 'punica' {
526
542
  stopLocalServer(): Promise<void>;
527
543
  }
528
544
 
545
+ /**
546
+ * One stored answer.
547
+ */
548
+ export interface LlmResponseCacheEntry {
549
+ /** The response as the provider returned it, `usage` included. */
550
+ response: LlmChatResponse;
551
+ /** Epoch ms — the store's own expiry clock. */
552
+ storedAtMs: number;
553
+ }
554
+
555
+ export interface LlmResponseCacheStats {
556
+ entries: number;
557
+ hits: number;
558
+ misses: number;
559
+ stores: number;
560
+ /**
561
+ * Tokens the hits did not spend, summed from each entry's own
562
+ * `usage.totalTokens` as it was served. Absent usage counts as 0, so
563
+ * this is a floor.
564
+ */
565
+ tokensSaved: number;
566
+ }
567
+
568
+ /**
569
+ * The development response cache (LLMOps 2.6).
570
+ *
571
+ * **The substrate computes the key; a store only holds what it is
572
+ * handed.** That division is the whole point of the type: the cache this
573
+ * replaces was keyed on `instructionClass:userPrompt`, which named
574
+ * neither the workspace, the user, the model nor the temperature, and two
575
+ * tenants asking one question shared an answer. A host store keyed by the
576
+ * host would put that decision back where no substrate test can see it.
577
+ *
578
+ * `get` and `set` may be async so a host can back this with disk or a
579
+ * shared service. A host store that is SHARED between people must be
580
+ * paired with `setHostIdentity`, because an unset identity keys every one
581
+ * of them as the same anonymous subject.
582
+ *
583
+ * See `docs/llm-response-cache.md`.
584
+ */
585
+ export interface LlmResponseCache {
586
+ get(
587
+ key: string
588
+ ):
589
+ | LlmResponseCacheEntry
590
+ | undefined
591
+ | Promise<LlmResponseCacheEntry | undefined>;
592
+ set(key: string, entry: LlmResponseCacheEntry): void | Promise<void>;
593
+ clear(): void | Promise<void>;
594
+ stats(): LlmResponseCacheStats;
595
+ }
596
+
597
+ export interface InMemoryLlmResponseCacheOptions {
598
+ /** LRU cap. Defaults to 200 — a development session, not a corpus. */
599
+ maxEntries?: number;
600
+ /**
601
+ * Entry lifetime in ms. Defaults to one hour. The prompt is in the key,
602
+ * so this is not about edited prompts; it is about the case the key
603
+ * cannot see — the same `modelId` pointing at different weights after a
604
+ * pull.
605
+ */
606
+ ttlMs?: number;
607
+ }
608
+
529
609
  export interface LlmApi {
530
610
  chat(req: LlmChatRequest): Promise<LlmChatResponse>;
531
611
  /**
@@ -22,9 +22,17 @@ declare module 'punica' {
22
22
  endpoint: string;
23
23
  runtimeId: string;
24
24
  /**
25
- * Tasks this started runtime can serve (e.g. ['text-generation',
26
- * 'embeddings']). Lets `runtime.inference` consumers know what a handle
27
- * supports before calling `infer`. See docs/model-runtimes.md §9.2.
25
+ * Tasks THIS started runtime can serve (e.g. ['text-generation',
26
+ * 'embeddings']), when they differ from what the engine can serve in
27
+ * general — a llama.cpp server started without `--embedding` is the
28
+ * case this exists for. `runtime.inference.infer` refuses a task that
29
+ * is not in the list.
30
+ *
31
+ * Absent means "not stated", not "none": the engine's static
32
+ * `ModelRuntime.tasks` is consulted next, and if that is absent too the
33
+ * request proceeds and the engine answers for itself. That ordering is
34
+ * what keeps runtimes written before either field kept working.
35
+ * See docs/model-runtimes.md §9.2.
28
36
  */
29
37
  tasks?: models.ModelTask[];
30
38
  /**
@@ -42,6 +50,19 @@ declare module 'punica' {
42
50
  export interface ModelRuntime {
43
51
  /** Stable id, e.g. "gguf", "transformers", "onnx". */
44
52
  id: string;
53
+ /**
54
+ * What this engine can serve, readable WITHOUT starting it.
55
+ *
56
+ * The per-handle `tasks` could only ever be read after a model was
57
+ * loaded, so "can this engine embed?" was a question with no cheap
58
+ * answer — a picker had to start a runtime to find out. Declare the
59
+ * engine's general capability here and narrow it per handle only when
60
+ * a particular start is more limited.
61
+ *
62
+ * An engine with no `infer` serves chat through
63
+ * `runtime.llm.chat({ route: 'local' })` and should still say so.
64
+ */
65
+ tasks?: models.ModelTask[];
45
66
  /** Format-based dispatch test against a model record. */
46
67
  canRun(record: models.LocalModelRecord): boolean;
47
68
  /** Start the engine for this record and return a live endpoint. */