@punica/editor 1.26.1 → 1.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@punica/editor",
3
- "version": "1.26.1",
3
+ "version": "1.28.0",
4
4
  "description": "Punica Editor",
5
5
  "private": false,
6
6
  "type": "module",
@@ -26,6 +26,7 @@
26
26
  "check:permissions": "node scripts/check-permissions.mjs",
27
27
  "check:empty-states": "node scripts/check-empty-states.mjs",
28
28
  "check:capability-handlers": "node scripts/check-capability-handlers.mjs",
29
+ "check:instruction-classes": "node scripts/check-instruction-classes.mjs",
29
30
  "report:capability-actions": "node scripts/report-capability-actions.mjs",
30
31
  "test": "vitest run",
31
32
  "test:watch": "vitest",
@@ -23,16 +23,42 @@ declare module 'punica' {
23
23
  maxTokens?: number;
24
24
  /** Response format: 'json_object' for structured output, 'text' for free-form */
25
25
  responseFormat?: 'json_object' | 'text';
26
+ /**
27
+ * How much of `maxTokens` this class may spend on reasoning.
28
+ *
29
+ * Absent means `none`, which is what twelve of the fleet's fifteen
30
+ * classes want: `maxTokens` reaches a local server as `num_predict`
31
+ * and caps thinking and answer together, so a formatting job that
32
+ * declares 400 tokens against a reasoning core returns nothing.
33
+ * Declare `auto` to keep the pre-1.27.0 wire and let the model think.
34
+ */
35
+ reasoning?: runtime.LlmReasoningMode;
26
36
  /**
27
37
  * Preferred model identifier for this instruction class (e.g. 'llama-3.2-8b-instruct-q4_k_m').
28
38
  * Used by expert routing to select a single expert model.
39
+ *
40
+ * **Reaches a model only when NO provider chain applies.** The chain —
41
+ * the class's own, or the default one the active AI core registers —
42
+ * is resolved first and replaces the single-model path entirely, and a
43
+ * host may additionally distrust a `modelId` on a request that named
44
+ * no target (the Electron host does, so a local GGUF name is never
45
+ * sent to a remote vendor). To say where a class runs, declare
46
+ * `providerChain`. Four shipped classes named a model installed
47
+ * nowhere here for months without anything noticing, which is why
48
+ * `scripts/check-instruction-classes.mjs` now refuses the field in an
49
+ * `.ic.yaml`.
29
50
  */
30
51
  expertModelId?: string;
31
52
  /**
32
- * Optional static risk level override for this instruction class.
33
- * When not provided, risk is determined dynamically via LLM-based assessment.
53
+ * This class's risk level, which decides single- vs multi-expert
54
+ * routing.
55
+ *
56
+ * Declare one. `assess` — and, for classes written before it existed,
57
+ * omitting the field — asks the kernel to work it out, and that is a
58
+ * full extra `llm.chat` round trip on every invocation whose answer
59
+ * the chain path then discards unread.
34
60
  */
35
- riskLevel?: 'low' | 'medium' | 'high';
61
+ riskLevel?: 'low' | 'medium' | 'high' | 'assess';
36
62
  /**
37
63
  * Optional explicit list of expert model identifiers for multi-expert scenarios.
38
64
  * When provided, auto-selection can use or extend this list.
@@ -106,6 +132,8 @@ declare module 'punica' {
106
132
  options?: {
107
133
  temperature?: number;
108
134
  maxTokens?: number;
135
+ /** Per-call override of the class's reasoning mode. */
136
+ reasoning?: runtime.LlmReasoningMode;
109
137
  timeoutMs?: number;
110
138
  /** Skip cache lookup (force fresh response) */
111
139
  skipCache?: boolean;
@@ -313,6 +313,49 @@ declare module 'punica' {
313
313
 
314
314
  /** The installed event exporter — never undefined; a no-op by default. */
315
315
  getEventTelemetryExporter: () => kernel.EventTelemetryExporter;
316
+
317
+ /**
318
+ * Install the development response cache, so a repeated LLM request is
319
+ * answered from the last answer instead of from the model.
320
+ *
321
+ * **There is no substrate default and the absent state is the shipped
322
+ * one**: with nothing installed every call reaches a model, which is
323
+ * how the product has behaved since the previous cache was deleted for
324
+ * leaking answers across tenants. That is the opposite default from the
325
+ * cost aggregator on purpose — an aggregator nobody enabled records
326
+ * numbers nobody reads, while a cache nobody enabled changes what the
327
+ * product answers.
328
+ *
329
+ * A hit skips the provider call and nothing else: the remote-policy
330
+ * gate runs first and is evaluated against the provider the stored
331
+ * answer names, so a hit is governed exactly as the call it replaces.
332
+ * It reports `usage` as zeros and `LlmCallMeta.cacheHit: true`.
333
+ *
334
+ * Pass `undefined` to turn it off; the store it replaces keeps its
335
+ * entries until someone calls `clear()`.
336
+ */
337
+ setLlmResponseCache: (cache: LlmResponseCache | undefined) => void;
338
+
339
+ /**
340
+ * The installed cache, or `undefined` when none is — which is also how
341
+ * a host asks it for `stats()`.
342
+ *
343
+ * On the facade because the setter does not survive without it: a
344
+ * written value that no published code reads is what the build
345
+ * eliminates, and `setApprovalTokenSigner` once shipped compiled down
346
+ * to an empty function for exactly that reason.
347
+ */
348
+ getLlmResponseCache: () => LlmResponseCache | undefined;
349
+
350
+ /**
351
+ * The substrate's store: bounded, expiring, process-local. Constructing
352
+ * one is on the facade beside the setter so that turning the cache on
353
+ * is one line — a host that had to write its own store first would be
354
+ * looking at a seam with no callers.
355
+ */
356
+ InMemoryLlmResponseCache: new (
357
+ options?: InMemoryLlmResponseCacheOptions
358
+ ) => LlmResponseCache;
316
359
  }
317
360
 
318
361
  /**
@@ -10,6 +10,31 @@ declare module 'punica' {
10
10
 
11
11
  export type LlmRoutePreference = 'auto' | 'local' | 'remote';
12
12
 
13
+ /**
14
+ * How much of the output budget a reasoning model may spend on thinking.
15
+ *
16
+ * `maxTokens` reaches a local server as `num_predict`, which caps thinking
17
+ * and answer TOGETHER — so a class that declares a small budget against a
18
+ * reasoning core spends all of it thinking and returns nothing. Measured
19
+ * 2026-09-01: a 125-character tutor hint costs 2,138 output tokens on
20
+ * `qwen3.5:9b`, against a class that declares 400.
21
+ *
22
+ * - `none` — ask the provider not to think. This is the DEFAULT for an
23
+ * instruction class that declares nothing, because twelve of the fleet's
24
+ * fifteen classes are formatting or extraction jobs.
25
+ * - `auto` — send nothing and let the provider decide, which is the wire
26
+ * exactly as it was before this field existed.
27
+ *
28
+ * There is deliberately no `extended`: no class asks for it today, and a
29
+ * value that cannot be measured is a claim rather than a capability. It
30
+ * arrives with the measurement that needs it.
31
+ *
32
+ * Only the local route and the `ollama` vendor put anything on the wire
33
+ * for `none`; for the others "off" is already the default and an
34
+ * unrecognised argument is an error, so nothing is sent.
35
+ */
36
+ export type LlmReasoningMode = 'none' | 'auto';
37
+
13
38
  export interface LlmMessage {
14
39
  role: LlmRole;
15
40
  content: string;
@@ -136,6 +161,28 @@ declare module 'punica' {
136
161
  * loop); translated to vendor syntax host-side.
137
162
  */
138
163
  cacheHints?: LlmCacheHints;
164
+ /**
165
+ * How much of `maxTokens` the provider may spend on reasoning. Absent is
166
+ * read as `none` by the LLM capability perimeter; a provider adapter
167
+ * that receives `undefined` sends nothing.
168
+ */
169
+ reasoning?: LlmReasoningMode;
170
+ /**
171
+ * Bypass the development response cache for this call — go to the
172
+ * model, and store what comes back.
173
+ *
174
+ * It skips the READ, not the write. A flag that skipped both would
175
+ * leave the stale entry in place for the next caller, so an entry could
176
+ * never be refreshed — only waited out.
177
+ *
178
+ * Only meaningful while a host has installed a cache
179
+ * (`setLlmResponseCache`); with none installed every call already
180
+ * reaches the model and this changes nothing. `kernel.llm`'s
181
+ * `options.skipCache` is forwarded here, which is what finally gave
182
+ * that flag a reader: it was accepted and ignored from the day the old
183
+ * cache was deleted until 2026-09-02.
184
+ */
185
+ skipCache?: boolean;
139
186
  }
140
187
 
141
188
  export interface LlmUsage {
@@ -495,6 +542,70 @@ declare module 'punica' {
495
542
  stopLocalServer(): Promise<void>;
496
543
  }
497
544
 
545
+ /**
546
+ * One stored answer.
547
+ */
548
+ export interface LlmResponseCacheEntry {
549
+ /** The response as the provider returned it, `usage` included. */
550
+ response: LlmChatResponse;
551
+ /** Epoch ms — the store's own expiry clock. */
552
+ storedAtMs: number;
553
+ }
554
+
555
+ export interface LlmResponseCacheStats {
556
+ entries: number;
557
+ hits: number;
558
+ misses: number;
559
+ stores: number;
560
+ /**
561
+ * Tokens the hits did not spend, summed from each entry's own
562
+ * `usage.totalTokens` as it was served. Absent usage counts as 0, so
563
+ * this is a floor.
564
+ */
565
+ tokensSaved: number;
566
+ }
567
+
568
+ /**
569
+ * The development response cache (LLMOps 2.6).
570
+ *
571
+ * **The substrate computes the key; a store only holds what it is
572
+ * handed.** That division is the whole point of the type: the cache this
573
+ * replaces was keyed on `instructionClass:userPrompt`, which named
574
+ * neither the workspace, the user, the model nor the temperature, and two
575
+ * tenants asking one question shared an answer. A host store keyed by the
576
+ * host would put that decision back where no substrate test can see it.
577
+ *
578
+ * `get` and `set` may be async so a host can back this with disk or a
579
+ * shared service. A host store that is SHARED between people must be
580
+ * paired with `setHostIdentity`, because an unset identity keys every one
581
+ * of them as the same anonymous subject.
582
+ *
583
+ * See `docs/llm-response-cache.md`.
584
+ */
585
+ export interface LlmResponseCache {
586
+ get(
587
+ key: string
588
+ ):
589
+ | LlmResponseCacheEntry
590
+ | undefined
591
+ | Promise<LlmResponseCacheEntry | undefined>;
592
+ set(key: string, entry: LlmResponseCacheEntry): void | Promise<void>;
593
+ clear(): void | Promise<void>;
594
+ stats(): LlmResponseCacheStats;
595
+ }
596
+
597
+ export interface InMemoryLlmResponseCacheOptions {
598
+ /** LRU cap. Defaults to 200 — a development session, not a corpus. */
599
+ maxEntries?: number;
600
+ /**
601
+ * Entry lifetime in ms. Defaults to one hour. The prompt is in the key,
602
+ * so this is not about edited prompts; it is about the case the key
603
+ * cannot see — the same `modelId` pointing at different weights after a
604
+ * pull.
605
+ */
606
+ ttlMs?: number;
607
+ }
608
+
498
609
  export interface LlmApi {
499
610
  chat(req: LlmChatRequest): Promise<LlmChatResponse>;
500
611
  /**
@@ -22,9 +22,17 @@ declare module 'punica' {
22
22
  endpoint: string;
23
23
  runtimeId: string;
24
24
  /**
25
- * Tasks this started runtime can serve (e.g. ['text-generation',
26
- * 'embeddings']). Lets `runtime.inference` consumers know what a handle
27
- * supports before calling `infer`. See docs/model-runtimes.md §9.2.
25
+ * Tasks THIS started runtime can serve (e.g. ['text-generation',
26
+ * 'embeddings']), when they differ from what the engine can serve in
27
+ * general — a llama.cpp server started without `--embedding` is the
28
+ * case this exists for. `runtime.inference.infer` refuses a task that
29
+ * is not in the list.
30
+ *
31
+ * Absent means "not stated", not "none": the engine's static
32
+ * `ModelRuntime.tasks` is consulted next, and if that is absent too the
33
+ * request proceeds and the engine answers for itself. That ordering is
34
+ * what keeps runtimes written before either field kept working.
35
+ * See docs/model-runtimes.md §9.2.
28
36
  */
29
37
  tasks?: models.ModelTask[];
30
38
  /**
@@ -42,6 +50,19 @@ declare module 'punica' {
42
50
  export interface ModelRuntime {
43
51
  /** Stable id, e.g. "gguf", "transformers", "onnx". */
44
52
  id: string;
53
+ /**
54
+ * What this engine can serve, readable WITHOUT starting it.
55
+ *
56
+ * The per-handle `tasks` could only ever be read after a model was
57
+ * loaded, so "can this engine embed?" was a question with no cheap
58
+ * answer — a picker had to start a runtime to find out. Declare the
59
+ * engine's general capability here and narrow it per handle only when
60
+ * a particular start is more limited.
61
+ *
62
+ * An engine with no `infer` serves chat through
63
+ * `runtime.llm.chat({ route: 'local' })` and should still say so.
64
+ */
65
+ tasks?: models.ModelTask[];
45
66
  /** Format-based dispatch test against a model record. */
46
67
  canRun(record: models.LocalModelRecord): boolean;
47
68
  /** Start the engine for this record and return a live endpoint. */