@vibgrate/cli 2026.722.1 → 2026.727.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/DOCS.md +418 -196
  2. package/README.md +34 -7
  3. package/dist/baseline-K7YBFKZI.js +7 -0
  4. package/dist/{baseline-J44M7MWP.js.map → baseline-K7YBFKZI.js.map} +1 -1
  5. package/dist/chunk-3BVSDMHV.js +2496 -0
  6. package/dist/chunk-3BVSDMHV.js.map +1 -0
  7. package/dist/{chunk-PY3DNX5H.js → chunk-3Q7QBWFW.js} +13 -2
  8. package/dist/chunk-3Q7QBWFW.js.map +1 -0
  9. package/dist/chunk-4CQ3MGNN.js +136 -0
  10. package/dist/chunk-4CQ3MGNN.js.map +1 -0
  11. package/dist/chunk-5I4VSBWQ.js +764 -0
  12. package/dist/chunk-5I4VSBWQ.js.map +1 -0
  13. package/dist/chunk-5Q54GEHM.js +29 -0
  14. package/dist/chunk-5Q54GEHM.js.map +1 -0
  15. package/dist/{chunk-XKXKWYJF.js → chunk-65MEOA4E.js} +21 -18
  16. package/dist/chunk-65MEOA4E.js.map +1 -0
  17. package/dist/{chunk-WNIIKCNF.js → chunk-6IOQJGFT.js} +29 -238
  18. package/dist/chunk-6IOQJGFT.js.map +1 -0
  19. package/dist/{chunk-CS37OBE3.js → chunk-7JHPQ6WF.js} +791 -783
  20. package/dist/chunk-7JHPQ6WF.js.map +1 -0
  21. package/dist/chunk-BPF3YX2U.js +64 -0
  22. package/dist/chunk-BPF3YX2U.js.map +1 -0
  23. package/dist/chunk-BWPBB45M.js +88 -0
  24. package/dist/chunk-BWPBB45M.js.map +1 -0
  25. package/dist/chunk-C4CKTAZP.js +1695 -0
  26. package/dist/chunk-C4CKTAZP.js.map +1 -0
  27. package/dist/chunk-EDAAG7BL.js +612 -0
  28. package/dist/chunk-EDAAG7BL.js.map +1 -0
  29. package/dist/chunk-EDBV33B7.js +57 -0
  30. package/dist/chunk-EDBV33B7.js.map +1 -0
  31. package/dist/chunk-HABUT7HW.js +233 -0
  32. package/dist/chunk-HABUT7HW.js.map +1 -0
  33. package/dist/{chunk-EXQV26MF.js → chunk-HYZVHJ3Q.js} +3 -3
  34. package/dist/{chunk-EXQV26MF.js.map → chunk-HYZVHJ3Q.js.map} +1 -1
  35. package/dist/chunk-IS7VTH2Q.js +627 -0
  36. package/dist/chunk-IS7VTH2Q.js.map +1 -0
  37. package/dist/chunk-IZDH67DX.js +289 -0
  38. package/dist/chunk-IZDH67DX.js.map +1 -0
  39. package/dist/chunk-J7LKEB2R.js +51 -0
  40. package/dist/chunk-J7LKEB2R.js.map +1 -0
  41. package/dist/chunk-JIBOJTGS.js +313 -0
  42. package/dist/chunk-JIBOJTGS.js.map +1 -0
  43. package/dist/chunk-JZYGEZTU.js +82 -0
  44. package/dist/chunk-JZYGEZTU.js.map +1 -0
  45. package/dist/chunk-K7REHDQS.js +776 -0
  46. package/dist/chunk-K7REHDQS.js.map +1 -0
  47. package/dist/chunk-KHJO4D5N.js +108 -0
  48. package/dist/chunk-KHJO4D5N.js.map +1 -0
  49. package/dist/chunk-NDMDSPFX.js +69 -0
  50. package/dist/chunk-NDMDSPFX.js.map +1 -0
  51. package/dist/chunk-OOIG3RUM.js +276 -0
  52. package/dist/chunk-OOIG3RUM.js.map +1 -0
  53. package/dist/{chunk-6CXTPC74.js → chunk-OXQNQW5J.js} +27 -6
  54. package/dist/chunk-OXQNQW5J.js.map +1 -0
  55. package/dist/chunk-PMYWZRRS.js +62 -0
  56. package/dist/chunk-PMYWZRRS.js.map +1 -0
  57. package/dist/chunk-PXHNTC6D.js +382 -0
  58. package/dist/chunk-PXHNTC6D.js.map +1 -0
  59. package/dist/{chunk-VXT6LUGP.js → chunk-S6CYDZEW.js} +1395 -1975
  60. package/dist/chunk-S6CYDZEW.js.map +1 -0
  61. package/dist/chunk-SZLF66FJ.js +843 -0
  62. package/dist/chunk-SZLF66FJ.js.map +1 -0
  63. package/dist/chunk-VXGV6NTD.js +476 -0
  64. package/dist/chunk-VXGV6NTD.js.map +1 -0
  65. package/dist/{chunk-JSNHZCIV.js → chunk-WLETRGAC.js} +16 -9
  66. package/dist/chunk-WLETRGAC.js.map +1 -0
  67. package/dist/{chunk-2EVQKZTO.js → chunk-YDAT7LD2.js} +190 -13
  68. package/dist/chunk-YDAT7LD2.js.map +1 -0
  69. package/dist/cli.js +3451 -509
  70. package/dist/cli.js.map +1 -1
  71. package/dist/execution-env-3BN7Z737.js +3 -0
  72. package/dist/execution-env-3BN7Z737.js.map +1 -0
  73. package/dist/federation-EPZDUWCK.js +6 -0
  74. package/dist/federation-EPZDUWCK.js.map +1 -0
  75. package/dist/git-ref-HPRLUFFW.js +3 -0
  76. package/dist/git-ref-HPRLUFFW.js.map +1 -0
  77. package/dist/graph-backend-ZMSYOZMW.js +10 -0
  78. package/dist/graph-backend-ZMSYOZMW.js.map +1 -0
  79. package/dist/index.d.ts +432 -126
  80. package/dist/index.js +14 -7
  81. package/dist/index.js.map +1 -1
  82. package/dist/{interactive-736PBC7J.js → interactive-ICL7D7LO.js} +185 -365
  83. package/dist/interactive-ICL7D7LO.js.map +1 -0
  84. package/dist/llm-host-2RAYQSDQ.js +8 -0
  85. package/dist/llm-host-2RAYQSDQ.js.map +1 -0
  86. package/dist/load-OHVKDX4R.js +7 -0
  87. package/dist/load-OHVKDX4R.js.map +1 -0
  88. package/dist/local-runtime-OJFLX2HI.js +4 -0
  89. package/dist/local-runtime-OJFLX2HI.js.map +1 -0
  90. package/dist/model-execution-profile-MQ2QCP24.js +3 -0
  91. package/dist/model-execution-profile-MQ2QCP24.js.map +1 -0
  92. package/dist/model-orchestrator-FFTZI3U2.js +8 -0
  93. package/dist/model-orchestrator-FFTZI3U2.js.map +1 -0
  94. package/dist/parse-worker.d.ts +1 -1
  95. package/dist/parse-worker.js +1 -1
  96. package/dist/paths-MCH75O3G.js +5 -0
  97. package/dist/paths-MCH75O3G.js.map +1 -0
  98. package/dist/resolve-gguf-4UPIGPEE.js +7 -0
  99. package/dist/resolve-gguf-4UPIGPEE.js.map +1 -0
  100. package/dist/runtime-session-IDU5BDUJ.js +17 -0
  101. package/dist/runtime-session-IDU5BDUJ.js.map +1 -0
  102. package/dist/session-L2OSDRPI.js +10 -0
  103. package/dist/{session-POJUDFPH.js.map → session-L2OSDRPI.js.map} +1 -1
  104. package/dist/{stream-json-Q2JGWPHB.js → stream-json-F74W5FCK.js} +16 -6
  105. package/dist/stream-json-F74W5FCK.js.map +1 -0
  106. package/dist/{types-Jl6RJ175.d.ts → types-ZMg1mSuu.d.ts} +3 -1
  107. package/dist/vgd-EGBQD7BW.js +16 -0
  108. package/dist/vgd-EGBQD7BW.js.map +1 -0
  109. package/package.json +9 -6
  110. package/dist/baseline-J44M7MWP.js +0 -7
  111. package/dist/chunk-2EVQKZTO.js.map +0 -1
  112. package/dist/chunk-6CXTPC74.js.map +0 -1
  113. package/dist/chunk-CS37OBE3.js.map +0 -1
  114. package/dist/chunk-GGJZA3Q6.js +0 -961
  115. package/dist/chunk-GGJZA3Q6.js.map +0 -1
  116. package/dist/chunk-JBXNQCGE.js +0 -484
  117. package/dist/chunk-JBXNQCGE.js.map +0 -1
  118. package/dist/chunk-JSNHZCIV.js.map +0 -1
  119. package/dist/chunk-PY3DNX5H.js.map +0 -1
  120. package/dist/chunk-VXT6LUGP.js.map +0 -1
  121. package/dist/chunk-WNIIKCNF.js.map +0 -1
  122. package/dist/chunk-XKXKWYJF.js.map +0 -1
  123. package/dist/interactive-736PBC7J.js.map +0 -1
  124. package/dist/session-POJUDFPH.js +0 -6
  125. package/dist/stream-json-Q2JGWPHB.js.map +0 -1
@@ -0,0 +1,776 @@
1
+ import { trieHas, trieHasPrefix, maskAllowedNextChars, trieEnumerateWords, extractIdentifiers, SessionPrefixCursor, KvBlockRegistry, annotateUnknownIdentifiers, scanIdentifiersAgainstTrie, contentHash, longestWarmPrefix } from './chunk-JIBOJTGS.js';
2
+
3
+ // src/runtime/llm-host/binding-capabilities.ts
4
+ function probeBindingCapabilities(lib, model, sequence, options = {}) {
5
+ const l = lib;
6
+ const m = model;
7
+ const seq = sequence;
8
+ const env = options.env ?? process.env;
9
+ const tokenBias = !!(l && typeof l.TokenBias === "function");
10
+ const llamaGrammar = !!(l && typeof l.LlamaGrammar === "function");
11
+ const vocabIterate = !!(m && typeof m.iterateAllTokens === "function" && typeof m.detokenize === "function");
12
+ const tokenize = !!(m && typeof m.tokenize === "function");
13
+ const evaluateWithoutGenerate = !!(seq && typeof seq.evaluateWithoutGeneratingNewTokens === "function");
14
+ const envOnToken = /^(1|true|yes)$/i.test(String(env.VG_LLM_ON_TOKEN ?? env.VIBGRATE_LLM_ON_TOKEN ?? ""));
15
+ const envCustom = /^(1|true|yes)$/i.test(String(env.VG_LLM_CUSTOM_SAMPLER ?? ""));
16
+ const onTokenCallback = !!(options.assumeOnToken || envOnToken);
17
+ const customSampler = !!(options.assumeCustomSampler || envCustom);
18
+ const summary = [];
19
+ if (tokenBias) summary.push("TokenBias");
20
+ if (llamaGrammar) summary.push("LlamaGrammar");
21
+ if (vocabIterate) summary.push("vocabIterate");
22
+ if (tokenize) summary.push("tokenize");
23
+ if (evaluateWithoutGenerate) summary.push("evaluateWithoutGenerate");
24
+ if (onTokenCallback) summary.push("onToken");
25
+ if (customSampler) summary.push("customSampler");
26
+ if (summary.length === 0) summary.push("none");
27
+ return {
28
+ tokenBias,
29
+ llamaGrammar,
30
+ vocabIterate,
31
+ tokenize,
32
+ evaluateWithoutGenerate,
33
+ onTokenCallback,
34
+ customSampler,
35
+ summary
36
+ };
37
+ }
38
+ function supportsDynamicIdentSampler(caps) {
39
+ return caps.onTokenCallback || caps.customSampler;
40
+ }
41
+ function supportsKvPrefixEvaluate(caps) {
42
+ return caps.tokenize && caps.evaluateWithoutGenerate;
43
+ }
44
+
45
+ // src/runtime/llm-host/compose-constraints.ts
46
+ function composePromptConstraints(input) {
47
+ const promptOpts = { ...input.base ?? {} };
48
+ if (input.grammar != null) promptOpts.grammar = input.grammar;
49
+ if (input.tokenBias != null) promptOpts.tokenBias = input.tokenBias;
50
+ if (input.dynamic) {
51
+ for (const [k, v] of Object.entries(input.dynamic)) {
52
+ if (v !== void 0) promptOpts[k] = v;
53
+ }
54
+ }
55
+ return {
56
+ promptOpts,
57
+ layers: {
58
+ grammar: input.grammar != null,
59
+ tokenBias: input.tokenBias != null,
60
+ dynamic: !!(input.dynamic && Object.keys(input.dynamic).length > 0)
61
+ }
62
+ };
63
+ }
64
+
65
+ // src/runtime/llm-host/grammar.ts
66
+ async function attachGrammar(lib, gbnf) {
67
+ if (!gbnf?.trim()) {
68
+ return { value: null, method: "none", applied: false, reason: "no GBNF provided" };
69
+ }
70
+ const l = lib;
71
+ if (!l) {
72
+ return { value: null, method: "none", applied: false, reason: "no llama binding loaded" };
73
+ }
74
+ if (l.LlamaGrammar) {
75
+ try {
76
+ if (typeof l.LlamaGrammar === "function") {
77
+ let llama;
78
+ try {
79
+ llama = typeof l.getLlama === "function" ? await l.getLlama() : void 0;
80
+ } catch {
81
+ llama = void 0;
82
+ }
83
+ const attempts = [
84
+ () => new l.LlamaGrammar(llama, { grammar: gbnf }),
85
+ () => new l.LlamaGrammar({ grammar: gbnf }),
86
+ () => new l.LlamaGrammar(gbnf)
87
+ ];
88
+ for (const attempt of attempts) {
89
+ try {
90
+ const g = attempt();
91
+ if (g) return { value: g, method: "LlamaGrammar", applied: true };
92
+ } catch {
93
+ }
94
+ }
95
+ }
96
+ if (typeof l.LlamaGrammar.fromString === "function") {
97
+ const g = await l.LlamaGrammar.fromString(gbnf);
98
+ if (g) return { value: g, method: "LlamaGrammar", applied: true };
99
+ }
100
+ } catch (e) {
101
+ return {
102
+ value: null,
103
+ method: "none",
104
+ applied: false,
105
+ reason: `LlamaGrammar failed: ${e instanceof Error ? e.message : String(e)}`
106
+ };
107
+ }
108
+ }
109
+ if (typeof l.createGrammar === "function") {
110
+ try {
111
+ const g = await l.createGrammar(gbnf);
112
+ if (g) return { value: g, method: "createGrammar", applied: true };
113
+ } catch (e) {
114
+ return {
115
+ value: null,
116
+ method: "none",
117
+ applied: false,
118
+ reason: `createGrammar failed: ${e instanceof Error ? e.message : String(e)}`
119
+ };
120
+ }
121
+ }
122
+ return {
123
+ value: gbnf,
124
+ method: "string-fallback",
125
+ applied: false,
126
+ reason: "binding has no LlamaGrammar/createGrammar; raw GBNF string is available but not verified as constrained"
127
+ };
128
+ }
129
+ function assertGrammarOrThrow(result, opts = {}) {
130
+ if (!opts.requireGrammar) return;
131
+ if (result.applied) return;
132
+ if (opts.allowStringFallback && result.method === "string-fallback" && result.value) {
133
+ return;
134
+ }
135
+ throw new Error(
136
+ `constrained decoding required but grammar could not be attached (${result.method}${result.reason ? `: ${result.reason}` : ""}). Use a node-llama-cpp build with LlamaGrammar, or choose a Code Mode without constrainedDecoding (e.g. Flow), or set VG_ALLOW_GRAMMAR_STRING_FALLBACK=1 only if your binding documents raw GBNF support.`
137
+ );
138
+ }
139
+ function allowGrammarStringFallback(env = process.env) {
140
+ return env.VG_ALLOW_GRAMMAR_STRING_FALLBACK === "1" || env.VIBGRATE_ALLOW_GRAMMAR_STRING_FALLBACK === "1";
141
+ }
142
+
143
+ // src/runtime/llm-host/identifier-sampler.ts
144
+ var IDENT_START = /[A-Za-z_]/;
145
+ var IDENT_CONT = /[A-Za-z0-9_]/;
146
+ var COMPLETE_IDENT = /^[A-Za-z_][A-Za-z0-9_]{2,}$/;
147
+ var STOP = /* @__PURE__ */ new Set([
148
+ "the",
149
+ "and",
150
+ "for",
151
+ "var",
152
+ "let",
153
+ "const",
154
+ "function",
155
+ "class",
156
+ "return",
157
+ "import",
158
+ "export",
159
+ "from",
160
+ "async",
161
+ "await",
162
+ "true",
163
+ "false",
164
+ "null",
165
+ "undefined",
166
+ "this",
167
+ "new",
168
+ "typeof",
169
+ "interface",
170
+ "type",
171
+ "string",
172
+ "number",
173
+ "boolean",
174
+ "void",
175
+ "any",
176
+ "unknown",
177
+ "Promise",
178
+ "Array",
179
+ "Object",
180
+ "Error",
181
+ "console",
182
+ "process",
183
+ "require",
184
+ "module",
185
+ "exports",
186
+ "default",
187
+ "extends",
188
+ "implements",
189
+ "public",
190
+ "private",
191
+ "protected",
192
+ "static",
193
+ "readonly",
194
+ "if",
195
+ "else",
196
+ "while",
197
+ "switch",
198
+ "case",
199
+ "break",
200
+ "continue",
201
+ "try",
202
+ "catch",
203
+ "finally",
204
+ "throw",
205
+ "with",
206
+ "yield",
207
+ "of",
208
+ "in",
209
+ "as",
210
+ "is",
211
+ "get",
212
+ "set",
213
+ "super",
214
+ "delete",
215
+ "void",
216
+ "enum",
217
+ "package"
218
+ ]);
219
+ function createIdentStreamState() {
220
+ return { mode: "free", prefix: "" };
221
+ }
222
+ function isStop(s) {
223
+ return STOP.has(s) || STOP.has(s.toLowerCase());
224
+ }
225
+ function stepIdentStream(state, tokenText, trie) {
226
+ if (!tokenText) {
227
+ return { next: state, allowed: true, allowedNextChars: allowedChars(state, trie) };
228
+ }
229
+ if (!trie) {
230
+ return { next: trackOnly(state, tokenText), allowed: true, allowedNextChars: null };
231
+ }
232
+ if (state.mode === "in_ident") {
233
+ const joined = state.prefix + tokenText;
234
+ const m = joined.match(/^[A-Za-z_][A-Za-z0-9_]*/);
235
+ const head = m?.[0] ?? "";
236
+ if (head.length >= 3 && !isStop(head) && !trieHas(trie, head) && !trieHasPrefix(trie, head)) {
237
+ return {
238
+ next: state,
239
+ allowed: false,
240
+ reason: `identifier '${head}' not in graph trie`,
241
+ allowedNextChars: maskAllowedNextChars(trie, state.prefix)
242
+ };
243
+ }
244
+ } else if (COMPLETE_IDENT.test(tokenText) && !isStop(tokenText)) {
245
+ if (!trieHas(trie, tokenText) && !trieHasPrefix(trie, tokenText)) {
246
+ return {
247
+ next: state,
248
+ allowed: false,
249
+ reason: `identifier '${tokenText}' not in graph trie`,
250
+ allowedNextChars: null
251
+ };
252
+ }
253
+ }
254
+ const next = trackOnly(state, tokenText);
255
+ return {
256
+ next,
257
+ allowed: true,
258
+ allowedNextChars: allowedChars(next, trie)
259
+ };
260
+ }
261
+ function trackOnly(state, tokenText) {
262
+ let mode = state.mode;
263
+ let prefix = state.prefix;
264
+ for (const ch of tokenText) {
265
+ if (mode === "free") {
266
+ if (IDENT_START.test(ch)) {
267
+ mode = "in_ident";
268
+ prefix = ch;
269
+ }
270
+ } else if (IDENT_CONT.test(ch)) {
271
+ prefix += ch;
272
+ } else {
273
+ mode = "free";
274
+ prefix = "";
275
+ }
276
+ }
277
+ return { mode, prefix };
278
+ }
279
+ function allowedChars(state, trie) {
280
+ if (!trie || state.mode !== "in_ident") return null;
281
+ return maskAllowedNextChars(trie, state.prefix);
282
+ }
283
+ function filterTokenTexts(state, candidates, trie) {
284
+ return candidates.filter((t) => stepIdentStream(state, t, trie).allowed);
285
+ }
286
+ function dynamicAllowedNextChars(state, trie) {
287
+ return allowedChars(state, trie);
288
+ }
289
+ function attachDynamicIdentFilter(promptOpts, trie, capabilities) {
290
+ const state = createIdentStreamState();
291
+ if (!trie) return { attached: false, method: "none", state };
292
+ if (capabilities.onTokenCallback) {
293
+ let stream = createIdentStreamState();
294
+ promptOpts.onToken = (tokenText) => {
295
+ const step = stepIdentStream(stream, tokenText, trie);
296
+ if (!step.allowed) return false;
297
+ stream = step.next;
298
+ return true;
299
+ };
300
+ return { attached: true, method: "onToken", state: stream };
301
+ }
302
+ if (capabilities.customSampler) {
303
+ let stream = createIdentStreamState();
304
+ promptOpts.customSampler = {
305
+ accept: (tokenText) => {
306
+ const step = stepIdentStream(stream, tokenText, trie);
307
+ if (!step.allowed) return false;
308
+ stream = step.next;
309
+ return true;
310
+ }
311
+ };
312
+ return { attached: true, method: "customSampler", state: stream };
313
+ }
314
+ return { attached: false, method: "none", state };
315
+ }
316
+
317
+ // src/runtime/llm-host/logit-mask.ts
318
+ var COMPLETE_IDENT2 = /^[A-Za-z_][A-Za-z0-9_]{2,}$/;
319
+ function buildIdentifierTokenBias(lib, model, trie, options = {}) {
320
+ if (!trie) {
321
+ return { applied: false, method: "none", tokenBias: null, boostedTokens: 0, suppressedTokens: 0, reason: "no trie" };
322
+ }
323
+ const l = lib;
324
+ const m = model;
325
+ if (!l || typeof l.TokenBias !== "function") {
326
+ if (m && typeof m.iterateAllTokens === "function") {
327
+ return {
328
+ applied: false,
329
+ method: "capability",
330
+ tokenBias: null,
331
+ boostedTokens: 0,
332
+ suppressedTokens: 0,
333
+ reason: "TokenBias class missing (binding can iterate tokens)"
334
+ };
335
+ }
336
+ return {
337
+ applied: false,
338
+ method: "none",
339
+ tokenBias: null,
340
+ boostedTokens: 0,
341
+ suppressedTokens: 0,
342
+ reason: "TokenBias unavailable"
343
+ };
344
+ }
345
+ const tokenizer = m?.tokenizer ?? m;
346
+ if (!tokenizer) {
347
+ return {
348
+ applied: false,
349
+ method: "none",
350
+ tokenBias: null,
351
+ boostedTokens: 0,
352
+ suppressedTokens: 0,
353
+ reason: "model.tokenizer missing"
354
+ };
355
+ }
356
+ let bias;
357
+ try {
358
+ bias = new l.TokenBias(tokenizer);
359
+ } catch (e) {
360
+ return {
361
+ applied: false,
362
+ method: "none",
363
+ tokenBias: null,
364
+ boostedTokens: 0,
365
+ suppressedTokens: 0,
366
+ reason: e instanceof Error ? e.message : "TokenBias construct failed"
367
+ };
368
+ }
369
+ const boost = options.boost ?? 0.35;
370
+ const maxBoostIds = options.maxBoostIds ?? 2048;
371
+ const words = trieEnumerateWords(trie, maxBoostIds);
372
+ let boostedTokens = 0;
373
+ for (const word of words) {
374
+ if (word.length < 3) continue;
375
+ const tokens = tokenizeBestEffort(m, word);
376
+ for (const tok of tokens) {
377
+ try {
378
+ if (typeof bias.set === "function") {
379
+ try {
380
+ bias.set(tok, boost);
381
+ } catch {
382
+ bias.set(tok, { logit: boost });
383
+ }
384
+ boostedTokens++;
385
+ }
386
+ } catch {
387
+ }
388
+ }
389
+ }
390
+ let suppressedTokens = 0;
391
+ if (options.suppressUnknown !== false && typeof m.iterateAllTokens === "function" && typeof m.detokenize === "function") {
392
+ const maxScan = options.maxVocabScan ?? 5e4;
393
+ let scanned = 0;
394
+ try {
395
+ for (const token of m.iterateAllTokens()) {
396
+ if (scanned++ >= maxScan) break;
397
+ let text = "";
398
+ try {
399
+ text = m.detokenize([token]);
400
+ } catch {
401
+ continue;
402
+ }
403
+ if (!text || !COMPLETE_IDENT2.test(text)) continue;
404
+ if (trieHas(trie, text) || trieHasPrefix(trie, text)) continue;
405
+ try {
406
+ bias.set(token, "never");
407
+ suppressedTokens++;
408
+ } catch {
409
+ try {
410
+ bias.set(token, -0.95);
411
+ suppressedTokens++;
412
+ } catch {
413
+ }
414
+ }
415
+ }
416
+ } catch {
417
+ }
418
+ }
419
+ const method = boostedTokens > 0 || suppressedTokens > 0 ? suppressedTokens > 0 ? "TokenBias" : "TokenBias-boost" : "none";
420
+ return {
421
+ applied: method !== "none",
422
+ method: method === "none" ? "none" : method,
423
+ tokenBias: method === "none" ? null : bias,
424
+ boostedTokens,
425
+ suppressedTokens,
426
+ reason: method === "none" ? "no tokens adjusted" : void 0
427
+ };
428
+ }
429
+ function tokenizeBestEffort(model, text) {
430
+ try {
431
+ if (typeof model.tokenize === "function") {
432
+ const t = model.tokenize(text);
433
+ if (Array.isArray(t)) return t;
434
+ if (t && typeof t[Symbol.iterator] === "function") return [...t];
435
+ }
436
+ } catch {
437
+ }
438
+ try {
439
+ if (model.tokenizer && typeof model.tokenizer.encode === "function") {
440
+ const t = model.tokenizer.encode(text);
441
+ if (Array.isArray(t)) return t;
442
+ }
443
+ } catch {
444
+ }
445
+ return [];
446
+ }
447
+ function planNextCharMask(trie, partialIdent) {
448
+ return maskAllowedNextChars(trie, partialIdent);
449
+ }
450
+ function tokenCompatibleWithTrie(tokenText, trie, openPrefix = "") {
451
+ const joined = openPrefix + tokenText;
452
+ if (trieHas(trie, joined) || trieHasPrefix(trie, joined)) return true;
453
+ const ids = extractIdentifiers(joined);
454
+ if (ids.length === 0) return true;
455
+ return ids.every((id) => trieHas(trie, id) || trieHasPrefix(trie, id));
456
+ }
457
+ function trieFingerprint(trie, limit = 256) {
458
+ const words = trieEnumerateWords(trie, limit);
459
+ return `${words.length}:${words.slice(0, 32).join("|")}`;
460
+ }
461
+
462
+ // src/runtime/llm-host/speculative.ts
463
+ function rankDraftCandidates(candidates, messages, max = 3) {
464
+ if (!candidates?.length) return [];
465
+ const last = messages[messages.length - 1]?.content?.toLowerCase() ?? "";
466
+ const ranked = [];
467
+ for (const raw of candidates) {
468
+ const d = raw.trim();
469
+ if (d.length < 8) continue;
470
+ let score = 0;
471
+ const reasons = [];
472
+ const head = d.slice(0, 48).toLowerCase().replace(/\s+/g, " ");
473
+ const tokens = head.split(/[^a-z0-9_]+/).filter((t) => t.length > 3);
474
+ for (const t of tokens) {
475
+ if (last.includes(t)) {
476
+ score += 3;
477
+ reasons.push(`token:${t}`);
478
+ }
479
+ }
480
+ if (/\b(signature|import|export function|export class|implements|extends)\b/i.test(last)) {
481
+ score += 2;
482
+ reasons.push("quotational-ask");
483
+ }
484
+ if (/^export\s+(async\s+)?(function|class|const|type|interface)\b/.test(d)) {
485
+ score += 1;
486
+ reasons.push("export-shape");
487
+ }
488
+ if (score > 0) {
489
+ ranked.push({ text: d, score, reason: reasons.join(",") || "match" });
490
+ }
491
+ }
492
+ if (ranked.length === 0 && /\b(signature|import|export function|export class)\b/i.test(last)) {
493
+ for (const d of candidates.slice(0, max)) {
494
+ if (d.trim().length >= 8) ranked.push({ text: d.trim(), score: 1, reason: "fallback-quotational" });
495
+ }
496
+ }
497
+ return ranked.sort((a, b) => b.score - a.score || b.text.length - a.text.length).slice(0, max);
498
+ }
499
+ function pickBestDraft(candidates, messages) {
500
+ return rankDraftCandidates(candidates, messages, 1)[0]?.text ?? null;
501
+ }
502
+
503
+ // src/runtime/llm-host/session.ts
504
+ var globalPool = /* @__PURE__ */ new Map();
505
+ function sessionKey(modelPath) {
506
+ return `embedded:${modelPath}`;
507
+ }
508
+ function clearHostSessionPool() {
509
+ globalPool.clear();
510
+ }
511
+ function hostSessionPoolSize() {
512
+ return globalPool.size;
513
+ }
514
+ function listHostSessions() {
515
+ return [...globalPool.values()].map((s) => ({
516
+ key: s.key,
517
+ modelPath: s.modelPath,
518
+ loads: s.loads,
519
+ lastUsed: s.lastUsed,
520
+ kvBlocks: s.kv.size(),
521
+ cursorBlocks: s.cursor.evaluatedHashes.length,
522
+ bindingCaps: s.caps?.summary ?? null
523
+ })).sort((a, b) => b.lastUsed - a.lastUsed);
524
+ }
525
+ function dropHostSession(modelPathOrKey) {
526
+ const key = modelPathOrKey.startsWith("embedded:") ? modelPathOrKey : sessionKey(modelPathOrKey);
527
+ return globalPool.delete(key);
528
+ }
529
+ async function acquireHostSession(lib, modelPath) {
530
+ const key = sessionKey(modelPath);
531
+ const existing = globalPool.get(key);
532
+ if (existing) {
533
+ existing.lastUsed = Date.now();
534
+ return existing;
535
+ }
536
+ const l = lib;
537
+ if (!l || typeof l.getLlama !== "function") {
538
+ throw new Error("llm-host session: invalid node-llama-cpp module (getLlama missing)");
539
+ }
540
+ const llama = await l.getLlama();
541
+ const model = await llama.loadModel({ modelPath });
542
+ const context = await model.createContext();
543
+ const sequence = context.getSequence();
544
+ const { LlamaChatSession } = l;
545
+ if (typeof LlamaChatSession !== "function") {
546
+ throw new Error("llm-host session: LlamaChatSession missing");
547
+ }
548
+ const chat = new LlamaChatSession({ contextSequence: sequence });
549
+ const sess = {
550
+ key,
551
+ lib: l,
552
+ modelPath,
553
+ model,
554
+ context,
555
+ sequence,
556
+ chat,
557
+ kv: new KvBlockRegistry(),
558
+ cursor: new SessionPrefixCursor(),
559
+ lastUsed: Date.now(),
560
+ loads: 1,
561
+ logitMaskCache: /* @__PURE__ */ new Map(),
562
+ caps: null
563
+ };
564
+ globalPool.set(key, sess);
565
+ return sess;
566
+ }
567
+ function evictIdleSessions(maxSessions = 2, maxIdleMs = 30 * 6e4, now = Date.now()) {
568
+ if (globalPool.size <= maxSessions) {
569
+ let dropped2 = 0;
570
+ for (const [k, s] of globalPool) {
571
+ if (now - s.lastUsed > maxIdleMs) {
572
+ globalPool.delete(k);
573
+ dropped2++;
574
+ }
575
+ }
576
+ return dropped2;
577
+ }
578
+ const ordered = [...globalPool.entries()].sort((a, b) => a[1].lastUsed - b[1].lastUsed);
579
+ let dropped = 0;
580
+ while (globalPool.size > maxSessions && ordered.length) {
581
+ const [k] = ordered.shift();
582
+ globalPool.delete(k);
583
+ dropped++;
584
+ }
585
+ return dropped;
586
+ }
587
+ async function generateOnSession(session, messages, extras = {}) {
588
+ const t0 = Date.now();
589
+ const warmStart = session.loads > 0 && session.lastUsed > 0;
590
+ session.loads++;
591
+ session.lastUsed = Date.now();
592
+ const env = extras.env ?? process.env;
593
+ const caps = session.caps ?? probeBindingCapabilities(session.lib, session.model, session.sequence, {
594
+ assumeOnToken: extras.assumeOnToken,
595
+ assumeCustomSampler: extras.assumeCustomSampler,
596
+ env
597
+ });
598
+ session.caps = caps;
599
+ const segments = extras.promptSegments ?? messages.map((m, i) => ({
600
+ kind: m.role === "system" ? "system" : i === messages.length - 1 ? "user" : "context",
601
+ text: m.content
602
+ }));
603
+ const kvPlan = session.kv.plan(segments, session.cursor);
604
+ const prefixReuse = await tryReuseWarmPrefix(session, segments, kvPlan, caps);
605
+ const ranked = rankDraftCandidates(extras.draftCandidates, messages, 3);
606
+ let draftAcceptedChars = 0;
607
+ let draftCandidatesTried = 0;
608
+ let text = "";
609
+ const stringOk = allowGrammarStringFallback(env);
610
+ const grammar = await attachGrammar(session.lib, extras.grammar);
611
+ assertGrammarOrThrow(grammar, {
612
+ requireGrammar: extras.requireGrammar,
613
+ allowStringFallback: stringOk
614
+ });
615
+ let promptGrammar = null;
616
+ if (grammar.applied) promptGrammar = grammar.value;
617
+ else if (grammar.method === "string-fallback" && grammar.value && (stringOk || !extras.requireGrammar)) {
618
+ promptGrammar = grammar.value;
619
+ }
620
+ const mask = resolveLogitMask(session, extras.identifierTrie, extras.suppressUnknownLogits !== false);
621
+ const baseOpts = {
622
+ temperature: extras.temperature ?? 0
623
+ };
624
+ if (typeof extras.maxTokens === "number") baseOpts.maxTokens = extras.maxTokens;
625
+ const dynBag = {};
626
+ let dynamicAttached = false;
627
+ let dynMethod = "none";
628
+ if (extras.identifierTrie && supportsDynamicIdentSampler(caps)) {
629
+ const dyn = attachDynamicIdentFilter(dynBag, extras.identifierTrie, {
630
+ onTokenCallback: caps.onTokenCallback,
631
+ customSampler: caps.customSampler
632
+ });
633
+ dynamicAttached = dyn.attached;
634
+ dynMethod = dyn.method;
635
+ }
636
+ const composed = composePromptConstraints({
637
+ base: baseOpts,
638
+ grammar: promptGrammar,
639
+ tokenBias: mask.tokenBias,
640
+ dynamic: dynamicAttached ? dynBag : null
641
+ });
642
+ const samplerMethod = resolveSamplerMethod(mask, dynamicAttached, dynMethod);
643
+ if (ranked.length) {
644
+ let accepted = null;
645
+ for (const d of ranked) {
646
+ draftCandidatesTried++;
647
+ if (await tryAcceptDraft(session, d.text, caps)) {
648
+ accepted = d.text;
649
+ break;
650
+ }
651
+ }
652
+ if (accepted) {
653
+ draftAcceptedChars = accepted.length;
654
+ text = accepted;
655
+ const continueMsgs = [
656
+ ...messages,
657
+ { role: "assistant", content: accepted },
658
+ { role: "user", content: "Continue from the draft above without repeating it." }
659
+ ];
660
+ const cont = await sessionPrompt(session, continueMsgs, composed.promptOpts);
661
+ text = accepted + (cont.startsWith(accepted) ? cont.slice(accepted.length) : cont);
662
+ } else {
663
+ text = await sessionPrompt(session, messages, composed.promptOpts);
664
+ }
665
+ } else {
666
+ text = await sessionPrompt(session, messages, composed.promptOpts);
667
+ }
668
+ session.kv.commit(segments);
669
+ session.cursor.commit(kvPlan.order);
670
+ let unknownIdentifiers;
671
+ if (extras.identifierTrie) {
672
+ if (extras.annotateUnknownIdentifiers !== false) {
673
+ const ann = annotateUnknownIdentifiers(text, extras.identifierTrie);
674
+ text = ann.text;
675
+ unknownIdentifiers = ann.unknown.length ? ann.unknown : void 0;
676
+ } else {
677
+ const scan = scanIdentifiersAgainstTrie(text, extras.identifierTrie);
678
+ unknownIdentifiers = scan.unknown.length ? scan.unknown : void 0;
679
+ }
680
+ }
681
+ return {
682
+ text,
683
+ model: session.modelPath,
684
+ isolation: "embedded",
685
+ constrained: grammar.applied || !!promptGrammar && stringOk || dynamicAttached,
686
+ unknownIdentifiers,
687
+ draftAcceptedChars,
688
+ draftCandidatesTried,
689
+ kvPlan,
690
+ grammarApplied: grammar.applied || !!promptGrammar && stringOk && grammar.method === "string-fallback",
691
+ grammarMethod: grammar.method,
692
+ samplerHook: mask.applied || dynamicAttached,
693
+ samplerMethod,
694
+ samplerBoostedTokens: mask.boostedTokens,
695
+ samplerSuppressedTokens: mask.suppressedTokens,
696
+ dynamicSampler: dynamicAttached,
697
+ kvPrefixReusedTokens: prefixReuse.tokens,
698
+ kvDeltaTokens: kvPlan.deltaTokens,
699
+ bindingCaps: caps.summary,
700
+ constraintLayers: composed.layers,
701
+ warmStart,
702
+ latencyMs: Date.now() - t0
703
+ };
704
+ }
705
+ function resolveSamplerMethod(mask, dynamicAttached, dynMethod) {
706
+ if (mask.applied && dynamicAttached) return "TokenBias+dynamic";
707
+ if (dynamicAttached && dynMethod === "onToken") return "dynamic-onToken";
708
+ if (dynamicAttached && dynMethod === "customSampler") return "dynamic-customSampler";
709
+ return mask.method;
710
+ }
711
+ function resolveLogitMask(session, trie, suppressUnknown) {
712
+ if (!trie) {
713
+ return { applied: false, method: "none", tokenBias: null, boostedTokens: 0, suppressedTokens: 0 };
714
+ }
715
+ const fp = `${trieFingerprint(trie)}|sup=${suppressUnknown ? 1 : 0}`;
716
+ const cached = session.logitMaskCache.get(fp);
717
+ if (cached) return cached;
718
+ const built = buildIdentifierTokenBias(session.lib, session.model, trie, {
719
+ suppressUnknown
720
+ });
721
+ if (session.logitMaskCache.size > 8) session.logitMaskCache.clear();
722
+ session.logitMaskCache.set(fp, built);
723
+ return built;
724
+ }
725
+ async function sessionPrompt(session, messages, promptOpts) {
726
+ const prompt = messages.map((m) => `${m.role}: ${m.content}`).join("\n\n");
727
+ const text = await session.chat.prompt(prompt, promptOpts);
728
+ return typeof text === "string" ? text : String(text ?? "");
729
+ }
730
+ async function tryAcceptDraft(session, draft, caps) {
731
+ try {
732
+ const seq = session.sequence;
733
+ if (caps.evaluateWithoutGenerate && caps.tokenize && seq && typeof seq.evaluateWithoutGeneratingNewTokens === "function") {
734
+ const model = session.model;
735
+ if (model && typeof model.tokenize === "function") {
736
+ const tokens = model.tokenize(draft);
737
+ await seq.evaluateWithoutGeneratingNewTokens(tokens);
738
+ return true;
739
+ }
740
+ }
741
+ } catch {
742
+ }
743
+ return session.kv.isWarm(contentHash(draft));
744
+ }
745
+ async function tryReuseWarmPrefix(session, segments, plan, caps) {
746
+ const prefix = longestWarmPrefix(session.kv, segments);
747
+ if (prefix.count === 0 && (plan.cursorSkipBlocks ?? 0) === 0) {
748
+ return { tokens: 0, method: "none" };
749
+ }
750
+ const canEval = caps.tokenize && caps.evaluateWithoutGenerate;
751
+ try {
752
+ const seq = session.sequence;
753
+ const model = session.model;
754
+ if (canEval && seq && typeof seq.evaluateWithoutGeneratingNewTokens === "function" && model && typeof model.tokenize === "function") {
755
+ const delta = session.cursor.delta(plan.order, segments);
756
+ if (delta.skipBlocks > 0 && delta.evaluateText) {
757
+ const tokens2 = model.tokenize(delta.evaluateText);
758
+ await seq.evaluateWithoutGeneratingNewTokens(tokens2);
759
+ return { tokens: plan.hitTokens, method: "evaluate-delta" };
760
+ }
761
+ if (prefix.text) {
762
+ const tokens2 = model.tokenize(prefix.text);
763
+ await seq.evaluateWithoutGeneratingNewTokens(tokens2);
764
+ return { tokens: prefix.tokens, method: "evaluate-prefix" };
765
+ }
766
+ }
767
+ } catch {
768
+ }
769
+ const tokens = prefix.tokens || plan.hitTokens;
770
+ if (tokens > 0) return { tokens, method: "plan-only" };
771
+ return { tokens: 0, method: "none" };
772
+ }
773
+
774
+ export { acquireHostSession, allowGrammarStringFallback, assertGrammarOrThrow, attachDynamicIdentFilter, attachGrammar, buildIdentifierTokenBias, clearHostSessionPool, composePromptConstraints, createIdentStreamState, dropHostSession, dynamicAllowedNextChars, evictIdleSessions, filterTokenTexts, generateOnSession, hostSessionPoolSize, listHostSessions, pickBestDraft, planNextCharMask, probeBindingCapabilities, rankDraftCandidates, sessionKey, stepIdentStream, supportsDynamicIdentSampler, supportsKvPrefixEvaluate, tokenCompatibleWithTrie, trieFingerprint };
775
+ //# sourceMappingURL=chunk-K7REHDQS.js.map
776
+ //# sourceMappingURL=chunk-K7REHDQS.js.map