@tryhamster/gerbil 1.8.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/dist/browser/index.d.ts.map +1 -1
  2. package/dist/browser/index.js +10 -0
  3. package/dist/browser/index.js.map +1 -1
  4. package/dist/cli.mjs +7 -7
  5. package/dist/cli.mjs.map +1 -1
  6. package/dist/frameworks/express.mjs +1 -1
  7. package/dist/frameworks/fastify.mjs +1 -1
  8. package/dist/frameworks/hono.mjs +1 -1
  9. package/dist/frameworks/next.d.mts +2 -2
  10. package/dist/frameworks/next.mjs +1 -1
  11. package/dist/frameworks/trpc.mjs +1 -1
  12. package/dist/{gerbil-DU1aRO6v.d.mts → gerbil-CSk3AHNN.d.mts} +10 -9
  13. package/dist/{gerbil-DU1aRO6v.d.mts.map → gerbil-CSk3AHNN.d.mts.map} +1 -1
  14. package/dist/gerbil-DrV6iNcD.mjs +4 -0
  15. package/dist/{gerbil-Cgmb4Dit.mjs → gerbil-DtREprR_.mjs} +32 -10
  16. package/dist/gerbil-DtREprR_.mjs.map +1 -0
  17. package/dist/gpu/hooks.d.mts +16 -1
  18. package/dist/gpu/hooks.d.mts.map +1 -1
  19. package/dist/gpu/hooks.mjs +29 -4
  20. package/dist/gpu/hooks.mjs.map +1 -1
  21. package/dist/gpu/index.d.mts +1 -1
  22. package/dist/gpu/index.mjs +2 -2
  23. package/dist/{gpu-kQVLpV3n.mjs → gpu-NPdk7pmr.mjs} +221 -25
  24. package/dist/gpu-NPdk7pmr.mjs.map +1 -0
  25. package/dist/{index-ElJKy9i9.d.mts → index-B56xBiR2.d.mts} +96 -4
  26. package/dist/index-B56xBiR2.d.mts.map +1 -0
  27. package/dist/index.d.mts +2 -2
  28. package/dist/index.d.mts.map +1 -1
  29. package/dist/index.mjs +4 -4
  30. package/dist/integrations/ai-sdk.mjs +1 -1
  31. package/dist/integrations/langchain.mjs +1 -1
  32. package/dist/integrations/llamaindex.mjs +1 -1
  33. package/dist/integrations/mcp.d.mts +2 -2
  34. package/dist/integrations/mcp.mjs +4 -4
  35. package/dist/{mcp-DTusP2m5.mjs → mcp-XU4quJ9w.mjs} +3 -3
  36. package/dist/{mcp-DTusP2m5.mjs.map → mcp-XU4quJ9w.mjs.map} +1 -1
  37. package/dist/moonshine-stt-BRTnAoQ8.mjs +4 -0
  38. package/dist/{moonshine-stt-9wc6t11v.mjs → moonshine-stt-BYbYwSz2.mjs} +301 -13
  39. package/dist/moonshine-stt-BYbYwSz2.mjs.map +1 -0
  40. package/dist/{one-liner-Bk5x7gYH.mjs → one-liner-1WoJGLlZ.mjs} +2 -2
  41. package/dist/{one-liner-Bk5x7gYH.mjs.map → one-liner-1WoJGLlZ.mjs.map} +1 -1
  42. package/dist/{repl-DV6l-jT8.mjs → repl-B41yROa-.mjs} +3 -3
  43. package/dist/skills/index.d.mts +4 -4
  44. package/dist/skills/index.mjs +3 -3
  45. package/dist/{skills-BhcwnL2l.mjs → skills-C9RgR8qU.mjs} +2 -2
  46. package/dist/{skills-BhcwnL2l.mjs.map → skills-C9RgR8qU.mjs.map} +1 -1
  47. package/dist/tune/index.mjs +1 -1
  48. package/package.json +1 -1
  49. package/dist/gerbil-BFk5jV0h.mjs +0 -4
  50. package/dist/gerbil-Cgmb4Dit.mjs.map +0 -1
  51. package/dist/gpu-kQVLpV3n.mjs.map +0 -1
  52. package/dist/index-ElJKy9i9.d.mts.map +0 -1
  53. package/dist/moonshine-stt-9wc6t11v.mjs.map +0 -1
  54. package/dist/moonshine-stt-B7rDn6Q9.mjs +0 -4
@@ -1,3 +1,3 @@
1
1
  import { a as OuteSpeakerWord, c as buildOutePromptString, i as OuteSpeaker, l as loadOuteSpeaker, n as OuteSpeakOptions, o as OuteTTS, r as OuteSpeakResult, s as OuteTTSOptions, t as OuteDType } from "../outetts-CAL3_K3j.mjs";
2
- import { $ as Gemma4VisionGridConfig, A as generateDacSpeechDecoderGraph, At as AdapterSource, B as generateNanoCodecDecoderGraph, Bt as GPUDiagnosticResult, C as TranscribeOptions, Ct as loadKaniTTS, D as generateQwen3_5VisionGraph, Dt as ChatMessage, E as Executor, Et as quantizeKaniBackbone, F as generateMoonshineEncoderGraph, Ft as KaniDType, G as generateGemma4VisionGraph, Gt as ModelArchConfig, H as Gemma4VisionGraphInfo, Ht as GraphDType, I as moonshineEncoderFrames, It as KaniTTS, J as DEFAULT_MODELS, K as patchGemma4VisionClips, Kt as ModelCapabilities, L as parseMoonshineConfig, Lt as KaniTTSOptions, M as parseOuteTtsConfig, Mt as applyLoRAToStore, N as MOONSHINE_REMAINING_WORK, Nt as buildLoRADeltas, O as audioTokensToDacCodes, Ot as Tokenizer, P as generateMoonshineDecoderGraph, Pt as fetchAdapter, Q as GEMMA4_IMAGE_PROCESSOR, R as audioTokensToCodes, Rt as SpeakOptions, S as MoonshineSTTOptions, St as LoadedMoonshine, T as MoonshineEncoderExecutor, Tt as loadMoonshine, U as dequantizeGemma4VisionProjection, Ut as KVDType, V as parseKaniConfig, Vt as initGPU, W as dequantizeMLXProjection, Wt as KvMode, X as OUTETTS_PRESET_VOICES, Y as OUTETTS_ASSETS, Z as resolveDefaultRepo, _ as ParlerSpeakOptions, _t as preprocessImage, a as GenerateObjectOptions, at as VisionPositionTensors, b as ParlerTTSOptions, bt as SamplingParams, c as GenerateResult, ct as buildGemma4RotaryCosSin, d as ObjectSchema, dt as buildMRoPEPositionIds, et as Gemma4VisionPositionTensors, f as ObjectValidator, ft as buildPosEmbeds, g as VisionInputs, gt as mropeFreqDims, h as VisionExecutor, ht as buildVisionPositionTensors, i as EncodeImageResult, it as VisionGridConfig, j as generateOuteTtsBackboneGraph, jt as LoRADelta, k as dacOutputLength, kt as AdapterConfig, l as IntegrityCheckEntry, lt as buildGemma4VisionPositionTensors, m as WebGPUEngineOptions, mt as buildRotaryCosSin, n as AgentTool, nt as PreprocessedImage, o as GenerateObjectResult, ot as buildGemma4PoolMatrix, p as WebGPUEngine, pt as buildPositionIds, q as resolveGemma4VisionInfo, qt as createDefaultHFKeyMapper, r as EmbedOptions, rt as QWEN3_5_IMAGE_PROCESSOR, s as GenerateOptions, st as buildGemma4PosEmbeds, t as AgentStep, tt as ImageProcessorConfig, u as IntegrityCheckResult, ut as buildMRoPECosSin, v as ParlerSpeakResult, vt as preprocessImageGemma4, w as TranscribeResult, wt as loadModel, x as MoonshineSTT, xt as LoadedKaniTTS, y as ParlerTTS, yt as smartResize, z as generateKaniTtsGraph, zt as SpeakResult } from "../index-ElJKy9i9.mjs";
2
+ import { $ as Gemma4VisionGridConfig, A as generateDacSpeechDecoderGraph, At as AdapterSource, B as generateNanoCodecDecoderGraph, Bt as GPUDiagnosticResult, C as TranscribeOptions, Ct as loadKaniTTS, D as generateQwen3_5VisionGraph, Dt as ChatMessage, E as Executor, Et as quantizeKaniBackbone, F as generateMoonshineEncoderGraph, Ft as KaniDType, G as generateGemma4VisionGraph, Gt as ModelArchConfig, H as Gemma4VisionGraphInfo, Ht as GraphDType, I as moonshineEncoderFrames, It as KaniTTS, J as DEFAULT_MODELS, K as patchGemma4VisionClips, Kt as ModelCapabilities, L as parseMoonshineConfig, Lt as KaniTTSOptions, M as parseOuteTtsConfig, Mt as applyLoRAToStore, N as MOONSHINE_REMAINING_WORK, Nt as buildLoRADeltas, O as audioTokensToDacCodes, Ot as Tokenizer, P as generateMoonshineDecoderGraph, Pt as fetchAdapter, Q as GEMMA4_IMAGE_PROCESSOR, R as audioTokensToCodes, Rt as SpeakOptions, S as MoonshineSTTOptions, St as LoadedMoonshine, T as MoonshineEncoderExecutor, Tt as loadMoonshine, U as dequantizeGemma4VisionProjection, Ut as KVDType, V as parseKaniConfig, Vt as initGPU, W as dequantizeMLXProjection, Wt as KvMode, X as OUTETTS_PRESET_VOICES, Y as OUTETTS_ASSETS, Z as resolveDefaultRepo, _ as ParlerSpeakOptions, _t as preprocessImage, a as GenerateObjectOptions, at as VisionPositionTensors, b as ParlerTTSOptions, bt as SamplingParams, c as GenerateResult, ct as buildGemma4RotaryCosSin, d as ObjectSchema, dt as buildMRoPEPositionIds, et as Gemma4VisionPositionTensors, f as ObjectValidator, ft as buildPosEmbeds, g as VisionInputs, gt as mropeFreqDims, h as VisionExecutor, ht as buildVisionPositionTensors, i as EncodeImageResult, it as VisionGridConfig, j as generateOuteTtsBackboneGraph, jt as LoRADelta, k as dacOutputLength, kt as AdapterConfig, l as IntegrityCheckEntry, lt as buildGemma4VisionPositionTensors, m as WebGPUEngineOptions, mt as buildRotaryCosSin, n as AgentTool, nt as PreprocessedImage, o as GenerateObjectResult, ot as buildGemma4PoolMatrix, p as WebGPUEngine, pt as buildPositionIds, q as resolveGemma4VisionInfo, qt as createDefaultHFKeyMapper, r as EmbedOptions, rt as QWEN3_5_IMAGE_PROCESSOR, s as GenerateOptions, st as buildGemma4PosEmbeds, t as AgentStep, tt as ImageProcessorConfig, u as IntegrityCheckResult, ut as buildMRoPECosSin, v as ParlerSpeakResult, vt as preprocessImageGemma4, w as TranscribeResult, wt as loadModel, x as MoonshineSTT, xt as LoadedKaniTTS, y as ParlerTTS, yt as smartResize, z as generateKaniTtsGraph, zt as SpeakResult } from "../index-B56xBiR2.mjs";
3
3
  export { AdapterConfig, AdapterSource, AgentStep, AgentTool, ChatMessage, DEFAULT_MODELS, EmbedOptions, EncodeImageResult, Executor, GEMMA4_IMAGE_PROCESSOR, GPUDiagnosticResult, Gemma4VisionGraphInfo, Gemma4VisionGridConfig, Gemma4VisionPositionTensors, GenerateObjectOptions, GenerateObjectResult, GenerateOptions, GenerateResult, GraphDType, ImageProcessorConfig, IntegrityCheckEntry, IntegrityCheckResult, KVDType, KaniDType, KaniTTS, KaniTTSOptions, KvMode, LoRADelta, LoadedKaniTTS, LoadedMoonshine, MOONSHINE_REMAINING_WORK, ModelArchConfig, ModelCapabilities, MoonshineEncoderExecutor, MoonshineSTT, MoonshineSTTOptions, OUTETTS_ASSETS, OUTETTS_PRESET_VOICES, ObjectSchema, ObjectValidator, OuteDType, OuteSpeakOptions, OuteSpeakResult, OuteSpeaker, OuteSpeakerWord, OuteTTS, OuteTTSOptions, ParlerSpeakOptions, ParlerSpeakResult, ParlerTTS, ParlerTTSOptions, PreprocessedImage, QWEN3_5_IMAGE_PROCESSOR, SamplingParams, SpeakOptions, SpeakResult, Tokenizer, TranscribeOptions, TranscribeResult, VisionExecutor, VisionGridConfig, VisionInputs, VisionPositionTensors, WebGPUEngine, WebGPUEngineOptions, applyLoRAToStore, audioTokensToCodes, audioTokensToDacCodes, buildGemma4PoolMatrix, buildGemma4PosEmbeds, buildGemma4RotaryCosSin, buildGemma4VisionPositionTensors, buildLoRADeltas, buildMRoPECosSin, buildMRoPEPositionIds, buildOutePromptString, buildPosEmbeds, buildPositionIds, buildRotaryCosSin, buildVisionPositionTensors, createDefaultHFKeyMapper, dacOutputLength, dequantizeGemma4VisionProjection, dequantizeMLXProjection, fetchAdapter, generateDacSpeechDecoderGraph, generateGemma4VisionGraph, generateKaniTtsGraph, generateMoonshineDecoderGraph, generateMoonshineEncoderGraph, generateNanoCodecDecoderGraph, generateOuteTtsBackboneGraph, generateQwen3_5VisionGraph, initGPU, loadKaniTTS, loadModel, loadMoonshine, loadOuteSpeaker, moonshineEncoderFrames, mropeFreqDims, parseKaniConfig, parseMoonshineConfig, parseOuteTtsConfig, patchGemma4VisionClips, preprocessImage, preprocessImageGemma4, quantizeKaniBackbone, resolveDefaultRepo, resolveGemma4VisionInfo, smartResize };
@@ -1,6 +1,6 @@
1
1
  import { i as resolveDefaultRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "../defaults-C_bJK9zs.mjs";
2
2
  import { D as createDefaultHFKeyMapper, a as generateMoonshineEncoderGraph, h as generateNanoCodecDecoderGraph, i as generateMoonshineDecoderGraph, m as generateKaniTtsGraph, o as moonshineEncoderFrames, r as MOONSHINE_REMAINING_WORK, s as parseMoonshineConfig, u as audioTokensToCodes, x as parseKaniConfig } from "../architectures-BHkqQ9xp.mjs";
3
- import { A as dequantizeGemma4VisionProjection, C as audioTokensToDacCodes, D as parseOuteTtsConfig, E as generateOuteTtsBackboneGraph, M as generateGemma4VisionGraph, N as patchGemma4VisionClips, O as KaniTTS, P as resolveGemma4VisionInfo, S as loadOuteSpeaker, T as generateDacSpeechDecoderGraph, _ as smartResize, a as buildGemma4PosEmbeds, b as OuteTTS, c as buildMRoPECosSin, d as buildPositionIds, f as buildRotaryCosSin, g as preprocessImageGemma4, h as preprocessImage, i as buildGemma4PoolMatrix, j as dequantizeMLXProjection, k as generateQwen3_5VisionGraph, l as buildMRoPEPositionIds, m as mropeFreqDims, n as GEMMA4_IMAGE_PROCESSOR, o as buildGemma4RotaryCosSin, p as buildVisionPositionTensors, r as QWEN3_5_IMAGE_PROCESSOR, s as buildGemma4VisionPositionTensors, t as WebGPUEngine, u as buildPosEmbeds, v as VisionExecutor, w as dacOutputLength, x as buildOutePromptString, y as ParlerTTS } from "../gpu-kQVLpV3n.mjs";
4
- import { a as loadMoonshine, d as Tokenizer, f as applyLoRAToStore, h as Executor, i as loadModel, l as quantizeKaniBackbone, m as fetchAdapter, n as MoonshineEncoderExecutor, p as buildLoRADeltas, r as loadKaniTTS, t as MoonshineSTT, w as initGPU } from "../moonshine-stt-9wc6t11v.mjs";
3
+ import { A as dequantizeGemma4VisionProjection, C as audioTokensToDacCodes, D as parseOuteTtsConfig, E as generateOuteTtsBackboneGraph, M as generateGemma4VisionGraph, N as patchGemma4VisionClips, O as KaniTTS, P as resolveGemma4VisionInfo, S as loadOuteSpeaker, T as generateDacSpeechDecoderGraph, _ as smartResize, a as buildGemma4PosEmbeds, b as OuteTTS, c as buildMRoPECosSin, d as buildPositionIds, f as buildRotaryCosSin, g as preprocessImageGemma4, h as preprocessImage, i as buildGemma4PoolMatrix, j as dequantizeMLXProjection, k as generateQwen3_5VisionGraph, l as buildMRoPEPositionIds, m as mropeFreqDims, n as GEMMA4_IMAGE_PROCESSOR, o as buildGemma4RotaryCosSin, p as buildVisionPositionTensors, r as QWEN3_5_IMAGE_PROCESSOR, s as buildGemma4VisionPositionTensors, t as WebGPUEngine, u as buildPosEmbeds, v as VisionExecutor, w as dacOutputLength, x as buildOutePromptString, y as ParlerTTS } from "../gpu-NPdk7pmr.mjs";
4
+ import { T as initGPU, a as loadModel, f as Tokenizer, g as Executor, h as fetchAdapter, i as loadKaniTTS, m as buildLoRADeltas, n as MoonshineEncoderExecutor, o as loadMoonshine, p as applyLoRAToStore, t as MoonshineSTT, u as quantizeKaniBackbone } from "../moonshine-stt-BYbYwSz2.mjs";
5
5
 
6
6
  export { DEFAULT_MODELS, Executor, GEMMA4_IMAGE_PROCESSOR, KaniTTS, MOONSHINE_REMAINING_WORK, MoonshineEncoderExecutor, MoonshineSTT, OUTETTS_ASSETS, OUTETTS_PRESET_VOICES, OuteTTS, ParlerTTS, QWEN3_5_IMAGE_PROCESSOR, Tokenizer, VisionExecutor, WebGPUEngine, applyLoRAToStore, audioTokensToCodes, audioTokensToDacCodes, buildGemma4PoolMatrix, buildGemma4PosEmbeds, buildGemma4RotaryCosSin, buildGemma4VisionPositionTensors, buildLoRADeltas, buildMRoPECosSin, buildMRoPEPositionIds, buildOutePromptString, buildPosEmbeds, buildPositionIds, buildRotaryCosSin, buildVisionPositionTensors, createDefaultHFKeyMapper, dacOutputLength, dequantizeGemma4VisionProjection, dequantizeMLXProjection, fetchAdapter, generateDacSpeechDecoderGraph, generateGemma4VisionGraph, generateKaniTtsGraph, generateMoonshineDecoderGraph, generateMoonshineEncoderGraph, generateNanoCodecDecoderGraph, generateOuteTtsBackboneGraph, generateQwen3_5VisionGraph, initGPU, loadKaniTTS, loadModel, loadMoonshine, loadOuteSpeaker, moonshineEncoderFrames, mropeFreqDims, parseKaniConfig, parseMoonshineConfig, parseOuteTtsConfig, patchGemma4VisionClips, preprocessImage, preprocessImageGemma4, quantizeKaniBackbone, resolveDefaultRepo, resolveGemma4VisionInfo, smartResize };
@@ -1,6 +1,6 @@
1
1
  import { i as resolveDefaultRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "./defaults-C_bJK9zs.mjs";
2
2
  import { E as GEMMA4_VIS_KEYS, S as DEFAULT_GROUP_SIZE, T as DTYPE_BYTES, _ as kaniCosTensor, c as KANI_END_OF_HUMAN, d as buildKaniLayerCosSin, f as computeKaniPositions, g as kaniAttentionLayerIndices, h as generateNanoCodecDecoderGraph, l as KANI_START_OF_HUMAN, m as generateKaniTtsGraph, u as audioTokensToCodes, v as kaniLayerAlpha, w as CANONICAL_KEYS, x as parseKaniConfig, y as kaniSinTensor } from "./architectures-BHkqQ9xp.mjs";
3
- import { C as getOrCreatePipeline, S as destroyBuffers, T as verifyGPU, _ as MATMUL_BIAS_F16C_SPEC, b as createStorageBuffer, c as quantizeBackboneInt4, g as KERNEL_REGISTRY, h as Executor, i as loadModel, l as quantizeKaniBackbone, o as loadOuteTTS, r as loadKaniTTS, s as loadParlerTTS, u as remapPrunedToken, v as clearPipelineCache, w as initGPU, x as createUniformBuffer, y as createBindGroup } from "./moonshine-stt-9wc6t11v.mjs";
3
+ import { C as destroyBuffers, E as verifyGPU, S as createUniformBuffer, T as initGPU, _ as KERNEL_REGISTRY, a as loadModel, b as createBindGroup, c as loadParlerTTS, d as remapPrunedToken, g as Executor, h as fetchAdapter, i as loadKaniTTS, l as quantizeBackboneInt4, m as buildLoRADeltas, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as MATMUL_BIAS_F16C_SPEC, w as getOrCreatePipeline, x as createStorageBuffer, y as clearPipelineCache } from "./moonshine-stt-BYbYwSz2.mjs";
4
4
 
5
5
  //#region src/gpu/architectures/gemma4_vision.ts
6
6
  /**
@@ -1123,6 +1123,108 @@ var KaniTTS = class KaniTTS {
1123
1123
  }
1124
1124
  };
1125
1125
 
1126
+ //#endregion
1127
+ //#region src/gpu/autocomplete-clean.ts
1128
+ const DEFAULT_MAX_SUGGESTION_CHARS = 140;
1129
+ const DEFAULT_RESTATEMENT_MIN_WORDS = 3;
1130
+ const DEFAULT_INTERNAL_REPEAT_MIN_RUN = 6;
1131
+ const LEADING_PUNCT = /^[.,;:!?)\]}'"”’%]/;
1132
+ const WRAPPING_QUOTES_START = /^["'“”‘’`]+/;
1133
+ const WRAPPING_QUOTES_END = /["'“”‘’`]+$/;
1134
+ const TRAILING_WHITESPACE = /\s+$/;
1135
+ const WHITESPACE_RUN = /\s+/g;
1136
+ const AFTER_FIRST_NEWLINE = /\r?\n[\s\S]*$/;
1137
+ const ENDS_WITH_SPACE = /\s$/;
1138
+ /**
1139
+ * Longest span of characters that is both a suffix of `typed` and a prefix of
1140
+ * `suggestion`, dropped from the start of `suggestion`. Trailing whitespace on
1141
+ * `typed` is ignored so a duplicated word at the cursor ("the " + "the store")
1142
+ * is caught. Returns `suggestion` unchanged when there is no overlap.
1143
+ */
1144
+ function stripLeadingOverlap(typed, suggestion) {
1145
+ const t = typed.replace(TRAILING_WHITESPACE, "");
1146
+ const max = Math.min(t.length, suggestion.length);
1147
+ for (let k = max; k > 0; k--) if (t.slice(t.length - k) === suggestion.slice(0, k)) return suggestion.slice(k);
1148
+ return suggestion;
1149
+ }
1150
+ /**
1151
+ * Drop a leading run of words from `suggestion` when that run already appears as
1152
+ * a contiguous phrase anywhere in `typed`. This catches the model restating an
1153
+ * EARLIER phrase (not just the immediate tail). Matching is case-insensitive,
1154
+ * whitespace-normalized, and word-boundary aware so it never strips on a
1155
+ * partial-word coincidence. Only strips runs of at least `minWords` words to
1156
+ * avoid nuking a legitimate short continuation.
1157
+ */
1158
+ function stripRestatement(typed, suggestion, minWords = DEFAULT_RESTATEMENT_MIN_WORDS) {
1159
+ const normTyped = ` ${typed.toLowerCase().replace(WHITESPACE_RUN, " ").trim()} `;
1160
+ const words = suggestion.split(WHITESPACE_RUN).filter(Boolean);
1161
+ let matched = 0;
1162
+ for (let n = 1; n <= words.length; n++) {
1163
+ const phrase = ` ${words.slice(0, n).join(" ").toLowerCase()} `;
1164
+ if (normTyped.includes(phrase)) matched = n;
1165
+ else break;
1166
+ }
1167
+ if (matched >= minWords) return words.slice(matched).join(" ");
1168
+ return suggestion;
1169
+ }
1170
+ /**
1171
+ * Truncate `text` right before the first point where a span of `minRun` or more
1172
+ * consecutive words repeats a span seen earlier — the classic small-model loop
1173
+ * ("…endless sands …endless sands…"). Returns `text` unchanged (spacing
1174
+ * preserved) when no such repeat exists.
1175
+ */
1176
+ function truncateInternalRepeat(text, minRun = DEFAULT_INTERNAL_REPEAT_MIN_RUN) {
1177
+ const words = text.split(WHITESPACE_RUN).filter(Boolean);
1178
+ const seen = /* @__PURE__ */ new Set();
1179
+ for (let j = 0; j + minRun <= words.length; j++) {
1180
+ const gram = words.slice(j, j + minRun).join(" ").toLowerCase();
1181
+ if (seen.has(gram)) return words.slice(0, j).join(" ");
1182
+ seen.add(gram);
1183
+ }
1184
+ return text;
1185
+ }
1186
+ /**
1187
+ * Cap `text` to at most `max` characters, cutting on a word boundary when a
1188
+ * reasonable one exists in the back half of the window.
1189
+ */
1190
+ function capLength(text, max) {
1191
+ if (text.length <= max) return text;
1192
+ const cut = text.slice(0, max);
1193
+ const lastSpace = cut.lastIndexOf(" ");
1194
+ return (lastSpace > max * .5 ? cut.slice(0, lastSpace) : cut).trimEnd();
1195
+ }
1196
+ /**
1197
+ * Turn a raw model completion into a clean inline ghost continuation:
1198
+ * 1. keep only the first line (when `singleLine`),
1199
+ * 2. strip wrapping quotes,
1200
+ * 3. drop a full verbatim echo of the typed text,
1201
+ * 4. drop a leading character overlap with the typed tail,
1202
+ * 5. drop a leading word-run that restates an earlier typed phrase,
1203
+ * 6. truncate an internal phrase loop,
1204
+ * 7. cap the length,
1205
+ * 8. add a single smart leading space so the ghost joins the caret naturally.
1206
+ *
1207
+ * Returns "" when nothing novel is left — the ghost then simply shows nothing,
1208
+ * which is the correct behavior for a suggestion that only repeats the input.
1209
+ */
1210
+ function cleanSuggestion(raw, typed, options = {}) {
1211
+ const { singleLine = true, maxChars = DEFAULT_MAX_SUGGESTION_CHARS } = options;
1212
+ let s = singleLine ? raw.replace(AFTER_FIRST_NEWLINE, "") : raw;
1213
+ s = s.replace(WRAPPING_QUOTES_START, "").replace(WRAPPING_QUOTES_END, "");
1214
+ s = s.trim();
1215
+ if (!s) return "";
1216
+ const typedTrim = typed.trim();
1217
+ if (typedTrim && s.startsWith(typedTrim)) s = s.slice(typedTrim.length).trimStart();
1218
+ s = stripLeadingOverlap(typed, s).trimStart();
1219
+ s = stripRestatement(typed, s);
1220
+ s = truncateInternalRepeat(s);
1221
+ s = capLength(s.trim(), maxChars).trim();
1222
+ if (!s) return "";
1223
+ const startsWithPunct = LEADING_PUNCT.test(s);
1224
+ const typedEndsWithSpace = ENDS_WITH_SPACE.test(typed) || typed.length === 0;
1225
+ return startsWithPunct || typedEndsWithSpace ? s : ` ${s}`;
1226
+ }
1227
+
1126
1228
  //#endregion
1127
1229
  //#region src/gpu/architectures/outetts.ts
1128
1230
  const OUTETTS_C1_BASE = 151669;
@@ -5516,22 +5618,6 @@ const AUTOCOMPLETE_SYSTEM = [
5516
5618
  "Do not answer questions; just continue the writing.",
5517
5619
  "Example — input: \"The quick brown fox\" → continuation: \" jumps over the lazy dog.\""
5518
5620
  ].join(" ");
5519
- /**
5520
- * Turn raw model output into a clean inline continuation: cut after the first
5521
- * newline (single-line), strip wrapping quotes, drop an echoed copy of the typed
5522
- * text, and add a single leading space unless the suggestion hugs punctuation or
5523
- * the typed text already ends with whitespace.
5524
- */
5525
- function normalizeContinuation(raw, typed, singleLine) {
5526
- let s = singleLine ? raw.replace(/\n[\s\S]*$/, "") : raw;
5527
- s = s.replace(/^["'“”']+/, "").replace(/["'“”']+$/, "");
5528
- if (s.startsWith(typed)) s = s.slice(typed.length);
5529
- s = s.replace(/^\s+/, "");
5530
- if (!s) return "";
5531
- const startsWithPunct = /^[.,;:!?)\]}'"”’%]/.test(s);
5532
- const typedEndsWithSpace = /\s$/.test(typed) || typed.length === 0;
5533
- return startsWithPunct || typedEndsWithSpace ? s : ` ${s}`;
5534
- }
5535
5621
  function formatAgentToolsPrompt(tools) {
5536
5622
  return `You are a helpful assistant with access to tools.
5537
5623
 
@@ -5592,6 +5678,8 @@ var WebGPUEngine = class WebGPUEngine {
5592
5678
  _outeTTS = null;
5593
5679
  /** The preset voice the lazily-built OuteTTS engine was created with. */
5594
5680
  _outeVoice = null;
5681
+ /** Source of the runtime LoRA adapter currently applied on the static base, if any. */
5682
+ _currentAdapter = null;
5595
5683
  /** Lazily-created Parler-TTS engine (Flan-T5 encoder + decoder LM + dac_44khz). */
5596
5684
  _parlerTTS = null;
5597
5685
  /**
@@ -5601,6 +5689,20 @@ var WebGPUEngine = class WebGPUEngine {
5601
5689
  * promotion runs at most once per session. Always false on Dawn/node.
5602
5690
  */
5603
5691
  _groupProbePending = false;
5692
+ /**
5693
+ * Tail of the per-engine generation queue. The shared {@link Executor} owns a
5694
+ * single set of readback/staging buffers (logits + argmax) and one KV cache, so
5695
+ * two overlapping generate()/embed()/describeImage() calls on the SAME engine
5696
+ * instance would call `mapAsync` on the logits/argmax readback while a prior map
5697
+ * is still pending ("Buffer already has an outstanding map pending") and corrupt
5698
+ * each other's KV state. The browser SDK shares one engine per model across
5699
+ * components (see SHARED_ENGINES in browser/use-engine.ts), so this overlap is
5700
+ * real — e.g. a playground's chat handle and its example-generator on the same
5701
+ * instance. We serialize whole generations through this promise chain: a second
5702
+ * caller awaits the first. Per-engine and only engaged on genuine overlap, so it
5703
+ * adds no per-token overhead to the decode hot path.
5704
+ */
5705
+ _genQueueTail = Promise.resolve();
5604
5706
  /** Model capabilities (text, vision, moe). */
5605
5707
  capabilities;
5606
5708
  /** Model architecture config. */
@@ -5649,6 +5751,33 @@ var WebGPUEngine = class WebGPUEngine {
5649
5751
  };
5650
5752
  }
5651
5753
  /**
5754
+ * Load, swap, or remove a runtime LoRA adapter on the loaded base.
5755
+ *
5756
+ * @param source Adapter spec (`hf:owner/repo`, a URL, or `file:./dir` holding
5757
+ * `adapter_config.json` + `adapter_model.safetensors`), or `null` to drop it.
5758
+ */
5759
+ async loadAdapter(source) {
5760
+ if (source === null) {
5761
+ this.executor.clearRuntimeLoRA();
5762
+ this._currentAdapter = null;
5763
+ return;
5764
+ }
5765
+ const adapter = await fetchAdapter(source, {
5766
+ hfToken: this._createOptions.hfToken,
5767
+ revision: this._createOptions.revision
5768
+ });
5769
+ if (!adapter) throw new Error(`No adapter found at ${source}.`);
5770
+ const deltas = buildLoRADeltas(adapter, createKeyMapperForArch(this._architecture));
5771
+ const { applied, skipped } = this.executor.applyRuntimeLoRA(deltas);
5772
+ if (applied === 0) throw new Error(`Adapter ${source} resolved to 0 applicable runtime targets on ${this._architecture} (skipped ${skipped}).`);
5773
+ console.log(`[engine] runtime adapter ${source}: applied ${applied} target(s)${skipped ? `, skipped ${skipped}` : ""}.`);
5774
+ this._currentAdapter = source;
5775
+ }
5776
+ /** The runtime LoRA adapter currently applied on the base, or null. */
5777
+ getAdapter() {
5778
+ return this._currentAdapter;
5779
+ }
5780
+ /**
5652
5781
  * Write a coarse crash-phase breadcrumb that survives a GPU-process kill / page
5653
5782
  * reload. The iPad harness reads `localStorage["gerbil-crash-phase"]` after a
5654
5783
  * crash; without these, a describe-time crash only shows the last load phase
@@ -5948,8 +6077,21 @@ var WebGPUEngine = class WebGPUEngine {
5948
6077
  }
5949
6078
  /**
5950
6079
  * Generate text from a prompt.
6080
+ *
6081
+ * Serialized per engine: the shared executor's readback/staging buffers and KV
6082
+ * cache cannot be shared by two overlapping generations, so a concurrent call on
6083
+ * the same instance awaits this one (see {@link _genQueueTail}).
5951
6084
  */
5952
6085
  async generate(prompt, options = {}) {
6086
+ this.checkDestroyed();
6087
+ const release = await this._acquireGenLock();
6088
+ try {
6089
+ return await this._generateLocked(prompt, options);
6090
+ } finally {
6091
+ release();
6092
+ }
6093
+ }
6094
+ async _generateLocked(prompt, options = {}) {
5953
6095
  this.checkDestroyed();
5954
6096
  const { maxTokens = 512, stopSequences = [], sampling = {}, systemPrompt, onToken } = options;
5955
6097
  this.executor.reset();
@@ -6017,7 +6159,7 @@ var WebGPUEngine = class WebGPUEngine {
6017
6159
  let mmLogicalPos = inputIds.length;
6018
6160
  const keepPlan = this.config.vocabKeepPlan;
6019
6161
  const remapTok = keepPlan ? (i) => remapPrunedToken(i, keepPlan) : (i) => i;
6020
- if (isGreedy && !this.executor.needsMultiEncoder && !mmDecode) {
6162
+ if (isGreedy && !this.executor.needsMultiEncoder && !mmDecode && !this.executor.hasRuntimeAdapter) {
6021
6163
  const firstToken = remapTok(sampleToken(logits, sampling, [...inputIds, ...generatedIds]));
6022
6164
  if (!consumeToken(firstToken)) {
6023
6165
  const depth = Executor.PIPELINE_DEPTH;
@@ -6062,9 +6204,17 @@ var WebGPUEngine = class WebGPUEngine {
6062
6204
  }
6063
6205
  /**
6064
6206
  * Inline autocomplete: continue `prefix` with a brief, single-line continuation.
6065
- * Wraps `generate` with low-latency defaults (16 tokens, temp 0.3, stop at the
6066
- * first newline) + a continuation system prompt, then cleans the output (strip
6067
- * after newline, dequote, drop an echoed prefix, smart leading space).
6207
+ * Wraps `generate` with low-latency defaults (16 tokens, temp 0.3, a repetition
6208
+ * penalty of 1.25, stop at the first newline) + a continuation system prompt,
6209
+ * then cleans the output: strip after newline, dequote, drop an echoed prefix,
6210
+ * drop a leading overlap with the typed tail, drop a leading word-run that
6211
+ * restates an earlier typed phrase, truncate an internal phrase loop, cap the
6212
+ * length, and add a smart leading space.
6213
+ *
6214
+ * The repetition penalty plus the overlap/restatement/loop cleanup keep small
6215
+ * quantized models (e.g. Qwen3.5-0.8B) from echoing the typed text or looping
6216
+ * on a phrase. Both are on by default; `repetitionPenalty` and `maxChars` are
6217
+ * overridable.
6068
6218
  *
6069
6219
  * ```ts
6070
6220
  * const suggestion = await engine.autocomplete("The quick brown fox");
@@ -6072,12 +6222,18 @@ var WebGPUEngine = class WebGPUEngine {
6072
6222
  * ```
6073
6223
  */
6074
6224
  async autocomplete(prefix, opts = {}) {
6075
- return normalizeContinuation((await this.generate(prefix, {
6225
+ return cleanSuggestion((await this.generate(prefix, {
6076
6226
  systemPrompt: AUTOCOMPLETE_SYSTEM,
6077
6227
  maxTokens: opts.maxTokens ?? 16,
6078
- sampling: { temperature: opts.temperature ?? .3 },
6228
+ sampling: {
6229
+ temperature: opts.temperature ?? .3,
6230
+ repetitionPenalty: opts.repetitionPenalty ?? 1.25
6231
+ },
6079
6232
  stopSequences: opts.stop ?? ["\n"]
6080
- })).text, prefix, opts.singleLine ?? true);
6233
+ })).text, prefix, {
6234
+ singleLine: opts.singleLine ?? true,
6235
+ maxChars: opts.maxChars
6236
+ });
6081
6237
  }
6082
6238
  /**
6083
6239
  * Rewrite `text` in a target tone (e.g. "professional", "friendly", "concise",
@@ -6380,6 +6536,15 @@ var WebGPUEngine = class WebGPUEngine {
6380
6536
  * HF-exact pixel_values from a reference; skips host preprocessing).
6381
6537
  */
6382
6538
  async describeImage(image, prompt = "Describe this image.", options = {}) {
6539
+ this.checkDestroyed();
6540
+ const release = await this._acquireGenLock();
6541
+ try {
6542
+ return await this._describeImageLocked(image, prompt, options);
6543
+ } finally {
6544
+ release();
6545
+ }
6546
+ }
6547
+ async _describeImageLocked(image, prompt = "Describe this image.", options = {}) {
6383
6548
  this.checkDestroyed();
6384
6549
  if (!this.visionExecutor) throw new Error("describeImage() requires a vision encoder. Load with { enableVision: true } on a vision-capable checkpoint (Qwen3.5 or Gemma 4).");
6385
6550
  if (this.visionExecutor.gemma4) {
@@ -6646,6 +6811,15 @@ var WebGPUEngine = class WebGPUEngine {
6646
6811
  * `{ taskPrompt }` for other tasks (clustering/classification/STS).
6647
6812
  */
6648
6813
  async embed(text, options = {}) {
6814
+ this.checkDestroyed();
6815
+ const release = await this._acquireGenLock();
6816
+ try {
6817
+ return await this._embedLocked(text, options);
6818
+ } finally {
6819
+ release();
6820
+ }
6821
+ }
6822
+ async _embedLocked(text, options = {}) {
6649
6823
  this.checkDestroyed();
6650
6824
  if (!this._isEmbedding) throw new Error("embed() requires an embedding model. Load with { embedding: true } (e.g. repo: 'Qwen/Qwen3-Embedding-0.6B' or 'mlx-community/embeddinggemma-300m-4bit').");
6651
6825
  this.executor.reset();
@@ -7072,8 +7246,30 @@ var WebGPUEngine = class WebGPUEngine {
7072
7246
  checkDestroyed() {
7073
7247
  if (this._destroyed) throw new Error("WebGPUEngine has been destroyed");
7074
7248
  }
7249
+ /**
7250
+ * Acquire the per-engine generation lock (see {@link _genQueueTail}). Returns a
7251
+ * release fn that MUST be called in a `finally` so the chain never wedges. This
7252
+ * is a plain promise-chain mutex: each caller links onto the current tail and
7253
+ * only proceeds once the previous holder releases.
7254
+ *
7255
+ * Non-reentrant: only the top-level generation entry points that directly drive
7256
+ * the shared executor readback — generate(), embed(), describeImage() — acquire
7257
+ * it. Helpers that delegate to generate() (autocomplete, rewrite,
7258
+ * generateWithTools, generateObject, stream) inherit serialization through that
7259
+ * single call and must NOT acquire it themselves, or they would self-deadlock.
7260
+ */
7261
+ async _acquireGenLock() {
7262
+ let release;
7263
+ const next = new Promise((resolve) => {
7264
+ release = resolve;
7265
+ });
7266
+ const prev = this._genQueueTail;
7267
+ this._genQueueTail = next;
7268
+ await prev;
7269
+ return release;
7270
+ }
7075
7271
  };
7076
7272
 
7077
7273
  //#endregion
7078
7274
  export { dequantizeGemma4VisionProjection as A, audioTokensToDacCodes as C, parseOuteTtsConfig as D, generateOuteTtsBackboneGraph as E, generateGemma4VisionGraph as M, patchGemma4VisionClips as N, KaniTTS as O, resolveGemma4VisionInfo as P, loadOuteSpeaker as S, generateDacSpeechDecoderGraph as T, smartResize as _, buildGemma4PosEmbeds as a, OuteTTS as b, buildMRoPECosSin as c, buildPositionIds as d, buildRotaryCosSin as f, preprocessImageGemma4 as g, preprocessImage as h, buildGemma4PoolMatrix as i, dequantizeMLXProjection as j, generateQwen3_5VisionGraph as k, buildMRoPEPositionIds as l, mropeFreqDims as m, GEMMA4_IMAGE_PROCESSOR as n, buildGemma4RotaryCosSin as o, buildVisionPositionTensors as p, QWEN3_5_IMAGE_PROCESSOR as r, buildGemma4VisionPositionTensors as s, WebGPUEngine as t, buildPosEmbeds as u, VisionExecutor as v, dacOutputLength as w, buildOutePromptString as x, ParlerTTS as y };
7079
- //# sourceMappingURL=gpu-kQVLpV3n.mjs.map
7275
+ //# sourceMappingURL=gpu-NPdk7pmr.mjs.map