@tryhamster/gerbil 1.8.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser/index.d.ts.map +1 -1
- package/dist/browser/index.js +10 -0
- package/dist/browser/index.js.map +1 -1
- package/dist/cli.mjs +7 -7
- package/dist/cli.mjs.map +1 -1
- package/dist/frameworks/express.mjs +1 -1
- package/dist/frameworks/fastify.mjs +1 -1
- package/dist/frameworks/hono.mjs +1 -1
- package/dist/frameworks/next.d.mts +2 -2
- package/dist/frameworks/next.mjs +1 -1
- package/dist/frameworks/trpc.mjs +1 -1
- package/dist/{gerbil-DU1aRO6v.d.mts → gerbil-CSk3AHNN.d.mts} +10 -9
- package/dist/{gerbil-DU1aRO6v.d.mts.map → gerbil-CSk3AHNN.d.mts.map} +1 -1
- package/dist/gerbil-DrV6iNcD.mjs +4 -0
- package/dist/{gerbil-Cgmb4Dit.mjs → gerbil-DtREprR_.mjs} +32 -10
- package/dist/gerbil-DtREprR_.mjs.map +1 -0
- package/dist/gpu/hooks.d.mts +16 -1
- package/dist/gpu/hooks.d.mts.map +1 -1
- package/dist/gpu/hooks.mjs +29 -4
- package/dist/gpu/hooks.mjs.map +1 -1
- package/dist/gpu/index.d.mts +1 -1
- package/dist/gpu/index.mjs +2 -2
- package/dist/{gpu-kQVLpV3n.mjs → gpu-NPdk7pmr.mjs} +221 -25
- package/dist/gpu-NPdk7pmr.mjs.map +1 -0
- package/dist/{index-ElJKy9i9.d.mts → index-B56xBiR2.d.mts} +96 -4
- package/dist/index-B56xBiR2.d.mts.map +1 -0
- package/dist/index.d.mts +2 -2
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +4 -4
- package/dist/integrations/ai-sdk.mjs +1 -1
- package/dist/integrations/langchain.mjs +1 -1
- package/dist/integrations/llamaindex.mjs +1 -1
- package/dist/integrations/mcp.d.mts +2 -2
- package/dist/integrations/mcp.mjs +4 -4
- package/dist/{mcp-DTusP2m5.mjs → mcp-XU4quJ9w.mjs} +3 -3
- package/dist/{mcp-DTusP2m5.mjs.map → mcp-XU4quJ9w.mjs.map} +1 -1
- package/dist/moonshine-stt-BRTnAoQ8.mjs +4 -0
- package/dist/{moonshine-stt-9wc6t11v.mjs → moonshine-stt-BYbYwSz2.mjs} +301 -13
- package/dist/moonshine-stt-BYbYwSz2.mjs.map +1 -0
- package/dist/{one-liner-Bk5x7gYH.mjs → one-liner-1WoJGLlZ.mjs} +2 -2
- package/dist/{one-liner-Bk5x7gYH.mjs.map → one-liner-1WoJGLlZ.mjs.map} +1 -1
- package/dist/{repl-DV6l-jT8.mjs → repl-B41yROa-.mjs} +3 -3
- package/dist/skills/index.d.mts +4 -4
- package/dist/skills/index.mjs +3 -3
- package/dist/{skills-BhcwnL2l.mjs → skills-C9RgR8qU.mjs} +2 -2
- package/dist/{skills-BhcwnL2l.mjs.map → skills-C9RgR8qU.mjs.map} +1 -1
- package/dist/tune/index.mjs +1 -1
- package/package.json +1 -1
- package/dist/gerbil-BFk5jV0h.mjs +0 -4
- package/dist/gerbil-Cgmb4Dit.mjs.map +0 -1
- package/dist/gpu-kQVLpV3n.mjs.map +0 -1
- package/dist/index-ElJKy9i9.d.mts.map +0 -1
- package/dist/moonshine-stt-9wc6t11v.mjs.map +0 -1
- package/dist/moonshine-stt-B7rDn6Q9.mjs +0 -4
package/dist/gpu/index.d.mts
CHANGED
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import { a as OuteSpeakerWord, c as buildOutePromptString, i as OuteSpeaker, l as loadOuteSpeaker, n as OuteSpeakOptions, o as OuteTTS, r as OuteSpeakResult, s as OuteTTSOptions, t as OuteDType } from "../outetts-CAL3_K3j.mjs";
|
|
2
|
-
import { $ as Gemma4VisionGridConfig, A as generateDacSpeechDecoderGraph, At as AdapterSource, B as generateNanoCodecDecoderGraph, Bt as GPUDiagnosticResult, C as TranscribeOptions, Ct as loadKaniTTS, D as generateQwen3_5VisionGraph, Dt as ChatMessage, E as Executor, Et as quantizeKaniBackbone, F as generateMoonshineEncoderGraph, Ft as KaniDType, G as generateGemma4VisionGraph, Gt as ModelArchConfig, H as Gemma4VisionGraphInfo, Ht as GraphDType, I as moonshineEncoderFrames, It as KaniTTS, J as DEFAULT_MODELS, K as patchGemma4VisionClips, Kt as ModelCapabilities, L as parseMoonshineConfig, Lt as KaniTTSOptions, M as parseOuteTtsConfig, Mt as applyLoRAToStore, N as MOONSHINE_REMAINING_WORK, Nt as buildLoRADeltas, O as audioTokensToDacCodes, Ot as Tokenizer, P as generateMoonshineDecoderGraph, Pt as fetchAdapter, Q as GEMMA4_IMAGE_PROCESSOR, R as audioTokensToCodes, Rt as SpeakOptions, S as MoonshineSTTOptions, St as LoadedMoonshine, T as MoonshineEncoderExecutor, Tt as loadMoonshine, U as dequantizeGemma4VisionProjection, Ut as KVDType, V as parseKaniConfig, Vt as initGPU, W as dequantizeMLXProjection, Wt as KvMode, X as OUTETTS_PRESET_VOICES, Y as OUTETTS_ASSETS, Z as resolveDefaultRepo, _ as ParlerSpeakOptions, _t as preprocessImage, a as GenerateObjectOptions, at as VisionPositionTensors, b as ParlerTTSOptions, bt as SamplingParams, c as GenerateResult, ct as buildGemma4RotaryCosSin, d as ObjectSchema, dt as buildMRoPEPositionIds, et as Gemma4VisionPositionTensors, f as ObjectValidator, ft as buildPosEmbeds, g as VisionInputs, gt as mropeFreqDims, h as VisionExecutor, ht as buildVisionPositionTensors, i as EncodeImageResult, it as VisionGridConfig, j as generateOuteTtsBackboneGraph, jt as LoRADelta, k as dacOutputLength, kt as AdapterConfig, l as IntegrityCheckEntry, lt as buildGemma4VisionPositionTensors, m as WebGPUEngineOptions, mt as buildRotaryCosSin, n as AgentTool, nt as PreprocessedImage, o as GenerateObjectResult, ot as buildGemma4PoolMatrix, p as WebGPUEngine, pt as buildPositionIds, q as resolveGemma4VisionInfo, qt as createDefaultHFKeyMapper, r as EmbedOptions, rt as QWEN3_5_IMAGE_PROCESSOR, s as GenerateOptions, st as buildGemma4PosEmbeds, t as AgentStep, tt as ImageProcessorConfig, u as IntegrityCheckResult, ut as buildMRoPECosSin, v as ParlerSpeakResult, vt as preprocessImageGemma4, w as TranscribeResult, wt as loadModel, x as MoonshineSTT, xt as LoadedKaniTTS, y as ParlerTTS, yt as smartResize, z as generateKaniTtsGraph, zt as SpeakResult } from "../index-
|
|
2
|
+
import { $ as Gemma4VisionGridConfig, A as generateDacSpeechDecoderGraph, At as AdapterSource, B as generateNanoCodecDecoderGraph, Bt as GPUDiagnosticResult, C as TranscribeOptions, Ct as loadKaniTTS, D as generateQwen3_5VisionGraph, Dt as ChatMessage, E as Executor, Et as quantizeKaniBackbone, F as generateMoonshineEncoderGraph, Ft as KaniDType, G as generateGemma4VisionGraph, Gt as ModelArchConfig, H as Gemma4VisionGraphInfo, Ht as GraphDType, I as moonshineEncoderFrames, It as KaniTTS, J as DEFAULT_MODELS, K as patchGemma4VisionClips, Kt as ModelCapabilities, L as parseMoonshineConfig, Lt as KaniTTSOptions, M as parseOuteTtsConfig, Mt as applyLoRAToStore, N as MOONSHINE_REMAINING_WORK, Nt as buildLoRADeltas, O as audioTokensToDacCodes, Ot as Tokenizer, P as generateMoonshineDecoderGraph, Pt as fetchAdapter, Q as GEMMA4_IMAGE_PROCESSOR, R as audioTokensToCodes, Rt as SpeakOptions, S as MoonshineSTTOptions, St as LoadedMoonshine, T as MoonshineEncoderExecutor, Tt as loadMoonshine, U as dequantizeGemma4VisionProjection, Ut as KVDType, V as parseKaniConfig, Vt as initGPU, W as dequantizeMLXProjection, Wt as KvMode, X as OUTETTS_PRESET_VOICES, Y as OUTETTS_ASSETS, Z as resolveDefaultRepo, _ as ParlerSpeakOptions, _t as preprocessImage, a as GenerateObjectOptions, at as VisionPositionTensors, b as ParlerTTSOptions, bt as SamplingParams, c as GenerateResult, ct as buildGemma4RotaryCosSin, d as ObjectSchema, dt as buildMRoPEPositionIds, et as Gemma4VisionPositionTensors, f as ObjectValidator, ft as buildPosEmbeds, g as VisionInputs, gt as mropeFreqDims, h as VisionExecutor, ht as buildVisionPositionTensors, i as EncodeImageResult, it as VisionGridConfig, j as generateOuteTtsBackboneGraph, jt as LoRADelta, k as dacOutputLength, kt as AdapterConfig, l as IntegrityCheckEntry, lt as buildGemma4VisionPositionTensors, m as WebGPUEngineOptions, mt as buildRotaryCosSin, n as AgentTool, nt as PreprocessedImage, o as GenerateObjectResult, ot as buildGemma4PoolMatrix, p as WebGPUEngine, pt as buildPositionIds, q as resolveGemma4VisionInfo, qt as createDefaultHFKeyMapper, r as EmbedOptions, rt as QWEN3_5_IMAGE_PROCESSOR, s as GenerateOptions, st as buildGemma4PosEmbeds, t as AgentStep, tt as ImageProcessorConfig, u as IntegrityCheckResult, ut as buildMRoPECosSin, v as ParlerSpeakResult, vt as preprocessImageGemma4, w as TranscribeResult, wt as loadModel, x as MoonshineSTT, xt as LoadedKaniTTS, y as ParlerTTS, yt as smartResize, z as generateKaniTtsGraph, zt as SpeakResult } from "../index-B56xBiR2.mjs";
|
|
3
3
|
export { AdapterConfig, AdapterSource, AgentStep, AgentTool, ChatMessage, DEFAULT_MODELS, EmbedOptions, EncodeImageResult, Executor, GEMMA4_IMAGE_PROCESSOR, GPUDiagnosticResult, Gemma4VisionGraphInfo, Gemma4VisionGridConfig, Gemma4VisionPositionTensors, GenerateObjectOptions, GenerateObjectResult, GenerateOptions, GenerateResult, GraphDType, ImageProcessorConfig, IntegrityCheckEntry, IntegrityCheckResult, KVDType, KaniDType, KaniTTS, KaniTTSOptions, KvMode, LoRADelta, LoadedKaniTTS, LoadedMoonshine, MOONSHINE_REMAINING_WORK, ModelArchConfig, ModelCapabilities, MoonshineEncoderExecutor, MoonshineSTT, MoonshineSTTOptions, OUTETTS_ASSETS, OUTETTS_PRESET_VOICES, ObjectSchema, ObjectValidator, OuteDType, OuteSpeakOptions, OuteSpeakResult, OuteSpeaker, OuteSpeakerWord, OuteTTS, OuteTTSOptions, ParlerSpeakOptions, ParlerSpeakResult, ParlerTTS, ParlerTTSOptions, PreprocessedImage, QWEN3_5_IMAGE_PROCESSOR, SamplingParams, SpeakOptions, SpeakResult, Tokenizer, TranscribeOptions, TranscribeResult, VisionExecutor, VisionGridConfig, VisionInputs, VisionPositionTensors, WebGPUEngine, WebGPUEngineOptions, applyLoRAToStore, audioTokensToCodes, audioTokensToDacCodes, buildGemma4PoolMatrix, buildGemma4PosEmbeds, buildGemma4RotaryCosSin, buildGemma4VisionPositionTensors, buildLoRADeltas, buildMRoPECosSin, buildMRoPEPositionIds, buildOutePromptString, buildPosEmbeds, buildPositionIds, buildRotaryCosSin, buildVisionPositionTensors, createDefaultHFKeyMapper, dacOutputLength, dequantizeGemma4VisionProjection, dequantizeMLXProjection, fetchAdapter, generateDacSpeechDecoderGraph, generateGemma4VisionGraph, generateKaniTtsGraph, generateMoonshineDecoderGraph, generateMoonshineEncoderGraph, generateNanoCodecDecoderGraph, generateOuteTtsBackboneGraph, generateQwen3_5VisionGraph, initGPU, loadKaniTTS, loadModel, loadMoonshine, loadOuteSpeaker, moonshineEncoderFrames, mropeFreqDims, parseKaniConfig, parseMoonshineConfig, parseOuteTtsConfig, patchGemma4VisionClips, preprocessImage, preprocessImageGemma4, quantizeKaniBackbone, resolveDefaultRepo, resolveGemma4VisionInfo, smartResize };
|
package/dist/gpu/index.mjs
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { i as resolveDefaultRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "../defaults-C_bJK9zs.mjs";
|
|
2
2
|
import { D as createDefaultHFKeyMapper, a as generateMoonshineEncoderGraph, h as generateNanoCodecDecoderGraph, i as generateMoonshineDecoderGraph, m as generateKaniTtsGraph, o as moonshineEncoderFrames, r as MOONSHINE_REMAINING_WORK, s as parseMoonshineConfig, u as audioTokensToCodes, x as parseKaniConfig } from "../architectures-BHkqQ9xp.mjs";
|
|
3
|
-
import { A as dequantizeGemma4VisionProjection, C as audioTokensToDacCodes, D as parseOuteTtsConfig, E as generateOuteTtsBackboneGraph, M as generateGemma4VisionGraph, N as patchGemma4VisionClips, O as KaniTTS, P as resolveGemma4VisionInfo, S as loadOuteSpeaker, T as generateDacSpeechDecoderGraph, _ as smartResize, a as buildGemma4PosEmbeds, b as OuteTTS, c as buildMRoPECosSin, d as buildPositionIds, f as buildRotaryCosSin, g as preprocessImageGemma4, h as preprocessImage, i as buildGemma4PoolMatrix, j as dequantizeMLXProjection, k as generateQwen3_5VisionGraph, l as buildMRoPEPositionIds, m as mropeFreqDims, n as GEMMA4_IMAGE_PROCESSOR, o as buildGemma4RotaryCosSin, p as buildVisionPositionTensors, r as QWEN3_5_IMAGE_PROCESSOR, s as buildGemma4VisionPositionTensors, t as WebGPUEngine, u as buildPosEmbeds, v as VisionExecutor, w as dacOutputLength, x as buildOutePromptString, y as ParlerTTS } from "../gpu-
|
|
4
|
-
import {
|
|
3
|
+
import { A as dequantizeGemma4VisionProjection, C as audioTokensToDacCodes, D as parseOuteTtsConfig, E as generateOuteTtsBackboneGraph, M as generateGemma4VisionGraph, N as patchGemma4VisionClips, O as KaniTTS, P as resolveGemma4VisionInfo, S as loadOuteSpeaker, T as generateDacSpeechDecoderGraph, _ as smartResize, a as buildGemma4PosEmbeds, b as OuteTTS, c as buildMRoPECosSin, d as buildPositionIds, f as buildRotaryCosSin, g as preprocessImageGemma4, h as preprocessImage, i as buildGemma4PoolMatrix, j as dequantizeMLXProjection, k as generateQwen3_5VisionGraph, l as buildMRoPEPositionIds, m as mropeFreqDims, n as GEMMA4_IMAGE_PROCESSOR, o as buildGemma4RotaryCosSin, p as buildVisionPositionTensors, r as QWEN3_5_IMAGE_PROCESSOR, s as buildGemma4VisionPositionTensors, t as WebGPUEngine, u as buildPosEmbeds, v as VisionExecutor, w as dacOutputLength, x as buildOutePromptString, y as ParlerTTS } from "../gpu-NPdk7pmr.mjs";
|
|
4
|
+
import { T as initGPU, a as loadModel, f as Tokenizer, g as Executor, h as fetchAdapter, i as loadKaniTTS, m as buildLoRADeltas, n as MoonshineEncoderExecutor, o as loadMoonshine, p as applyLoRAToStore, t as MoonshineSTT, u as quantizeKaniBackbone } from "../moonshine-stt-BYbYwSz2.mjs";
|
|
5
5
|
|
|
6
6
|
export { DEFAULT_MODELS, Executor, GEMMA4_IMAGE_PROCESSOR, KaniTTS, MOONSHINE_REMAINING_WORK, MoonshineEncoderExecutor, MoonshineSTT, OUTETTS_ASSETS, OUTETTS_PRESET_VOICES, OuteTTS, ParlerTTS, QWEN3_5_IMAGE_PROCESSOR, Tokenizer, VisionExecutor, WebGPUEngine, applyLoRAToStore, audioTokensToCodes, audioTokensToDacCodes, buildGemma4PoolMatrix, buildGemma4PosEmbeds, buildGemma4RotaryCosSin, buildGemma4VisionPositionTensors, buildLoRADeltas, buildMRoPECosSin, buildMRoPEPositionIds, buildOutePromptString, buildPosEmbeds, buildPositionIds, buildRotaryCosSin, buildVisionPositionTensors, createDefaultHFKeyMapper, dacOutputLength, dequantizeGemma4VisionProjection, dequantizeMLXProjection, fetchAdapter, generateDacSpeechDecoderGraph, generateGemma4VisionGraph, generateKaniTtsGraph, generateMoonshineDecoderGraph, generateMoonshineEncoderGraph, generateNanoCodecDecoderGraph, generateOuteTtsBackboneGraph, generateQwen3_5VisionGraph, initGPU, loadKaniTTS, loadModel, loadMoonshine, loadOuteSpeaker, moonshineEncoderFrames, mropeFreqDims, parseKaniConfig, parseMoonshineConfig, parseOuteTtsConfig, patchGemma4VisionClips, preprocessImage, preprocessImageGemma4, quantizeKaniBackbone, resolveDefaultRepo, resolveGemma4VisionInfo, smartResize };
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { i as resolveDefaultRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "./defaults-C_bJK9zs.mjs";
|
|
2
2
|
import { E as GEMMA4_VIS_KEYS, S as DEFAULT_GROUP_SIZE, T as DTYPE_BYTES, _ as kaniCosTensor, c as KANI_END_OF_HUMAN, d as buildKaniLayerCosSin, f as computeKaniPositions, g as kaniAttentionLayerIndices, h as generateNanoCodecDecoderGraph, l as KANI_START_OF_HUMAN, m as generateKaniTtsGraph, u as audioTokensToCodes, v as kaniLayerAlpha, w as CANONICAL_KEYS, x as parseKaniConfig, y as kaniSinTensor } from "./architectures-BHkqQ9xp.mjs";
|
|
3
|
-
import { C as
|
|
3
|
+
import { C as destroyBuffers, E as verifyGPU, S as createUniformBuffer, T as initGPU, _ as KERNEL_REGISTRY, a as loadModel, b as createBindGroup, c as loadParlerTTS, d as remapPrunedToken, g as Executor, h as fetchAdapter, i as loadKaniTTS, l as quantizeBackboneInt4, m as buildLoRADeltas, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as MATMUL_BIAS_F16C_SPEC, w as getOrCreatePipeline, x as createStorageBuffer, y as clearPipelineCache } from "./moonshine-stt-BYbYwSz2.mjs";
|
|
4
4
|
|
|
5
5
|
//#region src/gpu/architectures/gemma4_vision.ts
|
|
6
6
|
/**
|
|
@@ -1123,6 +1123,108 @@ var KaniTTS = class KaniTTS {
|
|
|
1123
1123
|
}
|
|
1124
1124
|
};
|
|
1125
1125
|
|
|
1126
|
+
//#endregion
|
|
1127
|
+
//#region src/gpu/autocomplete-clean.ts
|
|
1128
|
+
const DEFAULT_MAX_SUGGESTION_CHARS = 140;
|
|
1129
|
+
const DEFAULT_RESTATEMENT_MIN_WORDS = 3;
|
|
1130
|
+
const DEFAULT_INTERNAL_REPEAT_MIN_RUN = 6;
|
|
1131
|
+
const LEADING_PUNCT = /^[.,;:!?)\]}'"”’%]/;
|
|
1132
|
+
const WRAPPING_QUOTES_START = /^["'“”‘’`]+/;
|
|
1133
|
+
const WRAPPING_QUOTES_END = /["'“”‘’`]+$/;
|
|
1134
|
+
const TRAILING_WHITESPACE = /\s+$/;
|
|
1135
|
+
const WHITESPACE_RUN = /\s+/g;
|
|
1136
|
+
const AFTER_FIRST_NEWLINE = /\r?\n[\s\S]*$/;
|
|
1137
|
+
const ENDS_WITH_SPACE = /\s$/;
|
|
1138
|
+
/**
|
|
1139
|
+
* Longest span of characters that is both a suffix of `typed` and a prefix of
|
|
1140
|
+
* `suggestion`, dropped from the start of `suggestion`. Trailing whitespace on
|
|
1141
|
+
* `typed` is ignored so a duplicated word at the cursor ("the " + "the store")
|
|
1142
|
+
* is caught. Returns `suggestion` unchanged when there is no overlap.
|
|
1143
|
+
*/
|
|
1144
|
+
function stripLeadingOverlap(typed, suggestion) {
|
|
1145
|
+
const t = typed.replace(TRAILING_WHITESPACE, "");
|
|
1146
|
+
const max = Math.min(t.length, suggestion.length);
|
|
1147
|
+
for (let k = max; k > 0; k--) if (t.slice(t.length - k) === suggestion.slice(0, k)) return suggestion.slice(k);
|
|
1148
|
+
return suggestion;
|
|
1149
|
+
}
|
|
1150
|
+
/**
|
|
1151
|
+
* Drop a leading run of words from `suggestion` when that run already appears as
|
|
1152
|
+
* a contiguous phrase anywhere in `typed`. This catches the model restating an
|
|
1153
|
+
* EARLIER phrase (not just the immediate tail). Matching is case-insensitive,
|
|
1154
|
+
* whitespace-normalized, and word-boundary aware so it never strips on a
|
|
1155
|
+
* partial-word coincidence. Only strips runs of at least `minWords` words to
|
|
1156
|
+
* avoid nuking a legitimate short continuation.
|
|
1157
|
+
*/
|
|
1158
|
+
function stripRestatement(typed, suggestion, minWords = DEFAULT_RESTATEMENT_MIN_WORDS) {
|
|
1159
|
+
const normTyped = ` ${typed.toLowerCase().replace(WHITESPACE_RUN, " ").trim()} `;
|
|
1160
|
+
const words = suggestion.split(WHITESPACE_RUN).filter(Boolean);
|
|
1161
|
+
let matched = 0;
|
|
1162
|
+
for (let n = 1; n <= words.length; n++) {
|
|
1163
|
+
const phrase = ` ${words.slice(0, n).join(" ").toLowerCase()} `;
|
|
1164
|
+
if (normTyped.includes(phrase)) matched = n;
|
|
1165
|
+
else break;
|
|
1166
|
+
}
|
|
1167
|
+
if (matched >= minWords) return words.slice(matched).join(" ");
|
|
1168
|
+
return suggestion;
|
|
1169
|
+
}
|
|
1170
|
+
/**
|
|
1171
|
+
* Truncate `text` right before the first point where a span of `minRun` or more
|
|
1172
|
+
* consecutive words repeats a span seen earlier — the classic small-model loop
|
|
1173
|
+
* ("…endless sands …endless sands…"). Returns `text` unchanged (spacing
|
|
1174
|
+
* preserved) when no such repeat exists.
|
|
1175
|
+
*/
|
|
1176
|
+
function truncateInternalRepeat(text, minRun = DEFAULT_INTERNAL_REPEAT_MIN_RUN) {
|
|
1177
|
+
const words = text.split(WHITESPACE_RUN).filter(Boolean);
|
|
1178
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1179
|
+
for (let j = 0; j + minRun <= words.length; j++) {
|
|
1180
|
+
const gram = words.slice(j, j + minRun).join(" ").toLowerCase();
|
|
1181
|
+
if (seen.has(gram)) return words.slice(0, j).join(" ");
|
|
1182
|
+
seen.add(gram);
|
|
1183
|
+
}
|
|
1184
|
+
return text;
|
|
1185
|
+
}
|
|
1186
|
+
/**
|
|
1187
|
+
* Cap `text` to at most `max` characters, cutting on a word boundary when a
|
|
1188
|
+
* reasonable one exists in the back half of the window.
|
|
1189
|
+
*/
|
|
1190
|
+
function capLength(text, max) {
|
|
1191
|
+
if (text.length <= max) return text;
|
|
1192
|
+
const cut = text.slice(0, max);
|
|
1193
|
+
const lastSpace = cut.lastIndexOf(" ");
|
|
1194
|
+
return (lastSpace > max * .5 ? cut.slice(0, lastSpace) : cut).trimEnd();
|
|
1195
|
+
}
|
|
1196
|
+
/**
|
|
1197
|
+
* Turn a raw model completion into a clean inline ghost continuation:
|
|
1198
|
+
* 1. keep only the first line (when `singleLine`),
|
|
1199
|
+
* 2. strip wrapping quotes,
|
|
1200
|
+
* 3. drop a full verbatim echo of the typed text,
|
|
1201
|
+
* 4. drop a leading character overlap with the typed tail,
|
|
1202
|
+
* 5. drop a leading word-run that restates an earlier typed phrase,
|
|
1203
|
+
* 6. truncate an internal phrase loop,
|
|
1204
|
+
* 7. cap the length,
|
|
1205
|
+
* 8. add a single smart leading space so the ghost joins the caret naturally.
|
|
1206
|
+
*
|
|
1207
|
+
* Returns "" when nothing novel is left — the ghost then simply shows nothing,
|
|
1208
|
+
* which is the correct behavior for a suggestion that only repeats the input.
|
|
1209
|
+
*/
|
|
1210
|
+
function cleanSuggestion(raw, typed, options = {}) {
|
|
1211
|
+
const { singleLine = true, maxChars = DEFAULT_MAX_SUGGESTION_CHARS } = options;
|
|
1212
|
+
let s = singleLine ? raw.replace(AFTER_FIRST_NEWLINE, "") : raw;
|
|
1213
|
+
s = s.replace(WRAPPING_QUOTES_START, "").replace(WRAPPING_QUOTES_END, "");
|
|
1214
|
+
s = s.trim();
|
|
1215
|
+
if (!s) return "";
|
|
1216
|
+
const typedTrim = typed.trim();
|
|
1217
|
+
if (typedTrim && s.startsWith(typedTrim)) s = s.slice(typedTrim.length).trimStart();
|
|
1218
|
+
s = stripLeadingOverlap(typed, s).trimStart();
|
|
1219
|
+
s = stripRestatement(typed, s);
|
|
1220
|
+
s = truncateInternalRepeat(s);
|
|
1221
|
+
s = capLength(s.trim(), maxChars).trim();
|
|
1222
|
+
if (!s) return "";
|
|
1223
|
+
const startsWithPunct = LEADING_PUNCT.test(s);
|
|
1224
|
+
const typedEndsWithSpace = ENDS_WITH_SPACE.test(typed) || typed.length === 0;
|
|
1225
|
+
return startsWithPunct || typedEndsWithSpace ? s : ` ${s}`;
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1126
1228
|
//#endregion
|
|
1127
1229
|
//#region src/gpu/architectures/outetts.ts
|
|
1128
1230
|
const OUTETTS_C1_BASE = 151669;
|
|
@@ -5516,22 +5618,6 @@ const AUTOCOMPLETE_SYSTEM = [
|
|
|
5516
5618
|
"Do not answer questions; just continue the writing.",
|
|
5517
5619
|
"Example — input: \"The quick brown fox\" → continuation: \" jumps over the lazy dog.\""
|
|
5518
5620
|
].join(" ");
|
|
5519
|
-
/**
|
|
5520
|
-
* Turn raw model output into a clean inline continuation: cut after the first
|
|
5521
|
-
* newline (single-line), strip wrapping quotes, drop an echoed copy of the typed
|
|
5522
|
-
* text, and add a single leading space unless the suggestion hugs punctuation or
|
|
5523
|
-
* the typed text already ends with whitespace.
|
|
5524
|
-
*/
|
|
5525
|
-
function normalizeContinuation(raw, typed, singleLine) {
|
|
5526
|
-
let s = singleLine ? raw.replace(/\n[\s\S]*$/, "") : raw;
|
|
5527
|
-
s = s.replace(/^["'“”']+/, "").replace(/["'“”']+$/, "");
|
|
5528
|
-
if (s.startsWith(typed)) s = s.slice(typed.length);
|
|
5529
|
-
s = s.replace(/^\s+/, "");
|
|
5530
|
-
if (!s) return "";
|
|
5531
|
-
const startsWithPunct = /^[.,;:!?)\]}'"”’%]/.test(s);
|
|
5532
|
-
const typedEndsWithSpace = /\s$/.test(typed) || typed.length === 0;
|
|
5533
|
-
return startsWithPunct || typedEndsWithSpace ? s : ` ${s}`;
|
|
5534
|
-
}
|
|
5535
5621
|
function formatAgentToolsPrompt(tools) {
|
|
5536
5622
|
return `You are a helpful assistant with access to tools.
|
|
5537
5623
|
|
|
@@ -5592,6 +5678,8 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5592
5678
|
_outeTTS = null;
|
|
5593
5679
|
/** The preset voice the lazily-built OuteTTS engine was created with. */
|
|
5594
5680
|
_outeVoice = null;
|
|
5681
|
+
/** Source of the runtime LoRA adapter currently applied on the static base, if any. */
|
|
5682
|
+
_currentAdapter = null;
|
|
5595
5683
|
/** Lazily-created Parler-TTS engine (Flan-T5 encoder + decoder LM + dac_44khz). */
|
|
5596
5684
|
_parlerTTS = null;
|
|
5597
5685
|
/**
|
|
@@ -5601,6 +5689,20 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5601
5689
|
* promotion runs at most once per session. Always false on Dawn/node.
|
|
5602
5690
|
*/
|
|
5603
5691
|
_groupProbePending = false;
|
|
5692
|
+
/**
|
|
5693
|
+
* Tail of the per-engine generation queue. The shared {@link Executor} owns a
|
|
5694
|
+
* single set of readback/staging buffers (logits + argmax) and one KV cache, so
|
|
5695
|
+
* two overlapping generate()/embed()/describeImage() calls on the SAME engine
|
|
5696
|
+
* instance would call `mapAsync` on the logits/argmax readback while a prior map
|
|
5697
|
+
* is still pending ("Buffer already has an outstanding map pending") and corrupt
|
|
5698
|
+
* each other's KV state. The browser SDK shares one engine per model across
|
|
5699
|
+
* components (see SHARED_ENGINES in browser/use-engine.ts), so this overlap is
|
|
5700
|
+
* real — e.g. a playground's chat handle and its example-generator on the same
|
|
5701
|
+
* instance. We serialize whole generations through this promise chain: a second
|
|
5702
|
+
* caller awaits the first. Per-engine and only engaged on genuine overlap, so it
|
|
5703
|
+
* adds no per-token overhead to the decode hot path.
|
|
5704
|
+
*/
|
|
5705
|
+
_genQueueTail = Promise.resolve();
|
|
5604
5706
|
/** Model capabilities (text, vision, moe). */
|
|
5605
5707
|
capabilities;
|
|
5606
5708
|
/** Model architecture config. */
|
|
@@ -5649,6 +5751,33 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5649
5751
|
};
|
|
5650
5752
|
}
|
|
5651
5753
|
/**
|
|
5754
|
+
* Load, swap, or remove a runtime LoRA adapter on the loaded base.
|
|
5755
|
+
*
|
|
5756
|
+
* @param source Adapter spec (`hf:owner/repo`, a URL, or `file:./dir` holding
|
|
5757
|
+
* `adapter_config.json` + `adapter_model.safetensors`), or `null` to drop it.
|
|
5758
|
+
*/
|
|
5759
|
+
async loadAdapter(source) {
|
|
5760
|
+
if (source === null) {
|
|
5761
|
+
this.executor.clearRuntimeLoRA();
|
|
5762
|
+
this._currentAdapter = null;
|
|
5763
|
+
return;
|
|
5764
|
+
}
|
|
5765
|
+
const adapter = await fetchAdapter(source, {
|
|
5766
|
+
hfToken: this._createOptions.hfToken,
|
|
5767
|
+
revision: this._createOptions.revision
|
|
5768
|
+
});
|
|
5769
|
+
if (!adapter) throw new Error(`No adapter found at ${source}.`);
|
|
5770
|
+
const deltas = buildLoRADeltas(adapter, createKeyMapperForArch(this._architecture));
|
|
5771
|
+
const { applied, skipped } = this.executor.applyRuntimeLoRA(deltas);
|
|
5772
|
+
if (applied === 0) throw new Error(`Adapter ${source} resolved to 0 applicable runtime targets on ${this._architecture} (skipped ${skipped}).`);
|
|
5773
|
+
console.log(`[engine] runtime adapter ${source}: applied ${applied} target(s)${skipped ? `, skipped ${skipped}` : ""}.`);
|
|
5774
|
+
this._currentAdapter = source;
|
|
5775
|
+
}
|
|
5776
|
+
/** The runtime LoRA adapter currently applied on the base, or null. */
|
|
5777
|
+
getAdapter() {
|
|
5778
|
+
return this._currentAdapter;
|
|
5779
|
+
}
|
|
5780
|
+
/**
|
|
5652
5781
|
* Write a coarse crash-phase breadcrumb that survives a GPU-process kill / page
|
|
5653
5782
|
* reload. The iPad harness reads `localStorage["gerbil-crash-phase"]` after a
|
|
5654
5783
|
* crash; without these, a describe-time crash only shows the last load phase
|
|
@@ -5948,8 +6077,21 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
5948
6077
|
}
|
|
5949
6078
|
/**
|
|
5950
6079
|
* Generate text from a prompt.
|
|
6080
|
+
*
|
|
6081
|
+
* Serialized per engine: the shared executor's readback/staging buffers and KV
|
|
6082
|
+
* cache cannot be shared by two overlapping generations, so a concurrent call on
|
|
6083
|
+
* the same instance awaits this one (see {@link _genQueueTail}).
|
|
5951
6084
|
*/
|
|
5952
6085
|
async generate(prompt, options = {}) {
|
|
6086
|
+
this.checkDestroyed();
|
|
6087
|
+
const release = await this._acquireGenLock();
|
|
6088
|
+
try {
|
|
6089
|
+
return await this._generateLocked(prompt, options);
|
|
6090
|
+
} finally {
|
|
6091
|
+
release();
|
|
6092
|
+
}
|
|
6093
|
+
}
|
|
6094
|
+
async _generateLocked(prompt, options = {}) {
|
|
5953
6095
|
this.checkDestroyed();
|
|
5954
6096
|
const { maxTokens = 512, stopSequences = [], sampling = {}, systemPrompt, onToken } = options;
|
|
5955
6097
|
this.executor.reset();
|
|
@@ -6017,7 +6159,7 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6017
6159
|
let mmLogicalPos = inputIds.length;
|
|
6018
6160
|
const keepPlan = this.config.vocabKeepPlan;
|
|
6019
6161
|
const remapTok = keepPlan ? (i) => remapPrunedToken(i, keepPlan) : (i) => i;
|
|
6020
|
-
if (isGreedy && !this.executor.needsMultiEncoder && !mmDecode) {
|
|
6162
|
+
if (isGreedy && !this.executor.needsMultiEncoder && !mmDecode && !this.executor.hasRuntimeAdapter) {
|
|
6021
6163
|
const firstToken = remapTok(sampleToken(logits, sampling, [...inputIds, ...generatedIds]));
|
|
6022
6164
|
if (!consumeToken(firstToken)) {
|
|
6023
6165
|
const depth = Executor.PIPELINE_DEPTH;
|
|
@@ -6062,9 +6204,17 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6062
6204
|
}
|
|
6063
6205
|
/**
|
|
6064
6206
|
* Inline autocomplete: continue `prefix` with a brief, single-line continuation.
|
|
6065
|
-
* Wraps `generate` with low-latency defaults (16 tokens, temp 0.3,
|
|
6066
|
-
* first newline) + a continuation system prompt,
|
|
6067
|
-
* after newline, dequote, drop an echoed prefix,
|
|
6207
|
+
* Wraps `generate` with low-latency defaults (16 tokens, temp 0.3, a repetition
|
|
6208
|
+
* penalty of 1.25, stop at the first newline) + a continuation system prompt,
|
|
6209
|
+
* then cleans the output: strip after newline, dequote, drop an echoed prefix,
|
|
6210
|
+
* drop a leading overlap with the typed tail, drop a leading word-run that
|
|
6211
|
+
* restates an earlier typed phrase, truncate an internal phrase loop, cap the
|
|
6212
|
+
* length, and add a smart leading space.
|
|
6213
|
+
*
|
|
6214
|
+
* The repetition penalty plus the overlap/restatement/loop cleanup keep small
|
|
6215
|
+
* quantized models (e.g. Qwen3.5-0.8B) from echoing the typed text or looping
|
|
6216
|
+
* on a phrase. Both are on by default; `repetitionPenalty` and `maxChars` are
|
|
6217
|
+
* overridable.
|
|
6068
6218
|
*
|
|
6069
6219
|
* ```ts
|
|
6070
6220
|
* const suggestion = await engine.autocomplete("The quick brown fox");
|
|
@@ -6072,12 +6222,18 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6072
6222
|
* ```
|
|
6073
6223
|
*/
|
|
6074
6224
|
async autocomplete(prefix, opts = {}) {
|
|
6075
|
-
return
|
|
6225
|
+
return cleanSuggestion((await this.generate(prefix, {
|
|
6076
6226
|
systemPrompt: AUTOCOMPLETE_SYSTEM,
|
|
6077
6227
|
maxTokens: opts.maxTokens ?? 16,
|
|
6078
|
-
sampling: {
|
|
6228
|
+
sampling: {
|
|
6229
|
+
temperature: opts.temperature ?? .3,
|
|
6230
|
+
repetitionPenalty: opts.repetitionPenalty ?? 1.25
|
|
6231
|
+
},
|
|
6079
6232
|
stopSequences: opts.stop ?? ["\n"]
|
|
6080
|
-
})).text, prefix,
|
|
6233
|
+
})).text, prefix, {
|
|
6234
|
+
singleLine: opts.singleLine ?? true,
|
|
6235
|
+
maxChars: opts.maxChars
|
|
6236
|
+
});
|
|
6081
6237
|
}
|
|
6082
6238
|
/**
|
|
6083
6239
|
* Rewrite `text` in a target tone (e.g. "professional", "friendly", "concise",
|
|
@@ -6380,6 +6536,15 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6380
6536
|
* HF-exact pixel_values from a reference; skips host preprocessing).
|
|
6381
6537
|
*/
|
|
6382
6538
|
async describeImage(image, prompt = "Describe this image.", options = {}) {
|
|
6539
|
+
this.checkDestroyed();
|
|
6540
|
+
const release = await this._acquireGenLock();
|
|
6541
|
+
try {
|
|
6542
|
+
return await this._describeImageLocked(image, prompt, options);
|
|
6543
|
+
} finally {
|
|
6544
|
+
release();
|
|
6545
|
+
}
|
|
6546
|
+
}
|
|
6547
|
+
async _describeImageLocked(image, prompt = "Describe this image.", options = {}) {
|
|
6383
6548
|
this.checkDestroyed();
|
|
6384
6549
|
if (!this.visionExecutor) throw new Error("describeImage() requires a vision encoder. Load with { enableVision: true } on a vision-capable checkpoint (Qwen3.5 or Gemma 4).");
|
|
6385
6550
|
if (this.visionExecutor.gemma4) {
|
|
@@ -6646,6 +6811,15 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
6646
6811
|
* `{ taskPrompt }` for other tasks (clustering/classification/STS).
|
|
6647
6812
|
*/
|
|
6648
6813
|
async embed(text, options = {}) {
|
|
6814
|
+
this.checkDestroyed();
|
|
6815
|
+
const release = await this._acquireGenLock();
|
|
6816
|
+
try {
|
|
6817
|
+
return await this._embedLocked(text, options);
|
|
6818
|
+
} finally {
|
|
6819
|
+
release();
|
|
6820
|
+
}
|
|
6821
|
+
}
|
|
6822
|
+
async _embedLocked(text, options = {}) {
|
|
6649
6823
|
this.checkDestroyed();
|
|
6650
6824
|
if (!this._isEmbedding) throw new Error("embed() requires an embedding model. Load with { embedding: true } (e.g. repo: 'Qwen/Qwen3-Embedding-0.6B' or 'mlx-community/embeddinggemma-300m-4bit').");
|
|
6651
6825
|
this.executor.reset();
|
|
@@ -7072,8 +7246,30 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
7072
7246
|
checkDestroyed() {
|
|
7073
7247
|
if (this._destroyed) throw new Error("WebGPUEngine has been destroyed");
|
|
7074
7248
|
}
|
|
7249
|
+
/**
|
|
7250
|
+
* Acquire the per-engine generation lock (see {@link _genQueueTail}). Returns a
|
|
7251
|
+
* release fn that MUST be called in a `finally` so the chain never wedges. This
|
|
7252
|
+
* is a plain promise-chain mutex: each caller links onto the current tail and
|
|
7253
|
+
* only proceeds once the previous holder releases.
|
|
7254
|
+
*
|
|
7255
|
+
* Non-reentrant: only the top-level generation entry points that directly drive
|
|
7256
|
+
* the shared executor readback — generate(), embed(), describeImage() — acquire
|
|
7257
|
+
* it. Helpers that delegate to generate() (autocomplete, rewrite,
|
|
7258
|
+
* generateWithTools, generateObject, stream) inherit serialization through that
|
|
7259
|
+
* single call and must NOT acquire it themselves, or they would self-deadlock.
|
|
7260
|
+
*/
|
|
7261
|
+
async _acquireGenLock() {
|
|
7262
|
+
let release;
|
|
7263
|
+
const next = new Promise((resolve) => {
|
|
7264
|
+
release = resolve;
|
|
7265
|
+
});
|
|
7266
|
+
const prev = this._genQueueTail;
|
|
7267
|
+
this._genQueueTail = next;
|
|
7268
|
+
await prev;
|
|
7269
|
+
return release;
|
|
7270
|
+
}
|
|
7075
7271
|
};
|
|
7076
7272
|
|
|
7077
7273
|
//#endregion
|
|
7078
7274
|
export { dequantizeGemma4VisionProjection as A, audioTokensToDacCodes as C, parseOuteTtsConfig as D, generateOuteTtsBackboneGraph as E, generateGemma4VisionGraph as M, patchGemma4VisionClips as N, KaniTTS as O, resolveGemma4VisionInfo as P, loadOuteSpeaker as S, generateDacSpeechDecoderGraph as T, smartResize as _, buildGemma4PosEmbeds as a, OuteTTS as b, buildMRoPECosSin as c, buildPositionIds as d, buildRotaryCosSin as f, preprocessImageGemma4 as g, preprocessImage as h, buildGemma4PoolMatrix as i, dequantizeMLXProjection as j, generateQwen3_5VisionGraph as k, buildMRoPEPositionIds as l, mropeFreqDims as m, GEMMA4_IMAGE_PROCESSOR as n, buildGemma4RotaryCosSin as o, buildVisionPositionTensors as p, QWEN3_5_IMAGE_PROCESSOR as r, buildGemma4VisionPositionTensors as s, WebGPUEngine as t, buildPosEmbeds as u, VisionExecutor as v, dacOutputLength as w, buildOutePromptString as x, ParlerTTS as y };
|
|
7079
|
-
//# sourceMappingURL=gpu-
|
|
7275
|
+
//# sourceMappingURL=gpu-NPdk7pmr.mjs.map
|