@tryhamster/gerbil 1.10.1 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{architectures-BHkqQ9xp.mjs → architectures-DmZMEFsA.mjs} +652 -9
- package/dist/architectures-DmZMEFsA.mjs.map +1 -0
- package/dist/cli.mjs +8 -8
- package/dist/cli.mjs.map +1 -1
- package/dist/frameworks/express.mjs +1 -1
- package/dist/frameworks/fastify.mjs +1 -1
- package/dist/frameworks/hono.mjs +1 -1
- package/dist/frameworks/next.d.mts +2 -2
- package/dist/frameworks/next.mjs +1 -1
- package/dist/frameworks/trpc.mjs +1 -1
- package/dist/{gerbil-PEAdJdsH.d.mts → gerbil-6E0XH_8s.d.mts} +2 -2
- package/dist/{gerbil-PEAdJdsH.d.mts.map → gerbil-6E0XH_8s.d.mts.map} +1 -1
- package/dist/gerbil-BY5EW-Jk.mjs +4 -0
- package/dist/{gerbil-sQ7eqzn3.mjs → gerbil-BpYemKEH.mjs} +5 -4
- package/dist/{gerbil-sQ7eqzn3.mjs.map → gerbil-BpYemKEH.mjs.map} +1 -1
- package/dist/gpu/hooks.d.mts +1 -1
- package/dist/gpu/index.d.mts +1 -1
- package/dist/gpu/index.mjs +3 -3
- package/dist/{gpu-B1xk3xOJ.mjs → gpu-CzgbyVeq.mjs} +100 -650
- package/dist/gpu-CzgbyVeq.mjs.map +1 -0
- package/dist/{index-CJux7zbV.d.mts → index-DyqcKFfQ.d.mts} +13 -7
- package/dist/{index-CJux7zbV.d.mts.map → index-DyqcKFfQ.d.mts.map} +1 -1
- package/dist/index.d.mts +2 -2
- package/dist/index.mjs +5 -5
- package/dist/integrations/ai-sdk.mjs +1 -1
- package/dist/integrations/langchain.mjs +1 -1
- package/dist/integrations/llamaindex.mjs +1 -1
- package/dist/integrations/mcp.d.mts +2 -2
- package/dist/integrations/mcp.mjs +4 -4
- package/dist/{mcp-BX-ryGGJ.mjs → mcp-DSpjqWZC.mjs} +3 -3
- package/dist/{mcp-BX-ryGGJ.mjs.map → mcp-DSpjqWZC.mjs.map} +1 -1
- package/dist/moonshine-stt-CcVt4Vdd.mjs +4 -0
- package/dist/{moonshine-stt-BKeD6OYF.mjs → moonshine-stt-CgDXoLFd.mjs} +2 -2
- package/dist/{moonshine-stt-BKeD6OYF.mjs.map → moonshine-stt-CgDXoLFd.mjs.map} +1 -1
- package/dist/{one-liner-m6NjYWMg.mjs → one-liner-BgNDxJSJ.mjs} +2 -2
- package/dist/{one-liner-m6NjYWMg.mjs.map → one-liner-BgNDxJSJ.mjs.map} +1 -1
- package/dist/outetts-CAL3_K3j.d.mts.map +1 -1
- package/dist/repl-DSK2hDzu.mjs +9 -0
- package/dist/skills/index.d.mts +4 -4
- package/dist/skills/index.mjs +3 -3
- package/dist/{skills-DZ5OITy0.mjs → skills-Cd75ZGew.mjs} +2 -2
- package/dist/{skills-DZ5OITy0.mjs.map → skills-Cd75ZGew.mjs.map} +1 -1
- package/dist/tune/index.mjs +1 -1
- package/package.json +1 -1
- package/dist/architectures-BHkqQ9xp.mjs.map +0 -1
- package/dist/gerbil-IqsK32DM.mjs +0 -4
- package/dist/gpu-B1xk3xOJ.mjs.map +0 -1
- package/dist/moonshine-stt-DKC0PHF_.mjs +0 -4
- package/dist/repl-hN4jSHrb.mjs +0 -9
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { i as resolveDefaultRepo, n as OUTETTS_ASSETS, r as OUTETTS_PRESET_VOICES, t as DEFAULT_MODELS } from "./defaults-C_bJK9zs.mjs";
|
|
2
|
-
import { E as GEMMA4_VIS_KEYS,
|
|
3
|
-
import { C as destroyBuffers, E as verifyGPU, S as createUniformBuffer, T as initGPU, _ as KERNEL_REGISTRY, a as loadModel, b as createBindGroup, c as loadParlerTTS, d as remapPrunedToken, g as Executor, h as fetchAdapter, i as loadKaniTTS, l as quantizeBackboneInt4, m as buildLoRADeltas, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as MATMUL_BIAS_F16C_SPEC, w as getOrCreatePipeline, x as createStorageBuffer, y as clearPipelineCache } from "./moonshine-stt-
|
|
2
|
+
import { A as kaniSinTensor, C as computeKaniPositions, D as kaniAttentionLayerIndices, E as generateNanoCodecDecoderGraph, F as CANONICAL_KEYS, I as DTYPE_BYTES, L as GEMMA4_VIS_KEYS, M as parseKaniConfig, N as DEFAULT_GROUP_SIZE, O as kaniCosTensor, S as buildKaniLayerCosSin, T as generateKaniTtsGraph, a as PARLER_DAC_LATENT_DIM, b as KANI_START_OF_HUMAN, c as PARLER_SAMPLE_RATE, d as generateParlerEncoderGraph, f as parseParlerConfig, i as PARLER_DAC_DECODER_DIM, k as kaniLayerAlpha, l as buildT5RelativeBias, o as PARLER_DECODER_RATES, p as revertDelayPattern, r as PARLER_BOS_TOKEN_ID, s as PARLER_EOS_TOKEN_ID, u as generateParlerDecoderGraph, x as audioTokensToCodes, y as KANI_END_OF_HUMAN } from "./architectures-DmZMEFsA.mjs";
|
|
3
|
+
import { C as destroyBuffers, E as verifyGPU, S as createUniformBuffer, T as initGPU, _ as KERNEL_REGISTRY, a as loadModel, b as createBindGroup, c as loadParlerTTS, d as remapPrunedToken, g as Executor, h as fetchAdapter, i as loadKaniTTS, l as quantizeBackboneInt4, m as buildLoRADeltas, r as createKeyMapperForArch, s as loadOuteTTS, u as quantizeKaniBackbone, v as MATMUL_BIAS_F16C_SPEC, w as getOrCreatePipeline, x as createStorageBuffer, y as clearPipelineCache } from "./moonshine-stt-CgDXoLFd.mjs";
|
|
4
4
|
|
|
5
5
|
//#region src/gpu/architectures/gemma4_vision.ts
|
|
6
6
|
/**
|
|
@@ -806,7 +806,7 @@ function nucleusSample$1(probs, ids, topP) {
|
|
|
806
806
|
* In-place top-k filter over a probability array: keep only the `k` highest, zero
|
|
807
807
|
* the rest, and renormalize. No-op when k <= 0 or k >= length (keep everything).
|
|
808
808
|
*/
|
|
809
|
-
function applyTopK
|
|
809
|
+
function applyTopK(scores, k) {
|
|
810
810
|
const n = scores.length;
|
|
811
811
|
if (k <= 0 || k >= n) return;
|
|
812
812
|
const order = Array.from({ length: n }, (_, i) => i).sort((a, b) => scores[b] - scores[a]);
|
|
@@ -1065,7 +1065,7 @@ var KaniTTS = class KaniTTS {
|
|
|
1065
1065
|
scores[i] = s / p.temperature;
|
|
1066
1066
|
}
|
|
1067
1067
|
softmaxInPlace$1(scores);
|
|
1068
|
-
applyTopK
|
|
1068
|
+
applyTopK(scores, p.topK);
|
|
1069
1069
|
return nucleusSample$1(scores, ids, p.topP);
|
|
1070
1070
|
}
|
|
1071
1071
|
/**
|
|
@@ -2880,16 +2880,6 @@ function softmaxInPlace(scores) {
|
|
|
2880
2880
|
}
|
|
2881
2881
|
for (let i = 0; i < scores.length; i++) scores[i] /= sum;
|
|
2882
2882
|
}
|
|
2883
|
-
/** In-place top-k filter: keep the k highest, zero the rest, renormalize. */
|
|
2884
|
-
function applyTopK(scores, k) {
|
|
2885
|
-
const n = scores.length;
|
|
2886
|
-
if (k <= 0 || k >= n) return;
|
|
2887
|
-
const order = Array.from({ length: n }, (_, i) => i).sort((a, b) => scores[b] - scores[a]);
|
|
2888
|
-
let keepSum = 0;
|
|
2889
|
-
for (let r = 0; r < k; r++) keepSum += scores[order[r]];
|
|
2890
|
-
for (let r = k; r < n; r++) scores[order[r]] = 0;
|
|
2891
|
-
if (keepSum > 0) for (let i = 0; i < n; i++) scores[i] /= keepSum;
|
|
2892
|
-
}
|
|
2893
2883
|
/** Top-p nucleus sample from a softmaxed prob array over `ids`. */
|
|
2894
2884
|
function nucleusSample(probs, ids, topP) {
|
|
2895
2885
|
const n = probs.length;
|
|
@@ -2913,6 +2903,81 @@ function nucleusSample(probs, ids, topP) {
|
|
|
2913
2903
|
return ids[order[0]];
|
|
2914
2904
|
}
|
|
2915
2905
|
/**
|
|
2906
|
+
* Select the `k` highest-scoring token ids in a single O(n·logk) pass (bounded
|
|
2907
|
+
* min-heap), applying the repetition penalty to `previous` tokens inline so a
|
|
2908
|
+
* penalized token can drop out of the shortlist. Returns the kept ids and their
|
|
2909
|
+
* (penalized, pre-temperature) logits, unsorted.
|
|
2910
|
+
*
|
|
2911
|
+
* This replaces the two full-vocab `Array.from(...).sort(...)` passes the sampler
|
|
2912
|
+
* ran PER TOKEN over OuteTTS's ~157k vocab (the top-k filter + the nucleus sort):
|
|
2913
|
+
* a codec-LM emits hundreds-to-thousands of tokens per utterance, so those two
|
|
2914
|
+
* O(n·logn) sorts — plus their two fresh 157k-element allocations every step —
|
|
2915
|
+
* dominated host-side decode and made the browser SPEECH tab look stalled
|
|
2916
|
+
* ("SYNTHESIZING…" for minutes). Restricting the softmax + nucleus to the kept
|
|
2917
|
+
* top-k is distribution-identical to the old full-vocab path (softmax is
|
|
2918
|
+
* monotonic and the discarded tail is renormalized away either way).
|
|
2919
|
+
*/
|
|
2920
|
+
function selectTopKPenalized(logits, k, penalized, penalty) {
|
|
2921
|
+
const n = logits.length;
|
|
2922
|
+
const cap = Math.min(k, n);
|
|
2923
|
+
const heapVal = new Float32Array(cap);
|
|
2924
|
+
const heapId = new Int32Array(cap);
|
|
2925
|
+
let size = 0;
|
|
2926
|
+
const penalize = (i, v) => {
|
|
2927
|
+
if (penalized?.has(i)) return v > 0 ? v / penalty : v * penalty;
|
|
2928
|
+
return v;
|
|
2929
|
+
};
|
|
2930
|
+
const siftDown$1 = () => {
|
|
2931
|
+
let root = 0;
|
|
2932
|
+
for (;;) {
|
|
2933
|
+
let smallest = root;
|
|
2934
|
+
const l = 2 * root + 1;
|
|
2935
|
+
const r = 2 * root + 2;
|
|
2936
|
+
if (l < size && heapVal[l] < heapVal[smallest]) smallest = l;
|
|
2937
|
+
if (r < size && heapVal[r] < heapVal[smallest]) smallest = r;
|
|
2938
|
+
if (smallest === root) break;
|
|
2939
|
+
const tv = heapVal[root];
|
|
2940
|
+
const ti = heapId[root];
|
|
2941
|
+
heapVal[root] = heapVal[smallest];
|
|
2942
|
+
heapId[root] = heapId[smallest];
|
|
2943
|
+
heapVal[smallest] = tv;
|
|
2944
|
+
heapId[smallest] = ti;
|
|
2945
|
+
root = smallest;
|
|
2946
|
+
}
|
|
2947
|
+
};
|
|
2948
|
+
const siftUp = (node0) => {
|
|
2949
|
+
let node = node0;
|
|
2950
|
+
while (node > 0) {
|
|
2951
|
+
const parent = node - 1 >> 1;
|
|
2952
|
+
if (heapVal[parent] <= heapVal[node]) break;
|
|
2953
|
+
const tv = heapVal[node];
|
|
2954
|
+
const ti = heapId[node];
|
|
2955
|
+
heapVal[node] = heapVal[parent];
|
|
2956
|
+
heapId[node] = heapId[parent];
|
|
2957
|
+
heapVal[parent] = tv;
|
|
2958
|
+
heapId[parent] = ti;
|
|
2959
|
+
node = parent;
|
|
2960
|
+
}
|
|
2961
|
+
};
|
|
2962
|
+
for (let i = 0; i < n; i++) {
|
|
2963
|
+
const v = penalize(i, logits[i]);
|
|
2964
|
+
if (size < cap) {
|
|
2965
|
+
heapVal[size] = v;
|
|
2966
|
+
heapId[size] = i;
|
|
2967
|
+
size++;
|
|
2968
|
+
siftUp(size - 1);
|
|
2969
|
+
} else if (v > heapVal[0]) {
|
|
2970
|
+
heapVal[0] = v;
|
|
2971
|
+
heapId[0] = i;
|
|
2972
|
+
siftDown$1();
|
|
2973
|
+
}
|
|
2974
|
+
}
|
|
2975
|
+
return {
|
|
2976
|
+
ids: heapId.subarray(0, size),
|
|
2977
|
+
vals: heapVal.subarray(0, size)
|
|
2978
|
+
};
|
|
2979
|
+
}
|
|
2980
|
+
/**
|
|
2916
2981
|
* Resample interleaved-or-mono PCM to mono at `targetRate` via linear interpolation.
|
|
2917
2982
|
* Audio is assumed already mono (Float32 [-1,1]); if `srcRate === targetRate` the
|
|
2918
2983
|
* input is returned as-is. Linear resampling is sufficient for codec conditioning
|
|
@@ -3145,9 +3210,16 @@ var OuteTTS = class OuteTTS {
|
|
|
3145
3210
|
*/
|
|
3146
3211
|
sampleToken(logits, p) {
|
|
3147
3212
|
const n = logits.length;
|
|
3213
|
+
const prev = p.repetitionPenalty !== 1 ? new Set(p.previous) : null;
|
|
3214
|
+
if (p.topK > 0 && p.topK < n) {
|
|
3215
|
+
const { ids: kIds, vals: kVals } = selectTopKPenalized(logits, p.topK, prev, p.repetitionPenalty);
|
|
3216
|
+
const scores$1 = new Float32Array(kVals.length);
|
|
3217
|
+
for (let i = 0; i < kVals.length; i++) scores$1[i] = kVals[i] / p.temperature;
|
|
3218
|
+
softmaxInPlace(scores$1);
|
|
3219
|
+
return nucleusSample(scores$1, Int32Array.from(kIds), p.topP);
|
|
3220
|
+
}
|
|
3148
3221
|
const ids = new Int32Array(n);
|
|
3149
3222
|
const scores = new Float32Array(n);
|
|
3150
|
-
const prev = p.repetitionPenalty !== 1 ? new Set(p.previous) : null;
|
|
3151
3223
|
for (let i = 0; i < n; i++) {
|
|
3152
3224
|
ids[i] = i;
|
|
3153
3225
|
let s = logits[i];
|
|
@@ -3155,7 +3227,6 @@ var OuteTTS = class OuteTTS {
|
|
|
3155
3227
|
scores[i] = s / p.temperature;
|
|
3156
3228
|
}
|
|
3157
3229
|
softmaxInPlace(scores);
|
|
3158
|
-
applyTopK(scores, p.topK);
|
|
3159
3230
|
return nucleusSample(scores, ids, p.topP);
|
|
3160
3231
|
}
|
|
3161
3232
|
/**
|
|
@@ -3195,633 +3266,6 @@ var OuteTTS = class OuteTTS {
|
|
|
3195
3266
|
}
|
|
3196
3267
|
};
|
|
3197
3268
|
|
|
3198
|
-
//#endregion
|
|
3199
|
-
//#region src/gpu/architectures/parler.ts
|
|
3200
|
-
const PARLER_SAMPLE_RATE = 44100;
|
|
3201
|
-
const PARLER_NUM_CODEBOOKS = 9;
|
|
3202
|
-
const PARLER_DECODER_RATES = [
|
|
3203
|
-
8,
|
|
3204
|
-
8,
|
|
3205
|
-
4,
|
|
3206
|
-
2
|
|
3207
|
-
];
|
|
3208
|
-
const PARLER_DAC_LATENT_DIM = 1024;
|
|
3209
|
-
const PARLER_DAC_DECODER_DIM = 1536;
|
|
3210
|
-
/** Decoder audio-vocab: 1024 codes + eos(1024)+bos(1025)+pad(1024); lm_head vocab. */
|
|
3211
|
-
const PARLER_AUDIO_VOCAB = 1088;
|
|
3212
|
-
const PARLER_BOS_TOKEN_ID = 1025;
|
|
3213
|
-
const PARLER_EOS_TOKEN_ID = 1024;
|
|
3214
|
-
const PARLER_PAD_TOKEN_ID = 1024;
|
|
3215
|
-
const CAPS = {
|
|
3216
|
-
text: true,
|
|
3217
|
-
vision: false,
|
|
3218
|
-
moe: false
|
|
3219
|
-
};
|
|
3220
|
-
/** Pull Parler dims from the (nested) HF config. */
|
|
3221
|
-
function parseParlerConfig(raw) {
|
|
3222
|
-
const te = raw.text_encoder ?? {};
|
|
3223
|
-
const dec = raw.decoder ?? {};
|
|
3224
|
-
const enc_heads = te.num_heads ?? 16;
|
|
3225
|
-
const enc_d_model = te.d_model ?? 1024;
|
|
3226
|
-
const dec_heads = dec.num_attention_heads ?? 16;
|
|
3227
|
-
return {
|
|
3228
|
-
enc_d_model,
|
|
3229
|
-
enc_layers: te.num_layers ?? 24,
|
|
3230
|
-
enc_heads,
|
|
3231
|
-
enc_d_kv: te.d_kv ?? Math.floor(enc_d_model / enc_heads),
|
|
3232
|
-
enc_d_ff: te.d_ff ?? 2816,
|
|
3233
|
-
enc_vocab: te.vocab_size ?? 32128,
|
|
3234
|
-
rel_num_buckets: te.relative_attention_num_buckets ?? 32,
|
|
3235
|
-
rel_max_distance: te.relative_attention_max_distance ?? 128,
|
|
3236
|
-
ln_eps: te.layer_norm_epsilon ?? 1e-6,
|
|
3237
|
-
dec_hidden: dec.hidden_size ?? 1024,
|
|
3238
|
-
dec_layers: dec.num_hidden_layers ?? 24,
|
|
3239
|
-
dec_heads,
|
|
3240
|
-
dec_kv_heads: dec.num_key_value_heads ?? dec_heads,
|
|
3241
|
-
dec_ffn: dec.ffn_dim ?? 4096,
|
|
3242
|
-
num_codebooks: dec.num_codebooks ?? PARLER_NUM_CODEBOOKS,
|
|
3243
|
-
audio_vocab: dec.vocab_size ?? PARLER_AUDIO_VOCAB,
|
|
3244
|
-
embed_vocab: (dec.vocab_size ?? PARLER_AUDIO_VOCAB) + 1,
|
|
3245
|
-
max_position: dec.max_position_embeddings ?? 4096,
|
|
3246
|
-
bos_token_id: dec.bos_token_id ?? PARLER_BOS_TOKEN_ID,
|
|
3247
|
-
eos_token_id: dec.eos_token_id ?? PARLER_EOS_TOKEN_ID,
|
|
3248
|
-
pad_token_id: dec.pad_token_id ?? PARLER_PAD_TOKEN_ID
|
|
3249
|
-
};
|
|
3250
|
-
}
|
|
3251
|
-
function baseConfig(c, hidden, heads, layers) {
|
|
3252
|
-
return {
|
|
3253
|
-
hidden_size: hidden,
|
|
3254
|
-
num_layers: layers,
|
|
3255
|
-
num_heads: heads,
|
|
3256
|
-
num_kv_heads: heads,
|
|
3257
|
-
head_dim: Math.floor(hidden / heads),
|
|
3258
|
-
intermediate_size: c.dec_ffn,
|
|
3259
|
-
vocab_size: c.audio_vocab,
|
|
3260
|
-
context_length: c.max_position,
|
|
3261
|
-
rms_norm_eps: c.ln_eps,
|
|
3262
|
-
norm_type: "layernorm",
|
|
3263
|
-
rope_base: 0,
|
|
3264
|
-
rope_dim: 0,
|
|
3265
|
-
kv_layout: "LHSd",
|
|
3266
|
-
is_moe: false,
|
|
3267
|
-
has_vision_tower: false
|
|
3268
|
-
};
|
|
3269
|
-
}
|
|
3270
|
-
/**
|
|
3271
|
-
* T5's bidirectional relative-position bucket (verbatim port of
|
|
3272
|
-
* transformers' T5Attention._relative_position_bucket, bidirectional=True).
|
|
3273
|
-
* Maps a signed relative position (key − query) to a bucket id in [0, num_buckets).
|
|
3274
|
-
*/
|
|
3275
|
-
function relativePositionBucket(relativePosition, numBuckets, maxDistance) {
|
|
3276
|
-
let bucket = 0;
|
|
3277
|
-
let n = relativePosition;
|
|
3278
|
-
const halfBuckets = Math.floor(numBuckets / 2);
|
|
3279
|
-
if (n > 0) bucket += halfBuckets;
|
|
3280
|
-
else n = -n;
|
|
3281
|
-
const maxExact = Math.floor(halfBuckets / 2);
|
|
3282
|
-
const isSmall = n < maxExact;
|
|
3283
|
-
let largeVal = maxExact + Math.floor(Math.log(n / maxExact) / Math.log(maxDistance / maxExact) * (halfBuckets - maxExact));
|
|
3284
|
-
largeVal = Math.min(largeVal, halfBuckets - 1);
|
|
3285
|
-
bucket += isSmall ? n : largeVal;
|
|
3286
|
-
return bucket;
|
|
3287
|
-
}
|
|
3288
|
-
/**
|
|
3289
|
-
* Build the dense per-head relative-position bias B[head, q, k] = bias_table[bucket(k-q), head]
|
|
3290
|
-
* for a length-S encoder sequence. `biasTable` is the learned
|
|
3291
|
-
* relative_attention_bias.weight, row-major [num_buckets, num_heads].
|
|
3292
|
-
* Returns a flat Float32Array of length num_heads*S*S (head-major, then q, then k).
|
|
3293
|
-
*/
|
|
3294
|
-
function buildT5RelativeBias(biasTable, numBuckets, maxDistance, numHeads, S) {
|
|
3295
|
-
const out = new Float32Array(numHeads * S * S);
|
|
3296
|
-
const buckets = new Int32Array(S * S);
|
|
3297
|
-
for (let q = 0; q < S; q++) for (let k = 0; k < S; k++) buckets[q * S + k] = relativePositionBucket(k - q, numBuckets, maxDistance);
|
|
3298
|
-
for (let h = 0; h < numHeads; h++) {
|
|
3299
|
-
const headBase = h * S * S;
|
|
3300
|
-
for (let q = 0; q < S; q++) for (let k = 0; k < S; k++) {
|
|
3301
|
-
const bucket = buckets[q * S + k];
|
|
3302
|
-
out[headBase + q * S + k] = biasTable[bucket * numHeads + h];
|
|
3303
|
-
}
|
|
3304
|
-
}
|
|
3305
|
-
return out;
|
|
3306
|
-
}
|
|
3307
|
-
/**
|
|
3308
|
-
* Build the T5 encoder graph for a concrete description length S. The relative
|
|
3309
|
-
* bias `t5_rel_bias` enters as a host-written activation [num_heads, S, S]. Like
|
|
3310
|
-
* Moonshine, the graph also pre-projects encoder_out through every decoder layer's
|
|
3311
|
-
* encoder_attn.k/v_proj → enc_k_layer{i}/enc_v_layer{i} (frozen cross-attn K/V).
|
|
3312
|
-
*/
|
|
3313
|
-
function generateParlerEncoderGraph(raw, sDesc) {
|
|
3314
|
-
const c = parseParlerConfig(raw);
|
|
3315
|
-
const H = c.enc_d_model;
|
|
3316
|
-
const innerDim = c.enc_heads * c.enc_d_kv;
|
|
3317
|
-
const tensors = {};
|
|
3318
|
-
const nodes = [];
|
|
3319
|
-
const executionOrder = [];
|
|
3320
|
-
const addTensor = (t) => {
|
|
3321
|
-
tensors[t.name] = t;
|
|
3322
|
-
};
|
|
3323
|
-
const addNode = (n) => {
|
|
3324
|
-
nodes.push(n);
|
|
3325
|
-
executionOrder.push(n.id);
|
|
3326
|
-
};
|
|
3327
|
-
const w = (name, shape) => addTensor({
|
|
3328
|
-
name,
|
|
3329
|
-
shape,
|
|
3330
|
-
dtype: "f32",
|
|
3331
|
-
storage: "constant",
|
|
3332
|
-
safetensorsKey: name
|
|
3333
|
-
});
|
|
3334
|
-
const act = (name, shape) => addTensor({
|
|
3335
|
-
name,
|
|
3336
|
-
shape,
|
|
3337
|
-
dtype: "f32",
|
|
3338
|
-
storage: "activation"
|
|
3339
|
-
});
|
|
3340
|
-
const linear = (id, inp, weight, out, K, N) => {
|
|
3341
|
-
w(weight, [N, K]);
|
|
3342
|
-
addNode({
|
|
3343
|
-
id,
|
|
3344
|
-
opType: "MatMul",
|
|
3345
|
-
inputs: [inp, weight],
|
|
3346
|
-
outputs: [out],
|
|
3347
|
-
attributes: {
|
|
3348
|
-
M_tensor: inp,
|
|
3349
|
-
K,
|
|
3350
|
-
N
|
|
3351
|
-
}
|
|
3352
|
-
});
|
|
3353
|
-
};
|
|
3354
|
-
const t5norm = (id, inp, weight, out) => {
|
|
3355
|
-
w(weight, [H]);
|
|
3356
|
-
act(out, [sDesc, H]);
|
|
3357
|
-
addNode({
|
|
3358
|
-
id,
|
|
3359
|
-
opType: "RMSNorm",
|
|
3360
|
-
inputs: [inp, weight],
|
|
3361
|
-
outputs: [out],
|
|
3362
|
-
attributes: {
|
|
3363
|
-
hidden_size: H,
|
|
3364
|
-
eps: c.ln_eps,
|
|
3365
|
-
seq_len_tensor: inp
|
|
3366
|
-
}
|
|
3367
|
-
});
|
|
3368
|
-
};
|
|
3369
|
-
addTensor({
|
|
3370
|
-
name: "input_ids",
|
|
3371
|
-
shape: ["T"],
|
|
3372
|
-
dtype: "u32",
|
|
3373
|
-
storage: "activation"
|
|
3374
|
-
});
|
|
3375
|
-
addTensor({
|
|
3376
|
-
name: "text_encoder.shared.weight",
|
|
3377
|
-
shape: [c.enc_vocab, H],
|
|
3378
|
-
dtype: "f32",
|
|
3379
|
-
storage: "constant",
|
|
3380
|
-
safetensorsKey: "text_encoder.shared.weight"
|
|
3381
|
-
});
|
|
3382
|
-
act("enc_embed", [sDesc, H]);
|
|
3383
|
-
addNode({
|
|
3384
|
-
id: "enc_embed",
|
|
3385
|
-
opType: "Embedding",
|
|
3386
|
-
inputs: ["input_ids", "text_encoder.shared.weight"],
|
|
3387
|
-
outputs: ["enc_embed"],
|
|
3388
|
-
attributes: {
|
|
3389
|
-
vocab_size: c.enc_vocab,
|
|
3390
|
-
hidden_size: H
|
|
3391
|
-
}
|
|
3392
|
-
});
|
|
3393
|
-
addTensor({
|
|
3394
|
-
name: "t5_rel_bias",
|
|
3395
|
-
shape: [
|
|
3396
|
-
c.enc_heads,
|
|
3397
|
-
sDesc,
|
|
3398
|
-
sDesc
|
|
3399
|
-
],
|
|
3400
|
-
dtype: "f32",
|
|
3401
|
-
storage: "activation"
|
|
3402
|
-
});
|
|
3403
|
-
let prev = "enc_embed";
|
|
3404
|
-
for (let i = 0; i < c.enc_layers; i++) {
|
|
3405
|
-
const p = `enc${i}`;
|
|
3406
|
-
const k = (s) => `text_encoder.encoder.block.${i}.${s}`;
|
|
3407
|
-
t5norm(`${p}_norm0`, prev, k("layer.0.layer_norm.weight"), `${p}_n0`);
|
|
3408
|
-
act(`${p}_q`, [sDesc, innerDim]);
|
|
3409
|
-
act(`${p}_k`, [sDesc, innerDim]);
|
|
3410
|
-
act(`${p}_v`, [sDesc, innerDim]);
|
|
3411
|
-
linear(`${p}_qp`, `${p}_n0`, k("layer.0.SelfAttention.q.weight"), `${p}_q`, H, innerDim);
|
|
3412
|
-
linear(`${p}_kp`, `${p}_n0`, k("layer.0.SelfAttention.k.weight"), `${p}_k`, H, innerDim);
|
|
3413
|
-
linear(`${p}_vp`, `${p}_n0`, k("layer.0.SelfAttention.v.weight"), `${p}_v`, H, innerDim);
|
|
3414
|
-
act(`${p}_attn`, [sDesc, innerDim]);
|
|
3415
|
-
addNode({
|
|
3416
|
-
id: `${p}_sa`,
|
|
3417
|
-
opType: "T5Attention",
|
|
3418
|
-
inputs: [
|
|
3419
|
-
`${p}_q`,
|
|
3420
|
-
`${p}_k`,
|
|
3421
|
-
`${p}_v`,
|
|
3422
|
-
"t5_rel_bias"
|
|
3423
|
-
],
|
|
3424
|
-
outputs: [`${p}_attn`],
|
|
3425
|
-
attributes: {
|
|
3426
|
-
num_heads: c.enc_heads,
|
|
3427
|
-
head_dim: c.enc_d_kv,
|
|
3428
|
-
S: sDesc
|
|
3429
|
-
}
|
|
3430
|
-
});
|
|
3431
|
-
act(`${p}_o`, [sDesc, H]);
|
|
3432
|
-
linear(`${p}_op`, `${p}_attn`, k("layer.0.SelfAttention.o.weight"), `${p}_o`, innerDim, H);
|
|
3433
|
-
act(`${p}_res0`, [sDesc, H]);
|
|
3434
|
-
addNode({
|
|
3435
|
-
id: `${p}_add0`,
|
|
3436
|
-
opType: "Add",
|
|
3437
|
-
inputs: [prev, `${p}_o`],
|
|
3438
|
-
outputs: [`${p}_res0`],
|
|
3439
|
-
attributes: {
|
|
3440
|
-
count_tensor: prev,
|
|
3441
|
-
hidden_size: H
|
|
3442
|
-
}
|
|
3443
|
-
});
|
|
3444
|
-
t5norm(`${p}_norm1`, `${p}_res0`, k("layer.1.layer_norm.weight"), `${p}_n1`);
|
|
3445
|
-
act(`${p}_wi0`, [sDesc, c.enc_d_ff]);
|
|
3446
|
-
act(`${p}_wi1`, [sDesc, c.enc_d_ff]);
|
|
3447
|
-
linear(`${p}_wi0p`, `${p}_n1`, k("layer.1.DenseReluDense.wi_0.weight"), `${p}_wi0`, H, c.enc_d_ff);
|
|
3448
|
-
linear(`${p}_wi1p`, `${p}_n1`, k("layer.1.DenseReluDense.wi_1.weight"), `${p}_wi1`, H, c.enc_d_ff);
|
|
3449
|
-
act(`${p}_gelu`, [sDesc, c.enc_d_ff]);
|
|
3450
|
-
addNode({
|
|
3451
|
-
id: `${p}_act`,
|
|
3452
|
-
opType: "GELU",
|
|
3453
|
-
inputs: [`${p}_wi0`],
|
|
3454
|
-
outputs: [`${p}_gelu`],
|
|
3455
|
-
attributes: { count_tensor: `${p}_wi0` }
|
|
3456
|
-
});
|
|
3457
|
-
act(`${p}_gated`, [sDesc, c.enc_d_ff]);
|
|
3458
|
-
addNode({
|
|
3459
|
-
id: `${p}_gate`,
|
|
3460
|
-
opType: "Mul",
|
|
3461
|
-
inputs: [`${p}_gelu`, `${p}_wi1`],
|
|
3462
|
-
outputs: [`${p}_gated`],
|
|
3463
|
-
attributes: {
|
|
3464
|
-
count_tensor: `${p}_gelu`,
|
|
3465
|
-
hidden_size: c.enc_d_ff
|
|
3466
|
-
}
|
|
3467
|
-
});
|
|
3468
|
-
act(`${p}_wo`, [sDesc, H]);
|
|
3469
|
-
linear(`${p}_wop`, `${p}_gated`, k("layer.1.DenseReluDense.wo.weight"), `${p}_wo`, c.enc_d_ff, H);
|
|
3470
|
-
act(`${p}_res1`, [sDesc, H]);
|
|
3471
|
-
addNode({
|
|
3472
|
-
id: `${p}_add1`,
|
|
3473
|
-
opType: "Add",
|
|
3474
|
-
inputs: [`${p}_res0`, `${p}_wo`],
|
|
3475
|
-
outputs: [`${p}_res1`],
|
|
3476
|
-
attributes: {
|
|
3477
|
-
count_tensor: `${p}_res0`,
|
|
3478
|
-
hidden_size: H
|
|
3479
|
-
}
|
|
3480
|
-
});
|
|
3481
|
-
prev = `${p}_res1`;
|
|
3482
|
-
}
|
|
3483
|
-
w("text_encoder.encoder.final_layer_norm.weight", [H]);
|
|
3484
|
-
act("encoder_out", [sDesc, H]);
|
|
3485
|
-
addNode({
|
|
3486
|
-
id: "enc_final_norm",
|
|
3487
|
-
opType: "RMSNorm",
|
|
3488
|
-
inputs: [prev, "text_encoder.encoder.final_layer_norm.weight"],
|
|
3489
|
-
outputs: ["encoder_out"],
|
|
3490
|
-
attributes: {
|
|
3491
|
-
hidden_size: H,
|
|
3492
|
-
eps: c.ln_eps,
|
|
3493
|
-
seq_len_tensor: prev
|
|
3494
|
-
}
|
|
3495
|
-
});
|
|
3496
|
-
const outputs = ["encoder_out"];
|
|
3497
|
-
for (let i = 0; i < c.dec_layers; i++) {
|
|
3498
|
-
const kw = `decoder.model.decoder.layers.${i}.encoder_attn.k_proj.weight`;
|
|
3499
|
-
const vw = `decoder.model.decoder.layers.${i}.encoder_attn.v_proj.weight`;
|
|
3500
|
-
w(kw, [H, H]);
|
|
3501
|
-
w(vw, [H, H]);
|
|
3502
|
-
const encK = `enc_k_layer${i}`;
|
|
3503
|
-
const encV = `enc_v_layer${i}`;
|
|
3504
|
-
act(encK, [sDesc, H]);
|
|
3505
|
-
act(encV, [sDesc, H]);
|
|
3506
|
-
addNode({
|
|
3507
|
-
id: `enc_kproj${i}`,
|
|
3508
|
-
opType: "MatMul",
|
|
3509
|
-
inputs: ["encoder_out", kw],
|
|
3510
|
-
outputs: [encK],
|
|
3511
|
-
attributes: {
|
|
3512
|
-
M_tensor: "encoder_out",
|
|
3513
|
-
K: H,
|
|
3514
|
-
N: H
|
|
3515
|
-
}
|
|
3516
|
-
});
|
|
3517
|
-
addNode({
|
|
3518
|
-
id: `enc_vproj${i}`,
|
|
3519
|
-
opType: "MatMul",
|
|
3520
|
-
inputs: ["encoder_out", vw],
|
|
3521
|
-
outputs: [encV],
|
|
3522
|
-
attributes: {
|
|
3523
|
-
M_tensor: "encoder_out",
|
|
3524
|
-
K: H,
|
|
3525
|
-
N: H
|
|
3526
|
-
}
|
|
3527
|
-
});
|
|
3528
|
-
outputs.push(encK, encV);
|
|
3529
|
-
}
|
|
3530
|
-
return {
|
|
3531
|
-
architecture: "ParlerTextEncoder",
|
|
3532
|
-
config: baseConfig(c, H, c.enc_heads, c.enc_layers),
|
|
3533
|
-
capabilities: CAPS,
|
|
3534
|
-
tensors,
|
|
3535
|
-
nodes,
|
|
3536
|
-
executionOrder,
|
|
3537
|
-
inputs: ["input_ids", "t5_rel_bias"],
|
|
3538
|
-
outputs
|
|
3539
|
-
};
|
|
3540
|
-
}
|
|
3541
|
-
/**
|
|
3542
|
-
* Build the Parler decoder graph for a single decode step (T=1). The host supplies:
|
|
3543
|
-
* - `dec_input_embed` [1, H]: the SUMMED 9-codebook embedding of the current step's
|
|
3544
|
-
* codes PLUS the sinusoidal position (computed host-side from embed_tokens +
|
|
3545
|
-
* embed_positions, since the 9-way embedding sum is cheap on the host and avoids
|
|
3546
|
-
* 9 Embedding ops + an Add tree per step).
|
|
3547
|
-
* - `enc_k_layer{i}` / `enc_v_layer{i}`: frozen cross-attn K/V from the encoder.
|
|
3548
|
-
* Output: 9 logit rows `logits_cb{c}` [1, audio_vocab] (one per codebook head).
|
|
3549
|
-
*
|
|
3550
|
-
* The prompt prefix (transcript) is handled by prefilling its embeddings through the
|
|
3551
|
-
* SAME graph before audio decode begins (the driver feeds the prompt rows first, with
|
|
3552
|
-
* their sinusoidal positions, so the self-attn KV-cache holds [prompt ++ audio]).
|
|
3553
|
-
*/
|
|
3554
|
-
function generateParlerDecoderGraph(raw, sEnc) {
|
|
3555
|
-
const c = parseParlerConfig(raw);
|
|
3556
|
-
const H = c.dec_hidden;
|
|
3557
|
-
const heads = c.dec_heads;
|
|
3558
|
-
const headDim = Math.floor(H / heads);
|
|
3559
|
-
const tensors = {};
|
|
3560
|
-
const nodes = [];
|
|
3561
|
-
const executionOrder = [];
|
|
3562
|
-
const addTensor = (t) => {
|
|
3563
|
-
tensors[t.name] = t;
|
|
3564
|
-
};
|
|
3565
|
-
const addNode = (n) => {
|
|
3566
|
-
nodes.push(n);
|
|
3567
|
-
executionOrder.push(n.id);
|
|
3568
|
-
};
|
|
3569
|
-
const w = (name, shape) => addTensor({
|
|
3570
|
-
name,
|
|
3571
|
-
shape,
|
|
3572
|
-
dtype: "f32",
|
|
3573
|
-
storage: "constant",
|
|
3574
|
-
safetensorsKey: name
|
|
3575
|
-
});
|
|
3576
|
-
const act = (name, shape) => addTensor({
|
|
3577
|
-
name,
|
|
3578
|
-
shape,
|
|
3579
|
-
dtype: "f32",
|
|
3580
|
-
storage: "activation"
|
|
3581
|
-
});
|
|
3582
|
-
const linear = (id, inp, weight, out, K, N) => {
|
|
3583
|
-
w(weight, [N, K]);
|
|
3584
|
-
addNode({
|
|
3585
|
-
id,
|
|
3586
|
-
opType: "MatMul",
|
|
3587
|
-
inputs: [inp, weight],
|
|
3588
|
-
outputs: [out],
|
|
3589
|
-
attributes: {
|
|
3590
|
-
M_tensor: inp,
|
|
3591
|
-
K,
|
|
3592
|
-
N
|
|
3593
|
-
}
|
|
3594
|
-
});
|
|
3595
|
-
};
|
|
3596
|
-
const layernorm = (id, inp, prefix, out) => {
|
|
3597
|
-
w(`${prefix}.weight`, [H]);
|
|
3598
|
-
w(`${prefix}.bias`, [H]);
|
|
3599
|
-
act(out, ["T", H]);
|
|
3600
|
-
addNode({
|
|
3601
|
-
id,
|
|
3602
|
-
opType: "LayerNorm",
|
|
3603
|
-
inputs: [
|
|
3604
|
-
inp,
|
|
3605
|
-
`${prefix}.weight`,
|
|
3606
|
-
`${prefix}.bias`
|
|
3607
|
-
],
|
|
3608
|
-
outputs: [out],
|
|
3609
|
-
attributes: {
|
|
3610
|
-
hidden_size: H,
|
|
3611
|
-
eps: c.ln_eps,
|
|
3612
|
-
seq_len_tensor: inp,
|
|
3613
|
-
has_bias: true
|
|
3614
|
-
}
|
|
3615
|
-
});
|
|
3616
|
-
};
|
|
3617
|
-
const inputs = ["dec_input_embed"];
|
|
3618
|
-
addTensor({
|
|
3619
|
-
name: "dec_input_embed",
|
|
3620
|
-
shape: ["T", H],
|
|
3621
|
-
dtype: "f32",
|
|
3622
|
-
storage: "activation"
|
|
3623
|
-
});
|
|
3624
|
-
let prev = "dec_input_embed";
|
|
3625
|
-
for (let i = 0; i < c.dec_layers; i++) {
|
|
3626
|
-
const p = `dec${i}`;
|
|
3627
|
-
const lk = `decoder.model.decoder.layers.${i}`;
|
|
3628
|
-
layernorm(`${p}_n1`, prev, `${lk}.self_attn_layer_norm`, `${p}_ln1`);
|
|
3629
|
-
act(`${p}_q`, ["T", H]);
|
|
3630
|
-
act(`${p}_k`, ["T", H]);
|
|
3631
|
-
act(`${p}_v`, ["T", H]);
|
|
3632
|
-
linear(`${p}_qp`, `${p}_ln1`, `${lk}.self_attn.q_proj.weight`, `${p}_q`, H, H);
|
|
3633
|
-
linear(`${p}_kp`, `${p}_ln1`, `${lk}.self_attn.k_proj.weight`, `${p}_k`, H, H);
|
|
3634
|
-
linear(`${p}_vp`, `${p}_ln1`, `${lk}.self_attn.v_proj.weight`, `${p}_v`, H, H);
|
|
3635
|
-
addTensor({
|
|
3636
|
-
name: `${p}_kcache`,
|
|
3637
|
-
shape: ["L_max", H],
|
|
3638
|
-
dtype: "f32",
|
|
3639
|
-
storage: "kv_cache"
|
|
3640
|
-
});
|
|
3641
|
-
addTensor({
|
|
3642
|
-
name: `${p}_vcache`,
|
|
3643
|
-
shape: ["L_max", H],
|
|
3644
|
-
dtype: "f32",
|
|
3645
|
-
storage: "kv_cache"
|
|
3646
|
-
});
|
|
3647
|
-
addNode({
|
|
3648
|
-
id: `${p}_kappend`,
|
|
3649
|
-
opType: "KVCacheAppend",
|
|
3650
|
-
inputs: [`${p}_k`],
|
|
3651
|
-
outputs: [`${p}_kcache`],
|
|
3652
|
-
attributes: {
|
|
3653
|
-
width: H,
|
|
3654
|
-
T_tensor: `${p}_k`
|
|
3655
|
-
}
|
|
3656
|
-
});
|
|
3657
|
-
addNode({
|
|
3658
|
-
id: `${p}_vappend`,
|
|
3659
|
-
opType: "KVCacheAppend",
|
|
3660
|
-
inputs: [`${p}_v`],
|
|
3661
|
-
outputs: [`${p}_vcache`],
|
|
3662
|
-
attributes: {
|
|
3663
|
-
width: H,
|
|
3664
|
-
T_tensor: `${p}_v`
|
|
3665
|
-
}
|
|
3666
|
-
});
|
|
3667
|
-
act(`${p}_sa`, ["T", H]);
|
|
3668
|
-
addNode({
|
|
3669
|
-
id: `${p}_self_attn`,
|
|
3670
|
-
opType: "Attention",
|
|
3671
|
-
inputs: [
|
|
3672
|
-
`${p}_q`,
|
|
3673
|
-
`${p}_kcache`,
|
|
3674
|
-
`${p}_vcache`
|
|
3675
|
-
],
|
|
3676
|
-
outputs: [`${p}_sa`],
|
|
3677
|
-
attributes: {
|
|
3678
|
-
hidden_size: H,
|
|
3679
|
-
num_q_heads: heads,
|
|
3680
|
-
num_kv_heads: c.dec_kv_heads,
|
|
3681
|
-
head_dim: headDim,
|
|
3682
|
-
causal: true,
|
|
3683
|
-
layer_index: i
|
|
3684
|
-
}
|
|
3685
|
-
});
|
|
3686
|
-
act(`${p}_sao`, ["T", H]);
|
|
3687
|
-
linear(`${p}_sap`, `${p}_sa`, `${lk}.self_attn.out_proj.weight`, `${p}_sao`, H, H);
|
|
3688
|
-
act(`${p}_res1`, ["T", H]);
|
|
3689
|
-
addNode({
|
|
3690
|
-
id: `${p}_add1`,
|
|
3691
|
-
opType: "Add",
|
|
3692
|
-
inputs: [prev, `${p}_sao`],
|
|
3693
|
-
outputs: [`${p}_res1`],
|
|
3694
|
-
attributes: {
|
|
3695
|
-
count_tensor: prev,
|
|
3696
|
-
hidden_size: H
|
|
3697
|
-
}
|
|
3698
|
-
});
|
|
3699
|
-
layernorm(`${p}_n2`, `${p}_res1`, `${lk}.encoder_attn_layer_norm`, `${p}_ln2`);
|
|
3700
|
-
act(`${p}_cq`, ["T", H]);
|
|
3701
|
-
linear(`${p}_cqp`, `${p}_ln2`, `${lk}.encoder_attn.q_proj.weight`, `${p}_cq`, H, H);
|
|
3702
|
-
const encK = `enc_k_layer${i}`;
|
|
3703
|
-
const encV = `enc_v_layer${i}`;
|
|
3704
|
-
act(encK, [sEnc, H]);
|
|
3705
|
-
act(encV, [sEnc, H]);
|
|
3706
|
-
inputs.push(encK, encV);
|
|
3707
|
-
act(`${p}_ca`, ["T", H]);
|
|
3708
|
-
addNode({
|
|
3709
|
-
id: `${p}_cross_attn`,
|
|
3710
|
-
opType: "CrossAttention",
|
|
3711
|
-
inputs: [
|
|
3712
|
-
`${p}_cq`,
|
|
3713
|
-
encK,
|
|
3714
|
-
encV
|
|
3715
|
-
],
|
|
3716
|
-
outputs: [`${p}_ca`],
|
|
3717
|
-
attributes: {
|
|
3718
|
-
num_q_heads: heads,
|
|
3719
|
-
num_kv_heads: c.dec_kv_heads,
|
|
3720
|
-
head_dim: headDim
|
|
3721
|
-
}
|
|
3722
|
-
});
|
|
3723
|
-
act(`${p}_cao`, ["T", H]);
|
|
3724
|
-
linear(`${p}_cap`, `${p}_ca`, `${lk}.encoder_attn.out_proj.weight`, `${p}_cao`, H, H);
|
|
3725
|
-
act(`${p}_res2`, ["T", H]);
|
|
3726
|
-
addNode({
|
|
3727
|
-
id: `${p}_add2`,
|
|
3728
|
-
opType: "Add",
|
|
3729
|
-
inputs: [`${p}_res1`, `${p}_cao`],
|
|
3730
|
-
outputs: [`${p}_res2`],
|
|
3731
|
-
attributes: {
|
|
3732
|
-
count_tensor: `${p}_res1`,
|
|
3733
|
-
hidden_size: H
|
|
3734
|
-
}
|
|
3735
|
-
});
|
|
3736
|
-
layernorm(`${p}_n3`, `${p}_res2`, `${lk}.final_layer_norm`, `${p}_ln3`);
|
|
3737
|
-
act(`${p}_fc1`, ["T", c.dec_ffn]);
|
|
3738
|
-
linear(`${p}_fc1p`, `${p}_ln3`, `${lk}.fc1.weight`, `${p}_fc1`, H, c.dec_ffn);
|
|
3739
|
-
act(`${p}_gelu`, ["T", c.dec_ffn]);
|
|
3740
|
-
addNode({
|
|
3741
|
-
id: `${p}_act`,
|
|
3742
|
-
opType: "GeluErf",
|
|
3743
|
-
inputs: [`${p}_fc1`],
|
|
3744
|
-
outputs: [`${p}_gelu`],
|
|
3745
|
-
attributes: { count_tensor: `${p}_fc1` }
|
|
3746
|
-
});
|
|
3747
|
-
act(`${p}_fc2`, ["T", H]);
|
|
3748
|
-
linear(`${p}_fc2p`, `${p}_gelu`, `${lk}.fc2.weight`, `${p}_fc2`, c.dec_ffn, H);
|
|
3749
|
-
act(`${p}_res3`, ["T", H]);
|
|
3750
|
-
addNode({
|
|
3751
|
-
id: `${p}_add3`,
|
|
3752
|
-
opType: "Add",
|
|
3753
|
-
inputs: [`${p}_res2`, `${p}_fc2`],
|
|
3754
|
-
outputs: [`${p}_res3`],
|
|
3755
|
-
attributes: {
|
|
3756
|
-
count_tensor: `${p}_res2`,
|
|
3757
|
-
hidden_size: H
|
|
3758
|
-
}
|
|
3759
|
-
});
|
|
3760
|
-
prev = `${p}_res3`;
|
|
3761
|
-
}
|
|
3762
|
-
layernorm("dec_final_norm", prev, "decoder.model.decoder.layer_norm", "dec_normed");
|
|
3763
|
-
act("dec_last", [1, H]);
|
|
3764
|
-
addNode({
|
|
3765
|
-
id: "slice_last",
|
|
3766
|
-
opType: "SliceLastRow",
|
|
3767
|
-
inputs: ["dec_normed"],
|
|
3768
|
-
outputs: ["dec_last"],
|
|
3769
|
-
attributes: { width: H }
|
|
3770
|
-
});
|
|
3771
|
-
const outputs = [];
|
|
3772
|
-
for (let cb = 0; cb < c.num_codebooks; cb++) {
|
|
3773
|
-
const headW = `decoder.lm_heads.${cb}.weight`;
|
|
3774
|
-
w(headW, [c.audio_vocab, H]);
|
|
3775
|
-
const logit = `logits_cb${cb}`;
|
|
3776
|
-
act(logit, [1, c.audio_vocab]);
|
|
3777
|
-
addNode({
|
|
3778
|
-
id: `lm_head${cb}`,
|
|
3779
|
-
opType: "MatMul",
|
|
3780
|
-
inputs: ["dec_last", headW],
|
|
3781
|
-
outputs: [logit],
|
|
3782
|
-
attributes: {
|
|
3783
|
-
M_tensor: "dec_last",
|
|
3784
|
-
K: H,
|
|
3785
|
-
N: c.audio_vocab
|
|
3786
|
-
}
|
|
3787
|
-
});
|
|
3788
|
-
outputs.push(logit);
|
|
3789
|
-
}
|
|
3790
|
-
return {
|
|
3791
|
-
architecture: "ParlerTTSDecoder",
|
|
3792
|
-
config: baseConfig(c, H, heads, c.dec_layers),
|
|
3793
|
-
capabilities: CAPS,
|
|
3794
|
-
tensors,
|
|
3795
|
-
nodes,
|
|
3796
|
-
executionOrder,
|
|
3797
|
-
inputs,
|
|
3798
|
-
outputs
|
|
3799
|
-
};
|
|
3800
|
-
}
|
|
3801
|
-
/**
|
|
3802
|
-
* Apply the Parler/MusicGen delay pattern to a [numCodebooks, T] code grid produced
|
|
3803
|
-
* by AR decode. During generation, codebook `i` is delayed by `i` steps: at decode
|
|
3804
|
-
* step t the model emits, for each codebook i, the code for ACOUSTIC frame (t − i).
|
|
3805
|
-
* To recover the aligned acoustic grid we simply read codebook i's stream starting
|
|
3806
|
-
* at offset i. This returns the de-delayed [numCodebooks, frames] grid (codebook-major)
|
|
3807
|
-
* ready for the DAC decoder, where frames = T − (numCodebooks − 1).
|
|
3808
|
-
*
|
|
3809
|
-
* `delayed` is codebook-major [numCodebooks, T] (delayed[i*T + t]).
|
|
3810
|
-
*/
|
|
3811
|
-
function revertDelayPattern(delayed, numCodebooks, T) {
|
|
3812
|
-
const frames = T - (numCodebooks - 1);
|
|
3813
|
-
if (frames <= 0) return {
|
|
3814
|
-
codes: new Uint32Array(0),
|
|
3815
|
-
frames: 0
|
|
3816
|
-
};
|
|
3817
|
-
const codes = new Uint32Array(numCodebooks * frames);
|
|
3818
|
-
for (let i = 0; i < numCodebooks; i++) for (let f = 0; f < frames; f++) codes[i * frames + f] = delayed[i * T + (f + i)];
|
|
3819
|
-
return {
|
|
3820
|
-
codes,
|
|
3821
|
-
frames
|
|
3822
|
-
};
|
|
3823
|
-
}
|
|
3824
|
-
|
|
3825
3269
|
//#endregion
|
|
3826
3270
|
//#region src/gpu/parler-tts.ts
|
|
3827
3271
|
/**
|
|
@@ -3847,12 +3291,18 @@ function revertDelayPattern(delayed, numCodebooks, T) {
|
|
|
3847
3291
|
* 4. Revert the delay pattern → [9, frames] code grid → shared DAC decode → 44.1 kHz
|
|
3848
3292
|
* PCM (tanh applied host-side).
|
|
3849
3293
|
*
|
|
3850
|
-
* STATUS —
|
|
3851
|
-
*
|
|
3852
|
-
*
|
|
3853
|
-
*
|
|
3854
|
-
*
|
|
3855
|
-
*
|
|
3294
|
+
* STATUS — WIRED + RUNS END-TO-END. Typechecks + builds, and the full pipeline runs
|
|
3295
|
+
* on Dawn (node): parler-tts/parler-tts-mini-v1 loads, the T5 encoder runs, the prompt
|
|
3296
|
+
* prefills, the 9-codebook delay-pattern AR loop decodes, and the shared DAC decoder
|
|
3297
|
+
* produces finite, non-silent, in-range 44.1 kHz PCM (verified via
|
|
3298
|
+
* scripts/engine/test-parler-speak.mjs). Reached from the high-level engine through
|
|
3299
|
+
* `g.speak(text, { model: DEFAULT_MODELS.ttsParler, describeVoice })`, which routes
|
|
3300
|
+
* Parler repos here directly (bypassing the single-graph text loader, the same way
|
|
3301
|
+
* MoonshineSTT is reached for STT). A bit-exact reference diff vs transformers
|
|
3302
|
+
* ParlerTTSForConditionalGeneration (T5 encoder cosine + a decoder forward; see
|
|
3303
|
+
* scripts/engine/parler-*-ref.py stubs) remains as a follow-up for numerical parity.
|
|
3304
|
+
* The architecture, weight layout, tokenization, delay pattern, and DAC config are all
|
|
3305
|
+
* verified against the live config.json + safetensors header.
|
|
3856
3306
|
*/
|
|
3857
3307
|
const MAP_MODE_READ$1 = 1;
|
|
3858
3308
|
/** Default voice description when the caller doesn't supply one. */
|
|
@@ -7272,4 +6722,4 @@ var WebGPUEngine = class WebGPUEngine {
|
|
|
7272
6722
|
|
|
7273
6723
|
//#endregion
|
|
7274
6724
|
export { dequantizeGemma4VisionProjection as A, audioTokensToDacCodes as C, parseOuteTtsConfig as D, generateOuteTtsBackboneGraph as E, generateGemma4VisionGraph as M, patchGemma4VisionClips as N, KaniTTS as O, resolveGemma4VisionInfo as P, loadOuteSpeaker as S, generateDacSpeechDecoderGraph as T, smartResize as _, buildGemma4PosEmbeds as a, OuteTTS as b, buildMRoPECosSin as c, buildPositionIds as d, buildRotaryCosSin as f, preprocessImageGemma4 as g, preprocessImage as h, buildGemma4PoolMatrix as i, dequantizeMLXProjection as j, generateQwen3_5VisionGraph as k, buildMRoPEPositionIds as l, mropeFreqDims as m, GEMMA4_IMAGE_PROCESSOR as n, buildGemma4RotaryCosSin as o, buildVisionPositionTensors as p, QWEN3_5_IMAGE_PROCESSOR as r, buildGemma4VisionPositionTensors as s, WebGPUEngine as t, buildPosEmbeds as u, VisionExecutor as v, dacOutputLength as w, buildOutePromptString as x, ParlerTTS as y };
|
|
7275
|
-
//# sourceMappingURL=gpu-
|
|
6725
|
+
//# sourceMappingURL=gpu-CzgbyVeq.mjs.map
|