npm - whisper.rn - Versions diffs - 0.5.2 → 0.5.4 - Mend

whisper.rn 0.5.2 → 0.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (68) hide show

package/cpp/ggml-cpu/ops.cpp CHANGED Viewed

@@ -7,8 +7,10 @@
 #include "unary-ops.h"
 #include "vec.h"
-#include <float.h>
+#include <cfloat>
 #include <algorithm>
+#include <cmath>
+#include <functional>
 // wsp_ggml_compute_forward_dup
@@ -1394,6 +1396,56 @@ void wsp_ggml_compute_forward_sum(
     }
 }
+// wsp_ggml_compute_forward_cumsum
+static void wsp_ggml_compute_forward_cumsum_f32(
+        const wsp_ggml_compute_params * params,
+        wsp_ggml_tensor * dst) {
+    const wsp_ggml_tensor * src0 = dst->src[0];
+    WSP_GGML_ASSERT(src0->nb[0] == sizeof(float));
+    WSP_GGML_ASSERT(dst->nb[0] == sizeof(float));
+    WSP_GGML_TENSOR_UNARY_OP_LOCALS
+    WSP_GGML_ASSERT(ne0 == ne00);
+    WSP_GGML_ASSERT(ne1 == ne01);
+    WSP_GGML_ASSERT(ne2 == ne02);
+    WSP_GGML_ASSERT(ne3 == ne03);
+    const auto [ir0, ir1] = get_thread_range(params, src0);
+    for (int64_t ir = ir0; ir < ir1; ++ir) {
+        const int64_t i03 = ir/(ne02*ne01);
+        const int64_t i02 = (ir - i03*ne02*ne01)/ne01;
+        const int64_t i01 = (ir - i03*ne02*ne01 - i02*ne01);
+        float * src_row = (float *) ((char *) src0->data + i01*nb01 + i02*nb02 + i03*nb03);
+        float * dst_row = (float *) ((char *) dst->data  + i01*nb1  + i02*nb2  + i03*nb3);
+        wsp_ggml_vec_cumsum_f32(ne00, dst_row, src_row);
+    }
+}
+void wsp_ggml_compute_forward_cumsum(
+        const wsp_ggml_compute_params * params,
+        wsp_ggml_tensor * dst) {
+    const wsp_ggml_tensor * src0 = dst->src[0];
+    switch (src0->type) {
+        case WSP_GGML_TYPE_F32:
+            {
+                wsp_ggml_compute_forward_cumsum_f32(params, dst);
+            } break;
+        default:
+            {
+                WSP_GGML_ABORT("fatal error");
+            }
+    }
+}
 // wsp_ggml_compute_forward_sum_rows
 static void wsp_ggml_compute_forward_sum_rows_f32(
@@ -2140,6 +2192,83 @@ static void wsp_ggml_compute_forward_gelu(
     }
 }
+// wsp_ggml_compute_fill
+static void wsp_ggml_compute_forward_fill_f32(const wsp_ggml_compute_params * params, wsp_ggml_tensor * dst) {
+    const float c = wsp_ggml_get_op_params_f32(dst, 0);
+    WSP_GGML_TENSOR_LOCALS(int64_t, ne, dst, ne);
+    WSP_GGML_TENSOR_LOCALS(size_t,  nb, dst, nb);
+    const auto [ir0, ir1] = get_thread_range(params, dst);
+    for (int64_t ir = ir0; ir < ir1; ++ir) {
+        const int64_t i03 = ir/(ne2*ne1);
+        const int64_t i02 = (ir - i03*ne2*ne1)/ne1;
+        const int64_t i01 = (ir - i03*ne2*ne1 - i02*ne1);
+        float * dst_ptr  = (float *) ((char *) dst->data + i03*nb3 + i02*nb2 + i01*nb1);
+        wsp_ggml_vec_set_f32(ne0, dst_ptr, c);
+    }
+}
+void wsp_ggml_compute_forward_fill(const wsp_ggml_compute_params * params, wsp_ggml_tensor * dst) {
+    wsp_ggml_compute_forward_fill_f32(params, dst);
+}
+// wsp_ggml_compute_tri
+static void wsp_ggml_compute_forward_tri_f32(const wsp_ggml_compute_params * params, wsp_ggml_tensor * dst) {
+    const wsp_ggml_tensor * src0 = dst->src[0];
+    const wsp_ggml_tri_type ttype = (wsp_ggml_tri_type) wsp_ggml_get_op_params_i32(dst, 0);
+    WSP_GGML_ASSERT(wsp_ggml_is_contiguous(src0));
+    WSP_GGML_TENSOR_UNARY_OP_LOCALS
+    const auto [ir0, ir1] = get_thread_range(params, src0);
+    bool (*bipred)(int, int);
+    switch (ttype) {
+        case WSP_GGML_TRI_TYPE_LOWER:      bipred = [](int i, int r) { return i <  r; }; break;
+        case WSP_GGML_TRI_TYPE_LOWER_DIAG: bipred = [](int i, int r) { return i <= r; }; break;
+        case WSP_GGML_TRI_TYPE_UPPER:      bipred = [](int i, int r) { return i >  r; }; break;
+        case WSP_GGML_TRI_TYPE_UPPER_DIAG: bipred = [](int i, int r) { return i >= r; }; break;
+        default: WSP_GGML_ABORT("invalid tri type");
+    }
+    for (int64_t ir = ir0; ir < ir1; ++ir) {
+        const int64_t i03 = ir/(ne02*ne01);
+        const int64_t i02 = (ir - i03*ne02*ne01)/ne01;
+        const int64_t i01 = (ir - i03*ne02*ne01 - i02*ne01);
+        const float * src_ptr = (const float  *) ((const char *) src0->data + i03*nb03 + i02*nb02 + i01*nb01);
+              float * dst_ptr = (      float  *) ((      char *) dst->data  + i03*nb3  + i02*nb2  + i01*nb1);
+        for (int i0 = 0; i0 < ne0; ++i0) {
+            dst_ptr[i0] = bipred(i0, i01) ? src_ptr[i0] : 0.0f;
+        }
+    }
+}
+void wsp_ggml_compute_forward_tri(const wsp_ggml_compute_params * params, wsp_ggml_tensor * dst) {
+    const wsp_ggml_tensor * src0 = dst->src[0];
+    switch (src0->type) {
+        case WSP_GGML_TYPE_F32:
+            {
+                wsp_ggml_compute_forward_tri_f32(params, dst);
+            } break;
+        default:
+            {
+                WSP_GGML_ABORT("fatal error");
+            }
+    }
+}
 // wsp_ggml_compute_forward_gelu_erf
 static void wsp_ggml_compute_forward_gelu_erf_f32(
@@ -4455,46 +4584,6 @@ void wsp_ggml_compute_forward_cont(
     wsp_ggml_compute_forward_dup(params, dst);
 }
-// wsp_ggml_compute_forward_reshape
-void wsp_ggml_compute_forward_reshape(
-        const wsp_ggml_compute_params * params,
-        wsp_ggml_tensor * dst) {
-    // NOP
-    WSP_GGML_UNUSED(params);
-    WSP_GGML_UNUSED(dst);
-}
-// wsp_ggml_compute_forward_view
-void wsp_ggml_compute_forward_view(
-        const wsp_ggml_compute_params * params,
-        wsp_ggml_tensor * dst) {
-    // NOP
-    WSP_GGML_UNUSED(params);
-    WSP_GGML_UNUSED(dst);
-}
-// wsp_ggml_compute_forward_permute
-void wsp_ggml_compute_forward_permute(
-        const wsp_ggml_compute_params * params,
-        wsp_ggml_tensor * dst) {
-    // NOP
-    WSP_GGML_UNUSED(params);
-    WSP_GGML_UNUSED(dst);
-}
-// wsp_ggml_compute_forward_transpose
-void wsp_ggml_compute_forward_transpose(
-        const wsp_ggml_compute_params * params,
-        wsp_ggml_tensor * dst) {
-    // NOP
-    WSP_GGML_UNUSED(params);
-    WSP_GGML_UNUSED(dst);
-}
 // wsp_ggml_compute_forward_get_rows
 static void wsp_ggml_compute_forward_get_rows_q(
@@ -5474,7 +5563,7 @@ static void wsp_ggml_rope_cache_init(
 }
 static void wsp_ggml_mrope_cache_init(
-     float theta_base_t, float theta_base_h, float theta_base_w, float theta_base_e, int sections[4], bool indep_sects,
+     float theta_base_t, float theta_base_h, float theta_base_w, float theta_base_e, int sections[4], bool is_imrope, bool indep_sects,
      float freq_scale, const float * freq_factors, float corr_dims[2], int64_t ne0, float ext_factor, float mscale,
      float * cache, float sin_sign, float theta_scale) {
     // ref: https://github.com/jquesnelle/yarn/blob/master/scaled_rope/LlamaYaRNScaledRotaryEmbedding.py
@@ -5509,14 +5598,26 @@ static void wsp_ggml_mrope_cache_init(
         }
         float theta = theta_t;
-        if (sector >= sections[0] && sector < sec_w) {
-            theta = theta_h;
-        }
-        else if (sector >= sec_w && sector < sec_w + sections[2]) {
-            theta = theta_w;
-        }
-        else if (sector >= sec_w + sections[2]) {
-            theta = theta_e;
+        if (is_imrope) { // qwen3vl apply interleaved mrope
+            if (sector % 3 == 1 && sector < 3 * sections[1]) {
+                theta = theta_h;
+            } else if (sector % 3 == 2 && sector < 3 * sections[2]) {
+                theta = theta_w;
+            } else if (sector % 3 == 0 && sector < 3 * sections[0]) {
+                theta = theta_t;
+            } else {
+                theta = theta_e;
+            }
+        } else {
+            if (sector >= sections[0] && sector < sec_w) {
+                theta = theta_h;
+            }
+            else if (sector >= sec_w && sector < sec_w + sections[2]) {
+                theta = theta_w;
+            }
+            else if (sector >= sec_w + sections[2]) {
+                theta = theta_e;
+            }
         }
         rope_yarn(
@@ -5531,193 +5632,28 @@ static void wsp_ggml_mrope_cache_init(
     }
 }
-static void wsp_ggml_compute_forward_rope_f32(
-        const wsp_ggml_compute_params * params,
-        wsp_ggml_tensor * dst,
-        const bool forward) {
-    const wsp_ggml_tensor * src0 = dst->src[0];
-    const wsp_ggml_tensor * src1 = dst->src[1];
-    const wsp_ggml_tensor * src2 = dst->src[2];
-    float freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow;
-    int sections[4];
-    //const int n_past     = ((int32_t *) dst->op_params)[0];
-    const int n_dims     = ((int32_t *) dst->op_params)[1];
-    const int mode       = ((int32_t *) dst->op_params)[2];
-    //const int n_ctx      = ((int32_t *) dst->op_params)[3];
-    const int n_ctx_orig = ((int32_t *) dst->op_params)[4];
-    memcpy(&freq_base,   (int32_t *) dst->op_params +  5, sizeof(float));
-    memcpy(&freq_scale,  (int32_t *) dst->op_params +  6, sizeof(float));
-    memcpy(&ext_factor,  (int32_t *) dst->op_params +  7, sizeof(float));
-    memcpy(&attn_factor, (int32_t *) dst->op_params +  8, sizeof(float));
-    memcpy(&beta_fast,   (int32_t *) dst->op_params +  9, sizeof(float));
-    memcpy(&beta_slow,   (int32_t *) dst->op_params + 10, sizeof(float));
-    memcpy(&sections,    (int32_t *) dst->op_params + 11, sizeof(int)*4);
-    WSP_GGML_TENSOR_UNARY_OP_LOCALS
-    //printf("ne0: %d, ne1: %d, ne2: %d, ne3: %d\n", ne0, ne1, ne2, ne3);
-    //printf("n_past = %d, ne2 = %d\n", n_past, ne2);
-    WSP_GGML_ASSERT(nb00 == sizeof(float));
-    const int ith = params->ith;
-    const int nth = params->nth;
-    const int nr = wsp_ggml_nrows(dst);
-    WSP_GGML_ASSERT(n_dims <= ne0);
-    WSP_GGML_ASSERT(n_dims % 2 == 0);
-    // rows per thread
-    const int dr = (nr + nth - 1)/nth;
-    // row range for this thread
-    const int ir0 = dr*ith;
-    const int ir1 = MIN(ir0 + dr, nr);
-    // row index used to determine which thread to use
-    int ir = 0;
-    const float theta_scale = powf(freq_base, -2.0f/n_dims);
-    float corr_dims[2];
-    wsp_ggml_rope_yarn_corr_dims(n_dims, n_ctx_orig, freq_base, beta_fast, beta_slow, corr_dims);
-    const bool is_neox = mode & WSP_GGML_ROPE_TYPE_NEOX;
-    const bool is_mrope = mode & WSP_GGML_ROPE_TYPE_MROPE;  // wsp_ggml_rope_multi, multimodal rotary position embedding
-    const bool is_vision = mode == WSP_GGML_ROPE_TYPE_VISION;
-    if (is_mrope) {
-        WSP_GGML_ASSERT(sections[0] > 0 || sections[1] > 0 || sections[2] > 0);
-    }
-    if (is_vision) {
-        WSP_GGML_ASSERT(n_dims == ne0/2);
-    }
-    const float * freq_factors = NULL;
-    if (src2 != NULL) {
-        WSP_GGML_ASSERT(src2->type == WSP_GGML_TYPE_F32);
-        WSP_GGML_ASSERT(src2->ne[0] >= n_dims / 2);
-        freq_factors = (const float *) src2->data;
-    }
-    // backward process uses inverse rotation by cos and sin.
-    // cos and sin build a rotation matrix, where the inverse is the transpose.
-    // this essentially just switches the sign of sin.
-    const float sin_sign = forward ? 1.0f : -1.0f;
-    const int32_t * pos = (const int32_t *) src1->data;
-    for (int64_t i3 = 0; i3 < ne3; i3++) { // batch
-        for (int64_t i2 = 0; i2 < ne2; i2++) { // seq-len
-            float * cache = (float *) params->wdata + (ne0 + CACHE_LINE_SIZE_F32)*ith;
-            if (!is_mrope) {
-                const int64_t p = pos[i2];
-                wsp_ggml_rope_cache_init(p, freq_scale, freq_factors, corr_dims, ne0, ext_factor, attn_factor, cache, sin_sign, theta_scale);
-            }
-            else {
-                const int64_t p_t = pos[i2];
-                const int64_t p_h = pos[i2 + ne2];
-                const int64_t p_w = pos[i2 + ne2 * 2];
-                const int64_t p_e = pos[i2 + ne2 * 3];
-                wsp_ggml_mrope_cache_init(
-                    p_t, p_h, p_w, p_e, sections, is_vision,
-                    freq_scale, freq_factors, corr_dims, ne0, ext_factor, attn_factor, cache, sin_sign, theta_scale);
-            }
-            for (int64_t i1 = 0; i1 < ne1; i1++) { // attn-heads
-                if (ir++ < ir0) continue;
-                if (ir   > ir1) break;
-                if (is_neox || is_mrope) {
-                    if (is_vision){
-                        for (int64_t i0 = 0; i0 < n_dims; i0 += 2) {
-                            const int64_t ic = i0/2;
-                            const float cos_theta = cache[i0 + 0];
-                            const float sin_theta = cache[i0 + 1];
-                            const float * const src = (float *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00);
-                            float * dst_data  = (float *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + ic*nb0);
-                            const float x0 = src[0];
-                            const float x1 = src[n_dims];
-                            dst_data[0]      = x0*cos_theta - x1*sin_theta;
-                            dst_data[n_dims] = x0*sin_theta + x1*cos_theta;
-                        }
-                    } else {
-                        for (int64_t i0 = 0; i0 < n_dims; i0 += 2) {
-                            const int64_t ic = i0/2;
+template<typename T>
+static void rotate_pairs(const int64_t n, const int64_t n_offset, const float * cache, const T * src_data, T * dst_data, const int scale = 2) {
+  for (int64_t i0 = 0; i0 < n; i0 += 2) {
+    const int64_t ic = i0/scale; // hack for WSP_GGML_ROPE_TYPE_NORMAL, where we need ic = i0; for all other cases, ic = i0/2
-                            const float cos_theta = cache[i0 + 0];
-                            const float sin_theta = cache[i0 + 1];
+    const float cos_theta = cache[i0 + 0];
+    const float sin_theta = cache[i0 + 1];
-                            const float * const src = (float *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00);
-                            float * dst_data  = (float *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + ic*nb0);
+    const T * const src = src_data + ic;
+    T * dst             = dst_data + ic;
-                            const float x0 = src[0];
-                            const float x1 = src[n_dims/2];
+    const float x0 = type_conversion_table<T>::to_f32(src[0]);
+    const float x1 = type_conversion_table<T>::to_f32(src[n_offset]);
-                            dst_data[0]        = x0*cos_theta - x1*sin_theta;
-                            dst_data[n_dims/2] = x0*sin_theta + x1*cos_theta;
-                        }
-                    }
-                } else {
-                    for (int64_t i0 = 0; i0 < n_dims; i0 += 2) {
-                        const float cos_theta = cache[i0 + 0];
-                        const float sin_theta = cache[i0 + 1];
-                        const float * const src = (float *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00);
-                              float * dst_data  = (float *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + i0*nb0);
-                        const float x0 = src[0];
-                        const float x1 = src[1];
-                        dst_data[0] = x0*cos_theta - x1*sin_theta;
-                        dst_data[1] = x0*sin_theta + x1*cos_theta;
-                    }
-                }
-                if (is_vision) {
-                    for (int64_t i0 = n_dims; i0 < ne0; i0 += 2) {
-                        const int64_t ic = i0/2;
-                        const float cos_theta = cache[i0 + 0];
-                        const float sin_theta = cache[i0 + 1];
-                        const float * const src = (float *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00);
-                        float * dst_data  = (float *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + ic*nb0);
-                        const float x0 = src[0];
-                        const float x1 = src[n_dims];
-                        dst_data[0]      = x0*cos_theta - x1*sin_theta;
-                        dst_data[n_dims] = x0*sin_theta + x1*cos_theta;
-                    }
-                } else {
-                    // fill the remain channels with data from src tensor
-                    for (int64_t i0 = n_dims; i0 < ne0; i0 += 2) {
-                        const float * const src = (float *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00);
-                        float * dst_data  = (float *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + i0*nb0);
-                        dst_data[0] = src[0];
-                        dst_data[1] = src[1];
-                    }
-                }
-            }
-        }
-    }
+    dst[0]        = type_conversion_table<T>::from_f32(x0*cos_theta - x1*sin_theta);
+    dst[n_offset] = type_conversion_table<T>::from_f32(x0*sin_theta + x1*cos_theta);
+  }
 }
-// TODO: deduplicate f16/f32 code
-static void wsp_ggml_compute_forward_rope_f16(
+template<typename T> //float or wsp_ggml_fp16_t
+static void wsp_ggml_compute_forward_rope_flt(
         const wsp_ggml_compute_params * params,
         wsp_ggml_tensor * dst,
         const bool forward) {
@@ -5726,6 +5662,9 @@ static void wsp_ggml_compute_forward_rope_f16(
     const wsp_ggml_tensor * src1 = dst->src[1];
     const wsp_ggml_tensor * src2 = dst->src[2];
+    WSP_GGML_ASSERT(src0->type == WSP_GGML_TYPE_F32 || src0->type == WSP_GGML_TYPE_F16);
+    WSP_GGML_ASSERT(src1->type == WSP_GGML_TYPE_I32);
     float freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow;
     int sections[4];
@@ -5734,6 +5673,7 @@ static void wsp_ggml_compute_forward_rope_f16(
     const int mode       = ((int32_t *) dst->op_params)[2];
     //const int n_ctx      = ((int32_t *) dst->op_params)[3];
     const int n_ctx_orig = ((int32_t *) dst->op_params)[4];
     memcpy(&freq_base,   (int32_t *) dst->op_params +  5, sizeof(float));
     memcpy(&freq_scale,  (int32_t *) dst->op_params +  6, sizeof(float));
     memcpy(&ext_factor,  (int32_t *) dst->op_params +  7, sizeof(float));
@@ -5742,13 +5682,13 @@ static void wsp_ggml_compute_forward_rope_f16(
     memcpy(&beta_slow,   (int32_t *) dst->op_params + 10, sizeof(float));
     memcpy(&sections,    (int32_t *) dst->op_params + 11, sizeof(int)*4);
     WSP_GGML_TENSOR_UNARY_OP_LOCALS
     //printf("ne0: %d, ne1: %d, ne2: %d, ne3: %d\n", ne0, ne1, ne2, ne3);
     //printf("n_past = %d, ne2 = %d\n", n_past, ne2);
-    WSP_GGML_ASSERT(nb0 == sizeof(wsp_ggml_fp16_t));
+    WSP_GGML_ASSERT(nb0 == nb00);
+    WSP_GGML_ASSERT(nb0 == sizeof(T));
     const int ith = params->ith;
     const int nth = params->nth;
@@ -5773,11 +5713,11 @@ static void wsp_ggml_compute_forward_rope_f16(
     float corr_dims[2];
     wsp_ggml_rope_yarn_corr_dims(n_dims, n_ctx_orig, freq_base, beta_fast, beta_slow, corr_dims);
-    const bool is_neox = mode & WSP_GGML_ROPE_TYPE_NEOX;
-    const bool is_mrope = mode & WSP_GGML_ROPE_TYPE_MROPE;
+    const bool is_imrope = mode == WSP_GGML_ROPE_TYPE_IMROPE; // qwen3vl apply interleaved mrope
+    const bool mrope_used = mode & WSP_GGML_ROPE_TYPE_MROPE;  // wsp_ggml_rope_multi, note: also true for vision (24 & 8 == true) and for imrope
     const bool is_vision = mode == WSP_GGML_ROPE_TYPE_VISION;
-    if (is_mrope) {
+    if (mrope_used) {
         WSP_GGML_ASSERT(sections[0] > 0 || sections[1] > 0 || sections[2] > 0);
     }
@@ -5799,11 +5739,11 @@ static void wsp_ggml_compute_forward_rope_f16(
     const int32_t * pos = (const int32_t *) src1->data;
-    for (int64_t i3 = 0; i3 < ne3; i3++) {
-        for (int64_t i2 = 0; i2 < ne2; i2++) {
+    for (int64_t i3 = 0; i3 < ne3; i3++) { // batch
+        for (int64_t i2 = 0; i2 < ne2; i2++) { // seq-len
             float * cache = (float *) params->wdata + (ne0 + CACHE_LINE_SIZE_F32)*ith;
-            if (!is_mrope) {
+            if (!mrope_used) {
                 const int64_t p = pos[i2];
                 wsp_ggml_rope_cache_init(p, freq_scale, freq_factors, corr_dims, ne0, ext_factor, attn_factor, cache, sin_sign, theta_scale);
             }
@@ -5813,90 +5753,44 @@ static void wsp_ggml_compute_forward_rope_f16(
                 const int64_t p_w = pos[i2 + ne2 * 2];
                 const int64_t p_e = pos[i2 + ne2 * 3];
                 wsp_ggml_mrope_cache_init(
-                    p_t, p_h, p_w, p_e, sections, is_vision,
+                    p_t, p_h, p_w, p_e, sections, is_imrope, is_vision,
                     freq_scale, freq_factors, corr_dims, ne0, ext_factor, attn_factor, cache, sin_sign, theta_scale);
             }
-            for (int64_t i1 = 0; i1 < ne1; i1++) {
+            for (int64_t i1 = 0; i1 < ne1; i1++) { // attn-heads
                 if (ir++ < ir0) continue;
                 if (ir   > ir1) break;
-                if (is_neox || is_mrope) {
-                    if (is_vision) {
-                        for (int64_t i0 = 0; i0 < n_dims; i0 += 2) {
-                            const int64_t ic = i0/2;
-                            const float cos_theta = cache[i0 + 0];
-                            const float sin_theta = cache[i0 + 1];
-                            const wsp_ggml_fp16_t * const src = (wsp_ggml_fp16_t *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00);
-                            wsp_ggml_fp16_t * dst_data  = (wsp_ggml_fp16_t *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + ic*nb0);
-                            const float x0 = WSP_GGML_CPU_FP16_TO_FP32(src[0]);
-                            const float x1 = WSP_GGML_CPU_FP16_TO_FP32(src[n_dims]);
-                            dst_data[0]      = WSP_GGML_CPU_FP32_TO_FP16(x0*cos_theta - x1*sin_theta);
-                            dst_data[n_dims] = WSP_GGML_CPU_FP32_TO_FP16(x0*sin_theta + x1*cos_theta);
-                        }
-                    } else {
-                        for (int64_t i0 = 0; i0 < n_dims; i0 += 2) {
-                            const int64_t ic = i0/2;
-                            const float cos_theta = cache[i0 + 0];
-                            const float sin_theta = cache[i0 + 1];
-                            const wsp_ggml_fp16_t * const src = (wsp_ggml_fp16_t *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00);
-                            wsp_ggml_fp16_t * dst_data  = (wsp_ggml_fp16_t *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + ic*nb0);
-                            const float x0 = WSP_GGML_CPU_FP16_TO_FP32(src[0]);
-                            const float x1 = WSP_GGML_CPU_FP16_TO_FP32(src[n_dims/2]);
-                            dst_data[0]        = WSP_GGML_CPU_FP32_TO_FP16(x0*cos_theta - x1*sin_theta);
-                            dst_data[n_dims/2] = WSP_GGML_CPU_FP32_TO_FP16(x0*sin_theta + x1*cos_theta);
-                        }
-                    }
-                } else {
-                    for (int64_t i0 = 0; i0 < n_dims; i0 += 2) {
-                        const float cos_theta = cache[i0 + 0];
-                        const float sin_theta = cache[i0 + 1];
-                        const wsp_ggml_fp16_t * const src = (wsp_ggml_fp16_t *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00);
-                              wsp_ggml_fp16_t * dst_data  = (wsp_ggml_fp16_t *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + i0*nb0);
-                        const float x0 = WSP_GGML_CPU_FP16_TO_FP32(src[0]);
-                        const float x1 = WSP_GGML_CPU_FP16_TO_FP32(src[1]);
-                        dst_data[0] = WSP_GGML_CPU_FP32_TO_FP16(x0*cos_theta - x1*sin_theta);
-                        dst_data[1] = WSP_GGML_CPU_FP32_TO_FP16(x0*sin_theta + x1*cos_theta);
-                    }
+                T * src = (T *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01);
+                T * dst_data  = (T *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1);
+                switch (mode) {
+                    case WSP_GGML_ROPE_TYPE_NORMAL:
+                        rotate_pairs<T>(n_dims, 1, cache, src, dst_data, 1);
+                        break;
+                    case WSP_GGML_ROPE_TYPE_NEOX:
+                    case WSP_GGML_ROPE_TYPE_MROPE:
+                    case WSP_GGML_ROPE_TYPE_IMROPE:
+                        rotate_pairs<T>(n_dims, n_dims/2, cache, src, dst_data);
+                        break;
+                    case WSP_GGML_ROPE_TYPE_VISION:
+                        rotate_pairs<T>(ne0, n_dims, cache, src, dst_data);
+                        break;
+                    default:
+                        WSP_GGML_ABORT("rope type not supported");
                 }
-                if (is_vision) {
-                    for (int64_t i0 = n_dims; i0 < ne0; i0 += 2) {
-                        const int64_t ic = i0/2;
-                        const float cos_theta = cache[i0 + 0];
-                        const float sin_theta = cache[i0 + 1];
-                        const wsp_ggml_fp16_t * const src = (wsp_ggml_fp16_t *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00);
-                        wsp_ggml_fp16_t * dst_data  = (wsp_ggml_fp16_t *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + ic*nb0);
-                        const float x0 = WSP_GGML_CPU_FP16_TO_FP32(src[0]);
-                        const float x1 = WSP_GGML_CPU_FP16_TO_FP32(src[n_dims]);
-                        dst_data[0]      = WSP_GGML_CPU_FP32_TO_FP16(x0*cos_theta - x1*sin_theta);
-                        dst_data[n_dims] = WSP_GGML_CPU_FP32_TO_FP16(x0*sin_theta + x1*cos_theta);
-                    }
-                } else {
+                if (!is_vision) {
+                    // fill the remain channels with data from src tensor
                     for (int64_t i0 = n_dims; i0 < ne0; i0 += 2) {
-                        const wsp_ggml_fp16_t * const src = (wsp_ggml_fp16_t *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00);
-                        wsp_ggml_fp16_t * dst_data  = (wsp_ggml_fp16_t *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + i0*nb0);
+                        const T * const src = (T *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00);
+                        T * dst_data  = (T *)((char *)  dst->data + i3*nb3  + i2*nb2  + i1*nb1  + i0*nb0);
                         dst_data[0] = src[0];
                         dst_data[1] = src[1];
                     }
                 }
-            }
+            } //attn-heads
         }
     }
 }
@@ -5910,11 +5804,11 @@ void wsp_ggml_compute_forward_rope(
     switch (src0->type) {
         case WSP_GGML_TYPE_F16:
             {
-                wsp_ggml_compute_forward_rope_f16(params, dst, true);
+                wsp_ggml_compute_forward_rope_flt<wsp_ggml_fp16_t>(params, dst, true);
             } break;
         case WSP_GGML_TYPE_F32:
             {
-                wsp_ggml_compute_forward_rope_f32(params, dst, true);
+                wsp_ggml_compute_forward_rope_flt<float>(params, dst, true);
             } break;
         default:
             {
@@ -5934,11 +5828,11 @@ void wsp_ggml_compute_forward_rope_back(
     switch (src0->type) {
         case WSP_GGML_TYPE_F16:
             {
-                wsp_ggml_compute_forward_rope_f16(params, dst, false);
+                wsp_ggml_compute_forward_rope_flt<wsp_ggml_fp16_t>(params, dst, false);
             } break;
         case WSP_GGML_TYPE_F32:
             {
-                wsp_ggml_compute_forward_rope_f32(params, dst, false);
+                wsp_ggml_compute_forward_rope_flt<float>(params, dst, false);
             } break;
         default:
             {
@@ -7070,7 +6964,11 @@ static void wsp_ggml_compute_forward_conv_2d_dw_cwhn(
     const int64_t row_end = MIN(row_start + rows_per_thread, rows_total);
 #ifdef WSP_GGML_SIMD
-    const int64_t pkg_size = WSP_GGML_F32_EPR;
+    #if defined(__ARM_FEATURE_SVE)
+        const int64_t pkg_size = svcntw();
+    #else
+        const int64_t pkg_size = WSP_GGML_F32_EPR;
+    #endif
     const int64_t pkg_count = c / pkg_size;
     const int64_t c_pkg_end = pkg_count * pkg_size;
 #else
@@ -7493,10 +7391,17 @@ static void wsp_ggml_compute_forward_upscale_f32(
     float sf1 = (float)ne1/src0->ne[1];
     float sf2 = (float)ne2/src0->ne[2];
     float sf3 = (float)ne3/src0->ne[3];
+    float pixel_offset = 0.5f;
     const int32_t mode_flags = wsp_ggml_get_op_params_i32(dst, 0);
     const wsp_ggml_scale_mode mode = (wsp_ggml_scale_mode) (mode_flags & 0xFF);
+    if (mode_flags & WSP_GGML_SCALE_FLAG_ALIGN_CORNERS) {
+        pixel_offset = 0.0f;
+        sf0 = ne0 > 1 && ne00 > 1 ? (float)(ne0 - 1) / (ne00 - 1) : sf0;
+        sf1 = ne1 > 1 && ne01 > 1 ? (float)(ne1 - 1) / (ne01 - 1) : sf1;
+    }
     if (mode == WSP_GGML_SCALE_MODE_NEAREST) {
         for (int64_t i3 = 0; i3 < ne3; i3++) {
             const int64_t i03 = i3 / sf3;
@@ -7516,13 +7421,6 @@ static void wsp_ggml_compute_forward_upscale_f32(
             }
         }
     } else if (mode == WSP_GGML_SCALE_MODE_BILINEAR) {
-        float pixel_offset = 0.5f;
-        if (mode_flags & WSP_GGML_SCALE_FLAG_ALIGN_CORNERS) {
-            pixel_offset = 0.0f;
-            sf0 = (float)(ne0 - 1) / (src0->ne[0] - 1);
-            sf1 = (float)(ne1 - 1) / (src0->ne[1] - 1);
-        }
         for (int64_t i3 = 0; i3 < ne3; i3++) {
             const int64_t i03 = i3 / sf3;
             for (int64_t i2 = ith; i2 < ne2; i2 += nth) {
@@ -7557,6 +7455,51 @@ static void wsp_ggml_compute_forward_upscale_f32(
                         const float val = a*(1 - dx)*(1 - dy) + b*dx*(1 - dy) + c*(1 - dx)*dy + d*dx*dy;
+                        float * y_dst = (float *)((char *)dst->data + i0*nb0 + i1*nb1 + i2*nb2 + i3*nb3);
+                        *y_dst = val;
+                    }
+                }
+            }
+        }
+    } else if (mode == WSP_GGML_SCALE_MODE_BICUBIC) {
+        // https://en.wikipedia.org/wiki/Bicubic_interpolation#Bicubic_convolution_algorithm
+        const float a = -0.75f; // use alpha = -0.75 (same as PyTorch)
+        auto weight1 = [a](float x) { return ((a + 2) * x - (a + 3)) * x * x + 1; };
+        auto weight2 = [a](float x) { return ((a * x - 5 * a) * x + 8 * a) * x - 4 * a; };
+        auto bicubic = [=](float p0, float p1, float p2, float p3, float x) {
+            const float w0 = weight2(x + 1);
+            const float w1 = weight1(x + 0);
+            const float w2 = weight1(1 - x);
+            const float w3 = weight2(2 - x);
+            return p0*w0 + p1*w1 + p2*w2 + p3*w3;
+        };
+        for (int64_t i3 = 0; i3 < ne3; i3++) {
+            const int64_t i03 = i3 / sf3;
+            for (int64_t i2 = ith; i2 < ne2; i2 += nth) {
+                const int64_t i02 = i2 / sf2;
+                for (int64_t i1 = 0; i1 < ne1; i1++) {
+                    const float y = ((float)i1 + pixel_offset) / sf1 - pixel_offset;
+                    const int64_t y0 = (int64_t)floorf(y);
+                    const float dy = y - (float)y0;
+                    for (int64_t i0 = 0; i0 < ne0; i0++) {
+                        const float x = ((float)i0 + pixel_offset) / sf0 - pixel_offset;
+                        const int64_t x0 = (int64_t)floorf(x);
+                        const float dx = x - (float)x0;
+                        auto p = [=](int64_t x_off, int64_t y_off) -> float {
+                            int64_t i00 = std::max(int64_t(0), std::min(x0 + x_off, ne00 - 1));
+                            int64_t i01 = std::max(int64_t(0), std::min(y0 + y_off, ne01 - 1));
+                            return *(const float *)((const char *)src0->data + i00*nb00 + i01*nb01 + i02*nb02 + i03*nb03);
+                        };
+                        const float val = bicubic(
+                            bicubic(p(-1,-1), p(0,-1), p(1,-1), p(2,-1), dx),
+                            bicubic(p(-1, 0), p(0, 0), p(1, 0), p(2, 0), dx),
+                            bicubic(p(-1, 1), p(0, 1), p(1, 1), p(2, 1), dx),
+                            bicubic(p(-1, 2), p(0, 2), p(1, 2), p(2, 2), dx), dy);
                         float * y_dst = (float *)((char *)dst->data + i0*nb0 + i1*nb1 + i2*nb2 + i3*nb3);
                         *y_dst = val;
                     }
@@ -7850,6 +7793,18 @@ void wsp_ggml_compute_forward_timestep_embedding(
 // wsp_ggml_compute_forward_argsort
+template<enum wsp_ggml_sort_order order>
+struct argsort_cmp {
+    const float * data;
+    bool operator()(int32_t a, int32_t b) const {
+        if constexpr (order == WSP_GGML_SORT_ORDER_ASC) {
+            return data[a] < data[b];
+        } else {
+            return data[a] > data[b];
+        }
+    }
+};
 static void wsp_ggml_compute_forward_argsort_f32(
     const wsp_ggml_compute_params * params,
     wsp_ggml_tensor * dst) {
@@ -7868,23 +7823,25 @@ static void wsp_ggml_compute_forward_argsort_f32(
     wsp_ggml_sort_order order = (wsp_ggml_sort_order) wsp_ggml_get_op_params_i32(dst, 0);
     for (int64_t i = ith; i < nr; i += nth) {
-        int32_t * dst_data = (int32_t *)((char *) dst->data + i*nb1);
         const float * src_data = (float *)((char *) src0->data + i*nb01);
+        int32_t * dst_data = (int32_t *)((char *) dst->data + i*nb1);
         for (int64_t j = 0; j < ne0; j++) {
             dst_data[j] = j;
         }
-        // C doesn't have a functional sort, so we do a bubble sort instead
-        for (int64_t j = 0; j < ne0; j++) {
-            for (int64_t k = j + 1; k < ne0; k++) {
-                if ((order == WSP_GGML_SORT_ORDER_ASC  && src_data[dst_data[j]] > src_data[dst_data[k]]) ||
-                    (order == WSP_GGML_SORT_ORDER_DESC && src_data[dst_data[j]] < src_data[dst_data[k]])) {
-                    int32_t tmp = dst_data[j];
-                    dst_data[j] = dst_data[k];
-                    dst_data[k] = tmp;
-                }
-            }
+        switch (order) {
+            case WSP_GGML_SORT_ORDER_ASC:
+                std::sort(dst_data, dst_data + ne0, argsort_cmp<WSP_GGML_SORT_ORDER_ASC>{src_data});
+                break;
+            case WSP_GGML_SORT_ORDER_DESC:
+                std::sort(dst_data, dst_data + ne0, argsort_cmp<WSP_GGML_SORT_ORDER_DESC>{src_data});
+                break;
+            default:
+                WSP_GGML_ABORT("invalid sort order");
         }
     }
 }
@@ -7909,10 +7866,10 @@ void wsp_ggml_compute_forward_argsort(
 // wsp_ggml_compute_forward_flash_attn_ext
-static void wsp_ggml_compute_forward_flash_attn_ext_f16(
+static void wsp_ggml_compute_forward_flash_attn_ext_f16_one_chunk(
         const wsp_ggml_compute_params * params,
-        wsp_ggml_tensor * dst) {
+        wsp_ggml_tensor * dst,
+        int ir0, int ir1) {
     const wsp_ggml_tensor * q     = dst->src[0];
     const wsp_ggml_tensor * k     = dst->src[1];
     const wsp_ggml_tensor * v     = dst->src[2];
@@ -7928,9 +7885,6 @@ static void wsp_ggml_compute_forward_flash_attn_ext_f16(
     WSP_GGML_TENSOR_LOCALS(int64_t, ne,  dst, ne)
     WSP_GGML_TENSOR_LOCALS(size_t,  nb,  dst, nb)
-    const int ith = params->ith;
-    const int nth = params->nth;
     const int64_t DK = nek0;
     const int64_t DV = nev0;
     const int64_t N  = neq1;
@@ -7964,16 +7918,6 @@ static void wsp_ggml_compute_forward_flash_attn_ext_f16(
     // parallelize by q rows using wsp_ggml_vec_dot_f32
-    // total rows in q
-    const int nr = neq1*neq2*neq3;
-    // rows per thread
-    const int dr = (nr + nth - 1)/nth;
-    // row range for this thread
-    const int ir0 = dr*ith;
-    const int ir1 = MIN(ir0 + dr, nr);
     float scale         = 1.0f;
     float max_bias      = 0.0f;
     float logit_softcap = 0.0f;
@@ -8000,6 +7944,8 @@ static void wsp_ggml_compute_forward_flash_attn_ext_f16(
     WSP_GGML_ASSERT((                            q_to_vec_dot) && "fattn: unsupported K-type");
     WSP_GGML_ASSERT((v->type == WSP_GGML_TYPE_F32 || v_to_float  ) && "fattn: unsupported V-type");
+    int ith = params->ith;
     // loop over n_batch and n_head
     for (int ir = ir0; ir < ir1; ++ir) {
         // q indices
@@ -8147,6 +8093,91 @@ static void wsp_ggml_compute_forward_flash_attn_ext_f16(
     }
 }
+static void wsp_ggml_compute_forward_flash_attn_ext_f16(
+        const wsp_ggml_compute_params * params,
+        wsp_ggml_tensor * dst) {
+    const wsp_ggml_tensor * q     = dst->src[0];
+    const wsp_ggml_tensor * k     = dst->src[1];
+    const wsp_ggml_tensor * v     = dst->src[2];
+    WSP_GGML_TENSOR_LOCALS(int64_t, neq, q,   ne)
+    WSP_GGML_TENSOR_LOCALS(size_t,  nbq, q,   nb)
+    WSP_GGML_TENSOR_LOCALS(int64_t, nek, k,   ne)
+    WSP_GGML_TENSOR_LOCALS(size_t,  nbk, k,   nb)
+    WSP_GGML_TENSOR_LOCALS(int64_t, nev, v,   ne)
+    WSP_GGML_TENSOR_LOCALS(size_t,  nbv, v,   nb)
+    WSP_GGML_TENSOR_LOCALS(int64_t, ne,  dst, ne)
+    WSP_GGML_TENSOR_LOCALS(size_t,  nb,  dst, nb)
+    const int64_t DK = nek0;
+    const int64_t DV = nev0;
+    const int64_t N  = neq1;
+    WSP_GGML_ASSERT(ne0 == DV);
+    WSP_GGML_ASSERT(ne2 == N);
+    // input tensor rows must be contiguous
+    WSP_GGML_ASSERT(nbq0 == wsp_ggml_type_size(q->type));
+    WSP_GGML_ASSERT(nbk0 == wsp_ggml_type_size(k->type));
+    WSP_GGML_ASSERT(nbv0 == wsp_ggml_type_size(v->type));
+    WSP_GGML_ASSERT(neq0 == DK);
+    WSP_GGML_ASSERT(nek0 == DK);
+    WSP_GGML_ASSERT(nev0 == DV);
+    WSP_GGML_ASSERT(neq1 == N);
+    // dst cannot be transposed or permuted
+    WSP_GGML_ASSERT(nb0 == sizeof(float));
+    WSP_GGML_ASSERT(nb0 <= nb1);
+    WSP_GGML_ASSERT(nb1 <= nb2);
+    WSP_GGML_ASSERT(nb2 <= nb3);
+    // parallelize by q rows using wsp_ggml_vec_dot_f32
+    // total rows in q
+    const int64_t nr = neq1*neq2*neq3;
+    // rows per thread
+    const int ith = params->ith;
+    const int nth = params->nth;
+    // disable for NUMA
+    const bool disable_chunking = wsp_ggml_is_numa();
+    // 4x chunks per thread
+    int nth_scaled = nth * 4;
+    int64_t chunk_size = (nr + nth_scaled - 1) / nth_scaled;
+    int64_t nchunk     = (nr + chunk_size - 1) / chunk_size;
+    if (nth == 1 || nchunk < nth || disable_chunking) {
+        nchunk = nth;
+    }
+    if (ith == 0) {
+        // Every thread starts at ith, so the first unprocessed chunk is nth.  This save a bit of coordination right at the start.
+        wsp_ggml_threadpool_chunk_set(params->threadpool, nth);
+    }
+    wsp_ggml_barrier(params->threadpool);
+    // The number of elements in each chunk
+    const int64_t dr = (nr + nchunk - 1) / nchunk;
+    // The first chunk comes from our thread_id, the rest will get auto-assigned.
+    int current_chunk = ith;
+    while (current_chunk < nchunk) {
+        const int64_t ir0 = dr * current_chunk;
+        const int64_t ir1 = MIN(ir0 + dr, nr);
+        wsp_ggml_compute_forward_flash_attn_ext_f16_one_chunk(params, dst, ir0, ir1);
+        current_chunk = wsp_ggml_threadpool_chunk_add(params->threadpool, 1);
+    }
+}
 void wsp_ggml_compute_forward_flash_attn_ext(
         const wsp_ggml_compute_params * params,
         wsp_ggml_tensor * dst) {
@@ -8633,7 +8664,7 @@ static void wsp_ggml_compute_forward_ssm_scan_f32(
                 // n_head
                 for (int h = ih0; h < ih1; ++h) {
                     // ref: https://github.com/state-spaces/mamba/blob/62db608da60f6fc790b8ed9f4b3225e95ca15fde/mamba_ssm/ops/triton/softplus.py#L16
-                    const float dt_soft_plus = wsp_ggml_softplus(dt[h]);
+                    const float dt_soft_plus = wsp_ggml_compute_softplus_f32(dt[h]);
                     const float dA = expf(dt_soft_plus * A[h]);
                     const int g = h / (nh / ng); // repeat_interleave
@@ -8730,7 +8761,7 @@ static void wsp_ggml_compute_forward_ssm_scan_f32(
                 // n_head
                 for (int h = ih0; h < ih1; ++h) {
                     // ref: https://github.com/state-spaces/mamba/blob/62db608da60f6fc790b8ed9f4b3225e95ca15fde/mamba_ssm/ops/triton/softplus.py#L16
-                    const float dt_soft_plus = wsp_ggml_softplus(dt[h]);
+                    const float dt_soft_plus = wsp_ggml_compute_softplus_f32(dt[h]);
                     const int g = h / (nh / ng); // repeat_interleave
                     // dim
@@ -9013,6 +9044,14 @@ void wsp_ggml_compute_forward_unary(
             {
                 wsp_ggml_compute_forward_xielu(params, dst);
             } break;
+        case WSP_GGML_UNARY_OP_EXPM1:
+            {
+                wsp_ggml_compute_forward_expm1(params, dst);
+            } break;
+        case WSP_GGML_UNARY_OP_SOFTPLUS:
+            {
+                wsp_ggml_compute_forward_softplus(params, dst);
+            } break;
         default:
             {
                 WSP_GGML_ABORT("fatal error");
@@ -9609,6 +9648,76 @@ void wsp_ggml_compute_forward_gla(
     }
 }
+static void wsp_ggml_compute_forward_solve_tri_f32(const struct wsp_ggml_compute_params * params, struct wsp_ggml_tensor * dst) {
+    const struct wsp_ggml_tensor * src0 = dst->src[0];  // A (lower triangular)
+    const struct wsp_ggml_tensor * src1 = dst->src[1];  // B (RHS)
+    WSP_GGML_TENSOR_BINARY_OP_LOCALS;
+    WSP_GGML_ASSERT(src0->type == WSP_GGML_TYPE_F32);
+    WSP_GGML_ASSERT(src1->type == WSP_GGML_TYPE_F32);
+    WSP_GGML_ASSERT(dst->type  == WSP_GGML_TYPE_F32);
+    WSP_GGML_ASSERT(ne00 == ne01); // A must be square
+    WSP_GGML_ASSERT(ne0  == ne10); // solution cols == B cols
+    WSP_GGML_ASSERT(ne1  == ne11); // solution rows == B rows
+    WSP_GGML_ASSERT(ne02 == ne12 && ne12 == ne2);
+    WSP_GGML_ASSERT(ne03 == ne13 && ne13 == ne3);
+    const int ith = params->ith;
+    const int nth = params->nth;
+    const int64_t k = ne10;   // number of RHS columns
+    const int64_t n = ne11;   // A is n×n
+    const int64_t nr = ne02 * ne03 * k; // we're parallelizing on columns here, so seq x token x column will be the unit
+    // chunks per thread
+    const int64_t dr = (nr + nth - 1)/nth;
+    // chunk range for this thread
+    const int64_t ir0 = dr*ith;
+    const int64_t ir1 = MIN(ir0 + dr, nr);
+    const float * A = (const float *) src0->data;  // [n, n, B1, B2]
+    const float * B = (const float *) src1->data;  // [n, k, B1, B2]
+          float * X = (      float *) dst->data;   // [n, k, B1, B2]
+    for (int64_t ir = ir0; ir < ir1; ++ir) {
+        const int64_t i03 = ir/(ne02*k);
+        const int64_t i02 = (ir - i03*ne02*k)/k;
+        const int64_t i01 = (ir - i03*ne02*k - i02*k);
+        const float * A_batch = A + i02 * nb02 / sizeof(float) + i03 * nb03 / sizeof(float);
+        const float * B_batch = B + i02 * nb12 / sizeof(float) + i03 * nb13 / sizeof(float);
+        float * X_batch = X + i02 * nb2 / sizeof(float) + i03 * nb3 / sizeof(float);
+        for (int64_t i00 = 0; i00 < n; ++i00) {
+            float sum = 0.0f;
+            for (int64_t t = 0; t < i00; ++t) {
+                sum += A_batch[i00 * n + t] * X_batch[i01 * n + t];
+            }
+            const float diag = A_batch[i00 * n + i00];
+            WSP_GGML_ASSERT(diag != 0.0f && "Zero diagonal in triangular matrix");
+            X_batch[i01 * n + i00] = (B_batch[i00 * k + i01] - sum) / diag;
+        }
+    }
+}
+void wsp_ggml_compute_forward_solve_tri(const struct wsp_ggml_compute_params * params, struct wsp_ggml_tensor * dst) {
+    const wsp_ggml_tensor * src0 = dst->src[0];
+    const wsp_ggml_tensor * src1 = dst->src[1];
+    if (src0->type == WSP_GGML_TYPE_F32 && src1->type == WSP_GGML_TYPE_F32) {
+        wsp_ggml_compute_forward_solve_tri_f32(params, dst);
+    } else {
+        WSP_GGML_ABORT("fatal error");
+    }
+}
 // wsp_ggml_compute_forward_rwkv_wkv7
 static void wsp_ggml_compute_forward_rwkv_wkv7_f32(