vk_pipeline pipeline_pool2d_f32;
vk_pipeline pipeline_rwkv_wkv6_f32;
vk_pipeline pipeline_rwkv_wkv7_f32;
+ vk_pipeline pipeline_gated_linear_attn_f32;
// [size_idx][kda] where size_idx: 0=d16, 1=d32, 2=d64, 3=d128
vk_pipeline pipeline_gated_delta_net[4][2];
vk_pipeline pipeline_ssm_scan_f32_d128;
uint32_t C;
uint32_t H;
};
+struct vk_op_gated_linear_attn_push_constants {
+ uint32_t B;
+ uint32_t T;
+ uint32_t C;
+ uint32_t H;
+ float scale;
+};
struct vk_op_gated_delta_net_push_constants {
uint32_t H;
uint32_t n_tokens;
ggml_vk_create_pipeline(device, device->pipeline_rwkv_wkv7_f32, "rwkv_wkv7_f32", rwkv_wkv7_f32_len, rwkv_wkv7_f32_data, "main", 8, sizeof(vk_op_rwkv_wkv7_push_constants), {1, 1, 1}, {device->subgroup_size}, 1);
+ ggml_vk_create_pipeline(device, device->pipeline_gated_linear_attn_f32, "gated_linear_attn_f32", gated_linear_attn_f32_len, gated_linear_attn_f32_data, "main", 6, sizeof(vk_op_gated_linear_attn_push_constants), {1, 1, 1}, {}, 1);
+
{
const uint32_t gdn_sizes[] = {16, 32, 64, 128};
const char * gdn_names[][2] = {
return ctx->device->pipeline_rwkv_wkv7_f32;
}
return nullptr;
+ case GGML_OP_GATED_LINEAR_ATTN:
+ if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) {
+ return ctx->device->pipeline_gated_linear_attn_f32;
+ }
+ return nullptr;
case GGML_OP_GATED_DELTA_NET:
if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) {
const uint32_t S_v = dst->src[2]->ne[0];
);
}
+static void ggml_vk_gated_linear_attn(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) {
+ const size_t seq_length = dst->src[0]->ne[2];
+ const size_t n_embed = dst->ne[0];
+ const size_t n_heads = dst->src[0]->ne[1];
+ const size_t n_seqs = dst->src[4]->ne[1];
+
+ float scale;
+ memcpy(&scale, dst->op_params, sizeof(float));
+
+ GGML_ASSERT(dst->buffer != nullptr);
+
+ vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op);
+ GGML_ASSERT(pipeline != nullptr);
+
+ ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1);
+
+ vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst);
+ vk_subbuffer src_buf[5] = {};
+ for (int i = 0; i < 5; i++) {
+ src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]);
+ }
+
+ const vk_op_gated_linear_attn_push_constants pc = {
+ (uint32_t)n_seqs,
+ (uint32_t)seq_length,
+ (uint32_t)n_embed,
+ (uint32_t)n_heads,
+ scale,
+ };
+
+ ggml_vk_dispatch_pipeline(ctx, subctx, pipeline,
+ {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], dst_buf},
+ pc, { (uint32_t)(n_seqs * n_heads), 1, 1 });
+}
+
static void ggml_vk_gated_delta_net(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) {
const ggml_tensor * src_q = dst->src[0];
const ggml_tensor * src_v = dst->src[2];
break;
+ case GGML_OP_GATED_LINEAR_ATTN:
+ ggml_vk_gated_linear_attn(ctx, compute_ctx, node);
+
+ break;
+
case GGML_OP_GATED_DELTA_NET:
ggml_vk_gated_delta_net(ctx, compute_ctx, node);
case GGML_OP_RWKV_WKV6:
case GGML_OP_RWKV_WKV7:
return true; // all inputs are contiguous, see ggml.c
+ case GGML_OP_GATED_LINEAR_ATTN:
+ // the shader block size is hardcoded to head_size 64
+ return op->src[0]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32 && op->src[0]->ne[0] == 64;
case GGML_OP_GATED_DELTA_NET:
{
const uint32_t S_v = op->src[2]->ne[0];
} else if (tensor->op == GGML_OP_RWKV_WKV7) {
tensor_clone = ggml_rwkv_wkv7(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3],
src_clone[4], src_clone[5], src_clone[6]);
+ } else if (tensor->op == GGML_OP_GATED_LINEAR_ATTN) {
+ const float * op_params = (const float *)tensor->op_params;
+ tensor_clone = ggml_gated_linear_attn(ggml_ctx, src_clone[0], src_clone[1],
+ src_clone[2], src_clone[3], src_clone[4], op_params[0]);
} else if (tensor->op == GGML_OP_GATED_DELTA_NET) {
tensor_clone = ggml_gated_delta_net(ggml_ctx, src_clone[0], src_clone[1],
src_clone[2], src_clone[3], src_clone[4], src_clone[5],
--- /dev/null
+#version 450
+
+#extension GL_EXT_control_flow_attributes : require
+
+#define BLOCK_SIZE 64
+layout(local_size_x = BLOCK_SIZE, local_size_y = 1, local_size_z = 1) in;
+
+layout(push_constant) uniform Parameters {
+ uint B;
+ uint T;
+ uint C;
+ uint H;
+ float scale;
+};
+
+layout(binding = 0) readonly buffer KBuf { A_TYPE k[]; };
+layout(binding = 1) readonly buffer VBuf { A_TYPE v[]; };
+layout(binding = 2) readonly buffer QBuf { A_TYPE q[]; };
+layout(binding = 3) readonly buffer GBuf { A_TYPE g[]; };
+layout(binding = 4) readonly buffer StateBuf { A_TYPE state_in[]; };
+layout(binding = 5) buffer DstBuf { A_TYPE dst[]; };
+
+shared A_TYPE _k[BLOCK_SIZE], _q[BLOCK_SIZE], _g[BLOCK_SIZE];
+
+void main() {
+ const uint head_size = BLOCK_SIZE;
+ const uint batch_id = gl_WorkGroupID.x / H;
+ const uint head_id = gl_WorkGroupID.x % H;
+ const uint tid = gl_LocalInvocationID.x;
+
+ const uint state_size = C * head_size;
+ const uint n_seq_tokens = T / B;
+
+ if (batch_id >= B || head_id >= H) {
+ return;
+ }
+
+ // state[i] holds column tid of this head's state matrix: S[i][tid]
+ A_TYPE state[BLOCK_SIZE];
+ [[unroll]] for (uint i = 0; i < head_size; i++) {
+ state[i] = state_in[batch_id * state_size + head_id * head_size * head_size
+ + i * head_size + tid];
+ }
+
+ const uint start_t = batch_id * n_seq_tokens * C + head_id * head_size + tid;
+ const uint end_t = (batch_id + 1) * n_seq_tokens * C + head_id * head_size + tid;
+
+ for (uint t = start_t; t < end_t; t += C) {
+ barrier();
+ _k[tid] = k[t];
+ _q[tid] = q[t];
+ _g[tid] = g[t];
+ barrier();
+
+ const A_TYPE v_val = v[t];
+ A_TYPE y = 0.0;
+
+ [[unroll]] for (uint i = 0; i < head_size; i += 4) {
+ vec4 k_vec = vec4(_k[i], _k[i+1], _k[i+2], _k[i+3]);
+ vec4 q_vec = vec4(_q[i], _q[i+1], _q[i+2], _q[i+3]);
+ vec4 g_vec = vec4(_g[i], _g[i+1], _g[i+2], _g[i+3]);
+ vec4 s_vec = vec4(state[i], state[i+1], state[i+2], state[i+3]);
+
+ vec4 kv = k_vec * v_val;
+
+ s_vec = s_vec * g_vec + kv;
+ y += dot(q_vec, s_vec);
+
+ state[i] = s_vec.x;
+ state[i+1] = s_vec.y;
+ state[i+2] = s_vec.z;
+ state[i+3] = s_vec.w;
+ }
+
+ dst[t] = y * scale;
+ }
+
+ [[unroll]] for (uint i = 0; i < head_size; i++) {
+ dst[T * C + batch_id * state_size + head_id * head_size * head_size
+ + i * head_size + tid] = state[i];
+ }
+}