}
// operations with weights are preferably run on the same backend as the weights
- for (int i = 0; i < GGML_MAX_SRC; i++) {
- const struct ggml_tensor * src = tensor->src[i];
- if (src == NULL) {
- continue;
- }
- // skip ROPE since the rope freqs tensor is too small to choose a backend based on it
- // not an ideal solution
- if (tensor->op != GGML_OP_ROPE && src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {
- int src_backend_id = ggml_backend_sched_backend_from_buffer(sched, src, tensor);
- // check if a backend with higher prio wants to offload the op
- if (sched->op_offload && src_backend_id == sched->n_backends - 1 && ggml_backend_buffer_is_host(src->buffer)) {
- for (int b = 0; b < src_backend_id; b++) {
- if (ggml_backend_supports_op(sched->backends[b], tensor) && ggml_backend_offload_op(sched->backends[b], tensor)) {
- SET_CAUSE(tensor, "1.off");
- return b;
+ // TODO: there are exceptions (see below) - not an ideal solution
+ bool allow = true;
+
+ // skip ROPE since the rope freqs tensor is too small to choose a backend based on it
+ allow = allow && tensor->op != GGML_OP_ROPE;
+
+ // skip FLASH_ATTN_EXT since the sinks tensor is too small to choose a based based on it
+ allow = allow && tensor->op != GGML_OP_FLASH_ATTN_EXT;
+
+ if (allow) {
+ for (int i = 0; i < GGML_MAX_SRC; i++) {
+ const struct ggml_tensor * src = tensor->src[i];
+ if (src == NULL) {
+ continue;
+ }
+ if (src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {
+ int src_backend_id = ggml_backend_sched_backend_from_buffer(sched, src, tensor);
+ // check if a backend with higher prio wants to offload the op
+ if (sched->op_offload && src_backend_id == sched->n_backends - 1 && ggml_backend_buffer_is_host(src->buffer)) {
+ for (int b = 0; b < src_backend_id; b++) {
+ if (ggml_backend_supports_op(sched->backends[b], tensor) && ggml_backend_offload_op(sched->backends[b], tensor)) {
+ SET_CAUSE(tensor, "1.off");
+ return b;
+ }
}
}
+ SET_CAUSE(tensor, "1.wgt%d", i);
+ return src_backend_id;
}
- SET_CAUSE(tensor, "1.wgt%d", i);
- return src_backend_id;
}
}
ggml_set_name(cur, name);
}
- // norm may be automatically assigned to the backend of the previous layer, increasing data transfer between backends
+ // - norm may be automatically assigned to the backend of the previous layer, increasing data transfer between backends
+ // - force the last op of the layer on the specified backend to avoid running it on the backend of the next layer due to scheduling
// FIXME: fix in ggml_backend_sched
const bool full_offload = model.n_gpu_layers() > model.hparams.n_layer_all;
if (ubatch.n_tokens < 32 || full_offload) {
- if (il != -1 && strcmp(name, "norm") == 0) {
+ if (il != -1 && (strcmp(name, "norm") == 0 || strcmp(name, "l_last") == 0)) {
const auto & dev_layer = model.dev_layer(il);
for (const auto & backend : backends) {
if (ggml_backend_get_device(backend.get()) == dev_layer) {
&post, &comb, il);
cb(cur, "hc_ffn_pre", il);
+ ggml_build_forward_expand(gf, residual);
+ ggml_build_forward_expand(gf, post);
+ ggml_build_forward_expand(gf, comb);
+
cur = build_norm(cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);
cb(cur, "ffn_norm", il);
inpL = build_hc_post(cur, residual, post, comb, il);
inpL = build_cvec(inpL, il);
- cb(inpL, "l_out", il);
+ cb(inpL, "l_last", il);
}
if (inp_out_ids) {