int backend_id;
int i_start;
int i_end;
- struct ggml_tensor * inputs[GGML_SCHED_MAX_SPLIT_INPUTS];
+ struct ggml_tensor ** inputs;
int n_inputs;
+ int inputs_capacity;
// graph view of this split
struct ggml_cgraph graph;
};
int cur_copy;
int next_copy;
ggml_backend_event_t events[GGML_SCHED_MAX_BACKENDS][GGML_SCHED_MAX_COPIES];
- struct ggml_tensor * graph_inputs[GGML_SCHED_MAX_SPLIT_INPUTS];
+ struct ggml_tensor ** graph_inputs;
int n_graph_inputs;
+ int graph_inputs_capacity;
struct ggml_context * ctx;
#define tensor_id_copy(id, backend_id, copy_id) sched->hv_tensor_copies[(id) * sched->n_backends * sched->n_copies + (backend_id) * sched->n_copies + (copy_id)]
#define tensor_copy(tensor, backend_id, copy_id) tensor_id_copy(hash_id(tensor), backend_id, copy_id)
+static void ggml_backend_sched_split_inputs_grow(struct ggml_backend_sched_split * split) {
+ int new_cap = GGML_SCHED_MAX_SPLIT_INPUTS;
+ if (split->inputs_capacity > 0) {
+ new_cap = 2*split->inputs_capacity;
+ GGML_LOG_WARN("%s: increasing split inputs capacity from %d to %d\n", __func__, split->inputs_capacity, new_cap);
+ }
+ auto * pnew = (struct ggml_tensor **) realloc((void *) split->inputs, new_cap * sizeof(struct ggml_tensor *));
+ if (pnew == NULL) {
+ GGML_LOG_ERROR("%s: failed to allocate %zu bytes\n", __func__, new_cap * sizeof(struct ggml_tensor *));
+ GGML_ABORT("failed to grow split inputs container");
+ }
+ split->inputs = pnew;
+ split->inputs_capacity = new_cap;
+}
+
+static void ggml_backend_sched_graph_inputs_grow(ggml_backend_sched_t sched) {
+ int new_cap = GGML_SCHED_MAX_SPLIT_INPUTS;
+ if (sched->graph_inputs_capacity > 0) {
+ new_cap = 2*sched->graph_inputs_capacity;
+ GGML_LOG_WARN("%s: increasing graph inputs capacity from %d to %d\n", __func__, sched->graph_inputs_capacity, new_cap);
+ }
+ auto * pnew = (struct ggml_tensor **) realloc((void *) sched->graph_inputs, new_cap * sizeof(struct ggml_tensor *));
+ if (pnew == NULL) {
+ GGML_LOG_ERROR("%s: failed to allocate %zu bytes\n", __func__, new_cap * sizeof(struct ggml_tensor *));
+ GGML_ABORT("failed to grow graph inputs container");
+ }
+ sched->graph_inputs = pnew;
+ sched->graph_inputs_capacity = new_cap;
+}
+
// returns the priority of the backend, lower id is higher priority
static int ggml_backend_sched_backend_id(ggml_backend_sched_t sched, ggml_backend_t backend) {
for (int i = 0; i < sched->n_backends; i++) {
}
// check if the split has too many inputs
// FIXME: count the number of inputs instead of only checking when full
- if (split->n_inputs == GGML_SCHED_MAX_SPLIT_INPUTS) {
+ if (split->n_inputs >= split->inputs_capacity) {
const size_t id = hash_id(src);
int src_backend_id = sched->hv_tensor_backend_ids[id];
bool supported = ggml_backend_sched_buffer_supported(sched, src, cur_backend_id);
split->i_end = i;
i_split++;
if (i_split >= sched->splits_capacity) {
+ int old_cap = sched->splits_capacity;
sched->splits_capacity *= 2;
sched->splits = (ggml_backend_sched_split *)
realloc(sched->splits, sched->splits_capacity * sizeof(struct ggml_backend_sched_split));
GGML_ASSERT(sched->splits != NULL);
+ for (int k = old_cap; k < sched->splits_capacity; k++) {
+ memset(&sched->splits[k], 0, sizeof(struct ggml_backend_sched_split));
+ }
}
split = &sched->splits[i_split];
split->backend_id = node_backend_id;
SET_CAUSE(tensor_copy, "4.cpy");
}
int n_graph_inputs = sched->n_graph_inputs++;
- GGML_ASSERT(n_graph_inputs < GGML_SCHED_MAX_SPLIT_INPUTS);
+ if (n_graph_inputs >= sched->graph_inputs_capacity) {
+ ggml_backend_sched_graph_inputs_grow(sched);
+ }
sched->graph_inputs[n_graph_inputs] = src;
}
}
SET_CAUSE(tensor_copy, "4.cpy");
}
int n_inputs = split->n_inputs++;
- GGML_ASSERT(n_inputs < GGML_SCHED_MAX_SPLIT_INPUTS);
+ if (n_inputs >= split->inputs_capacity) {
+ ggml_backend_sched_split_inputs_grow(split);
+ }
split->inputs[n_inputs] = src;
}
node->src[j] = tensor_id_copy(src_id, cur_backend_id, sched->cur_copy);
sched->prev_leaf_backend_ids = tmp;
}
- int graph_size = std::max(graph->n_nodes, graph->n_leafs) + sched->n_splits*GGML_SCHED_MAX_SPLIT_INPUTS*2*sched->n_copies;
+ int total_inputs = sched->n_graph_inputs;
+ for (int i = 0; i < sched->n_splits; i++) {
+ total_inputs += sched->splits[i].n_inputs;
+ }
+ int graph_size = std::max(graph->n_nodes, graph->n_leafs) + total_inputs * 2 * sched->n_copies;
// remember the actual graph_size for performing reallocation checks later [GGML_SCHED_DEBUG_REALLOC]
sched->debug_prev_graph_size = sched->debug_graph_size;
sched->splits = (ggml_backend_sched_split *) calloc(initial_splits_capacity, sizeof(sched->splits[0]));
sched->splits_capacity = initial_splits_capacity;
+ sched->graph_inputs_capacity = GGML_SCHED_MAX_SPLIT_INPUTS;
+ sched->graph_inputs = (struct ggml_tensor **) calloc(sched->graph_inputs_capacity, sizeof(struct ggml_tensor *));
+
for (int b = 0; b < n_backends; b++) {
sched->backends[b] = backends[b];
sched->bufts[b] = bufts ? bufts[b] : ggml_backend_get_default_buffer_type(backends[b]);
ggml_gallocr_free(sched->galloc);
ggml_free(sched->ctx);
ggml_hash_set_free(&sched->hash_set);
+ for (int i = 0; i < sched->splits_capacity; i++) {
+ free(sched->splits[i].inputs);
+ }
free(sched->splits);
+ free(sched->graph_inputs);
free(sched->hv_tensor_backend_ids);
free(sched->hv_tensor_copies);
free(sched->node_backend_ids);