Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion ci/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -312,6 +312,8 @@ function gg_run_test_llama_archs_tensor_split {
GGML_CUDA_DEVICES=2 ./build-ci-release/bin/test-llama-archs -s 1 2>&1
GGML_CUDA_DEVICES=3 ./build-ci-release/bin/test-llama-archs -s 1 2>&1
GGML_CUDA_DEVICES=4 ./build-ci-release/bin/test-llama-archs -s 1 2>&1
# the scheduler names a copy after its source, and 8 device names fill the name field
GGML_CUDA_DEVICES=8 ./build-ci-release/bin/test-llama-archs -s 1 -a llama 2>&1
fi

if [ ! -z ${GG_BUILD_METAL} ]; then
Expand All @@ -327,7 +329,7 @@ function gg_run_test_llama_archs_tensor_split {
function gg_sum_test_llama_archs_tensor_split {
gg_printf '### %s\n\n' "${ci}"

gg_printf 'Runs test-llama-archs with 1 to 4 devices\n'
gg_printf 'Runs test-llama-archs with 1 to 4 and 8 devices\n'
gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"
gg_printf '```\n'
gg_printf '%s\n' "$(cat $OUT/${ci}.log)"
Expand Down
2 changes: 2 additions & 0 deletions docs/multi-gpu.md
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,8 @@ llama-cli -m model.gguf -sm tensor -ctk f16 -ctv f16
- `--flash-attn off` or (`--flash-attn auto` resolving to `off` when it isn't supported) is a hard error.
- KV cache types must be non-quantized: `f32`, `f16`, or `bf16`. Support for quantized KV cache is not implemented and trying to use it will result in an error.
- Mark this configuration as experimental in your tooling: validate output quality before deploying.
- `--no-kv-offload` works in this mode: the host-resident cache is split by attention head like the rest. Two limits: a backend without a native 2d copy (CPU, Metal) pays one transfer per cache cell, and Gemma 4 is less accurate this way (perplexity 235.03 with a host cache versus 227.23 with a device cache), so keep the cache on the devices for that architecture.
- A recurrent or hybrid model always keeps its recurrent state on the devices in this mode, even with `--no-recurrent-state-offload`. The linear-attention op writes the state back together with its output, so the two do not agree on a host-resident state.
- `--split-mode tensor`is not implemented for all architectures. The following will fail with *"LLAMA_SPLIT_MODE_TENSOR not implemented for architecture '...'"*:

- **MoE / hybrid:** Grok, MPT, OLMoE, DeepSeek2, GLM-DSA, Nemotron-H, Nemotron-H-MoE, Granite-Hybrid, LFM2-MoE, Minimax-M2, Mistral4, Kimi-Linear, Jamba, Falcon-H1
Expand Down
15 changes: 15 additions & 0 deletions ggml/src/ggml-backend-impl.h
Original file line number Diff line number Diff line change
Expand Up @@ -104,6 +104,21 @@ extern "C" {
// temporary workaround to statically allocate tensors from a context in a deduplicated way:
GGML_API struct ggml_backend_buffer * ggml_backend_meta_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft);

//
// Backend (sched)
//

// The scheduler names a copy of a graph input "<backend>#<source>#<copy>". <source> is the name of the
// tensor the copy was made from and carries any suffix that ggml appends for a view. Only <source>
// identifies the copy, so the backend label is cut when the name does not fit. A source name that
// does not fit on its own asserts, a cut one would name a different tensor.
GGML_API void ggml_backend_sched_name_copy(
struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c);

// write the <source> part of a name written by ggml_backend_sched_name_copy into buf,
// returns false and leaves buf alone if the name is not one
GGML_API bool ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size);

//
// Backend (stream)
//
Expand Down
126 changes: 117 additions & 9 deletions ggml/src/ggml-backend-meta.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -831,10 +831,29 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(
if (ggml_nelements(tensor) == 0) {
return {GGML_BACKEND_SPLIT_AXIS_UNKNOWN, {0}, {1}, 1};
}
if (ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE && tensor->view_src == nullptr) {
// A host-resident KV cache reaches the graph as a copied-in leaf of the compute buffer.
// Mirroring it while the queries stay split by head makes each device attend the wrong
// heads, so ask the callback; it still answers MIRRORED for names it does not know.
// The name tells a copy apart, the scheduler also flags one as an input when it keeps several.
char source_name[GGML_MAX_NAME];
const bool copied_in_leaf =
ggml_backend_buffer_get_usage(tensor->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE &&
tensor->op == GGML_OP_NONE &&
ggml_backend_sched_copy_source_name(tensor->name, source_name, sizeof(source_name));

if ((ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE || copied_in_leaf) &&
tensor->view_src == nullptr) {
ggml_backend_dev_t dev = ggml_backend_buft_get_device(ggml_backend_buffer_get_type(tensor->buffer));
const ggml_backend_meta_device_context * dev_ctx = (const ggml_backend_meta_device_context *) dev->context;
ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor, dev_ctx->get_split_state_ud);
// the callback classifies by name, so offer a copy under the name it was copied from
const ggml_tensor * tensor_query = tensor;
ggml_tensor tensor_named;
if (copied_in_leaf) {
tensor_named = *tensor;
ggml_set_name(&tensor_named, source_name);
tensor_query = &tensor_named;
}
ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor_query, dev_ctx->get_split_state_ud);
if (ret.axis >= 0 && ret.axis < GGML_MAX_DIMS) {
const int64_t granularity = ret.axis == GGML_BACKEND_SPLIT_AXIS_0 ? ggml_blck_size(tensor->type) : 1;
int64_t ne_sum = 0;
Expand Down Expand Up @@ -1316,6 +1335,32 @@ static void ggml_backend_meta_buffer_memset_tensor(
const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer);
const ggml_backend_meta_split_state split_state =
ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false);

// a host-resident attention cache is permuted, its heads are one run per cell, see set_tensor
const bool strided_head_split =
!ggml_is_contiguous(tensor) &&
split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 &&
split_state.n_segments == 1 && split_state.nr[0] == 1 &&
tensor->nb[1] > tensor->nb[2] &&
offset == 0 && size == ggml_nbytes(tensor);

if (strided_head_split) {
for (int64_t i3 = 0; i3 < tensor->ne[3]; i3++) {
for (size_t j = 0; j < n_bufs; j++) {
ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j);
const size_t nbytes = split_state.ne[j] * tensor->nb[2];
if (nbytes == 0) {
continue;
}
for (int64_t i1 = 0; i1 < tensor->ne[1]; i1++) {
ggml_backend_tensor_memset(simple_tensor, value,
i3*simple_tensor->nb[3] + i1*simple_tensor->nb[1], nbytes);
}
}
}
return;
}

GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED);

if (split_state.n_segments != 1 || split_state.nr[0] != 1) {
Expand Down Expand Up @@ -1416,6 +1461,37 @@ static void ggml_backend_meta_buffer_memset_tensor(
static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) {
const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer);
const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false);

// A host-resident attention cache reaches this permuted, as [head_dim, n_kv, n_head_kv, n_stream].
// The heads split, but interleaved per cell, so the chunk splice below cannot express the write.
// Each device's heads are one contiguous run per cell: ne[1] cells, from one stride to another.
// A backend without a native 2d copy (CPU, Metal) pays one transfer per cell here, CUDA does not.
const bool strided_head_split =
!ggml_is_contiguous(tensor) &&
split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 &&
split_state.n_segments == 1 && split_state.nr[0] == 1 &&
tensor->nb[1] > tensor->nb[2] &&
offset == 0 && size == ggml_nbytes(tensor);

if (strided_head_split) {
for (int64_t i3 = 0; i3 < tensor->ne[3]; i3++) {
size_t offset_data = i3 * tensor->nb[3];
for (size_t j = 0; j < n_bufs; j++) {
ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j);
const size_t nbytes = split_state.ne[j] * tensor->nb[2];
if (nbytes == 0) {
continue;
}
ggml_backend_tensor_set_2d(simple_tensor, (const char *) data + offset_data,
i3 * simple_tensor->nb[3], nbytes,
tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]);
offset_data += nbytes;
}
GGML_ASSERT(offset_data == i3 * tensor->nb[3] + (size_t) tensor->ne[2] * tensor->nb[2]);
}
return;
}

GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED);

if (split_state.n_segments != 1 || split_state.nr[0] != 1) {
Expand Down Expand Up @@ -1544,6 +1620,34 @@ static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, gg
static void ggml_backend_meta_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) {
const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer);
const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false);

// A fused QKV puts Kcur and Vcur in a strided view, which a host-resident cache reads back here.
// The rows split, so the chunk splice below cannot express the read. Each device's part of a row
// is contiguous: ne[1] rows, from the device's own stride to the view's.
// A backend without a native 2d copy (CPU, Metal) pays one transfer per row here, CUDA does not.
const bool strided_rows =
!ggml_is_contiguous(tensor) &&
split_state.axis == GGML_BACKEND_SPLIT_AXIS_0 &&
split_state.n_segments == 1 && split_state.nr[0] == 1 &&
tensor->ne[2] == 1 && tensor->ne[3] == 1 &&
offset == 0 && size == ggml_nbytes(tensor);

if (strided_rows) {
size_t offset_data = 0;
for (size_t j = 0; j < n_bufs; j++) {
const ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j);
const size_t nbytes = ggml_row_size(tensor->type, split_state.ne[j]);
if (nbytes == 0) {
continue;
}
ggml_backend_tensor_get_2d(simple_tensor, (char *) data + offset_data, 0, nbytes,
tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]);
offset_data += nbytes;
}
GGML_ASSERT(offset_data == ggml_row_size(tensor->type, tensor->ne[0]));
return;
}

GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED);

if (split_state.n_segments != 1 || split_state.nr[0] != 1) {
Expand Down Expand Up @@ -2182,14 +2286,18 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend,
int i_start = 0;
for (int i = 0; i < cgraph->n_nodes; i++) {
ggml_tensor * node = cgraph->nodes[i];
if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) {
continue;
}
const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(node, /*assume_sync =*/ false);
if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) {
max_tmp_size = std::max(max_tmp_size, ggml_nbytes(node));
// a host-resident KV cache ends a split with a view of itself, that view needs no split state
// but it is still the last node and must close the last subgraph
const bool host_view = node->view_src != nullptr && node->view_src->op == GGML_OP_NONE &&
ggml_backend_buffer_is_host(node->view_src->buffer);
bool new_subgraph = i + 1 == cgraph->n_nodes;
if (!host_view) {
const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(node, /*assume_sync =*/ false);
if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) {
max_tmp_size = std::max(max_tmp_size, ggml_nbytes(node));
new_subgraph = true;
}
}
const bool new_subgraph = i + 1 == cgraph->n_nodes || split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL;
if (!new_subgraph) {
continue;
}
Expand Down
58 changes: 55 additions & 3 deletions ggml/src/ggml-backend.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1034,6 +1034,57 @@ static void ggml_backend_sched_print_assignments(ggml_backend_sched_t sched, str
}
}

// The name field has a fixed size, so cut the backend label rather than the source name.
// See the contract at the declaration in ggml-backend-impl.h.
void ggml_backend_sched_name_copy(
struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c) {
const int n_tail = snprintf(NULL, 0, "#%s#%d", src->name, c);
const int n_max = GGML_MAX_NAME - 1 - n_tail;
GGML_ASSERT(n_max >= 0 && "source name too long to name a scheduler copy");
int n_head = (int) strlen(backend_name);
if (n_head > n_max) {
n_head = n_max;
}
ggml_format_name(copy, "%.*s#%s#%d", n_head, backend_name, src->name, c);
}

bool ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size) {
GGML_ASSERT(buf_size > 0);

const char * first = strchr(name, '#');
if (first == NULL) {
return false;
}
const char * src = first + 1;
size_t len = strlen(src);

// ggml writes a view suffix as " (...)", a graph name has no spaces
const char * suffix = strstr(src, " (");
if (suffix != NULL) {
len = suffix - src;
} else {
const char * copy = strrchr(src, '#');
if (copy == NULL) {
return false;
}
const char * digits = copy + 1;
while (*digits >= '0' && *digits <= '9') {
digits++;
}
if (digits == copy + 1 || *digits != '\0') {
return false;
}
len = copy - src;
}

if (len > buf_size - 1) {
return false;
}
memcpy(buf, src, len);
buf[len] = '\0';
return true;
}

static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) {
ggml_backend_buffer_t buf = t->view_src ? t->view_src->buffer : t->buffer;
ggml_backend_buffer_type_t buft = NULL;
Expand Down Expand Up @@ -1381,7 +1432,7 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
tensor_copy = src; // use the original tensor as the current copy
} else {
tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);
ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);
ggml_backend_sched_name_copy(tensor_copy, ggml_backend_name(backend), src, c);
}
ggml_set_input(tensor_copy);
ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
Expand All @@ -1402,7 +1453,7 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
ggml_backend_t backend = sched->backends[cur_backend_id];
for (int c = 0; c < sched->n_copies; c++) {
struct ggml_tensor * tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);
ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);
ggml_backend_sched_name_copy(tensor_copy, ggml_backend_name(backend), src, c);
if (sched->n_copies > 1) {
ggml_set_input(tensor_copy);
ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
Expand Down Expand Up @@ -1812,7 +1863,8 @@ static enum ggml_status ggml_backend_sched_compute_splits(ggml_backend_sched_t s
ggml_backend_buffer_t src_buf = input->view_src ? input->view_src->buffer : input->buffer;
struct ggml_backend_sched_ranges rg;
ggml_backend_sched_input_ranges(input, &rg);
const bool ranged = rg.n > 1 && src_buf != NULL && ggml_backend_buffer_is_host(src_buf);
// a meta backend writes a whole contiguous tensor, it cannot take one range per stream
const bool ranged = rg.n > 1 && src_buf != NULL && ggml_backend_buffer_is_host(src_buf) && !ggml_backend_is_meta(split_backend);

// try async copy, but if not possible, we can still use a sync copy without synchronizing the dst backend, since we handle the synchronization here with multiple copies and events
// TODO: add public function to facilitate this, since applications do not have direct access to the backend interface
Expand Down
9 changes: 9 additions & 0 deletions src/llama-context.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -144,6 +144,15 @@ llama_context::llama_context(
cparams.offload_kqv = params.offload_kqv;
cparams.kv_cpu_pinned = params.kv_cpu_pinned;
cparams.recurrent_state_offload = params.recurrent_state_offload;

// A linear-attention op packs the state it writes back together with its output, so that split
// does not line up with the one a host-resident state expects. Keep the state on its device.
if (!cparams.recurrent_state_offload && model.split_mode() == LLAMA_SPLIT_MODE_TENSOR &&
(llm_arch_is_recurrent(model.arch) || llm_arch_is_hybrid(model.arch))) {
LLAMA_LOG_WARN("%s: split mode tensor cannot keep the recurrent state in host memory - "
"overriding --no-recurrent-state-offload, which needs more VRAM\n", __func__);
cparams.recurrent_state_offload = true;
}
cparams.offload_attn_compute = params.offload_kqv || (params.op_offload && params.kv_cpu_pinned);
cparams.kv_gpu_layers = params.kv_gpu_layers;
cparams.phase_aware_workspace = params.phase_aware_workspace;
Expand Down
Loading
Loading