From 205cb8d0bc51d3904b4b8daedc2d083ea53744b9 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 3 Sep 2026 11:21:53 +0200 Subject: [PATCH 01/11] ggml : keep the source name of a scheduler copy A copy of an input is named "##" in a name field of fixed size. With many devices the backend label of the meta backend lists all of them and the source name is what gets cut, so a consumer can no longer tell which tensor the copy was made from. Cut the label instead. Assisted-by: Claude Opus 5 --- ggml/src/ggml-backend.cpp | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp index 8e7c9b08b329..fd9788ef025f 100644 --- a/ggml/src/ggml-backend.cpp +++ b/ggml/src/ggml-backend.cpp @@ -1034,6 +1034,19 @@ static void ggml_backend_sched_print_assignments(ggml_backend_sched_t sched, str } } +// Name a copy of an input "##". The name field has a fixed size, so cut the +// backend label rather than the source name - the source name is what identifies the copy. +static void ggml_backend_sched_name_copy( + struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c) { + const int n_tail = snprintf(NULL, 0, "#%s#%d", src->name, c); + const int n_max = GGML_MAX_NAME - 1 - n_tail; + int n_head = (int) strlen(backend_name); + if (n_head > n_max) { + n_head = n_max > 0 ? n_max : 0; + } + ggml_format_name(copy, "%.*s#%s#%d", n_head, backend_name, src->name, c); +} + static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) { ggml_backend_buffer_t buf = t->view_src ? t->view_src->buffer : t->buffer; ggml_backend_buffer_type_t buft = NULL; @@ -1381,7 +1394,7 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra tensor_copy = src; // use the original tensor as the current copy } else { tensor_copy = ggml_dup_tensor_layout(sched->ctx, src); - ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c); + ggml_backend_sched_name_copy(tensor_copy, ggml_backend_name(backend), src, c); } ggml_set_input(tensor_copy); ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor @@ -1402,7 +1415,7 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra ggml_backend_t backend = sched->backends[cur_backend_id]; for (int c = 0; c < sched->n_copies; c++) { struct ggml_tensor * tensor_copy = ggml_dup_tensor_layout(sched->ctx, src); - ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c); + ggml_backend_sched_name_copy(tensor_copy, ggml_backend_name(backend), src, c); if (sched->n_copies > 1) { ggml_set_input(tensor_copy); ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor From e1c5abef8b3586af189f2b5560f75657406c26b8 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 3 Sep 2026 11:21:53 +0200 Subject: [PATCH 02/11] ggml-meta : split a host-resident KV cache by head A KV cache in host memory reaches attention as a scheduler copy, which is a leaf in the compute buffer. Such a leaf never reached the device split-state callback and fell through to MIRRORED, while the queries stayed split by head: each device then attended heads whose keys live on the other device. With more than one KV head that aborts in the FlashAttention kernel, or returns wrong output where the query split happens to stay a multiple of the KV head count. Offer a copied-in leaf to the callback under the name the graph gave it, and let the callback recognise the cache there. The cache folds its heads into one flat axis but the copy arrives permuted, with the heads on an axis of their own, so its segments and granularity are rescaled to whole heads. Assisted-by: Claude Opus 5 --- ggml/src/ggml-backend-meta.cpp | 69 +++++++++++++++++++++++++++++++++- src/llama-model.cpp | 63 +++++++++++++++++++++++++------ 2 files changed, 118 insertions(+), 14 deletions(-) diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index 3ec40fb1af7f..ec0bde1c9275 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -485,6 +485,28 @@ static struct ggml_tensor * ggml_backend_meta_buffer_simple_tensor(const struct return it->second[index]; } +// A scheduler copy is named "##", where also carries the suffixes +// ggml appends for views. Recover the graph name the copy was made from. +static std::string ggml_backend_meta_copy_source_name(const char * name) { + std::string ret = name; + const size_t first = ret.find('#'); + if (first == std::string::npos) { + return ret; + } + ret.erase(0, first + 1); + // ggml writes a view suffix as " (...)", a graph name has no spaces + const size_t suffix = ret.find(" ("); + if (suffix != std::string::npos) { + ret.erase(suffix); + return ret; + } + const size_t copy = ret.rfind('#'); + if (copy != std::string::npos && ret.find_first_not_of("0123456789", copy + 1) == std::string::npos) { + ret.erase(copy); + } + return ret; +} + static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync); static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( @@ -831,10 +853,26 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( if (ggml_nelements(tensor) == 0) { return {GGML_BACKEND_SPLIT_AXIS_UNKNOWN, {0}, {1}, 1}; } - if (ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE && tensor->view_src == nullptr) { + // A host-resident KV cache reaches the graph as a copied-in leaf of the compute buffer. + // Mirroring it while the queries stay split by head makes each device attend the wrong + // heads, so ask the callback; it still answers MIRRORED for names it does not know. + const bool copied_in_leaf = + ggml_backend_buffer_get_usage(tensor->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE && + tensor->op == GGML_OP_NONE && (tensor->flags & GGML_TENSOR_FLAG_INPUT) == 0; + + if ((ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE || copied_in_leaf) && + tensor->view_src == nullptr) { ggml_backend_dev_t dev = ggml_backend_buft_get_device(ggml_backend_buffer_get_type(tensor->buffer)); const ggml_backend_meta_device_context * dev_ctx = (const ggml_backend_meta_device_context *) dev->context; - ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor, dev_ctx->get_split_state_ud); + // the callback classifies by name, so offer a copy under the name it was copied from + const ggml_tensor * tensor_query = tensor; + ggml_tensor tensor_named; + if (copied_in_leaf) { + tensor_named = *tensor; + ggml_set_name(&tensor_named, ggml_backend_meta_copy_source_name(tensor->name).c_str()); + tensor_query = &tensor_named; + } + ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor_query, dev_ctx->get_split_state_ud); if (ret.axis >= 0 && ret.axis < GGML_MAX_DIMS) { const int64_t granularity = ret.axis == GGML_BACKEND_SPLIT_AXIS_0 ? ggml_blck_size(tensor->type) : 1; int64_t ne_sum = 0; @@ -1416,6 +1454,33 @@ static void ggml_backend_meta_buffer_memset_tensor( static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + + // A host-resident attention cache reaches this permuted, as [head_dim, n_kv, n_head_kv, 1]. The + // heads split, but interleaved per cell, so the chunk splice below cannot express the write. + // Each device's heads are one contiguous run per cell: ne[1] cells, from one stride to another. + const bool strided_head_split = + !ggml_is_contiguous(tensor) && + split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 && + split_state.n_segments == 1 && split_state.nr[0] == 1 && + tensor->ne[3] == 1 && tensor->nb[1] > tensor->nb[2] && + offset == 0 && size == ggml_nbytes(tensor); + + if (strided_head_split) { + size_t offset_data = 0; + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = split_state.ne[j] * tensor->nb[2]; + if (nbytes == 0) { + continue; + } + ggml_backend_tensor_set_2d(simple_tensor, (const char *) data + offset_data, 0, nbytes, + tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]); + offset_data += nbytes; + } + GGML_ASSERT(offset_data == (size_t) tensor->ne[2] * tensor->nb[2]); + return; + } + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); if (split_state.n_segments != 1 || split_state.nr[0] != 1) { diff --git a/src/llama-model.cpp b/src/llama-model.cpp index a65abfc9ca64..1f2d7c5b4459 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -382,8 +382,8 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str static const std::regex pattern_kv_bias ("blk\\.\\d*\\.attn_(k|v)\\.bias"); static const std::regex pattern_qkv_bias ("blk\\.\\d*\\.attn_qkv.bias"); static const std::regex pattern_qk_norm ("blk\\.\\d*\\.attn_(q|k)_norm\\.weight"); - static const std::regex pattern_kv_cache ("cache_(k|v)_l\\d*"); - static const std::regex pattern_idx_cache ("cache_idx_(k|v)_l\\d*"); + static const std::regex pattern_kv_cache ("cache_(k|v)_l\\d+"); + static const std::regex pattern_idx_cache ("cache_idx_(k|v)_l\\d+"); static const std::regex pattern_dsv4_state ("dsv4_(csa|hca|lid)_state_(kv|score)_l\\d*"); static const std::regex pattern_attn_sinks ("blk\\.\\d*\\.attn_sinks.weight"); static const std::regex pattern_attn_out_weight ("blk\\.\\d*\\.attn_output.weight"); @@ -398,9 +398,9 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str static const std::regex pattern_ssm_alpha ("blk\\.\\d*\\.ssm_alpha.weight"); static const std::regex pattern_ssm_beta ("blk\\.\\d*\\.ssm_beta.weight"); static const std::regex pattern_ssm_beta_alpha ("blk\\.\\d*\\.ssm_ba.weight"); - static const std::regex pattern_r_cache ("cache_r_l\\d*"); - static const std::regex pattern_ple_r_cache ("cache_ple_r_l\\d*"); - static const std::regex pattern_s_cache ("cache_s_l\\d*"); + static const std::regex pattern_r_cache ("cache_r_l\\d+"); + static const std::regex pattern_ple_r_cache ("cache_ple_r_l\\d+"); + static const std::regex pattern_s_cache ("cache_s_l\\d+"); static const std::regex pattern_ssm_conv1d ("blk\\.\\d*\\.ssm_conv1d.weight"); static const std::regex pattern_ssm_out_weight ("blk\\.\\d*\\.ssm_out.weight"); @@ -470,7 +470,34 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return {axis, tensor_axis_0, il, rotation}; }; + // A host-resident cache reaches attention as a scheduler copy, permuted, with the heads on an + // axis of their own. The split follows the cache, so record that axis and its unit - one head. + struct host_cache_copy { + bool valid = false; + ggml_backend_meta_split_axis axis = GGML_BACKEND_SPLIT_AXIS_0; + int64_t unit = 1; // cache elements per element of axis + }; + + const host_cache_copy cache_copy = [&]() -> host_cache_copy { + if (is_dsv4 || !std::regex_match(tensor_name, pattern_kv_cache)) { + return {}; + } + // [head_dim, n_kv, n_head_kv, n_stream] + const uint32_t il = std::stoul(tensor_name.substr(tensor_name.find("_l", 6) + 2)); + const int64_t head_dim = tensor_name[6] == 'k' ? hparams.n_embd_head_k(il) : hparams.n_embd_head_v(il); + if (hparams.n_head_kv(il) > 1 && tensor->ne[0] == head_dim && + tensor->ne[2] == (int64_t) hparams.n_head_kv(il)) { + return {true, GGML_BACKEND_SPLIT_AXIS_2, head_dim}; + } + return {}; + }(); + auto get_tensor_config = [&]() -> tensor_config { + // a graph tensor only reaches here as a scheduler copy, and only a copy of a host-resident + // cache is split - everything else the graph copies in keeps the mirrored fallback + if (!cache_copy.valid && ggml_backend_buffer_get_usage(tensor->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE) { + return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_MIRRORED); + } if (is_dsv4) { if (std::regex_match(tensor_name, pattern_kv_cache) || std::regex_match(tensor_name, pattern_dsv4_state)) { @@ -594,7 +621,9 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_MIRRORED); }; - auto get_split_segments = [&](int axis, uint32_t il) -> std::vector> { + // ne_axis is the extent of the split axis as the cache sees it, which for a host-resident + // cache copy is not the extent of the axis the copy is split on + auto get_split_segments = [&](int axis, uint32_t il, int64_t ne_axis) -> std::vector> { if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_QWEN4EXP) { const int64_t head_k_dim = hparams.ssm_d_state; @@ -642,7 +671,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str GGML_ASSERT(tensor->ne[axis] == 2*n_ff_exp); return {{n_ff_exp, 2}}; } - return {{tensor->ne[axis], 1}}; + return {{ne_axis, 1}}; } if (std::regex_match(tensor_name, pattern_qkv_weight) || std::regex_match(tensor_name, pattern_qkv_bias)) { @@ -658,14 +687,14 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str if (tensor->ne[axis] == 2*n_ff) { return {{n_ff, 2}}; } - return {{tensor->ne[axis], 1}}; + return {{ne_axis, 1}}; } if (std::regex_match(tensor_name, pattern_ffn_gate_up_weight)) { const int64_t n_ff_exp = hparams.n_ff_exp(il); GGML_ASSERT(tensor->ne[axis] == 2*n_ff_exp); return {{n_ff_exp, 2}}; } - return {{tensor->ne[axis], 1}}; + return {{ne_axis, 1}}; }; auto get_split_granularity = [&](int64_t blck_size, uint32_t il, const std::vector> & segments) -> std::vector { @@ -788,6 +817,11 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str tensor_config tc = get_tensor_config(); split_state.axis = tc.axis; if (split_state.axis >= 0 && split_state.axis < GGML_MAX_DIMS) { + // a host-resident cache copy is split on the axis its heads arrive on, in whole heads, + // so the segments and the granularity of the cache are rescaled to that unit + const bool unfolded = cache_copy.valid && split_state.axis == GGML_BACKEND_SPLIT_AXIS_0; + const int64_t unit = unfolded ? cache_copy.unit : 1; + const int64_t ne_axis = unfolded ? tensor->ne[cache_copy.axis]*unit : tensor->ne[split_state.axis]; const int64_t blck_size = ggml_blck_size(tc.tensor_axis_0->type); const float * tensor_split = ud->model->tensor_split(); std::vector tensor_split_scan; @@ -798,12 +832,17 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str tensor_split_scan[j] += tensor_split_scan[j - 1]; } } - const std::vector> segments = get_split_segments(split_state.axis, tc.il); + const std::vector> segments = get_split_segments(split_state.axis, tc.il, ne_axis); const std::vector granularity = get_split_granularity(blck_size, tc.il, segments); + if (unfolded) { + split_state.axis = cache_copy.axis; + } for (size_t is = 0; is < segments.size(); is++) { - const int64_t ne_s = segments[is].first; + GGML_ASSERT(segments[is].first % unit == 0); + GGML_ASSERT(granularity[is] % unit == 0); + const int64_t ne_s = segments[is].first / unit; const uint32_t nr_s = segments[is].second; - const int64_t g_s = granularity[is]; + const int64_t g_s = granularity[is] / unit; int64_t low = 0; size_t j = 0; for (; j < ud->n_devices - 1; j++) { From fa56aca076dba1ae1267ebcfc39c363a75109452 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 3 Sep 2026 11:21:53 +0200 Subject: [PATCH 03/11] llama : split a V tensor on its own head size The KV granularity was derived from the query granularity through n_gqa, which assumes the V side has the head size of the K side. Count whole KV heads and scale each side by its own head size. Assisted-by: Claude Opus 5 --- src/llama-model.cpp | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 1f2d7c5b4459..66ae3b617fb6 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -781,12 +781,16 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return {granularity_q}; } - const int64_t granularity_kv = granularity_q / n_gqa; + const int64_t n_head_gran = granularity_q / n_embd_q; // KV heads per granule + const int64_t granularity_kv = n_head_gran * hparams.n_embd_head_k(il); if (std::regex_match(tensor_name, pattern_kv_weight) || std::regex_match(tensor_name, pattern_kv_bias) || std::regex_match(tensor_name, pattern_kv_cache)) { GGML_ASSERT(segments.size() == 1); - return {granularity_kv}; + // the V side can have a head size of its own, the split still lands on head boundaries + const bool is_v = tensor_name.compare(0, 8, "cache_v_") == 0 || + tensor_name.find(".attn_v.") != std::string::npos; + return {is_v ? n_head_gran * hparams.n_embd_head_v(il) : granularity_kv}; } if (std::regex_match(tensor_name, pattern_qkv_weight) || std::regex_match(tensor_name, pattern_qkv_bias)) { GGML_ASSERT(segments.size() == 2); From 5719cbfa0f50332206e3b2ca50a8fffa5ca05aa1 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 3 Sep 2026 11:21:53 +0200 Subject: [PATCH 04/11] ggml : read back a strided row split from a meta buffer A fused QKV puts Kcur and Vcur in a strided view, which a host-resident cache reads back through the meta buffer. The rows split, so the chunk splice could not express the read. Assisted-by: Claude Opus 5 --- ggml/src/ggml-backend-meta.cpp | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index ec0bde1c9275..30e6070fd4b4 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -1609,6 +1609,33 @@ static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, gg static void ggml_backend_meta_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + + // A fused QKV puts Kcur and Vcur in a strided view, which a host-resident cache reads back here. + // The rows split, so the chunk splice below cannot express the read. Each device's part of a row + // is contiguous: ne[1] rows, from the device's own stride to the view's. + const bool strided_rows = + !ggml_is_contiguous(tensor) && + split_state.axis == GGML_BACKEND_SPLIT_AXIS_0 && + split_state.n_segments == 1 && split_state.nr[0] == 1 && + tensor->ne[2] == 1 && tensor->ne[3] == 1 && + offset == 0 && size == ggml_nbytes(tensor); + + if (strided_rows) { + size_t offset_data = 0; + for (size_t j = 0; j < n_bufs; j++) { + const ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = ggml_row_size(tensor->type, split_state.ne[j]); + if (nbytes == 0) { + continue; + } + ggml_backend_tensor_get_2d(simple_tensor, (char *) data + offset_data, 0, nbytes, + tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]); + offset_data += nbytes; + } + GGML_ASSERT(offset_data == ggml_row_size(tensor->type, tensor->ne[0])); + return; + } + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); if (split_state.n_segments != 1 || split_state.nr[0] != 1) { From 5cdde605b04677fe6b26cd6b650cef88b83c8d98 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 3 Sep 2026 11:23:39 +0200 Subject: [PATCH 05/11] llama : keep the recurrent state on device under split mode tensor A linear-attention op packs the state it writes back together with its output, so that split does not line up with the one a host-resident state expects. On device the two orders agree; in host memory they disagree, and the split state of the fused op no longer resolves. The state is small next to the attention cache that -nkvo exists to move, so keep it device-resident. Assisted-by: Claude Opus 5 --- src/llama-context.cpp | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/src/llama-context.cpp b/src/llama-context.cpp index 9aaa0f2a5cf6..f085569d1054 100644 --- a/src/llama-context.cpp +++ b/src/llama-context.cpp @@ -144,6 +144,14 @@ llama_context::llama_context( cparams.offload_kqv = params.offload_kqv; cparams.kv_cpu_pinned = params.kv_cpu_pinned; cparams.recurrent_state_offload = params.recurrent_state_offload; + + // A linear-attention op packs the state it writes back together with its output, so that split + // does not line up with the one a host-resident state expects. Keep the state on its device. + if (!cparams.recurrent_state_offload && model.split_mode() == LLAMA_SPLIT_MODE_TENSOR && + (llm_arch_is_recurrent(model.arch) || llm_arch_is_hybrid(model.arch))) { + LLAMA_LOG_INFO("%s: split mode tensor: keeping the recurrent state device-resident\n", __func__); + cparams.recurrent_state_offload = true; + } cparams.offload_attn_compute = params.offload_kqv || (params.op_offload && params.kv_cpu_pinned); cparams.kv_gpu_layers = params.kv_gpu_layers; cparams.phase_aware_workspace = params.phase_aware_workspace; From 346b38b7529f5fcf4d8b39351fcf859fb3eefef0 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 3 Sep 2026 11:22:08 +0200 Subject: [PATCH 06/11] tests : cover a host-resident KV cache split by tensor Run the tensor-split architecture matrix a second time with the cache in host memory, and add an 8-device CI run, where the scheduler copy name is long enough to be truncated. Assisted-by: Claude Opus 5 --- ci/run.sh | 4 +++- tests/test-llama-archs.cpp | 26 +++++++++++++++++++++----- 2 files changed, 24 insertions(+), 6 deletions(-) diff --git a/ci/run.sh b/ci/run.sh index 1ceb19fd50aa..110335a4efa5 100755 --- a/ci/run.sh +++ b/ci/run.sh @@ -312,6 +312,8 @@ function gg_run_test_llama_archs_tensor_split { GGML_CUDA_DEVICES=2 ./build-ci-release/bin/test-llama-archs -s 1 2>&1 GGML_CUDA_DEVICES=3 ./build-ci-release/bin/test-llama-archs -s 1 2>&1 GGML_CUDA_DEVICES=4 ./build-ci-release/bin/test-llama-archs -s 1 2>&1 + # the scheduler names a copy after its source, and 8 device names fill the name field + GGML_CUDA_DEVICES=8 ./build-ci-release/bin/test-llama-archs -s 1 -a llama 2>&1 fi if [ ! -z ${GG_BUILD_METAL} ]; then @@ -327,7 +329,7 @@ function gg_run_test_llama_archs_tensor_split { function gg_sum_test_llama_archs_tensor_split { gg_printf '### %s\n\n' "${ci}" - gg_printf 'Runs test-llama-archs with 1 to 4 devices\n' + gg_printf 'Runs test-llama-archs with 1 to 4 and 8 devices\n' gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)" gg_printf '```\n' gg_printf '%s\n' "$(cat $OUT/${ci}.log)" diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index 3951dbbe8526..b25d932c150a 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -398,9 +398,14 @@ static bool silent_model_load_progress(float /*progress*/, void * /*user_data*/) return true; } +// with offload_kqv=false the cache lives in host memory +struct kv_config { + bool offload_kqv = true; +}; + static std::pair get_model_and_ctx( struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const std::vector & devs, - const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false) { + const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false, const kv_config & kvc = {}) { GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr)); llama_model_params model_params = llama_model_default_params(); model_params.progress_callback = silent_model_load_progress; @@ -413,6 +418,7 @@ static std::pair get_model_and_ctx( ctx_params.n_ctx = 0; ctx_params.n_threads = 4; ctx_params.n_threads_batch = 4; + ctx_params.offload_kqv = kvc.offload_kqv; if (!encode) { ctx_params.n_ubatch = 64; } @@ -1377,9 +1383,10 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in std::vector devs; std::string label; llama_split_mode split_mode; + kv_config kvc; - device_config(std::vector devs, std::string name, llama_split_mode split_mode) - : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode) {} + device_config(std::vector devs, std::string name, llama_split_mode split_mode, kv_config kvc = {}) + : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), kvc(std::move(kvc)) {} }; std::vector dev_configs; @@ -1401,6 +1408,15 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in } dev_configs.emplace_back(devices_meta, "Meta", LLAMA_SPLIT_MODE_TENSOR); + + // a host-resident cache reaches attention as a scheduler copy that is split by head + kv_config kvc_host; + kvc_host.offload_kqv = false; + dev_configs.emplace_back(devices_meta, "Meta -nkvo", LLAMA_SPLIT_MODE_TENSOR, kvc_host); + + for (const device_config & dc : dev_configs) { + max_device_label_length = std::max(max_device_label_length, dc.label.length()); + } } size_t max_arch_name_length = 0; @@ -1473,7 +1489,7 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in logits_cpu = get_logits(model_and_ctx_cpu.first.get(), model_and_ctx_cpu.second.get(), tokens, encode); } if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) { - model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, dc.devs, dc.split_mode, encode); + model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, dc.devs, dc.split_mode, encode, dc.kvc); logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode); const double nmse_val = nmse(logits_cpu, logits_dev); snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val); @@ -1504,7 +1520,7 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in ms.save(file); rewind(file); - auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, dc.devs, dc.split_mode, encode); + auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, dc.devs, dc.split_mode, encode, dc.kvc); const std::vector logits_roundtrip = get_logits( model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode); status_roundtrip = "\033[1;32mOK\033[0m"; From 1f90680dcaa0660e1b10ebc6925f70bde5753120 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Fri, 4 Sep 2026 23:57:21 +0200 Subject: [PATCH 07/11] ggml-meta : write every stream of a head-split host KV cache The strided head split path ran only for a single stream, so a host-resident cache built with --parallel N fell through to the chunk splice, which cannot express that write. Loop over ne[3] and offset each side by its own stride. Assisted-by: Claude Opus 5 --- ggml/src/ggml-backend-meta.cpp | 29 ++++++++++++++++------------- 1 file changed, 16 insertions(+), 13 deletions(-) diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index 30e6070fd4b4..8593ad02ef57 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -1455,29 +1455,32 @@ static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, gg const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); - // A host-resident attention cache reaches this permuted, as [head_dim, n_kv, n_head_kv, 1]. The - // heads split, but interleaved per cell, so the chunk splice below cannot express the write. + // A host-resident attention cache reaches this permuted, as [head_dim, n_kv, n_head_kv, n_stream]. + // The heads split, but interleaved per cell, so the chunk splice below cannot express the write. // Each device's heads are one contiguous run per cell: ne[1] cells, from one stride to another. const bool strided_head_split = !ggml_is_contiguous(tensor) && split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 && split_state.n_segments == 1 && split_state.nr[0] == 1 && - tensor->ne[3] == 1 && tensor->nb[1] > tensor->nb[2] && + tensor->nb[1] > tensor->nb[2] && offset == 0 && size == ggml_nbytes(tensor); if (strided_head_split) { - size_t offset_data = 0; - for (size_t j = 0; j < n_bufs; j++) { - ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); - const size_t nbytes = split_state.ne[j] * tensor->nb[2]; - if (nbytes == 0) { - continue; + for (int64_t i3 = 0; i3 < tensor->ne[3]; i3++) { + size_t offset_data = i3 * tensor->nb[3]; + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = split_state.ne[j] * tensor->nb[2]; + if (nbytes == 0) { + continue; + } + ggml_backend_tensor_set_2d(simple_tensor, (const char *) data + offset_data, + i3 * simple_tensor->nb[3], nbytes, + tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]); + offset_data += nbytes; } - ggml_backend_tensor_set_2d(simple_tensor, (const char *) data + offset_data, 0, nbytes, - tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]); - offset_data += nbytes; + GGML_ASSERT(offset_data == i3 * tensor->nb[3] + (size_t) tensor->ne[2] * tensor->nb[2]); } - GGML_ASSERT(offset_data == (size_t) tensor->ne[2] * tensor->nb[2]); return; } From 35c62439569352b54ae77a0a3a4248cd6d8e3df2 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Sun, 6 Sep 2026 01:19:07 +0200 Subject: [PATCH 08/11] ggml-meta : close the last subgraph on a host-resident cache A split that ends with a view of a host tensor left the subgraph bookkeeping short of the node count and aborted. That happens with more than one cache stream, which the test matrix now covers. Also make the scheduler copy name a stated contract instead of a grammar that two files reconstruct on their own, note the 2d transfer fallback at both strided cache paths, say out loud that split mode tensor overrides the recurrent state placement, and record the Gemma 4 host cache accuracy gap where it is skipped. Assisted-by: Claude Opus 5 --- docs/multi-gpu.md | 2 + ggml/src/ggml-backend-impl.h | 13 ++++++ ggml/src/ggml-backend-meta.cpp | 46 +++++++------------ ggml/src/ggml-backend.cpp | 37 +++++++++++++-- src/llama-context.cpp | 3 +- tests/CMakeLists.txt | 7 +++ tests/test-llama-archs.cpp | 84 ++++++++++++++++++++++++++++------ 7 files changed, 144 insertions(+), 48 deletions(-) diff --git a/docs/multi-gpu.md b/docs/multi-gpu.md index 0d9eea7c2fb8..3d57bfbbd461 100644 --- a/docs/multi-gpu.md +++ b/docs/multi-gpu.md @@ -86,6 +86,8 @@ llama-cli -m model.gguf -sm tensor -ctk f16 -ctv f16 - `--flash-attn off` or (`--flash-attn auto` resolving to `off` when it isn't supported) is a hard error. - KV cache types must be non-quantized: `f32`, `f16`, or `bf16`. Support for quantized KV cache is not implemented and trying to use it will result in an error. - Mark this configuration as experimental in your tooling: validate output quality before deploying. +- `--no-kv-offload` works in this mode: the host-resident cache is split by attention head like the rest. Two limits: a backend without a native 2d copy (CPU, Metal) pays one transfer per cache cell, and Gemma 4 is less accurate this way (perplexity 235.03 with a host cache versus 227.23 with a device cache), so keep the cache on the devices for that architecture. +- A recurrent or hybrid model always keeps its recurrent state on the devices in this mode, even with `--no-recurrent-state-offload`. The linear-attention op writes the state back together with its output, so the two do not agree on a host-resident state. - `--split-mode tensor`is not implemented for all architectures. The following will fail with *"LLAMA_SPLIT_MODE_TENSOR not implemented for architecture '...'"*: - **MoE / hybrid:** Grok, MPT, OLMoE, DeepSeek2, GLM-DSA, Nemotron-H, Nemotron-H-MoE, Granite-Hybrid, LFM2-MoE, Minimax-M2, Mistral4, Kimi-Linear, Jamba, Falcon-H1 diff --git a/ggml/src/ggml-backend-impl.h b/ggml/src/ggml-backend-impl.h index ef05905cf9ab..f60f44997051 100644 --- a/ggml/src/ggml-backend-impl.h +++ b/ggml/src/ggml-backend-impl.h @@ -104,6 +104,19 @@ extern "C" { // temporary workaround to statically allocate tensors from a context in a deduplicated way: GGML_API struct ggml_backend_buffer * ggml_backend_meta_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft); + // + // Backend (sched) + // + + // The scheduler names a copy of a graph input "##". is the name of the + // tensor the copy was made from and carries any suffix that ggml appends for a view. Only + // identifies the copy, so the backend label is cut when the name does not fit. + GGML_API void ggml_backend_sched_name_copy( + struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c); + + // write the part of a name written by ggml_backend_sched_name_copy into buf + GGML_API void ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size); + // // Backend (stream) // diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index 8593ad02ef57..f86712aff013 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -485,28 +485,6 @@ static struct ggml_tensor * ggml_backend_meta_buffer_simple_tensor(const struct return it->second[index]; } -// A scheduler copy is named "##", where also carries the suffixes -// ggml appends for views. Recover the graph name the copy was made from. -static std::string ggml_backend_meta_copy_source_name(const char * name) { - std::string ret = name; - const size_t first = ret.find('#'); - if (first == std::string::npos) { - return ret; - } - ret.erase(0, first + 1); - // ggml writes a view suffix as " (...)", a graph name has no spaces - const size_t suffix = ret.find(" ("); - if (suffix != std::string::npos) { - ret.erase(suffix); - return ret; - } - const size_t copy = ret.rfind('#'); - if (copy != std::string::npos && ret.find_first_not_of("0123456789", copy + 1) == std::string::npos) { - ret.erase(copy); - } - return ret; -} - static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync); static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( @@ -869,7 +847,9 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( ggml_tensor tensor_named; if (copied_in_leaf) { tensor_named = *tensor; - ggml_set_name(&tensor_named, ggml_backend_meta_copy_source_name(tensor->name).c_str()); + char source_name[GGML_MAX_NAME]; + ggml_backend_sched_copy_source_name(tensor->name, source_name, sizeof(source_name)); + ggml_set_name(&tensor_named, source_name); tensor_query = &tensor_named; } ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor_query, dev_ctx->get_split_state_ud); @@ -1458,6 +1438,7 @@ static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, gg // A host-resident attention cache reaches this permuted, as [head_dim, n_kv, n_head_kv, n_stream]. // The heads split, but interleaved per cell, so the chunk splice below cannot express the write. // Each device's heads are one contiguous run per cell: ne[1] cells, from one stride to another. + // A backend without a native 2d copy (CPU, Metal) pays one transfer per cell here, CUDA does not. const bool strided_head_split = !ggml_is_contiguous(tensor) && split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 && @@ -1616,6 +1597,7 @@ static void ggml_backend_meta_buffer_get_tensor(ggml_backend_buffer_t buffer, co // A fused QKV puts Kcur and Vcur in a strided view, which a host-resident cache reads back here. // The rows split, so the chunk splice below cannot express the read. Each device's part of a row // is contiguous: ne[1] rows, from the device's own stride to the view's. + // A backend without a native 2d copy (CPU, Metal) pays one transfer per row here, CUDA does not. const bool strided_rows = !ggml_is_contiguous(tensor) && split_state.axis == GGML_BACKEND_SPLIT_AXIS_0 && @@ -2277,14 +2259,18 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend, int i_start = 0; for (int i = 0; i < cgraph->n_nodes; i++) { ggml_tensor * node = cgraph->nodes[i]; - if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) { - continue; - } - const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(node, /*assume_sync =*/ false); - if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) { - max_tmp_size = std::max(max_tmp_size, ggml_nbytes(node)); + // a host-resident KV cache ends a split with a view of itself, that view needs no split state + // but it is still the last node and must close the last subgraph + const bool host_view = node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && + ggml_backend_buffer_is_host(node->view_src->buffer); + bool new_subgraph = i + 1 == cgraph->n_nodes; + if (!host_view) { + const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(node, /*assume_sync =*/ false); + if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) { + max_tmp_size = std::max(max_tmp_size, ggml_nbytes(node)); + new_subgraph = true; + } } - const bool new_subgraph = i + 1 == cgraph->n_nodes || split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL; if (!new_subgraph) { continue; } diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp index fd9788ef025f..94080c86fc73 100644 --- a/ggml/src/ggml-backend.cpp +++ b/ggml/src/ggml-backend.cpp @@ -1034,9 +1034,9 @@ static void ggml_backend_sched_print_assignments(ggml_backend_sched_t sched, str } } -// Name a copy of an input "##". The name field has a fixed size, so cut the -// backend label rather than the source name - the source name is what identifies the copy. -static void ggml_backend_sched_name_copy( +// The name field has a fixed size, so cut the backend label rather than the source name. +// See the contract at the declaration in ggml-backend-impl.h. +void ggml_backend_sched_name_copy( struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c) { const int n_tail = snprintf(NULL, 0, "#%s#%d", src->name, c); const int n_max = GGML_MAX_NAME - 1 - n_tail; @@ -1047,6 +1047,37 @@ static void ggml_backend_sched_name_copy( ggml_format_name(copy, "%.*s#%s#%d", n_head, backend_name, src->name, c); } +void ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size) { + GGML_ASSERT(buf_size > 0); + + const char * first = strchr(name, '#'); + const char * src = first != NULL ? first + 1 : name; + size_t len = strlen(src); + + // ggml writes a view suffix as " (...)", a graph name has no spaces + const char * suffix = strstr(src, " ("); + if (suffix != NULL) { + len = suffix - src; + } else { + const char * copy = strrchr(src, '#'); + if (copy != NULL) { + const char * digits = copy + 1; + while (*digits >= '0' && *digits <= '9') { + digits++; + } + if (*digits == '\0') { + len = copy - src; + } + } + } + + if (len > buf_size - 1) { + len = buf_size - 1; + } + memcpy(buf, src, len); + buf[len] = '\0'; +} + static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) { ggml_backend_buffer_t buf = t->view_src ? t->view_src->buffer : t->buffer; ggml_backend_buffer_type_t buft = NULL; diff --git a/src/llama-context.cpp b/src/llama-context.cpp index f085569d1054..c7ba2f400c08 100644 --- a/src/llama-context.cpp +++ b/src/llama-context.cpp @@ -149,7 +149,8 @@ llama_context::llama_context( // does not line up with the one a host-resident state expects. Keep the state on its device. if (!cparams.recurrent_state_offload && model.split_mode() == LLAMA_SPLIT_MODE_TENSOR && (llm_arch_is_recurrent(model.arch) || llm_arch_is_hybrid(model.arch))) { - LLAMA_LOG_INFO("%s: split mode tensor: keeping the recurrent state device-resident\n", __func__); + LLAMA_LOG_WARN("%s: split mode tensor cannot keep the recurrent state in host memory - " + "overriding --no-recurrent-state-offload, which needs more VRAM\n", __func__); cparams.recurrent_state_offload = true; } cparams.offload_attn_compute = params.offload_kqv || (params.op_offload && params.kv_cpu_pinned); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 04693bb0e6ed..aa42167bf30e 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -214,6 +214,13 @@ if (NOT WIN32 OR NOT BUILD_SHARED_LIBS) ARGS --test-live-context-workspace ) + llama_test( + test-llama-archs + NAME test-sched-copy-name + LABEL main + ARGS --test-sched-copy-name + ) + set(MODEL_DIR "${CMAKE_CURRENT_BINARY_DIR}/test-models/") file(MAKE_DIRECTORY "${MODEL_DIR}") diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index b25d932c150a..329fd951fa3c 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -69,7 +70,7 @@ static void set_tensor_data(struct ggml_tensor * tensor, void * userdata) { } static void usage(char ** argv) { - printf("Usage: %s [-a/--arch arch] [-s/--seed seed] [-o/--out dir] [-v N] [-h/--help] [--test-phase-workspace] [--test-live-context-workspace]\n", argv[0]); + printf("Usage: %s [-a/--arch arch] [-s/--seed seed] [-o/--out dir] [-v N] [-h/--help] [--test-phase-workspace] [--test-live-context-workspace] [--test-sched-copy-name]\n", argv[0]); } static std::vector get_tokens(const uint32_t n_tokens, const uint32_t n_vocab, const size_t seed){ @@ -399,8 +400,10 @@ static bool silent_model_load_progress(float /*progress*/, void * /*user_data*/) } // with offload_kqv=false the cache lives in host memory +// n_seq_max > 1 gives the cache one stream per sequence struct kv_config { - bool offload_kqv = true; + bool offload_kqv = true; + uint32_t n_seq_max = 1; }; static std::pair get_model_and_ctx( @@ -419,6 +422,7 @@ static std::pair get_model_and_ctx( ctx_params.n_threads = 4; ctx_params.n_threads_batch = 4; ctx_params.offload_kqv = kvc.offload_kqv; + ctx_params.n_seq_max = kvc.n_seq_max; if (!encode) { ctx_params.n_ubatch = 64; } @@ -1058,15 +1062,44 @@ static void test_phase_workspace_mismatched_placement(size_t seed) { GGML_ASSERT(llama_contexts_share_workspace(target.get(), draft.get()) == (status == 1)); } +// the meta backend recovers the source of a scheduler copy from its name, so both sides must agree +static void test_sched_copy_name() { + const ggml_init_params params = { + /*.mem_size =*/ 2*ggml_tensor_overhead(), + /*.mem_buffer =*/ nullptr, + /*.no_alloc =*/ true, + }; + ggml_context_ptr ctx(ggml_init(params)); + GGML_ASSERT(ctx); + ggml_tensor * src = ggml_new_tensor_1d(ctx.get(), GGML_TYPE_F32, 1); + ggml_tensor * copy = ggml_new_tensor_1d(ctx.get(), GGML_TYPE_F32, 1); + + auto check = [&](const char * name, const char * backend_name, const char * expected) { + char buf[GGML_MAX_NAME]; + ggml_set_name(src, name); + ggml_backend_sched_name_copy(copy, backend_name, src, 0); + ggml_backend_sched_copy_source_name(copy->name, buf, sizeof(buf)); + GGML_ASSERT(strcmp(buf, expected) == 0); + }; + + check("cache_k_l0", "CUDA0", "cache_k_l0"); + check("cache_k_l0 (view)", "CUDA0", "cache_k_l0"); + // the backend label is cut when the name does not fit, the source name survives + check("cache_k_l31", "Meta(CUDA0,CUDA1,CUDA2,CUDA3,CUDA4,CUDA5,CUDA6,CUDA7)", "cache_k_l31"); +} + +// the tokens are spread over n_seq sequences, each of which starts at position 0 static std::vector get_logits( - llama_model * model, llama_context * lctx, const std::vector & tokens, bool encode = false) { + llama_model * model, llama_context * lctx, const std::vector & tokens, bool encode = false, uint32_t n_seq = 1) { const uint32_t n_vocab = llama_vocab_n_tokens(llama_model_get_vocab(model)); const uint32_t n_ctx = llama_n_ctx(lctx); const uint32_t n_tokens = tokens.size(); - llama_batch batch = llama_batch_init(n_ctx, 0, 1); + GGML_ASSERT(n_seq >= 1 && n_tokens % n_seq == 0); + const uint32_t n_tokens_seq = n_tokens / n_seq; + llama_batch batch = llama_batch_init(n_ctx, 0, n_seq); GGML_ASSERT(n_tokens <= n_ctx); for (uint32_t pos = 0; pos < n_tokens; pos++) { - common_batch_add(batch, tokens[pos], pos, {0}, true); + common_batch_add(batch, tokens[pos], pos % n_tokens_seq, {llama_seq_id(pos / n_tokens_seq)}, true); } batch.n_tokens = n_tokens; if (encode) { @@ -1414,6 +1447,11 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in kvc_host.offload_kqv = false; dev_configs.emplace_back(devices_meta, "Meta -nkvo", LLAMA_SPLIT_MODE_TENSOR, kvc_host); + // with more than one stream that copy is 4d, one stride per stream + kv_config kvc_host_streams = kvc_host; + kvc_host_streams.n_seq_max = 2; + dev_configs.emplace_back(devices_meta, "Meta -nkvo -np 2", LLAMA_SPLIT_MODE_TENSOR, kvc_host_streams); + for (const device_config & dc : dev_configs) { max_device_label_length = std::max(max_device_label_length, dc.label.length()); } @@ -1448,7 +1486,10 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in continue; } if (arch == LLM_ARCH_GEMMA4 || arch == LLM_ARCH_GEMMA4_ASSISTANT) { - continue; // FIXME: ISWA KV cache initialization needs more fixture params + // FIXME: ISWA KV cache initialization needs more fixture params + // this arch also loses accuracy with a host-resident cache under split mode tensor + // (perplexity 235.03 versus 227.23), the other ISWA archs are clean + continue; } if (arch == LLM_ARCH_EAGLE3 || arch == LLM_ARCH_DFLASH) { continue; @@ -1468,7 +1509,7 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in GGML_ASSERT(gguf_remove_key(gguf_ctx.get(), "bailingmoe3.kda.safe_gate") >= 0); } std::pair model_and_ctx_cpu; - std::vector logits_cpu; + std::map> logits_cpu_per_n_seq; for (device_config & dc : dev_configs) { // print test config first; should anything fail during model loading or inference, at least we know which test case caused it printf(template_row_cfg.c_str(), @@ -1482,15 +1523,21 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in std::string status_parallel = "\033[1;33mSKIP\033[0m"; char nmse_str[12] = {0}; - bool skip = !arch_supported(arch) || (dc.split_mode == LLAMA_SPLIT_MODE_TENSOR && dc.devs.empty()); + // an encoder-decoder model needs its own batch layout, so it stays on one sequence + bool skip = !arch_supported(arch) || (dc.split_mode == LLAMA_SPLIT_MODE_TENSOR && dc.devs.empty()) || + (encode && dc.kvc.n_seq_max > 1); if (!skip) { - if (logits_cpu.empty()) { - model_and_ctx_cpu = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {}, LLAMA_SPLIT_MODE_LAYER, encode); - logits_cpu = get_logits(model_and_ctx_cpu.first.get(), model_and_ctx_cpu.second.get(), tokens, encode); - } if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) { + // the reference runs the same batch layout, one sequence per stream + std::vector & logits_cpu = logits_cpu_per_n_seq[dc.kvc.n_seq_max]; + if (logits_cpu.empty()) { + kv_config kvc_cpu; + kvc_cpu.n_seq_max = dc.kvc.n_seq_max; + model_and_ctx_cpu = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {}, LLAMA_SPLIT_MODE_LAYER, encode, kvc_cpu); + logits_cpu = get_logits(model_and_ctx_cpu.first.get(), model_and_ctx_cpu.second.get(), tokens, encode, dc.kvc.n_seq_max); + } model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, dc.devs, dc.split_mode, encode, dc.kvc); - logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode); + logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode, dc.kvc.n_seq_max); const double nmse_val = nmse(logits_cpu, logits_dev); snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val); status_nmse = "\033[1;32mOK\033[0m"; @@ -1522,7 +1569,7 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, dc.devs, dc.split_mode, encode, dc.kvc); const std::vector logits_roundtrip = get_logits( - model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode); + model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode, dc.kvc.n_seq_max); status_roundtrip = "\033[1;32mOK\033[0m"; GGML_ASSERT(logits_roundtrip.size() == logits_dev.size()); for (size_t i = 0; i < logits_roundtrip.size(); i++) { @@ -1557,6 +1604,7 @@ int main(int argc, char ** argv) { std::string out; bool test_phase_workspace = false; bool test_live_context_workspace = false; + bool test_copy_name = false; int verbosity = LOG_LEVEL_ERROR; @@ -1610,6 +1658,10 @@ int main(int argc, char ** argv) { test_live_context_workspace = true; continue; } + if (strcmp(argv[i], "--test-sched-copy-name") == 0) { + test_copy_name = true; + continue; + } } printf("%s: using seed %zu\n", __func__, seed); @@ -1627,6 +1679,10 @@ int main(int argc, char ** argv) { test_live_context_workspace_unsupported(seed); return 0; } + if (test_copy_name) { + test_sched_copy_name(); + return 0; + } if (!out.empty()) { return save_models(arch, seed, verbosity, out); } From 20b0891d143c54e7993d0c732f3f345953533c13 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Mon, 7 Sep 2026 21:40:53 +0200 Subject: [PATCH 09/11] ggml-meta : identify a scheduler copy by its name, not by its flags The scheduler flags a copy as an input when it keeps more than one, which hid a host-resident cache from the split state callback. A cut source name now asserts instead of naming another tensor, and memset writes a head split like set_tensor does. Assisted-by: Claude Opus 5 --- ggml/src/ggml-backend-impl.h | 8 +++++--- ggml/src/ggml-backend-meta.cpp | 33 ++++++++++++++++++++++++++++++--- ggml/src/ggml-backend.cpp | 31 +++++++++++++++++++------------ tests/test-llama-archs.cpp | 10 ++++++++-- 4 files changed, 62 insertions(+), 20 deletions(-) diff --git a/ggml/src/ggml-backend-impl.h b/ggml/src/ggml-backend-impl.h index f60f44997051..36abb660535b 100644 --- a/ggml/src/ggml-backend-impl.h +++ b/ggml/src/ggml-backend-impl.h @@ -110,12 +110,14 @@ extern "C" { // The scheduler names a copy of a graph input "##". is the name of the // tensor the copy was made from and carries any suffix that ggml appends for a view. Only - // identifies the copy, so the backend label is cut when the name does not fit. + // identifies the copy, so the backend label is cut when the name does not fit. A source name that + // does not fit on its own asserts, a cut one would name a different tensor. GGML_API void ggml_backend_sched_name_copy( struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c); - // write the part of a name written by ggml_backend_sched_name_copy into buf - GGML_API void ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size); + // write the part of a name written by ggml_backend_sched_name_copy into buf, + // returns false and leaves buf alone if the name is not one + GGML_API bool ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size); // // Backend (stream) diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index f86712aff013..23dc91dee3d4 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -834,9 +834,12 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( // A host-resident KV cache reaches the graph as a copied-in leaf of the compute buffer. // Mirroring it while the queries stay split by head makes each device attend the wrong // heads, so ask the callback; it still answers MIRRORED for names it does not know. + // The name tells a copy apart, the scheduler also flags one as an input when it keeps several. + char source_name[GGML_MAX_NAME]; const bool copied_in_leaf = ggml_backend_buffer_get_usage(tensor->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE && - tensor->op == GGML_OP_NONE && (tensor->flags & GGML_TENSOR_FLAG_INPUT) == 0; + tensor->op == GGML_OP_NONE && + ggml_backend_sched_copy_source_name(tensor->name, source_name, sizeof(source_name)); if ((ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE || copied_in_leaf) && tensor->view_src == nullptr) { @@ -847,8 +850,6 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( ggml_tensor tensor_named; if (copied_in_leaf) { tensor_named = *tensor; - char source_name[GGML_MAX_NAME]; - ggml_backend_sched_copy_source_name(tensor->name, source_name, sizeof(source_name)); ggml_set_name(&tensor_named, source_name); tensor_query = &tensor_named; } @@ -1334,6 +1335,32 @@ static void ggml_backend_meta_buffer_memset_tensor( const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + + // a host-resident attention cache is permuted, its heads are one run per cell, see set_tensor + const bool strided_head_split = + !ggml_is_contiguous(tensor) && + split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 && + split_state.n_segments == 1 && split_state.nr[0] == 1 && + tensor->nb[1] > tensor->nb[2] && + offset == 0 && size == ggml_nbytes(tensor); + + if (strided_head_split) { + for (int64_t i3 = 0; i3 < tensor->ne[3]; i3++) { + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = split_state.ne[j] * tensor->nb[2]; + if (nbytes == 0) { + continue; + } + for (int64_t i1 = 0; i1 < tensor->ne[1]; i1++) { + ggml_backend_tensor_memset(simple_tensor, value, + i3*simple_tensor->nb[3] + i1*simple_tensor->nb[1], nbytes); + } + } + } + return; + } + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); if (split_state.n_segments != 1 || split_state.nr[0] != 1) { diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp index 94080c86fc73..66f4d2dc28b9 100644 --- a/ggml/src/ggml-backend.cpp +++ b/ggml/src/ggml-backend.cpp @@ -1040,18 +1040,22 @@ void ggml_backend_sched_name_copy( struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c) { const int n_tail = snprintf(NULL, 0, "#%s#%d", src->name, c); const int n_max = GGML_MAX_NAME - 1 - n_tail; + GGML_ASSERT(n_max >= 0 && "source name too long to name a scheduler copy"); int n_head = (int) strlen(backend_name); if (n_head > n_max) { - n_head = n_max > 0 ? n_max : 0; + n_head = n_max; } ggml_format_name(copy, "%.*s#%s#%d", n_head, backend_name, src->name, c); } -void ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size) { +bool ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size) { GGML_ASSERT(buf_size > 0); const char * first = strchr(name, '#'); - const char * src = first != NULL ? first + 1 : name; + if (first == NULL) { + return false; + } + const char * src = first + 1; size_t len = strlen(src); // ggml writes a view suffix as " (...)", a graph name has no spaces @@ -1060,22 +1064,25 @@ void ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t b len = suffix - src; } else { const char * copy = strrchr(src, '#'); - if (copy != NULL) { - const char * digits = copy + 1; - while (*digits >= '0' && *digits <= '9') { - digits++; - } - if (*digits == '\0') { - len = copy - src; - } + if (copy == NULL) { + return false; } + const char * digits = copy + 1; + while (*digits >= '0' && *digits <= '9') { + digits++; + } + if (digits == copy + 1 || *digits != '\0') { + return false; + } + len = copy - src; } if (len > buf_size - 1) { - len = buf_size - 1; + return false; } memcpy(buf, src, len); buf[len] = '\0'; + return true; } static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) { diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index 329fd951fa3c..b4411522b9b9 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -1078,7 +1078,7 @@ static void test_sched_copy_name() { char buf[GGML_MAX_NAME]; ggml_set_name(src, name); ggml_backend_sched_name_copy(copy, backend_name, src, 0); - ggml_backend_sched_copy_source_name(copy->name, buf, sizeof(buf)); + GGML_ASSERT(ggml_backend_sched_copy_source_name(copy->name, buf, sizeof(buf))); GGML_ASSERT(strcmp(buf, expected) == 0); }; @@ -1086,6 +1086,13 @@ static void test_sched_copy_name() { check("cache_k_l0 (view)", "CUDA0", "cache_k_l0"); // the backend label is cut when the name does not fit, the source name survives check("cache_k_l31", "Meta(CUDA0,CUDA1,CUDA2,CUDA3,CUDA4,CUDA5,CUDA6,CUDA7)", "cache_k_l31"); + + // a name that was not written by the scheduler is not a copy + char buf[GGML_MAX_NAME]; + GGML_ASSERT(!ggml_backend_sched_copy_source_name("cache_k_l0", buf, sizeof(buf))); + GGML_ASSERT(!ggml_backend_sched_copy_source_name("CUDA0#cache_k_l0", buf, sizeof(buf))); + // a cut name lost its copy index, what is left of the source names another tensor + GGML_ASSERT(!ggml_backend_sched_copy_source_name("#cache_k_l3", buf, sizeof(buf))); } // the tokens are spread over n_seq sequences, each of which starts at position 0 @@ -1431,7 +1438,6 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in for (size_t i = 0; i < device_count; i++) { ggml_backend_dev_t dev = ggml_backend_dev_get(i); dev_configs.emplace_back(std::vector{dev}, ggml_backend_dev_description(dev), LLAMA_SPLIT_MODE_LAYER); - max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length()); // cpu-based devices cannot be used in tensor split mode if (ggml_backend_dev_buffer_type(dev) != ggml_backend_cpu_buffer_type()) { From 50b379a5f5d01d733a03d64b4b01df8e9505e3b6 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Mon, 14 Sep 2026 02:04:53 +0200 Subject: [PATCH 10/11] sched : write a multi-stream window to a meta backend in one copy The ranged copy of a host-resident window calls set_tensor_async once per stream. A meta backend only takes a whole contiguous tensor, so with two or more streams it aborts. Keep the whole-span copy for a meta destination. Assisted-by: Claude Opus 5 --- ggml/src/ggml-backend.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp index 66f4d2dc28b9..bea33db60bc2 100644 --- a/ggml/src/ggml-backend.cpp +++ b/ggml/src/ggml-backend.cpp @@ -1863,7 +1863,8 @@ static enum ggml_status ggml_backend_sched_compute_splits(ggml_backend_sched_t s ggml_backend_buffer_t src_buf = input->view_src ? input->view_src->buffer : input->buffer; struct ggml_backend_sched_ranges rg; ggml_backend_sched_input_ranges(input, &rg); - const bool ranged = rg.n > 1 && src_buf != NULL && ggml_backend_buffer_is_host(src_buf); + // a meta backend writes a whole contiguous tensor, it cannot take one range per stream + const bool ranged = rg.n > 1 && src_buf != NULL && ggml_backend_buffer_is_host(src_buf) && !ggml_backend_is_meta(split_backend); // try async copy, but if not possible, we can still use a sync copy without synchronizing the dst backend, since we handle the synchronization here with multiple copies and events // TODO: add public function to facilitate this, since applications do not have direct access to the backend interface From 9faffa99a4ba76fbaa4c6c23b874fcd4c74e9ed6 Mon Sep 17 00:00:00 2001 From: piggidragon Date: Thu, 17 Sep 2026 21:51:49 +0200 Subject: [PATCH 11/11] tests : align the parallel check with the host KV rows A multi-stream row decodes its reference as several sequences, so it cannot serve as a one-sequence reference - skip the check there. Give the check the row's offload_kqv, so it covers a host-resident cache with two streams. Assisted-by: Claude Opus 5 --- tests/test-llama-archs.cpp | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index b4411522b9b9..cf193bf168e6 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -1134,7 +1134,7 @@ static std::vector get_logits( // decode two sequences in one batch, compare each with a decode of it alone // logits_a: logits of tokens decoded alone with the same device config -static bool test_parallel_seqs(llama_model * model, const std::vector & tokens, const std::vector & logits_a, bool encode) { +static bool test_parallel_seqs(llama_model * model, const std::vector & tokens, const std::vector & logits_a, bool encode, bool offload_kqv) { const uint32_t n_vocab = llama_vocab_n_tokens(llama_model_get_vocab(model)); const uint32_t n_tokens = tokens.size(); @@ -1145,6 +1145,7 @@ static bool test_parallel_seqs(llama_model * model, const std::vector 1 aborts - if (arch != LLM_ARCH_T5) { + // a multi-stream row spreads the reference over several sequences, so it does not match one sequence at pos 0..n-1 + if (arch != LLM_ARCH_T5 && dc.kvc.n_seq_max == 1) { status_parallel = "\033[1;32mOK\033[0m"; - if (!test_parallel_seqs(model_and_ctx_dev.first.get(), tokens, logits_dev, encode)) { + if (!test_parallel_seqs(model_and_ctx_dev.first.get(), tokens, logits_dev, encode, dc.kvc.offload_kqv)) { all_ok = false; status_parallel = "\033[1;31mFAIL\033[0m"; }