diff --git a/ci/run.sh b/ci/run.sh index 1ceb19fd50aa..110335a4efa5 100755 --- a/ci/run.sh +++ b/ci/run.sh @@ -312,6 +312,8 @@ function gg_run_test_llama_archs_tensor_split { GGML_CUDA_DEVICES=2 ./build-ci-release/bin/test-llama-archs -s 1 2>&1 GGML_CUDA_DEVICES=3 ./build-ci-release/bin/test-llama-archs -s 1 2>&1 GGML_CUDA_DEVICES=4 ./build-ci-release/bin/test-llama-archs -s 1 2>&1 + # the scheduler names a copy after its source, and 8 device names fill the name field + GGML_CUDA_DEVICES=8 ./build-ci-release/bin/test-llama-archs -s 1 -a llama 2>&1 fi if [ ! -z ${GG_BUILD_METAL} ]; then @@ -327,7 +329,7 @@ function gg_run_test_llama_archs_tensor_split { function gg_sum_test_llama_archs_tensor_split { gg_printf '### %s\n\n' "${ci}" - gg_printf 'Runs test-llama-archs with 1 to 4 devices\n' + gg_printf 'Runs test-llama-archs with 1 to 4 and 8 devices\n' gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)" gg_printf '```\n' gg_printf '%s\n' "$(cat $OUT/${ci}.log)" diff --git a/docs/multi-gpu.md b/docs/multi-gpu.md index 0d9eea7c2fb8..3d57bfbbd461 100644 --- a/docs/multi-gpu.md +++ b/docs/multi-gpu.md @@ -86,6 +86,8 @@ llama-cli -m model.gguf -sm tensor -ctk f16 -ctv f16 - `--flash-attn off` or (`--flash-attn auto` resolving to `off` when it isn't supported) is a hard error. - KV cache types must be non-quantized: `f32`, `f16`, or `bf16`. Support for quantized KV cache is not implemented and trying to use it will result in an error. - Mark this configuration as experimental in your tooling: validate output quality before deploying. +- `--no-kv-offload` works in this mode: the host-resident cache is split by attention head like the rest. Two limits: a backend without a native 2d copy (CPU, Metal) pays one transfer per cache cell, and Gemma 4 is less accurate this way (perplexity 235.03 with a host cache versus 227.23 with a device cache), so keep the cache on the devices for that architecture. +- A recurrent or hybrid model always keeps its recurrent state on the devices in this mode, even with `--no-recurrent-state-offload`. The linear-attention op writes the state back together with its output, so the two do not agree on a host-resident state. - `--split-mode tensor`is not implemented for all architectures. The following will fail with *"LLAMA_SPLIT_MODE_TENSOR not implemented for architecture '...'"*: - **MoE / hybrid:** Grok, MPT, OLMoE, DeepSeek2, GLM-DSA, Nemotron-H, Nemotron-H-MoE, Granite-Hybrid, LFM2-MoE, Minimax-M2, Mistral4, Kimi-Linear, Jamba, Falcon-H1 diff --git a/ggml/src/ggml-backend-impl.h b/ggml/src/ggml-backend-impl.h index ef05905cf9ab..36abb660535b 100644 --- a/ggml/src/ggml-backend-impl.h +++ b/ggml/src/ggml-backend-impl.h @@ -104,6 +104,21 @@ extern "C" { // temporary workaround to statically allocate tensors from a context in a deduplicated way: GGML_API struct ggml_backend_buffer * ggml_backend_meta_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft); + // + // Backend (sched) + // + + // The scheduler names a copy of a graph input "##". is the name of the + // tensor the copy was made from and carries any suffix that ggml appends for a view. Only + // identifies the copy, so the backend label is cut when the name does not fit. A source name that + // does not fit on its own asserts, a cut one would name a different tensor. + GGML_API void ggml_backend_sched_name_copy( + struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c); + + // write the part of a name written by ggml_backend_sched_name_copy into buf, + // returns false and leaves buf alone if the name is not one + GGML_API bool ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size); + // // Backend (stream) // diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index 3ec40fb1af7f..23dc91dee3d4 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -831,10 +831,29 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( if (ggml_nelements(tensor) == 0) { return {GGML_BACKEND_SPLIT_AXIS_UNKNOWN, {0}, {1}, 1}; } - if (ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE && tensor->view_src == nullptr) { + // A host-resident KV cache reaches the graph as a copied-in leaf of the compute buffer. + // Mirroring it while the queries stay split by head makes each device attend the wrong + // heads, so ask the callback; it still answers MIRRORED for names it does not know. + // The name tells a copy apart, the scheduler also flags one as an input when it keeps several. + char source_name[GGML_MAX_NAME]; + const bool copied_in_leaf = + ggml_backend_buffer_get_usage(tensor->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE && + tensor->op == GGML_OP_NONE && + ggml_backend_sched_copy_source_name(tensor->name, source_name, sizeof(source_name)); + + if ((ggml_backend_buffer_get_usage(tensor->buffer) != GGML_BACKEND_BUFFER_USAGE_COMPUTE || copied_in_leaf) && + tensor->view_src == nullptr) { ggml_backend_dev_t dev = ggml_backend_buft_get_device(ggml_backend_buffer_get_type(tensor->buffer)); const ggml_backend_meta_device_context * dev_ctx = (const ggml_backend_meta_device_context *) dev->context; - ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor, dev_ctx->get_split_state_ud); + // the callback classifies by name, so offer a copy under the name it was copied from + const ggml_tensor * tensor_query = tensor; + ggml_tensor tensor_named; + if (copied_in_leaf) { + tensor_named = *tensor; + ggml_set_name(&tensor_named, source_name); + tensor_query = &tensor_named; + } + ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor_query, dev_ctx->get_split_state_ud); if (ret.axis >= 0 && ret.axis < GGML_MAX_DIMS) { const int64_t granularity = ret.axis == GGML_BACKEND_SPLIT_AXIS_0 ? ggml_blck_size(tensor->type) : 1; int64_t ne_sum = 0; @@ -1316,6 +1335,32 @@ static void ggml_backend_meta_buffer_memset_tensor( const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + + // a host-resident attention cache is permuted, its heads are one run per cell, see set_tensor + const bool strided_head_split = + !ggml_is_contiguous(tensor) && + split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 && + split_state.n_segments == 1 && split_state.nr[0] == 1 && + tensor->nb[1] > tensor->nb[2] && + offset == 0 && size == ggml_nbytes(tensor); + + if (strided_head_split) { + for (int64_t i3 = 0; i3 < tensor->ne[3]; i3++) { + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = split_state.ne[j] * tensor->nb[2]; + if (nbytes == 0) { + continue; + } + for (int64_t i1 = 0; i1 < tensor->ne[1]; i1++) { + ggml_backend_tensor_memset(simple_tensor, value, + i3*simple_tensor->nb[3] + i1*simple_tensor->nb[1], nbytes); + } + } + } + return; + } + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); if (split_state.n_segments != 1 || split_state.nr[0] != 1) { @@ -1416,6 +1461,37 @@ static void ggml_backend_meta_buffer_memset_tensor( static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + + // A host-resident attention cache reaches this permuted, as [head_dim, n_kv, n_head_kv, n_stream]. + // The heads split, but interleaved per cell, so the chunk splice below cannot express the write. + // Each device's heads are one contiguous run per cell: ne[1] cells, from one stride to another. + // A backend without a native 2d copy (CPU, Metal) pays one transfer per cell here, CUDA does not. + const bool strided_head_split = + !ggml_is_contiguous(tensor) && + split_state.axis == GGML_BACKEND_SPLIT_AXIS_2 && + split_state.n_segments == 1 && split_state.nr[0] == 1 && + tensor->nb[1] > tensor->nb[2] && + offset == 0 && size == ggml_nbytes(tensor); + + if (strided_head_split) { + for (int64_t i3 = 0; i3 < tensor->ne[3]; i3++) { + size_t offset_data = i3 * tensor->nb[3]; + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = split_state.ne[j] * tensor->nb[2]; + if (nbytes == 0) { + continue; + } + ggml_backend_tensor_set_2d(simple_tensor, (const char *) data + offset_data, + i3 * simple_tensor->nb[3], nbytes, + tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]); + offset_data += nbytes; + } + GGML_ASSERT(offset_data == i3 * tensor->nb[3] + (size_t) tensor->ne[2] * tensor->nb[2]); + } + return; + } + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); if (split_state.n_segments != 1 || split_state.nr[0] != 1) { @@ -1544,6 +1620,34 @@ static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, gg static void ggml_backend_meta_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + + // A fused QKV puts Kcur and Vcur in a strided view, which a host-resident cache reads back here. + // The rows split, so the chunk splice below cannot express the read. Each device's part of a row + // is contiguous: ne[1] rows, from the device's own stride to the view's. + // A backend without a native 2d copy (CPU, Metal) pays one transfer per row here, CUDA does not. + const bool strided_rows = + !ggml_is_contiguous(tensor) && + split_state.axis == GGML_BACKEND_SPLIT_AXIS_0 && + split_state.n_segments == 1 && split_state.nr[0] == 1 && + tensor->ne[2] == 1 && tensor->ne[3] == 1 && + offset == 0 && size == ggml_nbytes(tensor); + + if (strided_rows) { + size_t offset_data = 0; + for (size_t j = 0; j < n_bufs; j++) { + const ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = ggml_row_size(tensor->type, split_state.ne[j]); + if (nbytes == 0) { + continue; + } + ggml_backend_tensor_get_2d(simple_tensor, (char *) data + offset_data, 0, nbytes, + tensor->ne[1], simple_tensor->nb[1], tensor->nb[1]); + offset_data += nbytes; + } + GGML_ASSERT(offset_data == ggml_row_size(tensor->type, tensor->ne[0])); + return; + } + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); if (split_state.n_segments != 1 || split_state.nr[0] != 1) { @@ -2182,14 +2286,18 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend, int i_start = 0; for (int i = 0; i < cgraph->n_nodes; i++) { ggml_tensor * node = cgraph->nodes[i]; - if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) { - continue; - } - const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(node, /*assume_sync =*/ false); - if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) { - max_tmp_size = std::max(max_tmp_size, ggml_nbytes(node)); + // a host-resident KV cache ends a split with a view of itself, that view needs no split state + // but it is still the last node and must close the last subgraph + const bool host_view = node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && + ggml_backend_buffer_is_host(node->view_src->buffer); + bool new_subgraph = i + 1 == cgraph->n_nodes; + if (!host_view) { + const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(node, /*assume_sync =*/ false); + if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) { + max_tmp_size = std::max(max_tmp_size, ggml_nbytes(node)); + new_subgraph = true; + } } - const bool new_subgraph = i + 1 == cgraph->n_nodes || split_state.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL; if (!new_subgraph) { continue; } diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp index 8e7c9b08b329..bea33db60bc2 100644 --- a/ggml/src/ggml-backend.cpp +++ b/ggml/src/ggml-backend.cpp @@ -1034,6 +1034,57 @@ static void ggml_backend_sched_print_assignments(ggml_backend_sched_t sched, str } } +// The name field has a fixed size, so cut the backend label rather than the source name. +// See the contract at the declaration in ggml-backend-impl.h. +void ggml_backend_sched_name_copy( + struct ggml_tensor * copy, const char * backend_name, const struct ggml_tensor * src, int c) { + const int n_tail = snprintf(NULL, 0, "#%s#%d", src->name, c); + const int n_max = GGML_MAX_NAME - 1 - n_tail; + GGML_ASSERT(n_max >= 0 && "source name too long to name a scheduler copy"); + int n_head = (int) strlen(backend_name); + if (n_head > n_max) { + n_head = n_max; + } + ggml_format_name(copy, "%.*s#%s#%d", n_head, backend_name, src->name, c); +} + +bool ggml_backend_sched_copy_source_name(const char * name, char * buf, size_t buf_size) { + GGML_ASSERT(buf_size > 0); + + const char * first = strchr(name, '#'); + if (first == NULL) { + return false; + } + const char * src = first + 1; + size_t len = strlen(src); + + // ggml writes a view suffix as " (...)", a graph name has no spaces + const char * suffix = strstr(src, " ("); + if (suffix != NULL) { + len = suffix - src; + } else { + const char * copy = strrchr(src, '#'); + if (copy == NULL) { + return false; + } + const char * digits = copy + 1; + while (*digits >= '0' && *digits <= '9') { + digits++; + } + if (digits == copy + 1 || *digits != '\0') { + return false; + } + len = copy - src; + } + + if (len > buf_size - 1) { + return false; + } + memcpy(buf, src, len); + buf[len] = '\0'; + return true; +} + static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) { ggml_backend_buffer_t buf = t->view_src ? t->view_src->buffer : t->buffer; ggml_backend_buffer_type_t buft = NULL; @@ -1381,7 +1432,7 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra tensor_copy = src; // use the original tensor as the current copy } else { tensor_copy = ggml_dup_tensor_layout(sched->ctx, src); - ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c); + ggml_backend_sched_name_copy(tensor_copy, ggml_backend_name(backend), src, c); } ggml_set_input(tensor_copy); ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor @@ -1402,7 +1453,7 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra ggml_backend_t backend = sched->backends[cur_backend_id]; for (int c = 0; c < sched->n_copies; c++) { struct ggml_tensor * tensor_copy = ggml_dup_tensor_layout(sched->ctx, src); - ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c); + ggml_backend_sched_name_copy(tensor_copy, ggml_backend_name(backend), src, c); if (sched->n_copies > 1) { ggml_set_input(tensor_copy); ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor @@ -1812,7 +1863,8 @@ static enum ggml_status ggml_backend_sched_compute_splits(ggml_backend_sched_t s ggml_backend_buffer_t src_buf = input->view_src ? input->view_src->buffer : input->buffer; struct ggml_backend_sched_ranges rg; ggml_backend_sched_input_ranges(input, &rg); - const bool ranged = rg.n > 1 && src_buf != NULL && ggml_backend_buffer_is_host(src_buf); + // a meta backend writes a whole contiguous tensor, it cannot take one range per stream + const bool ranged = rg.n > 1 && src_buf != NULL && ggml_backend_buffer_is_host(src_buf) && !ggml_backend_is_meta(split_backend); // try async copy, but if not possible, we can still use a sync copy without synchronizing the dst backend, since we handle the synchronization here with multiple copies and events // TODO: add public function to facilitate this, since applications do not have direct access to the backend interface diff --git a/src/llama-context.cpp b/src/llama-context.cpp index 9aaa0f2a5cf6..c7ba2f400c08 100644 --- a/src/llama-context.cpp +++ b/src/llama-context.cpp @@ -144,6 +144,15 @@ llama_context::llama_context( cparams.offload_kqv = params.offload_kqv; cparams.kv_cpu_pinned = params.kv_cpu_pinned; cparams.recurrent_state_offload = params.recurrent_state_offload; + + // A linear-attention op packs the state it writes back together with its output, so that split + // does not line up with the one a host-resident state expects. Keep the state on its device. + if (!cparams.recurrent_state_offload && model.split_mode() == LLAMA_SPLIT_MODE_TENSOR && + (llm_arch_is_recurrent(model.arch) || llm_arch_is_hybrid(model.arch))) { + LLAMA_LOG_WARN("%s: split mode tensor cannot keep the recurrent state in host memory - " + "overriding --no-recurrent-state-offload, which needs more VRAM\n", __func__); + cparams.recurrent_state_offload = true; + } cparams.offload_attn_compute = params.offload_kqv || (params.op_offload && params.kv_cpu_pinned); cparams.kv_gpu_layers = params.kv_gpu_layers; cparams.phase_aware_workspace = params.phase_aware_workspace; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index a65abfc9ca64..66ae3b617fb6 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -382,8 +382,8 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str static const std::regex pattern_kv_bias ("blk\\.\\d*\\.attn_(k|v)\\.bias"); static const std::regex pattern_qkv_bias ("blk\\.\\d*\\.attn_qkv.bias"); static const std::regex pattern_qk_norm ("blk\\.\\d*\\.attn_(q|k)_norm\\.weight"); - static const std::regex pattern_kv_cache ("cache_(k|v)_l\\d*"); - static const std::regex pattern_idx_cache ("cache_idx_(k|v)_l\\d*"); + static const std::regex pattern_kv_cache ("cache_(k|v)_l\\d+"); + static const std::regex pattern_idx_cache ("cache_idx_(k|v)_l\\d+"); static const std::regex pattern_dsv4_state ("dsv4_(csa|hca|lid)_state_(kv|score)_l\\d*"); static const std::regex pattern_attn_sinks ("blk\\.\\d*\\.attn_sinks.weight"); static const std::regex pattern_attn_out_weight ("blk\\.\\d*\\.attn_output.weight"); @@ -398,9 +398,9 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str static const std::regex pattern_ssm_alpha ("blk\\.\\d*\\.ssm_alpha.weight"); static const std::regex pattern_ssm_beta ("blk\\.\\d*\\.ssm_beta.weight"); static const std::regex pattern_ssm_beta_alpha ("blk\\.\\d*\\.ssm_ba.weight"); - static const std::regex pattern_r_cache ("cache_r_l\\d*"); - static const std::regex pattern_ple_r_cache ("cache_ple_r_l\\d*"); - static const std::regex pattern_s_cache ("cache_s_l\\d*"); + static const std::regex pattern_r_cache ("cache_r_l\\d+"); + static const std::regex pattern_ple_r_cache ("cache_ple_r_l\\d+"); + static const std::regex pattern_s_cache ("cache_s_l\\d+"); static const std::regex pattern_ssm_conv1d ("blk\\.\\d*\\.ssm_conv1d.weight"); static const std::regex pattern_ssm_out_weight ("blk\\.\\d*\\.ssm_out.weight"); @@ -470,7 +470,34 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return {axis, tensor_axis_0, il, rotation}; }; + // A host-resident cache reaches attention as a scheduler copy, permuted, with the heads on an + // axis of their own. The split follows the cache, so record that axis and its unit - one head. + struct host_cache_copy { + bool valid = false; + ggml_backend_meta_split_axis axis = GGML_BACKEND_SPLIT_AXIS_0; + int64_t unit = 1; // cache elements per element of axis + }; + + const host_cache_copy cache_copy = [&]() -> host_cache_copy { + if (is_dsv4 || !std::regex_match(tensor_name, pattern_kv_cache)) { + return {}; + } + // [head_dim, n_kv, n_head_kv, n_stream] + const uint32_t il = std::stoul(tensor_name.substr(tensor_name.find("_l", 6) + 2)); + const int64_t head_dim = tensor_name[6] == 'k' ? hparams.n_embd_head_k(il) : hparams.n_embd_head_v(il); + if (hparams.n_head_kv(il) > 1 && tensor->ne[0] == head_dim && + tensor->ne[2] == (int64_t) hparams.n_head_kv(il)) { + return {true, GGML_BACKEND_SPLIT_AXIS_2, head_dim}; + } + return {}; + }(); + auto get_tensor_config = [&]() -> tensor_config { + // a graph tensor only reaches here as a scheduler copy, and only a copy of a host-resident + // cache is split - everything else the graph copies in keeps the mirrored fallback + if (!cache_copy.valid && ggml_backend_buffer_get_usage(tensor->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE) { + return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_MIRRORED); + } if (is_dsv4) { if (std::regex_match(tensor_name, pattern_kv_cache) || std::regex_match(tensor_name, pattern_dsv4_state)) { @@ -594,7 +621,9 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_MIRRORED); }; - auto get_split_segments = [&](int axis, uint32_t il) -> std::vector> { + // ne_axis is the extent of the split axis as the cache sees it, which for a host-resident + // cache copy is not the extent of the axis the copy is split on + auto get_split_segments = [&](int axis, uint32_t il, int64_t ne_axis) -> std::vector> { if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_QWEN4EXP) { const int64_t head_k_dim = hparams.ssm_d_state; @@ -642,7 +671,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str GGML_ASSERT(tensor->ne[axis] == 2*n_ff_exp); return {{n_ff_exp, 2}}; } - return {{tensor->ne[axis], 1}}; + return {{ne_axis, 1}}; } if (std::regex_match(tensor_name, pattern_qkv_weight) || std::regex_match(tensor_name, pattern_qkv_bias)) { @@ -658,14 +687,14 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str if (tensor->ne[axis] == 2*n_ff) { return {{n_ff, 2}}; } - return {{tensor->ne[axis], 1}}; + return {{ne_axis, 1}}; } if (std::regex_match(tensor_name, pattern_ffn_gate_up_weight)) { const int64_t n_ff_exp = hparams.n_ff_exp(il); GGML_ASSERT(tensor->ne[axis] == 2*n_ff_exp); return {{n_ff_exp, 2}}; } - return {{tensor->ne[axis], 1}}; + return {{ne_axis, 1}}; }; auto get_split_granularity = [&](int64_t blck_size, uint32_t il, const std::vector> & segments) -> std::vector { @@ -752,12 +781,16 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return {granularity_q}; } - const int64_t granularity_kv = granularity_q / n_gqa; + const int64_t n_head_gran = granularity_q / n_embd_q; // KV heads per granule + const int64_t granularity_kv = n_head_gran * hparams.n_embd_head_k(il); if (std::regex_match(tensor_name, pattern_kv_weight) || std::regex_match(tensor_name, pattern_kv_bias) || std::regex_match(tensor_name, pattern_kv_cache)) { GGML_ASSERT(segments.size() == 1); - return {granularity_kv}; + // the V side can have a head size of its own, the split still lands on head boundaries + const bool is_v = tensor_name.compare(0, 8, "cache_v_") == 0 || + tensor_name.find(".attn_v.") != std::string::npos; + return {is_v ? n_head_gran * hparams.n_embd_head_v(il) : granularity_kv}; } if (std::regex_match(tensor_name, pattern_qkv_weight) || std::regex_match(tensor_name, pattern_qkv_bias)) { GGML_ASSERT(segments.size() == 2); @@ -788,6 +821,11 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str tensor_config tc = get_tensor_config(); split_state.axis = tc.axis; if (split_state.axis >= 0 && split_state.axis < GGML_MAX_DIMS) { + // a host-resident cache copy is split on the axis its heads arrive on, in whole heads, + // so the segments and the granularity of the cache are rescaled to that unit + const bool unfolded = cache_copy.valid && split_state.axis == GGML_BACKEND_SPLIT_AXIS_0; + const int64_t unit = unfolded ? cache_copy.unit : 1; + const int64_t ne_axis = unfolded ? tensor->ne[cache_copy.axis]*unit : tensor->ne[split_state.axis]; const int64_t blck_size = ggml_blck_size(tc.tensor_axis_0->type); const float * tensor_split = ud->model->tensor_split(); std::vector tensor_split_scan; @@ -798,12 +836,17 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str tensor_split_scan[j] += tensor_split_scan[j - 1]; } } - const std::vector> segments = get_split_segments(split_state.axis, tc.il); + const std::vector> segments = get_split_segments(split_state.axis, tc.il, ne_axis); const std::vector granularity = get_split_granularity(blck_size, tc.il, segments); + if (unfolded) { + split_state.axis = cache_copy.axis; + } for (size_t is = 0; is < segments.size(); is++) { - const int64_t ne_s = segments[is].first; + GGML_ASSERT(segments[is].first % unit == 0); + GGML_ASSERT(granularity[is] % unit == 0); + const int64_t ne_s = segments[is].first / unit; const uint32_t nr_s = segments[is].second; - const int64_t g_s = granularity[is]; + const int64_t g_s = granularity[is] / unit; int64_t low = 0; size_t j = 0; for (; j < ud->n_devices - 1; j++) { diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 04693bb0e6ed..aa42167bf30e 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -214,6 +214,13 @@ if (NOT WIN32 OR NOT BUILD_SHARED_LIBS) ARGS --test-live-context-workspace ) + llama_test( + test-llama-archs + NAME test-sched-copy-name + LABEL main + ARGS --test-sched-copy-name + ) + set(MODEL_DIR "${CMAKE_CURRENT_BINARY_DIR}/test-models/") file(MAKE_DIRECTORY "${MODEL_DIR}") diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index 3951dbbe8526..cf193bf168e6 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -69,7 +70,7 @@ static void set_tensor_data(struct ggml_tensor * tensor, void * userdata) { } static void usage(char ** argv) { - printf("Usage: %s [-a/--arch arch] [-s/--seed seed] [-o/--out dir] [-v N] [-h/--help] [--test-phase-workspace] [--test-live-context-workspace]\n", argv[0]); + printf("Usage: %s [-a/--arch arch] [-s/--seed seed] [-o/--out dir] [-v N] [-h/--help] [--test-phase-workspace] [--test-live-context-workspace] [--test-sched-copy-name]\n", argv[0]); } static std::vector get_tokens(const uint32_t n_tokens, const uint32_t n_vocab, const size_t seed){ @@ -398,9 +399,16 @@ static bool silent_model_load_progress(float /*progress*/, void * /*user_data*/) return true; } +// with offload_kqv=false the cache lives in host memory +// n_seq_max > 1 gives the cache one stream per sequence +struct kv_config { + bool offload_kqv = true; + uint32_t n_seq_max = 1; +}; + static std::pair get_model_and_ctx( struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const std::vector & devs, - const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false) { + const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false, const kv_config & kvc = {}) { GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr)); llama_model_params model_params = llama_model_default_params(); model_params.progress_callback = silent_model_load_progress; @@ -413,6 +421,8 @@ static std::pair get_model_and_ctx( ctx_params.n_ctx = 0; ctx_params.n_threads = 4; ctx_params.n_threads_batch = 4; + ctx_params.offload_kqv = kvc.offload_kqv; + ctx_params.n_seq_max = kvc.n_seq_max; if (!encode) { ctx_params.n_ubatch = 64; } @@ -1052,15 +1062,51 @@ static void test_phase_workspace_mismatched_placement(size_t seed) { GGML_ASSERT(llama_contexts_share_workspace(target.get(), draft.get()) == (status == 1)); } +// the meta backend recovers the source of a scheduler copy from its name, so both sides must agree +static void test_sched_copy_name() { + const ggml_init_params params = { + /*.mem_size =*/ 2*ggml_tensor_overhead(), + /*.mem_buffer =*/ nullptr, + /*.no_alloc =*/ true, + }; + ggml_context_ptr ctx(ggml_init(params)); + GGML_ASSERT(ctx); + ggml_tensor * src = ggml_new_tensor_1d(ctx.get(), GGML_TYPE_F32, 1); + ggml_tensor * copy = ggml_new_tensor_1d(ctx.get(), GGML_TYPE_F32, 1); + + auto check = [&](const char * name, const char * backend_name, const char * expected) { + char buf[GGML_MAX_NAME]; + ggml_set_name(src, name); + ggml_backend_sched_name_copy(copy, backend_name, src, 0); + GGML_ASSERT(ggml_backend_sched_copy_source_name(copy->name, buf, sizeof(buf))); + GGML_ASSERT(strcmp(buf, expected) == 0); + }; + + check("cache_k_l0", "CUDA0", "cache_k_l0"); + check("cache_k_l0 (view)", "CUDA0", "cache_k_l0"); + // the backend label is cut when the name does not fit, the source name survives + check("cache_k_l31", "Meta(CUDA0,CUDA1,CUDA2,CUDA3,CUDA4,CUDA5,CUDA6,CUDA7)", "cache_k_l31"); + + // a name that was not written by the scheduler is not a copy + char buf[GGML_MAX_NAME]; + GGML_ASSERT(!ggml_backend_sched_copy_source_name("cache_k_l0", buf, sizeof(buf))); + GGML_ASSERT(!ggml_backend_sched_copy_source_name("CUDA0#cache_k_l0", buf, sizeof(buf))); + // a cut name lost its copy index, what is left of the source names another tensor + GGML_ASSERT(!ggml_backend_sched_copy_source_name("#cache_k_l3", buf, sizeof(buf))); +} + +// the tokens are spread over n_seq sequences, each of which starts at position 0 static std::vector get_logits( - llama_model * model, llama_context * lctx, const std::vector & tokens, bool encode = false) { + llama_model * model, llama_context * lctx, const std::vector & tokens, bool encode = false, uint32_t n_seq = 1) { const uint32_t n_vocab = llama_vocab_n_tokens(llama_model_get_vocab(model)); const uint32_t n_ctx = llama_n_ctx(lctx); const uint32_t n_tokens = tokens.size(); - llama_batch batch = llama_batch_init(n_ctx, 0, 1); + GGML_ASSERT(n_seq >= 1 && n_tokens % n_seq == 0); + const uint32_t n_tokens_seq = n_tokens / n_seq; + llama_batch batch = llama_batch_init(n_ctx, 0, n_seq); GGML_ASSERT(n_tokens <= n_ctx); for (uint32_t pos = 0; pos < n_tokens; pos++) { - common_batch_add(batch, tokens[pos], pos, {0}, true); + common_batch_add(batch, tokens[pos], pos % n_tokens_seq, {llama_seq_id(pos / n_tokens_seq)}, true); } batch.n_tokens = n_tokens; if (encode) { @@ -1088,7 +1134,7 @@ static std::vector get_logits( // decode two sequences in one batch, compare each with a decode of it alone // logits_a: logits of tokens decoded alone with the same device config -static bool test_parallel_seqs(llama_model * model, const std::vector & tokens, const std::vector & logits_a, bool encode) { +static bool test_parallel_seqs(llama_model * model, const std::vector & tokens, const std::vector & logits_a, bool encode, bool offload_kqv) { const uint32_t n_vocab = llama_vocab_n_tokens(llama_model_get_vocab(model)); const uint32_t n_tokens = tokens.size(); @@ -1099,6 +1145,7 @@ static bool test_parallel_seqs(llama_model * model, const std::vector devs; std::string label; llama_split_mode split_mode; + kv_config kvc; - device_config(std::vector devs, std::string name, llama_split_mode split_mode) - : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode) {} + device_config(std::vector devs, std::string name, llama_split_mode split_mode, kv_config kvc = {}) + : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), kvc(std::move(kvc)) {} }; std::vector dev_configs; @@ -1391,7 +1439,6 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in for (size_t i = 0; i < device_count; i++) { ggml_backend_dev_t dev = ggml_backend_dev_get(i); dev_configs.emplace_back(std::vector{dev}, ggml_backend_dev_description(dev), LLAMA_SPLIT_MODE_LAYER); - max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length()); // cpu-based devices cannot be used in tensor split mode if (ggml_backend_dev_buffer_type(dev) != ggml_backend_cpu_buffer_type()) { @@ -1401,6 +1448,20 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in } dev_configs.emplace_back(devices_meta, "Meta", LLAMA_SPLIT_MODE_TENSOR); + + // a host-resident cache reaches attention as a scheduler copy that is split by head + kv_config kvc_host; + kvc_host.offload_kqv = false; + dev_configs.emplace_back(devices_meta, "Meta -nkvo", LLAMA_SPLIT_MODE_TENSOR, kvc_host); + + // with more than one stream that copy is 4d, one stride per stream + kv_config kvc_host_streams = kvc_host; + kvc_host_streams.n_seq_max = 2; + dev_configs.emplace_back(devices_meta, "Meta -nkvo -np 2", LLAMA_SPLIT_MODE_TENSOR, kvc_host_streams); + + for (const device_config & dc : dev_configs) { + max_device_label_length = std::max(max_device_label_length, dc.label.length()); + } } size_t max_arch_name_length = 0; @@ -1432,7 +1493,10 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in continue; } if (arch == LLM_ARCH_GEMMA4 || arch == LLM_ARCH_GEMMA4_ASSISTANT) { - continue; // FIXME: ISWA KV cache initialization needs more fixture params + // FIXME: ISWA KV cache initialization needs more fixture params + // this arch also loses accuracy with a host-resident cache under split mode tensor + // (perplexity 235.03 versus 227.23), the other ISWA archs are clean + continue; } if (arch == LLM_ARCH_EAGLE3 || arch == LLM_ARCH_DFLASH) { continue; @@ -1452,7 +1516,7 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in GGML_ASSERT(gguf_remove_key(gguf_ctx.get(), "bailingmoe3.kda.safe_gate") >= 0); } std::pair model_and_ctx_cpu; - std::vector logits_cpu; + std::map> logits_cpu_per_n_seq; for (device_config & dc : dev_configs) { // print test config first; should anything fail during model loading or inference, at least we know which test case caused it printf(template_row_cfg.c_str(), @@ -1466,15 +1530,21 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in std::string status_parallel = "\033[1;33mSKIP\033[0m"; char nmse_str[12] = {0}; - bool skip = !arch_supported(arch) || (dc.split_mode == LLAMA_SPLIT_MODE_TENSOR && dc.devs.empty()); + // an encoder-decoder model needs its own batch layout, so it stays on one sequence + bool skip = !arch_supported(arch) || (dc.split_mode == LLAMA_SPLIT_MODE_TENSOR && dc.devs.empty()) || + (encode && dc.kvc.n_seq_max > 1); if (!skip) { - if (logits_cpu.empty()) { - model_and_ctx_cpu = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {}, LLAMA_SPLIT_MODE_LAYER, encode); - logits_cpu = get_logits(model_and_ctx_cpu.first.get(), model_and_ctx_cpu.second.get(), tokens, encode); - } if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) { - model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, dc.devs, dc.split_mode, encode); - logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode); + // the reference runs the same batch layout, one sequence per stream + std::vector & logits_cpu = logits_cpu_per_n_seq[dc.kvc.n_seq_max]; + if (logits_cpu.empty()) { + kv_config kvc_cpu; + kvc_cpu.n_seq_max = dc.kvc.n_seq_max; + model_and_ctx_cpu = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {}, LLAMA_SPLIT_MODE_LAYER, encode, kvc_cpu); + logits_cpu = get_logits(model_and_ctx_cpu.first.get(), model_and_ctx_cpu.second.get(), tokens, encode, dc.kvc.n_seq_max); + } + model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, dc.devs, dc.split_mode, encode, dc.kvc); + logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode, dc.kvc.n_seq_max); const double nmse_val = nmse(logits_cpu, logits_dev); snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val); status_nmse = "\033[1;32mOK\033[0m"; @@ -1484,9 +1554,10 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in } // FIXME: T5 kq_b does not broadcast over KV streams, so context init with n_seq_max > 1 aborts - if (arch != LLM_ARCH_T5) { + // a multi-stream row spreads the reference over several sequences, so it does not match one sequence at pos 0..n-1 + if (arch != LLM_ARCH_T5 && dc.kvc.n_seq_max == 1) { status_parallel = "\033[1;32mOK\033[0m"; - if (!test_parallel_seqs(model_and_ctx_dev.first.get(), tokens, logits_dev, encode)) { + if (!test_parallel_seqs(model_and_ctx_dev.first.get(), tokens, logits_dev, encode, dc.kvc.offload_kqv)) { all_ok = false; status_parallel = "\033[1;31mFAIL\033[0m"; } @@ -1504,9 +1575,9 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const in ms.save(file); rewind(file); - auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, dc.devs, dc.split_mode, encode); + auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, dc.devs, dc.split_mode, encode, dc.kvc); const std::vector logits_roundtrip = get_logits( - model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode); + model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode, dc.kvc.n_seq_max); status_roundtrip = "\033[1;32mOK\033[0m"; GGML_ASSERT(logits_roundtrip.size() == logits_dev.size()); for (size_t i = 0; i < logits_roundtrip.size(); i++) { @@ -1541,6 +1612,7 @@ int main(int argc, char ** argv) { std::string out; bool test_phase_workspace = false; bool test_live_context_workspace = false; + bool test_copy_name = false; int verbosity = LOG_LEVEL_ERROR; @@ -1594,6 +1666,10 @@ int main(int argc, char ** argv) { test_live_context_workspace = true; continue; } + if (strcmp(argv[i], "--test-sched-copy-name") == 0) { + test_copy_name = true; + continue; + } } printf("%s: using seed %zu\n", __func__, seed); @@ -1611,6 +1687,10 @@ int main(int argc, char ** argv) { test_live_context_workspace_unsupported(seed); return 0; } + if (test_copy_name) { + test_sched_copy_name(); + return 0; + } if (!out.empty()) { return save_models(arch, seed, verbosity, out); }