From b0aa2cfe27236891ce276e873a06615bcd8aaf84 Mon Sep 17 00:00:00 2001 From: thecodacus Date: Mon, 28 Sep 2026 00:35:17 +0530 Subject: [PATCH 1/3] fix: place each MoE expert-cache pack on the GPU that runs its layer init_moe_expert_cache picked the first GPU and allocated every layer's hot expert pack there. With a layer split across two GPUs that fills GPU0 while GPU1 sits partly empty, and layers on GPU1 bounce their activations to GPU0 and back for every cached matmul. Group the CPU-resident MoE layers by dev_layer(il) and build one pack context and buffer per GPU. Layers that are not on a GPU still use the first GPU, so single-GPU behaviour is unchanged. A failed pack allocation now disables the cache only for that device's layers and says which device failed, instead of silently turning the whole cache off. Co-Authored-By: Claude Opus 5.5 --- src/llama-model.cpp | 179 +++++++++++++++++++++++--------------------- 1 file changed, 94 insertions(+), 85 deletions(-) diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 90afd38a7eaf..3e09e0b2406a 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1887,117 +1887,126 @@ void llama_model_base::init_moe_expert_cache() { return; } - ggml_backend_dev_t dev = nullptr; + ggml_backend_dev_t dev_first = nullptr; for (const auto & d : devices) { - if (!d.is_meta) { dev = d.dev; break; } + if (!d.is_meta) { dev_first = d.dev; break; } } - if (dev == nullptr || ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU) { + if (dev_first == nullptr || ggml_backend_dev_type(dev_first) != GGML_BACKEND_DEVICE_TYPE_GPU) { LLAMA_LOG_WARN("%s: no GPU device - expert cache disabled\n", __func__); return; } - ggml_backend_buffer_type_t buft = ggml_backend_dev_buffer_type(dev); - // candidate layers: routed experts resident in host memory - std::vector pack_layers; + // candidate layers: routed experts resident in host memory, grouped by the GPU that runs the layer so the + // hot experts sit on the same device as the rest of the layer (layers left on the CPU use the first GPU) + std::map> pack_layers_by_dev; for (int il = 0; il < (int) layers.size(); il++) { const auto & l = layers[il]; if (l.ffn_gate_exps && l.ffn_up_exps && l.ffn_down_exps && freq.count(il) && l.ffn_gate_exps->buffer && ggml_backend_buft_is_host(ggml_backend_buffer_get_type(l.ffn_gate_exps->buffer))) { - pack_layers.push_back(il); + ggml_backend_dev_t dev = dev_layer(il); + if (ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU) { + dev = dev_first; + } + pack_layers_by_dev[dev].push_back(il); } } - if (pack_layers.empty()) { + if (pack_layers_by_dev.empty()) { LLAMA_LOG_INFO("%s: no CPU-resident MoE layers - expert cache not built\n", __func__); return; } - ggml_init_params ctx_params = { - /*.mem_size =*/ (5*pack_layers.size() + 1)*ggml_tensor_overhead(), - /*.mem_buffer =*/ nullptr, - /*.no_alloc =*/ true, - }; - ggml_context * ctx = ggml_init(ctx_params); - - for (int il : pack_layers) { - auto & l = layers[il]; - const ggml_tensor * g = l.ffn_gate_exps; - const ggml_tensor * u = l.ffn_up_exps; - const ggml_tensor * d = l.ffn_down_exps; - const int64_t n_expert = g->ne[2]; - const int64_t S = std::min(n_slots, n_expert); - l.ffn_gate_exps_hot = ggml_new_tensor_3d(ctx, g->type, g->ne[0], g->ne[1], S); - l.ffn_up_exps_hot = ggml_new_tensor_3d(ctx, u->type, u->ne[0], u->ne[1], S); - l.ffn_down_exps_hot = ggml_new_tensor_3d(ctx, d->type, d->ne[0], d->ne[1], S); - l.moe_map_hot = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); - l.moe_map_cold = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); - ggml_format_name(l.ffn_gate_exps_hot, "blk.%d.ffn_gate_exps_hot", il); - ggml_format_name(l.ffn_up_exps_hot, "blk.%d.ffn_up_exps_hot", il); - ggml_format_name(l.ffn_down_exps_hot, "blk.%d.ffn_down_exps_hot", il); - ggml_format_name(l.moe_map_hot, "blk.%d.moe_map_hot", il); - ggml_format_name(l.moe_map_cold, "blk.%d.moe_map_cold", il); - } - - ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, buft); - if (buf == nullptr) { - LLAMA_LOG_WARN("%s: pack allocation failed - expert cache disabled\n", __func__); - ggml_free(ctx); - for (int il : pack_layers) { - auto & l = layers[il]; - l.ffn_gate_exps_hot = l.ffn_up_exps_hot = l.ffn_down_exps_hot = nullptr; - l.moe_map_hot = l.moe_map_cold = nullptr; - } - return; - } - // weights usage pins the pack tensors to their backend during graph - // assignment - without it a CPU-assigned consumer can drag the hot - // matmuls (and a per-layer weight copy) onto the CPU - ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); - - // fill packs (expert dim is outermost: one contiguous slab per expert) std::vector slab; std::vector map_hot, map_cold; - size_t total_bytes = 0; - for (int il : pack_layers) { - auto & l = layers[il]; - const int64_t n_expert = l.ffn_gate_exps->ne[2]; - const int64_t S = l.ffn_gate_exps_hot->ne[2]; + for (const auto & [dev, pack_layers] : pack_layers_by_dev) { + ggml_backend_buffer_type_t buft = ggml_backend_dev_buffer_type(dev); - std::vector> ranked; // (-count, expert) - for (const auto & [e, c] : freq[il]) { - if (e < n_expert) { - ranked.push_back({-c, e}); + ggml_init_params ctx_params = { + /*.mem_size =*/ (5*pack_layers.size() + 1)*ggml_tensor_overhead(), + /*.mem_buffer =*/ nullptr, + /*.no_alloc =*/ true, + }; + ggml_context * ctx = ggml_init(ctx_params); + + for (int il : pack_layers) { + auto & l = layers[il]; + const ggml_tensor * g = l.ffn_gate_exps; + const ggml_tensor * u = l.ffn_up_exps; + const ggml_tensor * d = l.ffn_down_exps; + const int64_t n_expert = g->ne[2]; + const int64_t S = std::min(n_slots, n_expert); + l.ffn_gate_exps_hot = ggml_new_tensor_3d(ctx, g->type, g->ne[0], g->ne[1], S); + l.ffn_up_exps_hot = ggml_new_tensor_3d(ctx, u->type, u->ne[0], u->ne[1], S); + l.ffn_down_exps_hot = ggml_new_tensor_3d(ctx, d->type, d->ne[0], d->ne[1], S); + l.moe_map_hot = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); + l.moe_map_cold = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); + ggml_format_name(l.ffn_gate_exps_hot, "blk.%d.ffn_gate_exps_hot", il); + ggml_format_name(l.ffn_up_exps_hot, "blk.%d.ffn_up_exps_hot", il); + ggml_format_name(l.ffn_down_exps_hot, "blk.%d.ffn_down_exps_hot", il); + ggml_format_name(l.moe_map_hot, "blk.%d.moe_map_hot", il); + ggml_format_name(l.moe_map_cold, "blk.%d.moe_map_cold", il); + } + + ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, buft); + if (buf == nullptr) { + LLAMA_LOG_WARN("%s: pack allocation failed on %s - expert cache disabled for its %zu layers\n", + __func__, ggml_backend_buft_name(buft), pack_layers.size()); + ggml_free(ctx); + for (int il : pack_layers) { + auto & l = layers[il]; + l.ffn_gate_exps_hot = l.ffn_up_exps_hot = l.ffn_down_exps_hot = nullptr; + l.moe_map_hot = l.moe_map_cold = nullptr; } + continue; } - std::sort(ranked.begin(), ranked.end()); + // weights usage pins the pack tensors to their backend during graph + // assignment - without it a CPU-assigned consumer can drag the hot + // matmuls (and a per-layer weight copy) onto the CPU + ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); - map_hot.assign(n_expert, -1); - map_cold.resize(n_expert); - for (int64_t e = 0; e < n_expert; e++) { - map_cold[e] = (int32_t) e; - } - for (int64_t s = 0; s < S && s < (int64_t) ranked.size(); s++) { - const int32_t e = ranked[s].second; - map_hot[e] = (int32_t) s; - map_cold[e] = -1; - const ggml_tensor * srcs[3] = { l.ffn_gate_exps, l.ffn_up_exps, l.ffn_down_exps }; - ggml_tensor * dsts[3] = { l.ffn_gate_exps_hot, l.ffn_up_exps_hot, l.ffn_down_exps_hot }; - for (int t = 0; t < 3; t++) { - const size_t nb = srcs[t]->nb[2]; - slab.resize(nb); - ggml_backend_tensor_get(srcs[t], slab.data(), e*nb, nb); - ggml_backend_tensor_set(dsts[t], slab.data(), s*nb, nb); - total_bytes += nb; + // fill packs (expert dim is outermost: one contiguous slab per expert) + size_t total_bytes = 0; + for (int il : pack_layers) { + auto & l = layers[il]; + const int64_t n_expert = l.ffn_gate_exps->ne[2]; + const int64_t S = l.ffn_gate_exps_hot->ne[2]; + + std::vector> ranked; // (-count, expert) + for (const auto & [e, c] : freq[il]) { + if (e < n_expert) { + ranked.push_back({-c, e}); + } + } + std::sort(ranked.begin(), ranked.end()); + + map_hot.assign(n_expert, -1); + map_cold.resize(n_expert); + for (int64_t e = 0; e < n_expert; e++) { + map_cold[e] = (int32_t) e; + } + for (int64_t s = 0; s < S && s < (int64_t) ranked.size(); s++) { + const int32_t e = ranked[s].second; + map_hot[e] = (int32_t) s; + map_cold[e] = -1; + const ggml_tensor * srcs[3] = { l.ffn_gate_exps, l.ffn_up_exps, l.ffn_down_exps }; + ggml_tensor * dsts[3] = { l.ffn_gate_exps_hot, l.ffn_up_exps_hot, l.ffn_down_exps_hot }; + for (int t = 0; t < 3; t++) { + const size_t nb = srcs[t]->nb[2]; + slab.resize(nb); + ggml_backend_tensor_get(srcs[t], slab.data(), e*nb, nb); + ggml_backend_tensor_set(dsts[t], slab.data(), s*nb, nb); + total_bytes += nb; + } } + ggml_backend_tensor_set(l.moe_map_hot, map_hot.data(), 0, n_expert*sizeof(int32_t)); + ggml_backend_tensor_set(l.moe_map_cold, map_cold.data(), 0, n_expert*sizeof(int32_t)); } - ggml_backend_tensor_set(l.moe_map_hot, map_hot.data(), 0, n_expert*sizeof(int32_t)); - ggml_backend_tensor_set(l.moe_map_cold, map_cold.data(), 0, n_expert*sizeof(int32_t)); - } - pimpl->ctxs_bufs.emplace_back(ggml_context_ptr{ctx}, std::vector{}); - pimpl->ctxs_bufs.back().second.emplace_back(buf); + pimpl->ctxs_bufs.emplace_back(ggml_context_ptr{ctx}, std::vector{}); + pimpl->ctxs_bufs.back().second.emplace_back(buf); - LLAMA_LOG_INFO("%s: expert cache: %zu layers x %d slots, %.2f MiB uploaded to %s\n", - __func__, pack_layers.size(), n_slots, total_bytes/1024.0/1024.0, ggml_backend_buft_name(buft)); + LLAMA_LOG_INFO("%s: expert cache: %zu layers x %d slots, %.2f MiB uploaded to %s\n", + __func__, pack_layers.size(), n_slots, total_bytes/1024.0/1024.0, ggml_backend_buft_name(buft)); + } } ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM_TN_IMPL & tn, const std::initializer_list & ne, int flags) { From 0ed492f66e62187a64970b81f27c6c89d291587f Mon Sep 17 00:00:00 2001 From: thecodacus Date: Mon, 28 Sep 2026 02:27:14 +0530 Subject: [PATCH 2/3] feat(moe-trace): sample with the command-line sampling flags instead of greedy The trace tool always decoded greedily, so a routing profile recorded with it followed a different expert path than a server running with temperature, top-k or top-p. Use common_sampler with params.sampling so --temp, --top-k, --top-p, --min-p, --seed and the penalties apply to the traced decode. Pass --temp 0 to get the old greedy behaviour. Co-Authored-By: Claude Opus 5.5 --- tools/moe-trace/moe-trace.cpp | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/tools/moe-trace/moe-trace.cpp b/tools/moe-trace/moe-trace.cpp index 032d49d977f9..29636ff9a9bb 100644 --- a/tools/moe-trace/moe-trace.cpp +++ b/tools/moe-trace/moe-trace.cpp @@ -9,12 +9,13 @@ // separate prefill routing from decode routing. // // Usage: -// MOE_TRACE_OUT=trace.csv llama-moe-trace -m model.gguf -ngl 99 -ncmoe 26 -fa on \ +// MOE_TRACE_OUT=trace.csv llama-moe-trace -m model.gguf -ngl 99 -ncmoe 26 -fa on --temp 0 \ // -p "prompt text" -n 512 #include "arg.h" #include "common.h" #include "log.h" +#include "sampling.h" #include "llama.h" #include @@ -123,12 +124,14 @@ int main(int argc, char ** argv) { tc.pos += n_eval; } - // greedy decode + // decode with the sampling flags (--temp, --top-k, --top-p, --min-p, --seed, ...) so the trace + // follows the same expert routing a server with those defaults would see tc.in_prompt = false; - llama_sampler * smpl = llama_sampler_init_greedy(); + common_sampler * smpl = common_sampler_init(model, params.sampling); llama_token tok = 0; for (int i = 0; i < params.n_predict; i++) { - tok = llama_sampler_sample(smpl, lctx, -1); + tok = common_sampler_sample(smpl, lctx, -1); + common_sampler_accept(smpl, tok, true); if (llama_vocab_is_eog(vocab, tok)) { break; } @@ -141,7 +144,7 @@ int main(int argc, char ** argv) { LOG_INF("decoded %d/%d\n", i, params.n_predict); } } - llama_sampler_free(smpl); + common_sampler_free(smpl); fclose(tc.out); LOG_INF("trace written to %s\n", out_path); From f201e552799760e5286f6fe12f3e803105cd8313 Mon Sep 17 00:00:00 2001 From: thecodacus Date: Mon, 28 Sep 2026 02:30:17 +0530 Subject: [PATCH 3/3] docs: restructure README around a quick start for the fork Lead with what the fork adds and a five-step quick start (build, baseline, record a profile, serve with the cache, pick a slot count), then recipes for MTP, two GPUs and preset INIs. New sections cover recording profiles with real sampling settings and chat-templated long prompts, and per-GPU cache placement. The allocation-failure text now matches the log, and the claim that the warning prints the fit math is gone (it never did). The upstream README follows unchanged under its own heading. Co-Authored-By: Claude Opus 5.5 --- README.md | 282 +++++++++++++++++++++++++++++++++++++----------------- 1 file changed, 192 insertions(+), 90 deletions(-) diff --git a/README.md b/README.md index d66ee10c2022..3bdc8871336b 100644 --- a/README.md +++ b/README.md @@ -1,152 +1,254 @@ -# llama.cpp +# llama.cpp — `perf` fork + +Faster **mixture-of-experts (MoE) models on consumer GPUs** when the experts don't fit in VRAM. +This is [llama.cpp](https://github.com/ggml-org/llama.cpp) plus a handful of opt-in changes for the +`--n-cpu-moe` case (experts in system RAM, the rest on the GPU). Everything is **off by default** +and produces the **same tokens** as upstream. + +| Feature | What it does | How to turn it on | +| --- | --- | --- | +| **MoE expert cache** | Keeps each layer's most-used experts resident in VRAM; the rest stay in RAM. With several GPUs, each GPU caches the experts of its own layers. | `--moe-cache-profile FILE --moe-cache-slots N` | +| **Routing profiles** | `llama-moe-trace` records which experts a model picks, with your real sampling settings, so the cache knows what to keep. | `llama-moe-trace` (see below) | +| **Async CPU splits** | CPU and GPU halves of each MoE layer run at the same time. | on by default with the cache (`--no-sched-async-cpu` to disable) | +| **Prefill speedups** | Pins offloaded expert memory and prefetches it on a second CUDA stream. | `GGML_CUDA_REGISTER_HOST=1 GGML_SCHED_PREFETCH_EXPERTS=1` | +| **Qwen3.8-Flash-Next MTP** | `--spec-type draft-mtp` works for `qwen4exp` with its separate draft head. | `-md mtp-*.gguf --spec-type draft-mtp` | + +Measured decode speed on one **RTX 3060 12 GB** (`-ngl 99 -ncmoe 99 -fa 1`): + +| Model | Baseline | With the expert cache | +| --- | --- | --- | +| Qwen3.6-35B-A3B Q4_K_M (256 experts/layer) | 42.3 tok/s | **51.3 (+21%)** @ 124 slots | +| — same, plus `--spec-type draft-mtp` | 41.7 | **69.3 (+66%)** @ 112 slots | +| — same, plus async CPU splits (default on) | 41.7 | **74.2 (+78%)** @ 88 slots | +| GLM-4.7-Flash Q4_K_M (64 experts/layer) | 32.1 | **46.3 (+44%)** @ 40 slots | +| Laguna-S-2.1-118B-A8B IQ4_XS (256 experts/layer) | 11.5 | **12.1 (+5%)** @ 36 slots | +| Qwen3.8-Flash-Next UD-IQ3_XXS (512 experts/layer), with MTP | 16.6 | **24.4 (+47%)** @ 56 slots | + +On **two RTX 3060s**, Qwen3.8-Flash-Next with the cache split across both cards runs **~40 tok/s** +(see [Two GPUs](#two-gpus)). + +Cache-capable architectures: `qwen35moe`, `qwen4exp` (Qwen3.8-Flash-Next), `deepseek2`, `laguna`. +Other models run unchanged. -![llama](https://raw.githubusercontent.com/ggml-org/llama.brand/refs/heads/master/cover/llama-cpp/cover-llama-cpp-dark.svg) - -
- -LLM inference in C/C++ - -[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://opensource.org/licenses/MIT) -[![Release](https://img.shields.io/github/v/release/ggml-org/llama.cpp?filter=v*&color=brightgreen)](https://github.com/ggml-org/llama.cpp/releases?q=tag:v0) -[![Nightly](https://img.shields.io/github/v/release/ggml-org/llama.cpp?label=nightly&filter=b*&color=orange)](https://github.com/ggml-org/llama.cpp/releases?q=b) -[![Server](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/server.yml?label=Server)](https://github.com/ggml-org/llama.cpp/actions/workflows/server.yml) -[![Docker](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/docker.yml?label=Docker)](https://github.com/ggml-org/llama.cpp/actions/workflows/docker.yml) -[![Winget](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/winget.yml?label=Winget)](https://github.com/ggml-org/llama.cpp/actions/workflows/winget.yml) - -[ggml](https://github.com/ggml-org/ggml) / [ops](https://github.com/ggml-org/llama.cpp/blob/master/docs/ops.md) / [maintainer PRs](https://github.com/ggml-org/llama.cpp/issues?q=is%3Apr%20is%3Aopen%20draft%3AFalse%20(author%3Argerganov%20OR%20author%3AKitaitiMakoto%20OR%20author%3Adanbev%20OR%20author%3Aaldehir%20OR%20author%3Amax-krasnyansky%20OR%20author%3ACISC%20OR%20author%3Aggerganov%20OR%20author%3Aam17an%20OR%20author%3Abartowski1182%20OR%20author%3Anikwen%20OR%20author%3Ahipudding%20OR%20author%3AServeurpersoCom%20OR%20author%3Apwilkin%20OR%20author%3Areeselevine%20OR%20author%3Angxson%20OR%20author%3Ajeffbolznv%20OR%20author%3Amarty1885%20OR%20author%3A0cc4m%20OR%20author%3ATitaniumtown%20OR%20author%3Aangt%20OR%20author%3AIMbackK%20OR%20author%3Aarthw%20OR%20author%3AJohannesGaessler%20OR%20author%3AORippler%20OR%20author%3Aruixiang63%20OR%20author%3Axctan%20OR%20author%3Aallozaur%20OR%20author%3Ayomaytk%20OR%20author%3Aaendk%20OR%20author%3Agaugarg-nv%20OR%20author%3Ataronaeo%20OR%20author%3Aforforever73%20OR%20author%3Alhez%20OR%20author%3Anetrunnereve%20OR%20author%3Afairydreaming)%20sort%3Aupdated-desc) / [dev stats](https://github.com/ggml-org/llama.cpp-dev) / [lib llama API](https://github.com/ggml-org/llama.cpp/issues/9289) / [llama-server REST API](https://github.com/ggml-org/llama.cpp/issues/9291) +## Quick start -
+You need an NVIDIA GPU, the CUDA toolkit, CMake, and an MoE model in GGUF format. +**1. Build** +```bash +git clone --branch perf https://github.com/thecodacus/llama.cpp.git +cd llama.cpp +cmake -B build -DGGML_CUDA=ON -DCMAKE_BUILD_TYPE=Release +cmake --build build -j --target llama-server llama-moe-trace +``` +**2. Run it once without the cache** so you have a baseline. `-ncmoe 99` puts every expert in RAM: -## ⚡ This fork — Fable's MoE-offload prefill optimizations +```bash +./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa on -c 16384 +``` -Two **opt-in** optimizations for large MoE models whose experts are offloaded to system RAM -(`--n-cpu-moe`), found and implemented by Fable. Both are **off by default**, toggled via -environment variables, and produce **token-identical** output to mainline. +Open http://localhost:8080, ask something, and note the tokens per second. -| Env var | What it does | -| --- | --- | -| `GGML_CUDA_REGISTER_HOST=1` | Page-locks (pins) the mmap'd CPU expert weights so host→device copies go straight over DMA instead of through the driver's hidden bounce buffer (~6–7 → ~20 GB/s). | -| `GGML_SCHED_PREFETCH_EXPERTS=1` | Prefetches each layer's experts on a second CUDA stream, so the weight uploads overlap compute instead of stalling the GPU. | +**3. Record a routing profile** (once per model): -### Benchmark +```bash +MOE_TRACE_OUT=profile.csv ./build/bin/llama-moe-trace -m model.gguf \ + -ngl 99 -ncmoe 99 -fa on -c 4096 -n 512 --temp 0.7 \ + -p "Write a Python function that parses a CSV file and explain how it works." +``` -Measured on an **RTX 3060 12GB** with **Qwen3.6-35B-A3B** (`--n-cpu-moe 26`), prompt-processing at 2048 (`MODEL` = path to your `.gguf`): +**4. Serve with the cache.** Start around 64 slots: ```bash -# baseline (patches off): -./build/bin/llama-bench -m MODEL -ngl 99 -ncmoe 26 -p 2048 -n 0 -r 5 -b 2048 -ub 2048 +./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa on -c 16384 \ + --moe-cache-profile profile.csv --moe-cache-slots 64 +``` -# patched (both optimizations on): -GGML_CUDA_REGISTER_HOST=1 GGML_SCHED_PREFETCH_EXPERTS=1 \ -./build/bin/llama-bench -m MODEL -ngl 99 -ncmoe 26 -p 2048 -n 0 -r 5 -b 2048 -ub 2048 +The load log should show: + +``` +init_moe_expert_cache: expert cache: layers x 64 slots, MiB uploaded to CUDA0 ``` -Result: **~1143 → ~1880 t/s** prefill (**+64%**) — same GPU, same settings, token-identical. +**5. Find your slot count.** Raise `--moe-cache-slots` until the load prints +`pack allocation failed`, step back down, then leave about 1 GB of VRAM free (`nvidia-smi`) for +long prompts. More slots means more experts in VRAM and faster decode. -Branches: [`fable5/host-register`](https://github.com/thecodacus/llama.cpp/tree/fable5/host-register) (pinning only) · [`fable5/prefetch-experts`](https://github.com/thecodacus/llama.cpp/tree/fable5/prefetch-experts) (both — this branch). +That's it. Everything below is for squeezing out more. -## ⚡ This fork — MoE expert cache (VRAM-resident hot experts) +## Recipes -For MoE models whose routed experts live in system RAM (`--n-cpu-moe`), this fork can keep the -most-frequently-routed experts of each layer **resident in VRAM**. Decode runs the hot experts on -GPU and only the cold remainder on CPU; the two halves are merged exactly, so output is -**bit-identical** to baseline. Opt-in, off by default. +### One GPU, with MTP speculative decoding -Measured on an RTX 3060 12GB (`-ngl 99 -ncmoe 99 -fa 1`): +For models with an MTP head (Qwen3.6, Qwen3.8), stack it on top of the cache and drop a few slots +to make room: -| Model | Baseline tg | Cached tg | Prefill | -| --- | --- | --- | --- | -| Qwen3.6-35B-A3B Q4_K_M (256 experts/layer) | 42.3 | **51.3 (+21%)** @ 124 slots | +14% | -| — same, stacked with `--spec-type draft-mtp` | 41.7 | **69.3 (+66%)** @ 112 slots | — | -| — same, plus async CPU splits (default on) | 41.7 | **74.2 (+78%)** @ 88 slots | — | -| GLM-4.7-Flash Q4_K_M (64 experts/layer) | 32.1 | **46.3 (+44%)** @ 40 slots | +64% | -| Laguna-S-2.1-118B-A8B IQ4_XS (256 experts/layer) | 11.5 | **12.1 (+5%)** @ 36 slots | +12% | -| Qwen3.8-Flash-Next 177B UD-IQ3_XXS (512 experts/layer, 10 active) | 16.6 | **24.4 (+47%)** @ 56 slots, with `--spec-type draft-mtp` | — | +```bash +./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa on -c 16384 \ + --moe-cache-profile profile.csv --moe-cache-slots 56 \ + --spec-type draft-mtp --spec-draft-n-max 2 +``` -Supported architectures: `qwen35moe`, `qwen4exp` (Qwen3.8-Flash-Next), `deepseek2`, `laguna` (plain fused-SILU gated expert FFN, -separate gate/up/down tensors). Other architectures run unchanged. +Qwen3.8-Flash-Next keeps its MTP head in a separate file: add `-md mtp-Qwen3.8-Flash-Next-*.gguf`. -### Quick start +### Two GPUs -**1. Capture a routing profile** (one time per model — records which experts the router picks): +Split layers with `-sm layer`; each GPU caches the experts of its own layers, so `-ts` decides how +the cache is shared. The MTP draft head (`-ngld 99`) lands on the last GPU, so give that GPU fewer +layers: ```bash -MOE_TRACE_OUT=mymodel-code.csv ./build/bin/llama-moe-trace -m model.gguf \ - -ngl 99 -ncmoe 99 -fa 1 -c 4096 -n 512 -p "" +./build/bin/llama-server -m Qwen3.8-Flash-Next-UD-IQ3_XXS-00001-of-00003.gguf \ + -ngl 99 -ncmoe 99 -fa on -c 65536 -ctk q8_0 -ctv q8_0 \ + -sm layer -ts 28,20 --moe-cache-profile profile.csv --moe-cache-slots 136 \ + -md mtp-Qwen3.8-Flash-Next-shared-Q8_0.gguf -ngld 99 --spec-type draft-mtp --spec-draft-n-max 2 \ + --load-mode mmap --lazy-mode on --no-op-offload +``` + +On 2× RTX 3060 12 GB this used 11.2 + 11.4 GB and ran **39.9 tok/s** (temp 0, warm server), vs +35.0 for the best layout without the cache. Tune `-ts` until both cards end up about equally full. + +### Several models behind one server -MOE_TRACE_OUT=mymodel-chat.csv ./build/bin/llama-moe-trace -m model.gguf \ - -ngl 99 -ncmoe 99 -fa 1 -c 4096 -n 512 -p "" +Every flag works as a key in a `--models-preset` INI section: -cat mymodel-code.csv mymodel-chat.csv > mymodel-merged.csv +```ini +[qwen38-flash] +model = /models/Qwen3.8-Flash-Next-UD-IQ3_XXS-00001-of-00003.gguf +n-cpu-moe = 99 +moe-cache-profile = /models/traces/profile.csv +moe-cache-slots = 136 ``` -512 generated tokens per prompt is enough. Merge traces from contrasting workloads — a merged -profile measures within 1% of per-workload specialist profiles, so one merged CSV per model is -all you need. +The cache flags also read `LLAMA_ARG_MOE_CACHE_PROFILE` / `LLAMA_ARG_MOE_CACHE_SLOTS` (and the +legacy `GGML_MOE_CACHE_PROFILE` / `GGML_MOE_CACHE_SLOTS`, which `llama-bench` accepts too). -**2. Serve with the cache:** +## Better routing profiles -```bash -./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa 1 \ - --moe-cache-profile mymodel-merged.csv --moe-cache-slots 112 -``` +The quick-start profile is enough to start. For the best hit rate, trace the kind of work you +actually do. -Also works per model in a `--models-preset` INI section (`moe-cache-profile = ...`, -`moe-cache-slots = ...`), and as env vars `LLAMA_ARG_MOE_CACHE_PROFILE` / `LLAMA_ARG_MOE_CACHE_SLOTS` -(or legacy `GGML_MOE_CACHE_PROFILE` / `GGML_MOE_CACHE_SLOTS`, which `llama-bench` also accepts). +- **Merge contrasting workloads.** One trace per workload, then concatenate them: + `cat code.csv chat.csv long.csv > profile.csv`. A merged profile measures within 1% of + per-workload specialist profiles. +- **Trace with your server's sampling settings.** The tracer uses the normal sampling flags + (`--temp`, `--top-k`, `--top-p`, `--min-p`, `--seed`, penalties); `--temp 0` is greedy. +- **Render prompts with the model's chat template.** `-p`/`-f` take raw text, so a plain question + skips the template the server would apply. Let a running `llama-server` render it: -**3. Confirm it engaged** — look for this line at load: + ```bash + curl -s localhost:8080/apply-template -H 'Content-Type: application/json' \ + -d '{"messages":[{"role":"user","content":""}], + "chat_template_kwargs":{"enable_thinking":true}}' \ + | jq -r .prompt > review.txt -``` -init_moe_expert_cache: expert cache: 40 layers x 112 slots, 8164.00 MiB uploaded to CUDA0 -``` + MOE_TRACE_OUT=review.csv ./build/bin/llama-moe-trace -m model.gguf \ + -ngl 99 -ncmoe 99 -fa on -c 16384 -n 1024 \ + --temp 0.7 --top-p 0.95 --top-k 20 --min-p 0 --seed 101 -f review.txt + ``` + +- **Layout doesn't matter while tracing.** Routing doesn't depend on `-sm`, `-ts` or `-ncmoe`, so + trace with whatever layout loads. +- Only generated tokens count toward the profile; prompt rows are written with negative positions + and skipped at load. -A warning instead of this line means the cache fell back to baseline (see Tuning). +On Qwen3.8-Flash-Next (2× RTX 3060), a profile from eight 0.1k–15k-token prompts traced at temp 0.7 +averaged the same as a greedy short-prompt profile (30.0 vs 29.7 tok/s with server sampling at +temp 1.0), trading ~3 tok/s on code for ~1–3 tok/s on scripts, reasoning and long prompts. Build the +profile from the work you want fastest. -### Tuning +## Tuning - **`--moe-cache-slots` is the main knob** — experts cached per layer. Throughput rises with slot - count until the pack no longer fits in VRAM. The pack is all-or-nothing: an oversized request - logs `pack allocation failed - expert cache disabled` and runs at baseline speed (it does not - partially fill). The warning reports the per-slot cost and the maximum count that could fit — - set slots to that, minus headroom for KV/compute buffers which allocate afterwards. -- **The cold and hot chains overlap by default.** CPU graph splits run on a worker thread so the - GPU hot chain executes concurrently with the CPU cold chain (`--no-sched-async-cpu` to disable; - `llama-bench --sched-async-cpu 0,1` benches both). Worth +4-5% with speculative decoding, ~±2% - without it; outputs stay bit-identical either way. + count until the pack no longer fits in VRAM. The pack is all-or-nothing per GPU: an oversized + request logs `pack allocation failed on - expert cache disabled for its N layers` and + those layers run at baseline speed (it does not partially fill). The pack costs roughly + slots × (cached layers on that GPU) × (one expert's gate+up+down bytes). Check the GPU's memory + after load: a pack that fell back leaves it far below full. - **Leave ~900 MB of VRAM free beyond the pack.** A slot count that loads can still crash on the first large prompt: runtime CUDA pool growth allocates beyond what the load-time check sees. Size slots against the biggest prompt you will serve, not against "it loaded". -- **Fill VRAM to just under the ceiling, don't sweat the split.** Near the maximum, a marginal MB - is worth about the same as cache slots or as fully-resident layers (lower `--n-cpu-moe`). - Pure `-ncmoe 99` + max slots is the simple default; a hybrid (e.g. `-ncmoe 30` + fewer slots) - buys ~1% decode and ~3% prefill at best. +- **On one GPU, fill VRAM to just under the ceiling and don't sweat the split.** Near the maximum, + a marginal MB is worth about the same as cache slots or as fully-resident layers (lower + `--n-cpu-moe`). Pure `-ncmoe 99` + max slots is the simple default; a hybrid (e.g. `-ncmoe 30` + + fewer slots) buys ~1% decode and ~3% prefill at best. - **Context size competes with the pack.** KV grows with `-c` and shrinks the viable slot count. - Compressing the KV cache (`-ctk`/`-ctv`, e.g. TurboQuant types) frees VRAM that converts - directly into slots — often worth more than the KV precision costs. + Compressing the KV cache (`-ctk`/`-ctv`) frees VRAM that converts directly into slots — often + worth more than the KV precision costs. - **Speculative decoding stacks multiplicatively.** `--spec-type draft-mtp` composes with the cache (+48% cache × +12% MTP ≈ +66% on Qwen); reserve ~1 GB for the draft context by dropping a few slots. +- **The cold and hot chains overlap by default.** CPU graph splits run on a worker thread so the + GPU hot chain runs concurrently with the CPU cold chain (`--no-sched-async-cpu` to disable; + `llama-bench --sched-async-cpu 0,1` benches both). Worth +4-5% with speculative decoding, ~±2% + without it; outputs stay bit-identical either way. - **Expert size decides the payoff.** Big experts (GLM: ~5 MiB each) gain the most per slot; small experts need high slot counts before the win beats the dual-path overhead (~25% traffic coverage is roughly break-even). If VRAM only fits <15% of the expert count, expect single-digit gains (Laguna above). +- **Two GPUs: balance the split so both fill.** Filling both cards to the last few hundred MB was + slower again in testing (144 vs 136 slots above); stop a little short. +- **Benchmark on a warm server.** The first requests after a load run while expert pages are still + coming off disk; compare layouts over several rounds on a loaded server, not one request per + fresh start. - **Profiles are model-specific, workload-tolerant.** A wrong-workload profile still helps - (+28% measured on Qwen worst-case) but loses about half the win; the merged profile recovers + (+28% measured on Qwen worst-case) but loses about half the win; a merged profile recovers nearly all of it. Regenerate only if your usage changes character entirely. -### Troubleshooting +## Troubleshooting | Symptom | Cause | | --- | --- | | `cannot open profile '...'` | Path not visible to the process (e.g. not mounted into the container). | -| `pack allocation failed` | Slot count too big — read the fit math in the warning and reduce. | +| `pack allocation failed on ` | Slot count too big for that GPU — reduce slots, or move layers off it with `-ts`. | | `no CPU-resident MoE layers` | Experts are already on GPU (no `--n-cpu-moe`) — nothing to cache. | | No init line, no warning | Architecture not wired for the cache — model runs unchanged. | | Model loads, then context creation OOMs | Pack fits but KV/compute don't — drop a few slots or shrink/compress KV. | +| Crash on the first long prompt | Not enough free VRAM for the prompt's workspace — drop slots until ~1 GB stays free. | + +## Prefill speedups for offloaded experts + +Two environment variables speed up prompt processing when experts live in RAM, independent of the +cache. Both are off by default and token-identical. + +| Env var | What it does | +| --- | --- | +| `GGML_CUDA_REGISTER_HOST=1` | Page-locks (pins) the mmap'd CPU expert weights so host→device copies go straight over DMA instead of through the driver's hidden bounce buffer (~6–7 → ~20 GB/s). | +| `GGML_SCHED_PREFETCH_EXPERTS=1` | Prefetches each layer's experts on a second CUDA stream, so the weight uploads overlap compute instead of stalling the GPU. | + +On an RTX 3060 12 GB with Qwen3.6-35B-A3B (`--n-cpu-moe 26`), prefill at 2048 went from +**~1143 → ~1880 tok/s (+64%)**: + +```bash +GGML_CUDA_REGISTER_HOST=1 GGML_SCHED_PREFETCH_EXPERTS=1 \ +./build/bin/llama-bench -m model.gguf -ngl 99 -ncmoe 26 -p 2048 -n 0 -r 5 -b 2048 -ub 2048 +``` + +--- + +# Upstream llama.cpp + +Everything below is the upstream README. + +![llama](https://raw.githubusercontent.com/ggml-org/llama.brand/refs/heads/master/cover/llama-cpp/cover-llama-cpp-dark.svg) + +
+ +LLM inference in C/C++ + +[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://opensource.org/licenses/MIT) +[![Release](https://img.shields.io/github/v/release/ggml-org/llama.cpp?filter=v*&color=brightgreen)](https://github.com/ggml-org/llama.cpp/releases?q=tag:v0) +[![Nightly](https://img.shields.io/github/v/release/ggml-org/llama.cpp?label=nightly&filter=b*&color=orange)](https://github.com/ggml-org/llama.cpp/releases?q=b) +[![Server](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/server.yml?label=Server)](https://github.com/ggml-org/llama.cpp/actions/workflows/server.yml) +[![Docker](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/docker.yml?label=Docker)](https://github.com/ggml-org/llama.cpp/actions/workflows/docker.yml) +[![Winget](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/winget.yml?label=Winget)](https://github.com/ggml-org/llama.cpp/actions/workflows/winget.yml) + +[ggml](https://github.com/ggml-org/ggml) / [ops](https://github.com/ggml-org/llama.cpp/blob/master/docs/ops.md) / [maintainer PRs](https://github.com/ggml-org/llama.cpp/issues?q=is%3Apr%20is%3Aopen%20draft%3AFalse%20(author%3Argerganov%20OR%20author%3AKitaitiMakoto%20OR%20author%3Adanbev%20OR%20author%3Aaldehir%20OR%20author%3Amax-krasnyansky%20OR%20author%3ACISC%20OR%20author%3Aggerganov%20OR%20author%3Aam17an%20OR%20author%3Abartowski1182%20OR%20author%3Anikwen%20OR%20author%3Ahipudding%20OR%20author%3AServeurpersoCom%20OR%20author%3Apwilkin%20OR%20author%3Areeselevine%20OR%20author%3Angxson%20OR%20author%3Ajeffbolznv%20OR%20author%3Amarty1885%20OR%20author%3A0cc4m%20OR%20author%3ATitaniumtown%20OR%20author%3Aangt%20OR%20author%3AIMbackK%20OR%20author%3Aarthw%20OR%20author%3AJohannesGaessler%20OR%20author%3AORippler%20OR%20author%3Aruixiang63%20OR%20author%3Axctan%20OR%20author%3Aallozaur%20OR%20author%3Ayomaytk%20OR%20author%3Aaendk%20OR%20author%3Agaugarg-nv%20OR%20author%3Ataronaeo%20OR%20author%3Aforforever73%20OR%20author%3Alhez%20OR%20author%3Anetrunnereve%20OR%20author%3Afairydreaming)%20sort%3Aupdated-desc) / [dev stats](https://github.com/ggml-org/llama.cpp-dev) / [lib llama API](https://github.com/ggml-org/llama.cpp/issues/9289) / [llama-server REST API](https://github.com/ggml-org/llama.cpp/issues/9291) + +
+ ## Recent API changes