diff --git a/README.md b/README.md index d66ee10c2022..3bdc8871336b 100644 --- a/README.md +++ b/README.md @@ -1,152 +1,254 @@ -# llama.cpp +# llama.cpp — `perf` fork + +Faster **mixture-of-experts (MoE) models on consumer GPUs** when the experts don't fit in VRAM. +This is [llama.cpp](https://github.com/ggml-org/llama.cpp) plus a handful of opt-in changes for the +`--n-cpu-moe` case (experts in system RAM, the rest on the GPU). Everything is **off by default** +and produces the **same tokens** as upstream. + +| Feature | What it does | How to turn it on | +| --- | --- | --- | +| **MoE expert cache** | Keeps each layer's most-used experts resident in VRAM; the rest stay in RAM. With several GPUs, each GPU caches the experts of its own layers. | `--moe-cache-profile FILE --moe-cache-slots N` | +| **Routing profiles** | `llama-moe-trace` records which experts a model picks, with your real sampling settings, so the cache knows what to keep. | `llama-moe-trace` (see below) | +| **Async CPU splits** | CPU and GPU halves of each MoE layer run at the same time. | on by default with the cache (`--no-sched-async-cpu` to disable) | +| **Prefill speedups** | Pins offloaded expert memory and prefetches it on a second CUDA stream. | `GGML_CUDA_REGISTER_HOST=1 GGML_SCHED_PREFETCH_EXPERTS=1` | +| **Qwen3.8-Flash-Next MTP** | `--spec-type draft-mtp` works for `qwen4exp` with its separate draft head. | `-md mtp-*.gguf --spec-type draft-mtp` | + +Measured decode speed on one **RTX 3060 12 GB** (`-ngl 99 -ncmoe 99 -fa 1`): + +| Model | Baseline | With the expert cache | +| --- | --- | --- | +| Qwen3.6-35B-A3B Q4_K_M (256 experts/layer) | 42.3 tok/s | **51.3 (+21%)** @ 124 slots | +| — same, plus `--spec-type draft-mtp` | 41.7 | **69.3 (+66%)** @ 112 slots | +| — same, plus async CPU splits (default on) | 41.7 | **74.2 (+78%)** @ 88 slots | +| GLM-4.7-Flash Q4_K_M (64 experts/layer) | 32.1 | **46.3 (+44%)** @ 40 slots | +| Laguna-S-2.1-118B-A8B IQ4_XS (256 experts/layer) | 11.5 | **12.1 (+5%)** @ 36 slots | +| Qwen3.8-Flash-Next UD-IQ3_XXS (512 experts/layer), with MTP | 16.6 | **24.4 (+47%)** @ 56 slots | + +On **two RTX 3060s**, Qwen3.8-Flash-Next with the cache split across both cards runs **~40 tok/s** +(see [Two GPUs](#two-gpus)). + +Cache-capable architectures: `qwen35moe`, `qwen4exp` (Qwen3.8-Flash-Next), `deepseek2`, `laguna`. +Other models run unchanged. -![llama](https://raw.githubusercontent.com/ggml-org/llama.brand/refs/heads/master/cover/llama-cpp/cover-llama-cpp-dark.svg) - -
- -LLM inference in C/C++ - -[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://opensource.org/licenses/MIT) -[![Release](https://img.shields.io/github/v/release/ggml-org/llama.cpp?filter=v*&color=brightgreen)](https://github.com/ggml-org/llama.cpp/releases?q=tag:v0) -[![Nightly](https://img.shields.io/github/v/release/ggml-org/llama.cpp?label=nightly&filter=b*&color=orange)](https://github.com/ggml-org/llama.cpp/releases?q=b) -[![Server](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/server.yml?label=Server)](https://github.com/ggml-org/llama.cpp/actions/workflows/server.yml) -[![Docker](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/docker.yml?label=Docker)](https://github.com/ggml-org/llama.cpp/actions/workflows/docker.yml) -[![Winget](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/winget.yml?label=Winget)](https://github.com/ggml-org/llama.cpp/actions/workflows/winget.yml) - -[ggml](https://github.com/ggml-org/ggml) / [ops](https://github.com/ggml-org/llama.cpp/blob/master/docs/ops.md) / [maintainer PRs](https://github.com/ggml-org/llama.cpp/issues?q=is%3Apr%20is%3Aopen%20draft%3AFalse%20(author%3Argerganov%20OR%20author%3AKitaitiMakoto%20OR%20author%3Adanbev%20OR%20author%3Aaldehir%20OR%20author%3Amax-krasnyansky%20OR%20author%3ACISC%20OR%20author%3Aggerganov%20OR%20author%3Aam17an%20OR%20author%3Abartowski1182%20OR%20author%3Anikwen%20OR%20author%3Ahipudding%20OR%20author%3AServeurpersoCom%20OR%20author%3Apwilkin%20OR%20author%3Areeselevine%20OR%20author%3Angxson%20OR%20author%3Ajeffbolznv%20OR%20author%3Amarty1885%20OR%20author%3A0cc4m%20OR%20author%3ATitaniumtown%20OR%20author%3Aangt%20OR%20author%3AIMbackK%20OR%20author%3Aarthw%20OR%20author%3AJohannesGaessler%20OR%20author%3AORippler%20OR%20author%3Aruixiang63%20OR%20author%3Axctan%20OR%20author%3Aallozaur%20OR%20author%3Ayomaytk%20OR%20author%3Aaendk%20OR%20author%3Agaugarg-nv%20OR%20author%3Ataronaeo%20OR%20author%3Aforforever73%20OR%20author%3Alhez%20OR%20author%3Anetrunnereve%20OR%20author%3Afairydreaming)%20sort%3Aupdated-desc) / [dev stats](https://github.com/ggml-org/llama.cpp-dev) / [lib llama API](https://github.com/ggml-org/llama.cpp/issues/9289) / [llama-server REST API](https://github.com/ggml-org/llama.cpp/issues/9291) +## Quick start -
+You need an NVIDIA GPU, the CUDA toolkit, CMake, and an MoE model in GGUF format. +**1. Build** +```bash +git clone --branch perf https://github.com/thecodacus/llama.cpp.git +cd llama.cpp +cmake -B build -DGGML_CUDA=ON -DCMAKE_BUILD_TYPE=Release +cmake --build build -j --target llama-server llama-moe-trace +``` +**2. Run it once without the cache** so you have a baseline. `-ncmoe 99` puts every expert in RAM: -## ⚡ This fork — Fable's MoE-offload prefill optimizations +```bash +./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa on -c 16384 +``` -Two **opt-in** optimizations for large MoE models whose experts are offloaded to system RAM -(`--n-cpu-moe`), found and implemented by Fable. Both are **off by default**, toggled via -environment variables, and produce **token-identical** output to mainline. +Open http://localhost:8080, ask something, and note the tokens per second. -| Env var | What it does | -| --- | --- | -| `GGML_CUDA_REGISTER_HOST=1` | Page-locks (pins) the mmap'd CPU expert weights so host→device copies go straight over DMA instead of through the driver's hidden bounce buffer (~6–7 → ~20 GB/s). | -| `GGML_SCHED_PREFETCH_EXPERTS=1` | Prefetches each layer's experts on a second CUDA stream, so the weight uploads overlap compute instead of stalling the GPU. | +**3. Record a routing profile** (once per model): -### Benchmark +```bash +MOE_TRACE_OUT=profile.csv ./build/bin/llama-moe-trace -m model.gguf \ + -ngl 99 -ncmoe 99 -fa on -c 4096 -n 512 --temp 0.7 \ + -p "Write a Python function that parses a CSV file and explain how it works." +``` -Measured on an **RTX 3060 12GB** with **Qwen3.6-35B-A3B** (`--n-cpu-moe 26`), prompt-processing at 2048 (`MODEL` = path to your `.gguf`): +**4. Serve with the cache.** Start around 64 slots: ```bash -# baseline (patches off): -./build/bin/llama-bench -m MODEL -ngl 99 -ncmoe 26 -p 2048 -n 0 -r 5 -b 2048 -ub 2048 +./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa on -c 16384 \ + --moe-cache-profile profile.csv --moe-cache-slots 64 +``` -# patched (both optimizations on): -GGML_CUDA_REGISTER_HOST=1 GGML_SCHED_PREFETCH_EXPERTS=1 \ -./build/bin/llama-bench -m MODEL -ngl 99 -ncmoe 26 -p 2048 -n 0 -r 5 -b 2048 -ub 2048 +The load log should show: + +``` +init_moe_expert_cache: expert cache: layers x 64 slots, MiB uploaded to CUDA0 ``` -Result: **~1143 → ~1880 t/s** prefill (**+64%**) — same GPU, same settings, token-identical. +**5. Find your slot count.** Raise `--moe-cache-slots` until the load prints +`pack allocation failed`, step back down, then leave about 1 GB of VRAM free (`nvidia-smi`) for +long prompts. More slots means more experts in VRAM and faster decode. -Branches: [`fable5/host-register`](https://github.com/thecodacus/llama.cpp/tree/fable5/host-register) (pinning only) · [`fable5/prefetch-experts`](https://github.com/thecodacus/llama.cpp/tree/fable5/prefetch-experts) (both — this branch). +That's it. Everything below is for squeezing out more. -## ⚡ This fork — MoE expert cache (VRAM-resident hot experts) +## Recipes -For MoE models whose routed experts live in system RAM (`--n-cpu-moe`), this fork can keep the -most-frequently-routed experts of each layer **resident in VRAM**. Decode runs the hot experts on -GPU and only the cold remainder on CPU; the two halves are merged exactly, so output is -**bit-identical** to baseline. Opt-in, off by default. +### One GPU, with MTP speculative decoding -Measured on an RTX 3060 12GB (`-ngl 99 -ncmoe 99 -fa 1`): +For models with an MTP head (Qwen3.6, Qwen3.8), stack it on top of the cache and drop a few slots +to make room: -| Model | Baseline tg | Cached tg | Prefill | -| --- | --- | --- | --- | -| Qwen3.6-35B-A3B Q4_K_M (256 experts/layer) | 42.3 | **51.3 (+21%)** @ 124 slots | +14% | -| — same, stacked with `--spec-type draft-mtp` | 41.7 | **69.3 (+66%)** @ 112 slots | — | -| — same, plus async CPU splits (default on) | 41.7 | **74.2 (+78%)** @ 88 slots | — | -| GLM-4.7-Flash Q4_K_M (64 experts/layer) | 32.1 | **46.3 (+44%)** @ 40 slots | +64% | -| Laguna-S-2.1-118B-A8B IQ4_XS (256 experts/layer) | 11.5 | **12.1 (+5%)** @ 36 slots | +12% | -| Qwen3.8-Flash-Next 177B UD-IQ3_XXS (512 experts/layer, 10 active) | 16.6 | **24.4 (+47%)** @ 56 slots, with `--spec-type draft-mtp` | — | +```bash +./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa on -c 16384 \ + --moe-cache-profile profile.csv --moe-cache-slots 56 \ + --spec-type draft-mtp --spec-draft-n-max 2 +``` -Supported architectures: `qwen35moe`, `qwen4exp` (Qwen3.8-Flash-Next), `deepseek2`, `laguna` (plain fused-SILU gated expert FFN, -separate gate/up/down tensors). Other architectures run unchanged. +Qwen3.8-Flash-Next keeps its MTP head in a separate file: add `-md mtp-Qwen3.8-Flash-Next-*.gguf`. -### Quick start +### Two GPUs -**1. Capture a routing profile** (one time per model — records which experts the router picks): +Split layers with `-sm layer`; each GPU caches the experts of its own layers, so `-ts` decides how +the cache is shared. The MTP draft head (`-ngld 99`) lands on the last GPU, so give that GPU fewer +layers: ```bash -MOE_TRACE_OUT=mymodel-code.csv ./build/bin/llama-moe-trace -m model.gguf \ - -ngl 99 -ncmoe 99 -fa 1 -c 4096 -n 512 -p "" +./build/bin/llama-server -m Qwen3.8-Flash-Next-UD-IQ3_XXS-00001-of-00003.gguf \ + -ngl 99 -ncmoe 99 -fa on -c 65536 -ctk q8_0 -ctv q8_0 \ + -sm layer -ts 28,20 --moe-cache-profile profile.csv --moe-cache-slots 136 \ + -md mtp-Qwen3.8-Flash-Next-shared-Q8_0.gguf -ngld 99 --spec-type draft-mtp --spec-draft-n-max 2 \ + --load-mode mmap --lazy-mode on --no-op-offload +``` + +On 2× RTX 3060 12 GB this used 11.2 + 11.4 GB and ran **39.9 tok/s** (temp 0, warm server), vs +35.0 for the best layout without the cache. Tune `-ts` until both cards end up about equally full. + +### Several models behind one server -MOE_TRACE_OUT=mymodel-chat.csv ./build/bin/llama-moe-trace -m model.gguf \ - -ngl 99 -ncmoe 99 -fa 1 -c 4096 -n 512 -p "" +Every flag works as a key in a `--models-preset` INI section: -cat mymodel-code.csv mymodel-chat.csv > mymodel-merged.csv +```ini +[qwen38-flash] +model = /models/Qwen3.8-Flash-Next-UD-IQ3_XXS-00001-of-00003.gguf +n-cpu-moe = 99 +moe-cache-profile = /models/traces/profile.csv +moe-cache-slots = 136 ``` -512 generated tokens per prompt is enough. Merge traces from contrasting workloads — a merged -profile measures within 1% of per-workload specialist profiles, so one merged CSV per model is -all you need. +The cache flags also read `LLAMA_ARG_MOE_CACHE_PROFILE` / `LLAMA_ARG_MOE_CACHE_SLOTS` (and the +legacy `GGML_MOE_CACHE_PROFILE` / `GGML_MOE_CACHE_SLOTS`, which `llama-bench` accepts too). -**2. Serve with the cache:** +## Better routing profiles -```bash -./build/bin/llama-server -m model.gguf -ngl 99 -ncmoe 99 -fa 1 \ - --moe-cache-profile mymodel-merged.csv --moe-cache-slots 112 -``` +The quick-start profile is enough to start. For the best hit rate, trace the kind of work you +actually do. -Also works per model in a `--models-preset` INI section (`moe-cache-profile = ...`, -`moe-cache-slots = ...`), and as env vars `LLAMA_ARG_MOE_CACHE_PROFILE` / `LLAMA_ARG_MOE_CACHE_SLOTS` -(or legacy `GGML_MOE_CACHE_PROFILE` / `GGML_MOE_CACHE_SLOTS`, which `llama-bench` also accepts). +- **Merge contrasting workloads.** One trace per workload, then concatenate them: + `cat code.csv chat.csv long.csv > profile.csv`. A merged profile measures within 1% of + per-workload specialist profiles. +- **Trace with your server's sampling settings.** The tracer uses the normal sampling flags + (`--temp`, `--top-k`, `--top-p`, `--min-p`, `--seed`, penalties); `--temp 0` is greedy. +- **Render prompts with the model's chat template.** `-p`/`-f` take raw text, so a plain question + skips the template the server would apply. Let a running `llama-server` render it: -**3. Confirm it engaged** — look for this line at load: + ```bash + curl -s localhost:8080/apply-template -H 'Content-Type: application/json' \ + -d '{"messages":[{"role":"user","content":""}], + "chat_template_kwargs":{"enable_thinking":true}}' \ + | jq -r .prompt > review.txt -``` -init_moe_expert_cache: expert cache: 40 layers x 112 slots, 8164.00 MiB uploaded to CUDA0 -``` + MOE_TRACE_OUT=review.csv ./build/bin/llama-moe-trace -m model.gguf \ + -ngl 99 -ncmoe 99 -fa on -c 16384 -n 1024 \ + --temp 0.7 --top-p 0.95 --top-k 20 --min-p 0 --seed 101 -f review.txt + ``` + +- **Layout doesn't matter while tracing.** Routing doesn't depend on `-sm`, `-ts` or `-ncmoe`, so + trace with whatever layout loads. +- Only generated tokens count toward the profile; prompt rows are written with negative positions + and skipped at load. -A warning instead of this line means the cache fell back to baseline (see Tuning). +On Qwen3.8-Flash-Next (2× RTX 3060), a profile from eight 0.1k–15k-token prompts traced at temp 0.7 +averaged the same as a greedy short-prompt profile (30.0 vs 29.7 tok/s with server sampling at +temp 1.0), trading ~3 tok/s on code for ~1–3 tok/s on scripts, reasoning and long prompts. Build the +profile from the work you want fastest. -### Tuning +## Tuning - **`--moe-cache-slots` is the main knob** — experts cached per layer. Throughput rises with slot - count until the pack no longer fits in VRAM. The pack is all-or-nothing: an oversized request - logs `pack allocation failed - expert cache disabled` and runs at baseline speed (it does not - partially fill). The warning reports the per-slot cost and the maximum count that could fit — - set slots to that, minus headroom for KV/compute buffers which allocate afterwards. -- **The cold and hot chains overlap by default.** CPU graph splits run on a worker thread so the - GPU hot chain executes concurrently with the CPU cold chain (`--no-sched-async-cpu` to disable; - `llama-bench --sched-async-cpu 0,1` benches both). Worth +4-5% with speculative decoding, ~±2% - without it; outputs stay bit-identical either way. + count until the pack no longer fits in VRAM. The pack is all-or-nothing per GPU: an oversized + request logs `pack allocation failed on - expert cache disabled for its N layers` and + those layers run at baseline speed (it does not partially fill). The pack costs roughly + slots × (cached layers on that GPU) × (one expert's gate+up+down bytes). Check the GPU's memory + after load: a pack that fell back leaves it far below full. - **Leave ~900 MB of VRAM free beyond the pack.** A slot count that loads can still crash on the first large prompt: runtime CUDA pool growth allocates beyond what the load-time check sees. Size slots against the biggest prompt you will serve, not against "it loaded". -- **Fill VRAM to just under the ceiling, don't sweat the split.** Near the maximum, a marginal MB - is worth about the same as cache slots or as fully-resident layers (lower `--n-cpu-moe`). - Pure `-ncmoe 99` + max slots is the simple default; a hybrid (e.g. `-ncmoe 30` + fewer slots) - buys ~1% decode and ~3% prefill at best. +- **On one GPU, fill VRAM to just under the ceiling and don't sweat the split.** Near the maximum, + a marginal MB is worth about the same as cache slots or as fully-resident layers (lower + `--n-cpu-moe`). Pure `-ncmoe 99` + max slots is the simple default; a hybrid (e.g. `-ncmoe 30` + + fewer slots) buys ~1% decode and ~3% prefill at best. - **Context size competes with the pack.** KV grows with `-c` and shrinks the viable slot count. - Compressing the KV cache (`-ctk`/`-ctv`, e.g. TurboQuant types) frees VRAM that converts - directly into slots — often worth more than the KV precision costs. + Compressing the KV cache (`-ctk`/`-ctv`) frees VRAM that converts directly into slots — often + worth more than the KV precision costs. - **Speculative decoding stacks multiplicatively.** `--spec-type draft-mtp` composes with the cache (+48% cache × +12% MTP ≈ +66% on Qwen); reserve ~1 GB for the draft context by dropping a few slots. +- **The cold and hot chains overlap by default.** CPU graph splits run on a worker thread so the + GPU hot chain runs concurrently with the CPU cold chain (`--no-sched-async-cpu` to disable; + `llama-bench --sched-async-cpu 0,1` benches both). Worth +4-5% with speculative decoding, ~±2% + without it; outputs stay bit-identical either way. - **Expert size decides the payoff.** Big experts (GLM: ~5 MiB each) gain the most per slot; small experts need high slot counts before the win beats the dual-path overhead (~25% traffic coverage is roughly break-even). If VRAM only fits <15% of the expert count, expect single-digit gains (Laguna above). +- **Two GPUs: balance the split so both fill.** Filling both cards to the last few hundred MB was + slower again in testing (144 vs 136 slots above); stop a little short. +- **Benchmark on a warm server.** The first requests after a load run while expert pages are still + coming off disk; compare layouts over several rounds on a loaded server, not one request per + fresh start. - **Profiles are model-specific, workload-tolerant.** A wrong-workload profile still helps - (+28% measured on Qwen worst-case) but loses about half the win; the merged profile recovers + (+28% measured on Qwen worst-case) but loses about half the win; a merged profile recovers nearly all of it. Regenerate only if your usage changes character entirely. -### Troubleshooting +## Troubleshooting | Symptom | Cause | | --- | --- | | `cannot open profile '...'` | Path not visible to the process (e.g. not mounted into the container). | -| `pack allocation failed` | Slot count too big — read the fit math in the warning and reduce. | +| `pack allocation failed on ` | Slot count too big for that GPU — reduce slots, or move layers off it with `-ts`. | | `no CPU-resident MoE layers` | Experts are already on GPU (no `--n-cpu-moe`) — nothing to cache. | | No init line, no warning | Architecture not wired for the cache — model runs unchanged. | | Model loads, then context creation OOMs | Pack fits but KV/compute don't — drop a few slots or shrink/compress KV. | +| Crash on the first long prompt | Not enough free VRAM for the prompt's workspace — drop slots until ~1 GB stays free. | + +## Prefill speedups for offloaded experts + +Two environment variables speed up prompt processing when experts live in RAM, independent of the +cache. Both are off by default and token-identical. + +| Env var | What it does | +| --- | --- | +| `GGML_CUDA_REGISTER_HOST=1` | Page-locks (pins) the mmap'd CPU expert weights so host→device copies go straight over DMA instead of through the driver's hidden bounce buffer (~6–7 → ~20 GB/s). | +| `GGML_SCHED_PREFETCH_EXPERTS=1` | Prefetches each layer's experts on a second CUDA stream, so the weight uploads overlap compute instead of stalling the GPU. | + +On an RTX 3060 12 GB with Qwen3.6-35B-A3B (`--n-cpu-moe 26`), prefill at 2048 went from +**~1143 → ~1880 tok/s (+64%)**: + +```bash +GGML_CUDA_REGISTER_HOST=1 GGML_SCHED_PREFETCH_EXPERTS=1 \ +./build/bin/llama-bench -m model.gguf -ngl 99 -ncmoe 26 -p 2048 -n 0 -r 5 -b 2048 -ub 2048 +``` + +--- + +# Upstream llama.cpp + +Everything below is the upstream README. + +![llama](https://raw.githubusercontent.com/ggml-org/llama.brand/refs/heads/master/cover/llama-cpp/cover-llama-cpp-dark.svg) + +
+ +LLM inference in C/C++ + +[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://opensource.org/licenses/MIT) +[![Release](https://img.shields.io/github/v/release/ggml-org/llama.cpp?filter=v*&color=brightgreen)](https://github.com/ggml-org/llama.cpp/releases?q=tag:v0) +[![Nightly](https://img.shields.io/github/v/release/ggml-org/llama.cpp?label=nightly&filter=b*&color=orange)](https://github.com/ggml-org/llama.cpp/releases?q=b) +[![Server](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/server.yml?label=Server)](https://github.com/ggml-org/llama.cpp/actions/workflows/server.yml) +[![Docker](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/docker.yml?label=Docker)](https://github.com/ggml-org/llama.cpp/actions/workflows/docker.yml) +[![Winget](https://img.shields.io/github/actions/workflow/status/ggml-org/llama.cpp/winget.yml?label=Winget)](https://github.com/ggml-org/llama.cpp/actions/workflows/winget.yml) + +[ggml](https://github.com/ggml-org/ggml) / [ops](https://github.com/ggml-org/llama.cpp/blob/master/docs/ops.md) / [maintainer PRs](https://github.com/ggml-org/llama.cpp/issues?q=is%3Apr%20is%3Aopen%20draft%3AFalse%20(author%3Argerganov%20OR%20author%3AKitaitiMakoto%20OR%20author%3Adanbev%20OR%20author%3Aaldehir%20OR%20author%3Amax-krasnyansky%20OR%20author%3ACISC%20OR%20author%3Aggerganov%20OR%20author%3Aam17an%20OR%20author%3Abartowski1182%20OR%20author%3Anikwen%20OR%20author%3Ahipudding%20OR%20author%3AServeurpersoCom%20OR%20author%3Apwilkin%20OR%20author%3Areeselevine%20OR%20author%3Angxson%20OR%20author%3Ajeffbolznv%20OR%20author%3Amarty1885%20OR%20author%3A0cc4m%20OR%20author%3ATitaniumtown%20OR%20author%3Aangt%20OR%20author%3AIMbackK%20OR%20author%3Aarthw%20OR%20author%3AJohannesGaessler%20OR%20author%3AORippler%20OR%20author%3Aruixiang63%20OR%20author%3Axctan%20OR%20author%3Aallozaur%20OR%20author%3Ayomaytk%20OR%20author%3Aaendk%20OR%20author%3Agaugarg-nv%20OR%20author%3Ataronaeo%20OR%20author%3Aforforever73%20OR%20author%3Alhez%20OR%20author%3Anetrunnereve%20OR%20author%3Afairydreaming)%20sort%3Aupdated-desc) / [dev stats](https://github.com/ggml-org/llama.cpp-dev) / [lib llama API](https://github.com/ggml-org/llama.cpp/issues/9289) / [llama-server REST API](https://github.com/ggml-org/llama.cpp/issues/9291) + +
+ ## Recent API changes diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 90afd38a7eaf..3e09e0b2406a 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1887,117 +1887,126 @@ void llama_model_base::init_moe_expert_cache() { return; } - ggml_backend_dev_t dev = nullptr; + ggml_backend_dev_t dev_first = nullptr; for (const auto & d : devices) { - if (!d.is_meta) { dev = d.dev; break; } + if (!d.is_meta) { dev_first = d.dev; break; } } - if (dev == nullptr || ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU) { + if (dev_first == nullptr || ggml_backend_dev_type(dev_first) != GGML_BACKEND_DEVICE_TYPE_GPU) { LLAMA_LOG_WARN("%s: no GPU device - expert cache disabled\n", __func__); return; } - ggml_backend_buffer_type_t buft = ggml_backend_dev_buffer_type(dev); - // candidate layers: routed experts resident in host memory - std::vector pack_layers; + // candidate layers: routed experts resident in host memory, grouped by the GPU that runs the layer so the + // hot experts sit on the same device as the rest of the layer (layers left on the CPU use the first GPU) + std::map> pack_layers_by_dev; for (int il = 0; il < (int) layers.size(); il++) { const auto & l = layers[il]; if (l.ffn_gate_exps && l.ffn_up_exps && l.ffn_down_exps && freq.count(il) && l.ffn_gate_exps->buffer && ggml_backend_buft_is_host(ggml_backend_buffer_get_type(l.ffn_gate_exps->buffer))) { - pack_layers.push_back(il); + ggml_backend_dev_t dev = dev_layer(il); + if (ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU) { + dev = dev_first; + } + pack_layers_by_dev[dev].push_back(il); } } - if (pack_layers.empty()) { + if (pack_layers_by_dev.empty()) { LLAMA_LOG_INFO("%s: no CPU-resident MoE layers - expert cache not built\n", __func__); return; } - ggml_init_params ctx_params = { - /*.mem_size =*/ (5*pack_layers.size() + 1)*ggml_tensor_overhead(), - /*.mem_buffer =*/ nullptr, - /*.no_alloc =*/ true, - }; - ggml_context * ctx = ggml_init(ctx_params); - - for (int il : pack_layers) { - auto & l = layers[il]; - const ggml_tensor * g = l.ffn_gate_exps; - const ggml_tensor * u = l.ffn_up_exps; - const ggml_tensor * d = l.ffn_down_exps; - const int64_t n_expert = g->ne[2]; - const int64_t S = std::min(n_slots, n_expert); - l.ffn_gate_exps_hot = ggml_new_tensor_3d(ctx, g->type, g->ne[0], g->ne[1], S); - l.ffn_up_exps_hot = ggml_new_tensor_3d(ctx, u->type, u->ne[0], u->ne[1], S); - l.ffn_down_exps_hot = ggml_new_tensor_3d(ctx, d->type, d->ne[0], d->ne[1], S); - l.moe_map_hot = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); - l.moe_map_cold = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); - ggml_format_name(l.ffn_gate_exps_hot, "blk.%d.ffn_gate_exps_hot", il); - ggml_format_name(l.ffn_up_exps_hot, "blk.%d.ffn_up_exps_hot", il); - ggml_format_name(l.ffn_down_exps_hot, "blk.%d.ffn_down_exps_hot", il); - ggml_format_name(l.moe_map_hot, "blk.%d.moe_map_hot", il); - ggml_format_name(l.moe_map_cold, "blk.%d.moe_map_cold", il); - } - - ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, buft); - if (buf == nullptr) { - LLAMA_LOG_WARN("%s: pack allocation failed - expert cache disabled\n", __func__); - ggml_free(ctx); - for (int il : pack_layers) { - auto & l = layers[il]; - l.ffn_gate_exps_hot = l.ffn_up_exps_hot = l.ffn_down_exps_hot = nullptr; - l.moe_map_hot = l.moe_map_cold = nullptr; - } - return; - } - // weights usage pins the pack tensors to their backend during graph - // assignment - without it a CPU-assigned consumer can drag the hot - // matmuls (and a per-layer weight copy) onto the CPU - ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); - - // fill packs (expert dim is outermost: one contiguous slab per expert) std::vector slab; std::vector map_hot, map_cold; - size_t total_bytes = 0; - for (int il : pack_layers) { - auto & l = layers[il]; - const int64_t n_expert = l.ffn_gate_exps->ne[2]; - const int64_t S = l.ffn_gate_exps_hot->ne[2]; + for (const auto & [dev, pack_layers] : pack_layers_by_dev) { + ggml_backend_buffer_type_t buft = ggml_backend_dev_buffer_type(dev); - std::vector> ranked; // (-count, expert) - for (const auto & [e, c] : freq[il]) { - if (e < n_expert) { - ranked.push_back({-c, e}); + ggml_init_params ctx_params = { + /*.mem_size =*/ (5*pack_layers.size() + 1)*ggml_tensor_overhead(), + /*.mem_buffer =*/ nullptr, + /*.no_alloc =*/ true, + }; + ggml_context * ctx = ggml_init(ctx_params); + + for (int il : pack_layers) { + auto & l = layers[il]; + const ggml_tensor * g = l.ffn_gate_exps; + const ggml_tensor * u = l.ffn_up_exps; + const ggml_tensor * d = l.ffn_down_exps; + const int64_t n_expert = g->ne[2]; + const int64_t S = std::min(n_slots, n_expert); + l.ffn_gate_exps_hot = ggml_new_tensor_3d(ctx, g->type, g->ne[0], g->ne[1], S); + l.ffn_up_exps_hot = ggml_new_tensor_3d(ctx, u->type, u->ne[0], u->ne[1], S); + l.ffn_down_exps_hot = ggml_new_tensor_3d(ctx, d->type, d->ne[0], d->ne[1], S); + l.moe_map_hot = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); + l.moe_map_cold = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, 1, n_expert); + ggml_format_name(l.ffn_gate_exps_hot, "blk.%d.ffn_gate_exps_hot", il); + ggml_format_name(l.ffn_up_exps_hot, "blk.%d.ffn_up_exps_hot", il); + ggml_format_name(l.ffn_down_exps_hot, "blk.%d.ffn_down_exps_hot", il); + ggml_format_name(l.moe_map_hot, "blk.%d.moe_map_hot", il); + ggml_format_name(l.moe_map_cold, "blk.%d.moe_map_cold", il); + } + + ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft(ctx, buft); + if (buf == nullptr) { + LLAMA_LOG_WARN("%s: pack allocation failed on %s - expert cache disabled for its %zu layers\n", + __func__, ggml_backend_buft_name(buft), pack_layers.size()); + ggml_free(ctx); + for (int il : pack_layers) { + auto & l = layers[il]; + l.ffn_gate_exps_hot = l.ffn_up_exps_hot = l.ffn_down_exps_hot = nullptr; + l.moe_map_hot = l.moe_map_cold = nullptr; } + continue; } - std::sort(ranked.begin(), ranked.end()); + // weights usage pins the pack tensors to their backend during graph + // assignment - without it a CPU-assigned consumer can drag the hot + // matmuls (and a per-layer weight copy) onto the CPU + ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); - map_hot.assign(n_expert, -1); - map_cold.resize(n_expert); - for (int64_t e = 0; e < n_expert; e++) { - map_cold[e] = (int32_t) e; - } - for (int64_t s = 0; s < S && s < (int64_t) ranked.size(); s++) { - const int32_t e = ranked[s].second; - map_hot[e] = (int32_t) s; - map_cold[e] = -1; - const ggml_tensor * srcs[3] = { l.ffn_gate_exps, l.ffn_up_exps, l.ffn_down_exps }; - ggml_tensor * dsts[3] = { l.ffn_gate_exps_hot, l.ffn_up_exps_hot, l.ffn_down_exps_hot }; - for (int t = 0; t < 3; t++) { - const size_t nb = srcs[t]->nb[2]; - slab.resize(nb); - ggml_backend_tensor_get(srcs[t], slab.data(), e*nb, nb); - ggml_backend_tensor_set(dsts[t], slab.data(), s*nb, nb); - total_bytes += nb; + // fill packs (expert dim is outermost: one contiguous slab per expert) + size_t total_bytes = 0; + for (int il : pack_layers) { + auto & l = layers[il]; + const int64_t n_expert = l.ffn_gate_exps->ne[2]; + const int64_t S = l.ffn_gate_exps_hot->ne[2]; + + std::vector> ranked; // (-count, expert) + for (const auto & [e, c] : freq[il]) { + if (e < n_expert) { + ranked.push_back({-c, e}); + } + } + std::sort(ranked.begin(), ranked.end()); + + map_hot.assign(n_expert, -1); + map_cold.resize(n_expert); + for (int64_t e = 0; e < n_expert; e++) { + map_cold[e] = (int32_t) e; + } + for (int64_t s = 0; s < S && s < (int64_t) ranked.size(); s++) { + const int32_t e = ranked[s].second; + map_hot[e] = (int32_t) s; + map_cold[e] = -1; + const ggml_tensor * srcs[3] = { l.ffn_gate_exps, l.ffn_up_exps, l.ffn_down_exps }; + ggml_tensor * dsts[3] = { l.ffn_gate_exps_hot, l.ffn_up_exps_hot, l.ffn_down_exps_hot }; + for (int t = 0; t < 3; t++) { + const size_t nb = srcs[t]->nb[2]; + slab.resize(nb); + ggml_backend_tensor_get(srcs[t], slab.data(), e*nb, nb); + ggml_backend_tensor_set(dsts[t], slab.data(), s*nb, nb); + total_bytes += nb; + } } + ggml_backend_tensor_set(l.moe_map_hot, map_hot.data(), 0, n_expert*sizeof(int32_t)); + ggml_backend_tensor_set(l.moe_map_cold, map_cold.data(), 0, n_expert*sizeof(int32_t)); } - ggml_backend_tensor_set(l.moe_map_hot, map_hot.data(), 0, n_expert*sizeof(int32_t)); - ggml_backend_tensor_set(l.moe_map_cold, map_cold.data(), 0, n_expert*sizeof(int32_t)); - } - pimpl->ctxs_bufs.emplace_back(ggml_context_ptr{ctx}, std::vector{}); - pimpl->ctxs_bufs.back().second.emplace_back(buf); + pimpl->ctxs_bufs.emplace_back(ggml_context_ptr{ctx}, std::vector{}); + pimpl->ctxs_bufs.back().second.emplace_back(buf); - LLAMA_LOG_INFO("%s: expert cache: %zu layers x %d slots, %.2f MiB uploaded to %s\n", - __func__, pack_layers.size(), n_slots, total_bytes/1024.0/1024.0, ggml_backend_buft_name(buft)); + LLAMA_LOG_INFO("%s: expert cache: %zu layers x %d slots, %.2f MiB uploaded to %s\n", + __func__, pack_layers.size(), n_slots, total_bytes/1024.0/1024.0, ggml_backend_buft_name(buft)); + } } ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM_TN_IMPL & tn, const std::initializer_list & ne, int flags) { diff --git a/tools/moe-trace/moe-trace.cpp b/tools/moe-trace/moe-trace.cpp index 032d49d977f9..29636ff9a9bb 100644 --- a/tools/moe-trace/moe-trace.cpp +++ b/tools/moe-trace/moe-trace.cpp @@ -9,12 +9,13 @@ // separate prefill routing from decode routing. // // Usage: -// MOE_TRACE_OUT=trace.csv llama-moe-trace -m model.gguf -ngl 99 -ncmoe 26 -fa on \ +// MOE_TRACE_OUT=trace.csv llama-moe-trace -m model.gguf -ngl 99 -ncmoe 26 -fa on --temp 0 \ // -p "prompt text" -n 512 #include "arg.h" #include "common.h" #include "log.h" +#include "sampling.h" #include "llama.h" #include @@ -123,12 +124,14 @@ int main(int argc, char ** argv) { tc.pos += n_eval; } - // greedy decode + // decode with the sampling flags (--temp, --top-k, --top-p, --min-p, --seed, ...) so the trace + // follows the same expert routing a server with those defaults would see tc.in_prompt = false; - llama_sampler * smpl = llama_sampler_init_greedy(); + common_sampler * smpl = common_sampler_init(model, params.sampling); llama_token tok = 0; for (int i = 0; i < params.n_predict; i++) { - tok = llama_sampler_sample(smpl, lctx, -1); + tok = common_sampler_sample(smpl, lctx, -1); + common_sampler_accept(smpl, tok, true); if (llama_vocab_is_eog(vocab, tok)) { break; } @@ -141,7 +144,7 @@ int main(int argc, char ** argv) { LOG_INF("decoded %d/%d\n", i, params.n_predict); } } - llama_sampler_free(smpl); + common_sampler_free(smpl); fclose(tc.out); LOG_INF("trace written to %s\n", out_path);