From 86fc8665e81a0194816f7db2da056cdb2f52744b Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 28 Sep 2026 08:04:06 +0000 Subject: [PATCH 1/4] mel: skip the zero bins of each mel filter Each triangular mel filter is non-zero over a narrow band of the 257 FFT bins, but the frontend multiplied the full row for every frame. The product now runs over each filter's non-zero band only. The skipped terms are exact zeros, so the output stays bit-identical. The mel frontend runs on the host. With the encoder on a GPU it was most of the latency: on a GB10 it took 101 of 117 ms for ced-base on a 36 s clip. With this change the full classify takes 62 ms. Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- src/mel.cpp | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/src/mel.cpp b/src/mel.cpp index 58cb940..11ade16 100644 --- a/src/mel.cpp +++ b/src/mel.cpp @@ -34,6 +34,19 @@ void mel_spectrogram(const CedConfig& cfg, const float* window, const float* fil input_values.assign((size_t)n_mels * T, 0.0f); std::vector frame(n_fft), re, im, power(n_freqs); + // Each triangular mel filter is non-zero over a narrow band only; restrict + // the filterbank product to [lo, hi) per row. The skipped terms are exact + // zeros, so the sums (and their order) are unchanged. + std::vector lo(n_mels, 0), hi(n_mels, 0); + for (int m = 0; m < n_mels; ++m) { + const float* fb = filterbank + (size_t)m * n_freqs; + int f0 = 0, f1 = n_freqs; + while (f0 < n_freqs && fb[f0] == 0.0f) ++f0; + while (f1 > f0 && fb[f1 - 1] == 0.0f) --f1; + lo[m] = f0; + hi[m] = f1; + } + for (int t = 0; t < T; ++t) { const int start = t * hop; for (int i = 0; i < n_fft; ++i) { @@ -46,7 +59,7 @@ void mel_spectrogram(const CedConfig& cfg, const float* window, const float* fil for (int m = 0; m < n_mels; ++m) { const float* fb = filterbank + (size_t)m * n_freqs; float acc = 0.0f; - for (int f = 0; f < n_freqs; ++f) acc += fb[f] * power[f]; + for (int f = lo[m]; f < hi[m]; ++f) acc += fb[f] * power[f]; input_values[(size_t)m * T + t] = acc; // mel power, dB applied below } } From fe7481c3a3e4a5523e6f4d2fae38de92d86c0223 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 28 Sep 2026 08:04:06 +0000 Subject: [PATCH 2/4] runtime: run the model on any ggml GPU backend The runner always used a static CPU backend and created and freed a graph allocator for every graph. GPU builds compiled but never used the GPU. Each loaded model now owns a backend. It picks the first GPU or integrated GPU, or the device named in CED_DEVICE ("cpu", "CUDA0", "Vulkan0", "MTL0"), and falls back to the CPU. One ggml_gallocr is kept for the model's lifetime. A graph goes through ggml_backend_sched with a CPU fallback only when the device has no kernel for one of its ops. On a GPU the weights are uploaded to one device buffer at load. The tensors the host reads (mel window, filterbank, init_bn stats) keep a host copy, and the init_bn scale and shift are folded once at load. classify now runs each chunk as one graph instead of two with a host round trip in between. CPU output is bit-identical to the previous code (all 527 probs, four models, two clips) and CPU speed is unchanged. On GPU the top-5 tags match the CPU on CUDA (GB10), Vulkan (Radeon 8060S) and Metal (M4). The GPU matmuls move intermediate activations by up to 1.15e-2, so the per-stage parity tests use a 2e-2 tolerance off the CPU. The CPU tolerances and all end-to-end probability checks are unchanged. Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- README.md | 2 + docs/BENCHMARKS.md | 32 ++++ examples/cli/main.cpp | 4 +- src/ced.cpp | 337 +++++++++++++++++++++------------------- src/ced.hpp | 16 +- src/ced_runner.cpp | 128 ++++++++++++--- src/ced_runner.hpp | 47 +++++- src/model_loader.cpp | 72 +++++++-- src/model_loader.hpp | 17 +- tests/parity.hpp | 11 ++ tests/test_blocks.cpp | 5 +- tests/test_frontend.cpp | 9 +- tests/test_quant.cpp | 1 + 13 files changed, 473 insertions(+), 208 deletions(-) diff --git a/README.md b/README.md index 27aa21f..454f869 100644 --- a/README.md +++ b/README.md @@ -114,6 +114,8 @@ cmake --build build-shared -j To build for a GPU backend, forward its flag, e.g. `cmake -B build -DCED_GGML_METAL=ON`. +At runtime a GPU build picks the first GPU it finds and falls back to CPU. Set `CED_DEVICE` to choose: `CED_DEVICE=cpu` forces the CPU, and a device name such as `CUDA0`, `Vulkan0` or `MTL0` selects that device (`ced-cli info` prints the one in use). The weights are uploaded to the device once at load. If the device has no kernel for an op, that graph runs through ggml's scheduler with a CPU fallback. + --- ## Running inference diff --git a/docs/BENCHMARKS.md b/docs/BENCHMARKS.md index 5d4ca9b..204f9ab 100644 --- a/docs/BENCHMARKS.md +++ b/docs/BENCHMARKS.md @@ -41,6 +41,38 @@ classified per wall-second). before the first inference — a large practical gap for serving and CLI use. - **Parity holds throughout**: identical top-5 tags across all variants. +## GPU + +The same build runs on any ggml GPU backend (see `CED_DEVICE` in the README). +Numbers below are mean `ced-cli bench` latency, 30 iterations after 3 warmup, +next to the CPU of the same machine at 4 threads. The 36 s clip is split into +four ~10 s chunks, so it measures the multi-chunk path. + +| Device | Model | 6 s clip | 36 s clip | same host CPU, 36 s | +|---|---|--:|--:|--:| +| Apple M4, Metal | base f32 | 17.5 ms | 99.5 ms | 1151.7 ms | +| Apple M4, Metal | base q8_0 | 16.9 ms | 97.6 ms | 419.5 ms | +| Apple M4, Metal | tiny q8_0 | 5.7 ms | 30.2 ms | 63.6 ms | +| Radeon 8060S, Vulkan (RADV) | base f32 | 14.7 ms | 70.6 ms | 446.4 ms | +| Radeon 8060S, Vulkan (RADV) | base q8_0 | 10.3 ms | 48.7 ms | 425.1 ms | +| Radeon 8060S, Vulkan (RADV) | tiny q8_0 | 7.0 ms | 33.2 ms | 73.7 ms | +| NVIDIA GB10, CUDA 13 | base f32 | 11.3 ms | 62.6 ms | 2393.7 ms | +| NVIDIA GB10, CUDA 13 | base q8_0 | 10.5 ms | 61.8 ms | 2332.7 ms | +| NVIDIA GB10, CUDA 13 | tiny q8_0 | 9.9 ms | 55.4 ms | 289.6 ms | + +- **ced-base is 6x (Vulkan) to 12x (Metal) faster than the CPU of the same + machine.** The GB10 CPU column is from a container build that is slow on + CPU in general, so do not read the CUDA ratio as representative. +- **The GPU output keeps the CPU tags.** Over tiny and base at f32, f16 and + q8_0 on both clips, the top-5 tags are the same on every backend. f32 + probabilities agree with the CPU to about 2e-4. Quantized models move more + (up to ~1e-2 for tiny q8_0), because the CPU also quantizes the activations + of q8_0 matmuls and the GPU backends do not. +- **Small models are bound by the host.** The mel frontend runs on the CPU, + and for ced-tiny it is a large part of the GPU latency (on the GB10 about + 47 of the 55 ms on the 36 s clip). The GPU compute itself is a few + milliseconds per chunk. + ## Reproduce ```sh diff --git a/examples/cli/main.cpp b/examples/cli/main.cpp index a068311..248607d 100644 --- a/examples/cli/main.cpp +++ b/examples/cli/main.cpp @@ -29,6 +29,7 @@ static int cmd_info(const char* path) { std::printf(" mel : n_mels=%u n_fft=%u hop=%u win=%u sr=%u\n", c.n_mels, c.n_fft, c.hop_size, c.win_size, c.sample_rate); std::printf(" target_length : %u frames\n", c.target_length); + std::printf(" device : %s\n", m.device_name().c_str()); return 0; } @@ -122,7 +123,8 @@ static int cmd_bench(int argc, char** argv) { double median = ms[ms.size() / 2]; double minv = ms.front(), maxv = ms.back(); - std::printf("model=%s clip=%.2fs threads=%d iters=%d\n", model, clip_s, threads, iters); + std::printf("model=%s device=%s clip=%.2fs threads=%d iters=%d\n", model, + m.device_name().c_str(), clip_s, threads, iters); std::printf(" latency ms: min=%.2f median=%.2f mean=%.2f max=%.2f\n", minv, median, mean, maxv); std::printf(" RTF (clip_s/mean_s): %.1fx realtime (%.1f clips/s)\n", clip_s / (mean / 1000.0), 1000.0 / mean); diff --git a/src/ced.cpp b/src/ced.cpp index 2926121..a9c9c13 100644 --- a/src/ced.cpp +++ b/src/ced.cpp @@ -1,6 +1,5 @@ #include "ced.hpp" -#include "ced_runner.hpp" #include "mel.hpp" #include "ggml.h" @@ -11,9 +10,40 @@ namespace ced { +struct Ced::Embed { + ggml_tensor* bn = nullptr; // init_bn output [T, n_mels] + ggml_tensor* conv = nullptr; // patch_embed [OW, OH, OC] + ggml_tensor* pos = nullptr; // + positional [OW, OH, OC] + ggml_tensor* tokens = nullptr; // [OC, OW*OH] +}; + +struct Ced::Head { + ggml_tensor* enc = nullptr; // post final LayerNorm [C, N] + ggml_tensor* logits = nullptr; // [outputdim, 1] + ggml_tensor* probs = nullptr; // [outputdim, 1] +}; + bool Ced::load(const std::string& path) { if (!loader_.load(path)) return false; - return loader_.realize_weights_cpu(); + backend_ = std::make_unique(); + if (!backend_->ok() || !loader_.realize_weights(*backend_)) return false; + + const CedConfig& c = loader_.config(); + const float* bw = loader_.host_f32("encoder.init_bn.weight"); + const float* bb = loader_.host_f32("encoder.init_bn.bias"); + const float* bm = loader_.host_f32("encoder.init_bn.running_mean"); + const float* bv = loader_.host_f32("encoder.init_bn.running_var"); + if (!bw || !bb || !bm || !bv) { + std::fprintf(stderr, "ced: missing init_bn tensors\n"); + return false; + } + bn_scale_.resize(c.n_mels); + bn_shift_.resize(c.n_mels); + for (uint32_t m = 0; m < c.n_mels; ++m) { + bn_scale_[m] = bw[m] / std::sqrt(bv[m] + c.bn_eps); + bn_shift_[m] = bb[m] - bm[m] * bn_scale_[m]; + } + return true; } // LayerNorm over ne[0] (channel dim): y = norm(x, eps) * w + b. w/b are [C] and @@ -26,98 +56,154 @@ static ggml_tensor* layernorm(ggml_context* ctx, ggml_tensor* x, ggml_tensor* w, return y; } -bool Ced::forward_from_tokens(const std::vector& tokens, int n_tokens, - std::vector& enc_norm, - std::vector& logits, - std::vector& probs, int n_threads) { +// init_bn + patch_embed + positional + flatten: input_values [n_mels*T] -> +// tokens [C, n_tokens]. +Ced::Embed Ced::build_embed(ggml_context* ctx, std::vector& inputs, + const std::vector& input_values, int T) const { + const CedConfig& c = loader_.config(); + const int n_mels = (int)c.n_mels; + const ModelLoader& L = loader_; + auto W = [&](const std::string& n) { return L.tensor(n); }; + Embed e; + + // input_values leaf ne=[T, n_mels] (element (t,m) at t + m*T == mel-major flat) + ggml_tensor* x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, T, n_mels); + ggml_set_input(x); + inputs.push_back({x, input_values.data(), input_values.size() * sizeof(float)}); + + ggml_tensor* scaleT = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 1, n_mels); + ggml_set_input(scaleT); + inputs.push_back({scaleT, bn_scale_.data(), bn_scale_.size() * sizeof(float)}); + ggml_tensor* shiftT = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 1, n_mels); + ggml_set_input(shiftT); + inputs.push_back({shiftT, bn_shift_.data(), bn_shift_.size() * sizeof(float)}); + + e.bn = ggml_add(ctx, ggml_mul(ctx, x, scaleT), shiftT); // [T,n_mels] + + // patch_embed conv2d: kernel [KW,KH,IC,OC]=[16,16,1,768], input [W=T,H=mels,1,1] + ggml_tensor* bn4 = ggml_reshape_4d(ctx, e.bn, T, n_mels, 1, 1); + ggml_tensor* conv = ggml_conv_2d(ctx, W("encoder.patch_embed.proj.weight"), bn4, + c.patch_stride, c.patch_stride, 0, 0, 1, 1); // [OW,OH,OC,1] + ggml_tensor* pbias = + ggml_reshape_3d(ctx, W("encoder.patch_embed.proj.bias"), 1, 1, c.embed_dim); + e.conv = ggml_add(ctx, conv, pbias); + + const int OW = (int)e.conv->ne[0], OH = (int)e.conv->ne[1]; + + // positional. time_pos_embed is stored at the full target time grid + // (ne=[63,1,OC]); the reference slices it to the actual time grid + // (time_pos_embed[..., :OW]). For OW < 63 (any clip shorter than + // target_length, or the padded last chunk) we must slice too, else the + // broadcast add fails ggml_can_repeat. freq_pos_embed (ne=[1,OH,OC]) + // already broadcasts over OW, so it needs no slice. + ggml_tensor* tpe = W("encoder.time_pos_embed"); + ggml_tensor* tpe_v = tpe; + if (OW != (int)tpe->ne[0]) + tpe_v = ggml_cont(ctx, ggml_view_3d(ctx, tpe, OW, tpe->ne[1], tpe->ne[2], + tpe->nb[1], tpe->nb[2], 0)); + e.pos = ggml_add(ctx, e.conv, tpe_v); + e.pos = ggml_add(ctx, e.pos, W("encoder.freq_pos_embed")); + + // flatten freq-major to tokens: [OW,OH,OC] -> [OC,OW,OH] -> [OC, OW*OH] + ggml_tensor* perm = ggml_cont(ctx, ggml_permute(ctx, e.pos, 1, 2, 0, 3)); // [OC,OW,OH] + e.tokens = ggml_reshape_2d(ctx, perm, c.embed_dim, OW * OH); + return e; +} + +// 12 ViT blocks + final norm + mean-pool head over tokens [C, N]. +Ced::Head Ced::build_blocks(ggml_context* ctx, ggml_tensor* x, int N) const { const CedConfig& c = loader_.config(); const int C = (int)c.embed_dim; const int H = (int)c.num_heads; const int D = C / H; - const int N = n_tokens; const float eps_e = c.ln_eps_encoder; const float eps_h = c.ln_eps_head; - if ((int)tokens.size() != N * C) { - std::fprintf(stderr, "ced: tokens size %zu != %d*%d\n", tokens.size(), N, C); - return false; - } - - ModelLoader& L = loader_; + const ModelLoader& L = loader_; auto W = [&](const std::string& n) -> ggml_tensor* { return L.tensor(n); }; auto blk = [&](int i, const char* suffix) -> ggml_tensor* { return L.tensor("encoder.blocks." + std::to_string(i) + "." + suffix); }; + const float scale = 1.0f / std::sqrt((float)D); + for (int i = 0; i < (int)c.depth; ++i) { + // --- attention --- + ggml_tensor* n1 = layernorm(ctx, x, blk(i, "norm1.weight"), + blk(i, "norm1.bias"), eps_e); + ggml_tensor* qkv = ggml_mul_mat(ctx, blk(i, "attn.qkv.weight"), n1); // [3C,N] + qkv = ggml_add(ctx, qkv, blk(i, "attn.qkv.bias")); + + ggml_tensor* q = ggml_cont( + ctx, ggml_view_2d(ctx, qkv, C, N, qkv->nb[1], 0 * (size_t)C * sizeof(float))); + ggml_tensor* k = ggml_cont( + ctx, ggml_view_2d(ctx, qkv, C, N, qkv->nb[1], 1 * (size_t)C * sizeof(float))); + ggml_tensor* v = ggml_cont( + ctx, ggml_view_2d(ctx, qkv, C, N, qkv->nb[1], 2 * (size_t)C * sizeof(float))); + + q = ggml_permute(ctx, ggml_reshape_3d(ctx, q, D, H, N), 0, 2, 1, 3); // [D,N,H] + k = ggml_permute(ctx, ggml_reshape_3d(ctx, k, D, H, N), 0, 2, 1, 3); // [D,N,H] + ggml_tensor* kq = ggml_mul_mat(ctx, k, q); // [N,N,H] + kq = ggml_scale(ctx, kq, scale); + kq = ggml_soft_max(ctx, kq); + v = ggml_cont(ctx, ggml_permute(ctx, ggml_reshape_3d(ctx, v, D, H, N), + 1, 2, 0, 3)); // [N,D,H] + ggml_tensor* kqv = ggml_mul_mat(ctx, v, kq); // [D,N,H] + kqv = ggml_permute(ctx, kqv, 0, 2, 1, 3); // [D,H,N] + ggml_tensor* attn = ggml_cont_2d(ctx, kqv, C, N); + attn = ggml_mul_mat(ctx, blk(i, "attn.proj.weight"), attn); + attn = ggml_add(ctx, attn, blk(i, "attn.proj.bias")); + x = ggml_add(ctx, x, attn); + + // --- mlp --- + ggml_tensor* n2 = layernorm(ctx, x, blk(i, "norm2.weight"), + blk(i, "norm2.bias"), eps_e); + ggml_tensor* h = ggml_mul_mat(ctx, blk(i, "mlp.fc1.weight"), n2); + h = ggml_add(ctx, h, blk(i, "mlp.fc1.bias")); + h = ggml_gelu_erf(ctx, h); + h = ggml_mul_mat(ctx, blk(i, "mlp.fc2.weight"), h); + h = ggml_add(ctx, h, blk(i, "mlp.fc2.bias")); + x = ggml_add(ctx, x, h); + } + + Head out; + // final encoder norm + out.enc = layernorm(ctx, x, W("encoder.norm.weight"), W("encoder.norm.bias"), eps_e); + + // mean pool over tokens: [C,N] -> [C,1] + ggml_tensor* xt = ggml_cont(ctx, ggml_transpose(ctx, out.enc)); // [N,C] + ggml_tensor* pooled = ggml_sum_rows(ctx, xt); // [1,C] + pooled = ggml_scale(ctx, pooled, 1.0f / (float)N); + pooled = ggml_reshape_2d(ctx, pooled, C, 1); // [C,1] + + // head: LayerNorm(eps_head) -> Linear(C->outputdim) -> sigmoid + ggml_tensor* hn = ggml_norm(ctx, pooled, eps_h); + hn = ggml_mul(ctx, hn, W("outputlayer.0.weight")); + hn = ggml_add(ctx, hn, W("outputlayer.0.bias")); + out.logits = ggml_mul_mat(ctx, W("outputlayer.1.weight"), hn); // [outputdim,1] + out.logits = ggml_add(ctx, out.logits, W("outputlayer.1.bias")); + out.probs = ggml_sigmoid(ctx, out.logits); + return out; +} + +bool Ced::forward_from_tokens(const std::vector& tokens, int n_tokens, + std::vector& enc_norm, + std::vector& logits, + std::vector& probs, int n_threads) { + const int C = (int)loader_.config().embed_dim; + if ((int)tokens.size() != n_tokens * C) { + std::fprintf(stderr, "ced: tokens size %zu != %d*%d\n", tokens.size(), n_tokens, C); + return false; + } BuildFn build = [&](ggml_context* ctx, std::vector& inputs) -> std::vector { // input tokens leaf: ne = [C, N] - ggml_tensor* x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, C, N); + ggml_tensor* x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, C, n_tokens); ggml_set_input(x); inputs.push_back({x, tokens.data(), tokens.size() * sizeof(float)}); - - const float scale = 1.0f / std::sqrt((float)D); - for (int i = 0; i < (int)c.depth; ++i) { - // --- attention --- - ggml_tensor* n1 = layernorm(ctx, x, blk(i, "norm1.weight"), - blk(i, "norm1.bias"), eps_e); - ggml_tensor* qkv = ggml_mul_mat(ctx, blk(i, "attn.qkv.weight"), n1); // [3C,N] - qkv = ggml_add(ctx, qkv, blk(i, "attn.qkv.bias")); - - ggml_tensor* q = ggml_cont( - ctx, ggml_view_2d(ctx, qkv, C, N, qkv->nb[1], 0 * (size_t)C * sizeof(float))); - ggml_tensor* k = ggml_cont( - ctx, ggml_view_2d(ctx, qkv, C, N, qkv->nb[1], 1 * (size_t)C * sizeof(float))); - ggml_tensor* v = ggml_cont( - ctx, ggml_view_2d(ctx, qkv, C, N, qkv->nb[1], 2 * (size_t)C * sizeof(float))); - - q = ggml_permute(ctx, ggml_reshape_3d(ctx, q, D, H, N), 0, 2, 1, 3); // [D,N,H] - k = ggml_permute(ctx, ggml_reshape_3d(ctx, k, D, H, N), 0, 2, 1, 3); // [D,N,H] - ggml_tensor* kq = ggml_mul_mat(ctx, k, q); // [N,N,H] - kq = ggml_scale(ctx, kq, scale); - kq = ggml_soft_max(ctx, kq); - v = ggml_cont(ctx, ggml_permute(ctx, ggml_reshape_3d(ctx, v, D, H, N), - 1, 2, 0, 3)); // [N,D,H] - ggml_tensor* kqv = ggml_mul_mat(ctx, v, kq); // [D,N,H] - kqv = ggml_permute(ctx, kqv, 0, 2, 1, 3); // [D,H,N] - ggml_tensor* attn = ggml_cont_2d(ctx, kqv, C, N); - attn = ggml_mul_mat(ctx, blk(i, "attn.proj.weight"), attn); - attn = ggml_add(ctx, attn, blk(i, "attn.proj.bias")); - x = ggml_add(ctx, x, attn); - - // --- mlp --- - ggml_tensor* n2 = layernorm(ctx, x, blk(i, "norm2.weight"), - blk(i, "norm2.bias"), eps_e); - ggml_tensor* h = ggml_mul_mat(ctx, blk(i, "mlp.fc1.weight"), n2); - h = ggml_add(ctx, h, blk(i, "mlp.fc1.bias")); - h = ggml_gelu_erf(ctx, h); - h = ggml_mul_mat(ctx, blk(i, "mlp.fc2.weight"), h); - h = ggml_add(ctx, h, blk(i, "mlp.fc2.bias")); - x = ggml_add(ctx, x, h); - } - - // final encoder norm - ggml_tensor* enc = layernorm(ctx, x, W("encoder.norm.weight"), - W("encoder.norm.bias"), eps_e); - - // mean pool over tokens: [C,N] -> [C,1] - ggml_tensor* xt = ggml_cont(ctx, ggml_transpose(ctx, enc)); // [N,C] - ggml_tensor* pooled = ggml_sum_rows(ctx, xt); // [1,C] - pooled = ggml_scale(ctx, pooled, 1.0f / (float)N); - pooled = ggml_reshape_2d(ctx, pooled, C, 1); // [C,1] - - // head: LayerNorm(eps_head) -> Linear(C->outputdim) -> sigmoid - ggml_tensor* hn = ggml_norm(ctx, pooled, eps_h); - hn = ggml_mul(ctx, hn, W("outputlayer.0.weight")); - hn = ggml_add(ctx, hn, W("outputlayer.0.bias")); - ggml_tensor* lg = ggml_mul_mat(ctx, W("outputlayer.1.weight"), hn); // [outputdim,1] - lg = ggml_add(ctx, lg, W("outputlayer.1.bias")); - ggml_tensor* pr = ggml_sigmoid(ctx, lg); - - return {enc, lg, pr}; + Head h = build_blocks(ctx, x, n_tokens); + return {h.enc, h.logits, h.probs}; }; - std::vector> outs; - if (!run_graph(n_threads, build, outs)) return false; + if (!backend_->compute(n_threads, build, outs)) return false; enc_norm = std::move(outs[0]); logits = std::move(outs[1]); probs = std::move(outs[2]); @@ -126,14 +212,13 @@ bool Ced::forward_from_tokens(const std::vector& tokens, int n_tokens, bool Ced::mel_frontend(const std::vector& wav, std::vector& input_values, int& T) { - ggml_tensor* win = loader_.tensor("ced.mel_window"); - ggml_tensor* fb = loader_.tensor("ced.mel_filterbank"); + const float* win = loader_.host_f32("ced.mel_window"); + const float* fb = loader_.host_f32("ced.mel_filterbank"); if (!win || !fb) { std::fprintf(stderr, "ced: missing baked mel window/filterbank\n"); return false; } - mel_spectrogram(loader_.config(), (const float*)win->data, (const float*)fb->data, wav, - input_values, T); + mel_spectrogram(loader_.config(), win, fb, wav, input_values, T); return true; } @@ -142,83 +227,18 @@ bool Ced::embed_from_input_values(const std::vector& input_values, int T, std::vector& patch_embed, std::vector& pos_out, std::vector& tokens, int& n_tokens, int n_threads) { const CedConfig& c = loader_.config(); - const int n_mels = (int)c.n_mels; - if ((int)input_values.size() != n_mels * T) { + if ((int)input_values.size() != (int)c.n_mels * T) { std::fprintf(stderr, "ced: input_values size %zu != %d*%d\n", input_values.size(), - n_mels, T); - return false; - } - ModelLoader& L = loader_; - auto W = [&](const std::string& n) { return L.tensor(n); }; - - // Fold init_bn (BatchNorm2d, eval) into per-mel scale/shift on host. - auto data = [&](const char* n) -> const float* { - ggml_tensor* t = L.tensor(n); - return t ? (const float*)t->data : nullptr; - }; - const float* bw = data("encoder.init_bn.weight"); - const float* bb = data("encoder.init_bn.bias"); - const float* bm = data("encoder.init_bn.running_mean"); - const float* bv = data("encoder.init_bn.running_var"); - if (!bw || !bb || !bm || !bv) { - std::fprintf(stderr, "ced: missing init_bn tensors\n"); + (int)c.n_mels, T); return false; } - std::vector bn_scale(n_mels), bn_shift(n_mels); - for (int m = 0; m < n_mels; ++m) { - bn_scale[m] = bw[m] / std::sqrt(bv[m] + c.bn_eps); - bn_shift[m] = bb[m] - bm[m] * bn_scale[m]; - } - BuildFn build = [&](ggml_context* ctx, std::vector& inputs) -> std::vector { - // input_values leaf ne=[T, n_mels] (element (t,m) at t + m*T == mel-major flat) - ggml_tensor* x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, T, n_mels); - ggml_set_input(x); - inputs.push_back({x, input_values.data(), input_values.size() * sizeof(float)}); - - ggml_tensor* scaleT = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 1, n_mels); - ggml_set_input(scaleT); - inputs.push_back({scaleT, bn_scale.data(), bn_scale.size() * sizeof(float)}); - ggml_tensor* shiftT = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 1, n_mels); - ggml_set_input(shiftT); - inputs.push_back({shiftT, bn_shift.data(), bn_shift.size() * sizeof(float)}); - - ggml_tensor* bn = ggml_add(ctx, ggml_mul(ctx, x, scaleT), shiftT); // [T,n_mels] - - // patch_embed conv2d: kernel [KW,KH,IC,OC]=[16,16,1,768], input [W=T,H=mels,1,1] - ggml_tensor* bn4 = ggml_reshape_4d(ctx, bn, T, n_mels, 1, 1); - ggml_tensor* conv = ggml_conv_2d(ctx, W("encoder.patch_embed.proj.weight"), bn4, - c.patch_stride, c.patch_stride, 0, 0, 1, 1); // [OW,OH,OC,1] - ggml_tensor* pbias = - ggml_reshape_3d(ctx, W("encoder.patch_embed.proj.bias"), 1, 1, c.embed_dim); - conv = ggml_add(ctx, conv, pbias); - - const int OW = (int)conv->ne[0], OH = (int)conv->ne[1]; - - // positional. time_pos_embed is stored at the full target time grid - // (ne=[63,1,OC]); the reference slices it to the actual time grid - // (time_pos_embed[..., :OW]). For OW < 63 (any clip shorter than - // target_length, or the padded last chunk) we must slice too, else the - // broadcast add fails ggml_can_repeat. freq_pos_embed (ne=[1,OH,OC]) - // already broadcasts over OW, so it needs no slice. - ggml_tensor* tpe = W("encoder.time_pos_embed"); - ggml_tensor* tpe_v = tpe; - if (OW != (int)tpe->ne[0]) - tpe_v = ggml_cont(ctx, ggml_view_3d(ctx, tpe, OW, tpe->ne[1], tpe->ne[2], - tpe->nb[1], tpe->nb[2], 0)); - ggml_tensor* pos = ggml_add(ctx, conv, tpe_v); - pos = ggml_add(ctx, pos, W("encoder.freq_pos_embed")); - - // flatten freq-major to tokens: [OW,OH,OC] -> [OC,OW,OH] -> [OC, OW*OH] - ggml_tensor* perm = ggml_cont(ctx, ggml_permute(ctx, pos, 1, 2, 0, 3)); // [OC,OW,OH] - ggml_tensor* toks = ggml_reshape_2d(ctx, perm, c.embed_dim, OW * OH); - - return {bn, conv, pos, toks}; + Embed e = build_embed(ctx, inputs, input_values, T); + return {e.bn, e.conv, e.pos, e.tokens}; }; - std::vector> outs; - if (!run_graph(n_threads, build, outs)) return false; + if (!backend_->compute(n_threads, build, outs)) return false; init_bn_out = std::move(outs[0]); patch_embed = std::move(outs[1]); pos_out = std::move(outs[2]); @@ -252,14 +272,17 @@ bool Ced::classify(const std::vector& wav, std::vector& logits, for (int t = 0; t < clen; ++t) civ[(size_t)m * clen + t] = input_values[(size_t)m * T + t0 + t]; - std::vector bn, pe, po, tk; - int n_tokens = 0; - if (!embed_from_input_values(civ, clen, bn, pe, po, tk, n_tokens, n_threads)) return false; - std::vector en, lg, pr; - if (!forward_from_tokens(tk, n_tokens, en, lg, pr, n_threads)) return false; + BuildFn build = [&](ggml_context* ctx, + std::vector& inputs) -> std::vector { + Embed e = build_embed(ctx, inputs, civ, clen); + Head h = build_blocks(ctx, e.tokens, (int)e.tokens->ne[1]); + return {h.logits, h.probs}; + }; + std::vector> outs; + if (!backend_->compute(n_threads, build, outs)) return false; for (int i = 0; i < (int)c.outputdim; ++i) { - logit_acc[i] += lg[i]; - prob_acc[i] += pr[i]; + logit_acc[i] += outs[0][i]; + prob_acc[i] += outs[1][i]; } ++used; } diff --git a/src/ced.hpp b/src/ced.hpp index 3321110..f3e50d1 100644 --- a/src/ced.hpp +++ b/src/ced.hpp @@ -1,7 +1,9 @@ #pragma once +#include #include #include +#include "ced_runner.hpp" #include "model_loader.hpp" namespace ced { @@ -10,6 +12,8 @@ class Ced { public: bool load(const std::string& path); const CedConfig& config() const { return loader_.config(); } + // Compute device the model runs on ("cpu", "CUDA0", "Vulkan0", ...). + const std::string& device_name() const { return backend_->device_name(); } // Parity entry point: run the 12 ViT blocks + final norm + mean-pool head // from precomputed patch tokens. `tokens` is n_tokens * embed_dim, token- @@ -36,12 +40,22 @@ class Ced { std::vector& pos_out, std::vector& tokens, int& n_tokens, int n_threads = 4); - // End-to-end: waveform -> logits/probs over the 527 AudioSet classes. + // End-to-end: waveform -> logits/probs over the 527 AudioSet classes. Each + // target_length chunk runs as ONE graph (embed + blocks + head). bool classify(const std::vector& wav, std::vector& logits, std::vector& probs, int n_threads = 4); private: + struct Embed; + struct Head; + Embed build_embed(ggml_context* ctx, std::vector& inputs, + const std::vector& input_values, int T) const; + Head build_blocks(ggml_context* ctx, ggml_tensor* tokens, int n_tokens) const; + ModelLoader loader_; + std::unique_ptr backend_; + // init_bn (BatchNorm2d, eval) folded into a per-mel scale/shift at load. + std::vector bn_scale_, bn_shift_; }; } // namespace ced diff --git a/src/ced_runner.cpp b/src/ced_runner.cpp index 1786be4..5e60da6 100644 --- a/src/ced_runner.cpp +++ b/src/ced_runner.cpp @@ -5,21 +5,84 @@ #include "ggml-cpu.h" #include "ggml.h" +#include #include +#include namespace ced { -bool run_graph(int n_threads, const BuildFn& build, - std::vector>& outs) { - static ggml_backend_t backend = ggml_backend_cpu_init(); - if (!backend) { - std::fprintf(stderr, "ced: cpu backend init failed\n"); - return false; +// Generous metadata context (no_alloc: data lives in allocator buffers). The +// fused classify graph (embed + 12 blocks + head) is well under 1k nodes. +static constexpr size_t kGraphSize = 8192; + +struct Backend::Impl { + ggml_backend_t backend = nullptr; // selected device (GPU or CPU) + ggml_backend_t cpu_fallback = nullptr; // GPU path only, for unsupported ops + ggml_gallocr_t galloc = nullptr; // persistent, reused by every graph + ggml_backend_sched_t sched = nullptr; // created only if a graph needs it +}; + +static bool iequals(const std::string& a, const char* b) { + if (!b) return false; + size_t i = 0; + for (; i < a.size() && b[i]; ++i) + if (std::tolower((unsigned char)a[i]) != std::tolower((unsigned char)b[i])) + return false; + return i == a.size() && b[i] == '\0'; +} + +Backend::Backend() : impl_(new Impl()) { + const char* env = std::getenv("CED_DEVICE"); + const std::string want = env ? env : ""; + if (!iequals(want, "cpu")) { + for (size_t i = 0; i < ggml_backend_dev_count(); ++i) { + ggml_backend_dev_t dev = ggml_backend_dev_get(i); + const auto type = ggml_backend_dev_type(dev); + const char* name = ggml_backend_dev_name(dev); + const bool selected = want.empty() + ? (type == GGML_BACKEND_DEVICE_TYPE_GPU || + type == GGML_BACKEND_DEVICE_TYPE_IGPU) + : iequals(want, name); + if (!selected) continue; + impl_->backend = ggml_backend_dev_init(dev, nullptr); + if (impl_->backend) { + device_name_ = name ? name : ""; + is_cpu_ = type == GGML_BACKEND_DEVICE_TYPE_CPU; + break; + } + } + if (!want.empty() && !impl_->backend) + std::fprintf(stderr, "ced: CED_DEVICE=%s not found, using CPU\n", want.c_str()); } + if (!impl_->backend) { + impl_->backend = ggml_backend_cpu_init(); + device_name_ = "cpu"; + is_cpu_ = true; + } + if (!impl_->backend) { + std::fprintf(stderr, "ced: backend init failed\n"); + return; + } + if (!is_cpu_) impl_->cpu_fallback = ggml_backend_cpu_init(); +} + +Backend::~Backend() { + // Allocators before the backends they reference. + if (impl_->sched) ggml_backend_sched_free(impl_->sched); + if (impl_->galloc) ggml_gallocr_free(impl_->galloc); + if (impl_->cpu_fallback) ggml_backend_free(impl_->cpu_fallback); + if (impl_->backend) ggml_backend_free(impl_->backend); + delete impl_; +} + +bool Backend::ok() const { return impl_->backend != nullptr; } + +ggml_backend_t Backend::handle() const { return impl_->backend; } + +bool Backend::compute(int n_threads, const BuildFn& build, + std::vector>& outs) { + if (!impl_->backend) return false; - // Generous metadata context (no_alloc: data lives in gallocr buffers). The - // 12-block ViT graph is a few hundred nodes; size for headroom. - constexpr size_t kGraphSize = 8192; size_t mem = ggml_tensor_overhead() * (kGraphSize * 2) + ggml_graph_overhead_custom(kGraphSize, false); struct ggml_init_params ip { mem, nullptr, /*no_alloc=*/true }; @@ -34,7 +97,7 @@ bool run_graph(int n_threads, const BuildFn& build, } ggml_cgraph* gf = ggml_new_graph_custom(ctx, kGraphSize, false); - // Mark captured tensors as outputs so the gallocr does NOT reuse their + // Mark captured tensors as outputs so the allocator does NOT reuse their // buffers for downstream nodes (intermediates like enc_norm/logits feed // later ops and would otherwise read back freed/overwritten memory). for (ggml_tensor* t : capture) { @@ -42,23 +105,48 @@ bool run_graph(int n_threads, const BuildFn& build, ggml_build_forward_expand(gf, t); } - ggml_gallocr_t alloc = - ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!alloc || !ggml_gallocr_alloc_graph(alloc, gf)) { - std::fprintf(stderr, "ced: gallocr alloc failed\n"); - if (alloc) ggml_gallocr_free(alloc); + // A GPU graph goes through the scheduler only when the device lacks a + // kernel for one of its ops; otherwise it takes the gallocr path. + bool use_sched = false; + if (impl_->cpu_fallback) { + const int n = ggml_graph_n_nodes(gf); + for (int i = 0; i < n && !use_sched; ++i) + use_sched = !ggml_backend_supports_op(impl_->backend, ggml_graph_node(gf, i)); + } + + const int nt = n_threads > 0 ? n_threads : 4; + bool alloc_ok = false; + if (use_sched) { + if (!impl_->sched) { + ggml_backend_t backs[2] = {impl_->backend, impl_->cpu_fallback}; + impl_->sched = ggml_backend_sched_new(backs, nullptr, 2, kGraphSize, + /*parallel=*/false, /*op_offload=*/true); + } + if (impl_->sched) { + ggml_backend_cpu_set_n_threads(impl_->cpu_fallback, nt); + ggml_backend_sched_reset(impl_->sched); + alloc_ok = ggml_backend_sched_alloc_graph(impl_->sched, gf); + } + } else { + if (!impl_->galloc) + impl_->galloc = + ggml_gallocr_new(ggml_backend_get_default_buffer_type(impl_->backend)); + alloc_ok = impl_->galloc && ggml_gallocr_alloc_graph(impl_->galloc, gf); + } + if (!alloc_ok) { + std::fprintf(stderr, "ced: graph alloc failed\n"); ggml_free(ctx); return false; } - // Inputs are gallocr-allocated; upload host data now. for (const GraphInput& in : inputs) ggml_backend_tensor_set(in.t, in.data, 0, in.nbytes); - ggml_backend_cpu_set_n_threads(backend, n_threads > 0 ? n_threads : 4); - if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { + if (is_cpu_) ggml_backend_cpu_set_n_threads(impl_->backend, nt); + const ggml_status st = use_sched ? ggml_backend_sched_graph_compute(impl_->sched, gf) + : ggml_backend_graph_compute(impl_->backend, gf); + if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "ced: graph compute failed\n"); - ggml_gallocr_free(alloc); ggml_free(ctx); return false; } @@ -69,8 +157,6 @@ bool run_graph(int n_threads, const BuildFn& build, outs[i].resize(n); ggml_backend_tensor_get(capture[i], outs[i].data(), 0, n * sizeof(float)); } - - ggml_gallocr_free(alloc); ggml_free(ctx); return true; } diff --git a/src/ced_runner.hpp b/src/ced_runner.hpp index c73a7ed..8142cc9 100644 --- a/src/ced_runner.hpp +++ b/src/ced_runner.hpp @@ -1,15 +1,18 @@ #pragma once #include #include +#include #include struct ggml_context; struct ggml_tensor; +struct ggml_backend; +typedef struct ggml_backend* ggml_backend_t; namespace ced { -// An input leaf to be filled with host data AFTER the gallocr allocates the -// graph (gallocr decides the final address, so data is set post-alloc). +// An input leaf to be filled with host data AFTER the graph is allocated (the +// allocator decides the final address, so data is set post-alloc). struct GraphInput { ggml_tensor* t = nullptr; const void* data = nullptr; @@ -17,13 +20,43 @@ struct GraphInput { }; // build() creates input leaves (registering each in `inputs`), builds the graph, -// and returns the tensors to capture. run_graph allocates the graph on a CPU -// backend, uploads the inputs, computes, and reads each captured tensor's f32 -// data into `outs` (parallel to the returned vector). +// and returns the tensors to capture. Backend::compute allocates the graph, +// uploads the inputs, computes, and reads each captured tensor's f32 data into +// `outs` (parallel to the returned vector). using BuildFn = std::function(ggml_context*, std::vector&)>; -bool run_graph(int n_threads, const BuildFn& build, - std::vector>& outs); +// Persistent compute backend + reusable graph allocator, one per loaded model. +// +// Device: CED_DEVICE names a registry device ("cpu", "CUDA0", "Vulkan0", +// "Metal", case-insensitive); unset picks the first GPU / integrated GPU and +// falls back to CPU. +// +// Every graph runs through ONE persistent ggml_gallocr, so the compute buffer +// is kept across calls instead of being allocated and freed per graph. On a GPU +// device, a graph that contains an op the device has no kernel for is routed +// through ggml_backend_sched with a CPU fallback instead; graphs the device +// fully supports stay on the gallocr path. +class Backend { +public: + Backend(); + ~Backend(); + Backend(const Backend&) = delete; + Backend& operator=(const Backend&) = delete; + + bool ok() const; + bool is_cpu() const { return is_cpu_; } + const std::string& device_name() const { return device_name_; } + ggml_backend_t handle() const; + + bool compute(int n_threads, const BuildFn& build, + std::vector>& outs); + +private: + struct Impl; + Impl* impl_; + bool is_cpu_ = true; + std::string device_name_ = "cpu"; +}; } // namespace ced diff --git a/src/model_loader.cpp b/src/model_loader.cpp index 8228dc8..213ddf4 100644 --- a/src/model_loader.cpp +++ b/src/model_loader.cpp @@ -1,11 +1,14 @@ #include "model_loader.hpp" +#include "ced_runner.hpp" + #include "ggml-backend.h" #include "ggml-cpu.h" #include "ggml.h" #include "gguf.h" #include +#include namespace ced { @@ -29,24 +32,71 @@ static std::string kv_str(gguf_context* g, const char* k, const char* d = "") { ModelLoader::~ModelLoader() { if (weights_buf_) ggml_backend_buffer_free(weights_buf_); if (gguf_) gguf_free(gguf_); + if (dev_ctx_) ggml_free(dev_ctx_); if (ctx_) ggml_free(ctx_); } -bool ModelLoader::realize_weights_cpu() { +// Tensors read on the host (the mel frontend and the init_bn fold), kept as +// host copies so they stay readable once the weights move to a device. +static bool host_side(const std::string& n) { + return n.rfind("ced.mel_", 0) == 0 || n.rfind("encoder.init_bn.", 0) == 0; +} + +bool ModelLoader::realize_weights(const Backend& backend) { if (weights_buf_) return true; // idempotent - if (!ctx_) return false; - // The GGUF was loaded no_alloc=false, so every tensor's data lives in one - // contiguous ctx mem buffer. Wrap that exact memory as a CPU backend buffer - // (zero-copy) and point every tensor's ->buffer at it; graphs then reference - // the loader's tensors directly as already-allocated leaves. - void* base = ggml_get_mem_buffer(ctx_); - size_t size = ggml_get_mem_size(ctx_); - weights_buf_ = ggml_backend_cpu_buffer_from_ptr(base, size); - if (!weights_buf_) return false; - for (auto& kv : tensors_) kv.second->buffer = weights_buf_; + if (!ctx_ || !backend.ok()) return false; + + for (auto& kv : tensors_) { + ggml_tensor* t = kv.second; + if (!host_side(kv.first) || t->type != GGML_TYPE_F32) continue; + const float* p = (const float*)t->data; + host_[kv.first].assign(p, p + ggml_nelements(t)); + } + + if (backend.is_cpu()) { + // The GGUF was loaded no_alloc=false, so every tensor's data lives in + // one contiguous ctx mem buffer. Wrap that exact memory as a CPU + // backend buffer (zero-copy) and point every tensor's ->buffer at it; + // graphs then reference the loader's tensors directly as leaves. + void* base = ggml_get_mem_buffer(ctx_); + size_t size = ggml_get_mem_size(ctx_); + weights_buf_ = ggml_backend_cpu_buffer_from_ptr(base, size); + if (!weights_buf_) return false; + for (auto& kv : tensors_) kv.second->buffer = weights_buf_; + return true; + } + + // Device: mirror every tensor into a no_alloc context, allocate them all in + // one device buffer, upload once, then drop the host copy. + struct ggml_init_params ip { ggml_tensor_overhead() * (tensors_.size() + 1), nullptr, + /*no_alloc=*/true }; + dev_ctx_ = ggml_init(ip); + if (!dev_ctx_) return false; + std::unordered_map dev; + for (auto& kv : tensors_) { + ggml_tensor* d = ggml_dup_tensor(dev_ctx_, kv.second); + ggml_set_name(d, kv.first.c_str()); + dev[kv.first] = d; + } + weights_buf_ = ggml_backend_alloc_ctx_tensors(dev_ctx_, backend.handle()); + if (!weights_buf_) { + std::fprintf(stderr, "ced: device weight allocation failed\n"); + return false; + } + ggml_backend_buffer_set_usage(weights_buf_, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); + for (auto& kv : tensors_) + ggml_backend_tensor_set(dev[kv.first], kv.second->data, 0, ggml_nbytes(kv.second)); + tensors_ = std::move(dev); + ggml_free(ctx_); + ctx_ = nullptr; return true; } +const float* ModelLoader::host_f32(const std::string& n) const { + auto it = host_.find(n); + return it == host_.end() ? nullptr : it->second.data(); +} + bool ModelLoader::load(const std::string& path) { struct gguf_init_params p { /*no_alloc=*/false, /*ctx=*/&ctx_ }; gguf_ = gguf_init_from_file(path.c_str(), p); diff --git a/src/model_loader.hpp b/src/model_loader.hpp index c997966..7c406bf 100644 --- a/src/model_loader.hpp +++ b/src/model_loader.hpp @@ -12,6 +12,8 @@ typedef struct ggml_backend_buffer* ggml_backend_buffer_t; namespace ced { +class Backend; + // All config is read from the GGUF (metadata-driven); nothing is hardcoded. struct CedConfig { std::string arch; @@ -38,17 +40,24 @@ class ModelLoader { bool load(const std::string& path); const CedConfig& config() const { return cfg_; } ggml_tensor* tensor(const std::string& name) const; // nullptr if absent - ggml_context* ggml_ctx() const { return ctx_; } - // Give every weight tensor a CPU backend buffer (zero-copy: wraps the ctx - // mem buffer) so graphs can reference them directly as leaves. Idempotent. - bool realize_weights_cpu(); + // Make every weight usable as a graph leaf on `backend`. CPU: zero-copy + // (wraps the ctx mem buffer). GPU: one upload into a device buffer at load, + // after which the host copy is released; tensor() then returns the device + // tensors. Idempotent. + bool realize_weights(const Backend& backend); + // Host f32 copy of a small tensor the CPU side reads directly (mel window, + // mel filterbank, init_bn stats). Valid after realize_weights, on any + // device. nullptr if absent or not f32. + const float* host_f32(const std::string& name) const; private: CedConfig cfg_; gguf_context* gguf_ = nullptr; ggml_context* ctx_ = nullptr; + ggml_context* dev_ctx_ = nullptr; // no_alloc mirror of ctx_ (GPU path) ggml_backend_buffer_t weights_buf_ = nullptr; std::unordered_map tensors_; + std::unordered_map> host_; }; } // namespace ced diff --git a/tests/parity.hpp b/tests/parity.hpp index 1a71dc0..cb6a2e5 100644 --- a/tests/parity.hpp +++ b/tests/parity.hpp @@ -62,4 +62,15 @@ inline bool compare(const std::vector& got, const std::vector& ref return ok; } +// Tolerance for an intermediate stage (activations of magnitude ~1-10). The +// reference values are CPU f32 and are gated tightly on CPU. GPU backends run +// their matmuls at lower internal precision (ggml's Metal matmul stages tiles +// in half), which moves intermediates by up to ~1e-2 while the final +// probabilities still agree to ~1e-4, so off-CPU the stage gate is widened. +// The end-to-end probability gates are the same on every device. +inline float stage_tol(const std::string& device, float cpu_tol) { + const float gpu_tol = 2e-2f; + return device == "cpu" ? cpu_tol : (cpu_tol > gpu_tol ? cpu_tol : gpu_tol); +} + } // namespace cedtest diff --git a/tests/test_blocks.cpp b/tests/test_blocks.cpp index 548144e..58af5c3 100644 --- a/tests/test_blocks.cpp +++ b/tests/test_blocks.cpp @@ -22,6 +22,7 @@ int main(int argc, char** argv) { std::fprintf(stderr, "FAIL: could not load model %s\n", model.c_str()); return 1; } + std::fprintf(stderr, "device: %s\n", m.device_name().c_str()); std::vector tokens; std::vector shape; @@ -44,11 +45,11 @@ int main(int argc, char** argv) { std::vector rshape; if (cedtest::load_baseline(baseline, "enc_norm", ref, rshape)) - ok &= cedtest::compare(enc, ref, "enc_norm", 2e-3f, 0.0f); + ok &= cedtest::compare(enc, ref, "enc_norm", cedtest::stage_tol(m.device_name(), 2e-3f), 0.0f); else ok = false; if (cedtest::load_baseline(baseline, "logits", ref, rshape)) - ok &= cedtest::compare(logits, ref, "logits", 2e-3f, 0.0f); + ok &= cedtest::compare(logits, ref, "logits", cedtest::stage_tol(m.device_name(), 2e-3f), 0.0f); else ok = false; if (cedtest::load_baseline(baseline, "probs", ref, rshape)) diff --git a/tests/test_frontend.cpp b/tests/test_frontend.cpp index 7fe357e..db927e6 100644 --- a/tests/test_frontend.cpp +++ b/tests/test_frontend.cpp @@ -20,6 +20,7 @@ int main(int argc, char** argv) { std::fprintf(stderr, "FAIL: load %s\n", model.c_str()); return 1; } + std::fprintf(stderr, "device: %s\n", m.device_name().c_str()); std::vector wav, ref; std::vector shp; @@ -44,16 +45,16 @@ int main(int argc, char** argv) { return 1; } if (cedtest::load_baseline(baseline, "init_bn_out", ref, shp)) - ok &= cedtest::compare(init_bn_out, ref, "init_bn_out", 2e-3f, 0.0f); + ok &= cedtest::compare(init_bn_out, ref, "init_bn_out", cedtest::stage_tol(m.device_name(), 2e-3f), 0.0f); else ok = false; if (cedtest::load_baseline(baseline, "patch_embed", ref, shp)) - ok &= cedtest::compare(patch_embed, ref, "patch_embed", 3e-3f, 0.0f); + ok &= cedtest::compare(patch_embed, ref, "patch_embed", cedtest::stage_tol(m.device_name(), 3e-3f), 0.0f); else ok = false; if (cedtest::load_baseline(baseline, "pos_out", ref, shp)) - ok &= cedtest::compare(pos_out, ref, "pos_out", 3e-3f, 0.0f); + ok &= cedtest::compare(pos_out, ref, "pos_out", cedtest::stage_tol(m.device_name(), 3e-3f), 0.0f); else ok = false; if (cedtest::load_baseline(baseline, "tokens_in", ref, shp)) - ok &= cedtest::compare(tokens, ref, "tokens_in", 3e-3f, 0.0f); + ok &= cedtest::compare(tokens, ref, "tokens_in", cedtest::stage_tol(m.device_name(), 3e-3f), 0.0f); else ok = false; // --- end-to-end: waveform -> probs --- diff --git a/tests/test_quant.cpp b/tests/test_quant.cpp index 9e80236..b03d751 100644 --- a/tests/test_quant.cpp +++ b/tests/test_quant.cpp @@ -30,6 +30,7 @@ int main(int argc, char** argv) { ced::Ced m; if (!m.load(model)) { std::fprintf(stderr, "FAIL: load %s\n", model.c_str()); return 1; } + std::fprintf(stderr, "device: %s\n", m.device_name().c_str()); std::vector wav, refp; std::vector shp; From b7dc30dac05a8a6ed44211684cf3d6731d585790 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 28 Sep 2026 11:09:47 +0000 Subject: [PATCH 3/4] build: allow embedding ced.cpp in another ggml project parakeet.cpp adds ced.cpp as a submodule for scene-sound classification, which needs ced to build inside a host project that already has its own ggml target and dr_wav implementation. Guard the GGML_* cache variable forwarding and the ggml submodule with a CED_TOP_LEVEL / TARGET ggml check so an embedding project's own ggml config and build are left alone. Add CED_EXTERNAL_DR_WAV so the host's dr_wav implementation can be reused instead of compiling a second copy into the same binary, which would otherwise fail to link with duplicate drwav_ symbols. tests/CMakeLists.txt now resolves paths from PROJECT_SOURCE_DIR instead of CMAKE_SOURCE_DIR so the ctest fixtures still resolve when ced is not the top-level project. Also add ced_capi_classify_pcm_probs, which writes every class score in class-index order with no sorting and no allocation, for callers (such as parakeet.cpp's streaming scene detector) that want the raw distribution on every window instead of a sorted top-k. Assisted-by: Claude:claude-sonnet-5 [Claude Code] --- CMakeLists.txt | 37 +++++++++++++++++++++++++------------ README.md | 5 ++++- include/ced_capi.h | 7 +++++++ src/audio_io.cpp | 4 +++- src/ced_capi.cpp | 11 +++++++++++ tests/CMakeLists.txt | 18 +++++++++--------- tests/test_capi.cpp | 24 ++++++++++++++++++++++++ 7 files changed, 83 insertions(+), 23 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 7c77723..52c4bfc 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -15,21 +15,31 @@ option(CED_GGML_METAL "Forward GGML_METAL" OFF) option(CED_GGML_VULKAN "Forward GGML_VULKAN" OFF) option(CED_GGML_HIP "Forward GGML_HIP (ROCm)" OFF) -set(GGML_CUDA ${CED_GGML_CUDA} CACHE BOOL "" FORCE) -set(GGML_METAL ${CED_GGML_METAL} CACHE BOOL "" FORCE) -set(GGML_VULKAN ${CED_GGML_VULKAN} CACHE BOOL "" FORCE) -set(GGML_HIP ${CED_GGML_HIP} CACHE BOOL "" FORCE) +set(CED_TOP_LEVEL OFF) +if(CMAKE_SOURCE_DIR STREQUAL CMAKE_CURRENT_SOURCE_DIR) + set(CED_TOP_LEVEL ON) +endif() + +option(CED_EXTERNAL_DR_WAV "Use the embedding project's dr_wav implementation" OFF) + +if(CED_TOP_LEVEL) + set(GGML_CUDA ${CED_GGML_CUDA} CACHE BOOL "" FORCE) + set(GGML_METAL ${CED_GGML_METAL} CACHE BOOL "" FORCE) + set(GGML_VULKAN ${CED_GGML_VULKAN} CACHE BOOL "" FORCE) + set(GGML_HIP ${CED_GGML_HIP} CACHE BOOL "" FORCE) -# Free perf wins (per parakeet.cpp / rt-detr.cpp): -march=native + tinyBLAS. -if(NOT DEFINED GGML_NATIVE) - set(GGML_NATIVE ON CACHE BOOL "ggml: optimize for the current system" FORCE) + # Free perf wins (per parakeet.cpp / rt-detr.cpp): -march=native + tinyBLAS. + if(NOT DEFINED GGML_NATIVE) + set(GGML_NATIVE ON CACHE BOOL "ggml: optimize for the current system" FORCE) + endif() + if(NOT DEFINED GGML_LLAMAFILE) + set(GGML_LLAMAFILE ON CACHE BOOL "ggml: use LLAMAFILE (tinyBLAS SGEMM)" FORCE) + endif() endif() -if(NOT DEFINED GGML_LLAMAFILE) - set(GGML_LLAMAFILE ON CACHE BOOL "ggml: use LLAMAFILE (tinyBLAS SGEMM)" FORCE) +if(NOT TARGET ggml) + add_subdirectory(third_party/ggml) endif() -add_subdirectory(third_party/ggml) - set(CED_SRC src/model_loader.cpp src/ced_runner.cpp @@ -45,7 +55,10 @@ if(CED_SHARED) else() add_library(ced STATIC ${CED_SRC}) endif() -target_include_directories(ced PUBLIC include PRIVATE src ${CMAKE_SOURCE_DIR}/third_party) +target_include_directories(ced PUBLIC include PRIVATE src ${CMAKE_CURRENT_SOURCE_DIR}/third_party) +if(CED_EXTERNAL_DR_WAV) + target_compile_definitions(ced PRIVATE CED_EXTERNAL_DR_WAV) +endif() target_compile_definitions(ced PUBLIC $<$:_USE_MATH_DEFINES>) target_link_libraries(ced PUBLIC ggml) diff --git a/README.md b/README.md index 454f869..83e185f 100644 --- a/README.md +++ b/README.md @@ -111,9 +111,12 @@ cmake --build build-shared -j | `CED_GGML_METAL` | OFF | Forward GGML_METAL to the submodule | | `CED_GGML_VULKAN` | OFF | Forward GGML_VULKAN to the submodule | | `CED_GGML_HIP` | OFF | Forward GGML_HIP (ROCm) to the submodule | +| `CED_EXTERNAL_DR_WAV` | OFF | Do not compile dr_wav; the embedding project provides it | To build for a GPU backend, forward its flag, e.g. `cmake -B build -DCED_GGML_METAL=ON`. +When ced.cpp is added with `add_subdirectory` by a project that already has a `ggml` target, it reuses that ggml. + At runtime a GPU build picks the first GPU it finds and falls back to CPU. Set `CED_DEVICE` to choose: `CED_DEVICE=cpu` forces the CPU, and a device name such as `CUDA0`, `Vulkan0` or `MTL0` selects that device (`ced-cli info` prints the one in use). The weights are uploaded to the device once at load. If the device has no kernel for an op, that graph runs through ggml's scheduler with a CPU fallback. --- @@ -158,7 +161,7 @@ if (json) { printf("%s\n", json); ced_capi_free_string(json); } ced_capi_free(ctx); ``` -The per-PCM entry points take an arbitrary mono window, so a realtime consumer can call them on a sliding buffer for live recognition. There is also a struct-array variant (`ced_capi_classify_pcm`) and a WAV-path variant (`ced_capi_classify_path_json`). See `include/ced_capi.h` for the full API. +The per-PCM entry points take an arbitrary mono window, so a realtime consumer can call them on a sliding buffer for live recognition. There is also a struct-array variant (`ced_capi_classify_pcm`), a WAV-path variant (`ced_capi_classify_path_json`), and `ced_capi_classify_pcm_probs`, which writes every class score in class-index order (no sorting, no allocation) for callers that want the raw distribution. See `include/ced_capi.h` for the full API. --- diff --git a/include/ced_capi.h b/include/ced_capi.h index 36b7c23..1d3cdb4 100644 --- a/include/ced_capi.h +++ b/include/ced_capi.h @@ -66,6 +66,13 @@ typedef struct { int ced_capi_classify_pcm(ced_ctx* ctx, const float* samples, int n_samples, int sample_rate, ced_tag* out, int max_tags); +// All class scores for mono float PCM, in class-index order (no sorting, no +// allocation). Writes min(n_out, ced_capi_num_classes(ctx)) floats into `out` +// and returns that count, or -1 on error (see ced_capi_last_error). Resamples +// like ced_capi_classify_pcm when `sample_rate` differs from the model rate. +int ced_capi_classify_pcm_probs(ced_ctx* ctx, const float* samples, int n_samples, + int sample_rate, float* out, int n_out); + // Free a string returned by a *_json function. Safe on NULL. void ced_capi_free_string(char* s); diff --git a/src/audio_io.cpp b/src/audio_io.cpp index 8decdcb..4f105c7 100644 --- a/src/audio_io.cpp +++ b/src/audio_io.cpp @@ -1,4 +1,6 @@ -#define DR_WAV_IMPLEMENTATION +#ifndef CED_EXTERNAL_DR_WAV +#define DR_WAV_IMPLEMENTATION // an embedding project may provide it instead +#endif #include "dr_wav.h" #include "audio_io.hpp" diff --git a/src/ced_capi.cpp b/src/ced_capi.cpp index d1a7ea5..0ce6c8d 100644 --- a/src/ced_capi.cpp +++ b/src/ced_capi.cpp @@ -175,6 +175,17 @@ int ced_capi_classify_pcm(ced_ctx* ctx, const float* samples, int n_samples, int return n; } +int ced_capi_classify_pcm_probs(ced_ctx* ctx, const float* samples, int n_samples, + int sample_rate, float* out, int n_out) { + auto* c = reinterpret_cast(ctx); + if (!c || !samples || !out || n_out <= 0) return -1; + std::vector probs; + if (!do_classify(c, samples, n_samples, sample_rate, probs)) return -1; + const int n = std::min(n_out, (int)probs.size()); + std::memcpy(out, probs.data(), (size_t)n * sizeof(float)); + return n; +} + void ced_capi_free_string(char* s) { std::free(s); } } // extern "C" diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 8cf000b..ba9e679 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -1,13 +1,13 @@ -set(CED_MODEL_GGUF "${CMAKE_SOURCE_DIR}/models/ced-base-f32.gguf") -set(CED_BASELINE_GGUF "${CMAKE_SOURCE_DIR}/tests/fixtures/ced-base.baseline.gguf") +set(CED_MODEL_GGUF "${PROJECT_SOURCE_DIR}/models/ced-base-f32.gguf") +set(CED_BASELINE_GGUF "${PROJECT_SOURCE_DIR}/tests/fixtures/ced-base.baseline.gguf") foreach(t blocks frontend quant capi) add_executable(test_${t} test_${t}.cpp) target_link_libraries(test_${t} PRIVATE ced) target_include_directories(test_${t} PRIVATE - ${CMAKE_SOURCE_DIR}/src - ${CMAKE_SOURCE_DIR}/include - ${CMAKE_SOURCE_DIR}/third_party/ggml/include) + ${PROJECT_SOURCE_DIR}/src + ${PROJECT_SOURCE_DIR}/include + ${PROJECT_SOURCE_DIR}/third_party/ggml/include) endforeach() add_test(NAME capi COMMAND test_capi "${CED_MODEL_GGUF}" "${CED_BASELINE_GGUF}") @@ -20,9 +20,9 @@ add_test(NAME frontend COMMAND test_frontend "${CED_MODEL_GGUF}" "${CED_BASELINE # python scripts/gen_ced_baseline.py --model mispeech/ced-base \ # --out tests/fixtures/ced-base-short.baseline.gguf --n-samples 64000 add_test(NAME frontend_short COMMAND test_frontend "${CED_MODEL_GGUF}" - "${CMAKE_SOURCE_DIR}/tests/fixtures/ced-base-short.baseline.gguf") + "${PROJECT_SOURCE_DIR}/tests/fixtures/ced-base-short.baseline.gguf") # End-to-end quantization parity (probs close + top-5 labels unchanged). -add_test(NAME e2e_f32 COMMAND test_quant "${CMAKE_SOURCE_DIR}/models/ced-base-f32.gguf" "${CED_BASELINE_GGUF}" 1e-3) -add_test(NAME e2e_f16 COMMAND test_quant "${CMAKE_SOURCE_DIR}/models/ced-base-f16.gguf" "${CED_BASELINE_GGUF}" 5e-3) -add_test(NAME e2e_q8 COMMAND test_quant "${CMAKE_SOURCE_DIR}/models/ced-base-q8_0.gguf" "${CED_BASELINE_GGUF}" 3e-2) +add_test(NAME e2e_f32 COMMAND test_quant "${PROJECT_SOURCE_DIR}/models/ced-base-f32.gguf" "${CED_BASELINE_GGUF}" 1e-3) +add_test(NAME e2e_f16 COMMAND test_quant "${PROJECT_SOURCE_DIR}/models/ced-base-f16.gguf" "${CED_BASELINE_GGUF}" 5e-3) +add_test(NAME e2e_q8 COMMAND test_quant "${PROJECT_SOURCE_DIR}/models/ced-base-q8_0.gguf" "${CED_BASELINE_GGUF}" 3e-2) diff --git a/tests/test_capi.cpp b/tests/test_capi.cpp index f835b5b..1a15a47 100644 --- a/tests/test_capi.cpp +++ b/tests/test_capi.cpp @@ -54,6 +54,30 @@ int main(int argc, char** argv) { ced_capi_free_string(json); } + // all-scores path: same numbers as the sorted top-k path, in index order + { + const int nc = ced_capi_num_classes(ctx); + std::vector all(nc, -1.0f); + int w = ced_capi_classify_pcm_probs(ctx, wav.data(), (int)wav.size(), 16000, + all.data(), nc); + ok &= (w == nc); + std::vector sorted(nc); + int ns = ced_capi_classify_pcm(ctx, wav.data(), (int)wav.size(), 16000, + sorted.data(), nc); + ok &= (ns == nc); + for (int i = 0; i < ns; ++i) ok &= (all[sorted[i].index] == sorted[i].score); + // short buffer: only the first n_out classes + float three[3]; + ok &= (ced_capi_classify_pcm_probs(ctx, wav.data(), (int)wav.size(), 16000, + three, 3) == 3); + ok &= (three[0] == all[0] && three[2] == all[2]); + // errors + ok &= (ced_capi_classify_pcm_probs(ctx, nullptr, 10, 16000, three, 3) == -1); + ok &= (ced_capi_classify_pcm_probs(ctx, wav.data(), (int)wav.size(), 16000, + nullptr, 3) == -1); + std::fprintf(stderr, "classify_pcm_probs: %s\n", ok ? "ok" : "FAIL"); + } + ced_capi_free(ctx); std::fprintf(stderr, "%s\n", ok ? "PASS" : "FAIL"); return ok ? 0 : 1; From e1a3cfa365feeb429cfd0373abb689647552c1ed Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 28 Sep 2026 11:15:22 +0000 Subject: [PATCH 4/4] build: fix ced-cli's include paths and dr_wav linkage when embedded examples/cli/CMakeLists.txt still resolved its include dirs from CMAKE_SOURCE_DIR, which is the outermost project's root, not ced's own. CED_BUILD_CLI defaults ON, so any host project doing add_subdirectory(ced) without explicitly turning it off got a build break, the same class of bug the previous commit fixed elsewhere. Switch to PROJECT_SOURCE_DIR. Fixing that path uncovered a second problem: with CED_EXTERNAL_DR_WAV on, libced.a is built without a dr_wav implementation, on the assumption that the embedding host's own binary provides one. ced-cli is not that binary, it is a separate executable built by ced.cpp itself, so it linked with undefined drwav_* references. Give ced-cli its own dr_wav_impl.cpp, compiled in only when CED_EXTERNAL_DR_WAV is set, so it stays a self-contained tool regardless of how the library is configured. Assisted-by: Claude:claude-sonnet-5 [Claude Code] --- examples/cli/CMakeLists.txt | 10 ++++++++-- examples/cli/dr_wav_impl.cpp | 8 ++++++++ 2 files changed, 16 insertions(+), 2 deletions(-) create mode 100644 examples/cli/dr_wav_impl.cpp diff --git a/examples/cli/CMakeLists.txt b/examples/cli/CMakeLists.txt index e635df7..e522bb6 100644 --- a/examples/cli/CMakeLists.txt +++ b/examples/cli/CMakeLists.txt @@ -1,5 +1,11 @@ add_executable(ced-cli main.cpp) +if(CED_EXTERNAL_DR_WAV) + # libced.a was built without a dr_wav implementation (the embedding host + # supplies one for its own binary), but ced-cli is a separate executable + # that still needs the real symbols. + target_sources(ced-cli PRIVATE dr_wav_impl.cpp) +endif() target_link_libraries(ced-cli PRIVATE ced) target_include_directories(ced-cli PRIVATE - ${CMAKE_SOURCE_DIR}/src - ${CMAKE_SOURCE_DIR}/third_party) + ${PROJECT_SOURCE_DIR}/src + ${PROJECT_SOURCE_DIR}/third_party) diff --git a/examples/cli/dr_wav_impl.cpp b/examples/cli/dr_wav_impl.cpp new file mode 100644 index 0000000..c112740 --- /dev/null +++ b/examples/cli/dr_wav_impl.cpp @@ -0,0 +1,8 @@ +// ced-cli is its own executable, never the embedding host's binary, so it +// needs a real dr_wav implementation even when the `ced` library is built +// with CED_EXTERNAL_DR_WAV (which only promises that *the host's* binary +// supplies one). Compiled into ced-cli only when that option is set; the +// non-embedded build already gets the implementation from src/audio_io.cpp +// via libced.a. +#define DR_WAV_IMPLEMENTATION +#include "dr_wav.h"