From 341155845d0112539c88f423297afa2474b849f0 Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Sun, 27 Sep 2026 14:58:30 -0300 Subject: [PATCH] Pin llama.cpp 00adc2b: HRX decode-split publishes next_q8 after a global barrier (#123, #140) Brings 1bit-MONSTER/llama.cpp#28 and #29: in every decode-split reduce_fused variant the barrier between the reduce's global output stores and pack_completed_q8's output loads was kernel.barrier (LDS only), so the next_q8 copy could be packed from stale output. It is now kernel.barrier followed by the LDS barrier (5 sites, ops and qwen3_moe corpora; #29 restores the LDS fence #28 dropped and adds the missing global barrier to the undispatched standalone reduce_f32 kernels). The pin moves fa226f9 -> 00adc2b, fast-forward. Co-Authored-By: Claude Opus 5.5 --- registry/architectures.json | 2 +- third_party/llama.cpp | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/registry/architectures.json b/registry/architectures.json index 7fcec4c..f38f0a2 100644 --- a/registry/architectures.json +++ b/registry/architectures.json @@ -2,7 +2,7 @@ "about": "HF architecture -> GGUF architecture and the backends whose code accepts it. Generated by tools/registry_build.py from the pinned sources; do not edit.", "sources": { "llama.cpp (vulkan)": "cecf3ee01d9d99378e98bfea95f51cac714b04b8", - "llama.cpp (hrx)": "fa226f9377a3025a171aea3eb8839c2f9f052d37", + "llama.cpp (hrx)": "00adc2b9d6676f39cb3fcca8b5ad97621e653fae", "zinc": "29bc350ac4cbf9f110ec628b9e177ea04ac816fb" }, "counts": { diff --git a/third_party/llama.cpp b/third_party/llama.cpp index fa226f9..00adc2b 160000 --- a/third_party/llama.cpp +++ b/third_party/llama.cpp @@ -1 +1 @@ -Subproject commit fa226f9377a3025a171aea3eb8839c2f9f052d37 +Subproject commit 00adc2b9d6676f39cb3fcca8b5ad97621e653fae