opencoti-llamafile / patches /0099-dca-gemma4-wiring.patch
ManniX-ITA's picture
Upload folder using huggingface_hub
b69d9d8 verified
Raw
History Blame
7.25 kB
opencoti patch 0099 — DCA: wire Gemma-4 builders (bug-2118, task #626 unblock)
`--dca on` was a SILENT NO-OP on all Gemma-4 archs: 0078-dca.patch wired only
the Qwen builders; dca.cpp's iswa build_attn_dca overload sat uncalled by
gemma4.cpp / gemma4-assistant.cpp. Consequence: every historical Gemma x DCA
eval measured DCA-off (contamination catalog: docs/evaluations bug-2118 notes).
Wiring: build_attn_inp_dca on the iswa base cache; per-layer dca_l gates
GLOBAL (non-SWA) layers only; Q/K rope moves inside build_attn_dca via the
dca_rope bundle (chunked-regime rope must happen per-band); the assistant's
Q-only shared-KV path passes (Qcur, nullptr, nullptr). SWA layers and DCA-off
are byte-identical (else-branch keeps the original build_attn call).
Gate (bs2, A4B + assistant, 32k, 24k prompt): INTRA regime (chunk==ctx) text
byte-identical to DCA-off with identical accept (f16 0.677 both; q8q4 text
identical); cross-chunk f16 accept 0.939, needle PRESENT.
--- a/llama.cpp/src/models/gemma4.cpp
+++ b/llama.cpp/src/models/gemma4.cpp
@@ -1,4 +1,6 @@
#include "models.h"
+#include "dca.h" // opencoti F5 dca bug-2118
+#include "llama-kv-cache-iswa.h" // bug-2118: get_base() on the iswa mctx
void llama_model_gemma4::load_arch_hparams(llama_model_loader & ml) {
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
@@ -161,6 +163,14 @@
// TODO: is causal == true correct? might need some changes
auto * inp_attn = build_attn_inp_kv_iswa();
+ // opencoti F5 dca bug-2118 — Gemma-4 DCA wiring. 0078 wired only the Qwen builders;
+ // the iswa build_attn_dca overload (dca.cpp) existed uncalled, so '--dca on' was a
+ // silent no-op on Gemma. DCA applies to the GLOBAL (non-SWA) layers only (the SWA
+ // window is <= chunk already); the overload stores PRE-rope K + handles has_kv==false
+ // shared-KV globals (null k/v -> read the earlier layer's cache).
+ const bool use_dca = cparams.dca_enabled;
+ llm_graph_input_dca * inp_dca = use_dca ? build_attn_inp_dca(inp_attn->mctx->get_base()) : nullptr;
+
ggml_tensor * inp_out_ids = build_inp_out_ids();
// opencoti-hook: gemma4-mtp-hidden (bug-858 dual-context MTP) — when a context extracts the
@@ -202,6 +212,10 @@
freq_factors = model.layers[il].rope_freqs;
}
+ // opencoti F5 dca bug-2118 — global layers route through build_attn_dca (which
+ // ropes Q per regime and pre-ropes K at insert), so skip the standard pre-rope.
+ const bool dca_l = use_dca && !hparams.is_swa(il);
+
// Q projection (shared for both non-KV and KV layers)
// this is to mirror Gemma4Attention in pytorch code
ggml_tensor * Qcur;
@@ -214,9 +228,11 @@
Qcur = build_norm(Qcur, model.layers[il].attn_q_norm, nullptr, LLM_NORM_RMS, il);
cb(Qcur, "Qcur_normed", il);
+ if (!dca_l) { // bug-2118: DCA ropes Q per regime in build_attn_dca
Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, freq_factors, n_rot_l, rope_type, n_ctx_orig, freq_base_l, freq_scale_l,
ext_factor, attn_factor, beta_fast, beta_slow);
cb(Qcur, "Qcur_pos", il);
+ }
}
// self-attention
@@ -238,14 +254,28 @@
cb(Kcur, "Kcur_normed", il);
cb(Vcur, "Vcur_normed", il);
+ if (!dca_l) { // bug-2118: DCA pre-ropes K at insert in build_attn_dca
Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, freq_factors, n_rot_l, rope_type, n_ctx_orig, freq_base_l, freq_scale_l,
ext_factor, attn_factor, beta_fast, beta_slow);
cb(Kcur, "Kcur_pos", il);
+ }
+ if (dca_l) { // bug-2118 (wo_s dropped like the qwen3 DCA path — null on these GGUFs)
+ const dca_rope rp = { freq_base_l, freq_scale_l, n_rot_l, freq_factors };
+ cur = build_attn_dca(inp_attn, inp_dca, model.layers[il].wo, nullptr,
+ Qcur, Kcur, Vcur, rp, hparams.f_attention_scale, il);
+ } else {
cur = build_attn(inp_attn, model.layers[il].wo,
nullptr, model.layers[il].wo_s, Qcur, Kcur, Vcur, nullptr, nullptr, nullptr,
hparams.f_attention_scale, il);
+ }
+ } else if (dca_l) {
+ // bug-2118: shared-KV global layer under DCA — Q-only read of the earlier
+ // layer's DCA-formatted cache (null k/v -> no store).
+ const dca_rope rp = { freq_base_l, freq_scale_l, n_rot_l, freq_factors };
+ cur = build_attn_dca(inp_attn, inp_dca, model.layers[il].wo, nullptr,
+ Qcur, nullptr, nullptr, rp, hparams.f_attention_scale, il);
} else {
// reuse KV cache of earlier layers
cur = build_attn(inp_attn,
--- a/llama.cpp/src/models/gemma4-assistant.cpp
+++ b/llama.cpp/src/models/gemma4-assistant.cpp
@@ -1,5 +1,7 @@
// opencoti F5 M6-S4 mtp
#include "models.h"
+#include "dca.h" // opencoti F5 dca bug-2118
+#include "llama-kv-cache-iswa.h" // bug-2118: get_base() on the iswa mctx
#include <cmath>
#include <vector>
@@ -190,6 +192,13 @@
ggml_tensor * inp_pos = build_inp_pos();
ggml_tensor * inp_out_ids = build_inp_out_ids();
+ // opencoti F5 dca bug-2118 — when the TARGET runs DCA, the drafter's global layers
+ // read the target's DCA-formatted (pre-roped, position-remapped) shared K/V, so its
+ // Q must be roped per regime through the same build_attn_dca path. Without this the
+ // drafter reads remapped keys with raw-extrapolated Q beyond native ctx -> accept 0.
+ const bool use_dca = cparams.dca_enabled;
+ llm_graph_input_dca * inp_dca = use_dca ? build_attn_inp_dca(inp_attn->mctx->get_base()) : nullptr;
+
ggml_tensor * inpL = cur;
for (int il = 0; il < n_layer; ++il) {
@@ -212,13 +221,22 @@
cb(Qcur, "Qcur_normed", il);
ggml_tensor * freq_factors = is_swa ? nullptr : model.layers[il].rope_freqs;
+ const bool dca_l = use_dca && !is_swa; // bug-2118
+ if (!dca_l) {
Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, freq_factors, n_rot_l, rope_type, n_ctx_orig,
freq_base_l, freq_scale_l, ext_factor, attn_factor, beta_fast, beta_slow);
cb(Qcur, "Qcur_pos", il);
+ }
// KV-less cross-attention into the shared/aliased target K/V (null k_cur/v_cur -> no store).
+ if (dca_l) { // bug-2118: Q-only per-regime read of the target's DCA cache
+ const dca_rope rp = { freq_base_l, freq_scale_l, n_rot_l, freq_factors };
+ cur = build_attn_dca(inp_attn, inp_dca, model.layers[il].wo, nullptr,
+ Qcur, nullptr, nullptr, rp, hparams.f_attention_scale, il);
+ } else {
cur = build_attn(inp_attn, model.layers[il].wo, nullptr, nullptr,
Qcur, nullptr, nullptr, nullptr, nullptr, nullptr, hparams.f_attention_scale, il);
+ }
if (il == n_layer - 1 && inp_out_ids) {
cur = ggml_get_rows(ctx0, cur, inp_out_ids);