Commit 6184e92c5 for llama.cpp
commit 6184e92c57dcd34de8a3e381d7641a5e75250d5f
Author: bosh <98094229+boshjerns@users.noreply.github.com>
Date: Fri Oct 9 22:26:56 2026 +0700
model : use exact GELU for ModernBERT encoders (#30108)
* model : use exact GELU for ModernBERT encoders
Assisted-by: Codex
* model : keep tanh GELU aliases on ggml_geglu
Assisted-by: Claude Opus 5.5
* model : map gelu_python to ggml_geglu_erf
Assisted-by: Claude Opus 5.5
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index f49e5ae88..5a6a2b320 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -1963,6 +1963,11 @@ ggml_tensor * llm_graph_context::build_ffn(
cur = ggml_geglu(ctx0, cur);
cb(cur, "ffn_geglu", il);
} break;
+ case LLM_FFN_GEGLU_ERF:
+ {
+ cur = ggml_geglu_erf(ctx0, cur);
+ cb(cur, "ffn_geglu_erf", il);
+ } break;
case LLM_FFN_REGLU:
{
cur = ggml_reglu(ctx0, cur);
diff --git a/src/llama-graph.h b/src/llama-graph.h
index 79e8409ac..c469847a8 100644
--- a/src/llama-graph.h
+++ b/src/llama-graph.h
@@ -62,6 +62,7 @@ enum llm_ffn_op_type : int {
LLM_FFN_RELU_SQR,
LLM_FFN_SWIGLU,
LLM_FFN_GEGLU,
+ LLM_FFN_GEGLU_ERF,
LLM_FFN_REGLU,
LLM_FFN_SWIGLU_OAI_MOE,
LLM_FFN_SITU, // kimi-k3
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index 2e86914bc..b3a0cf5d5 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -1064,18 +1064,21 @@ static llama_rope_scaling_type llama_rope_scaling_type_from_string(const std::st
return LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED;
}
-// Maps the GGUF `<arch>.hidden_activation` string to the FFN op type used by the
-// graph builders. Only gated activations that map cleanly to llm_ffn_op_type are
-// listed; unrecognized values fall back to GeGLU, which matches the historical
-// default for ModernBert-style architectures.
+// Maps GGUF activation names to the FFN op type used by the graph builders.
static const std::map<std::string, llm_ffn_op_type> LLM_FFN_OP_TYPES_FROM_STRING = {
- { "gelu", LLM_FFN_GEGLU },
- { "geglu", LLM_FFN_GEGLU },
- { "silu", LLM_FFN_SWIGLU },
- { "swish", LLM_FFN_SWIGLU },
- { "swiglu", LLM_FFN_SWIGLU },
- { "relu", LLM_FFN_RELU },
- { "reglu", LLM_FFN_REGLU },
+ { "gelu", LLM_FFN_GEGLU_ERF },
+ { "gelu_python", LLM_FFN_GEGLU_ERF },
+ { "gelu_pytorch_tanh", LLM_FFN_GEGLU },
+ { "gelu_new", LLM_FFN_GEGLU },
+ { "gelu_fast", LLM_FFN_GEGLU },
+ { "gelu_accurate", LLM_FFN_GEGLU },
+ { "gelu_python_tanh", LLM_FFN_GEGLU },
+ { "geglu", LLM_FFN_GEGLU },
+ { "silu", LLM_FFN_SWIGLU },
+ { "swish", LLM_FFN_SWIGLU },
+ { "swiglu", LLM_FFN_SWIGLU },
+ { "relu", LLM_FFN_RELU },
+ { "reglu", LLM_FFN_REGLU },
};
// transformers names, "gelu" is the exact (erf) variant
diff --git a/src/models/modern-bert.cpp b/src/models/modern-bert.cpp
index d265b03bf..84c85f869 100644
--- a/src/models/modern-bert.cpp
+++ b/src/models/modern-bert.cpp
@@ -17,10 +17,10 @@ void llama_model_modern_bert::load_arch_hparams(llama_model_loader & ml) {
// Some ModernBert derivatives (e.g. IBM Granite Embedding 97m R2) use
// SiLU/SwiGLU in the FFN instead of the default GELU/GeGLU.
- hparams.llm_ffn_op = LLM_FFN_GEGLU;
+ hparams.llm_ffn_op = LLM_FFN_GEGLU_ERF;
std::string hidden_act;
if (ml.get_key(LLM_KV_HIDDEN_ACT, hidden_act, false)) {
- hparams.llm_ffn_op = llm_ffn_op_type_from_string(hidden_act, LLM_FFN_GEGLU);
+ hparams.llm_ffn_op = llm_ffn_op_type_from_string(hidden_act, LLM_FFN_GEGLU_ERF);
}
// GGUFs without a classifier pooling type use mean (gte-reranker-modernbert-base)