Commit 37ac63456 for llama.cpp
commit 37ac634566439e680959b90e1c960714ca6b8b92
Author: bosh <98094229+boshjerns@users.noreply.github.com>
Date: Thu Oct 8 13:29:20 2026 +0700
model : support classifier_activation for rerankers (#29692)
* model : support classifier_activation for rerankers
Assisted-by: Claude Opus 5.5
* model : map classifier gelu to gelu_erf and accept tanh
Assisted-by: Claude Opus 5.5
* model : default act_cls to tanh, ModernBERT falls back to gelu_erf
Assisted-by: Claude Opus 5.5
diff --git a/conversion/base.py b/conversion/base.py
index 786e7faa1..7a93855c8 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -2352,6 +2352,10 @@ class TextModel(ModelBase):
if classifier_pooling not in ("cls", "mean"):
raise NotImplementedError(f"Unsupported classifier_pooling: {classifier_pooling}")
self.gguf_writer.add_classifier_pooling_type(mode_mapping[classifier_pooling])
+ if (classifier_activation := self.hparams.get("classifier_activation")) is not None:
+ if classifier_activation not in ("gelu", "silu", "tanh"):
+ raise NotImplementedError(f"Unsupported classifier_activation: {classifier_activation}")
+ self.gguf_writer.add_classifier_activation(classifier_activation)
def _set_vocab_glmedge(self):
from transformers import AutoTokenizer
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index fcf325fe9..c003ed2b6 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -318,6 +318,7 @@ class Keys:
class Classifier:
OUTPUT_LABELS = "{arch}.classifier.output_labels"
POOLING_TYPE = "{arch}.classifier.pooling_type"
+ ACTIVATION = "{arch}.classifier.activation"
class ShortConv:
L_CACHE = "{arch}.shortconv.l_cache"
diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
index f33a8a551..3d063bb3a 100644
--- a/gguf-py/gguf/gguf_writer.py
+++ b/gguf-py/gguf/gguf_writer.py
@@ -1340,6 +1340,9 @@ class GGUFWriter:
def add_classifier_pooling_type(self, value: PoolingType) -> None:
self.add_uint32(Keys.Classifier.POOLING_TYPE.format(arch=self.arch), value.value)
+ def add_classifier_activation(self, value: str) -> None:
+ self.add_string(Keys.Classifier.ACTIVATION.format(arch=self.arch), value)
+
def add_decision_type(self, value: str) -> None:
self.add_string(Keys.Decision.TYPE.format(arch=self.arch), value)
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index eea2ef589..821479064 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -370,6 +370,7 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
{ LLM_KV_CLASSIFIER_POOLING_TYPE, "%s.classifier.pooling_type" },
+ { LLM_KV_CLASSIFIER_ACTIVATION, "%s.classifier.activation" },
{ LLM_KV_DECISION_BLOCK_COUNT, "%s.decision.block_count" },
{ LLM_KV_DECISION_ROUTING_BLOCK_COUNT, "%s.decision.routing_block_count" },
diff --git a/src/llama-arch.h b/src/llama-arch.h
index c8abee234..1bf4744ff 100644
--- a/src/llama-arch.h
+++ b/src/llama-arch.h
@@ -416,6 +416,7 @@ enum llm_kv {
LLM_KV_CLASSIFIER_OUTPUT_LABELS,
LLM_KV_CLASSIFIER_POOLING_TYPE,
+ LLM_KV_CLASSIFIER_ACTIVATION,
LLM_KV_DECISION_BLOCK_COUNT,
LLM_KV_DECISION_ROUTING_BLOCK_COUNT,
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index e0b7a47a9..f030eb9e6 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -3903,11 +3903,7 @@ void llm_graph_context::build_pooling(
if (cls_b) {
cur = ggml_add(ctx0, cur, cls_b);
}
- if (arch == LLM_ARCH_MODERN_BERT) {
- cur = ggml_gelu(ctx0, cur);
- } else {
- cur = ggml_tanh(ctx0, cur);
- }
+ cur = ggml_unary(ctx0, cur, hparams.act_cls);
if (cls_norm) {
// head norm
cur = build_norm(cur, cls_norm, NULL, LLM_NORM, -1);
diff --git a/src/llama-hparams.h b/src/llama-hparams.h
index 848428706..98afe8a62 100644
--- a/src/llama-hparams.h
+++ b/src/llama-hparams.h
@@ -371,6 +371,7 @@ struct llama_hparams {
// llm_ffn_op_type_from_string() in llama-model.cpp, mirroring how
// rope_scaling_type_train is handled.
enum llm_ffn_op_type llm_ffn_op;
+ enum ggml_unary_op act_cls = GGML_UNARY_OP_TANH; // activation of the classifier head (RANK)
// Step35: optional per-layer clamps for (Swi)GLU
std::array<float, LLAMA_MAX_LAYERS> swiglu_clamp_exp; // clamping for expert FFN
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index bf04945bf..2e86914bc 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -1078,6 +1078,13 @@ static const std::map<std::string, llm_ffn_op_type> LLM_FFN_OP_TYPES_FROM_STRING
{ "reglu", LLM_FFN_REGLU },
};
+// transformers names, "gelu" is the exact (erf) variant
+static const std::map<std::string, ggml_unary_op> LLM_CLS_ACT_TYPES_FROM_STRING = {
+ { "gelu", GGML_UNARY_OP_GELU_ERF },
+ { "silu", GGML_UNARY_OP_SILU },
+ { "tanh", GGML_UNARY_OP_TANH },
+};
+
llm_ffn_op_type llm_ffn_op_type_from_string(const std::string & name, llm_ffn_op_type fallback) {
const auto it = LLM_FFN_OP_TYPES_FROM_STRING.find(name);
if (it != LLM_FFN_OP_TYPES_FROM_STRING.end()) {
@@ -1336,6 +1343,12 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_ATTENTION_CAUSAL, hparams.causal_attn, false);
ml.get_key(LLM_KV_POOLING_TYPE, hparams.pooling_type, false);
ml.get_key(LLM_KV_CLASSIFIER_POOLING_TYPE, hparams.pooling_type_cls, false);
+ std::string act_cls;
+ if (ml.get_key(LLM_KV_CLASSIFIER_ACTIVATION, act_cls, false)) {
+ const auto it = LLM_CLS_ACT_TYPES_FROM_STRING.find(act_cls);
+ GGML_ASSERT(it != LLM_CLS_ACT_TYPES_FROM_STRING.end() && "unsupported classifier activation");
+ hparams.act_cls = it->second;
+ }
ml.get_key(LLM_KV_BLOCK_COUNT, hparams.n_layer_all);
GGML_ASSERT(hparams.n_layer_all > 0 && hparams.n_layer_all <= LLAMA_MAX_LAYERS);
ml.get_key(LLM_KV_NEXTN_PREDICT_LAYERS, hparams.n_layer_nextn, false);
diff --git a/src/models/modern-bert.cpp b/src/models/modern-bert.cpp
index 45455acbb..d265b03bf 100644
--- a/src/models/modern-bert.cpp
+++ b/src/models/modern-bert.cpp
@@ -28,6 +28,12 @@ void llama_model_modern_bert::load_arch_hparams(llama_model_loader & ml) {
hparams.pooling_type_cls = LLAMA_POOLING_TYPE_MEAN;
}
+ // GGUFs without a classifier activation use gelu, the transformers default
+ std::string act_cls;
+ if (!ml.get_key(LLM_KV_CLASSIFIER_ACTIVATION, act_cls, false)) {
+ hparams.act_cls = GGML_UNARY_OP_GELU_ERF;
+ }
+
ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, hparams.n_layer_decision, false);
if (hparams.n_layer_decision > 0) {
if (hparams.n_layer_decision >= hparams.n_layer()) {