use split activations when possible (#12293)

* use ggml_*_split activations when possible * forward qkv
2025-12-23 07:03:57 +00:00 · 2025-09-16 09:51:19 -07:00
parent c253433d68
commit ad95d5b30b
16 changed files with 59 additions and 50 deletions
--- a/model/models/llama/model.go
+++ b/model/models/llama/model.go
@@ -118,7 +118,7 @@ type MLP struct {
 }

 func (mlp *MLP) Forward(ctx ml.Context, hiddenState ml.Tensor, opts *Options) ml.Tensor {
-	hiddenState = mlp.Gate.Forward(ctx, hiddenState).SILU(ctx).Mul(ctx, mlp.Up.Forward(ctx, hiddenState))
+	hiddenState = mlp.Gate.Forward(ctx, hiddenState).SILU(ctx, mlp.Up.Forward(ctx, hiddenState))
 	return mlp.Down.Forward(ctx, hiddenState)
 }