fix: fixes a memory leak in bfloat16 package

This change vendors in the bfloat16 package from github.com/d4l3k/go-bfloat16/ and fixes a memory leak which was being caused by using unsafe pointers instead of the math package.
Merge pull request #9824 from ollama/mxyng/sched
2025-03-17 21:46:12 -07:00 · 2025-03-17 15:41:37 -07:00 · 2025-03-17 15:33:16 -07:00 · 2025-03-17 15:11:15 -07:00 · 2025-03-17 13:32:40 -07:00 · 2025-03-17 11:24:18 -07:00
76 changed files with 3433 additions and 1401 deletions
--- a/CMakePresets.json
+++ b/CMakePresets.json
@@ -56,7 +56,7 @@
      "name": "ROCm 6",
      "inherits": [ "ROCm" ],
      "cacheVariables": {
-        "AMDGPU_TARGETS": "gfx900;gfx940;gfx941;gfx942;gfx1010;gfx1012;gfx1030;gfx1100;gfx1101;gfx1102;gfx906:xnack-;gfx908:xnack-;gfx90a:xnack+;gfx90a:xnack-"
+        "AMDGPU_TARGETS": "gfx900;gfx940;gfx941;gfx942;gfx1010;gfx1012;gfx1030;gfx1100;gfx1101;gfx1102;gfx1151;gfx906:xnack-;gfx908:xnack-;gfx90a:xnack+;gfx90a:xnack-"
      }
    }
  ],
--- a/README.md
+++ b/README.md
@@ -54,6 +54,10 @@ Here are some example models that can be downloaded:

 | Model              | Parameters | Size  | Download                         |
 | ------------------ | ---------- | ----- | -------------------------------- |
+| Gemma 3            | 1B         | 815MB | `ollama run gemma3:1b`           |
+| Gemma 3            | 4B         | 3.3GB | `ollama run gemma3`              |
+| Gemma 3            | 12B        | 8.1GB | `ollama run gemma3:12b`          |
+| Gemma 3            | 27B        | 17GB  | `ollama run gemma3:27b`          |
 | QwQ                | 32B        | 20GB  | `ollama run qwq`                 |
 | DeepSeek-R1        | 7B         | 4.7GB | `ollama run deepseek-r1`         |
 | DeepSeek-R1        | 671B       | 404GB | `ollama run deepseek-r1:671b`    |
@@ -66,9 +70,6 @@ Here are some example models that can be downloaded:
 | Llama 3.1          | 405B       | 231GB | `ollama run llama3.1:405b`       |
 | Phi 4              | 14B        | 9.1GB | `ollama run phi4`                |
 | Phi 4 Mini         | 3.8B       | 2.5GB | `ollama run phi4-mini`           |
-| Gemma 2            | 2B         | 1.6GB | `ollama run gemma2:2b`           |
-| Gemma 2            | 9B         | 5.5GB | `ollama run gemma2`              |
-| Gemma 2            | 27B        | 16GB  | `ollama run gemma2:27b`          |
 | Mistral            | 7B         | 4.1GB | `ollama run mistral`             |
 | Moondream 2        | 1.4B       | 829MB | `ollama run moondream`           |
 | Neural Chat        | 7B         | 4.1GB | `ollama run neural-chat`         |
@@ -391,6 +392,8 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [1Panel](https://github.com/1Panel-dev/1Panel/) (Web-based Linux Server Management Tool)
 - [AstrBot](https://github.com/Soulter/AstrBot/) (User-friendly LLM-based multi-platform chatbot with a WebUI, supporting RAG, LLM agents, and plugins integration)
 - [Reins](https://github.com/ibrahimcetin/reins) (Easily tweak parameters, customize system prompts per chat, and enhance your AI experiments with reasoning model support.)
+- [Ellama](https://github.com/zeozeozeo/ellama) (Friendly native app to chat with an Ollama instance)
+- [screenpipe](https://github.com/mediar-ai/screenpipe) Build agents powered by your screen history

 ### Cloud

@@ -571,6 +574,7 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [llama.cpp](https://github.com/ggerganov/llama.cpp) project founded by Georgi Gerganov.

 ### Observability
+- [Opik](https://www.comet.com/docs/opik/cookbook/ollama) is an open-source platform to debug, evaluate, and monitor your LLM applications, RAG systems, and agentic workflows with comprehensive tracing, automated evaluations, and production-ready dashboards. Opik supports native intergration to Ollama.
 - [Lunary](https://lunary.ai/docs/integrations/ollama) is the leading open-source LLM observability platform. It provides a variety of enterprise-grade features such as real-time analytics, prompt templates management, PII masking, and comprehensive agent tracing.
 - [OpenLIT](https://github.com/openlit/openlit) is an OpenTelemetry-native tool for monitoring Ollama Applications & GPUs using traces and metrics.
 - [HoneyHive](https://docs.honeyhive.ai/integrations/ollama) is an AI observability and evaluation platform for AI agents. Use HoneyHive to evaluate agent performance, interrogate failures, and monitor quality in production.
--- a/api/types.go
+++ b/api/types.go
@@ -349,6 +349,7 @@ type ShowResponse struct {
 	Messages      []Message      `json:"messages,omitempty"`
 	ModelInfo     map[string]any `json:"model_info,omitempty"`
 	ProjectorInfo map[string]any `json:"projector_info,omitempty"`
+	Tensors       []Tensor       `json:"tensors,omitempty"`
 	ModifiedAt    time.Time      `json:"modified_at,omitempty"`
 }

@@ -467,6 +468,13 @@ type ModelDetails struct {
 	QuantizationLevel string   `json:"quantization_level"`
 }

+// Tensor describes the metadata for a given tensor.
+type Tensor struct {
+	Name  string   `json:"name"`
+	Type  string   `json:"type"`
+	Shape []uint64 `json:"shape"`
+}
+
 func (m *Metrics) Summary() {
 	if m.TotalDuration > 0 {
 		fmt.Fprintf(os.Stderr, "total duration:       %v\n", m.TotalDuration)
--- a/cmd/cmd.go
+++ b/cmd/cmd.go
@@ -18,6 +18,7 @@ import (
 	"os/signal"
 	"path/filepath"
 	"runtime"
+	"sort"
 	"strconv"
 	"strings"
 	"sync/atomic"
@@ -568,8 +569,9 @@ func ShowHandler(cmd *cobra.Command, args []string) error {
 	parameters, errParams := cmd.Flags().GetBool("parameters")
 	system, errSystem := cmd.Flags().GetBool("system")
 	template, errTemplate := cmd.Flags().GetBool("template")
+	verbose, errVerbose := cmd.Flags().GetBool("verbose")

-	for _, boolErr := range []error{errLicense, errModelfile, errParams, errSystem, errTemplate} {
+	for _, boolErr := range []error{errLicense, errModelfile, errParams, errSystem, errTemplate, errVerbose} {
 		if boolErr != nil {
 			return errors.New("error retrieving flags")
 		}
@@ -607,7 +609,7 @@ func ShowHandler(cmd *cobra.Command, args []string) error {
 		return errors.New("only one of '--license', '--modelfile', '--parameters', '--system', or '--template' can be specified")
 	}

-	req := api.ShowRequest{Name: args[0]}
+	req := api.ShowRequest{Name: args[0], Verbose: verbose}
 	resp, err := client.Show(cmd.Context(), &req)
 	if err != nil {
 		return err
@@ -630,10 +632,10 @@ func ShowHandler(cmd *cobra.Command, args []string) error {
 		return nil
 	}

-	return showInfo(resp, os.Stdout)
+	return showInfo(resp, verbose, os.Stdout)
 }

-func showInfo(resp *api.ShowResponse, w io.Writer) error {
+func showInfo(resp *api.ShowResponse, verbose bool, w io.Writer) error {
 	tableRender := func(header string, rows func() [][]string) {
 		fmt.Fprintln(w, " ", header)
 		table := tablewriter.NewWriter(w)
@@ -690,6 +692,45 @@ func showInfo(resp *api.ShowResponse, w io.Writer) error {
 		})
 	}

+	if resp.ModelInfo != nil && verbose {
+		tableRender("Metadata", func() (rows [][]string) {
+			keys := make([]string, 0, len(resp.ModelInfo))
+			for k := range resp.ModelInfo {
+				keys = append(keys, k)
+			}
+			sort.Strings(keys)
+
+			for _, k := range keys {
+				var v string
+				switch vData := resp.ModelInfo[k].(type) {
+				case string:
+					v = vData
+				case float64:
+					v = fmt.Sprintf("%g", vData)
+				case []any:
+					n := 3
+					if len(vData) < n {
+						n = len(vData)
+					}
+					v = fmt.Sprintf("%v", vData[:n])
+				default:
+					v = fmt.Sprintf("%T", vData)
+				}
+				rows = append(rows, []string{"", k, v})
+			}
+			return
+		})
+	}
+
+	if len(resp.Tensors) > 0 && verbose {
+		tableRender("Tensors", func() (rows [][]string) {
+			for _, t := range resp.Tensors {
+				rows = append(rows, []string{"", t.Name, t.Type, fmt.Sprint(t.Shape)})
+			}
+			return
+		})
+	}
+
 	head := func(s string, n int) (rows [][]string) {
 		scanner := bufio.NewScanner(strings.NewReader(s))
 		for scanner.Scan() && (len(rows) < n || n < 0) {
@@ -1196,6 +1237,7 @@ func NewCLI() *cobra.Command {
 	showCmd.Flags().Bool("parameters", false, "Show parameters of a model")
 	showCmd.Flags().Bool("template", false, "Show template of a model")
 	showCmd.Flags().Bool("system", false, "Show system message of a model")
+	showCmd.Flags().BoolP("verbose", "v", false, "Show detailed model information")

 	runCmd := &cobra.Command{
 		Use:     "run MODEL [PROMPT]",
--- a/cmd/cmd_test.go
+++ b/cmd/cmd_test.go
@@ -27,7 +27,7 @@ func TestShowInfo(t *testing.T) {
 				ParameterSize:     "7B",
 				QuantizationLevel: "FP16",
 			},
-		}, &b); err != nil {
+		}, false, &b); err != nil {
 			t.Fatal(err)
 		}

@@ -57,7 +57,7 @@ func TestShowInfo(t *testing.T) {
 				ParameterSize:     "7B",
 				QuantizationLevel: "FP16",
 			},
-		}, &b); err != nil {
+		}, false, &b); err != nil {
 			t.Fatal(err)
 		}

@@ -68,6 +68,56 @@ func TestShowInfo(t *testing.T) {
    embedding length    0       
    quantization        FP16    

+`
+		if diff := cmp.Diff(expect, b.String()); diff != "" {
+			t.Errorf("unexpected output (-want +got):\n%s", diff)
+		}
+	})
+
+	t.Run("verbose model", func(t *testing.T) {
+		var b bytes.Buffer
+		if err := showInfo(&api.ShowResponse{
+			Details: api.ModelDetails{
+				Family:            "test",
+				ParameterSize:     "8B",
+				QuantizationLevel: "FP16",
+			},
+			Parameters: `
+			stop up`,
+			ModelInfo: map[string]any{
+				"general.architecture":    "test",
+				"general.parameter_count": float64(8_000_000_000),
+				"test.context_length":     float64(1000),
+				"test.embedding_length":   float64(11434),
+			},
+			Tensors: []api.Tensor{
+				{Name: "blk.0.attn_k.weight", Type: "BF16", Shape: []uint64{42, 3117}},
+				{Name: "blk.0.attn_q.weight", Type: "FP16", Shape: []uint64{3117, 42}},
+			},
+		}, true, &b); err != nil {
+			t.Fatal(err)
+		}
+
+		expect := `  Model
+    architecture        test     
+    parameters          8B       
+    context length      1000     
+    embedding length    11434    
+    quantization        FP16     
+
+  Parameters
+    stop    up    
+
+  Metadata
+    general.architecture       test     
+    general.parameter_count    8e+09    
+    test.context_length        1000     
+    test.embedding_length      11434    
+
+  Tensors
+    blk.0.attn_k.weight    BF16    [42 3117]    
+    blk.0.attn_q.weight    FP16    [3117 42]    
+
 `
 		if diff := cmp.Diff(expect, b.String()); diff != "" {
 			t.Errorf("unexpected output (-want +got):\n%s", diff)
@@ -89,7 +139,7 @@ func TestShowInfo(t *testing.T) {
 			stop you
 			stop up
 			temperature 99`,
-		}, &b); err != nil {
+		}, false, &b); err != nil {
 			t.Fatal(err)
 		}

@@ -126,7 +176,7 @@ func TestShowInfo(t *testing.T) {
 				"clip.vision.embedding_length": float64(0),
 				"clip.vision.projection_dim":   float64(0),
 			},
-		}, &b); err != nil {
+		}, false, &b); err != nil {
 			t.Fatal(err)
 		}

@@ -159,7 +209,7 @@ func TestShowInfo(t *testing.T) {
 Ahoy, matey!
 Weigh anchor!
 			`,
-		}, &b); err != nil {
+		}, false, &b); err != nil {
 			t.Fatal(err)
 		}

@@ -188,7 +238,7 @@ Weigh anchor!
 				QuantizationLevel: "FP16",
 			},
 			License: license,
-		}, &b); err != nil {
+		}, false, &b); err != nil {
 			t.Fatal(err)
 		}

@@ -707,3 +757,132 @@ func TestCreateHandler(t *testing.T) {
 		})
 	}
 }
+
+func TestNewCreateRequest(t *testing.T) {
+	tests := []struct {
+		name     string
+		from     string
+		opts     runOptions
+		expected *api.CreateRequest
+	}{
+		{
+			"basic test",
+			"newmodel",
+			runOptions{
+				Model:       "mymodel",
+				ParentModel: "",
+				Prompt:      "You are a fun AI agent",
+				Messages:    []api.Message{},
+				WordWrap:    true,
+			},
+			&api.CreateRequest{
+				From:  "mymodel",
+				Model: "newmodel",
+			},
+		},
+		{
+			"parent model test",
+			"newmodel",
+			runOptions{
+				Model:       "mymodel",
+				ParentModel: "parentmodel",
+				Messages:    []api.Message{},
+				WordWrap:    true,
+			},
+			&api.CreateRequest{
+				From:  "parentmodel",
+				Model: "newmodel",
+			},
+		},
+		{
+			"parent model as filepath test",
+			"newmodel",
+			runOptions{
+				Model:       "mymodel",
+				ParentModel: "/some/file/like/etc/passwd",
+				Messages:    []api.Message{},
+				WordWrap:    true,
+			},
+			&api.CreateRequest{
+				From:  "mymodel",
+				Model: "newmodel",
+			},
+		},
+		{
+			"parent model as windows filepath test",
+			"newmodel",
+			runOptions{
+				Model:       "mymodel",
+				ParentModel: "D:\\some\\file\\like\\etc\\passwd",
+				Messages:    []api.Message{},
+				WordWrap:    true,
+			},
+			&api.CreateRequest{
+				From:  "mymodel",
+				Model: "newmodel",
+			},
+		},
+		{
+			"options test",
+			"newmodel",
+			runOptions{
+				Model:       "mymodel",
+				ParentModel: "parentmodel",
+				Options: map[string]any{
+					"temperature": 1.0,
+				},
+			},
+			&api.CreateRequest{
+				From:  "parentmodel",
+				Model: "newmodel",
+				Parameters: map[string]any{
+					"temperature": 1.0,
+				},
+			},
+		},
+		{
+			"messages test",
+			"newmodel",
+			runOptions{
+				Model:       "mymodel",
+				ParentModel: "parentmodel",
+				System:      "You are a fun AI agent",
+				Messages: []api.Message{
+					{
+						Role:    "user",
+						Content: "hello there!",
+					},
+					{
+						Role:    "assistant",
+						Content: "hello to you!",
+					},
+				},
+				WordWrap: true,
+			},
+			&api.CreateRequest{
+				From:   "parentmodel",
+				Model:  "newmodel",
+				System: "You are a fun AI agent",
+				Messages: []api.Message{
+					{
+						Role:    "user",
+						Content: "hello there!",
+					},
+					{
+						Role:    "assistant",
+						Content: "hello to you!",
+					},
+				},
+			},
+		},
+	}
+
+	for _, tt := range tests {
+		t.Run(tt.name, func(t *testing.T) {
+			actual := NewCreateRequest(tt.from, tt.opts)
+			if !cmp.Equal(actual, tt.expected) {
+				t.Errorf("expected output %#v, got %#v", tt.expected, actual)
+			}
+		})
+	}
+}
--- a/cmd/interactive.go
+++ b/cmd/interactive.go
@@ -18,6 +18,7 @@ import (
 	"github.com/ollama/ollama/envconfig"
 	"github.com/ollama/ollama/readline"
 	"github.com/ollama/ollama/types/errtypes"
+	"github.com/ollama/ollama/types/model"
 )

 type MultilineState int
@@ -195,6 +196,10 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 			opts.Messages = []api.Message{}
 			fmt.Printf("Loading model '%s'\n", opts.Model)
 			if err := loadOrUnloadModel(cmd, &opts); err != nil {
+				if strings.Contains(err.Error(), "not found") {
+					fmt.Printf("error: %v\n", err)
+					continue
+				}
 				return err
 			}
 			continue
@@ -343,7 +348,7 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {

 				switch args[1] {
 				case "info":
-					_ = showInfo(resp, os.Stderr)
+					_ = showInfo(resp, false, os.Stderr)
 				case "license":
 					if resp.License == "" {
 						fmt.Println("No license was specified for this model.")
@@ -455,9 +460,16 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 }

 func NewCreateRequest(name string, opts runOptions) *api.CreateRequest {
+	parentModel := opts.ParentModel
+
+	modelName := model.ParseName(parentModel)
+	if !modelName.IsValid() {
+		parentModel = ""
+	}
+
 	req := &api.CreateRequest{
-		Name: name,
-		From: cmp.Or(opts.ParentModel, opts.Model),
+		Model: name,
+		From:  cmp.Or(parentModel, opts.Model),
 	}

 	if opts.System != "" {
--- a/convert/convert.go
+++ b/convert/convert.go
@@ -13,8 +13,13 @@ import (
 )

 type ModelParameters struct {
-	Architectures []string `json:"architectures"`
-	VocabSize     uint32   `json:"vocab_size"`
+	Architectures []string       `json:"architectures"`
+	VocabSize     uint32         `json:"vocab_size"`
+	TextModel     TextParameters `json:"text_config"`
+}
+
+type TextParameters struct {
+	VocabSize uint32 `json:"vocab_size"`
 }

 type AdapterParameters struct {
@@ -185,6 +190,8 @@ func ConvertModel(fsys fs.FS, ws io.WriteSeeker) error {
 		conv = &gemmaModel{}
 	case "Gemma2ForCausalLM":
 		conv = &gemma2Model{}
+	case "Gemma3ForCausalLM", "Gemma3ForConditionalGeneration":
+		conv = &gemma3Model{Architecture: p.Architectures[0]}
 	case "Phi3ForCausalLM":
 		conv = &phi3Model{}
 	case "Qwen2ForCausalLM":
@@ -213,7 +220,14 @@ func ConvertModel(fsys fs.FS, ws io.WriteSeeker) error {
 	}

 	vocabSize := int(p.VocabSize)
+	if vocabSize == 0 {
+		tVocabSize := int(p.TextModel.VocabSize)
+		vocabSize = tVocabSize
+	}
+
 	switch {
+	case vocabSize == 0:
+		slog.Warn("vocabulary size was not explicitly set by the model", "default size", len(t.Vocabulary.Tokens))
 	case vocabSize > len(t.Vocabulary.Tokens):
 		slog.Warn("vocabulary is smaller than expected, padding with dummy tokens", "expect", vocabSize, "actual", len(t.Vocabulary.Tokens))
 		for i := range vocabSize - len(t.Vocabulary.Tokens) {
--- a/convert/convert_gemma.go
+++ b/convert/convert_gemma.go
@@ -45,7 +45,7 @@ func (p *gemmaModel) KV(t *Tokenizer) ggml.KV {
 func (p *gemmaModel) Tensors(ts []Tensor) []ggml.Tensor {
 	var out []ggml.Tensor
 	for _, t := range ts {
-		if strings.HasSuffix(t.Name(), "_norm.weight") {
+		if !strings.HasPrefix(t.Name(), "v.") && strings.HasSuffix(t.Name(), "_norm.weight") {
 			t.SetRepacker(p.addOne)
 		}

--- a/convert/convert_gemma3.go
+++ b/convert/convert_gemma3.go
@@ -0,0 +1,142 @@
+package convert
+
+import (
+	"cmp"
+
+	"github.com/ollama/ollama/fs/ggml"
+)
+
+type gemma3Model struct {
+	gemmaModel
+	Architecture string
+	TextModel    struct {
+		HeadDim          uint32 `json:"head_dim"`
+		HiddenSize       uint32 `json:"hidden_size"`
+		HiddenLayers     uint32 `json:"num_hidden_layers"`
+		IntermediateSize uint32 `json:"intermediate_size"`
+		SlidingWindow    uint32 `json:"sliding_window"`
+	} `json:"text_config"`
+	VisionModel struct {
+		NumAttentionHeads uint32  `json:"num_attention_heads"` // attention.head_count 16
+		LayerNormEpsilon  float32 `json:"layer_norm_eps"`      // attention.layer_norm_epsilon 1e-05
+		NumHiddenLayers   uint32  `json:"num_hidden_layers"`   // block_count 32
+		HiddenSize        uint32  `json:"hidden_size"`         // embedding_length 1280
+		IntermediateSize  uint32  `json:"intermediate_size"`   // feed_forward_length 5120
+		ImageSize         uint32  `json:"image_size"`          // image_size 560
+		NumChannels       uint32  `json:"num_channels"`        // num_channels 3
+		PatchSize         uint32  `json:"patch_size"`          // patch_size 14
+	} `json:"vision_config"`
+	MaxPositionEmbeddings    uint32  `json:"max_position_embeddings"`
+	NumAttentionHeads        uint32  `json:"num_attention_heads"`
+	NumKeyValueHeads         uint32  `json:"num_key_value_heads"`
+	RMSNormEPS               float32 `json:"rms_norm_eps"`
+	HeadDim                  uint32  `json:"head_dim"`
+	FinalLogitSoftcap        float32 `json:"final_logit_softcapping"`
+	RopeLocalTheta           float32 `json:"rope_local_base_freq"`
+	RopeGlobalTheta          float32 `json:"rope_global_base_freq"`
+	SlidingWindow            uint32  `json:"sliding_window"`
+	MultiModalTokensPerImage uint32  `json:"mm_tokens_per_image"`
+}
+
+const (
+	gemma4BLayerCount  = 34
+	gemma12BLayerCount = 48
+	gemma27BLayerCount = 62
+)
+
+func (p *gemma3Model) KV(t *Tokenizer) ggml.KV {
+	kv := p.ModelParameters.KV(t)
+	kv["general.architecture"] = "gemma3"
+
+	numBlocks := cmp.Or(p.HiddenLayers, p.TextModel.HiddenLayers)
+	kv["gemma3.block_count"] = numBlocks
+
+	var (
+		numHeads   uint32
+		numKVHeads uint32
+	)
+
+	switch numBlocks {
+	case gemma4BLayerCount:
+		numHeads = 8
+		numKVHeads = 4
+	case gemma12BLayerCount:
+		numHeads = 16
+		numKVHeads = 8
+	case gemma27BLayerCount:
+		numHeads = 32
+		numKVHeads = 16
+	default:
+		numHeads = p.NumAttentionHeads
+		numKVHeads = p.NumKeyValueHeads
+	}
+
+	kv["gemma3.attention.head_count"] = numHeads
+	kv["gemma3.attention.head_count_kv"] = numKVHeads
+
+	switch p.Architecture {
+	case "Gemma3ForCausalLM":
+		kv["gemma3.context_length"] = p.MaxPositionEmbeddings
+		kv["gemma3.attention.layer_norm_rms_epsilon"] = p.RMSNormEPS
+		kv["gemma3.attention.key_length"] = p.HeadDim
+		kv["gemma3.attention.value_length"] = p.HeadDim
+		kv["gemma3.attention.sliding_window"] = p.SlidingWindow
+		kv["gemma3.final_logit_softcapping"] = cmp.Or(p.FinalLogitSoftcap, 30)
+		kv["gemma3.rope.local.freq_base"] = cmp.Or(p.RopeLocalTheta, 10000.0)
+		kv["gemma3.rope.global.freq_base"] = cmp.Or(p.RopeGlobalTheta, 1000000.0)
+		kv["gemma3.embedding_length"] = p.HiddenSize
+		kv["gemma3.feed_forward_length"] = p.IntermediateSize
+	default:
+		kv["gemma3.context_length"] = cmp.Or(p.MaxPositionEmbeddings, 131072)
+		kv["gemma3.embedding_length"] = p.TextModel.HiddenSize
+		kv["gemma3.feed_forward_length"] = p.TextModel.IntermediateSize
+		kv["gemma3.attention.sliding_window"] = p.TextModel.SlidingWindow
+		kv["gemma3.vision.block_count"] = p.VisionModel.NumHiddenLayers
+		kv["gemma3.vision.embedding_length"] = p.VisionModel.HiddenSize
+		kv["gemma3.vision.feed_forward_length"] = p.VisionModel.IntermediateSize
+		kv["gemma3.vision.image_size"] = p.VisionModel.ImageSize
+		kv["gemma3.vision.patch_size"] = p.VisionModel.PatchSize
+		kv["gemma3.vision.num_channels"] = cmp.Or(p.VisionModel.NumChannels, 3)
+		kv["gemma3.vision.attention.head_count"] = p.VisionModel.NumAttentionHeads
+		kv["gemma3.vision.attention.layer_norm_epsilon"] = cmp.Or(p.VisionModel.LayerNormEpsilon, 1e-6)
+		kv["gemma3.attention.key_length"] = cmp.Or(p.TextModel.HeadDim, 256)
+		kv["gemma3.attention.value_length"] = cmp.Or(p.TextModel.HeadDim, 256)
+	}
+
+	if p.MultiModalTokensPerImage > 0 {
+		kv["gemma3.mm.tokens_per_image"] = p.MultiModalTokensPerImage
+	}
+
+	return kv
+}
+
+func (p *gemma3Model) Replacements() []string {
+	return []string{
+		"lm_head", "output",
+		"model.embed_tokens", "token_embd",
+		"model.norm", "output_norm",
+		"vision_tower.vision_model.embeddings", "v",
+		"vision_tower.vision_model", "v",
+		"vision_model.vision_model.embeddings", "v",
+		"vision_model.vision_model", "v",
+		"language_model.", "",
+		"model.layers", "blk",
+		"encoder.layers", "blk",
+		"input_layernorm", "attn_norm",
+		"self_attn.q_proj", "attn_q",
+		"self_attn.q_norm", "attn_q_norm",
+		"self_attn.k_proj", "attn_k",
+		"self_attn.k_norm", "attn_k_norm",
+		"self_attn.v_proj", "attn_v",
+		"self_attn.o_proj", "attn_output",
+		"self_attn.out_proj", "attn_output",
+		"mlp.gate_proj", "ffn_gate",
+		"mlp.down_proj", "ffn_down",
+		"mlp.up_proj", "ffn_up",
+		"post_attention_layernorm", "post_attention_norm",
+		"pre_feedforward_layernorm", "ffn_norm",
+		"post_feedforward_layernorm", "post_ffw_norm",
+		"input_projection_weight", "input_projection.weight",
+		"multi_modal_projector", "mm",
+	}
+}
--- a/convert/reader_safetensors.go
+++ b/convert/reader_safetensors.go
@@ -11,9 +11,10 @@ import (
 	"slices"
 	"strings"

-	"github.com/d4l3k/go-bfloat16"
 	"github.com/x448/float16"
 	"golang.org/x/exp/maps"
+
+	"github.com/ollama/ollama/types/bfloat16"
 )

 type safetensorMetadata struct {
--- a/convert/tokenizer_spm.go
+++ b/convert/tokenizer_spm.go
@@ -6,7 +6,9 @@ import (
 	"errors"
 	"fmt"
 	"io/fs"
+	"log/slog"
 	"os"
+	"reflect"
 	"slices"

 	"google.golang.org/protobuf/proto"
@@ -15,6 +17,8 @@ import (
 )

 func parseSentencePiece(fsys fs.FS) (*Vocabulary, error) {
+	slog.Debug("using spm vocabulary")
+
 	ast, err := parseAdditionalSpecialTokens(fsys)
 	if err != nil {
 		return nil, err
@@ -43,10 +47,19 @@ func parseSentencePiece(fsys fs.FS) (*Vocabulary, error) {
 			v.Types = append(v.Types, int32(t))
 		default:
 			tt := int32(sentencepiece.ModelProto_SentencePiece_NORMAL)
-			if slices.Contains(ast, piece.GetPiece()) {
+
+			// temporary fix to handle gemma3 broken configs
+			if slices.Contains([]string{"<end_of_turn>", "<start_of_turn>"}, piece.GetPiece()) {
 				tt = int32(sentencepiece.ModelProto_SentencePiece_CONTROL)
 			}

+			for _, t := range ast {
+				if t.Content == piece.GetPiece() {
+					tt = int32(sentencepiece.ModelProto_SentencePiece_CONTROL)
+					break
+				}
+			}
+
 			v.Types = append(v.Types, tt)
 		}
 	}
@@ -78,10 +91,16 @@ func parseSentencePiece(fsys fs.FS) (*Vocabulary, error) {
 		return cmp.Compare(i.id, j.id)
 	})

-	n := len(v.Tokens)
-	for i, t := range ts {
-		if t.id != i+n {
-			return nil, fmt.Errorf("invalid token id: %d", t.id)
+	for _, t := range ts {
+		if t.id < len(v.Tokens) {
+			if v.Tokens[t.id] == t.content {
+				slog.Warn("tokenizer", "duplicate token", t.content, "id", t.id)
+				continue
+			}
+			return nil, fmt.Errorf("token mismatch: %s != %s at pos [%d]", t.content, v.Tokens[t.id], t.id)
+		}
+		if t.id != len(v.Tokens) {
+			return nil, fmt.Errorf("invalid token id: [%d] as pos [%d]", t.id, len(v.Tokens))
 		}

 		v.Tokens = append(v.Tokens, t.content)
@@ -92,7 +111,15 @@ func parseSentencePiece(fsys fs.FS) (*Vocabulary, error) {
 	return &v, nil
 }

-func parseAdditionalSpecialTokens(fsys fs.FS) ([]string, error) {
+type specialToken struct {
+	Content    string `json:"content"`
+	Lstrip     bool   `json:"lstrip"`
+	Normalized bool   `json:"normalized"`
+	Rstrip     bool   `json:"rstrip"`
+	SingleWord bool   `json:"single_word"`
+}
+
+func parseAdditionalSpecialTokens(fsys fs.FS) ([]specialToken, error) {
 	f, err := fsys.Open("special_tokens_map.json")
 	if errors.Is(err, os.ErrNotExist) {
 		return nil, nil
@@ -102,12 +129,43 @@ func parseAdditionalSpecialTokens(fsys fs.FS) ([]string, error) {
 	defer f.Close()

 	var m struct {
-		AdditionalSpecialTokens []string `json:"additional_special_tokens"`
+		AdditionalSpecialTokens any `json:"additional_special_tokens"`
 	}

 	if err := json.NewDecoder(f).Decode(&m); err != nil {
 		return nil, err
 	}

-	return m.AdditionalSpecialTokens, nil
+	var ast []specialToken
+
+	switch st := m.AdditionalSpecialTokens.(type) {
+	case []string:
+		for _, s := range st {
+			ast = append(ast, specialToken{Content: s})
+		}
+	case []any:
+		for _, s := range st {
+			// marshal and unmarshal the object to get the special token
+			tMap := s.(map[string]any)
+			data, err := json.Marshal(tMap)
+			if err != nil {
+				return nil, err
+			}
+
+			var token specialToken
+			err = json.Unmarshal(data, &token)
+			if err != nil {
+				return nil, err
+			}
+
+			ast = append(ast, token)
+		}
+
+	default:
+		slog.Warn("special token", "unknown token", reflect.TypeOf(st))
+	}
+
+	slog.Debug("spm tokenizer", "additional tokens", ast)
+
+	return ast, nil
 }
--- a/docs/faq.md
+++ b/docs/faq.md
@@ -187,6 +187,13 @@ cloudflared tunnel --url http://localhost:11434 --http-host-header="localhost:11

 Ollama allows cross-origin requests from `127.0.0.1` and `0.0.0.0` by default. Additional origins can be configured with `OLLAMA_ORIGINS`.

+For browser extensions, you'll need to explicitly allow the extension's origin pattern. Set `OLLAMA_ORIGINS` to include `chrome-extension://*`, `moz-extension://*`, and `safari-web-extension://*` if you wish to allow all browser extensions access, or specific extensions as needed:
+
+```
+# Allow all Chrome, Firefox, and Safari extensions
+OLLAMA_ORIGINS=chrome-extension://*,moz-extension://*,safari-web-extension://* ollama serve
+```
+
 Refer to the section [above](#how-do-i-configure-ollama-server) for how to set environment variables on your platform.

 ## Where are models stored?
--- a/docs/linux.md
+++ b/docs/linux.md
@@ -75,7 +75,7 @@ RestartSec=3
 Environment="PATH=$PATH"

 [Install]
-WantedBy=default.target
+WantedBy=multi-user.target
 ```

 Then start the service:
--- a/fs/ggml/ggml.go
+++ b/fs/ggml/ggml.go
@@ -124,6 +124,19 @@ func (kv KV) Uints(key string, defaultValue ...[]uint32) []uint32 {
 	return s
 }

+func (kv KV) Floats(key string, defaultValue ...[]float32) []float32 {
+	r := keyValue(kv, key, &array{})
+	s := make([]float32, r.size)
+	for i := range r.size {
+		s[i] = float32(r.values[i].(float32))
+	}
+	return s
+}
+
+func (kv KV) OllamaEngineRequired() bool {
+	return kv.Architecture() == "gemma3"
+}
+
 func keyValue[T string | uint32 | uint64 | float32 | *array | bool](kv KV, key string, defaultValue ...T) T {
 	if !strings.HasPrefix(key, "tokenizer.") && !strings.HasPrefix(key, "general.") {
 		key = kv.Architecture() + "." + key
@@ -314,6 +327,10 @@ func (t Tensor) Size() uint64 {
 	return t.parameters() * t.typeSize() / t.blockSize()
 }

+func (t Tensor) Type() string {
+	return fileType(t.Kind).String()
+}
+
 type container interface {
 	Name() string
 	Decode(io.ReadSeeker) (model, error)
@@ -476,7 +493,7 @@ func (f GGML) GraphSize(context, batch uint64, kvCacheType string) (kv, partialO
 			// vocab graph
 			4*batch*(embedding+vocab)+embedding*vocab*105/128,
 		)
-	case "gemma", "gemma2":
+	case "gemma", "gemma2", "gemma3":
 		fullOffload = max(
 			4*batch*(embedding+vocab),
 			4*batch*(2+context+context*heads+2*embedding+2*embeddingHeadsK*heads),
@@ -566,39 +583,52 @@ func (f GGML) GraphSize(context, batch uint64, kvCacheType string) (kv, partialO
 }

 func (llm GGML) VisionGraphSize() (weights, graphSize uint64) {
+	if llm.KV().Uint("vision.block_count") == 0 {
+		return
+	}
+
+	for name, layer := range llm.Tensors().GroupLayers() {
+		if name == "v" || strings.HasPrefix(name, "v.") {
+			for _, tensor := range layer {
+				weights += tensor.Size()
+			}
+		}
+	}
+
+	imageSize := uint64(llm.KV().Uint("vision.image_size"))
+	patchSize := uint64(llm.KV().Uint("vision.patch_size"))
+	if patchSize == 0 {
+		slog.Warn("unknown patch size for vision model")
+		return
+	}
+
+	numChannels := uint64(llm.KV().Uint("vision.num_channels"))
+
+	numPatches := (imageSize / patchSize) * (imageSize / patchSize)
+	if _, ok := llm.Tensors().GroupLayers()["v"]["class_embd"]; ok {
+		numPatches++
+	}
+
+	headCount := uint64(llm.KV().Uint("vision.attention.head_count"))
+	embeddingLength := uint64(llm.KV().Uint("vision.embedding_length"))
+
 	switch llm.KV().Architecture() {
 	case "mllama":
-		for _, layer := range llm.Tensors().GroupLayers()["v"] {
-			weights += layer.Size()
-		}
-
-		kv := func(n string) uint64 {
-			if v, ok := llm.KV()["mllama.vision."+n].(uint32); ok {
-				return uint64(v)
-			}
-
-			return 0
-		}
-
-		imageSize := kv("image_size")
-
-		maxNumTiles := kv("max_num_tiles")
-		embeddingLength := kv("embedding_length")
-		headCount := kv("attention.head_count")
-
-		numPatches := (imageSize / kv("patch_size")) * (imageSize / kv("patch_size"))
-		if _, ok := llm.Tensors().GroupLayers()["v"]["class_embd"]; ok {
-			numPatches++
-		}
-
 		numPaddedPatches := numPatches + 8 - (numPatches%8)%8

+		maxNumTiles := uint64(llm.KV().Uint("vision.max_num_tiles"))
+
 		graphSize = 4 * (8 +
-			imageSize*imageSize*kv("num_channels")*maxNumTiles +
+			imageSize*imageSize*numChannels*maxNumTiles +
 			embeddingLength*numPatches*maxNumTiles +
 			9*embeddingLength*numPaddedPatches*maxNumTiles +
 			numPaddedPatches*maxNumTiles*numPaddedPatches*maxNumTiles*headCount)
+	case "gemma3":
+		graphSize = 4 * (imageSize*imageSize*numChannels +
+			embeddingLength*patchSize +
+			numPatches*numPatches*headCount)
 	}
+
 	return weights, graphSize
 }

--- a/go.mod
+++ b/go.mod
@@ -16,7 +16,6 @@ require (

 require (
 	github.com/agnivade/levenshtein v1.1.1
-	github.com/d4l3k/go-bfloat16 v0.0.0-20211005043715-690c3bdd05f1
 	github.com/dlclark/regexp2 v1.11.4
 	github.com/emirpasic/gods/v2 v2.0.0-alpha
 	github.com/google/go-cmp v0.6.0
--- a/go.sum
+++ b/go.sum
@@ -35,8 +35,6 @@ github.com/containerd/console v1.0.3 h1:lIr7SlA5PxZyMV30bDW0MGbiOPXwc63yRuCP0ARu
 github.com/containerd/console v1.0.3/go.mod h1:7LqA/THxQ86k76b8c/EMSiaJ3h1eZkMkXar0TQ1gf3U=
 github.com/cpuguy83/go-md2man/v2 v2.0.2/go.mod h1:tgQtvFlXSQOSOSIRvRPT7W67SCa46tRHOmNcaadrF8o=
 github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E=
-github.com/d4l3k/go-bfloat16 v0.0.0-20211005043715-690c3bdd05f1 h1:cBzrdJPAFBsgCrDPnZxlp1dF2+k4r1kVpD7+1S1PVjY=
-github.com/d4l3k/go-bfloat16 v0.0.0-20211005043715-690c3bdd05f1/go.mod h1:uw2gLcxEuYUlAd/EXyjc/v55nd3+47YAgWbSXVxPrNI=
 github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
 github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
 github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
--- a/integration/llm_image_test.go
+++ b/integration/llm_image_test.go
@@ -66,6 +66,35 @@ func TestIntegrationMllama(t *testing.T) {
 	DoGenerate(ctx, t, client, req, []string{resp}, 240*time.Second, 30*time.Second)
 }

+func TestIntegrationSplitBatch(t *testing.T) {
+	image, err := base64.StdEncoding.DecodeString(imageEncoding)
+	require.NoError(t, err)
+	req := api.GenerateRequest{
+		Model: "gemma3:4b",
+		// Fill up a chunk of the batch so the image will partially spill over into the next one
+		System: "Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed aliquet, justo in malesuada lobortis, odio ligula volutpat quam, quis faucibus ipsum magna quis sapien. Aliquam in venenatis diam, eu viverra magna. Phasellus imperdiet hendrerit volutpat. Vivamus sem ex, facilisis placerat felis non, dictum elementum est. Phasellus aliquam imperdiet lacus, eget placerat ligula sodales vel. Pellentesque nec auctor mi. Curabitur arcu nisi, faucibus eget nunc id, viverra interdum mi. Curabitur ornare ipsum ex, ac euismod ex aliquam in. Vestibulum id magna at purus accumsan fermentum. Proin scelerisque posuere nunc quis interdum. Maecenas sed mollis nisl. Etiam vitae ipsum interdum, placerat est quis, tincidunt velit. Nullam tempor nibh non lorem volutpat efficitur. Cras laoreet diam imperdiet ipsum auctor bibendum. Suspendisse ultrices urna sed metus sagittis suscipit. Quisque ullamcorper aliquam nibh ut mollis. Aenean dapibus mauris pharetra, venenatis elit ac, hendrerit odio. Cras vestibulum erat tempor, lobortis justo eu, lobortis ipsum. Nam laoreet dapibus sem. Proin vel diam ultrices, elementum ante et, ornare lectus. Proin eu accumsan nisl. Praesent ac ex vitae ipsum vulputate tristique facilisis sit amet lacus. Nullam faucibus magna a pellentesque pretium. Nunc lacinia ullamcorper sollicitudin. Donec vitae accumsan turpis, sed porttitor est. Donec porttitor mi vitae augue faucibus, vel mollis diam tincidunt.",
+		Prompt: "what does the text in this image say?",
+		Stream: &stream,
+		Options: map[string]interface{}{
+			"seed":        42,
+			"temperature": 0.0,
+		},
+		Images: []api.ImageData{
+			image,
+		},
+	}
+
+	// Note: sometimes it returns "the ollamas" sometimes "the ollams"
+	resp := "the ollam"
+	ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute)
+	defer cancel()
+	client, _, cleanup := InitServerConnection(ctx, t)
+	defer cleanup()
+	require.NoError(t, PullIfMissing(ctx, client, req.Model))
+	// llava models on CPU can be quite slow to start,
+	DoGenerate(ctx, t, client, req, []string{resp}, 120*time.Second, 30*time.Second)
+}
+
 const imageEncoding = `iVBORw0KGgoAAAANSUhEUgAAANIAAAB4CAYAAACHHqzKAAAAAXNSR0IArs4c6QAAAIRlWElmTU0AKgAAAAgABQESAAMAAAABAAEAAAEaAAUAAAABAAAASgEb
 AAUAAAABAAAAUgEoAAMAAAABAAIAAIdpAAQAAAABAAAAWgAAAAAAAABIAAAAAQAAAEgAAAABAAOgAQADAAAAAQABAACgAgAEAAAAAQAAANKgAwAEAAAAAQAA
 AHgAAAAAXdsepgAAAAlwSFlzAAALEwAACxMBAJqcGAAAAVlpVFh0WE1MOmNvbS5hZG9iZS54bXAAAAAAADx4OnhtcG1ldGEgeG1sbnM6eD0iYWRvYmU6bnM6
--- a/kvcache/causal.go
+++ b/kvcache/causal.go
@@ -21,9 +21,10 @@ type shiftFn func(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, e
 type Causal struct {
 	DType      ml.DType
 	Capacity   int32
-	causal     bool
 	windowSize int32

+	opts CausalOptions
+
 	// config controls mostly backend-specific optimizations
 	config *ml.CacheConfig

@@ -79,7 +80,6 @@ type cellRange struct {

 func NewCausalCache(shift shiftFn) *Causal {
 	return &Causal{
-		causal:     true,
 		windowSize: math.MaxInt32,
 		shiftFn:    shift,
 		ctxs:       make(map[int]ml.Context),
@@ -90,7 +90,6 @@ func NewCausalCache(shift shiftFn) *Causal {

 func NewSWACache(windowSize int32, shift shiftFn) *Causal {
 	return &Causal{
-		causal:     true,
 		windowSize: windowSize,
 		shiftFn:    shift,
 		ctxs:       make(map[int]ml.Context),
@@ -145,6 +144,7 @@ func (c *Causal) StartForward(ctx ml.Context, opts input.Options) error {
 	c.curBatchSize = len(opts.Positions)
 	c.curSequences = opts.Sequences
 	c.curPositions = opts.Positions
+	c.opts.Except = nil

 	var err error
 	c.curLoc, err = c.findStartLoc()
@@ -235,9 +235,10 @@ func (c *Causal) buildMask(ctx ml.Context) (ml.Tensor, error) {
 	mask := make([]float32, batchSize*length)

 	for i := range c.curBatchSize {
+		enabled := !slices.Contains(c.opts.Except, i)
 		for j := c.curCellRange.min; j <= c.curCellRange.max; j++ {
 			if !slices.Contains(c.cells[j].sequences, c.curSequences[i]) ||
-				(c.causal && c.cells[j].pos > c.curPositions[i]) ||
+				(enabled && c.cells[j].pos > c.curPositions[i]) ||
 				c.cells[j].pos < c.curPositions[i]-c.windowSize {
 				mask[i*length+(j-c.curCellRange.min)] = float32(math.Inf(-1))
 			}
@@ -404,15 +405,16 @@ func (c *Causal) SetLayer(layer int) {
 	c.curLayer = layer
 }

-// SetCausal enables or disables causal mask generation for subsequent calls to Get.
-// This state carries over to future forward passes. The default value is true.
-//
-// ctx may be set to nil if this is called from outside of a forward pass, for
-// example, when initializing the cache.
-func (c *Causal) SetCausal(ctx ml.Context, causal bool) {
-	if c.causal != causal {
-		c.causal = causal
+type CausalOptions struct {
+	// Enabled controls whether the causal mask is generated for a particular index in a batch
+	Except []int
+}

+// SetCausal disables causal mask generation for a particular range of indicies in
+// the current batch for subsequent calls to Get. The state resets for the next forward pass.
+func (c *Causal) SetCausal(ctx ml.Context, opts CausalOptions) {
+	if !slices.Equal(c.opts.Except, opts.Except) {
+		c.opts = opts
 		if ctx != nil {
 			var err error
 			c.curMask, err = c.buildMask(ctx)
--- a/kvcache/causal_test.go
+++ b/kvcache/causal_test.go
@@ -441,11 +441,19 @@ func (t *testTensor) Scale(ctx ml.Context, s float64) ml.Tensor {
 	panic("not implemented")
 }

+func (t *testTensor) AvgPool1D(ctx ml.Context, k, s, p int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) AvgPool2D(ctx ml.Context, k, s int, p float32) ml.Tensor {
+	panic("not implemented")
+}
+
 func (t *testTensor) Conv2D(ctx ml.Context, weight ml.Tensor, s0, s1, p0, p1, d0, d1 int) ml.Tensor {
 	panic("not implemented")
 }

-func (t *testTensor) RoPE(ctx ml.Context, positionIDs, ropeFactors ml.Tensor, dim uint32, base, scale float32) ml.Tensor {
+func (t *testTensor) RoPE(ctx ml.Context, positionIDs, ropeFactors ml.Tensor, dim, ropeType uint32, base, scale float32) ml.Tensor {
 	panic("not implemented")
 }

@@ -495,6 +503,10 @@ func (t *testTensor) Contiguous(ctx ml.Context) ml.Tensor {
 	panic("not implemented")
 }

+func (t *testTensor) Set(ctx ml.Context, t2 ml.Tensor, offset int, strides ...int) ml.Tensor {
+	panic("not implemented")
+}
+
 func (t *testTensor) Pad(ctx ml.Context, shape ...int) ml.Tensor {
 	panic("not implemented")
 }
--- a/llama/llama.cpp/src/llama-arch.cpp
+++ b/llama/llama.cpp/src/llama-arch.cpp
@@ -37,6 +37,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
    { LLM_ARCH_MINICPM3,         "minicpm3"         },
    { LLM_ARCH_GEMMA,            "gemma"            },
    { LLM_ARCH_GEMMA2,           "gemma2"           },
+    { LLM_ARCH_GEMMA3,           "gemma3"           },
    { LLM_ARCH_STARCODER2,       "starcoder2"       },
    { LLM_ARCH_MAMBA,            "mamba"            },
    { LLM_ARCH_XVERSE,           "xverse"           },
@@ -804,6 +805,24 @@ static const std::map<llm_arch, std::map<llm_tensor, const char *>> LLM_TENSOR_N
            { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
        },
    },
+    {
+        LLM_ARCH_GEMMA3,
+        {
+            { LLM_TENSOR_TOKEN_EMBD,      "token_embd" },
+            { LLM_TENSOR_OUTPUT_NORM,     "output_norm" },
+            { LLM_TENSOR_ATTN_NORM,       "blk.%d.attn_norm" },
+            { LLM_TENSOR_ATTN_Q,          "blk.%d.attn_q" },
+            { LLM_TENSOR_ATTN_K,          "blk.%d.attn_k" },
+            { LLM_TENSOR_ATTN_V,          "blk.%d.attn_v" },
+            { LLM_TENSOR_ATTN_OUT,        "blk.%d.attn_output" },
+            { LLM_TENSOR_ATTN_POST_NORM,  "blk.%d.post_attention_norm" },
+            { LLM_TENSOR_FFN_NORM,        "blk.%d.ffn_norm" },
+            { LLM_TENSOR_FFN_GATE,        "blk.%d.ffn_gate" },
+            { LLM_TENSOR_FFN_DOWN,        "blk.%d.ffn_down" },
+            { LLM_TENSOR_FFN_UP,          "blk.%d.ffn_up" },
+            { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
+        },
+    },
    {
        LLM_ARCH_STARCODER2,
        {
--- a/llama/llama.cpp/src/llama-arch.h
+++ b/llama/llama.cpp/src/llama-arch.h
@@ -41,6 +41,7 @@ enum llm_arch {
    LLM_ARCH_MINICPM3,
    LLM_ARCH_GEMMA,
    LLM_ARCH_GEMMA2,
+    LLM_ARCH_GEMMA3,
    LLM_ARCH_STARCODER2,
    LLM_ARCH_MAMBA,
    LLM_ARCH_XVERSE,
--- a/llama/llama.cpp/src/llama-model.cpp
+++ b/llama/llama.cpp/src/llama-model.cpp
@@ -878,6 +878,9 @@ void llama_model::load_hparams(llama_model_loader & ml) {
                    default: type = LLM_TYPE_UNKNOWN;
               }
            } break;
+        case LLM_ARCH_GEMMA3:
+            {
+            } break;
        case LLM_ARCH_STARCODER2:
            {
                ml.get_key(LLM_KV_ATTENTION_LAYERNORM_EPS, hparams.f_norm_eps);
@@ -2537,6 +2540,9 @@ bool llama_model::load_tensors(llama_model_loader & ml) {
                        layer.ffn_post_norm = create_tensor(tn(LLM_TENSOR_FFN_POST_NORM, "weight", i), {n_embd}, 0);
                    }
                } break;
+            case LLM_ARCH_GEMMA3:
+                {
+                } break;
            case LLM_ARCH_STARCODER2:
                {
                    tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
@@ -4029,6 +4035,7 @@ enum llama_rope_type llama_model_rope_type(const struct llama_model * model) {
        case LLM_ARCH_PHIMOE:
        case LLM_ARCH_GEMMA:
        case LLM_ARCH_GEMMA2:
+        case LLM_ARCH_GEMMA3:
        case LLM_ARCH_STARCODER2:
        case LLM_ARCH_OPENELM:
        case LLM_ARCH_GPTNEOX:
--- a/llama/llama.cpp/src/llama-quant.cpp
+++ b/llama/llama.cpp/src/llama-quant.cpp
@@ -737,6 +737,15 @@ static void llama_model_quantize_impl(const std::string & fname_inp, const std::
        // This used to be a regex, but <regex> has an extreme cost to compile times.
        bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?

+        // don't quantize vision stuff
+        quantize &= name.find("v.blk.") == std::string::npos;
+
+        quantize &= name.find("mm.mm_input_projection.weight") == std::string::npos;
+        quantize &= name.find("mm.mm_soft_emb_norm.weight") == std::string::npos;
+        quantize &= name.find("v.patch_embedding.weight") == std::string::npos;
+        quantize &= name.find("v.position_embedding.weight") == std::string::npos;
+        quantize &= name.find("v.post_layernorm.weight") == std::string::npos;
+
        // quantize only 2D and 3D tensors (experts)
        quantize &= (ggml_n_dims(tensor) >= 2);

--- a/llama/patches/0020-ollama-debug-tensor.patch
+++ b/llama/patches/0020-ollama-debug-tensor.patch
@@ -0,0 +1,33 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Michael Yang <mxyng@pm.me>
+Date: Sun, 9 Mar 2025 14:44:16 -0700
+Subject: [PATCH] ollama debug tensor
+
+---
+ ggml/src/ggml-cpu/ggml-cpu.c | 6 ++++++
+ 1 file changed, 6 insertions(+)
+
+diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c
+index 2f606d82..ec60e8fc 100644
+--- a/ggml/src/ggml-cpu/ggml-cpu.c
+++ b/ggml/src/ggml-cpu/ggml-cpu.c
+@@ -11,6 +11,8 @@
+ #include "ggml-threading.h"
+ #include "ggml.h"
+ 
+#include "ollama-debug.h"
+
+ #if defined(_MSC_VER) || defined(__MINGW32__)
+ #include <malloc.h> // using malloc.h with MSC/MINGW
+ #elif !defined(__FreeBSD__) && !defined(__NetBSD__) && !defined(__OpenBSD__)
+@@ -14103,6 +14105,10 @@ static thread_ret_t ggml_graph_compute_thread(void * data) {
+ 
+         ggml_compute_forward(&params, node);
+ 
+#ifdef OLLAMA_DEBUG
+        ollama_debug(node, true);
+#endif
+
+         if (state->ith == 0 && cplan->abort_callback &&
+                 cplan->abort_callback(cplan->abort_callback_data)) {
+             atomic_store_explicit(&tp->abort, node_n + 1, memory_order_relaxed);
--- a/llama/patches/0021-gemma3-quantization.patch
+++ b/llama/patches/0021-gemma3-quantization.patch
@@ -0,0 +1,113 @@
+From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
+From: Patrick Devine <patrick@infrahq.com>
+Date: Fri, 14 Mar 2025 16:33:23 -0700
+Subject: [PATCH] gemma3 quantization
+
+---
+ src/llama-arch.cpp  | 19 +++++++++++++++++++
+ src/llama-arch.h    |  1 +
+ src/llama-model.cpp |  7 +++++++
+ src/llama-quant.cpp |  9 +++++++++
+ 4 files changed, 36 insertions(+)
+
+diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
+index b6f20286..b443fcd3 100644
+--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
+@@ -37,6 +37,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
+     { LLM_ARCH_MINICPM3,         "minicpm3"         },
+     { LLM_ARCH_GEMMA,            "gemma"            },
+     { LLM_ARCH_GEMMA2,           "gemma2"           },
+    { LLM_ARCH_GEMMA3,           "gemma3"           },
+     { LLM_ARCH_STARCODER2,       "starcoder2"       },
+     { LLM_ARCH_MAMBA,            "mamba"            },
+     { LLM_ARCH_XVERSE,           "xverse"           },
+@@ -804,6 +805,24 @@ static const std::map<llm_arch, std::map<llm_tensor, const char *>> LLM_TENSOR_N
+             { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
+         },
+     },
+    {
+        LLM_ARCH_GEMMA3,
+        {
+            { LLM_TENSOR_TOKEN_EMBD,      "token_embd" },
+            { LLM_TENSOR_OUTPUT_NORM,     "output_norm" },
+            { LLM_TENSOR_ATTN_NORM,       "blk.%d.attn_norm" },
+            { LLM_TENSOR_ATTN_Q,          "blk.%d.attn_q" },
+            { LLM_TENSOR_ATTN_K,          "blk.%d.attn_k" },
+            { LLM_TENSOR_ATTN_V,          "blk.%d.attn_v" },
+            { LLM_TENSOR_ATTN_OUT,        "blk.%d.attn_output" },
+            { LLM_TENSOR_ATTN_POST_NORM,  "blk.%d.post_attention_norm" },
+            { LLM_TENSOR_FFN_NORM,        "blk.%d.ffn_norm" },
+            { LLM_TENSOR_FFN_GATE,        "blk.%d.ffn_gate" },
+            { LLM_TENSOR_FFN_DOWN,        "blk.%d.ffn_down" },
+            { LLM_TENSOR_FFN_UP,          "blk.%d.ffn_up" },
+            { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
+        },
+    },
+     {
+         LLM_ARCH_STARCODER2,
+         {
+diff --git a/src/llama-arch.h b/src/llama-arch.h
+index ec742224..aad92a5d 100644
+--- a/src/llama-arch.h
+++ b/src/llama-arch.h
+@@ -41,6 +41,7 @@ enum llm_arch {
+     LLM_ARCH_MINICPM3,
+     LLM_ARCH_GEMMA,
+     LLM_ARCH_GEMMA2,
+    LLM_ARCH_GEMMA3,
+     LLM_ARCH_STARCODER2,
+     LLM_ARCH_MAMBA,
+     LLM_ARCH_XVERSE,
+diff --git a/src/llama-model.cpp b/src/llama-model.cpp
+index ab1a07d1..70183041 100644
+--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
+@@ -878,6 +878,9 @@ void llama_model::load_hparams(llama_model_loader & ml) {
+                     default: type = LLM_TYPE_UNKNOWN;
+                }
+             } break;
+        case LLM_ARCH_GEMMA3:
+            {
+            } break;
+         case LLM_ARCH_STARCODER2:
+             {
+                 ml.get_key(LLM_KV_ATTENTION_LAYERNORM_EPS, hparams.f_norm_eps);
+@@ -2537,6 +2540,9 @@ bool llama_model::load_tensors(llama_model_loader & ml) {
+                         layer.ffn_post_norm = create_tensor(tn(LLM_TENSOR_FFN_POST_NORM, "weight", i), {n_embd}, 0);
+                     }
+                 } break;
+            case LLM_ARCH_GEMMA3:
+                {
+                } break;
+             case LLM_ARCH_STARCODER2:
+                 {
+                     tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
+@@ -4029,6 +4035,7 @@ enum llama_rope_type llama_model_rope_type(const struct llama_model * model) {
+         case LLM_ARCH_PHIMOE:
+         case LLM_ARCH_GEMMA:
+         case LLM_ARCH_GEMMA2:
+        case LLM_ARCH_GEMMA3:
+         case LLM_ARCH_STARCODER2:
+         case LLM_ARCH_OPENELM:
+         case LLM_ARCH_GPTNEOX:
+diff --git a/src/llama-quant.cpp b/src/llama-quant.cpp
+index 6eb1da08..d2f3a510 100644
+--- a/src/llama-quant.cpp
+++ b/src/llama-quant.cpp
+@@ -737,6 +737,15 @@ static void llama_model_quantize_impl(const std::string & fname_inp, const std::
+         // This used to be a regex, but <regex> has an extreme cost to compile times.
+         bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?
+ 
+        // don't quantize vision stuff
+        quantize &= name.find("v.blk.") == std::string::npos;
+
+        quantize &= name.find("mm.mm_input_projection.weight") == std::string::npos;
+        quantize &= name.find("mm.mm_soft_emb_norm.weight") == std::string::npos;
+        quantize &= name.find("v.patch_embedding.weight") == std::string::npos;
+        quantize &= name.find("v.position_embedding.weight") == std::string::npos;
+        quantize &= name.find("v.post_layernorm.weight") == std::string::npos;
+
+         // quantize only 2D and 3D tensors (experts)
+         quantize &= (ggml_n_dims(tensor) >= 2);
+ 
--- a/llm/memory.go
+++ b/llm/memory.go
@@ -218,8 +218,8 @@ func EstimateGPULayers(gpus []discover.GpuInfo, f *ggml.GGML, projectors []strin
 		if blk, ok := layers[fmt.Sprintf("blk.%d", i)]; ok {
 			layerSize = blk.Size()
 			layerSize += kv / f.KV().BlockCount()
+			memoryWeights += blk.Size()
 		}
-		memoryWeights += layerSize

 		if opts.NumGPU >= 0 && layerCount >= opts.NumGPU {
 			// Stop allocating on GPU(s) once we hit the users target NumGPU
@@ -376,7 +376,7 @@ func (m MemoryEstimate) LogValue() slog.Value {
 				// memory of the weights
 				"total", format.HumanBytes2(m.memoryWeights),
 				// memory of repeating layers
-				"repeating", format.HumanBytes2(m.memoryWeights-m.memoryLayerOutput),
+				"repeating", format.HumanBytes2(m.memoryWeights),
 				// memory of non-repeating layers
 				"nonrepeating", format.HumanBytes2(m.memoryLayerOutput),
 			),
--- a/llm/server.go
+++ b/llm/server.go
@@ -271,7 +271,7 @@ func NewLlamaServer(gpus discover.GpuInfoList, modelPath string, f *ggml.GGML, a

 	var llamaModel *llama.Model
 	var textProcessor model.TextProcessor
-	if envconfig.NewEngine() {
+	if envconfig.NewEngine() || f.KV().OllamaEngineRequired() {
 		textProcessor, err = model.NewTextProcessor(modelPath)
 		if err != nil {
 			// To prepare for opt-out mode, instead of treating this as an error, we fallback to the old runner
@@ -402,7 +402,7 @@ func NewLlamaServer(gpus discover.GpuInfoList, modelPath string, f *ggml.GGML, a
 			s.cmd.Env = append(s.cmd.Env, visibleDevicesEnv+"="+visibleDevicesEnvVal)
 		}

-		slog.Info("starting llama server", "cmd", s.cmd.String())
+		slog.Info("starting llama server", "cmd", s.cmd)
 		if envconfig.Debug() {
 			filteredEnv := []string{}
 			for _, ev := range s.cmd.Env {
@@ -470,7 +470,7 @@ const ( // iota is reset to 0
 	ServerStatusError
 )

-func (s ServerStatus) ToString() string {
+func (s ServerStatus) String() string {
 	switch s {
 	case ServerStatusReady:
 		return "llm server ready"
@@ -485,12 +485,9 @@ func (s ServerStatus) ToString() string {
 	}
 }

-type ServerStatusResp struct {
-	Status          string  `json:"status"`
-	SlotsIdle       int     `json:"slots_idle"`
-	SlotsProcessing int     `json:"slots_processing"`
-	Error           string  `json:"error"`
-	Progress        float32 `json:"progress"`
+type ServerStatusResponse struct {
+	Status   ServerStatus `json:"status"`
+	Progress float32      `json:"progress"`
 }

 func (s *llmServer) getServerStatus(ctx context.Context) (ServerStatus, error) {
@@ -502,7 +499,7 @@ func (s *llmServer) getServerStatus(ctx context.Context) (ServerStatus, error) {
 		}
 		if s.cmd.ProcessState.ExitCode() == -1 {
 			// Most likely a signal killed it, log some more details to try to help troubleshoot
-			slog.Warn("llama runner process no longer running", "sys", s.cmd.ProcessState.Sys(), "string", s.cmd.ProcessState.String())
+			slog.Warn("llama runner process no longer running", "sys", s.cmd.ProcessState.Sys(), "string", s.cmd.ProcessState)
 		}
 		return ServerStatusError, fmt.Errorf("llama runner process no longer running: %d %s", s.cmd.ProcessState.ExitCode(), msg)
 	}
@@ -527,21 +524,19 @@ func (s *llmServer) getServerStatus(ctx context.Context) (ServerStatus, error) {
 		return ServerStatusError, fmt.Errorf("read health request: %w", err)
 	}

-	var status ServerStatusResp
-	if err := json.Unmarshal(body, &status); err != nil {
+	var ssr ServerStatusResponse
+	if err := json.Unmarshal(body, &ssr); err != nil {
 		return ServerStatusError, fmt.Errorf("health unmarshal encode response: %w", err)
 	}

-	switch status.Status {
-	case "ok":
-		return ServerStatusReady, nil
-	case "no slot available":
-		return ServerStatusNoSlotsAvailable, nil
-	case "loading model":
-		s.loadProgress = status.Progress
-		return ServerStatusLoadingModel, nil
+	switch ssr.Status {
+	case ServerStatusLoadingModel:
+		s.loadProgress = ssr.Progress
+		return ssr.Status, nil
+	case ServerStatusReady, ServerStatusNoSlotsAvailable:
+		return ssr.Status, nil
 	default:
-		return ServerStatusError, fmt.Errorf("server error: %+v", status)
+		return ssr.Status, fmt.Errorf("server error: %+v", ssr)
 	}
 }

@@ -616,7 +611,7 @@ func (s *llmServer) WaitUntilRunning(ctx context.Context) error {
 		status, _ := s.getServerStatus(ctx)
 		if lastStatus != status && status != ServerStatusReady {
 			// Only log on status changes
-			slog.Info("waiting for server to become available", "status", status.ToString())
+			slog.Info("waiting for server to become available", "status", status)
 		}
 		switch status {
 		case ServerStatusReady:
@@ -630,7 +625,7 @@ func (s *llmServer) WaitUntilRunning(ctx context.Context) error {
 				slog.Debug(fmt.Sprintf("model load progress %0.2f", s.loadProgress))
 				stallTimer = time.Now().Add(stallDuration)
 			} else if !fullyLoaded && int(s.loadProgress*100.0) >= 100 {
-				slog.Debug("model load completed, waiting for server to become available", "status", status.ToString())
+				slog.Debug("model load completed, waiting for server to become available", "status", status)
 				stallTimer = time.Now().Add(stallDuration)
 				fullyLoaded = true
 			}
@@ -671,63 +666,26 @@ type ImageData struct {
 	AspectRatioID int    `json:"aspect_ratio_id"`
 }

-type completion struct {
-	Content      string `json:"content"`
-	Model        string `json:"model"`
-	Prompt       string `json:"prompt"`
-	Stop         bool   `json:"stop"`
-	StoppedLimit bool   `json:"stopped_limit"`
-
-	Timings struct {
-		PredictedN  int     `json:"predicted_n"`
-		PredictedMS float64 `json:"predicted_ms"`
-		PromptN     int     `json:"prompt_n"`
-		PromptMS    float64 `json:"prompt_ms"`
-	}
-}
-
 type CompletionRequest struct {
 	Prompt  string
 	Format  json.RawMessage
 	Images  []ImageData
 	Options *api.Options
+
+	Grammar string // set before sending the request to the subprocess
 }

 type CompletionResponse struct {
-	Content            string
-	DoneReason         string
-	Done               bool
-	PromptEvalCount    int
-	PromptEvalDuration time.Duration
-	EvalCount          int
-	EvalDuration       time.Duration
+	Content            string        `json:"content"`
+	DoneReason         string        `json:"done_reason"`
+	Done               bool          `json:"done"`
+	PromptEvalCount    int           `json:"prompt_eval_count"`
+	PromptEvalDuration time.Duration `json:"prompt_eval_duration"`
+	EvalCount          int           `json:"eval_count"`
+	EvalDuration       time.Duration `json:"eval_duration"`
 }

 func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn func(CompletionResponse)) error {
-	request := map[string]any{
-		"prompt":            req.Prompt,
-		"stream":            true,
-		"n_predict":         req.Options.NumPredict,
-		"n_keep":            req.Options.NumKeep,
-		"main_gpu":          req.Options.MainGPU,
-		"temperature":       req.Options.Temperature,
-		"top_k":             req.Options.TopK,
-		"top_p":             req.Options.TopP,
-		"min_p":             req.Options.MinP,
-		"typical_p":         req.Options.TypicalP,
-		"repeat_last_n":     req.Options.RepeatLastN,
-		"repeat_penalty":    req.Options.RepeatPenalty,
-		"presence_penalty":  req.Options.PresencePenalty,
-		"frequency_penalty": req.Options.FrequencyPenalty,
-		"mirostat":          req.Options.Mirostat,
-		"mirostat_tau":      req.Options.MirostatTau,
-		"mirostat_eta":      req.Options.MirostatEta,
-		"seed":              req.Options.Seed,
-		"stop":              req.Options.Stop,
-		"image_data":        req.Images,
-		"cache_prompt":      true,
-	}
-
 	if len(req.Format) > 0 {
 		switch string(req.Format) {
 		case `null`, `""`:
@@ -735,7 +693,7 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 			// these as "not set".
 			break
 		case `"json"`:
-			request["grammar"] = grammarJSON
+			req.Grammar = grammarJSON
 		default:
 			if req.Format[0] != '{' {
 				return fmt.Errorf("invalid format: %q; expected \"json\" or a valid JSON Schema object", req.Format)
@@ -746,10 +704,15 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 			if g == nil {
 				return fmt.Errorf("invalid JSON schema in format")
 			}
-			request["grammar"] = string(g)
+			req.Grammar = string(g)
 		}
 	}

+	if req.Options == nil {
+		opts := api.DefaultOptions()
+		req.Options = &opts
+	}
+
 	if err := s.sem.Acquire(ctx, 1); err != nil {
 		if errors.Is(err, context.Canceled) {
 			slog.Info("aborting completion request due to client closing the connection")
@@ -770,7 +733,7 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 	if err != nil {
 		return err
 	} else if status != ServerStatusReady {
-		return fmt.Errorf("unexpected server status: %s", status.ToString())
+		return fmt.Errorf("unexpected server status: %s", status)
 	}

 	// Handling JSON marshaling with special characters unescaped.
@@ -778,7 +741,7 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 	enc := json.NewEncoder(buffer)
 	enc.SetEscapeHTML(false)

-	if err := enc.Encode(request); err != nil {
+	if err := enc.Encode(req); err != nil {
 		return fmt.Errorf("failed to marshal data: %v", err)
 	}

@@ -829,7 +792,7 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 				evt = line
 			}

-			var c completion
+			var c CompletionResponse
 			if err := json.Unmarshal(evt, &c); err != nil {
 				return fmt.Errorf("error unmarshalling llm prediction response: %v", err)
 			}
@@ -853,20 +816,8 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 				})
 			}

-			if c.Stop {
-				doneReason := "stop"
-				if c.StoppedLimit {
-					doneReason = "length"
-				}
-
-				fn(CompletionResponse{
-					Done:               true,
-					DoneReason:         doneReason,
-					PromptEvalCount:    c.Timings.PromptN,
-					PromptEvalDuration: parseDurationMs(c.Timings.PromptMS),
-					EvalCount:          c.Timings.PredictedN,
-					EvalDuration:       parseDurationMs(c.Timings.PredictedMS),
-				})
+			if c.Done {
+				fn(c)
 				return nil
 			}
 		}
@@ -914,7 +865,7 @@ func (s *llmServer) Embedding(ctx context.Context, input string) ([]float32, err
 	if err != nil {
 		return nil, err
 	} else if status != ServerStatusReady {
-		return nil, fmt.Errorf("unexpected server status: %s", status.ToString())
+		return nil, fmt.Errorf("unexpected server status: %s", status)
 	}

 	data, err := json.Marshal(EmbeddingRequest{Content: input})
@@ -1059,12 +1010,3 @@ func (s *llmServer) EstimatedVRAMByGPU(gpuID string) uint64 {
 	}
 	return 0
 }
-
-func parseDurationMs(ms float64) time.Duration {
-	dur, err := time.ParseDuration(fmt.Sprintf("%fms", ms))
-	if err != nil {
-		panic(err)
-	}
-
-	return dur
-}
--- a/ml/backend.go
+++ b/ml/backend.go
@@ -5,6 +5,7 @@ import (
 	"encoding/binary"
 	"fmt"
 	"os"
+	"slices"
 	"strconv"
 	"strings"
 )
@@ -18,6 +19,7 @@ type Config interface {

 	Strings(string, ...[]string) []string
 	Uints(string, ...[]uint32) []uint32
+	Floats(string, ...[]float32) []float32
 }

 type Backend interface {
@@ -133,8 +135,10 @@ type Tensor interface {
 	RMSNorm(ctx Context, weight Tensor, eps float32) Tensor
 	Scale(ctx Context, s float64) Tensor

+	AvgPool2D(ctx Context, k, s int, p float32) Tensor
 	Conv2D(ctx Context, weight Tensor, s0, s1, p0, p1, d0, d1 int) Tensor
-	RoPE(ctx Context, positionIDs, ropeFactors Tensor, dim uint32, base, scale float32) Tensor
+
+	RoPE(ctx Context, positionIDs, ropeFactors Tensor, dim, ropeType uint32, base, scale float32) Tensor

 	Tanh(ctx Context) Tensor
 	GELU(ctx Context) Tensor
@@ -144,6 +148,7 @@ type Tensor interface {
 	View(ctx Context, offset int, shape ...int) Tensor
 	Permute(ctx Context, shape ...int) Tensor
 	Contiguous(ctx Context) Tensor
+	Set(ctx Context, t2 Tensor, offset int, strides ...int) Tensor

 	Pad(ctx Context, shape ...int) Tensor
 	Unpad(ctx Context, shape ...int) Tensor
@@ -241,16 +246,17 @@ func dump[S ~[]E, E number](ctx Context, t Tensor, items int, fn func(E) string)
 	}

 	shape := t.Shape()
+	slices.Reverse(shape)

 	var sb strings.Builder
 	var f func([]int, int)
 	f = func(dims []int, stride int) {
 		prefix := strings.Repeat(" ", len(shape)-len(dims)+1)
-		fmt.Fprint(&sb, "[")
-		defer func() { fmt.Fprint(&sb, "]") }()
+		sb.WriteString("[")
+		defer func() { sb.WriteString("]") }()
 		for i := 0; i < dims[0]; i++ {
 			if i >= items && i < dims[0]-items {
-				fmt.Fprint(&sb, "..., ")
+				sb.WriteString("..., ")
 				// skip to next printable element
 				skip := dims[0] - 2*items
 				if len(dims) > 1 {
@@ -265,9 +271,14 @@ func dump[S ~[]E, E number](ctx Context, t Tensor, items int, fn func(E) string)
 					fmt.Fprint(&sb, ",", strings.Repeat("\n", len(dims)-1), prefix)
 				}
 			} else {
-				fmt.Fprint(&sb, fn(s[stride+i]))
+				text := fn(s[stride+i])
+				if len(text) > 0 && text[0] != '-' {
+					sb.WriteString(" ")
+				}
+
+				sb.WriteString(text)
 				if i < dims[0]-1 {
-					fmt.Fprint(&sb, ", ")
+					sb.WriteString(", ")
 				}
 			}
 		}
--- a/ml/backend/ggml/ggml.go
+++ b/ml/backend/ggml/ggml.go
@@ -240,11 +240,22 @@ func New(r *os.File, params ml.BackendParams) (ml.Backend, error) {
 		switch {
 		case contains(t.Name, "position_embd", "token_embd", "token_norm_embd", "token_types"):
 			createTensor(tensor{source: t}, input.bts)
+			if _, ok := meta.Tensors().GroupLayers()["output"]; !ok && t.Name == "token_embd.weight" {
+				createTensor(tensor{source: t, target: "output.weight"}, output.bts)
+			}
 		case contains(t.Name, "cls", "output", "output_norm"):
 			createTensor(tensor{source: t}, output.bts)
 		case strings.HasPrefix(t.Name, "v.") || strings.HasPrefix(t.Name, "mm."):
 			// TODO: assign vision tensors to the gpu if possible
-			createTensor(tensor{source: t}, input.bts)
+			createTensor(tensor{source: t}, output.bts)
+		case contains(t.Name, "rope_freqs", "rope_factors_long", "rope_factors_short"):
+			// these tensors should be repeated per layer
+			for i, layer := range layers {
+				createTensor(tensor{
+					source: t,
+					target: "blk." + strconv.Itoa(i) + "." + t.Name,
+				}, layer.bts)
+			}
 		default:
 			layerIndex := -1
 			if fields := strings.FieldsFunc(t.Name, func(r rune) bool { return !unicode.IsNumber(r) }); len(fields) > 0 {
@@ -256,14 +267,8 @@ func New(r *os.File, params ml.BackendParams) (ml.Backend, error) {
 			if layerIndex >= 0 {
 				createTensor(tensor{source: t}, layers[layerIndex].bts)
 			} else {
-				// this is a repeating tensor that doesn't explicitly associated with a layer so
-				// duplicate it for each layer
-				for i, layer := range layers {
-					createTensor(tensor{
-						source: t,
-						target: "blk." + strconv.Itoa(i) + "." + t.Name,
-					}, layer.bts)
-				}
+				// load all other tensors on the cpu
+				createTensor(tensor{source: t}, input.bts)
 			}
 		}
 	}
@@ -307,17 +312,19 @@ func New(r *os.File, params ml.BackendParams) (ml.Backend, error) {
 					return fmt.Errorf("unassigned tensor: %s", t.Name)
 				}

-				bts := make([]byte, t.Size())
-				n, err := io.ReadFull(io.NewSectionReader(sr, int64(t.Offset), int64(t.Size())), bts)
-				if err != nil {
-					return err
+				bts := C.malloc(C.size_t(t.Size()))
+				if bts == nil {
+					return errors.New("failed to allocate tensor buffer")
+				}
+				defer C.free(bts)
+
+				buf := unsafe.Slice((*byte)(bts), t.Size())
+				n, err := io.ReadFull(io.NewSectionReader(sr, int64(t.Offset), int64(t.Size())), buf)
+				if err != nil || n != len(buf) {
+					return errors.New("read failed")
 				}

-				if n != len(bts) {
-					return errors.New("short read")
-				}
-
-				C.ggml_backend_tensor_set(tt, unsafe.Pointer(&bts[0]), 0, C.size_t(t.Size()))
+				C.ggml_backend_tensor_set(tt, bts, 0, C.size_t(t.Size()))
 				return nil
 			})
 		}
@@ -352,7 +359,7 @@ func New(r *os.File, params ml.BackendParams) (ml.Backend, error) {

 		if C.ggml_backend_is_cpu(b) {
 			// set number of threads for cpu backend
-			C.ggml_backend_cpu_set_n_threads(b, C.int(params.NumThreads))
+			C.ggml_backend_cpu_set_n_threads(b, C.int(Threads(params.NumThreads)))
 		}
 	}

@@ -366,7 +373,7 @@ func New(r *os.File, params ml.BackendParams) (ml.Backend, error) {
 			(*C.ggml_backend_buffer_type_t)(unsafe.Pointer(&schedBufts[0])),
 			C.int(len(schedBackends)),
 			C.size_t(maxGraphNodes),
-			true,
+			C._Bool(len(gpus) > 1 && slices.Contains(gpus, output.d)),
 		),
 		input:  deviceBufferTypes[input.d],
 		output: deviceBufferTypes[output.d],
@@ -893,10 +900,13 @@ func (t *Tensor) View(ctx ml.Context, offset int, shape ...int) ml.Tensor {
 }

 const (
-	ropeTypeNorm C.int = iota
+	ropeTypeNorm   C.int = 0
+	ropeTypeNeox   C.int = 2
+	ropeTypeMrope  C.int = 8
+	ropeTypeVision C.int = 24
 )

-func (t *Tensor) RoPE(ctx ml.Context, positionIDs, ropeFactors ml.Tensor, ropeDim uint32, ropeBase, ropeScale float32) ml.Tensor {
+func (t *Tensor) RoPE(ctx ml.Context, positionIDs, ropeFactors ml.Tensor, ropeDim, ropeType uint32, ropeBase, ropeScale float32) ml.Tensor {
 	if ropeFactors == nil {
 		ropeFactors = &Tensor{b: t.b}
 	}
@@ -911,8 +921,8 @@ func (t *Tensor) RoPE(ctx ml.Context, positionIDs, ropeFactors ml.Tensor, ropeDi
 		t: C.ggml_rope_ext(
 			ctx.(*Context).ctx, dequant, positionIDs.(*Tensor).t, ropeFactors.(*Tensor).t,
 			C.int(ropeDim),
-			131072,       // YaRN n_ctx_train
-			ropeTypeNorm, // ROPE_TYPE_NORM
+			C.int(ropeType),
+			131072, // YaRN n_ctx_train
 			C.float(ropeBase),
 			C.float(ropeScale),
 			0.,  // YaRN ext_factor
@@ -944,6 +954,27 @@ func (t *Tensor) Conv2D(ctx ml.Context, t2 ml.Tensor, s0, s1, p0, p1, d0, d1 int
 	}
 }

+func (t *Tensor) AvgPool2D(ctx ml.Context, k, s int, p float32) ml.Tensor {
+	return &Tensor{
+		b: t.b,
+		t: C.ggml_pool_2d(ctx.(*Context).ctx, t.t, C.GGML_OP_POOL_AVG, C.int(k), C.int(k), C.int(s), C.int(s), C.float(p), C.float(p)),
+	}
+}
+
+func (t *Tensor) Set(ctx ml.Context, t2 ml.Tensor, offset int, strides ...int) ml.Tensor {
+	var tt *C.struct_ggml_tensor
+	switch len(strides) {
+	case 0:
+		tt = C.ggml_set_1d(ctx.(*Context).ctx, t.t, t2.(*Tensor).t, C.size_t(offset))
+	case 1:
+		tt = C.ggml_set_2d(ctx.(*Context).ctx, t.t, t2.(*Tensor).t, C.size_t(offset), C.size_t(strides[0]))
+	default:
+		panic("unsupported number of dimensions")
+	}
+
+	return &Tensor{b: t.b, t: tt}
+}
+
 func (t *Tensor) ScaledDotProductAttention(ctx ml.Context, key, value, mask ml.Tensor, scale float64) ml.Tensor {
 	var kqMask *C.struct_ggml_tensor
 	if mask != nil {
--- a/ml/backend/ggml/ggml/include/ollama-debug.h
+++ b/ml/backend/ggml/ggml/include/ollama-debug.h
@@ -0,0 +1,11 @@
+#include "ggml.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+void ollama_debug(const struct ggml_tensor *tensor, bool verbose);
+
+#ifdef __cplusplus
+}
+#endif
--- a/ml/backend/ggml/ggml/src/ggml-cpu/cpu_debug.go
+++ b/ml/backend/ggml/ggml/src/ggml-cpu/cpu_debug.go
@@ -0,0 +1,6 @@
+//go:build debug
+
+package cpu
+
+// #cgo CPPFLAGS: -DOLLAMA_DEBUG
+import "C"
--- a/ml/backend/ggml/ggml/src/ggml-cpu/ggml-cpu.c
+++ b/ml/backend/ggml/ggml/src/ggml-cpu/ggml-cpu.c
@@ -11,6 +11,8 @@
 #include "ggml-threading.h"
 #include "ggml.h"

+#include "ollama-debug.h"
+
 #if defined(_MSC_VER) || defined(__MINGW32__)
 #include <malloc.h> // using malloc.h with MSC/MINGW
 #elif !defined(__FreeBSD__) && !defined(__NetBSD__) && !defined(__OpenBSD__)
@@ -14103,6 +14105,10 @@ static thread_ret_t ggml_graph_compute_thread(void * data) {

        ggml_compute_forward(&params, node);

+#ifdef OLLAMA_DEBUG
+        ollama_debug(node, true);
+#endif
+
        if (state->ith == 0 && cplan->abort_callback &&
                cplan->abort_callback(cplan->abort_callback_data)) {
            atomic_store_explicit(&tp->abort, node_n + 1, memory_order_relaxed);
--- a/ml/backend/ggml/ggml/src/ollama-debug.c
+++ b/ml/backend/ggml/ggml/src/ollama-debug.c
@@ -0,0 +1,116 @@
+#include <string.h>
+#include <inttypes.h>
+
+#include "ollama-debug.h"
+
+static int mul(int64_t *dims, int ndims) {
+    int result = 1;
+    for (int i = 0; i < ndims; i++) {
+        result *= dims[i];
+    }
+
+    return result;
+}
+
+static void repeat(char c, int n) {
+    for (int i = 0; i < n; i++) {
+        fprintf(stderr, "%c", c);
+    }
+}
+
+static void print_tensor(const void *tensor, void (*cb)(const void *, int),
+                         int shape,
+                         int64_t *dims, int ndims, int stride,
+                         int nitems, int pad) {
+    fprintf(stderr, "[");
+    for (int i = 0; i < dims[0]; i++) {
+        if (i >= nitems && i < dims[0] - nitems) {
+            fprintf(stderr, "... (%" PRIi64 " more), ", dims[0] - 2 * nitems);
+            int skip = dims[0] - 2 * nitems;
+            if (ndims > 1) {
+                stride += mul(dims + 1, ndims - 1) * skip;
+                repeat('\n', ndims - 1);
+                repeat(' ', shape - ndims + 1 + pad);
+            }
+            i += skip - 1;
+        } else if (ndims > 1) {
+            print_tensor(tensor, cb, shape, dims + 1, ndims - 1, stride,
+                         nitems, pad);
+            stride += mul(dims + 1, ndims - 1);
+            if (i < dims[0] - 1) {
+                fprintf(stderr, ", ");
+                repeat('\n', ndims - 1);
+                repeat(' ', shape - ndims + 1 + pad);
+            }
+        } else {
+            cb(tensor, stride + i);
+            if (i < dims[0] - 1) {
+                fprintf(stderr, ", ");
+            }
+        }
+    }
+    fprintf(stderr, "]");
+}
+
+static void print_tensor_f16(const void *tensor, int i) {
+    float value = ggml_fp16_to_fp32(((const ggml_fp16_t *)tensor)[i]);
+    fprintf(stderr, "%s%f", value < 0 ? "" : " ", value);
+}
+
+static void print_tensor_f32(const void *tensor, int i) {
+    float value = ((const float *)tensor)[i];
+    fprintf(stderr, "%s%f", value < 0 ? "" : " ", value);
+}
+
+static void print_tensor_i32(const void *tensor, int i) {
+    int32_t value = ((const int32_t *)tensor)[i];
+    fprintf(stderr, "%s%d", value < 0 ? "" : " ", value);
+}
+
+static void ollama_debug_tensor(const struct ggml_tensor *tensor, bool verbose, const char *prefix, int indent) {
+    fprintf(stderr, "%s%s %s (%s): [%" PRIi64 " %" PRIi64 " %" PRIi64 " %" PRIi64 "]\n", prefix, tensor->name,
+            ggml_op_name(tensor->op), ggml_type_name(tensor->type), tensor->ne[0],
+            tensor->ne[1], tensor->ne[2], tensor->ne[3]);
+
+    if (!verbose) {
+        return;
+    }
+
+    for (int i = 0; i < indent; i++) {
+        fprintf(stderr, " ");
+    }
+
+    switch (tensor->type) {
+    case GGML_TYPE_F16:
+        print_tensor(ggml_get_data(tensor), print_tensor_f16, ggml_n_dims(tensor),
+                     (int64_t *)tensor->ne, ggml_n_dims(tensor), 0, 3, indent);
+        break;
+    case GGML_TYPE_F32:
+        print_tensor(ggml_get_data(tensor), print_tensor_f32, ggml_n_dims(tensor),
+                     (int64_t *)tensor->ne, ggml_n_dims(tensor), 0, 3, indent);
+        break;
+    case GGML_TYPE_I32:
+        print_tensor(ggml_get_data(tensor), print_tensor_i32, ggml_n_dims(tensor),
+                     (int64_t *)tensor->ne, ggml_n_dims(tensor), 0, 3, indent);
+        break;
+    default:
+        fprintf(stderr, "<unsupported type>\n");
+        return;
+    }
+
+    fprintf(stderr, "\n");
+}
+
+void ollama_debug(const struct ggml_tensor *tensor, bool verbose) {
+    ollama_debug_tensor(tensor, verbose, ">>> ", 4);
+
+    for (int i = 0; i < GGML_MAX_SRC && tensor->src[i] != NULL; ++i) {
+        char src[8];
+        const int n = snprintf(src, sizeof(src), " src%d ", i);
+        if (n >= sizeof(src)) {
+            src[sizeof(src) - 1] = '\0';
+        }
+
+        ollama_debug_tensor(tensor->src[i], verbose, src, 4);
+    }
+}
--- a/ml/backend/ggml/threads.go
+++ b/ml/backend/ggml/threads.go
@@ -0,0 +1,7 @@
+//go:build !debug
+
+package ggml
+
+func Threads(n int) int {
+	return n
+}
--- a/ml/backend/ggml/threads_debug.go
+++ b/ml/backend/ggml/threads_debug.go
@@ -0,0 +1,7 @@
+//go:build debug
+
+package ggml
+
+func Threads(_ int) int {
+	return 1
+}
--- a/model/input/input.go
+++ b/model/input/input.go
@@ -15,6 +15,12 @@ type Input struct {
 	// stored in Multimodal, used for caching and comparing
 	// equality.
 	MultimodalHash uint64
+
+	// SameBatch forces the following number of tokens to be processed
+	// in a single batch, breaking and extending batches as needed.
+	// Useful for things like images that must be processed in one
+	// shot.
+	SameBatch int
 }

 // MultimodalIndex is a multimodal element (such as an image)
--- a/model/model.go
+++ b/model/model.go
@@ -22,6 +22,8 @@ import (
 	"github.com/ollama/ollama/model/input"
 )

+var ErrNoVisionModel = errors.New("this model is missing data required for image input")
+
 // Model implements a specific model architecture, defining the forward pass and any model-specific configuration
 type Model interface {
 	Forward(ml.Context, input.Options) (ml.Tensor, error)
@@ -58,7 +60,7 @@ type MultimodalProcessor interface {
 	// This function is also responsible for updating MultimodalHash for any Multimodal
 	// that is modified to ensure that there is a unique hash value that accurately
 	// represents the contents.
-	PostTokenize(ml.Context, []input.Input) ([]input.Input, error)
+	PostTokenize([]input.Input) ([]input.Input, error)
 }

 // Base implements the common fields and methods for all models
--- a/model/models/gemma2/model.go
+++ b/model/models/gemma2/model.go
@@ -0,0 +1,220 @@
+package gemma2
+
+import (
+	"math"
+
+	"github.com/ollama/ollama/kvcache"
+	"github.com/ollama/ollama/ml"
+	"github.com/ollama/ollama/ml/nn"
+	"github.com/ollama/ollama/model"
+	"github.com/ollama/ollama/model/input"
+)
+
+type Options struct {
+	hiddenSize, numHeads, numKVHeads int
+	attnKeyLen, attnValLen           int
+	eps, ropeBase, ropeScale         float32
+	attnLogitSoftcap                 float32
+	finalLogitSoftcap                float32
+	largeModelScaling                bool
+}
+
+type Model struct {
+	model.Base
+	model.SentencePieceModel
+
+	TokenEmbedding *nn.Embedding `gguf:"token_embd"`
+	Layers         []Layer       `gguf:"blk"`
+	OutputNorm     *nn.RMSNorm   `gguf:"output_norm"`
+	Output         *nn.Linear    `gguf:"output,alt:token_embd"` // just set to token_embd?
+
+	*Options
+}
+
+const (
+	gemma27BLayerCount = 46
+)
+
+func New(c ml.Config) (model.Model, error) {
+	m := Model{
+		SentencePieceModel: model.NewSentencePieceModel(
+			c.String("tokenizer.ggml.pretokenizer", `(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+`),
+			&model.Vocabulary{
+				Values: c.Strings("tokenizer.ggml.tokens"),
+				Scores: c.Floats("tokenizer.ggml.scores"),
+				Types:  c.Uints("tokenizer.ggml.token_type"),
+				BOS:    int32(c.Uint("tokenizer.ggml.bos_token_id")),
+				EOS:    int32(c.Uint("tokenizer.ggml.eos_token_id")),
+			},
+		),
+		Layers: make([]Layer, c.Uint("block_count")),
+		Options: &Options{
+			hiddenSize:        int(c.Uint("embedding_length")),
+			numHeads:          int(c.Uint("attention.head_count")),
+			numKVHeads:        int(c.Uint("attention.head_count_kv")),
+			attnKeyLen:        int(c.Uint("attention.key_length")),
+			attnValLen:        int(c.Uint("attention.value_length")),
+			eps:               c.Float("attention.layer_norm_rms_epsilon"),
+			ropeBase:          c.Float("rope.freq_base", 10000.0),
+			ropeScale:         c.Float("rope.freq_scale", 1.0),
+			attnLogitSoftcap:  c.Float("attn_logit_softcapping"),
+			finalLogitSoftcap: c.Float("final_logit_softcapping"),
+		},
+	}
+
+	slidingWindowLen := int32(c.Uint("attention.sliding_window"))
+	m.Cache = kvcache.NewWrapperCache(kvcache.NewSWACache(slidingWindowLen, m.Shift), kvcache.NewCausalCache(m.Shift))
+	m.Cache.SetConfig(ml.CacheConfig{})
+
+	return &m, nil
+}
+
+type SelfAttention struct {
+	Query  *nn.Linear `gguf:"attn_q"`
+	Key    *nn.Linear `gguf:"attn_k"`
+	Value  *nn.Linear `gguf:"attn_v"`
+	Output *nn.Linear `gguf:"attn_output"`
+}
+
+func (sa *SelfAttention) Forward(ctx ml.Context, hiddenState, positionIDs ml.Tensor, cache kvcache.Cache, opts *Options) ml.Tensor {
+	batchSize := hiddenState.Dim(1)
+	ropeType := uint32(2)
+
+	q := sa.Query.Forward(ctx, hiddenState)
+	q = q.Reshape(ctx, opts.attnKeyLen, opts.numHeads, batchSize)
+	q = q.RoPE(ctx, positionIDs, nil, uint32(opts.attnKeyLen), ropeType, opts.ropeBase, opts.ropeScale)
+
+	if opts.largeModelScaling {
+		q = q.Scale(ctx, 1.0/math.Sqrt(float64(opts.hiddenSize/opts.numHeads)))
+	} else {
+		q = q.Scale(ctx, 1.0/math.Sqrt(float64(opts.attnKeyLen)))
+	}
+
+	k := sa.Key.Forward(ctx, hiddenState)
+	k = k.Reshape(ctx, opts.attnKeyLen, opts.numKVHeads, batchSize)
+	k = k.RoPE(ctx, positionIDs, nil, uint32(opts.attnKeyLen), ropeType, opts.ropeBase, opts.ropeScale)
+
+	v := sa.Value.Forward(ctx, hiddenState)
+	v = v.Reshape(ctx, opts.attnValLen, opts.numKVHeads, batchSize)
+
+	cache.Put(ctx, k, v)
+	k, v, mask := cache.Get(ctx)
+
+	q = q.Permute(ctx, 0, 2, 1, 3)
+	k = k.Permute(ctx, 0, 2, 1, 3)
+	v = v.Permute(ctx, 1, 2, 0, 3).Contiguous(ctx)
+
+	kq := k.Mulmat(ctx, q)
+
+	// logit softcap
+	kq = kq.Scale(ctx, 1.0/float64(opts.attnLogitSoftcap))
+	kq = kq.Tanh(ctx)
+	kq = kq.Scale(ctx, float64(opts.attnLogitSoftcap))
+
+	kq = kq.Add(ctx, mask)
+	kq = kq.Softmax(ctx)
+
+	kqv := v.Mulmat(ctx, kq)
+	kqv = kqv.Permute(ctx, 0, 2, 1, 3).Contiguous(ctx)
+	kqv = kqv.Reshape(ctx, opts.attnValLen*opts.numHeads, batchSize)
+
+	return sa.Output.Forward(ctx, kqv)
+}
+
+func (m *Model) Shift(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, error) {
+	return key.RoPE(ctx, shift, nil, uint32(m.Options.attnKeyLen), uint32(2), m.Options.ropeBase, m.Options.ropeScale), nil
+}
+
+type MLP struct {
+	Up   *nn.Linear `gguf:"ffn_up"`
+	Down *nn.Linear `gguf:"ffn_down"`
+	Gate *nn.Linear `gguf:"ffn_gate"`
+}
+
+func (mlp *MLP) Forward(ctx ml.Context, hiddenState ml.Tensor, opts *Options) ml.Tensor {
+	hiddenState = mlp.Gate.Forward(ctx, hiddenState).GELU(ctx).Mul(ctx, mlp.Up.Forward(ctx, hiddenState))
+	return mlp.Down.Forward(ctx, hiddenState)
+}
+
+type Layer struct {
+	AttentionNorm     *nn.RMSNorm `gguf:"attn_norm"`
+	SelfAttention     *SelfAttention
+	PostAttentionNorm *nn.RMSNorm `gguf:"post_attention_norm"`
+	MLPNorm           *nn.RMSNorm `gguf:"ffn_norm"`
+	MLP               *MLP
+	PostMLPNorm       *nn.RMSNorm `gguf:"post_ffw_norm"`
+}
+
+func (l *Layer) Forward(ctx ml.Context, hiddenState, positionIDs, outputs ml.Tensor, cache kvcache.Cache, opts *Options) ml.Tensor {
+	residual := hiddenState
+
+	hiddenState = l.AttentionNorm.Forward(ctx, hiddenState, opts.eps)
+	hiddenState = l.SelfAttention.Forward(ctx, hiddenState, positionIDs, cache, opts)
+	hiddenState = l.PostAttentionNorm.Forward(ctx, hiddenState, opts.eps)
+
+	// In the final layer (outputs != nil), optimize by pruning to just the token positions
+	// we need logits for.
+	if outputs != nil {
+		hiddenState = hiddenState.Rows(ctx, outputs)
+		residual = residual.Rows(ctx, outputs)
+	}
+
+	hiddenState = hiddenState.Add(ctx, residual)
+	residual = hiddenState
+
+	hiddenState = l.MLPNorm.Forward(ctx, hiddenState, opts.eps)
+	hiddenState = l.MLP.Forward(ctx, hiddenState, opts)
+	hiddenState = l.PostMLPNorm.Forward(ctx, hiddenState, opts.eps)
+	return hiddenState.Add(ctx, residual)
+}
+
+func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
+	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
+	if err != nil {
+		return nil, err
+	}
+
+	positions, err := ctx.Input().FromIntSlice(opts.Positions, len(opts.Positions))
+	if err != nil {
+		return nil, err
+	}
+
+	outputs, err := ctx.Output().FromIntSlice(opts.Outputs, len(opts.Outputs))
+	if err != nil {
+		return nil, err
+	}
+
+	hiddenState := m.TokenEmbedding.Forward(ctx, inputs)
+	hiddenState = hiddenState.Scale(ctx, math.Sqrt(float64(m.Options.hiddenSize)))
+
+	if len(m.Layers) == gemma27BLayerCount {
+		m.Options.largeModelScaling = true
+	}
+
+	for i, layer := range m.Layers {
+		cacheType := i % 2
+		m.Cache.SetLayer(i)
+		wc := m.Cache.(*kvcache.WrapperCache)
+		wc.SetLayerType(cacheType)
+
+		var lastLayerOutputs ml.Tensor
+		if i == len(m.Layers)-1 {
+			lastLayerOutputs = outputs
+		}
+
+		hiddenState = layer.Forward(ctx, hiddenState, positions, lastLayerOutputs, m.Cache, m.Options)
+	}
+
+	hiddenState = m.OutputNorm.Forward(ctx, hiddenState, m.eps)
+	hiddenState = m.Output.Forward(ctx, hiddenState)
+
+	// final logit softcap
+	hiddenState = hiddenState.Scale(ctx, 1.0/float64(m.Options.finalLogitSoftcap))
+	hiddenState = hiddenState.Tanh(ctx)
+	hiddenState = hiddenState.Scale(ctx, float64(m.Options.finalLogitSoftcap))
+	return hiddenState.Rows(ctx, outputs), nil
+}
+
+func init() {
+	model.Register("gemma2", New)
+}
--- a/model/models/gemma3/model.go
+++ b/model/models/gemma3/model.go
@@ -0,0 +1,163 @@
+package gemma3
+
+import (
+	"bytes"
+	"image"
+	"math"
+	"slices"
+
+	"github.com/ollama/ollama/kvcache"
+	"github.com/ollama/ollama/ml"
+	"github.com/ollama/ollama/ml/nn"
+	"github.com/ollama/ollama/model"
+	"github.com/ollama/ollama/model/input"
+)
+
+type Model struct {
+	model.Base
+	model.SentencePieceModel
+
+	*VisionModel `gguf:"v,vision"`
+	*TextModel
+
+	*MultiModalProjector `gguf:"mm"`
+
+	ImageProcessor
+}
+
+var _ model.MultimodalProcessor = (*Model)(nil)
+
+type MultiModalProjector struct {
+	SoftEmbNorm     *nn.RMSNorm `gguf:"mm_soft_emb_norm"`
+	InputProjection *nn.Linear  `gguf:"mm_input_projection"`
+
+	tokensPerImage int
+}
+
+func (p *MultiModalProjector) Forward(ctx ml.Context, visionOutputs ml.Tensor, imageSize, patchSize int, eps float32) ml.Tensor {
+	l := visionOutputs.Dim(0)
+
+	visionOutputs = visionOutputs.Permute(ctx, 1, 0, 2, 3).Contiguous(ctx)
+	patchesPerImage := imageSize / patchSize
+	visionOutputs = visionOutputs.Reshape(ctx, patchesPerImage, patchesPerImage, l)
+
+	kernelSize := patchesPerImage / int(math.Sqrt(float64(p.tokensPerImage)))
+	visionOutputs = visionOutputs.AvgPool2D(ctx, kernelSize, kernelSize, 0)
+	visionOutputs = visionOutputs.Reshape(ctx, visionOutputs.Dim(0)*visionOutputs.Dim(1), l)
+	visionOutputs = visionOutputs.Permute(ctx, 1, 0, 2, 3).Contiguous(ctx)
+	visionOutputs = p.SoftEmbNorm.Forward(ctx, visionOutputs, eps)
+
+	// TODO: inputProjection must be transposed since they're incompatible with visionOutputs
+	visionOutputs = p.InputProjection.Weight.Permute(ctx, 1, 0, 2, 3).Contiguous(ctx).Mulmat(ctx, visionOutputs)
+	return visionOutputs
+}
+
+func New(c ml.Config) (model.Model, error) {
+	m := Model{
+		SentencePieceModel: model.NewSentencePieceModel(
+			c.String("tokenizer.ggml.pretokenizer", `(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+`),
+			&model.Vocabulary{
+				Values: c.Strings("tokenizer.ggml.tokens"),
+				Scores: c.Floats("tokenizer.ggml.scores"),
+				Types:  c.Uints("tokenizer.ggml.token_type"),
+				BOS:    int32(c.Uint("tokenizer.ggml.bos_token_id")),
+				AddBOS: c.Bool("tokenizer.ggml.add_bos_token", true),
+				EOS:    int32(1),
+				AddEOS: c.Bool("tokenizer.ggml.add_eos_token", false),
+				EOT:    int32(106),
+				AddEOT: c.Bool("tokenizer.ggml.add_eot_token", false),
+			},
+		),
+		ImageProcessor: newImageProcessor(c),
+		VisionModel:    newVisionModel(c),
+		TextModel:      newTextModel(c),
+		MultiModalProjector: &MultiModalProjector{
+			tokensPerImage: int(c.Uint("mm_tokens_per_image", 256)),
+		},
+	}
+
+	slidingWindowLen := int32(c.Uint("attention.sliding_window"))
+	m.Cache = kvcache.NewWrapperCache(kvcache.NewSWACache(slidingWindowLen, m.Shift), kvcache.NewCausalCache(m.Shift))
+
+	return &m, nil
+}
+
+func (m *Model) EncodeMultimodal(ctx ml.Context, multimodalData []byte) (any, error) {
+	if len(m.VisionModel.Layers) == 0 {
+		return nil, model.ErrNoVisionModel
+	}
+
+	image, _, err := image.Decode(bytes.NewReader(multimodalData))
+	if err != nil {
+		return nil, err
+	}
+
+	f32s, err := m.ImageProcessor.ProcessImage(image)
+	if err != nil {
+		return nil, err
+	}
+
+	pixelValues, err := ctx.Input().FromFloatSlice(f32s,
+		m.ImageProcessor.imageSize,
+		m.ImageProcessor.imageSize,
+		m.ImageProcessor.numChannels,
+	)
+	if err != nil {
+		return nil, err
+	}
+
+	visionOutputs := m.VisionModel.Forward(ctx, pixelValues)
+	visionOutputs = m.MultiModalProjector.Forward(ctx, visionOutputs, m.imageSize, m.patchSize, m.VisionModel.eps)
+	return visionOutputs, nil
+}
+
+func (m *Model) PostTokenize(inputs []input.Input) ([]input.Input, error) {
+	var result []input.Input
+
+	for _, inp := range inputs {
+		if inp.Multimodal == nil {
+			result = append(result, inp)
+		} else {
+			inputMultimodal := inp.Multimodal.(ml.Tensor)
+
+			result = append(result,
+				input.Input{Token: 108, SameBatch: inputMultimodal.Dim(1) + 3},               // "\n\n"
+				input.Input{Token: 255999},                                                   // "<start_of_image>""
+				input.Input{Multimodal: inputMultimodal, MultimodalHash: inp.MultimodalHash}, // image data is on the first placeholder
+			)
+
+			// add image token placeholders
+			result = append(result, slices.Repeat([]input.Input{{Token: 0}}, inputMultimodal.Dim(1)-1)...)
+
+			result = append(result,
+				input.Input{Token: 256000}, // <end_of_image>
+				input.Input{Token: 108},    // "\n\n"
+			)
+		}
+	}
+
+	return result, nil
+}
+
+func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
+	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
+	if err != nil {
+		return nil, err
+	}
+
+	positions, err := ctx.Input().FromIntSlice(opts.Positions, len(opts.Positions))
+	if err != nil {
+		return nil, err
+	}
+
+	outputs, err := ctx.Output().FromIntSlice(opts.Outputs, len(opts.Outputs))
+	if err != nil {
+		return nil, err
+	}
+
+	return m.TextModel.Forward(ctx, inputs, positions, outputs, opts, m.Cache), nil
+}
+
+func init() {
+	model.Register("gemma3", New)
+}
--- a/model/models/gemma3/model_text.go
+++ b/model/models/gemma3/model_text.go
@@ -0,0 +1,214 @@
+package gemma3
+
+import (
+	"math"
+
+	"github.com/ollama/ollama/kvcache"
+	"github.com/ollama/ollama/ml"
+	"github.com/ollama/ollama/ml/nn"
+	"github.com/ollama/ollama/model"
+	"github.com/ollama/ollama/model/input"
+)
+
+type TextOptions struct {
+	hiddenSize, numHeads, numKVHeads int
+	attnKeyLen, attnValLen           int
+	eps, ropeScale                   float32
+	ropeLocalBase, ropeGlobalBase    float32
+	largeModelScaling                bool
+}
+
+type TextModel struct {
+	model.Base
+	model.SentencePieceModel
+
+	TokenEmbedding *nn.Embedding `gguf:"token_embd"`
+	Layers         []TextLayer   `gguf:"blk"`
+	OutputNorm     *nn.RMSNorm   `gguf:"output_norm"`
+	Output         *nn.Linear    `gguf:"output,alt:token_embd"`
+
+	*TextOptions
+}
+
+const (
+	gemmaGlobalCacheCount = 6
+	gemma27BLayerCount    = 62
+)
+
+const (
+	cacheTypeSWA = iota
+	cacheTypeCausal
+)
+
+func newTextModel(c ml.Config) *TextModel {
+	numBlocks := int(c.Uint("block_count"))
+
+	m := TextModel{
+		SentencePieceModel: model.NewSentencePieceModel(
+			c.String("tokenizer.ggml.pretokenizer", `(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+`),
+			&model.Vocabulary{
+				Values: c.Strings("tokenizer.ggml.tokens"),
+				Scores: c.Floats("tokenizer.ggml.scores"),
+				Types:  c.Uints("tokenizer.ggml.token_type"),
+				BOS:    int32(c.Uint("tokenizer.ggml.bos_token_id")),
+				EOS:    int32(c.Uint("tokenizer.ggml.eos_token_id")),
+			},
+		),
+		Layers: make([]TextLayer, numBlocks),
+		TextOptions: &TextOptions{
+			hiddenSize:     int(c.Uint("embedding_length")),
+			numHeads:       int(c.Uint("attention.head_count")),
+			numKVHeads:     int(c.Uint("attention.head_count_kv")),
+			attnKeyLen:     int(c.Uint("attention.key_length", 256)),
+			attnValLen:     int(c.Uint("attention.value_length", 256)),
+			eps:            c.Float("attention.layer_norm_rms_epsilon", 1e-06),
+			ropeLocalBase:  c.Float("rope.local.freq_base", 10000.0),
+			ropeGlobalBase: c.Float("rope.global.freq_base", 1000000.0),
+			ropeScale:      c.Float("rope.freq_scale", 1.0),
+		},
+	}
+
+	if numBlocks == gemma27BLayerCount {
+		m.largeModelScaling = true
+	}
+
+	return &m
+}
+
+type TextSelfAttention struct {
+	Query     *nn.Linear  `gguf:"attn_q"`
+	QueryNorm *nn.RMSNorm `gguf:"attn_q_norm"`
+	Key       *nn.Linear  `gguf:"attn_k"`
+	KeyNorm   *nn.RMSNorm `gguf:"attn_k_norm"`
+	Value     *nn.Linear  `gguf:"attn_v"`
+	Output    *nn.Linear  `gguf:"attn_output"`
+}
+
+func (sa *TextSelfAttention) Forward(ctx ml.Context, layer int, hiddenState, positionIDs ml.Tensor, cache kvcache.Cache, opts *TextOptions) ml.Tensor {
+	batchSize := hiddenState.Dim(1)
+	ropeType := uint32(2)
+
+	ropeBase := opts.ropeLocalBase
+	if (layer+1)%gemmaGlobalCacheCount == 0 {
+		ropeBase = opts.ropeGlobalBase
+	}
+
+	q := sa.Query.Forward(ctx, hiddenState)
+	q = q.Reshape(ctx, opts.attnKeyLen, opts.numHeads, batchSize)
+	q = sa.QueryNorm.Forward(ctx, q, opts.eps)
+	q = q.RoPE(ctx, positionIDs, nil, uint32(opts.attnKeyLen), ropeType, ropeBase, opts.ropeScale)
+
+	if opts.largeModelScaling {
+		q = q.Scale(ctx, 1.0/math.Sqrt(float64(opts.hiddenSize/opts.numHeads)))
+	} else {
+		q = q.Scale(ctx, 1.0/math.Sqrt(float64(opts.attnKeyLen)))
+	}
+
+	k := sa.Key.Forward(ctx, hiddenState)
+	k = k.Reshape(ctx, opts.attnKeyLen, opts.numKVHeads, batchSize)
+	k = sa.KeyNorm.Forward(ctx, k, opts.eps)
+	k = k.RoPE(ctx, positionIDs, nil, uint32(opts.attnKeyLen), ropeType, ropeBase, opts.ropeScale)
+
+	v := sa.Value.Forward(ctx, hiddenState)
+	v = v.Reshape(ctx, opts.attnValLen, opts.numKVHeads, batchSize)
+
+	scaleFactor := 1.0
+	kqv := nn.Attention(ctx, q, k, v, scaleFactor, cache)
+	kqv = kqv.Reshape(ctx, opts.attnValLen*opts.numHeads, batchSize)
+
+	return sa.Output.Forward(ctx, kqv)
+}
+
+func (m *TextModel) Shift(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, error) {
+	ropeBase := m.TextOptions.ropeLocalBase
+	if (layer+1)%gemmaGlobalCacheCount == 0 {
+		ropeBase = m.TextOptions.ropeGlobalBase
+	}
+
+	return key.RoPE(ctx, shift, nil, uint32(m.TextOptions.attnKeyLen), uint32(2), ropeBase, m.TextOptions.ropeScale), nil
+}
+
+type TextMLP struct {
+	Up   *nn.Linear `gguf:"ffn_up"`
+	Down *nn.Linear `gguf:"ffn_down"`
+	Gate *nn.Linear `gguf:"ffn_gate"`
+}
+
+func (mlp *TextMLP) Forward(ctx ml.Context, hiddenState ml.Tensor, opts *TextOptions) ml.Tensor {
+	hiddenState = mlp.Gate.Forward(ctx, hiddenState).GELU(ctx).Mul(ctx, mlp.Up.Forward(ctx, hiddenState))
+	return mlp.Down.Forward(ctx, hiddenState)
+}
+
+type TextLayer struct {
+	AttentionNorm     *nn.RMSNorm `gguf:"attn_norm"`
+	SelfAttention     *TextSelfAttention
+	PostAttentionNorm *nn.RMSNorm `gguf:"post_attention_norm"`
+	MLPNorm           *nn.RMSNorm `gguf:"ffn_norm"`
+	MLP               *TextMLP
+	PostMLPNorm       *nn.RMSNorm `gguf:"post_ffw_norm"`
+}
+
+func (l *TextLayer) Forward(ctx ml.Context, layer int, hiddenState, positionIDs, outputs ml.Tensor, cache kvcache.Cache, opts *TextOptions) ml.Tensor {
+	residual := hiddenState
+
+	hiddenState = l.AttentionNorm.Forward(ctx, hiddenState, opts.eps)
+	hiddenState = l.SelfAttention.Forward(ctx, layer, hiddenState, positionIDs, cache, opts)
+	hiddenState = l.PostAttentionNorm.Forward(ctx, hiddenState, opts.eps)
+
+	// In the final layer (outputs != nil), optimize by pruning to just the token positions
+	// we need logits for.
+	if outputs != nil {
+		hiddenState = hiddenState.Rows(ctx, outputs)
+		residual = residual.Rows(ctx, outputs)
+	}
+
+	hiddenState = hiddenState.Add(ctx, residual)
+	residual = hiddenState
+
+	hiddenState = l.MLPNorm.Forward(ctx, hiddenState, opts.eps)
+	hiddenState = l.MLP.Forward(ctx, hiddenState, opts)
+	hiddenState = l.PostMLPNorm.Forward(ctx, hiddenState, opts.eps)
+	return hiddenState.Add(ctx, residual)
+}
+
+func (m *TextModel) Forward(ctx ml.Context, inputs, positions, outputs ml.Tensor, opts input.Options, cache kvcache.Cache) ml.Tensor {
+	hiddenState := m.TokenEmbedding.Forward(ctx, inputs)
+	hiddenState = hiddenState.Scale(ctx, math.Sqrt(float64(m.TextOptions.hiddenSize)))
+
+	// set image embeddings
+	var except []int
+	for _, image := range opts.Multimodal {
+		visionOutputs := image.Multimodal.(ml.Tensor)
+		ctx.Forward(visionOutputs.Copy(ctx, hiddenState.View(ctx, image.Index*hiddenState.Stride(1), visionOutputs.Dim(0)*visionOutputs.Dim(1))))
+
+		for i := range visionOutputs.Dim(1) {
+			except = append(except, image.Index+i)
+		}
+	}
+
+	for i, layer := range m.Layers {
+		// gemma alternates between the sliding window (local) and causal (global)
+		// kv cache every 6 layers
+		cacheType := cacheTypeSWA
+		if (i+1)%gemmaGlobalCacheCount == 0 {
+			cacheType = cacheTypeCausal
+		}
+		cache.SetLayer(i)
+		wc := cache.(*kvcache.WrapperCache)
+		wc.SetLayerType(cacheType)
+
+		if causal, ok := wc.UnderlyingCache().(*kvcache.Causal); ok {
+			causal.SetCausal(ctx, kvcache.CausalOptions{Except: except})
+		}
+
+		var lastLayerOutputs ml.Tensor
+		if i == len(m.Layers)-1 {
+			lastLayerOutputs = outputs
+		}
+
+		hiddenState = layer.Forward(ctx, i, hiddenState, positions, lastLayerOutputs, cache, m.TextOptions)
+	}
+
+	hiddenState = m.OutputNorm.Forward(ctx, hiddenState, m.eps)
+	return m.Output.Forward(ctx, hiddenState)
+}
--- a/model/models/gemma3/model_vision.go
+++ b/model/models/gemma3/model_vision.go
@@ -0,0 +1,127 @@
+package gemma3
+
+import (
+	"math"
+
+	"github.com/ollama/ollama/ml"
+	"github.com/ollama/ollama/ml/nn"
+)
+
+var batchSize int = 1
+
+type VisionSelfAttention struct {
+	Query  *nn.Linear `gguf:"attn_q"`
+	Key    *nn.Linear `gguf:"attn_k"`
+	Value  *nn.Linear `gguf:"attn_v"`
+	Output *nn.Linear `gguf:"attn_output"`
+}
+
+func (sa *VisionSelfAttention) Forward(ctx ml.Context, hiddenState ml.Tensor, opts *VisionModelOptions) ml.Tensor {
+	headDim := opts.hiddenSize / opts.numHeads
+
+	query := sa.Query.Forward(ctx, hiddenState)
+	key := sa.Key.Forward(ctx, hiddenState)
+	value := sa.Value.Forward(ctx, hiddenState)
+
+	query = query.Reshape(ctx, headDim, opts.numHeads, query.Dim(1), batchSize)
+	key = key.Reshape(ctx, headDim, opts.numHeads, key.Dim(1), batchSize)
+	value = value.Reshape(ctx, headDim, opts.numHeads, value.Dim(1), batchSize)
+
+	attention := nn.Attention(ctx, query, key, value, 1.0/math.Sqrt(float64(headDim)), nil)
+	attention = attention.Reshape(ctx, opts.hiddenSize, attention.Dim(2), batchSize)
+
+	hiddenState = sa.Output.Forward(ctx, attention)
+	return hiddenState
+}
+
+type VisionMLP struct {
+	FC1 *nn.Linear `gguf:"fc1"`
+	FC2 *nn.Linear `gguf:"fc2"`
+}
+
+func (mlp *VisionMLP) Forward(ctx ml.Context, hiddenState ml.Tensor, opts *VisionModelOptions) ml.Tensor {
+	hiddenState = mlp.FC1.Forward(ctx, hiddenState).GELU(ctx)
+	hiddenState = mlp.FC2.Forward(ctx, hiddenState)
+	return hiddenState
+}
+
+type VisionEncoderLayer struct {
+	LayerNorm1    *nn.LayerNorm `gguf:"layer_norm1"`
+	SelfAttention *VisionSelfAttention
+
+	LayerNorm2 *nn.LayerNorm `gguf:"layer_norm2"`
+	MLP        *VisionMLP    `gguf:"mlp"`
+}
+
+func (e *VisionEncoderLayer) Forward(ctx ml.Context, hiddenState ml.Tensor, opts *VisionModelOptions) ml.Tensor {
+	residual := hiddenState
+
+	// self attention
+	hiddenState = e.LayerNorm1.Forward(ctx, hiddenState, opts.eps)
+	hiddenState = e.SelfAttention.Forward(ctx, hiddenState, opts)
+	hiddenState = hiddenState.Add(ctx, residual)
+	residual = hiddenState
+
+	// feed forward
+	hiddenState = e.LayerNorm2.Forward(ctx, hiddenState, opts.eps)
+	hiddenState = e.MLP.Forward(ctx, hiddenState, opts)
+	return hiddenState.Add(ctx, residual)
+}
+
+type VisionModelOptions struct {
+	hiddenSize, numHeads int
+	imageSize, patchSize int
+	eps                  float32
+}
+
+type VisionModel struct {
+	PatchEmbedding    *nn.Conv2D    `gguf:"patch_embedding"`
+	PositionEmbedding *nn.Embedding `gguf:"position_embedding"`
+	PostLayerNorm     *nn.LayerNorm `gguf:"post_layernorm"`
+
+	Layers []VisionEncoderLayer `gguf:"blk"`
+
+	*VisionModelOptions
+}
+
+func (m *VisionModel) Forward(ctx ml.Context, pixelValues ml.Tensor) ml.Tensor {
+	numPatches := (m.imageSize / m.patchSize) * (m.imageSize / m.patchSize)
+
+	hiddenState := m.PatchEmbedding.Forward(ctx, pixelValues, m.patchSize, m.patchSize, 0, 0, 1, 1)
+	hiddenState = hiddenState.Reshape(ctx, numPatches, m.hiddenSize)
+	hiddenState = hiddenState.Permute(ctx, 1, 0, 2, 3).Contiguous(ctx)
+
+	positions := make([]int32, numPatches)
+	for i := range positions {
+		positions[i] = int32(i)
+	}
+
+	positionIDs, err := ctx.Input().FromIntSlice(positions, len(positions))
+	if err != nil {
+		panic(err)
+	}
+
+	hiddenState = hiddenState.Add(ctx, m.PositionEmbedding.Forward(ctx, positionIDs))
+
+	for _, layer := range m.Layers {
+		hiddenState = layer.Forward(ctx, hiddenState, m.VisionModelOptions)
+	}
+
+	hiddenState = m.PostLayerNorm.Forward(ctx, hiddenState, m.eps)
+	return hiddenState
+}
+
+func newVisionModel(c ml.Config) *VisionModel {
+	return &VisionModel{
+		Layers: make([]VisionEncoderLayer, c.Uint("vision.block_count")),
+		VisionModelOptions: &VisionModelOptions{
+			hiddenSize: int(c.Uint("vision.embedding_length")),
+			numHeads:   int(c.Uint("vision.attention.head_count")),
+
+			imageSize: int(c.Uint("vision.image_size")),
+			patchSize: int(c.Uint("vision.patch_size")),
+
+			eps: c.Float("vision.attention.layer_norm_epsilon"),
+		},
+	}
+}
--- a/model/models/gemma3/process_image.go
+++ b/model/models/gemma3/process_image.go
@@ -0,0 +1,58 @@
+package gemma3
+
+import (
+	"image"
+
+	"github.com/ollama/ollama/ml"
+	"github.com/ollama/ollama/model/imageproc"
+)
+
+type ImageProcessor struct {
+	imageSize, patchSize, numChannels int
+}
+
+func newImageProcessor(c ml.Config) ImageProcessor {
+	return ImageProcessor{
+		imageSize:   int(c.Uint("vision.image_size")),
+		patchSize:   int(c.Uint("vision.patch_size")),
+		numChannels: int(c.Uint("vision.num_channels")),
+	}
+}
+
+func (p *ImageProcessor) pack(img image.Image, mean, std [3]float32) []float32 {
+	var pixelVals, rVals, gVals, bVals []float32
+
+	bounds := img.Bounds()
+	for y := bounds.Min.Y; y < bounds.Max.Y; y++ {
+		for x := bounds.Min.X; x < bounds.Max.X; x++ {
+			c := img.At(x, y)
+			r, g, b, _ := c.RGBA()
+			rVal := float32(r>>8) / 255.0
+			gVal := float32(g>>8) / 255.0
+			bVal := float32(b>>8) / 255.0
+
+			rVal = (rVal - mean[0]) / std[0]
+			gVal = (gVal - mean[1]) / std[1]
+			bVal = (bVal - mean[2]) / std[2]
+
+			rVals = append(rVals, rVal)
+			gVals = append(gVals, gVal)
+			bVals = append(bVals, bVal)
+		}
+	}
+
+	pixelVals = append(pixelVals, rVals...)
+	pixelVals = append(pixelVals, gVals...)
+	pixelVals = append(pixelVals, bVals...)
+
+	return pixelVals
+}
+
+func (p ImageProcessor) ProcessImage(img image.Image) ([]float32, error) {
+	outputSize := image.Point{p.imageSize, p.imageSize}
+	newImage := imageproc.Composite(img)
+	newImage = imageproc.Resize(newImage, outputSize, imageproc.ResizeBilinear)
+
+	data := p.pack(newImage, imageproc.ImageNetStandardMean, imageproc.ImageNetStandardSTD)
+	return data, nil
+}
--- a/model/models/llama/model.go
+++ b/model/models/llama/model.go
@@ -76,14 +76,15 @@ type SelfAttention struct {
 func (sa *SelfAttention) Forward(ctx ml.Context, hiddenState, positionIDs ml.Tensor, cache kvcache.Cache, opts *Options) ml.Tensor {
 	batchSize := hiddenState.Dim(1)
 	headDim := opts.hiddenSize / opts.numHeads
+	ropeType := uint32(0)

 	q := sa.Query.Forward(ctx, hiddenState)
 	q = q.Reshape(ctx, headDim, opts.numHeads, batchSize)
-	q = q.RoPE(ctx, positionIDs, sa.RopeFactors, opts.ropeDim, opts.ropeBase, opts.ropeScale)
+	q = q.RoPE(ctx, positionIDs, sa.RopeFactors, opts.ropeDim, ropeType, opts.ropeBase, opts.ropeScale)

 	k := sa.Key.Forward(ctx, hiddenState)
 	k = k.Reshape(ctx, headDim, opts.numKVHeads, batchSize)
-	k = k.RoPE(ctx, positionIDs, sa.RopeFactors, opts.ropeDim, opts.ropeBase, opts.ropeScale)
+	k = k.RoPE(ctx, positionIDs, sa.RopeFactors, opts.ropeDim, ropeType, opts.ropeBase, opts.ropeScale)

 	v := sa.Value.Forward(ctx, hiddenState)
 	v = v.Reshape(ctx, headDim, opts.numKVHeads, batchSize)
@@ -96,7 +97,7 @@ func (sa *SelfAttention) Forward(ctx ml.Context, hiddenState, positionIDs ml.Ten
 }

 func (m *Model) Shift(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, error) {
-	return key.RoPE(ctx, shift, m.Layers[layer].SelfAttention.RopeFactors, m.ropeDim, m.ropeBase, m.ropeScale), nil
+	return key.RoPE(ctx, shift, m.Layers[layer].SelfAttention.RopeFactors, uint32(0), m.ropeDim, m.ropeBase, m.ropeScale), nil
 }

 type MLP struct {
--- a/model/models/mllama/model.go
+++ b/model/models/mllama/model.go
@@ -63,6 +63,10 @@ func New(c ml.Config) (model.Model, error) {
 }

 func (m *Model) EncodeMultimodal(ctx ml.Context, multimodalData []byte) (any, error) {
+	if len(m.VisionModel.Transformer.Layers) == 0 || len(m.GlobalTransformer.Layers) == 0 {
+		return nil, model.ErrNoVisionModel
+	}
+
 	image, _, err := image.Decode(bytes.NewReader(multimodalData))
 	if err != nil {
 		return nil, err
@@ -102,17 +106,17 @@ func (m *Model) EncodeMultimodal(ctx ml.Context, multimodalData []byte) (any, er
 	return m.Projector.Forward(ctx, crossAttentionStates), nil
 }

-func (m *Model) PostTokenize(ctx ml.Context, inputs []input.Input) ([]input.Input, error) {
+func (m *Model) PostTokenize(inputs []input.Input) ([]input.Input, error) {
 	var images []input.Input
 	fnvHash := fnv.New64a()

 	for i := range inputs {
 		if inputs[i].Multimodal == nil {
 			if len(images) > 0 {
-				inputs[i].Multimodal = images[0].Multimodal
+				inputs[i].Multimodal = []ml.Tensor{images[0].Multimodal.(ml.Tensor)}
 				inputs[i].MultimodalHash = images[0].MultimodalHash
 				for j := 1; j < len(images); j++ {
-					inputs[i].Multimodal = inputs[i].Multimodal.(ml.Tensor).Concat(ctx, images[j].Multimodal.(ml.Tensor), 3)
+					inputs[i].Multimodal = append(inputs[i].Multimodal.([]ml.Tensor), images[0].Multimodal.(ml.Tensor))
 					fnvHash.Reset()
 					binary.Write(fnvHash, binary.NativeEndian, inputs[i].MultimodalHash)
 					binary.Write(fnvHash, binary.NativeEndian, inputs[j].MultimodalHash)
@@ -134,7 +138,10 @@ func (m *Model) PostTokenize(ctx ml.Context, inputs []input.Input) ([]input.Inpu
 func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
 	var crossAttentionStates ml.Tensor
 	if len(opts.Multimodal) > 0 {
-		crossAttentionStates = opts.Multimodal[len(opts.Multimodal)-1].Multimodal.(ml.Tensor)
+		images := opts.Multimodal[len(opts.Multimodal)-1].Multimodal.([]ml.Tensor)
+		if len(images) > 0 {
+			crossAttentionStates = images[len(images)-1]
+		}
 	}

 	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
--- a/model/models/mllama/model_text.go
+++ b/model/models/mllama/model_text.go
@@ -20,14 +20,15 @@ type TextSelfAttention struct {
 func (sa *TextSelfAttention) Forward(ctx ml.Context, hiddenState, positions, _ ml.Tensor, cache *kvcache.WrapperCache, opts *TextModelOptions) ml.Tensor {
 	batchSize := hiddenState.Dim(1)
 	headDim := opts.hiddenSize / opts.numHeads
+	ropeType := uint32(0)

 	query := sa.Query.Forward(ctx, hiddenState)
 	query = query.Reshape(ctx, headDim, opts.numHeads, batchSize)
-	query = query.RoPE(ctx, positions, sa.RopeFactors, opts.ropeDim, opts.ropeBase, opts.ropeScale)
+	query = query.RoPE(ctx, positions, sa.RopeFactors, opts.ropeDim, ropeType, opts.ropeBase, opts.ropeScale)

 	key := sa.Key.Forward(ctx, hiddenState)
 	key = key.Reshape(ctx, headDim, opts.numKVHeads, batchSize)
-	key = key.RoPE(ctx, positions, sa.RopeFactors, opts.ropeDim, opts.ropeBase, opts.ropeScale)
+	key = key.RoPE(ctx, positions, sa.RopeFactors, opts.ropeDim, ropeType, opts.ropeBase, opts.ropeScale)

 	value := sa.Value.Forward(ctx, hiddenState)
 	value = value.Reshape(ctx, headDim, opts.numKVHeads, batchSize)
@@ -40,8 +41,9 @@ func (sa *TextSelfAttention) Forward(ctx ml.Context, hiddenState, positions, _ m
 }

 func (m *TextModel) Shift(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, error) {
+	// This will only get called for layers in the cache, which are just the self attention layers
 	if sa, ok := m.Transformer.Layers[layer].(*TextSelfAttentionDecoderLayer); ok {
-		return key.RoPE(ctx, shift, sa.SelfAttention.RopeFactors, m.ropeDim, m.ropeBase, m.ropeScale), nil
+		return key.RoPE(ctx, shift, sa.SelfAttention.RopeFactors, m.ropeDim, uint32(0), m.ropeBase, m.ropeScale), nil
 	}

 	return key, nil
--- a/model/models/mllama/process_image.go
+++ b/model/models/mllama/process_image.go
@@ -144,8 +144,6 @@ func (p *ImageProcessor) splitToTiles(img image.Image, numTilesSize image.Point)
 	return images
 }

-// remove the "alpha" channel by drawing over a prefilled image
-//
 // remove the "alpha" channel by drawing over a prefilled image
 //
 //nolint:unused
--- a/model/models/models.go
+++ b/model/models/models.go
@@ -1,6 +1,8 @@
 package models

 import (
+	_ "github.com/ollama/ollama/model/models/gemma2"
+	_ "github.com/ollama/ollama/model/models/gemma3"
 	_ "github.com/ollama/ollama/model/models/llama"
 	_ "github.com/ollama/ollama/model/models/mllama"
 )
--- a/model/process_text.go
+++ b/model/process_text.go
@@ -4,6 +4,7 @@ import (
 	"cmp"
 	"iter"
 	"log/slog"
+	"slices"
 	"strings"
 	"sync"

@@ -18,6 +19,15 @@ const (
 	SpecialEOS
 )

+const (
+	TOKEN_TYPE_NORMAL = iota + 1
+	TOKEN_TYPE_UNKNOWN
+	TOKEN_TYPE_CONTROL
+	TOKEN_TYPE_USER_DEFINED
+	TOKEN_TYPE_UNUSED
+	TOKEN_TYPE_BYTE
+)
+
 type TextProcessor interface {
 	Encode(s string, addSpecial bool) ([]int32, error)
 	Decode([]int32) (string, error)
@@ -27,11 +37,11 @@ type TextProcessor interface {
 type Vocabulary struct {
 	Values []string
 	Types  []uint32
-	Scores []uint32
+	Scores []float32
 	Merges []string

-	BOS, EOS       int32
-	AddBOS, AddEOS bool
+	BOS, EOS, EOT          int32
+	AddBOS, AddEOS, AddEOT bool

 	specialOnce sync.Once
 	special     []string
@@ -48,7 +58,7 @@ func (v *Vocabulary) Is(id int32, special Special) bool {
 	case SpecialBOS:
 		return id == v.BOS
 	case SpecialEOS:
-		return id == v.EOS
+		return id == v.EOS || id == v.EOT
 	default:
 		return false
 	}
@@ -76,7 +86,9 @@ func (v *Vocabulary) Decode(id int32) string {
 func (v *Vocabulary) SpecialVocabulary() []string {
 	v.specialOnce.Do(func() {
 		for i := range v.Values {
-			if v.Types[i] == 3 {
+			if slices.Contains([]int{105, 106}, i) {
+				v.special = append(v.special, v.Values[i])
+			} else if v.Types[i] == TOKEN_TYPE_CONTROL {
 				v.special = append(v.special, v.Values[i])
 			}
 		}
--- a/model/process_text_spm.go
+++ b/model/process_text_spm.go
@@ -0,0 +1,246 @@
+package model
+
+import (
+	"iter"
+	"log/slog"
+	"strings"
+
+	"github.com/dlclark/regexp2"
+	queue "github.com/emirpasic/gods/v2/queues/priorityqueue"
+)
+
+const spmWhitespaceSep = "▁"
+
+func replaceWhitespaceBySeperator(s string) string {
+	return strings.ReplaceAll(s, " ", spmWhitespaceSep)
+}
+
+type SentencePieceModel struct {
+	maxTokenLen int
+	pre         *regexp2.Regexp
+	vocab       *Vocabulary
+}
+
+var _ TextProcessor = (*SentencePieceModel)(nil)
+
+func NewSentencePieceModel(pre string, vocab *Vocabulary) SentencePieceModel {
+	slog.Debug("Tokens", "num tokens", len(vocab.Values), "vals", vocab.Values[:5], "scores", vocab.Scores[:5], "types", vocab.Types[:5])
+
+	counter := map[int]int{}
+	var maxTokenLen int
+	for cnt := range vocab.Types {
+		switch vocab.Types[cnt] {
+		case TOKEN_TYPE_NORMAL, TOKEN_TYPE_USER_DEFINED, TOKEN_TYPE_UNUSED:
+			maxTokenLen = max(maxTokenLen, len(vocab.Values[cnt]))
+			fallthrough
+		default:
+			counter[int(vocab.Types[cnt])] += 1
+		}
+	}
+
+	slog.Debug("Token counts", "normal", counter[TOKEN_TYPE_NORMAL], "unknown", counter[TOKEN_TYPE_UNKNOWN], "control", counter[TOKEN_TYPE_CONTROL],
+		"user defined", counter[TOKEN_TYPE_USER_DEFINED], "unused", counter[TOKEN_TYPE_UNUSED], "byte", counter[TOKEN_TYPE_BYTE],
+		"max token len", maxTokenLen)
+
+	return SentencePieceModel{
+		maxTokenLen: maxTokenLen,
+		pre:         regexp2.MustCompile(pre, regexp2.Unicode|regexp2.RE2),
+		vocab:       vocab,
+	}
+}
+
+func (spm SentencePieceModel) Is(id int32, special Special) bool {
+	return spm.vocab.Is(id, special)
+}
+
+func (spm *SentencePieceModel) split(s string) iter.Seq[string] {
+	return func(yield func(string) bool) {
+		for m, _ := spm.pre.FindStringMatch(s); m != nil; m, _ = spm.pre.FindNextMatch(m) {
+			if !yield(m.String()) {
+				break
+			}
+		}
+	}
+}
+
+func (spm SentencePieceModel) Encode(s string, addSpecial bool) ([]int32, error) {
+	fragments := []fragment{{value: s}}
+	for _, special := range spm.vocab.SpecialVocabulary() {
+		// TODO: process special tokens concurrently
+		id := spm.vocab.Encode(special)
+		for i := 0; i < len(fragments); i++ {
+			frag := fragments[i]
+			if len(frag.ids) > 0 {
+				continue
+			}
+
+			var middle []fragment
+			switch i := strings.Index(frag.value, special); {
+			case i < 0:
+				middle = append(middle, frag)
+			case i > 0:
+				middle = append(middle, fragment{value: frag.value[:i]})
+				fallthrough
+			default:
+				middle = append(middle, fragment{value: special, ids: []int32{id}})
+				if rest := frag.value[i+len(special):]; rest != "" {
+					middle = append(middle, fragment{value: rest})
+				}
+			}
+
+			fragments = append(fragments[:i], append(middle, fragments[i+1:]...)...)
+		}
+	}
+	slog.Debug("fragments", "frags", fragments)
+
+	var ids []int32
+	for _, frag := range fragments {
+		if len(frag.ids) > 0 {
+			ids = append(ids, frag.ids...)
+			continue
+		}
+
+		for split := range spm.split(frag.value) {
+			split = replaceWhitespaceBySeperator(split)
+
+			var sb strings.Builder
+			sb.Write([]byte(split))
+			if id := spm.vocab.Encode(sb.String()); id >= 0 {
+				ids = append(ids, id)
+				continue
+			}
+
+			runes := []rune(sb.String())
+			pq := queue.NewWith(func(a, b any) int {
+				priA := a.(*candidate)
+				priB := b.(*candidate)
+				if priA.score > priB.score || (priA.score == priB.score && priA.a < priB.a) {
+					return -1
+				}
+				return 1
+			})
+
+			merges := make([]merge, len(runes))
+			for r := range runes {
+				merges[r] = merge{
+					p:     r - 1,
+					n:     r + 1,
+					runes: []rune{runes[r]},
+				}
+			}
+
+			slog.Debug("tokenizer", "merges", merges)
+
+			pairwise := func(a, b int) *candidate {
+				if a < 0 || b >= len(runes) {
+					return nil
+				}
+
+				left, right := string(merges[a].runes), string(merges[b].runes)
+				if id := spm.vocab.Encode(left + right); id >= 0 {
+					return &candidate{
+						a:     a,
+						b:     b,
+						score: spm.vocab.Scores[id],
+					}
+				}
+				return nil
+			}
+
+			for i := range len(runes) - 1 {
+				if pair := pairwise(i, i+1); pair != nil {
+					pq.Enqueue(pair)
+				}
+			}
+
+			pqv := pq.Values()
+			for _, v := range pqv {
+				e := v.(*candidate)
+				slog.Debug("candidate", "candidate", e)
+			}
+
+			for !pq.Empty() {
+				v, _ := pq.Dequeue()
+				pair := v.(*candidate)
+				left, right := merges[pair.a], merges[pair.b]
+
+				slog.Debug("pair", "left", left, "right", right)
+				if len(left.runes) == 0 || len(right.runes) == 0 {
+					continue
+				}
+
+				if id := spm.vocab.Encode(string(left.runes) + string(right.runes)); id < 0 {
+					continue
+				}
+
+				merges[pair.a].runes = append(left.runes, right.runes...)
+				merges[pair.b].runes = nil
+				merges[pair.a].n = right.n
+				if right.n < len(merges) {
+					merges[right.n].p = pair.a
+				}
+
+				if pair := pairwise(merges[pair.a].p, pair.a); pair != nil {
+					pq.Enqueue(pair)
+				}
+
+				if pair := pairwise(pair.a, merges[pair.a].n); pair != nil {
+					pq.Enqueue(pair)
+				}
+			}
+
+			slog.Debug("merges", "merges", merges)
+
+			for _, merge := range merges {
+				if len(merge.runes) > 0 {
+					if id := spm.vocab.Encode(string(merge.runes)); id >= 0 {
+						ids = append(ids, id)
+					} else {
+						slog.Debug("missing token", "token", string(merge.runes))
+					}
+				}
+			}
+		}
+	}
+
+	if addSpecial && len(ids) > 0 {
+		if spm.vocab.AddBOS {
+			if ids[0] == spm.vocab.BOS {
+				slog.Warn("adding bos token to prompt which already has it", "id", spm.vocab.BOS)
+			}
+
+			slog.Debug("adding bos token to prompt", "id", spm.vocab.BOS)
+			ids = append([]int32{spm.vocab.BOS}, ids...)
+		}
+
+		if spm.vocab.AddEOS {
+			if ids[len(ids)-1] == spm.vocab.EOS {
+				slog.Warn("adding eos token to prompt which already has it", "id", spm.vocab.EOS)
+			}
+
+			slog.Debug("adding eos token to prompt", "id", spm.vocab.EOS)
+			ids = append(ids, spm.vocab.EOS)
+		}
+	}
+
+	return ids, nil
+}
+
+type candidate struct {
+	a, b  int
+	score float32
+}
+
+func (spm SentencePieceModel) Decode(ids []int32) (string, error) {
+	var sb strings.Builder
+	for _, id := range ids {
+		data := spm.vocab.Decode(id)
+		data = strings.ReplaceAll(data, spmWhitespaceSep, " ")
+		if _, err := sb.WriteString(data); err != nil {
+			return "", err
+		}
+	}
+
+	slog.Debug("decoded", "ids", ids, "text", sb.String())
+	return sb.String(), nil
+}
--- a/model/process_text_spm_test.go
+++ b/model/process_text_spm_test.go
@@ -0,0 +1,118 @@
+package model
+
+import (
+	"log/slog"
+	"os"
+	"path/filepath"
+	"slices"
+	"testing"
+
+	"google.golang.org/protobuf/proto"
+
+	"github.com/ollama/ollama/convert/sentencepiece"
+)
+
+func loadSentencePieceVocab(t *testing.T) SentencePieceModel {
+	t.Helper()
+
+	bts, err := os.ReadFile(filepath.Join("testdata", "gemma2", "tokenizer.model"))
+	if err != nil {
+		t.Fatal(err)
+	}
+
+	var spm sentencepiece.ModelProto
+	if err := proto.Unmarshal(bts, &spm); err != nil {
+		t.Fatal(err)
+	}
+
+	preTokenizer := `(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+`
+
+	var v Vocabulary
+
+	for _, piece := range spm.GetPieces() {
+		v.Values = append(v.Values, piece.GetPiece())
+		v.Scores = append(v.Scores, piece.GetScore())
+		switch t := piece.GetType(); t {
+		case sentencepiece.ModelProto_SentencePiece_UNKNOWN,
+			sentencepiece.ModelProto_SentencePiece_CONTROL,
+			sentencepiece.ModelProto_SentencePiece_UNUSED,
+			sentencepiece.ModelProto_SentencePiece_BYTE:
+			v.Types = append(v.Types, uint32(t))
+		default:
+			tt := uint32(sentencepiece.ModelProto_SentencePiece_NORMAL)
+			// todo parse the special tokens file
+			//   - this will roundtrip correctly but the <start_of_turn> and
+			//     <end_of_turn> tokens aren't processed
+			v.Types = append(v.Types, tt)
+		}
+	}
+
+	return NewSentencePieceModel(preTokenizer, &v)
+}
+
+func TestSentencePieceEncode(t *testing.T) {
+	logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelDebug}))
+	slog.SetDefault(logger)
+
+	tokenizer := loadSentencePieceVocab(t)
+
+	t.Run("basic roundtrip", func(t *testing.T) {
+		t.Parallel()
+
+		cases := []string{
+			"hello",
+			"hello ",
+			"hello  ",
+			" hello",
+			" hello ",
+			" hello  ",
+			"hello world",
+			"请考试我的软件！12345",
+			"你好",
+			"Hello 你好 world!",
+			"Special characters: !@#$%^&*()_+-=[]{}|;':\",./<>?",
+			"Multilingual: 你好 こんにちは Привет Hola مرحبا",
+			"Numbers and symbols: 123456789 +- */",
+			"Special tokens: <bos> text <eos>",
+			"Code snippets: func main() { fmt.Println(\"Hello World\") }",
+			"Long text: " + "Lorem ipsum dolor sit amet, consectetur adipiscing elit. " +
+				"Sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. " +
+				"Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris.",
+		}
+
+		for _, want := range cases {
+			ids, err := tokenizer.Encode(want, true)
+			if err != nil {
+				t.Fatal(err)
+			}
+
+			if got, err := tokenizer.Decode(ids); err != nil {
+				t.Fatal(err)
+			} else if got != want {
+				t.Errorf("got %q, want %q [%#v]", got, want, ids)
+			}
+		}
+	})
+
+	t.Run("special tokens", func(t *testing.T) {
+		type candidate struct {
+			token string
+			ids   []int32
+		}
+
+		cases := []candidate{
+			{"<bos>", []int32{2}},
+			{"<eos>", []int32{1}},
+		}
+
+		for _, want := range cases {
+			ids, err := tokenizer.Encode(want.token, true)
+			if err != nil {
+				t.Fatal(err)
+			}
+			if !slices.Equal(ids, want.ids) {
+				t.Errorf("got %#v, want %#v", ids, want.ids)
+			}
+		}
+	})
+}
--- a/model/testdata/gemma2/tokenizer.model
+++ b/model/testdata/gemma2/tokenizer.model
--- a/readline/readline.go
+++ b/readline/readline.go
@@ -116,19 +116,9 @@ func (i *Instance) Readline() (string, error) {

 			switch r {
 			case KeyUp:
-				if i.History.Pos > 0 {
-					if i.History.Pos == i.History.Size() {
-						currentLineBuf = []rune(buf.String())
-					}
-					buf.Replace([]rune(i.History.Prev()))
-				}
+				i.historyPrev(buf, &currentLineBuf)
 			case KeyDown:
-				if i.History.Pos < i.History.Size() {
-					buf.Replace([]rune(i.History.Next()))
-					if i.History.Pos == i.History.Size() {
-						buf.Replace(currentLineBuf)
-					}
-				}
+				i.historyNext(buf, &currentLineBuf)
 			case KeyLeft:
 				buf.MoveLeft()
 			case KeyRight:
@@ -185,6 +175,10 @@ func (i *Instance) Readline() (string, error) {
 			esc = true
 		case CharInterrupt:
 			return "", ErrInterrupt
+		case CharPrev:
+			i.historyPrev(buf, &currentLineBuf)
+		case CharNext:
+			i.historyNext(buf, &currentLineBuf)
 		case CharLineStart:
 			buf.MoveToStart()
 		case CharLineEnd:
@@ -246,6 +240,24 @@ func (i *Instance) HistoryDisable() {
 	i.History.Enabled = false
 }

+func (i *Instance) historyPrev(buf *Buffer, currentLineBuf *[]rune) {
+	if i.History.Pos > 0 {
+		if i.History.Pos == i.History.Size() {
+			*currentLineBuf = []rune(buf.String())
+		}
+		buf.Replace([]rune(i.History.Prev()))
+	}
+}
+
+func (i *Instance) historyNext(buf *Buffer, currentLineBuf *[]rune) {
+	if i.History.Pos < i.History.Size() {
+		buf.Replace([]rune(i.History.Next()))
+		if i.History.Pos == i.History.Size() {
+			buf.Replace(*currentLineBuf)
+		}
+	}
+}
+
 func NewTerminal() (*Terminal, error) {
 	fd := os.Stdin.Fd()
 	termios, err := SetRawMode(fd)
--- a/runner/llamarunner/runner.go
+++ b/runner/llamarunner/runner.go
@@ -24,6 +24,7 @@ import (

 	"github.com/ollama/ollama/api"
 	"github.com/ollama/ollama/llama"
+	"github.com/ollama/ollama/llm"
 	"github.com/ollama/ollama/runner/common"
 )

@@ -99,7 +100,7 @@ type NewSequenceParams struct {
 	embedding      bool
 }

-func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequenceParams) (*Sequence, error) {
+func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSequenceParams) (*Sequence, error) {
 	s.ready.Wait()

 	startTime := time.Now()
@@ -163,7 +164,7 @@ func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequen
 // inputs processes the prompt and images into a list of inputs
 // by splitting the prompt on [img-<n>] tags, tokenizing text and
 // generating image embeddings for each image
-func (s *Server) inputs(prompt string, images []ImageData) ([]input, error) {
+func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input, error) {
 	var inputs []input
 	var parts []string
 	var matches [][]string
@@ -229,7 +230,7 @@ type Server struct {
 	image *ImageContext

 	// status for external health reporting - loading, ready to serve, etc.
-	status ServerStatus
+	status llm.ServerStatus

 	// current progress on loading the model
 	progress float32
@@ -541,75 +542,18 @@ func (s *Server) processBatch(tokenBatch *llama.Batch, embedBatch *llama.Batch)
 	return nil
 }

-// TODO (jmorganca): use structs from the api package to avoid duplication
-// this way the api acts as a proxy instead of using a different api for the
-// runner
-type Options struct {
-	api.Runner
-
-	NumKeep          int      `json:"n_keep"`
-	Seed             int      `json:"seed"`
-	NumPredict       int      `json:"n_predict"`
-	TopK             int      `json:"top_k"`
-	TopP             float32  `json:"top_p"`
-	MinP             float32  `json:"min_p"`
-	TypicalP         float32  `json:"typical_p"`
-	RepeatLastN      int      `json:"repeat_last_n"`
-	Temperature      float32  `json:"temperature"`
-	RepeatPenalty    float32  `json:"repeat_penalty"`
-	PresencePenalty  float32  `json:"presence_penalty"`
-	FrequencyPenalty float32  `json:"frequency_penalty"`
-	Mirostat         int      `json:"mirostat"`
-	MirostatTau      float32  `json:"mirostat_tau"`
-	MirostatEta      float32  `json:"mirostat_eta"`
-	Stop             []string `json:"stop"`
-}
-
-type ImageData struct {
-	Data          []byte `json:"data"`
-	ID            int    `json:"id"`
-	AspectRatioID int    `json:"aspect_ratio_id"`
-}
-
-type CompletionRequest struct {
-	Prompt      string      `json:"prompt"`
-	Images      []ImageData `json:"image_data"`
-	Grammar     string      `json:"grammar"`
-	CachePrompt bool        `json:"cache_prompt"`
-
-	Options
-}
-
-type Timings struct {
-	PredictedN  int     `json:"predicted_n"`
-	PredictedMS float64 `json:"predicted_ms"`
-	PromptN     int     `json:"prompt_n"`
-	PromptMS    float64 `json:"prompt_ms"`
-}
-
-type CompletionResponse struct {
-	Content string `json:"content"`
-	Stop    bool   `json:"stop"`
-
-	Model        string  `json:"model,omitempty"`
-	Prompt       string  `json:"prompt,omitempty"`
-	StoppedLimit bool    `json:"stopped_limit,omitempty"`
-	PredictedN   int     `json:"predicted_n,omitempty"`
-	PredictedMS  float64 `json:"predicted_ms,omitempty"`
-	PromptN      int     `json:"prompt_n,omitempty"`
-	PromptMS     float64 `json:"prompt_ms,omitempty"`
-
-	Timings Timings `json:"timings"`
-}
-
 func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
-	var req CompletionRequest
-	req.Options = Options(api.DefaultOptions())
+	var req llm.CompletionRequest
 	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
 		http.Error(w, "Bad request", http.StatusBadRequest)
 		return
 	}

+	if req.Options == nil {
+		opts := api.DefaultOptions()
+		req.Options = &opts
+	}
+
 	// Set the headers to indicate streaming
 	w.Header().Set("Content-Type", "application/json")
 	w.Header().Set("Transfer-Encoding", "chunked")
@@ -620,26 +564,28 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 		return
 	}

-	var samplingParams llama.SamplingParams
-	samplingParams.TopK = req.TopK
-	samplingParams.TopP = req.TopP
-	samplingParams.MinP = req.MinP
-	samplingParams.TypicalP = req.TypicalP
-	samplingParams.Temp = req.Temperature
-	samplingParams.RepeatLastN = req.RepeatLastN
-	samplingParams.PenaltyRepeat = req.RepeatPenalty
-	samplingParams.PenaltyFreq = req.FrequencyPenalty
-	samplingParams.PenaltyPresent = req.PresencePenalty
-	samplingParams.Mirostat = req.Mirostat
-	samplingParams.MirostatTau = req.MirostatTau
-	samplingParams.MirostatEta = req.MirostatEta
-	samplingParams.Seed = uint32(req.Seed)
-	samplingParams.Grammar = req.Grammar
+	// Extract options from the CompletionRequest
+	samplingParams := llama.SamplingParams{
+		TopK:           req.Options.TopK,
+		TopP:           req.Options.TopP,
+		MinP:           req.Options.MinP,
+		TypicalP:       req.Options.TypicalP,
+		Temp:           req.Options.Temperature,
+		RepeatLastN:    req.Options.RepeatLastN,
+		PenaltyRepeat:  req.Options.RepeatPenalty,
+		PenaltyFreq:    req.Options.FrequencyPenalty,
+		PenaltyPresent: req.Options.PresencePenalty,
+		Mirostat:       req.Options.Mirostat,
+		MirostatTau:    req.Options.MirostatTau,
+		MirostatEta:    req.Options.MirostatEta,
+		Seed:           uint32(req.Options.Seed),
+		Grammar:        req.Grammar,
+	}

 	seq, err := s.NewSequence(req.Prompt, req.Images, NewSequenceParams{
-		numPredict:     req.NumPredict,
-		stop:           req.Stop,
-		numKeep:        req.NumKeep,
+		numPredict:     req.Options.NumPredict,
+		stop:           req.Options.Stop,
+		numKeep:        req.Options.NumKeep,
 		samplingParams: &samplingParams,
 		embedding:      false,
 	})
@@ -662,7 +608,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	found := false
 	for i, sq := range s.seqs {
 		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
+			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, true)
 			if err != nil {
 				s.mu.Unlock()
 				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
@@ -691,7 +637,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 			return
 		case content, ok := <-seq.responses:
 			if ok {
-				if err := json.NewEncoder(w).Encode(&CompletionResponse{
+				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
 					Content: content,
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -702,15 +648,17 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 				flusher.Flush()
 			} else {
 				// Send the final response
-				if err := json.NewEncoder(w).Encode(&CompletionResponse{
-					Stop:         true,
-					StoppedLimit: seq.doneReason == "limit",
-					Timings: Timings{
-						PromptN:     seq.numPromptInputs,
-						PromptMS:    float64(seq.startGenerationTime.Sub(seq.startProcessingTime).Milliseconds()),
-						PredictedN:  seq.numDecoded,
-						PredictedMS: float64(time.Since(seq.startGenerationTime).Milliseconds()),
-					},
+				doneReason := "stop"
+				if seq.doneReason == "limit" {
+					doneReason = "length"
+				}
+				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
+					Done:               true,
+					DoneReason:         doneReason,
+					PromptEvalCount:    seq.numPromptInputs,
+					PromptEvalDuration: seq.startGenerationTime.Sub(seq.startProcessingTime),
+					EvalCount:          seq.numDecoded,
+					EvalDuration:       time.Since(seq.startGenerationTime),
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode final response: %v", err), http.StatusInternalServerError)
 				}
@@ -721,17 +669,8 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	}
 }

-type EmbeddingRequest struct {
-	Content     string `json:"content"`
-	CachePrompt bool   `json:"cache_prompt"`
-}
-
-type EmbeddingResponse struct {
-	Embedding []float32 `json:"embedding"`
-}
-
 func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {
-	var req EmbeddingRequest
+	var req llm.EmbeddingRequest
 	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
 		http.Error(w, fmt.Sprintf("bad request: %s", err), http.StatusBadRequest)
 		return
@@ -761,7 +700,7 @@ func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {
 	found := false
 	for i, sq := range s.seqs {
 		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
+			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, false)
 			if err != nil {
 				s.mu.Unlock()
 				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
@@ -782,41 +721,17 @@ func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {

 	embedding := <-seq.embedding

-	if err := json.NewEncoder(w).Encode(&EmbeddingResponse{
+	if err := json.NewEncoder(w).Encode(&llm.EmbeddingResponse{
 		Embedding: embedding,
 	}); err != nil {
 		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
 	}
 }

-type HealthResponse struct {
-	Status   string  `json:"status"`
-	Progress float32 `json:"progress"`
-}
-
-type ServerStatus int
-
-const (
-	ServerStatusReady ServerStatus = iota
-	ServerStatusLoadingModel
-	ServerStatusError
-)
-
-func (s ServerStatus) ToString() string {
-	switch s {
-	case ServerStatusReady:
-		return "ok"
-	case ServerStatusLoadingModel:
-		return "loading model"
-	default:
-		return "server error"
-	}
-}
-
 func (s *Server) health(w http.ResponseWriter, r *http.Request) {
 	w.Header().Set("Content-Type", "application/json")
-	if err := json.NewEncoder(w).Encode(&HealthResponse{
-		Status:   s.status.ToString(),
+	if err := json.NewEncoder(w).Encode(&llm.ServerStatusResponse{
+		Status:   s.status,
 		Progress: s.progress,
 	}); err != nil {
 		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -879,7 +794,7 @@ func (s *Server) loadModel(
 		panic(err)
 	}

-	s.status = ServerStatusReady
+	s.status = llm.ServerStatusReady
 	s.ready.Done()
 }

@@ -937,7 +852,7 @@ func Execute(args []string) error {
 		parallel:  *parallel,
 		seqs:      make([]*Sequence, *parallel),
 		seqsSem:   semaphore.NewWeighted(int64(*parallel)),
-		status:    ServerStatusLoadingModel,
+		status:    llm.ServerStatusLoadingModel,
 	}

 	var tensorSplitFloats []float32
--- a/runner/ollamarunner/cache.go
+++ b/runner/ollamarunner/cache.go
@@ -89,7 +89,7 @@ type InputCacheSlot struct {
 	lastUsed time.Time
 }

-func (c *InputCache) LoadCacheSlot(prompt []input.Input, cachePrompt bool) (*InputCacheSlot, []input.Input, error) {
+func (c *InputCache) LoadCacheSlot(prompt []input.Input) (*InputCacheSlot, []input.Input, error) {
 	var slot *InputCacheSlot
 	var numPast int32
 	var err error
@@ -107,10 +107,6 @@ func (c *InputCache) LoadCacheSlot(prompt []input.Input, cachePrompt bool) (*Inp
 		return nil, nil, err
 	}

-	if !cachePrompt {
-		numPast = 0
-	}
-
 	slot.InUse = true
 	slot.lastUsed = time.Now()

--- a/runner/ollamarunner/cache_test.go
+++ b/runner/ollamarunner/cache_test.go
@@ -297,3 +297,131 @@ func TestShiftDiscard(t *testing.T) {
 		})
 	}
 }
+
+func TestLoadCacheSlot(t *testing.T) {
+	tests := []struct {
+		name           string
+		cache          InputCache
+		prompt         []input.Input
+		wantErr        bool
+		expectedSlotId int
+		expectedPrompt int // expected length of remaining prompt
+	}{
+		{
+			name: "Basic cache hit - single user",
+			cache: InputCache{
+				multiUserCache: false,
+				slots: []InputCacheSlot{
+					{
+						Id:       0,
+						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
+						InUse:    false,
+						lastUsed: time.Now().Add(-time.Second),
+					},
+					{
+						Id:       1,
+						Inputs:   []input.Input{},
+						InUse:    false,
+						lastUsed: time.Now().Add(-2 * time.Second),
+					},
+				},
+			},
+			prompt:         []input.Input{{Token: 1}, {Token: 2}, {Token: 3}},
+			wantErr:        false,
+			expectedSlotId: 0,
+			expectedPrompt: 1, // Only token 3 remains
+		},
+		{
+			name: "Basic cache hit - multi user",
+			cache: InputCache{
+				multiUserCache: true,
+				slots: []InputCacheSlot{
+					{
+						Id:       0,
+						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
+						InUse:    false,
+						lastUsed: time.Now().Add(-time.Second),
+					},
+					{
+						Id:       1,
+						Inputs:   []input.Input{},
+						InUse:    false,
+						lastUsed: time.Now().Add(-2 * time.Second),
+					},
+				},
+			},
+			prompt:         []input.Input{{Token: 1}, {Token: 2}, {Token: 3}},
+			wantErr:        false,
+			expectedSlotId: 0,
+			expectedPrompt: 1, // Only token 3 remains
+		},
+		{
+			name: "Exact match - leave one input",
+			cache: InputCache{
+				multiUserCache: false,
+				slots: []InputCacheSlot{
+					{
+						Id:       0,
+						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
+						InUse:    false,
+						lastUsed: time.Now().Add(-time.Second),
+					},
+				},
+			},
+			prompt:         []input.Input{{Token: 1}, {Token: 2}},
+			wantErr:        false,
+			expectedSlotId: 0,
+			expectedPrompt: 1, // Should leave 1 token for sampling
+		},
+		{
+			name: "No available slots",
+			cache: InputCache{
+				multiUserCache: false,
+				slots: []InputCacheSlot{
+					{
+						Id:       0,
+						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
+						InUse:    true,
+						lastUsed: time.Now().Add(-time.Second),
+					},
+				},
+			},
+			prompt:         []input.Input{{Token: 1}, {Token: 2}, {Token: 3}},
+			wantErr:        true,
+			expectedSlotId: -1,
+			expectedPrompt: -1,
+		},
+	}
+
+	for _, tt := range tests {
+		t.Run(tt.name, func(t *testing.T) {
+			slot, remainingPrompt, err := tt.cache.LoadCacheSlot(tt.prompt)
+
+			// Check error state
+			if (err != nil) != tt.wantErr {
+				t.Errorf("LoadCacheSlot() error = %v, wantErr %v", err, tt.wantErr)
+				return
+			}
+
+			if tt.wantErr {
+				return // Skip further checks if we expected an error
+			}
+
+			// Verify slot ID
+			if slot.Id != tt.expectedSlotId {
+				t.Errorf("LoadCacheSlot() slot ID = %v, expected %v", slot.Id, tt.expectedSlotId)
+			}
+
+			// Verify slot is now marked in use
+			if !slot.InUse {
+				t.Errorf("LoadCacheSlot() slot not marked InUse")
+			}
+
+			// Verify remaining prompt length
+			if len(remainingPrompt) != tt.expectedPrompt {
+				t.Errorf("LoadCacheSlot() remaining prompt length = %v, expected %v",
+					len(remainingPrompt), tt.expectedPrompt)
+			}
+		})
+	}
+}
--- a/runner/ollamarunner/runner.go
+++ b/runner/ollamarunner/runner.go
@@ -24,6 +24,7 @@ import (
 	"golang.org/x/sync/semaphore"

 	"github.com/ollama/ollama/api"
+	"github.com/ollama/ollama/llm"
 	"github.com/ollama/ollama/ml"
 	"github.com/ollama/ollama/model"
 	"github.com/ollama/ollama/model/input"
@@ -33,10 +34,14 @@ import (
 	_ "github.com/ollama/ollama/model/models"
 )

+type contextList struct {
+	list []ml.Context
+}
+
 type Sequence struct {
-	// ctx for allocating tensors that last the lifetime of the sequence, such as
+	// ctxs are used for allocating tensors that last the lifetime of the sequence, such as
 	// multimodal embeddings
-	ctx ml.Context
+	ctxs *contextList

 	// batch index
 	iBatch int
@@ -94,13 +99,12 @@ type NewSequenceParams struct {
 	embedding  bool
 }

-func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequenceParams) (*Sequence, error) {
+func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSequenceParams) (*Sequence, error) {
 	s.ready.Wait()

 	startTime := time.Now()
-	ctx := s.model.Backend().NewContext()

-	inputs, err := s.inputs(ctx, prompt, images)
+	inputs, ctxs, err := s.inputs(prompt, images)
 	if err != nil {
 		return nil, fmt.Errorf("failed to process inputs: %w", err)
 	} else if len(inputs) == 0 {
@@ -111,6 +115,9 @@ func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequen
 		params.numKeep = int32(len(inputs))
 	}

+	// TODO(jessegross): We should ensure that we always leave minBatch of context space to shift,
+	// otherwise we might truncate or split the batch against the model's wishes
+
 	// Ensure that at least 1 input can be discarded during shift
 	params.numKeep = min(params.numKeep, s.cache.numCtx-1)

@@ -126,7 +133,7 @@ func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequen
 	// TODO(jessegross): Ingest cached history for grammar

 	return &Sequence{
-		ctx:                 ctx,
+		ctxs:                ctxs,
 		inputs:              inputs,
 		numPromptInputs:     len(inputs),
 		startProcessingTime: startTime,
@@ -145,7 +152,7 @@ func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequen
 // inputs processes the prompt and images into a list of inputs
 // by splitting the prompt on [img-<n>] tags, tokenizing text and
 // decoding images
-func (s *Server) inputs(ctx ml.Context, prompt string, images []ImageData) ([]input.Input, error) {
+func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input.Input, *contextList, error) {
 	var inputs []input.Input
 	var parts []string
 	var matches [][]string
@@ -160,12 +167,19 @@ func (s *Server) inputs(ctx ml.Context, prompt string, images []ImageData) ([]in
 		parts = []string{prompt}
 	}

+	var contexts contextList
+	runtime.AddCleanup(&contexts, func(ctxs []ml.Context) {
+		for _, ctx := range ctxs {
+			ctx.Close()
+		}
+	}, contexts.list)
+
 	postTokenize := false
 	for i, part := range parts {
 		// text - tokenize
 		tokens, err := s.model.(model.TextProcessor).Encode(part, i == 0)
 		if err != nil {
-			return nil, err
+			return nil, nil, err
 		}

 		for _, t := range tokens {
@@ -185,12 +199,14 @@ func (s *Server) inputs(ctx ml.Context, prompt string, images []ImageData) ([]in
 			}

 			if imageIndex < 0 {
-				return nil, fmt.Errorf("invalid image index: %d", n)
+				return nil, nil, fmt.Errorf("invalid image index: %d", n)
 			}

+			ctx := s.model.Backend().NewContext()
+			contexts.list = append(contexts.list, ctx)
 			imageEmbeddings, err := multimodalProcessor.EncodeMultimodal(ctx, images[imageIndex].Data)
 			if err != nil {
-				return nil, err
+				return nil, nil, err
 			}

 			s.multimodalHash.Reset()
@@ -204,13 +220,13 @@ func (s *Server) inputs(ctx ml.Context, prompt string, images []ImageData) ([]in

 	if visionModel && postTokenize {
 		var err error
-		inputs, err = multimodalProcessor.PostTokenize(ctx, inputs)
+		inputs, err = multimodalProcessor.PostTokenize(inputs)
 		if err != nil {
-			return nil, err
+			return nil, nil, err
 		}
 	}

-	return inputs, nil
+	return inputs, &contexts, nil
 }

 type Server struct {
@@ -222,7 +238,7 @@ type Server struct {
 	model model.Model

 	// status for external health reporting - loading, ready to serve, etc.
-	status ServerStatus
+	status llm.ServerStatus

 	// current progress on loading the model
 	progress float32
@@ -305,7 +321,6 @@ func (s *Server) removeSequence(seqIndex int, reason string) {
 	close(seq.responses)
 	close(seq.embedding)
 	seq.cache.InUse = false
-	seq.ctx.Close()
 	s.seqs[seqIndex] = nil
 	s.seqsSem.Release(1)
 }
@@ -351,22 +366,35 @@ func (s *Server) processBatch() error {
 			seq.cache.Inputs = []input.Input{}
 		}

+		batchSize := s.batchSize
+
 		for j, inp := range seq.inputs {
-			if int32(len(seq.cache.Inputs)+len(seq.pendingInputs)+1) > s.cache.numCtx {
-				if len(seq.pendingInputs) == 0 {
-					err := s.cache.ShiftCacheSlot(seq.cache, seq.numKeep)
-					if err != nil {
-						return err
-					}
-				} else {
-					break
-				}
+			// If we are required to put following inputs into a single batch then extend the
+			// batch size. Since we are only extending the size the minimum amount possible, this
+			// will cause a break if we have pending inputs.
+			minBatch := 1 + inp.SameBatch
+			if minBatch > batchSize {
+				batchSize = minBatch
 			}

-			if j >= s.batchSize {
+			if len(seq.pendingInputs)+minBatch > batchSize {
 				break
 			}

+			// If the sum of our working set (already processed tokens, tokens we added to this
+			// batch, required following tokens) exceeds the context size, then trigger a shift
+			// now so we don't have to do one later when we can't break the batch.
+			if int32(len(seq.cache.Inputs)+len(seq.pendingInputs)+minBatch) > s.cache.numCtx {
+				if len(seq.pendingInputs) != 0 {
+					break
+				}
+
+				err := s.cache.ShiftCacheSlot(seq.cache, seq.numKeep)
+				if err != nil {
+					return err
+				}
+			}
+
 			options.Inputs = append(options.Inputs, inp.Token)
 			if inp.Multimodal != nil {
 				options.Multimodal = append(options.Multimodal, input.MultimodalIndex{Index: len(options.Inputs) - 1, Multimodal: inp.Multimodal})
@@ -501,75 +529,18 @@ func (s *Server) processBatch() error {
 	return nil
 }

-// TODO (jmorganca): use structs from the api package to avoid duplication
-// this way the api acts as a proxy instead of using a different api for the
-// runner
-type Options struct {
-	api.Runner
-
-	NumKeep          int      `json:"n_keep"`
-	Seed             int      `json:"seed"`
-	NumPredict       int      `json:"n_predict"`
-	TopK             int      `json:"top_k"`
-	TopP             float32  `json:"top_p"`
-	MinP             float32  `json:"min_p"`
-	TypicalP         float32  `json:"typical_p"`
-	RepeatLastN      int      `json:"repeat_last_n"`
-	Temperature      float32  `json:"temperature"`
-	RepeatPenalty    float32  `json:"repeat_penalty"`
-	PresencePenalty  float32  `json:"presence_penalty"`
-	FrequencyPenalty float32  `json:"frequency_penalty"`
-	Mirostat         int      `json:"mirostat"`
-	MirostatTau      float32  `json:"mirostat_tau"`
-	MirostatEta      float32  `json:"mirostat_eta"`
-	Stop             []string `json:"stop"`
-}
-
-type ImageData struct {
-	Data          []byte `json:"data"`
-	ID            int    `json:"id"`
-	AspectRatioID int    `json:"aspect_ratio_id"`
-}
-
-type CompletionRequest struct {
-	Prompt      string      `json:"prompt"`
-	Images      []ImageData `json:"image_data"`
-	Grammar     string      `json:"grammar"`
-	CachePrompt bool        `json:"cache_prompt"`
-
-	Options
-}
-
-type Timings struct {
-	PredictedN  int     `json:"predicted_n"`
-	PredictedMS float64 `json:"predicted_ms"`
-	PromptN     int     `json:"prompt_n"`
-	PromptMS    float64 `json:"prompt_ms"`
-}
-
-type CompletionResponse struct {
-	Content string `json:"content"`
-	Stop    bool   `json:"stop"`
-
-	Model        string  `json:"model,omitempty"`
-	Prompt       string  `json:"prompt,omitempty"`
-	StoppedLimit bool    `json:"stopped_limit,omitempty"`
-	PredictedN   int     `json:"predicted_n,omitempty"`
-	PredictedMS  float64 `json:"predicted_ms,omitempty"`
-	PromptN      int     `json:"prompt_n,omitempty"`
-	PromptMS     float64 `json:"prompt_ms,omitempty"`
-
-	Timings Timings `json:"timings"`
-}
-
 func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
-	var req CompletionRequest
-	req.Options = Options(api.DefaultOptions())
+	var req llm.CompletionRequest
 	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
 		http.Error(w, "Bad request", http.StatusBadRequest)
 		return
 	}

+	if req.Options == nil {
+		opts := api.DefaultOptions()
+		req.Options = &opts
+	}
+
 	// Set the headers to indicate streaming
 	w.Header().Set("Content-Type", "application/json")
 	w.Header().Set("Transfer-Encoding", "chunked")
@@ -591,18 +562,18 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	}

 	sampler := sample.NewSampler(
-		req.Temperature,
-		req.TopK,
-		req.TopP,
-		req.MinP,
-		req.Seed,
+		req.Options.Temperature,
+		req.Options.TopK,
+		req.Options.TopP,
+		req.Options.MinP,
+		req.Options.Seed,
 		grammar,
 	)

 	seq, err := s.NewSequence(req.Prompt, req.Images, NewSequenceParams{
-		numPredict: req.NumPredict,
-		stop:       req.Stop,
-		numKeep:    int32(req.NumKeep),
+		numPredict: req.Options.NumPredict,
+		stop:       req.Options.Stop,
+		numKeep:    int32(req.Options.NumKeep),
 		sampler:    sampler,
 		embedding:  false,
 	})
@@ -625,7 +596,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	found := false
 	for i, sq := range s.seqs {
 		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
+			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs)
 			if err != nil {
 				s.mu.Unlock()
 				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
@@ -652,7 +623,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 			return
 		case content, ok := <-seq.responses:
 			if ok {
-				if err := json.NewEncoder(w).Encode(&CompletionResponse{
+				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
 					Content: content,
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -663,15 +634,17 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 				flusher.Flush()
 			} else {
 				// Send the final response
-				if err := json.NewEncoder(w).Encode(&CompletionResponse{
-					Stop:         true,
-					StoppedLimit: seq.doneReason == "limit",
-					Timings: Timings{
-						PromptN:     seq.numPromptInputs,
-						PromptMS:    float64(seq.startGenerationTime.Sub(seq.startProcessingTime).Milliseconds()),
-						PredictedN:  seq.numPredicted,
-						PredictedMS: float64(time.Since(seq.startGenerationTime).Milliseconds()),
-					},
+				doneReason := "stop"
+				if seq.doneReason == "limit" {
+					doneReason = "length"
+				}
+				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
+					Done:               true,
+					DoneReason:         doneReason,
+					PromptEvalCount:    seq.numPromptInputs,
+					PromptEvalDuration: seq.startGenerationTime.Sub(seq.startProcessingTime),
+					EvalCount:          seq.numPredicted,
+					EvalDuration:       time.Since(seq.startGenerationTime),
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode final response: %v", err), http.StatusInternalServerError)
 				}
@@ -682,102 +655,10 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	}
 }

-type EmbeddingRequest struct {
-	Content     string `json:"content"`
-	CachePrompt bool   `json:"cache_prompt"`
-}
-
-type EmbeddingResponse struct {
-	Embedding []float32 `json:"embedding"`
-}
-
-func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {
-	var req EmbeddingRequest
-	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
-		http.Error(w, fmt.Sprintf("bad request: %s", err), http.StatusBadRequest)
-		return
-	}
-
-	w.Header().Set("Content-Type", "application/json")
-
-	slog.Debug("embedding request", "content", req.Content)
-
-	seq, err := s.NewSequence(req.Content, nil, NewSequenceParams{embedding: true})
-	if err != nil {
-		http.Error(w, fmt.Sprintf("Failed to create new sequence: %v", err), http.StatusInternalServerError)
-		return
-	}
-
-	// Ensure there is a place to put the sequence, released when removed from s.seqs
-	if err := s.seqsSem.Acquire(r.Context(), 1); err != nil {
-		if errors.Is(err, context.Canceled) {
-			slog.Info("aborting embeddings request due to client closing the connection")
-		} else {
-			slog.Error("Failed to acquire semaphore", "error", err)
-		}
-		return
-	}
-
-	s.mu.Lock()
-	found := false
-	for i, sq := range s.seqs {
-		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
-			if err != nil {
-				s.mu.Unlock()
-				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
-				return
-			}
-			s.seqs[i] = seq
-			s.cond.Signal()
-			found = true
-			break
-		}
-	}
-	s.mu.Unlock()
-
-	if !found {
-		http.Error(w, "could not find an available sequence", http.StatusInternalServerError)
-		return
-	}
-
-	embedding := <-seq.embedding
-
-	if err := json.NewEncoder(w).Encode(&EmbeddingResponse{
-		Embedding: embedding,
-	}); err != nil {
-		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
-	}
-}
-
-type HealthResponse struct {
-	Status   string  `json:"status"`
-	Progress float32 `json:"progress"`
-}
-
-type ServerStatus int
-
-const (
-	ServerStatusReady ServerStatus = iota
-	ServerStatusLoadingModel
-	ServerStatusError
-)
-
-func (s ServerStatus) ToString() string {
-	switch s {
-	case ServerStatusReady:
-		return "ok"
-	case ServerStatusLoadingModel:
-		return "loading model"
-	default:
-		return "server error"
-	}
-}
-
 func (s *Server) health(w http.ResponseWriter, r *http.Request) {
 	w.Header().Set("Content-Type", "application/json")
-	if err := json.NewEncoder(w).Encode(&HealthResponse{
-		Status:   s.status.ToString(),
+	if err := json.NewEncoder(w).Encode(&llm.ServerStatusResponse{
+		Status:   s.status,
 		Progress: s.progress,
 	}); err != nil {
 		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -831,7 +712,7 @@ func (s *Server) loadModel(
 	s.seqs = make([]*Sequence, s.parallel)
 	s.seqsSem = semaphore.NewWeighted(int64(s.parallel))

-	s.status = ServerStatusReady
+	s.status = llm.ServerStatusReady
 	s.ready.Done()
 }

@@ -883,7 +764,7 @@ func Execute(args []string) error {

 	server := &Server{
 		batchSize: *batchSize,
-		status:    ServerStatusLoadingModel,
+		status:    llm.ServerStatusLoadingModel,
 	}

 	// TODO(jessegross): Parameters that need to be implemented:
@@ -927,9 +808,13 @@ func Execute(args []string) error {
 	defer listener.Close()

 	mux := http.NewServeMux()
-	mux.HandleFunc("/embedding", server.embeddings)
-	mux.HandleFunc("/completion", server.completion)
-	mux.HandleFunc("/health", server.health)
+	// TODO: support embeddings
+	mux.HandleFunc("POST /embedding", func(w http.ResponseWriter, r *http.Request) {
+		http.Error(w, "this model does not support embeddings", http.StatusNotImplemented)
+	})
+
+	mux.HandleFunc("POST /completion", server.completion)
+	mux.HandleFunc("GET /health", server.health)

 	httpServer := http.Server{
 		Handler: mux,
--- a/sample/samplers.go
+++ b/sample/samplers.go
@@ -84,14 +84,13 @@ func (s *Sampler) sample(tokens []token) (token, error) {
 		return greedy(tokens), nil
 	}

-	if s.topK > 0 {
-		tokens = topK(tokens, s.topK)
-	} else {
-		sortLogits(tokens)
-	}
+	// topK also sorts the tokens in descending order of logits
+	tokens = topK(tokens, s.topK)
+
+	// scale and normalize the tokens in place
+	temperature(tokens, s.temperature)
+	softmax(tokens)

-	tokens = temperature(tokens, s.temperature)
-	tokens = softmax(tokens)
 	tokens = topP(tokens, s.topP)
 	tokens = minP(tokens, s.minP)

--- a/sample/transforms.go
+++ b/sample/transforms.go
@@ -1,104 +1,102 @@
 package sample

 import (
+	"container/heap"
 	"math"
 	"slices"
 )

-func softmax(ts []token) []token {
+// tokenHeap implements heap.Interface and holds tokens as a min-heap to track k largest elements
+type tokenHeap []token
+
+func (h tokenHeap) Len() int           { return len(h) }
+func (h tokenHeap) Less(i, j int) bool { return h[i].value < h[j].value }
+func (h tokenHeap) Swap(i, j int)      { h[i], h[j] = h[j], h[i] }
+
+func (h *tokenHeap) Push(x any) {
+	*h = append(*h, x.(token))
+}
+
+func (h *tokenHeap) Pop() any {
+	old := *h
+	n := len(old)
+	x := old[n-1]
+	*h = old[0 : n-1]
+	return x
+}
+
+// temperature applies scaling to the logits
+func temperature(ts []token, temp float32) {
+	// Ensure temperature clipping near 0 to avoid numerical instability
+	temp = max(temp, 1e-7)
+	for i := range ts {
+		ts[i].value = ts[i].value / temp
+	}
+}
+
+// softmax applies normalization to the logits
+func softmax(ts []token) {
+	// Find max logit for numerical stability
+	maxLogit := float32(math.Inf(-1))
+	for _, t := range ts {
+		if t.value > maxLogit {
+			maxLogit = t.value
+		}
+	}
+
+	// Compute exp(x - max)
 	var sum float32
 	for i, v := range ts {
-		ts[i].value = float32(math.Exp(float64(v.value)))
+		ts[i].value = float32(math.Exp(float64(v.value - maxLogit)))
 		sum += ts[i].value
 	}

+	// exp(x - max) / sum(exp(x - max))
 	for i := range ts {
 		ts[i].value /= sum
 	}
-
-	return ts
-}
-
-func temperature(ti []token, t float32) []token {
-	if t == 1 {
-		return ti
-	}
-
-	temp := max(t, 1e-7)
-	maxLogit := float32(math.Inf(-1))
-	for _, token := range ti {
-		if token.value > maxLogit {
-			maxLogit = token.value
-		}
-	}
-
-	// subtracting max logit to avoid under/overflow
-	for i := range ti {
-		ti[i].value = (ti[i].value - maxLogit) / temp
-	}
-
-	return ti
-}
-
-// siftDown maintains a min-heap property by recursively moving larger elements down the heap.
-//
-// The heap is represented as an array where for any node at index i:
-// - Left child is at index 2i + 1
-// - Right child is at index 2i + 2
-// - Parent is at index (i-1)/2
-//
-// The function compares a node with its children and:
-// 1. Finds the smallest value between the node and its children
-// 2. If the node is not the smallest, swaps it with its smallest child
-// 3. Continues this process down the affected path until the min-heap property is restored
-func siftDown(data []token, start, end int) {
-	root := start
-	for {
-		child := 2*root + 1
-		if child >= end {
-			break
-		}
-		// Find smaller child (we want min heap)
-		if child+1 < end && data[child+1].value < data[child].value {
-			child++
-		}
-		// Exit if root is already smaller than children
-		if data[root].value <= data[child].value {
-			break
-		}
-		// Swap with smaller child and continue
-		data[root], data[child] = data[child], data[root]
-		root = child
-	}
 }

 // topK limits the number of tokens considered to the k highest logits
 func topK(ts []token, k int) []token {
-	if k >= len(ts) {
+	if k >= len(ts) || k <= 0 {
+		slices.SortFunc(ts, func(a, b token) int {
+			switch {
+			case a.value < b.value:
+				return 1
+			case a.value > b.value:
+				return -1
+			default:
+				return 0
+			}
+		})
 		return ts
 	}
-	// Heapify + siftDown - O(nlog(k))
-	// Build min-heap of first k elements
-	heap := ts[:k]
-	for i := k/2 - 1; i >= 0; i-- {
-		siftDown(heap, i, k)
-	}

-	// Process remaining elements - if larger than heap root, replace root
+	// Initialize min-heap with first k elements
+	h := make(tokenHeap, k)
+	copy(h, ts[:k])
+	heap.Init(&h)
+
+	// Process remaining elements
 	for i := k; i < len(ts); i++ {
-		if ts[i].value > heap[0].value {
-			heap[0] = ts[i]
-			siftDown(heap, 0, k)
+		if ts[i].value > h[0].value {
+			heap.Pop(&h)
+			heap.Push(&h, ts[i])
 		}
 	}

-	slices.Reverse(heap)
+	// Convert heap to sorted slice in descending order
+	result := make([]token, len(h))
+	for i := k - 1; i >= 0; i-- {
+		result[i] = heap.Pop(&h).(token)
+	}

-	ts = heap
-	return ts
+	return result
 }

 // topP limits tokens to those with cumulative probability p
+// requires ts to be sorted in descending order of probabilities
 func topP(ts []token, p float32) []token {
 	if p == 1.0 {
 		return ts
@@ -109,95 +107,24 @@ func topP(ts []token, p float32) []token {
 	for i, t := range ts {
 		sum += t.value
 		if sum > float32(p) {
-			ts = ts[:i+1]
-			return ts
+			return ts[:i+1]
 		}
 	}

 	return ts
 }

-// minP limits tokens to those with cumulative probability p
+// minP filters tokens with probabilities >= p * max_prob
+// requires ts to be sorted in descending order of probabilities
 func minP(ts []token, p float32) []token {
-	if p == 1.0 {
-		return ts
-	}
+	maxProb := ts[0].value

-	maxProb := float32(math.Inf(-1))
-	for _, token := range ts {
-		if token.value > maxProb {
-			maxProb = token.value
+	threshold := maxProb * p
+
+	for i, t := range ts {
+		if t.value < threshold {
+			return ts[:i]
 		}
 	}
-
-	threshold := maxProb * float32(p)
-
-	// Filter tokens in-place
-	validTokens := ts[:0]
-	for i, token := range ts {
-		if token.value >= threshold {
-			validTokens = append(validTokens, ts[i])
-		}
-	}
-
-	ts = validTokens
 	return ts
 }
-
-// TODO(parthsareen): possibly replace with simpler implementation https://github.com/ollama/ollama/issues/9584
-// Conting sort implementation to sort tokens by logits
-func sortLogits(tokens []token) {
-	if len(tokens) <= 1 {
-		return
-	}
-
-	// Find max/min in a single pass
-	minLogit, maxLogit := tokens[0].value, tokens[0].value
-	for _, t := range tokens[1:] {
-		if t.value < minLogit {
-			minLogit = t.value
-		} else if t.value > maxLogit {
-			maxLogit = t.value
-		}
-	}
-
-	// Calculate scaling to map to uint32 range
-	logitRange := maxLogit - minLogit
-	if logitRange < 1e-6 {
-		return // All values effectively equal
-	}
-
-	// Count frequencies directly from tokens
-	const maxInt = (1 << 24) - 1 // Use 24 bits for good granularity
-	var counts [256]int          // For first byte
-
-	// First pass: count frequencies
-	for _, t := range tokens {
-		// Map to [0, maxInt] range
-		score := min(uint32((t.value-minLogit)*float32(maxInt)/logitRange), maxInt)
-		counts[score>>16]++
-	}
-
-	// Calculate offsets
-	var offset int
-	for i := range counts {
-		count := counts[i]
-		counts[i] = offset
-		offset += count
-	}
-
-	// Second pass: place elements in correct position
-	output := make([]token, len(tokens))
-	// Track current positions
-	countsCopy := counts
-
-	for i, t := range tokens {
-		score := min(uint32((t.value-minLogit)*float32(maxInt)/logitRange), maxInt)
-
-		pos := countsCopy[score>>16]
-		countsCopy[score>>16]++
-		output[len(tokens)-1-pos] = tokens[i]
-	}
-
-	copy(tokens, output)
-}
--- a/sample/transforms_test.go
+++ b/sample/transforms_test.go
@@ -6,119 +6,239 @@ import (
 	"testing"
 )

-// Helper to convert float64 slice to logit slice
-func toTokens(values []float64) []token {
+// Helper to convert float32 slice to logit slice
+func toTokens(values []float32) []token {
 	tokens := make([]token, len(values))
 	for i, v := range values {
 		tokens[i] = token{
 			id:    int32(i),
-			value: float32(v),
+			value: v,
 		}
 	}
 	return tokens
 }

 // Helper to compare logit slices
-func compareLogits(t *testing.T, name string, want []float64, got []token) {
+func compareLogits(t *testing.T, name string, want []float32, got []token) {
 	t.Helper()
 	if len(want) != len(got) {
 		t.Errorf("%s: length mismatch: want %d, got %d", name, len(want), len(got))
 		return
 	}
 	for i := range want {
-		if math.Abs(float64(got[i].value)-want[i]) > 1e-6 {
+		if math.Abs(float64(got[i].value-want[i])) > 1e-6 {
 			t.Errorf("%s: index %d: want %f, got %f", name, i, want[i], got[i].value)
 		}
 	}
 }

 func TestTemperature(t *testing.T) {
-	input := []float64{2, -1, 4, -3, 1, -2, 0}
-	want := []float64{-4, -10, 0, -14, -6, -12, -8} // (logit - max logit) / temp
+	input := []float32{1.0, 4.0, -2.0, 0.0}
+	tokens := toTokens(input)
+	temperature(tokens, 0.5)
+	want := []float32{2.0, 8.0, -4.0, 0.0}
+	compareLogits(t, "temperature(0.5)", want, tokens)

-	got := temperature(toTokens(input), 0.5)
-	compareLogits(t, "Temperature", want, got)
+	input = []float32{1.0, 4.0, -2.0, 0.0}
+	tokens = toTokens(input)
+	temperature(tokens, 1.0)
+	want = []float32{1.0, 4.0, -2.0, 0.0}
+	compareLogits(t, "temperature(1)", want, tokens)
+
+	input = []float32{1.0, 4.0, -2.0, 0.0}
+	tokens = toTokens(input)
+	temperature(tokens, 0.0)
+	want = []float32{1e7, 4e7, -2e7, 0.0}
+	compareLogits(t, "temperature(0)", want, tokens)
 }

 func TestSoftmax(t *testing.T) {
-	input := []float64{-3, -2, -1, 0, 1, 2, 4}
-	got := softmax(toTokens(input))
-
-	// Check probabilities sum to 1
-	var sum float32
-	for _, token := range got {
-		sum += token.value
-	}
-	if math.Abs(float64(sum)-1.0) > 1e-6 {
-		t.Errorf("probabilities don't sum to 1: got %f", sum)
+	tests := []struct {
+		name     string
+		input    []float32
+		expected []float32
+	}{
+		{
+			name:     "correctness softmax",
+			input:    []float32{1, -2, 3, 0},
+			expected: []float32{0.113550, 0.005653, 0.839024, 0.041773},
+		},
+		{
+			name:  "normal distribution",
+			input: []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367},
+		},
+		{
+			name:  "single value",
+			input: []float32{1.0},
+		},
+		{
+			name:  "identical values",
+			input: []float32{0.9, 0.9, 0.9},
+		},
+		{
+			name:  "large values",
+			input: []float32{1000.0, 2000.0, 3000.0},
+		},
+		{
+			name:  "small values",
+			input: []float32{1e-6, 2e-6, 3e-6},
+		},
+		{
+			name:  "negative values",
+			input: []float32{-1.0, -2.0, -3.0},
+		},
+		{
+			name:  "mixed values",
+			input: []float32{-100.0, 0.0, 100.0},
+		},
 	}

-	// Check relative ordering is preserved
-	for i := 1; i < len(got); i++ {
-		if got[i].value < got[i-1].value {
-			t.Errorf("probability ordering not preserved at index %d", i)
-		}
+	for _, tt := range tests {
+		t.Run(tt.name, func(t *testing.T) {
+			tokens := toTokens(tt.input)
+			softmax(tokens)
+
+			if tt.expected != nil {
+				compareLogits(t, tt.name, tt.expected, tokens)
+				return
+			}
+
+			// Check probabilities sum to 1
+			var sum float32
+			for _, token := range tokens {
+				sum += token.value
+				if token.value < 0 || token.value > 1 {
+					t.Errorf("probability out of range [0,1]: got %f", token.value)
+				}
+			}
+			if math.Abs(float64(sum-1.0)) > 1e-6 {
+				t.Errorf("probabilities don't sum to 1: got %f", sum)
+			}
+		})
 	}
 }

 func TestTopK(t *testing.T) {
-	input := []float64{-3, -2, -1, 0, 1, 2, 4}
-
-	// Test k=3
-	got := topK(toTokens(input), 3)
-	if len(got) != 3 {
-		t.Errorf("topK(3): wrong length: want 3, got %d", len(got))
+	input := []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
+	tokens := toTokens(input)
+	tokens = topK(tokens, 5)
+	if len(tokens) != 5 {
+		t.Errorf("topK(5): wrong length: want 5, got %d", len(tokens))
 	}
-	// Should keep highest 3 values: 4, 2, 1
-	want := []float64{4, 2, 1}
-	compareLogits(t, "topK(3)", want, got)
+	want := []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154}
+	compareLogits(t, "topK(3)", want, tokens)

-	// Test k > len
-	got = topK(toTokens(input), 10)
-	compareLogits(t, "topK(10)", input, got)
+	tokens = toTokens(input)
+	tokens = topK(tokens, 20)
+	if len(tokens) != len(input) {
+		t.Errorf("topK(20): wrong length: want %d, got %d", len(input), len(tokens))
+	}
+
+	input = []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
+	want = []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154, 0.043722924, 0.036774673, 0.026986899, 0.01681367, 0.0046718004, 0.00412893, 0.0030491839}
+	tokens = toTokens(input)
+	tokens = topK(tokens, -1)
+	if len(tokens) != len(input) {
+		t.Errorf("topK(-1): wrong length: want %d, got %d", len(input), len(tokens))
+	}
+	compareLogits(t, "topK(-1)", want, tokens)
+
+	input = []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
+	want = []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154, 0.043722924, 0.036774673, 0.026986899, 0.01681367, 0.0046718004, 0.00412893, 0.0030491839}
+	tokens = toTokens(input)
+	tokens = topK(tokens, 0)
+	if len(tokens) != len(input) {
+		t.Errorf("topK(-1): wrong length: want %d, got %d", len(input), len(tokens))
+	}
+	compareLogits(t, "topK(-1)", want, tokens)
+
+	input = []float32{-1e7, -2e7, -3e7, -4e7}
+	tokens = toTokens(input)
+	tokens = topK(tokens, 1)
+	if len(tokens) < 1 {
+		t.Error("topK should keep at least one token")
+	}
 }

 func TestTopP(t *testing.T) {
-	input := []float64{-3, -2, -1, 0, 1, 2, 4}
+	input := []float32{-3, -2, -1, 0, 1, 2, 4}
 	tokens := toTokens(input)

 	// First apply temperature and softmax to get probabilities
-	tokens = temperature(tokens, 1)
-	tokens = softmax(tokens)
-	sortLogits(tokens)
+	softmax(tokens)
+	tokens = topK(tokens, 20)

 	// Then apply topP
-	got := topP(tokens, 0.95)
+	tokens = topP(tokens, 0.95)

 	// Should keep tokens until cumsum > 0.95
-	if len(got) > 3 {
-		t.Errorf("topP(0.95): kept too many tokens: got %d", len(got))
-		t.Logf("got: %v", got)
+	if len(tokens) > 3 {
+		t.Errorf("topP(0.95): kept too many tokens: got %d", len(tokens))
+		t.Logf("got: %v", tokens)
+	}
+
+	// Test edge case - ensure at least one token remains
+	input = []float32{-1e6, -1e6, -1e6} // One dominant token
+	tokens = toTokens(input)
+	softmax(tokens)
+	tokens = topP(tokens, 0.0) // Very small p
+	if len(tokens) < 1 {
+		t.Error("topP should keep at least one token")
 	}
 }

 func TestMinP(t *testing.T) {
-	input := []float64{-3, -2, -1, 0, 1, 2, 4, 3}
+	input := []float32{-3, -2, -1, 0, 1, 2, 4, 3}
 	tokens := toTokens(input)

 	// First apply temperature and softmax
-	tokens = temperature(tokens, 1)
-	tokens = softmax(tokens)
+	tokens = topK(tokens, 20)
+	softmax(tokens)

-	// Then apply minP
-	got := minP(tokens, 0.2)
+	tokens = minP(tokens, 1.0)
+
+	if len(tokens) != 1 {
+		t.Errorf("minP(1.0): should keep all tokens, got %d, want %d", len(tokens), len(tokens))
+	}
+
+	// Test with normal p value
+	tokens = toTokens(input) // Reset tokens
+	tokens = topK(tokens, 20)
+	softmax(tokens)
+	tokens = minP(tokens, 0.2)

 	// Should keep tokens with prob >= 0.2 * max_prob
-	if len(got) > 3 {
-		t.Errorf("minP(0.2): kept too many tokens: got %d", len(got))
+	if len(tokens) > 3 {
+		t.Errorf("minP(0.2): kept too many tokens: got %d", len(tokens))
+		t.Logf("got: %v", tokens)
+	}
+
+	// Test with zero p value
+	tokens = toTokens(input) // Reset tokens
+	tokens = topK(tokens, 20)
+	softmax(tokens)
+	tokens = minP(tokens, 0.0)
+
+	// Should keep only the highest probability token
+	if len(tokens) != len(input) {
+		t.Errorf("minP(0.0): should keep only one token, got %d", len(tokens))
+		t.Logf("got: %v", tokens)
+	}
+
+	input = []float32{1e-10, 1e-10, 1e-10}
+	tokens = toTokens(input)
+	softmax(tokens)
+	tokens = minP(tokens, 1.0)
+	if len(tokens) < 1 {
+		t.Error("minP should keep at least one token even with extreme probabilities")
 	}
 }

 func TestSortLogits(t *testing.T) {
-	input := []float64{3, 1, 4, 2, -1, 0, -2}
+	input := []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
 	tokens := toTokens(input)

-	sortLogits(tokens)
+	tokens = topK(tokens, 20)

 	for i := 1; i < len(tokens); i++ {
 		if tokens[i].value > tokens[i-1].value {
@@ -127,7 +247,7 @@ func TestSortLogits(t *testing.T) {
 		}
 	}

-	want := []float64{4, 3, 2, 1, 0, -1, -2}
+	want := []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154, 0.043722924, 0.036774673, 0.026986899, 0.01681367, 0.0046718004, 0.00412893, 0.0030491839}
 	compareLogits(t, "sortLogits", want, tokens)
 }

@@ -151,11 +271,19 @@ func BenchmarkTransforms(b *testing.B) {
 		}
 	})

+	b.Run("Softmax", func(b *testing.B) {
+		b.ResetTimer()
+		for b.Loop() {
+			copy(tokensCopy, tokens)
+			softmax(tokensCopy)
+		}
+	})
+
 	b.Run("TopK", func(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			topK(tokensCopy, 10)
+			tokens = topK(tokensCopy, 10)
 		}
 	})

@@ -163,7 +291,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			topP(tokensCopy, 0.9)
+			tokens = topP(tokensCopy, 0.9)
 		}
 	})

@@ -171,7 +299,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			minP(tokensCopy, 0.2)
+			tokens = minP(tokensCopy, 0.2)
 		}
 	})

@@ -179,7 +307,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			sortLogits(tokensCopy)
+			tokens = topK(tokensCopy, 200000)
 		}
 	})
 }
--- a/scripts/build_darwin.sh
+++ b/scripts/build_darwin.sh
@@ -8,7 +8,7 @@ usage() {
    exit 1
 }

-export VERSION=${VERSION:-$(git describe --tags --dirty)}
+export VERSION=${VERSION:-$(git describe --tags --first-parent --abbrev=7 --long --dirty --always | sed -e "s/^v//g")}
 export GOFLAGS="'-ldflags=-w -s \"-X=github.com/ollama/ollama/version.Version=${VERSION#v}\" \"-X=github.com/ollama/ollama/server.mode=release\"'"
 export CGO_CPPFLAGS='-mmacosx-version-min=11.3'

--- a/scripts/build_windows.ps1
+++ b/scripts/build_windows.ps1
@@ -80,13 +80,14 @@ function checkEnv() {


 function buildOllama() {
+    mkdir -Force -path "${script:DIST_DIR}\"
    if ($script:ARCH -ne "arm64") {
        Remove-Item -ea 0 -recurse -force -path "${script:SRC_DIR}\dist\windows-${script:ARCH}"
        New-Item "${script:SRC_DIR}\dist\windows-${script:ARCH}\lib\ollama\" -ItemType Directory -ea 0

        & cmake --fresh --preset CPU --install-prefix $script:DIST_DIR
        if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
-        & cmake --build --preset CPU --parallel $script:JOBS
+        & cmake --build --preset CPU  --config Release --parallel $script:JOBS
        if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
        & cmake --install build --component CPU --strip
        if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
@@ -101,7 +102,7 @@ function buildOllama() {
            # to avoid 2022 (or newer) from being used as the default
            & cmake --fresh --preset "CUDA 11" -G "Visual Studio 16 2019" --install-prefix $script:DIST_DIR
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
-            & cmake --build --preset "CUDA 11" --parallel $script:JOBS
+            & cmake --build --preset "CUDA 11"  --config Release --parallel $script:JOBS
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
            & cmake --install build --component "CUDA" --strip
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
@@ -112,7 +113,7 @@ function buildOllama() {
            write-host "Building CUDA v12 backend libraries"
            & cmake --fresh --preset "CUDA 12" --install-prefix $script:DIST_DIR
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
-            & cmake --build --preset "CUDA 12" --parallel $script:JOBS
+            & cmake --build --preset "CUDA 12"  --config Release --parallel $script:JOBS
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
            & cmake --install build --component "CUDA" --strip
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
@@ -131,7 +132,7 @@ function buildOllama() {
            $env:HIPCXX=""
            $env:HIP_PLATFORM=""
            $env:CMAKE_PREFIX_PATH=""
-            & cmake --build --preset "ROCm" --parallel $script:JOBS
+            & cmake --build --preset "ROCm"  --config Release --parallel $script:JOBS
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
            & cmake --install build --component "HIP" --strip
            if ($LASTEXITCODE -ne 0) { exit($LASTEXITCODE)}
--- a/server/internal/cache/blob/cache.go
+++ b/server/internal/cache/blob/cache.go
@@ -146,7 +146,7 @@ func debugger(err *error) func(step string) {
 // be in either of the following forms:
 //
 //	@<digest>
-//	<name>
+//	<name>@<digest>
 //	<name>
 //
 // If a digest is provided, it is returned as is and nothing else happens.
@@ -160,8 +160,6 @@ func debugger(err *error) func(step string) {
 // hashed is passed to a PutBytes call to ensure that the manifest is in the
 // blob store. This is done to ensure that future calls to [Get] succeed in
 // these cases.
-//
-// TODO(bmizerany): Move Links/Resolve/etc. out of this package.
 func (c *DiskCache) Resolve(name string) (Digest, error) {
 	name, digest := splitNameDigest(name)
 	if digest != "" {
@@ -279,18 +277,6 @@ func (c *DiskCache) Get(d Digest) (Entry, error) {
 // It returns an error if either the name or digest is invalid, or if link
 // creation encounters any issues.
 func (c *DiskCache) Link(name string, d Digest) error {
-	// TODO(bmizerany): Move link handling from cache to registry.
-	//
-	// We originally placed links in the cache due to its storage
-	// knowledge. However, the registry likely offers better context for
-	// naming concerns, and our API design shouldn't be tightly coupled to
-	// our on-disk format.
-	//
-	// Links work effectively when independent from physical location -
-	// they can reference content with matching SHA regardless of storage
-	// location. In an upcoming change, we plan to shift this
-	// responsibility to the registry where it better aligns with the
-	// system's conceptual model.
 	manifest, err := c.manifestPath(name)
 	if err != nil {
 		return err
@@ -341,7 +327,9 @@ func (c *DiskCache) GetFile(d Digest) string {
 	return absJoin(c.dir, "blobs", filename)
 }

-// Links returns a sequence of links in the cache in lexical order.
+// Links returns a sequence of link names. The sequence is in lexical order.
+// Names are converted from their relative path form to their name form but are
+// not guaranteed to be valid. Callers should validate the names before using.
 func (c *DiskCache) Links() iter.Seq2[string, error] {
 	return func(yield func(string, error) bool) {
 		for path, err := range c.links() {
@@ -414,12 +402,14 @@ func (c *DiskCache) links() iter.Seq2[string, error] {
 }

 type checkWriter struct {
-	d    Digest
 	size int64
-	n    int64
-	h    hash.Hash
+	d    Digest
 	f    *os.File
-	err  error
+	h    hash.Hash
+
+	w   io.Writer // underlying writer; set by creator
+	n   int64
+	err error

 	testHookBeforeFinalWrite func(*os.File)
 }
@@ -435,6 +425,10 @@ func (w *checkWriter) seterr(err error) error {
 // underlying writer is guaranteed to be the last byte of p as verified by the
 // hash.
 func (w *checkWriter) Write(p []byte) (int, error) {
+	if w.err != nil {
+		return 0, w.err
+	}
+
 	_, err := w.h.Write(p)
 	if err != nil {
 		return 0, w.seterr(err)
@@ -453,7 +447,7 @@ func (w *checkWriter) Write(p []byte) (int, error) {
 	if nextSize > w.size {
 		return 0, w.seterr(fmt.Errorf("content exceeds expected size: %d > %d", nextSize, w.size))
 	}
-	n, err := w.f.Write(p)
+	n, err := w.w.Write(p)
 	w.n += int64(n)
 	return n, w.seterr(err)
 }
@@ -493,10 +487,12 @@ func (c *DiskCache) copyNamedFile(name string, file io.Reader, out Digest, size

 	// Copy file to f, but also into h to double-check hash.
 	cw := &checkWriter{
-		d:                        out,
-		size:                     size,
-		h:                        sha256.New(),
-		f:                        f,
+		d:    out,
+		size: size,
+		h:    sha256.New(),
+		f:    f,
+		w:    f,
+
 		testHookBeforeFinalWrite: c.testHookBeforeFinalWrite,
 	}
 	n, err := io.Copy(cw, file)
@@ -532,11 +528,6 @@ func splitNameDigest(s string) (name, digest string) {
 var errInvalidName = errors.New("invalid name")

 func nameToPath(name string) (_ string, err error) {
-	if strings.Contains(name, "@") {
-		// TODO(bmizerany): HACK: Fix names.Parse to validate.
-		// TODO(bmizerany): merge with default parts (maybe names.Merge(a, b))
-		return "", errInvalidName
-	}
 	n := names.Parse(name)
 	if !n.IsFullyQualified() {
 		return "", errInvalidName
@@ -547,8 +538,7 @@ func nameToPath(name string) (_ string, err error) {
 func absJoin(pp ...string) string {
 	abs, err := filepath.Abs(filepath.Join(pp...))
 	if err != nil {
-		// Likely a bug bug or a bad OS problem. Just panic.
-		panic(err)
+		panic(err) // this should never happen
 	}
 	return abs
 }
--- a/server/internal/cache/blob/chunked.go
+++ b/server/internal/cache/blob/chunked.go
@@ -0,0 +1,73 @@
+package blob
+
+import (
+	"crypto/sha256"
+	"errors"
+	"io"
+	"os"
+)
+
+// Chunk represents a range of bytes in a blob.
+type Chunk struct {
+	Start int64
+	End   int64
+}
+
+// Size returns end minus start plus one.
+func (c Chunk) Size() int64 {
+	return c.End - c.Start + 1
+}
+
+// Chunker writes to a blob in chunks.
+// Its zero value is invalid. Use [DiskCache.Chunked] to create a new Chunker.
+type Chunker struct {
+	digest Digest
+	size   int64
+	f      *os.File // nil means pre-validated
+}
+
+// Chunked returns a new Chunker, ready for use storing a blob of the given
+// size in chunks.
+//
+// Use [Chunker.Put] to write data to the blob at specific offsets.
+func (c *DiskCache) Chunked(d Digest, size int64) (*Chunker, error) {
+	name := c.GetFile(d)
+	info, err := os.Stat(name)
+	if err == nil && info.Size() == size {
+		return &Chunker{}, nil
+	}
+	f, err := os.OpenFile(name, os.O_CREATE|os.O_WRONLY, 0o666)
+	if err != nil {
+		return nil, err
+	}
+	return &Chunker{digest: d, size: size, f: f}, nil
+}
+
+// Put copies chunk.Size() bytes from r to the blob at the given offset,
+// merging the data with the existing blob. It returns an error if any. As a
+// special case, if r has less than chunk.Size() bytes, Put returns
+// io.ErrUnexpectedEOF.
+func (c *Chunker) Put(chunk Chunk, d Digest, r io.Reader) error {
+	if c.f == nil {
+		return nil
+	}
+
+	cw := &checkWriter{
+		d:    d,
+		size: chunk.Size(),
+		h:    sha256.New(),
+		f:    c.f,
+		w:    io.NewOffsetWriter(c.f, chunk.Start),
+	}
+
+	_, err := io.CopyN(cw, r, chunk.Size())
+	if err != nil && errors.Is(err, io.EOF) {
+		return io.ErrUnexpectedEOF
+	}
+	return err
+}
+
+// Close closes the underlying file.
+func (c *Chunker) Close() error {
+	return c.f.Close()
+}
--- a/server/internal/cache/blob/digest.go
+++ b/server/internal/cache/blob/digest.go
@@ -63,6 +63,10 @@ func (d Digest) Short() string {
 	return fmt.Sprintf("%x", d.sum[:4])
 }

+func (d Digest) Sum() [32]byte {
+	return d.sum
+}
+
 func (d Digest) Compare(other Digest) int {
 	return slices.Compare(d.sum[:], other.sum[:])
 }
--- a/server/internal/chunks/chunks.go
+++ b/server/internal/chunks/chunks.go
@@ -1,78 +0,0 @@
-package chunks
-
-import (
-	"fmt"
-	"iter"
-	"strconv"
-	"strings"
-)
-
-type Chunk struct {
-	Start, End int64
-}
-
-func New(start, end int64) Chunk {
-	return Chunk{start, end}
-}
-
-// ParseRange parses a string in the form "unit=range" where unit is a string
-// and range is a string in the form "start-end". It returns the unit and the
-// range as a Chunk.
-func ParseRange(s string) (unit string, _ Chunk, _ error) {
-	unit, r, _ := strings.Cut(s, "=")
-	if r == "" {
-		return unit, Chunk{}, nil
-	}
-	c, err := Parse(r)
-	if err != nil {
-		return "", Chunk{}, err
-	}
-	return unit, c, err
-}
-
-// Parse parses a string in the form "start-end" and returns the Chunk.
-func Parse(s string) (Chunk, error) {
-	startStr, endStr, _ := strings.Cut(s, "-")
-	start, err := strconv.ParseInt(startStr, 10, 64)
-	if err != nil {
-		return Chunk{}, fmt.Errorf("invalid start: %v", err)
-	}
-	end, err := strconv.ParseInt(endStr, 10, 64)
-	if err != nil {
-		return Chunk{}, fmt.Errorf("invalid end: %v", err)
-	}
-	if start > end {
-		return Chunk{}, fmt.Errorf("invalid range %d-%d: start > end", start, end)
-	}
-	return Chunk{start, end}, nil
-}
-
-// Of returns a sequence of contiguous Chunks of size chunkSize that cover
-// the range [0, size), in order.
-func Of(size, chunkSize int64) iter.Seq[Chunk] {
-	return func(yield func(Chunk) bool) {
-		for start := int64(0); start < size; start += chunkSize {
-			end := min(start+chunkSize-1, size-1)
-			if !yield(Chunk{start, end}) {
-				break
-			}
-		}
-	}
-}
-
-// Count returns the number of Chunks of size chunkSize needed to cover the
-// range [0, size).
-func Count(size, chunkSize int64) int64 {
-	return (size + chunkSize - 1) / chunkSize
-}
-
-// Size returns end minus start plus one.
-func (c Chunk) Size() int64 {
-	return c.End - c.Start + 1
-}
-
-// String returns the string representation of the Chunk in the form
-// "{start}-{end}".
-func (c Chunk) String() string {
-	return fmt.Sprintf("%d-%d", c.Start, c.End)
-}
--- a/server/internal/chunks/chunks_test.go
+++ b/server/internal/chunks/chunks_test.go
@@ -1,65 +0,0 @@
-package chunks
-
-import (
-	"slices"
-	"testing"
-)
-
-func TestOf(t *testing.T) {
-	cases := []struct {
-		total     int64
-		chunkSize int64
-		want      []Chunk
-	}{
-		{0, 1, nil},
-		{1, 1, []Chunk{{0, 0}}},
-		{1, 2, []Chunk{{0, 0}}},
-		{2, 1, []Chunk{{0, 0}, {1, 1}}},
-		{10, 9, []Chunk{{0, 8}, {9, 9}}},
-	}
-
-	for _, tt := range cases {
-		got := slices.Collect(Of(tt.total, tt.chunkSize))
-		if !slices.Equal(got, tt.want) {
-			t.Errorf("[%d/%d]: got %v; want %v", tt.total, tt.chunkSize, got, tt.want)
-		}
-	}
-}
-
-func TestSize(t *testing.T) {
-	cases := []struct {
-		c    Chunk
-		want int64
-	}{
-		{Chunk{0, 0}, 1},
-		{Chunk{0, 1}, 2},
-		{Chunk{3, 4}, 2},
-	}
-
-	for _, tt := range cases {
-		got := tt.c.Size()
-		if got != tt.want {
-			t.Errorf("%v: got %d; want %d", tt.c, got, tt.want)
-		}
-	}
-}
-
-func TestCount(t *testing.T) {
-	cases := []struct {
-		total     int64
-		chunkSize int64
-		want      int64
-	}{
-		{0, 1, 0},
-		{1, 1, 1},
-		{1, 2, 1},
-		{2, 1, 2},
-		{10, 9, 2},
-	}
-	for _, tt := range cases {
-		got := Count(tt.total, tt.chunkSize)
-		if got != tt.want {
-			t.Errorf("[%d/%d]: got %d; want %d", tt.total, tt.chunkSize, got, tt.want)
-		}
-	}
-}
--- a/server/internal/client/ollama/registry.go
+++ b/server/internal/client/ollama/registry.go
@@ -19,11 +19,13 @@ import (
 	"fmt"
 	"io"
 	"io/fs"
+	"iter"
 	"log/slog"
 	"net/http"
 	"os"
 	"path/filepath"
 	"runtime"
+	"runtime/debug"
 	"slices"
 	"strconv"
 	"strings"
@@ -35,10 +37,8 @@ import (
 	"golang.org/x/sync/errgroup"

 	"github.com/ollama/ollama/server/internal/cache/blob"
-	"github.com/ollama/ollama/server/internal/chunks"
 	"github.com/ollama/ollama/server/internal/internal/backoff"
 	"github.com/ollama/ollama/server/internal/internal/names"
-	"github.com/ollama/ollama/server/internal/internal/syncs"

 	_ "embed"
 )
@@ -66,12 +66,7 @@ var (
 const (
 	// DefaultChunkingThreshold is the threshold at which a layer should be
 	// split up into chunks when downloading.
-	DefaultChunkingThreshold = 128 << 20
-
-	// DefaultMaxChunkSize is the default maximum size of a chunk to
-	// download. It is configured based on benchmarks and aims to strike a
-	// balance between download speed and memory usage.
-	DefaultMaxChunkSize = 8 << 20
+	DefaultChunkingThreshold = 64 << 20
 )

 var defaultCache = sync.OnceValues(func() (*blob.DiskCache, error) {
@@ -211,8 +206,7 @@ type Registry struct {
 	// pushing or pulling models. If zero, the number of streams is
 	// determined by [runtime.GOMAXPROCS].
 	//
-	// Clients that want "unlimited" streams should set this to a large
-	// number.
+	// A negative value means no limit.
 	MaxStreams int

 	// ChunkingThreshold is the maximum size of a layer to download in a single
@@ -266,6 +260,7 @@ func DefaultRegistry() (*Registry, error) {
 	}

 	var rc Registry
+	rc.UserAgent = UserAgent()
 	rc.Key, err = ssh.ParseRawPrivateKey(keyPEM)
 	if err != nil {
 		return nil, err
@@ -281,25 +276,24 @@ func DefaultRegistry() (*Registry, error) {
 	return &rc, nil
 }

-func (r *Registry) maxStreams() int {
-	n := cmp.Or(r.MaxStreams, runtime.GOMAXPROCS(0))
+func UserAgent() string {
+	buildinfo, _ := debug.ReadBuildInfo()
+	return fmt.Sprintf("ollama/%s (%s %s) Go/%s",
+		buildinfo.Main.Version,
+		runtime.GOARCH,
+		runtime.GOOS,
+		runtime.Version(),
+	)
+}

-	// Large downloads require a writter stream, so ensure we have at least
-	// two streams to avoid a deadlock.
-	return max(n, 2)
+func (r *Registry) maxStreams() int {
+	return cmp.Or(r.MaxStreams, runtime.GOMAXPROCS(0))
 }

 func (r *Registry) maxChunkingThreshold() int64 {
 	return cmp.Or(r.ChunkingThreshold, DefaultChunkingThreshold)
 }

-// chunkSizeFor returns the chunk size for a layer of the given size. If the
-// size is less than or equal to the max chunking threshold, the size is
-// returned; otherwise, the max chunk size is returned.
-func (r *Registry) maxChunkSize() int64 {
-	return cmp.Or(r.MaxChunkSize, DefaultMaxChunkSize)
-}
-
 type PushParams struct {
 	// From is an optional destination name for the model. If empty, the
 	// destination name is the same as the source name.
@@ -426,6 +420,21 @@ func canRetry(err error) bool {
 	return re.Status >= 500
 }

+// trackingReader is an io.Reader that tracks the number of bytes read and
+// calls the update function with the layer, the number of bytes read.
+//
+// It always calls update with a nil error.
+type trackingReader struct {
+	r io.Reader
+	n *atomic.Int64
+}
+
+func (r *trackingReader) Read(p []byte) (n int, err error) {
+	n, err = r.r.Read(p)
+	r.n.Add(int64(n))
+	return
+}
+
 // Pull pulls the model with the given name from the remote registry into the
 // cache.
 //
@@ -434,11 +443,6 @@ func canRetry(err error) bool {
 // typically slower than splitting the model up across layers, and is mostly
 // utilized for layers of type equal to "application/vnd.ollama.image".
 func (r *Registry) Pull(ctx context.Context, name string) error {
-	scheme, n, _, err := r.parseNameExtended(name)
-	if err != nil {
-		return err
-	}
-
 	m, err := r.Resolve(ctx, name)
 	if err != nil {
 		return err
@@ -457,126 +461,95 @@ func (r *Registry) Pull(ctx context.Context, name string) error {
 		return err == nil && info.Size == l.Size
 	}

-	t := traceFromContext(ctx)
-
-	g, ctx := errgroup.WithContext(ctx)
-	g.SetLimit(r.maxStreams())
-
 	layers := m.Layers
 	if m.Config != nil && m.Config.Digest.IsValid() {
 		layers = append(layers, m.Config)
 	}

-	for _, l := range layers {
+	// Send initial layer trace events to allow clients to have an
+	// understanding of work to be done before work starts.
+	t := traceFromContext(ctx)
+	skip := make([]bool, len(layers))
+	for i, l := range layers {
+		t.update(l, 0, nil)
 		if exists(l) {
+			skip[i] = true
 			t.update(l, l.Size, ErrCached)
+		}
+	}
+
+	g, ctx := errgroup.WithContext(ctx)
+	g.SetLimit(r.maxStreams())
+	for i, l := range layers {
+		if skip[i] {
 			continue
 		}

-		blobURL := fmt.Sprintf("%s://%s/v2/%s/%s/blobs/%s", scheme, n.Host(), n.Namespace(), n.Model(), l.Digest)
-		req, err := r.newRequest(ctx, "GET", blobURL, nil)
+		chunked, err := c.Chunked(l.Digest, l.Size)
 		if err != nil {
 			t.update(l, 0, err)
 			continue
 		}
+		defer chunked.Close()

-		t.update(l, 0, nil)
-
-		if l.Size <= r.maxChunkingThreshold() {
-			g.Go(func() error {
-				// TODO(bmizerany): retry/backoff like below in
-				// the chunking case
-				res, err := sendRequest(r.client(), req)
-				if err != nil {
-					return err
-				}
-				defer res.Body.Close()
-				err = c.Put(l.Digest, res.Body, l.Size)
-				if err == nil {
-					t.update(l, l.Size, nil)
-				}
-				return err
-			})
-		} else {
-			q := syncs.NewRelayReader()
+		var progress atomic.Int64
+		for cs, err := range r.chunksums(ctx, name, l) {
+			if err != nil {
+				t.update(l, progress.Load(), err)
+				break
+			}

 			g.Go(func() (err error) {
-				defer func() { q.CloseWithError(err) }()
-				return c.Put(l.Digest, q, l.Size)
-			})
+				defer func() { t.update(l, progress.Load(), err) }()

-			var progress atomic.Int64
-
-			// We want to avoid extra round trips per chunk due to
-			// redirects from the registry to the blob store, so
-			// fire an initial request to get the final URL and
-			// then use that URL for the chunk requests.
-			req.Header.Set("Range", "bytes=0-0")
-			res, err := sendRequest(r.client(), req)
-			if err != nil {
-				return err
-			}
-			res.Body.Close()
-			req = res.Request.WithContext(req.Context())
-
-			wp := writerPool{size: r.maxChunkSize()}
-
-			for chunk := range chunks.Of(l.Size, r.maxChunkSize()) {
-				if ctx.Err() != nil {
-					break
-				}
-
-				ticket := q.Take()
-				g.Go(func() (err error) {
-					defer func() {
-						if err != nil {
-							q.CloseWithError(err)
-						}
-						ticket.Close()
-						t.update(l, progress.Load(), err)
-					}()
-
-					for _, err := range backoff.Loop(ctx, 3*time.Second) {
-						if err != nil {
-							return err
-						}
-						err := func() error {
-							req := req.Clone(req.Context())
-							req.Header.Set("Range", fmt.Sprintf("bytes=%s", chunk))
-							res, err := sendRequest(r.client(), req)
-							if err != nil {
-								return err
-							}
-							defer res.Body.Close()
-
-							tw := wp.get()
-							tw.Reset(ticket)
-							defer wp.put(tw)
-
-							_, err = io.CopyN(tw, res.Body, chunk.Size())
-							if err != nil {
-								return maybeUnexpectedEOF(err)
-							}
-							if err := tw.Flush(); err != nil {
-								return err
-							}
-
-							total := progress.Add(chunk.Size())
-							if total >= l.Size {
-								q.Close()
-							}
-							return nil
-						}()
-						if !canRetry(err) {
-							return err
-						}
+				for _, err := range backoff.Loop(ctx, 3*time.Second) {
+					if err != nil {
+						return err
 					}
-					return nil
-				})
-			}
+					err := func() error {
+						req, err := http.NewRequestWithContext(ctx, "GET", cs.URL, nil)
+						if err != nil {
+							return err
+						}
+						req.Header.Set("Range", fmt.Sprintf("bytes=%d-%d", cs.Chunk.Start, cs.Chunk.End))
+						res, err := sendRequest(r.client(), req)
+						if err != nil {
+							return err
+						}
+						defer res.Body.Close()
+
+						// Count bytes towards
+						// progress, as they arrive, so
+						// that our bytes piggyback
+						// other chunk updates on
+						// completion.
+						//
+						// This tactic is enough to
+						// show "smooth" progress given
+						// the current CLI client. In
+						// the near future, the server
+						// should report download rate
+						// since it knows better than
+						// a client that is measuring
+						// rate based on wall-clock
+						// time-since-last-update.
+						body := &trackingReader{r: res.Body, n: &progress}
+
+						err = chunked.Put(cs.Chunk, cs.Digest, body)
+						if err != nil {
+							return err
+						}
+
+						return nil
+					}()
+					if !canRetry(err) {
+						return err
+					}
+				}
+				return nil
+			})
 		}
 	}
-
 	if err := g.Wait(); err != nil {
 		return err
 	}
@@ -615,8 +588,6 @@ type Manifest struct {
 	Config *Layer `json:"config"`
 }

-var emptyDigest, _ = blob.ParseDigest("sha256:0000000000000000000000000000000000000000000000000000000000000000")
-
 // Layer returns the layer with the given
 // digest, or nil if not found.
 func (m *Manifest) Layer(d blob.Digest) *Layer {
@@ -643,10 +614,9 @@ func (m Manifest) MarshalJSON() ([]byte, error) {
 		// last phase of the commit which expects it, but does nothing
 		// with it. This will be fixed in a future release of
 		// ollama.com.
-		Config *Layer `json:"config"`
+		Config Layer `json:"config"`
 	}{
-		M:      M(m),
-		Config: &Layer{Digest: emptyDigest},
+		M: M(m),
 	}
 	return json.Marshal(v)
 }
@@ -736,6 +706,123 @@ func (r *Registry) Resolve(ctx context.Context, name string) (*Manifest, error)
 	return m, nil
 }

+type chunksum struct {
+	URL    string
+	Chunk  blob.Chunk
+	Digest blob.Digest
+}
+
+// chunksums returns a sequence of chunksums for the given layer. If the layer is under the
+// chunking threshold, a single chunksum is returned that covers the entire layer. If the layer
+// is over the chunking threshold, the chunksums are read from the chunksums endpoint.
+func (r *Registry) chunksums(ctx context.Context, name string, l *Layer) iter.Seq2[chunksum, error] {
+	return func(yield func(chunksum, error) bool) {
+		scheme, n, _, err := r.parseNameExtended(name)
+		if err != nil {
+			yield(chunksum{}, err)
+			return
+		}
+
+		if l.Size < r.maxChunkingThreshold() {
+			// any layer under the threshold should be downloaded
+			// in one go.
+			cs := chunksum{
+				URL: fmt.Sprintf("%s://%s/v2/%s/%s/blobs/%s",
+					scheme,
+					n.Host(),
+					n.Namespace(),
+					n.Model(),
+					l.Digest,
+				),
+				Chunk:  blob.Chunk{Start: 0, End: l.Size - 1},
+				Digest: l.Digest,
+			}
+			yield(cs, nil)
+			return
+		}
+
+		// A chunksums response is a sequence of chunksums in a
+		// simple, easy to parse line-oriented format.
+		//
+		// Example:
+		//
+		//     >> GET /v2/<namespace>/<model>/chunksums/<digest>
+		//
+		//     << HTTP/1.1 200 OK
+		//     << Content-Location: <blobURL>
+		//     <<
+		//     << <digest> <start>-<end>
+		//     << ...
+		//
+		// The blobURL is the URL to download the chunks from.
+
+		chunksumsURL := fmt.Sprintf("%s://%s/v2/%s/%s/chunksums/%s",
+			scheme,
+			n.Host(),
+			n.Namespace(),
+			n.Model(),
+			l.Digest,
+		)
+
+		req, err := r.newRequest(ctx, "GET", chunksumsURL, nil)
+		if err != nil {
+			yield(chunksum{}, err)
+			return
+		}
+		res, err := sendRequest(r.client(), req)
+		if err != nil {
+			yield(chunksum{}, err)
+			return
+		}
+		defer res.Body.Close()
+		if res.StatusCode != 200 {
+			err := fmt.Errorf("chunksums: unexpected status code %d", res.StatusCode)
+			yield(chunksum{}, err)
+			return
+		}
+		blobURL := res.Header.Get("Content-Location")
+
+		s := bufio.NewScanner(res.Body)
+		s.Split(bufio.ScanWords)
+		for {
+			if !s.Scan() {
+				if s.Err() != nil {
+					yield(chunksum{}, s.Err())
+				}
+				return
+			}
+			d, err := blob.ParseDigest(s.Bytes())
+			if err != nil {
+				yield(chunksum{}, fmt.Errorf("invalid digest: %q", s.Bytes()))
+				return
+			}
+
+			if !s.Scan() {
+				err := s.Err()
+				if err == nil {
+					err = fmt.Errorf("missing chunk range for digest %s", d)
+				}
+				yield(chunksum{}, err)
+				return
+			}
+			chunk, err := parseChunk(s.Bytes())
+			if err != nil {
+				yield(chunksum{}, fmt.Errorf("invalid chunk range for digest %s: %q", d, s.Bytes()))
+				return
+			}
+
+			cs := chunksum{
+				URL:    blobURL,
+				Chunk:  chunk,
+				Digest: d,
+			}
+			if !yield(cs, nil) {
+				return
+			}
+		}
+	}
+}
+
 func (r *Registry) client() *http.Client {
 	if r.HTTPClient != nil {
 		return r.HTTPClient
@@ -898,13 +985,6 @@ func checkData(url string) string {
 	return fmt.Sprintf("GET,%s,%s", url, zeroSum)
 }

-func maybeUnexpectedEOF(err error) error {
-	if errors.Is(err, io.EOF) {
-		return io.ErrUnexpectedEOF
-	}
-	return err
-}
-
 type publicError struct {
 	wrapped error
 	message string
@@ -991,27 +1071,22 @@ func splitExtended(s string) (scheme, name, digest string) {
 	return scheme, s, digest
 }

-type writerPool struct {
-	size int64 // set by the caller
-
-	mu sync.Mutex
-	ws []*bufio.Writer
-}
-
-func (p *writerPool) get() *bufio.Writer {
-	p.mu.Lock()
-	defer p.mu.Unlock()
-	if len(p.ws) == 0 {
-		return bufio.NewWriterSize(nil, int(p.size))
+// parseChunk parses a string in the form "start-end" and returns the Chunk.
+func parseChunk[S ~string | ~[]byte](s S) (blob.Chunk, error) {
+	startPart, endPart, found := strings.Cut(string(s), "-")
+	if !found {
+		return blob.Chunk{}, fmt.Errorf("chunks: invalid range %q: missing '-'", s)
 	}
-	w := p.ws[len(p.ws)-1]
-	p.ws = p.ws[:len(p.ws)-1]
-	return w
-}
-
-func (p *writerPool) put(w *bufio.Writer) {
-	p.mu.Lock()
-	defer p.mu.Unlock()
-	w.Reset(nil)
-	p.ws = append(p.ws, w)
+	start, err := strconv.ParseInt(startPart, 10, 64)
+	if err != nil {
+		return blob.Chunk{}, fmt.Errorf("chunks: invalid start to %q: %v", s, err)
+	}
+	end, err := strconv.ParseInt(endPart, 10, 64)
+	if err != nil {
+		return blob.Chunk{}, fmt.Errorf("chunks: invalid end to %q: %v", s, err)
+	}
+	if start > end {
+		return blob.Chunk{}, fmt.Errorf("chunks: invalid range %q: start > end", s)
+	}
+	return blob.Chunk{Start: start, End: end}, nil
 }
--- a/server/internal/client/ollama/registry_test.go
+++ b/server/internal/client/ollama/registry_test.go
@@ -21,7 +21,6 @@ import (
 	"time"

 	"github.com/ollama/ollama/server/internal/cache/blob"
-	"github.com/ollama/ollama/server/internal/chunks"
 	"github.com/ollama/ollama/server/internal/testutil"
 )

@@ -428,7 +427,7 @@ func TestRegistryPullCached(t *testing.T) {
 	err := rc.Pull(ctx, "single")
 	testutil.Check(t, err)

-	want := []int64{6}
+	want := []int64{0, 6}
 	if !errors.Is(errors.Join(errs...), ErrCached) {
 		t.Errorf("errs = %v; want %v", errs, ErrCached)
 	}
@@ -531,54 +530,6 @@ func TestRegistryPullMixedCachedNotCached(t *testing.T) {
 	}
 }

-func TestRegistryPullChunking(t *testing.T) {
-	rc, _ := newClient(t, func(w http.ResponseWriter, r *http.Request) {
-		t.Log("request:", r.URL.Host, r.Method, r.URL.Path, r.Header.Get("Range"))
-		if r.URL.Host != "blob.store" {
-			// The production registry redirects to the blob store.
-			http.Redirect(w, r, "http://blob.store"+r.URL.Path, http.StatusFound)
-			return
-		}
-		if strings.Contains(r.URL.Path, "/blobs/") {
-			rng := r.Header.Get("Range")
-			if rng == "" {
-				http.Error(w, "missing range", http.StatusBadRequest)
-				return
-			}
-			_, c, err := chunks.ParseRange(r.Header.Get("Range"))
-			if err != nil {
-				panic(err)
-			}
-			io.WriteString(w, "remote"[c.Start:c.End+1])
-			return
-		}
-		fmt.Fprintf(w, `{"layers":[{"digest":%q,"size":6}]}`, blob.DigestFromBytes("remote"))
-	})
-
-	// Force chunking by setting the threshold to less than the size of the
-	// layer.
-	rc.ChunkingThreshold = 3
-	rc.MaxChunkSize = 3
-
-	var reads []int64
-	ctx := WithTrace(t.Context(), &Trace{
-		Update: func(d *Layer, n int64, err error) {
-			if err != nil {
-				t.Errorf("update %v %d %v", d, n, err)
-			}
-			reads = append(reads, n)
-		},
-	})
-
-	err := rc.Pull(ctx, "remote")
-	testutil.Check(t, err)
-
-	want := []int64{0, 3, 6}
-	if !slices.Equal(reads, want) {
-		t.Errorf("reads = %v; want %v", reads, want)
-	}
-}
-
 func TestRegistryResolveByDigest(t *testing.T) {
 	check := testutil.Checker(t)

--- a/server/internal/cmd/oppbench/oppbench.go
+++ b/server/internal/cmd/oppbench/oppbench.go
@@ -1,11 +0,0 @@
-package main
-
-import (
-	"fmt"
-	"os"
-)
-
-func main() {
-	fmt.Println("Run as 'go test -bench=.' to run the benchmarks")
-	os.Exit(1)
-}
--- a/server/internal/cmd/oppbench/oppbench_test.go
+++ b/server/internal/cmd/oppbench/oppbench_test.go
@@ -1,107 +0,0 @@
-package main
-
-import (
-	"bytes"
-	"context"
-	"fmt"
-	"io"
-	"net/http"
-	"os"
-	"path/filepath"
-	"runtime"
-	"sync/atomic"
-	"testing"
-	"time"
-
-	"github.com/ollama/ollama/server/internal/chunks"
-	"golang.org/x/sync/errgroup"
-)
-
-func BenchmarkDownload(b *testing.B) {
-	run := func(fileSize, chunkSize int64) {
-		name := fmt.Sprintf("size=%d/chunksize=%d", fileSize, chunkSize)
-		b.Run(name, func(b *testing.B) { benchmarkDownload(b, fileSize, chunkSize) })
-	}
-
-	run(100<<20, 8<<20)
-	run(100<<20, 16<<20)
-	run(100<<20, 32<<20)
-	run(100<<20, 64<<20)
-	run(100<<20, 128<<20) // 1 chunk
-}
-
-func run(ctx context.Context, c *http.Client, chunk chunks.Chunk) error {
-	const blobURL = "https://ollama.com/v2/x/x/blobs/sha256-4824460d29f2058aaf6e1118a63a7a197a09bed509f0e7d4e2efb1ee273b447d"
-	req, err := http.NewRequestWithContext(ctx, "GET", blobURL, nil)
-	if err != nil {
-		return err
-	}
-	req.Header.Set("Range", fmt.Sprintf("bytes=%s", chunk))
-	res, err := c.Do(req)
-	if err != nil {
-		return err
-	}
-	defer res.Body.Close()
-
-	_, err = io.CopyN(io.Discard, res.Body, chunk.Size()) // will io.EOF on short read
-	return err
-}
-
-var sleepTime atomic.Int64
-
-func benchmarkDownload(b *testing.B, fileSize, chunkSize int64) {
-	client := &http.Client{
-		Transport: func() http.RoundTripper {
-			tr := http.DefaultTransport.(*http.Transport).Clone()
-			tr.DisableKeepAlives = true
-			return tr
-		}(),
-	}
-	defer client.CloseIdleConnections()
-
-	// warm up the client
-	run(context.Background(), client, chunks.New(0, 1<<20))
-
-	b.SetBytes(fileSize)
-	b.ReportAllocs()
-
-	// Give our CDN a min to breathe between benchmarks.
-	time.Sleep(time.Duration(sleepTime.Swap(3)))
-
-	for b.Loop() {
-		g, ctx := errgroup.WithContext(b.Context())
-		g.SetLimit(runtime.GOMAXPROCS(0))
-		for chunk := range chunks.Of(fileSize, chunkSize) {
-			g.Go(func() error { return run(ctx, client, chunk) })
-		}
-		if err := g.Wait(); err != nil {
-			b.Fatal(err)
-		}
-	}
-}
-
-func BenchmarkWrite(b *testing.B) {
-	b.Run("chunksize=1MiB", func(b *testing.B) { benchmarkWrite(b, 1<<20) })
-}
-
-func benchmarkWrite(b *testing.B, chunkSize int) {
-	b.ReportAllocs()
-
-	dir := b.TempDir()
-	f, err := os.Create(filepath.Join(dir, "write-single"))
-	if err != nil {
-		b.Fatal(err)
-	}
-	defer f.Close()
-
-	data := make([]byte, chunkSize)
-	b.SetBytes(int64(chunkSize))
-	r := bytes.NewReader(data)
-	for b.Loop() {
-		r.Reset(data)
-		_, err := io.Copy(f, r)
-		if err != nil {
-			b.Fatal(err)
-		}
-	}
-}
--- a/server/internal/registry/server.go
+++ b/server/internal/registry/server.go
@@ -1,6 +1,5 @@
-// Package registry provides an http.Handler for handling local Ollama API
-// requests for performing tasks related to the ollama.com model registry and
-// the local disk cache.
+// Package registry implements an http.Handler for handling local Ollama API
+// model management requests. See [Local] for details.
 package registry

 import (
@@ -10,6 +9,7 @@ import (
 	"fmt"
 	"io"
 	"log/slog"
+	"maps"
 	"net/http"
 	"sync"
 	"time"
@@ -18,16 +18,11 @@ import (
 	"github.com/ollama/ollama/server/internal/client/ollama"
 )

-// Local is an http.Handler for handling local Ollama API requests for
-// performing tasks related to the ollama.com model registry combined with the
-// local disk cache.
+// Local implements an http.Handler for handling local Ollama API model
+// management requests, such as pushing, pulling, and deleting models.
 //
-// It is not concern of Local, or this package, to handle model creation, which
-// proceeds any registry operations for models it produces.
-//
-// NOTE: The package built for dealing with model creation should use
-// [DefaultCache] to access the blob store and not attempt to read or write
-// directly to the blob disk cache.
+// It can be arranged for all unknown requests to be passed through to a
+// fallback handler, if one is provided.
 type Local struct {
 	Client *ollama.Registry // required
 	Logger *slog.Logger     // required
@@ -63,6 +58,7 @@ func (e serverError) Error() string {
 var (
 	errMethodNotAllowed = &serverError{405, "method_not_allowed", "method not allowed"}
 	errNotFound         = &serverError{404, "not_found", "not found"}
+	errModelNotFound    = &serverError{404, "not_found", "model not found"}
 	errInternalError    = &serverError{500, "internal_error", "internal server error"}
 )

@@ -175,8 +171,16 @@ func (s *Local) serveHTTP(rec *statusCodeRecorder, r *http.Request) {
 }

 type params struct {
-	DeprecatedName string `json:"name"`  // Use [params.model]
-	Model          string `json:"model"` // Use [params.model]
+	// DeprecatedName is the name of the model to push, pull, or delete,
+	// but is deprecated. New clients should use [Model] instead.
+	//
+	// Use [model()] to get the model name for both old and new API requests.
+	DeprecatedName string `json:"name"`
+
+	// Model is the name of the model to push, pull, or delete.
+	//
+	// Use [model()] to get the model name for both old and new API requests.
+	Model string `json:"model"`

 	// AllowNonTLS is a flag that indicates a client using HTTP
 	// is doing so, deliberately.
@@ -189,9 +193,18 @@ type params struct {
 	// confusing flags such as this.
 	AllowNonTLS bool `json:"insecure"`

-	// ProgressStream is a flag that indicates the client is expecting a stream of
-	// progress updates.
-	ProgressStream bool `json:"stream"`
+	// Stream, if true, will make the server send progress updates in a
+	// streaming of JSON objects. If false, the server will send a single
+	// JSON object with the final status as "success", or an error object
+	// if an error occurred.
+	//
+	// Unfortunately, this API was designed to be a bit awkward. Stream is
+	// defined to default to true if not present, so we need a way to check
+	// if the client decisively it to false. So, we use a pointer to a
+	// bool. Gross.
+	//
+	// Use [stream()] to get the correct value for this field.
+	Stream *bool `json:"stream"`
 }

 // model returns the model name for both old and new API requests.
@@ -199,6 +212,13 @@ func (p params) model() string {
 	return cmp.Or(p.Model, p.DeprecatedName)
 }

+func (p params) stream() bool {
+	if p.Stream == nil {
+		return true
+	}
+	return *p.Stream
+}
+
 func (s *Local) handleDelete(_ http.ResponseWriter, r *http.Request) error {
 	if r.Method != "DELETE" {
 		return errMethodNotAllowed
@@ -212,16 +232,16 @@ func (s *Local) handleDelete(_ http.ResponseWriter, r *http.Request) error {
 		return err
 	}
 	if !ok {
-		return &serverError{404, "not_found", "model not found"}
+		return errModelNotFound
 	}
-	if s.Prune == nil {
-		return nil
+	if s.Prune != nil {
+		return s.Prune()
 	}
-	return s.Prune()
+	return nil
 }

 type progressUpdateJSON struct {
-	Status    string      `json:"status"`
+	Status    string      `json:"status,omitempty,omitzero"`
 	Digest    blob.Digest `json:"digest,omitempty,omitzero"`
 	Total     int64       `json:"total,omitempty,omitzero"`
 	Completed int64       `json:"completed,omitempty,omitzero"`
@@ -237,6 +257,17 @@ func (s *Local) handlePull(w http.ResponseWriter, r *http.Request) error {
 		return err
 	}

+	enc := json.NewEncoder(w)
+	if !p.stream() {
+		if err := s.Client.Pull(r.Context(), p.model()); err != nil {
+			if errors.Is(err, ollama.ErrModelNotFound) {
+				return errModelNotFound
+			}
+			return err
+		}
+		return enc.Encode(progressUpdateJSON{Status: "success"})
+	}
+
 	maybeFlush := func() {
 		fl, _ := w.(http.Flusher)
 		if fl != nil {
@@ -246,69 +277,67 @@ func (s *Local) handlePull(w http.ResponseWriter, r *http.Request) error {
 	defer maybeFlush()

 	var mu sync.Mutex
-	enc := json.NewEncoder(w)
-	enc.Encode(progressUpdateJSON{Status: "pulling manifest"})
+	progress := make(map[*ollama.Layer]int64)

-	ctx := ollama.WithTrace(r.Context(), &ollama.Trace{
-		Update: func(l *ollama.Layer, n int64, err error) {
-			mu.Lock()
-			defer mu.Unlock()
+	progressCopy := make(map[*ollama.Layer]int64, len(progress))
+	pushUpdate := func() {
+		defer maybeFlush()

-			// TODO(bmizerany): coalesce these updates; writing per
-			// update is expensive
+		// TODO(bmizerany): This scales poorly with more layers due to
+		// needing to flush out them all in one big update. We _could_
+		// just flush on the changed ones, or just track the whole
+		// download. Needs more thought. This is fine for now.
+		mu.Lock()
+		maps.Copy(progressCopy, progress)
+		mu.Unlock()
+		for l, n := range progress {
 			enc.Encode(progressUpdateJSON{
 				Digest:    l.Digest,
-				Status:    "pulling",
 				Total:     l.Size,
 				Completed: n,
 			})
+		}
+	}
+
+	t := time.NewTicker(time.Hour) // "unstarted" timer
+	start := sync.OnceFunc(func() {
+		pushUpdate()
+		t.Reset(100 * time.Millisecond)
+	})
+	ctx := ollama.WithTrace(r.Context(), &ollama.Trace{
+		Update: func(l *ollama.Layer, n int64, err error) {
+			if n > 0 {
+				start() // flush initial state
+			}
+			mu.Lock()
+			progress[l] = n
+			mu.Unlock()
 		},
 	})

 	done := make(chan error, 1)
 	go func() {
-		// TODO(bmizerany): continue to support non-streaming responses
 		done <- s.Client.Pull(ctx, p.model())
 	}()

-	func() {
-		t := time.NewTicker(100 * time.Millisecond)
-		defer t.Stop()
-		for {
-			select {
-			case <-t.C:
-				mu.Lock()
-				maybeFlush()
-				mu.Unlock()
-			case err := <-done:
-				if err != nil {
-					var status string
-					if errors.Is(err, ollama.ErrModelNotFound) {
-						status = fmt.Sprintf("error: model %q not found", p.model())
-						enc.Encode(progressUpdateJSON{Status: status})
-					} else {
-						status = fmt.Sprintf("error: %v", err)
-						enc.Encode(progressUpdateJSON{Status: status})
-					}
-					return
+	for {
+		select {
+		case <-t.C:
+			pushUpdate()
+		case err := <-done:
+			pushUpdate()
+			if err != nil {
+				var status string
+				if errors.Is(err, ollama.ErrModelNotFound) {
+					status = fmt.Sprintf("error: model %q not found", p.model())
+				} else {
+					status = fmt.Sprintf("error: %v", err)
 				}
-
-				// These final updates are not strictly necessary, because they have
-				// already happened at this point. Our pull handler code used to do
-				// these steps after, not during, the pull, and they were slow, so we
-				// wanted to provide feedback to users what was happening. For now, we
-				// keep them to not jar users who are used to seeing them. We can phase
-				// them out with a new and nicer UX later. One without progress bars
-				// and digests that no one cares about.
-				enc.Encode(progressUpdateJSON{Status: "verifying layers"})
-				enc.Encode(progressUpdateJSON{Status: "writing manifest"})
-				enc.Encode(progressUpdateJSON{Status: "success"})
-				return
+				enc.Encode(progressUpdateJSON{Status: status})
 			}
+			return nil
 		}
-	}()
-
-	return nil
+	}
 }

 func decodeUserJSON[T any](r io.Reader) (T, error) {
--- a/server/internal/registry/server_test.go
+++ b/server/internal/registry/server_test.go
@@ -4,7 +4,6 @@ import (
 	"bytes"
 	"context"
 	"encoding/json"
-	"fmt"
 	"io"
 	"io/fs"
 	"net"
@@ -160,7 +159,6 @@ var registryFS = sync.OnceValue(func() fs.FS {
 	// to \n when parsing the txtar on Windows.
 	data := bytes.ReplaceAll(registryTXT, []byte("\r\n"), []byte("\n"))
 	a := txtar.Parse(data)
-	fmt.Printf("%q\n", a.Comment)
 	fsys, err := txtar.FS(a)
 	if err != nil {
 		panic(err)
@@ -179,7 +177,7 @@ func TestServerPull(t *testing.T) {
 			w.WriteHeader(404)
 			io.WriteString(w, `{"errors": [{"code": "MANIFEST_UNKNOWN", "message": "manifest unknown"}]}`)
 		default:
-			t.Logf("serving file: %s", r.URL.Path)
+			t.Logf("serving blob: %s", r.URL.Path)
 			modelsHandler.ServeHTTP(w, r)
 		}
 	})
@@ -188,7 +186,7 @@ func TestServerPull(t *testing.T) {
 		t.Helper()

 		if got.Code != 200 {
-			t.Fatalf("Code = %d; want 200", got.Code)
+			t.Errorf("Code = %d; want 200", got.Code)
 		}
 		gotlines := got.Body.String()
 		t.Logf("got:\n%s", gotlines)
@@ -197,35 +195,29 @@ func TestServerPull(t *testing.T) {
 			want, unwanted := strings.CutPrefix(want, "!")
 			want = strings.TrimSpace(want)
 			if !unwanted && !strings.Contains(gotlines, want) {
-				t.Fatalf("! missing %q in body", want)
+				t.Errorf("! missing %q in body", want)
 			}
 			if unwanted && strings.Contains(gotlines, want) {
-				t.Fatalf("! unexpected %q in body", want)
+				t.Errorf("! unexpected %q in body", want)
 			}
 		}
 	}

 	got := s.send(t, "POST", "/api/pull", `{"model": "BOOM"}`)
 	checkResponse(got, `
-		{"status":"pulling manifest"}
 		{"status":"error: request error https://example.com/v2/library/BOOM/manifests/latest: registry responded with status 999: boom"}
 	`)

 	got = s.send(t, "POST", "/api/pull", `{"model": "smol"}`)
 	checkResponse(got, `
-		{"status":"pulling manifest"}
-		{"status":"pulling","digest":"sha256:68e0ec597aee59d35f8dc44942d7b17d471ade10d3aca07a5bb7177713950312","total":5}
-		{"status":"pulling","digest":"sha256:ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356","total":3}
-		{"status":"pulling","digest":"sha256:68e0ec597aee59d35f8dc44942d7b17d471ade10d3aca07a5bb7177713950312","total":5,"completed":5}
-		{"status":"pulling","digest":"sha256:ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356","total":3,"completed":3}
-		{"status":"verifying layers"}
-		{"status":"writing manifest"}
-		{"status":"success"}
+		{"digest":"sha256:68e0ec597aee59d35f8dc44942d7b17d471ade10d3aca07a5bb7177713950312","total":5}
+		{"digest":"sha256:ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356","total":3}
+		{"digest":"sha256:68e0ec597aee59d35f8dc44942d7b17d471ade10d3aca07a5bb7177713950312","total":5,"completed":5}
+		{"digest":"sha256:ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356","total":3,"completed":3}
 	`)

 	got = s.send(t, "POST", "/api/pull", `{"model": "unknown"}`)
 	checkResponse(got, `
-		{"status":"pulling manifest"}
 		{"status":"error: model \"unknown\" not found"}
 	`)

@@ -240,19 +232,39 @@ func TestServerPull(t *testing.T) {

 	got = s.send(t, "POST", "/api/pull", `{"model": "://"}`)
 	checkResponse(got, `
-		{"status":"pulling manifest"}
 		{"status":"error: invalid or missing name: \"\""}
-
-		!verifying
-		!writing
-		!success
 	`)
+
+	// Non-streaming pulls
+	got = s.send(t, "POST", "/api/pull", `{"model": "://", "stream": false}`)
+	checkErrorResponse(t, got, 400, "bad_request", "invalid or missing name")
+	got = s.send(t, "POST", "/api/pull", `{"model": "smol", "stream": false}`)
+	checkResponse(got, `
+		{"status":"success"}
+		!digest
+		!total
+		!completed
+	`)
+	got = s.send(t, "POST", "/api/pull", `{"model": "unknown", "stream": false}`)
+	checkErrorResponse(t, got, 404, "not_found", "model not found")
 }

 func TestServerUnknownPath(t *testing.T) {
 	s := newTestServer(t, nil)
 	got := s.send(t, "DELETE", "/api/unknown", `{}`)
 	checkErrorResponse(t, got, 404, "not_found", "not found")
+
+	var fellback bool
+	s.Fallback = http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+		fellback = true
+	})
+	got = s.send(t, "DELETE", "/api/unknown", `{}`)
+	if !fellback {
+		t.Fatal("expected Fallback to be called")
+	}
+	if got.Code != 200 {
+		t.Fatalf("Code = %d; want 200", got.Code)
+	}
 }

 func checkErrorResponse(t *testing.T, got *httptest.ResponseRecorder, status int, code, msg string) {
--- a/server/routes.go
+++ b/server/routes.go
@@ -435,7 +435,7 @@ func (s *Server) EmbedHandler(c *gin.Context) {
 		return
 	}

-	kvData, err := getKVData(m.ModelPath, false)
+	kvData, _, err := getModelData(m.ModelPath, false)
 	if err != nil {
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
@@ -483,8 +483,7 @@ func (s *Server) EmbedHandler(c *gin.Context) {
 	}

 	if err := g.Wait(); err != nil {
-		slog.Error("embedding generation failed", "error", err)
-		c.JSON(http.StatusInternalServerError, gin.H{"error": fmt.Errorf("failed to generate embeddings: %v", err)})
+		c.AbortWithStatusJSON(http.StatusInternalServerError, gin.H{"error": strings.TrimSpace(err.Error())})
 		return
 	}

@@ -545,8 +544,7 @@ func (s *Server) EmbeddingsHandler(c *gin.Context) {

 	embedding, err := r.Embedding(c.Request.Context(), req.Prompt)
 	if err != nil {
-		slog.Info(fmt.Sprintf("embedding generation failed: %v", err))
-		c.JSON(http.StatusInternalServerError, gin.H{"error": fmt.Errorf("failed to generate embedding: %v", err)})
+		c.AbortWithStatusJSON(http.StatusInternalServerError, gin.H{"error": strings.TrimSpace(err.Error())})
 		return
 	}

@@ -850,16 +848,23 @@ func GetModelInfo(req api.ShowRequest) (*api.ShowResponse, error) {
 	fmt.Fprint(&sb, m.String())
 	resp.Modelfile = sb.String()

-	kvData, err := getKVData(m.ModelPath, req.Verbose)
+	kvData, tensors, err := getModelData(m.ModelPath, req.Verbose)
 	if err != nil {
 		return nil, err
 	}
+
 	delete(kvData, "general.name")
 	delete(kvData, "tokenizer.chat_template")
 	resp.ModelInfo = kvData

+	tensorData := make([]api.Tensor, len(tensors.Items()))
+	for cnt, t := range tensors.Items() {
+		tensorData[cnt] = api.Tensor{Name: t.Name, Type: t.Type(), Shape: t.Shape}
+	}
+	resp.Tensors = tensorData
+
 	if len(m.ProjectorPaths) > 0 {
-		projectorData, err := getKVData(m.ProjectorPaths[0], req.Verbose)
+		projectorData, _, err := getModelData(m.ProjectorPaths[0], req.Verbose)
 		if err != nil {
 			return nil, err
 		}
@@ -869,17 +874,17 @@ func GetModelInfo(req api.ShowRequest) (*api.ShowResponse, error) {
 	return resp, nil
 }

-func getKVData(digest string, verbose bool) (ggml.KV, error) {
+func getModelData(digest string, verbose bool) (ggml.KV, ggml.Tensors, error) {
 	maxArraySize := 0
 	if verbose {
 		maxArraySize = -1
 	}
-	kvData, err := llm.LoadModel(digest, maxArraySize)
+	data, err := llm.LoadModel(digest, maxArraySize)
 	if err != nil {
-		return nil, err
+		return nil, ggml.Tensors{}, err
 	}

-	kv := kvData.KV()
+	kv := data.KV()

 	if !verbose {
 		for k := range kv {
@@ -889,7 +894,7 @@ func getKVData(digest string, verbose bool) (ggml.KV, error) {
 		}
 	}

-	return kv, nil
+	return kv, data.Tensors(), nil
 }

 func (s *Server) ListHandler(c *gin.Context) {
--- a/types/bfloat16/LICENSE
+++ b/types/bfloat16/LICENSE
@@ -0,0 +1,21 @@
+MIT License
+
+Copyright (c) 2021 Tristan Rice
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
--- a/types/bfloat16/bfloat16.go
+++ b/types/bfloat16/bfloat16.go
@@ -0,0 +1,57 @@
+// Vendored code from https://github.com/d4l3k/go-bfloat16
+// unsafe pointer replaced by "math"
+package bfloat16
+
+import "math"
+
+type BF16 uint16
+
+func FromBytes(buf []byte) BF16 {
+	return BF16(uint16(buf[0]) + uint16(buf[1])<<8)
+}
+
+func ToBytes(b BF16) []byte {
+	return []byte{byte(b & 0xFF), byte(b >> 8)}
+}
+
+func Decode(buf []byte) []BF16 {
+	var out []BF16
+	for i := 0; i < len(buf); i += 2 {
+		out = append(out, FromBytes(buf[i:]))
+	}
+	return out
+}
+
+func Encode(f []BF16) []byte {
+	var out []byte
+	for _, a := range f {
+		out = append(out, ToBytes(a)...)
+	}
+	return out
+}
+
+func DecodeFloat32(buf []byte) []float32 {
+	var out []float32
+	for i := 0; i < len(buf); i += 2 {
+		out = append(out, ToFloat32(FromBytes(buf[i:])))
+	}
+	return out
+}
+
+func EncodeFloat32(f []float32) []byte {
+	var out []byte
+	for _, a := range f {
+		out = append(out, ToBytes(FromFloat32(a))...)
+	}
+	return out
+}
+
+func ToFloat32(b BF16) float32 {
+	u32 := uint32(b) << 16
+	return math.Float32frombits(u32)
+}
+
+func FromFloat32(f float32) BF16 {
+	u32 := math.Float32bits(f)
+	return BF16(u32 >> 16)
+}
--- a/types/bfloat16/bfloat16_test.go
+++ b/types/bfloat16/bfloat16_test.go
@@ -0,0 +1,53 @@
+package bfloat16
+
+import (
+	"crypto/rand"
+	"reflect"
+	"testing"
+)
+
+func randomBytes(n int) []byte {
+	out := make([]byte, n)
+	if _, err := rand.Read(out); err != nil {
+		panic(err)
+	}
+	return out
+}
+
+func TestEncodeDecode(t *testing.T) {
+	b := randomBytes(1024)
+	bf16 := Decode(b)
+	out := Encode(bf16)
+	if !reflect.DeepEqual(b, out) {
+		t.Fatalf("%+v != %+v", b, out)
+	}
+}
+
+func TestEncodeDecodeFloat32(t *testing.T) {
+	b := randomBytes(1024)
+	bf16 := DecodeFloat32(b)
+	out := EncodeFloat32(bf16)
+	if !reflect.DeepEqual(b, out) {
+		t.Fatalf("%+v != %+v", b, out)
+	}
+}
+
+func TestBasicFloat32(t *testing.T) {
+	var in float32 = 1.0
+	out := ToFloat32(FromFloat32(in))
+	if !reflect.DeepEqual(in, out) {
+		t.Fatalf("%+v != %+v", in, out)
+	}
+}
+
+func TestComplexFloat32(t *testing.T) {
+	var in float32 = 123456789123456789.123456789
+	var want float32 = 123286039799267328.0
+	out := ToFloat32(FromFloat32(in))
+	if in == out {
+		t.Fatalf("no loss of precision")
+	}
+	if out != want {
+		t.Fatalf("%.16f != %.16f", want, out)
+	}
+}
Author	SHA1	Message	Date
Patrick Devine	c75b428249	fix: fixes a memory leak in bfloat16 package This change vendors in the bfloat16 package from github.com/d4l3k/go-bfloat16/ and fixes a memory leak which was being caused by using unsafe pointers instead of the math package.	2025-03-17 21:46:12 -07:00
Michael Yang	021dcf089d	Merge pull request #9824 from ollama/mxyng/sched conditionally enable parallel pipelines	2025-03-17 15:41:37 -07:00
Jesse Gross	bf24498b1e	ollamarunner: Check for minBatch of context space when shifting Models can specify that a group of inputs need to be handled a single batch. However, context shifting didn't respect this and could trigger a break anyways. In this case, we should instead trigger a context shift earlier so that it occurs before the grouped batch. Note that there still some corner cases: - A long prompt that exceeds the context window can get truncated in the middle of an image. With the current models, this will result in the model not recognizing the image at all, which is pretty much the expected result with truncation. - The context window is set less than the minimum batch size. The only solution to this is to refuse to load the model with these settings. However, this can never occur with current models and default settings. Since users are unlikely to run into these scenarios, fixing them is left as a follow up.	2025-03-17 15:33:16 -07:00
Bruce MacDonald	95e271d98f	runner: remove cache prompt flag from ollama runner (#9826 ) We do not need to bypass the prompt caching in the ollama runner yet, as only embedding models needed to bypass the prompt caching. When embedding models are implemented they can skip initializing this cache completely.	2025-03-17 15:11:15 -07:00
Jeffrey Morgan	364629b8d6	ml/backend/ggml: allocate memory with malloc when loading model (#9822 )	2025-03-17 13:32:40 -07:00
Parth Sareen	108fe02165	sample: make mutations in transforms explicit (#9743 ) * updated minP to use early exit making use of sorted tokens	2025-03-17 11:24:18 -07:00
Michael Yang	4561fff36e	conditionally enable parallel pipelines	2025-03-17 09:46:07 -07:00
Daniel Hiltgen	50b5962042	Add support for ROCm gfx1151 (#9773 )	2025-03-17 09:33:57 -07:00
Louis Beaumont	e27e4a3c1b	readme: add screenpipe to community integrations (#9786 )	2025-03-16 21:56:42 -04:00
zeo	088514bbd4	readme: add Ellama to list of community integrations (#9800 )	2025-03-16 21:54:43 -04:00
Patrick Devine	2c8b484643	fix: correctly save in interactive mode (#9788 ) This fixes the case where a FROM line in previous modelfile points to a file which may/may not be present in a different ollama instance. We shouldn't be relying on the filename though and instead just check if the FROM line was instead a valid model name and point to that instead.	2025-03-15 12:09:02 -07:00
Blake Mizerany	8294676150	server/internal/client/ollama: set User-Agent for registry client (#9775 ) This sets the agent header in DefaultRegistry to include the version of the client, OS, and architecture in the previous format, with a minor twist. Note: The version is obtained from the build info, instead of the version in version.Version, which should not longer be necessary, but we can remove in a future commit. Using the build info is more accurate and also provides extra build information if the build is not tagged, and if it is "dirty". Previously, the version was just "0.0.0" with no other helpful information. The ollama.com registry and others handle this swimmingly.	2025-03-14 18:33:07 -07:00
Patrick Devine	ef378ad673	gemma3 quantization (#9776 )	2025-03-14 17:41:07 -07:00
Daniel Hiltgen	2d2247e59e	Align versions for local builds (#9635 ) Darwin was using a different pattern for the version string than linux or windows.	2025-03-14 15:44:08 -07:00
Jesse Gross	7bf793a600	gemma3: Allow multiple image in a single input Previously processing multiple images in a batch would trigger segfaults so sending images together was disabled as a way to mitigate this. The trigger was processing one image on the CPU and one on the GPU. This can no longer happen: - The vision encoder is now on the GPU so both images would be processed on the GPU. - We require images to be fully contained in a batch and each image including its special tokens is over half the batch size. As a result, we will never get two images in the same batch. Fixes #9731	2025-03-14 15:38:54 -07:00
Jesse Gross	282bfaaa95	ollamarunner: Use a separate context per multimodal input Currently there is a single context per sequence, shared all by all multimodal inputs. Since we build a vision encoder graph per image, with a large number of inputs we can eventually hit the maximum number of graph nodes per context. This changes to use a separate context for each image, ensuring that available resource limits are consistent.	2025-03-14 15:38:54 -07:00
Jesse Gross	9679f40146	ml: Allow models to constrain inputs to a single batch Models may require that a set of inputs all be processed as part of the same batch. For example, if an image has multiple patches with fully connected attention between them, we should not split the batch in the middle of an image. Fixes #9697	2025-03-14 15:38:54 -07:00
Bruce MacDonald	3892c3a703	llm: remove internal subprocess req and resp types (#9324 ) This commit refactors the LLM subsystem by removing internal subprocess request and response types. It consolidates duplicate type definitions across the codebase, moving them to centralized locations. The change also standardizes interfaces between components, simplifies the ServerStatusResp struct, and moves the ParseDurationMs function to a common package. This cleanup reduces code duplication between different runner implementations (llamarunner and ollamarunner).	2025-03-14 15:21:53 -07:00
Blake Mizerany	4e320b8b90	server/internal/chunks: remove chunks package (#9755 )	2025-03-14 08:57:59 -07:00
Blake Mizerany	eb2b22b042	server/internal/client: use chunksums for concurrent blob verification (#9746 ) Replace large-chunk blob downloads with parallel small-chunk verification to solve timeout and performance issues. Registry users experienced progressively slowing download speeds as large-chunk transfers aged, often timing out completely. The previous approach downloaded blobs in a few large chunks but required a separate, single-threaded pass to read the entire blob back from disk for verification after download completion. This change uses the new chunksums API to fetch many smaller chunk+digest pairs, allowing concurrent downloads and immediate verification as each chunk arrives. Chunks are written directly to their final positions, eliminating the entire separate verification pass. The result is more reliable downloads that maintain speed throughout the transfer process and significantly faster overall completion, especially over unstable connections or with large blobs.	2025-03-13 22:18:29 -07:00
Michael Yang	4ea4d2b189	Merge pull request #9703 from ollama/mxyng/gemma3-memory count gemma3 vision tensors	2025-03-13 16:56:34 -07:00
Michael Yang	8d76fa23ef	count non-repeating vision layers	2025-03-13 16:53:29 -07:00
Bradley Erickson	74b44fdf8f	docs: Add OLLAMA_ORIGINS for browser extension support (#9643 )	2025-03-13 16:35:20 -07:00
Michael Yang	65b88c544f	fix divide by zero	2025-03-13 16:35:00 -07:00
Michael Yang	a422ba39c9	roughly count gemma3 graph the largest operation is by far (q @ k) so just count that for simplicity	2025-03-13 16:35:00 -07:00
Michael Yang	d2ec22371e	count all vision tensors	2025-03-13 16:35:00 -07:00
Michael Yang	033cec232a	count gemma3 vision tensors	2025-03-13 16:34:42 -07:00
Michael Yang	543240fb5f	Merge pull request #9741 from ollama/mxyng/visionless fix: error if image requested without vision model	2025-03-13 15:03:25 -07:00
Patrick Devine	4bed739259	add verbose mode to the show command (#9640 ) Add metadata and tensor information to the show command to be able to see more information about a model. This outputs the same data as shown on the model details page on ollama.com	2025-03-13 14:24:27 -07:00
Patrick Devine	80c7ce381b	fix: change default context size for gemma3 (#9744 )	2025-03-13 13:59:19 -07:00
Michael Yang	ccfd41c4f0	Merge pull request #9742 from ollama/mxyng/engine-error-embeddings fix: error on models that don't support embeddings	2025-03-13 13:12:33 -07:00
Michael Yang	3e102b7dad	Update model/model.go Co-authored-by: Jeffrey Morgan <jmorganca@gmail.com>	2025-03-13 13:11:52 -07:00
Michael Yang	ec46f3286c	engine: error on embeddings; not currently implemented	2025-03-13 11:40:55 -07:00
Michael Yang	5e2e0b46b1	fix: error if image requested without vision model	2025-03-13 10:52:09 -07:00
Michael Yang	45a13b1dec	Merge pull request #9688 from Shane-XB-Qian/debug_mistype_lld ollama-debug.c: correct mistype	2025-03-13 10:12:44 -07:00
Parth Sareen	5c0b663969	sample: separate softmax and temperature transforms (#9732 )	2025-03-13 09:53:27 -07:00
shane.xb.qian	30d7a59ba8	ollama-debug.c: change 'ld' to 'PRIi64' * macOS has different definition per info from @mxyng	2025-03-13 17:10:37 +08:00
ParthSareen	4aeb67ef4c	sample: do all sorting in topK	2025-03-12 11:59:17 -07:00
ParthSareen	3ba91634c1	sample: simplify top_k=0 sorting	2025-03-12 11:59:17 -07:00
ParthSareen	1b7433b71e	sample: use container/heap for top_k	2025-03-12 11:59:17 -07:00
Bruce MacDonald	a70820daa0	models/gemma3: remove final logit softcap (#9692 ) Softcap isn't in the whitepaper/implementation for the language model so we should remove it. There is no discernible difference in output with it removed.	2025-03-12 10:17:57 -07:00
Shane-XB-Qian	6b45b1d6b4	cli: adding support ctrl-n/p like general cli (#9136 ) Signed-off-by: shane.xb.qian <shane.qian@foxmail.com>	2025-03-12 08:51:56 -07:00
shane.xb.qian	85ab552028	ollama-debug.c: correct mistype Signed-off-by: shane.xb.qian <shane.qian@foxmail.com>	2025-03-12 22:32:30 +08:00
frob	b3af953a55	cli: don't exit for invalid model during /load. (#9576 ) Co-authored-by: Richard Lyons <frob@cloudstaff.com>	2025-03-11 23:42:53 -07:00
Michael	ad4e0bf3be	Adding Gemma 3 to readme (#9671 )	2025-03-12 07:39:25 +01:00
Michael Yang	aee28501b5	Merge pull request #9661 from ollama/gemma engine: add gemma support	2025-03-11 15:07:50 -07:00
jmorganca	83f0ec8269	all: address linter errors	2025-03-11 14:49:20 -07:00
jmorganca	c6b6938b3a	kvcache: fix tests by adding AvgPool2D stub	2025-03-11 14:49:20 -07:00
jmorganca	fb4664fcec	model: add more spm tokenizer tests	2025-03-11 14:49:20 -07:00
jmorganca	20e3593863	model: validate left and right pairs before merging them	2025-03-11 14:49:20 -07:00
Michael Yang	63a394068c	use 2d pooling	2025-03-11 14:49:20 -07:00
Daniel Hiltgen	ab39e08eb9	llm: auto detect models that require Ollama Engine (#1 )	2025-03-11 14:49:20 -07:00
jmorganca	11bfa62796	add trailing \n\n after <end_of_image> to match reference implementation	2025-03-11 14:49:20 -07:00
jmorganca	f63e62e546	reduce kernel size, add TODO for loading from config	2025-03-11 14:49:20 -07:00
jmorganca	65b0f329d1	Revert "Allow models to force a new batch" This reverts commit c7eae586b899083acebcd9b3847b89ea78c2850c.	2025-03-11 14:49:20 -07:00
Jesse Gross	06007c0a18	Allow models to force a new batch This is useful for a few things: - Work around bugs, such as having 2 images in one batch - Keep the image in a single batch for fully connected attention - Improve performance by not evaluating embeddings multiple times	2025-03-11 14:49:20 -07:00
Jesse Gross	a8e83a7654	Disable causal attention based on batch index Currently we are using positions, which are relative to a sequence and may not be unique.	2025-03-11 14:49:20 -07:00
Jesse Gross	475005504e	Restrict Gemma to a single image per request	2025-03-11 14:49:20 -07:00
Jesse Gross	2c40c4d35e	Fix follow up images and images split across batches	2025-03-11 14:49:19 -07:00
Michael Yang	e95278932b	use non-causal mask only for image positions	2025-03-11 14:49:19 -07:00
Michael Yang	9d2a20a763	use non-causal mask for inputs with images	2025-03-11 14:49:19 -07:00
Patrick Devine	2e54d72fc3	fix gemma3 1b conversion	2025-03-11 14:49:19 -07:00
Michael Yang	6b32a2d549	compat with upstream gguf	2025-03-11 14:49:19 -07:00
Michael Yang	c5cbe4fc2a	fallback to cpu	2025-03-11 14:49:19 -07:00
Michael Yang	f888912870	fix vision encoder	2025-03-11 14:49:19 -07:00
Michael Yang	9e4642e9b3	ollama debug tensor	2025-03-11 14:49:19 -07:00
Michael Yang	6b0486c216	duplicate token_embd to output	2025-03-11 14:49:19 -07:00
Michael Yang	d368c039f0	skip repacking vision tensors	2025-03-11 14:49:19 -07:00
Patrick Devine	9b54267e69	fix configs	2025-03-11 14:49:19 -07:00
Michael Yang	46bb0169c4	update model	2025-03-11 14:49:19 -07:00
Michael Yang	8934324b72	use fast attention	2025-03-11 14:49:18 -07:00
Jesse Gross	0e886595bf	Fix tests and drift from main	2025-03-11 14:49:18 -07:00
Patrick Devine	c62861f4fa	fix conversion	2025-03-11 14:49:18 -07:00
Michael Yang	0df1800436	set non-causal attention	2025-03-11 14:49:18 -07:00
Patrick Devine	631fecc6d9	temporary work around for converting spm	2025-03-11 14:49:18 -07:00
Jesse Gross	4346c2409d	fix drift from main	2025-03-11 14:49:18 -07:00
Michael Yang	4b037a97dc	add gemma vision encoder	2025-03-11 14:49:17 -07:00
Patrick Devine	5f74d1fd47	gemma2 impl	2025-03-11 14:35:08 -07:00
Daniel Hiltgen	4dcf80167a	Build release for windows with local script (#9636 )	2025-03-11 08:34:20 -07:00
Michael Yang	26a26998fb	Merge pull request #9590 from ollama/mxyng/dump-pad fix: pad tensor item if ge zero	2025-03-10 16:34:55 -07:00
Michael Yang	9926eae015	fix: pad tensor item if ge zero this produces a nicer output since both positive and negative values produces the same width	2025-03-10 16:18:12 -07:00
Vincent Koc	8585b7b151	docs: add opik to observability integrations (#9626 )	2025-03-10 16:15:10 -07:00
Parth Sareen	7e34f4fbfa	sample: add numerical stability to temperature/softmax transform (#9631 )	2025-03-10 14:43:53 -07:00
Michael Yang	fe776293f7	Merge pull request #9569 from dwt/patch-1 Better WantedBy declaration	2025-03-10 14:09:37 -07:00
‮rekcäH nitraM‮	25248f4bd5	Better WantedBy declaration The problem with default.target is that it always points to the target that is currently started. So if you boot into single user mode or the rescue mode still Ollama tries to start. I noticed this because either tried (and failed) to start all the time during a system update, where Ollama definitely is not wanted.	2025-03-07 10:26:31 +01:00