model: support for mistral-small in the ollama runner

Mistral is a popular research lab making open source models. This updates the forward pass of llama architecture models to support both llama models and mistral models by accounting for additional metadata present in mistral models, and finding the correct dimensions for the output projection.
2025-03-14 16:56:39 -07:00
78 changed files with 1145 additions and 3327 deletions
--- a/CMakePresets.json
+++ b/CMakePresets.json
@@ -56,7 +56,7 @@
      "name": "ROCm 6",
      "inherits": [ "ROCm" ],
      "cacheVariables": {
-        "AMDGPU_TARGETS": "gfx900;gfx940;gfx941;gfx942;gfx1010;gfx1012;gfx1030;gfx1100;gfx1101;gfx1102;gfx1151;gfx906:xnack-;gfx908:xnack-;gfx90a:xnack+;gfx90a:xnack-"
+        "AMDGPU_TARGETS": "gfx900;gfx940;gfx941;gfx942;gfx1010;gfx1012;gfx1030;gfx1100;gfx1101;gfx1102;gfx906:xnack-;gfx908:xnack-;gfx90a:xnack+;gfx90a:xnack-"
      }
    }
  ],
--- a/README.md
+++ b/README.md
@@ -392,8 +392,6 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [1Panel](https://github.com/1Panel-dev/1Panel/) (Web-based Linux Server Management Tool)
 - [AstrBot](https://github.com/Soulter/AstrBot/) (User-friendly LLM-based multi-platform chatbot with a WebUI, supporting RAG, LLM agents, and plugins integration)
 - [Reins](https://github.com/ibrahimcetin/reins) (Easily tweak parameters, customize system prompts per chat, and enhance your AI experiments with reasoning model support.)
- [Ellama](https://github.com/zeozeozeo/ellama) (Friendly native app to chat with an Ollama instance)
- [screenpipe](https://github.com/mediar-ai/screenpipe) Build agents powered by your screen history

 ### Cloud

@@ -512,7 +510,6 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Ollama for Zig](https://github.com/dravenk/ollama-zig)
 - [Abso](https://github.com/lunary-ai/abso) (OpenAI-compatible TypeScript SDK for any LLM provider)
 - [Nichey](https://github.com/goodreasonai/nichey) is a Python package for generating custom wikis for your research topic
- [Ollama for D](https://github.com/kassane/ollama-d)

 ### Mobile

--- a/benchmark/server_benchmark_test.go
+++ b/benchmark/server_benchmark_test.go
@@ -1,178 +0,0 @@
-package benchmark
-
-import (
-	"context"
-	"flag"
-	"fmt"
-	"testing"
-	"time"
-
-	"github.com/ollama/ollama/api"
-)
-
-// Command line flags
-var modelFlag string
-
-func init() {
-	flag.StringVar(&modelFlag, "m", "", "Name of the model to benchmark")
-	flag.Lookup("m").DefValue = "model"
-}
-
-// modelName returns the model name from flags, failing the test if not set
-func modelName(b *testing.B) string {
-	if modelFlag == "" {
-		b.Fatal("Error: -m flag is required for benchmark tests")
-	}
-	return modelFlag
-}
-
-type TestCase struct {
-	name      string
-	prompt    string
-	maxTokens int
-}
-
-// runGenerateBenchmark contains the common generate and metrics logic
-func runGenerateBenchmark(b *testing.B, ctx context.Context, client *api.Client, req *api.GenerateRequest) {
-	start := time.Now()
-	var ttft time.Duration
-	var metrics api.Metrics
-
-	err := client.Generate(ctx, req, func(resp api.GenerateResponse) error {
-		if ttft == 0 && resp.Response != "" {
-			ttft = time.Since(start)
-		}
-		if resp.Done {
-			metrics = resp.Metrics
-		}
-		return nil
-	})
-
-	// Report custom metrics as part of the benchmark results
-	b.ReportMetric(float64(ttft.Milliseconds()), "ttft_ms")
-	b.ReportMetric(float64(metrics.LoadDuration.Milliseconds()), "load_ms")
-
-	// Token throughput metrics
-	promptThroughput := float64(metrics.PromptEvalCount) / metrics.PromptEvalDuration.Seconds()
-	genThroughput := float64(metrics.EvalCount) / metrics.EvalDuration.Seconds()
-	b.ReportMetric(promptThroughput, "prompt_tok/s")
-	b.ReportMetric(genThroughput, "gen_tok/s")
-
-	// Token counts
-	b.ReportMetric(float64(metrics.PromptEvalCount), "prompt_tokens")
-	b.ReportMetric(float64(metrics.EvalCount), "gen_tokens")
-	if err != nil {
-		b.Fatal(err)
-	}
-}
-
-// BenchmarkColdStart runs benchmarks with model loading from cold state
-func BenchmarkColdStart(b *testing.B) {
-	client := setup(b)
-	tests := []TestCase{
-		{"short_prompt", "Write a long story", 100},
-		{"medium_prompt", "Write a detailed economic analysis", 500},
-		{"long_prompt", "Write a comprehensive AI research paper", 1000},
-	}
-	m := modelName(b)
-
-	for _, tt := range tests {
-		b.Run(fmt.Sprintf("%s/cold/%s", m, tt.name), func(b *testing.B) {
-			ctx := context.Background()
-
-			// Set number of tokens as our throughput metric
-			b.SetBytes(int64(tt.maxTokens))
-
-			for b.Loop() {
-				b.StopTimer()
-				// Ensure model is unloaded before each iteration
-				unload(client, m, b)
-				b.StartTimer()
-
-				req := &api.GenerateRequest{
-					Model:   m,
-					Prompt:  tt.prompt,
-					Options: map[string]interface{}{"num_predict": tt.maxTokens, "temperature": 0.1},
-				}
-
-				runGenerateBenchmark(b, ctx, client, req)
-			}
-		})
-	}
-}
-
-// BenchmarkWarmStart runs benchmarks with pre-loaded model
-func BenchmarkWarmStart(b *testing.B) {
-	client := setup(b)
-	tests := []TestCase{
-		{"short_prompt", "Write a long story", 100},
-		{"medium_prompt", "Write a detailed economic analysis", 500},
-		{"long_prompt", "Write a comprehensive AI research paper", 1000},
-	}
-	m := modelName(b)
-
-	for _, tt := range tests {
-		b.Run(fmt.Sprintf("%s/warm/%s", m, tt.name), func(b *testing.B) {
-			ctx := context.Background()
-
-			// Pre-warm the model
-			warmup(client, m, tt.prompt, b)
-
-			// Set number of tokens as our throughput metric
-			b.SetBytes(int64(tt.maxTokens))
-
-			for b.Loop() {
-				req := &api.GenerateRequest{
-					Model:   m,
-					Prompt:  tt.prompt,
-					Options: map[string]any{"num_predict": tt.maxTokens, "temperature": 0.1},
-				}
-
-				runGenerateBenchmark(b, ctx, client, req)
-			}
-		})
-	}
-}
-
-// setup verifies server and model availability
-func setup(b *testing.B) *api.Client {
-	client, err := api.ClientFromEnvironment()
-	if err != nil {
-		b.Fatal(err)
-	}
-	if _, err := client.Show(context.Background(), &api.ShowRequest{Model: modelName(b)}); err != nil {
-		b.Fatalf("Model unavailable: %v", err)
-	}
-
-	return client
-}
-
-// warmup ensures the model is loaded and warmed up
-func warmup(client *api.Client, model string, prompt string, b *testing.B) {
-	for range 3 {
-		err := client.Generate(
-			context.Background(),
-			&api.GenerateRequest{
-				Model:   model,
-				Prompt:  prompt,
-				Options: map[string]interface{}{"num_predict": 50, "temperature": 0.1},
-			},
-			func(api.GenerateResponse) error { return nil },
-		)
-		if err != nil {
-			b.Logf("Error during model warm-up: %v", err)
-		}
-	}
-}
-
-// unload forces model unloading using KeepAlive: 0 parameter
-func unload(client *api.Client, model string, b *testing.B) {
-	req := &api.GenerateRequest{
-		Model:     model,
-		KeepAlive: &api.Duration{Duration: 0},
-	}
-	if err := client.Generate(context.Background(), req, func(api.GenerateResponse) error { return nil }); err != nil {
-		b.Logf("Unload error: %v", err)
-	}
-	time.Sleep(1 * time.Second)
-}
--- a/cmd/cmd.go
+++ b/cmd/cmd.go
@@ -703,8 +703,6 @@ func showInfo(resp *api.ShowResponse, verbose bool, w io.Writer) error {
 			for _, k := range keys {
 				var v string
 				switch vData := resp.ModelInfo[k].(type) {
-				case bool:
-					v = fmt.Sprintf("%t", vData)
 				case string:
 					v = vData
 				case float64:
--- a/cmd/cmd_test.go
+++ b/cmd/cmd_test.go
@@ -87,8 +87,6 @@ func TestShowInfo(t *testing.T) {
 			ModelInfo: map[string]any{
 				"general.architecture":    "test",
 				"general.parameter_count": float64(8_000_000_000),
-				"some.true_bool":          true,
-				"some.false_bool":         false,
 				"test.context_length":     float64(1000),
 				"test.embedding_length":   float64(11434),
 			},
@@ -113,8 +111,6 @@ func TestShowInfo(t *testing.T) {
  Metadata
    general.architecture       test     
    general.parameter_count    8e+09    
-    some.false_bool            false    
-    some.true_bool             true     
    test.context_length        1000     
    test.embedding_length      11434    

@@ -761,132 +757,3 @@ func TestCreateHandler(t *testing.T) {
 		})
 	}
 }
-
-func TestNewCreateRequest(t *testing.T) {
-	tests := []struct {
-		name     string
-		from     string
-		opts     runOptions
-		expected *api.CreateRequest
-	}{
-		{
-			"basic test",
-			"newmodel",
-			runOptions{
-				Model:       "mymodel",
-				ParentModel: "",
-				Prompt:      "You are a fun AI agent",
-				Messages:    []api.Message{},
-				WordWrap:    true,
-			},
-			&api.CreateRequest{
-				From:  "mymodel",
-				Model: "newmodel",
-			},
-		},
-		{
-			"parent model test",
-			"newmodel",
-			runOptions{
-				Model:       "mymodel",
-				ParentModel: "parentmodel",
-				Messages:    []api.Message{},
-				WordWrap:    true,
-			},
-			&api.CreateRequest{
-				From:  "parentmodel",
-				Model: "newmodel",
-			},
-		},
-		{
-			"parent model as filepath test",
-			"newmodel",
-			runOptions{
-				Model:       "mymodel",
-				ParentModel: "/some/file/like/etc/passwd",
-				Messages:    []api.Message{},
-				WordWrap:    true,
-			},
-			&api.CreateRequest{
-				From:  "mymodel",
-				Model: "newmodel",
-			},
-		},
-		{
-			"parent model as windows filepath test",
-			"newmodel",
-			runOptions{
-				Model:       "mymodel",
-				ParentModel: "D:\\some\\file\\like\\etc\\passwd",
-				Messages:    []api.Message{},
-				WordWrap:    true,
-			},
-			&api.CreateRequest{
-				From:  "mymodel",
-				Model: "newmodel",
-			},
-		},
-		{
-			"options test",
-			"newmodel",
-			runOptions{
-				Model:       "mymodel",
-				ParentModel: "parentmodel",
-				Options: map[string]any{
-					"temperature": 1.0,
-				},
-			},
-			&api.CreateRequest{
-				From:  "parentmodel",
-				Model: "newmodel",
-				Parameters: map[string]any{
-					"temperature": 1.0,
-				},
-			},
-		},
-		{
-			"messages test",
-			"newmodel",
-			runOptions{
-				Model:       "mymodel",
-				ParentModel: "parentmodel",
-				System:      "You are a fun AI agent",
-				Messages: []api.Message{
-					{
-						Role:    "user",
-						Content: "hello there!",
-					},
-					{
-						Role:    "assistant",
-						Content: "hello to you!",
-					},
-				},
-				WordWrap: true,
-			},
-			&api.CreateRequest{
-				From:   "parentmodel",
-				Model:  "newmodel",
-				System: "You are a fun AI agent",
-				Messages: []api.Message{
-					{
-						Role:    "user",
-						Content: "hello there!",
-					},
-					{
-						Role:    "assistant",
-						Content: "hello to you!",
-					},
-				},
-			},
-		},
-	}
-
-	for _, tt := range tests {
-		t.Run(tt.name, func(t *testing.T) {
-			actual := NewCreateRequest(tt.from, tt.opts)
-			if !cmp.Equal(actual, tt.expected) {
-				t.Errorf("expected output %#v, got %#v", tt.expected, actual)
-			}
-		})
-	}
-}
--- a/cmd/interactive.go
+++ b/cmd/interactive.go
@@ -18,7 +18,6 @@ import (
 	"github.com/ollama/ollama/envconfig"
 	"github.com/ollama/ollama/readline"
 	"github.com/ollama/ollama/types/errtypes"
-	"github.com/ollama/ollama/types/model"
 )

 type MultilineState int
@@ -460,16 +459,9 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 }

 func NewCreateRequest(name string, opts runOptions) *api.CreateRequest {
-	parentModel := opts.ParentModel
-
-	modelName := model.ParseName(parentModel)
-	if !modelName.IsValid() {
-		parentModel = ""
-	}
-
 	req := &api.CreateRequest{
-		Model: name,
-		From:  cmp.Or(parentModel, opts.Model),
+		Name: name,
+		From: cmp.Or(opts.ParentModel, opts.Model),
 	}

 	if opts.System != "" {
--- a/convert/convert.go
+++ b/convert/convert.go
@@ -201,7 +201,7 @@ func ConvertModel(fsys fs.FS, ws io.WriteSeeker) error {
 	case "CohereForCausalLM":
 		conv = &commandrModel{}
 	default:
-		return fmt.Errorf("unsupported architecture %q", p.Architectures[0])
+		return errors.New("unsupported architecture")
 	}

 	if err := json.Unmarshal(bts, conv); err != nil {
--- a/docs/api.md
+++ b/docs/api.md
@@ -558,10 +558,6 @@ Final response:
 {
  "model": "llama3.2",
  "created_at": "2023-08-04T19:22:45.499127Z",
-  "message": {
-    "role": "assistant",
-    "content": ""
-  },
  "done": true,
  "total_duration": 4883583458,
  "load_duration": 1334875,
--- a/docs/benchmark.md
+++ b/docs/benchmark.md
@@ -1,59 +0,0 @@
-# Benchmark
-
-Go benchmark tests that measure end-to-end performance of a running Ollama server. Run these tests to evaluate model inference performance on your hardware and measure the impact of code changes.
-
-## When to use
-
-Run these benchmarks when:
- Making changes to the model inference engine
- Modifying model loading/unloading logic
- Changing prompt processing or token generation code
- Implementing a new model architecture
- Testing performance across different hardware setups
-
-## Prerequisites
- Ollama server running locally with `ollama serve` on `127.0.0.1:11434`
-## Usage and Examples
-
->[!NOTE]
->All commands must be run from the root directory of the Ollama project.
-
-Basic syntax:
-```bash
-go test -bench=. ./benchmark/... -m $MODEL_NAME
-```
-
-Required flags:
- `-bench=.`: Run all benchmarks
- `-m`: Model name to benchmark
-
-Optional flags:
- `-count N`: Number of times to run the benchmark (useful for statistical analysis)
- `-timeout T`: Maximum time for the benchmark to run (e.g. "10m" for 10 minutes)
-
-Common usage patterns:
-
-Single benchmark run with a model specified:
-```bash
-go test -bench=. ./benchmark/... -m llama3.3
-```
-
-## Output metrics
-
-The benchmark reports several key metrics:
-
- `gen_tok/s`: Generated tokens per second
- `prompt_tok/s`: Prompt processing tokens per second
- `ttft_ms`: Time to first token in milliseconds
- `load_ms`: Model load time in milliseconds
- `gen_tokens`: Total tokens generated
- `prompt_tokens`: Total prompt tokens processed
-
-Each benchmark runs two scenarios:
- Cold start: Model is loaded from disk for each test
- Warm start: Model is pre-loaded in memory
-
-Three prompt lengths are tested for each scenario:
- Short prompt (100 tokens)
- Medium prompt (500 tokens)
- Long prompt (1000 tokens)
--- a/docs/troubleshooting.md
+++ b/docs/troubleshooting.md
@@ -9,7 +9,7 @@ cat ~/.ollama/logs/server.log
 On **Linux** systems with systemd, the logs can be found with this command:

 ```shell
-journalctl -u ollama --no-pager --follow --pager-end 
+journalctl -u ollama --no-pager
 ```

 When you run Ollama in a **container**, the logs go to stdout/stderr in the container:
--- a/grammar/bench_test.go
+++ b/grammar/bench_test.go
@@ -1,22 +0,0 @@
-//go:build go1.24
-
-package grammar
-
-import "testing"
-
-func BenchmarkFromSchema(b *testing.B) {
-	for tt := range testCases(b) {
-		b.Run("", func(b *testing.B) {
-			s := []byte(tt.schema)
-
-			b.ReportAllocs()
-			for b.Loop() {
-				_, err := FromSchema(nil, s)
-				if err != nil {
-					b.Fatalf("GrammarFromSchema: %v", err)
-				}
-			}
-		})
-		return
-	}
-}
--- a/grammar/grammar.go
+++ b/grammar/grammar.go
@@ -1,227 +0,0 @@
-package grammar
-
-import (
-	"bytes"
-	"encoding/json"
-	"fmt"
-	"iter"
-	"strconv"
-
-	"github.com/ollama/ollama/grammar/jsonschema"
-)
-
-const jsonTerms = `
-# Unicode
-#
-# Unicode characters can be specified directly in the grammar, for example
-# hiragana ::= [ぁ-ゟ], or with escapes: 8-bit (\xXX), 16-bit (\uXXXX) or 32-bit
-# (\UXXXXXXXX).
-unicode ::= \x{hex}{2} | \u{hex}{4} | \U{hex}{8}
-
-# JSON grammar from RFC 7159
-null    ::= "null"
-object  ::= "{" (kv ("," kv)*)? "}"
-array   ::= "[" (value ("," value)*)? "]"
-kv      ::= string ":" value
-integer ::= "0" | [1-9] [0-9]*
-number  ::= "-"? integer frac? exp?
-frac    ::= "." [0-9]+
-exp     ::= ("e" | "E") ("+" | "-") [0-9]+
-string  ::= "\"" char* "\""
-escape  ::= ["/" | "b" | "f" | "n" | "r" | "t" | unicode]
-char    ::= [^"\\] | escape
-space   ::= (" " | "\t" | "\n" | "\r")*
-hex     ::= [0-9] | [a-f] | [A-F]
-boolean ::= "true" | "false"
-value   ::= object | array | string | number | boolean | "null"
-
-# User-defined
-`
-
-// FromSchema generates a grammar from a JSON schema.
-func FromSchema(buf []byte, jsonSchema []byte) ([]byte, error) {
-	var s *jsonschema.Schema
-	if err := json.Unmarshal(jsonSchema, &s); err != nil {
-		return nil, err
-	}
-
-	var g builder
-
-	// "root" is the only rule that is guaranteed to exist, so we start
-	// with its length for padding, and then adjust it as we go.
-	g.pad = len("root")
-	for id := range dependencies("root", s) {
-		g.pad = max(g.pad, len(id))
-	}
-
-	g.b.WriteString(jsonTerms)
-
-	ids := make(map[*jsonschema.Schema]string)
-	for id, s := range dependencies("root", s) {
-		ids[s] = id
-		g.define(id)
-		if err := fromSchema(&g, ids, s); err != nil {
-			return nil, err
-		}
-	}
-	g.define("root")
-	if err := fromSchema(&g, ids, s); err != nil {
-		return nil, err
-	}
-	g.define("") // finalize the last rule
-	return g.b.Bytes(), nil
-}
-
-func fromSchema(g *builder, ids map[*jsonschema.Schema]string, s *jsonschema.Schema) error {
-	switch typ := s.EffectiveType(); typ {
-	case "array":
-		if len(s.PrefixItems) == 0 && s.Items == nil {
-			g.u("array")
-		} else {
-			g.q("[")
-			for i, s := range s.PrefixItems {
-				if i > 0 {
-					g.q(",")
-				}
-				g.u(ids[s])
-			}
-			if s.Items != nil {
-				g.u("(")
-				if len(s.PrefixItems) > 0 {
-					g.q(",")
-				}
-				g.u(ids[s.Items])
-				g.u(")*")
-			}
-			g.q("]")
-		}
-	case "object":
-		if len(s.Properties) == 0 {
-			g.u("object")
-		} else {
-			g.q("{")
-			for i, p := range s.Properties {
-				name := ids[p]
-				if i > 0 {
-					g.q(",")
-				}
-				g.q(p.Name)
-				g.q(":")
-				g.u(name)
-			}
-			g.q("}")
-		}
-	case "number":
-		buildConstrainedNumber(g, s)
-	case "string":
-		if len(s.Enum) == 0 {
-			g.u("string")
-		} else {
-			g.u("(")
-			for i, e := range s.Enum {
-				if i > 0 {
-					g.q("|")
-				}
-				g.q(string(e))
-			}
-			g.u(")")
-		}
-	case "boolean", "value", "null", "integer":
-		g.u(typ)
-	default:
-		return fmt.Errorf("%s: unsupported type %q", s.Name, typ)
-	}
-	return nil
-}
-
-// dependencies returns a sequence of all child dependencies of the schema in
-// post-order.
-//
-// The first value is the id/pointer to the dependency, and the second value
-// is the schema.
-func dependencies(id string, s *jsonschema.Schema) iter.Seq2[string, *jsonschema.Schema] {
-	return func(yield func(string, *jsonschema.Schema) bool) {
-		for i, p := range s.Properties {
-			id := fmt.Sprintf("%s_%d", id, i)
-			for did, d := range dependencies(id, p) {
-				if !yield(did, d) {
-					return
-				}
-			}
-			if !yield(id, p) {
-				return
-			}
-		}
-		for i, p := range s.PrefixItems {
-			id := fmt.Sprintf("tuple_%d", i)
-			for did, d := range dependencies(id, p) {
-				id := fmt.Sprintf("%s_%s", id, did)
-				if !yield(id, d) {
-					return
-				}
-			}
-			if !yield(id, p) {
-				return
-			}
-		}
-		if s.Items != nil {
-			id := fmt.Sprintf("%s_tuple_%d", id, len(s.PrefixItems))
-			for did, d := range dependencies(id, s.Items) {
-				if !yield(did, d) {
-					return
-				}
-			}
-			if !yield(id, s.Items) {
-				return
-			}
-		}
-	}
-}
-
-type builder struct {
-	b     bytes.Buffer
-	pad   int
-	rules int
-	items int
-}
-
-// define terminates the current rule, if any, and then either starts a new
-// rule or does nothing else if the name is empty.
-func (b *builder) define(name string) {
-	if b.rules > 0 {
-		b.b.WriteString(";\n")
-	}
-	if name == "" {
-		return
-	}
-	fmt.Fprintf(&b.b, "% -*s", b.pad, name)
-	b.b.WriteString(" ::=")
-	b.rules++
-	b.items = 0
-}
-
-// quote appends a terminal to the current rule.
-func (b *builder) q(s string) {
-	if b.items > 0 {
-		b.b.WriteString(" ")
-	}
-	b.b.WriteString(" ")
-	b.b.WriteString(strconv.Quote(s))
-}
-
-// u appends a non-terminal to the current rule.
-func (b *builder) u(s string) {
-	if b.items > 0 {
-		b.b.WriteString(" ")
-	}
-	b.b.WriteString(" ")
-	b.b.WriteString(s)
-}
-
-func buildConstrainedNumber(b *builder, s *jsonschema.Schema) {
-	if s.Minimum == 0 && s.Maximum == 0 {
-		b.u("TODO")
-	} else {
-		b.u("number")
-	}
-}
--- a/grammar/grammar_test.go
+++ b/grammar/grammar_test.go
@@ -1,75 +0,0 @@
-package grammar
-
-import (
-	"bufio"
-	"cmp"
-	"iter"
-	"strings"
-	"testing"
-
-	_ "embed"
-
-	"github.com/ollama/ollama/grammar/internal/diff"
-)
-
-func TestFromSchema(t *testing.T) {
-	for tt := range testCases(t) {
-		t.Run(tt.name, func(t *testing.T) {
-			g, err := FromSchema(nil, []byte(tt.schema))
-			if err != nil {
-				t.Fatalf("FromSchema: %v", err)
-			}
-			got := string(g)
-			got = strings.TrimPrefix(got, jsonTerms)
-			if got != tt.want {
-				t.Logf("schema:\n%s", tt.schema)
-				t.Fatal(string(diff.Diff("got", []byte(got), "want", []byte(tt.want))))
-			}
-		})
-	}
-}
-
-type testCase struct {
-	name   string
-	schema string
-	want   string
-}
-
-//go:embed testdata/schemas.txt
-var tests string
-
-func testCases(t testing.TB) iter.Seq[testCase] {
-	t.Helper()
-	return func(yield func(testCase) bool) {
-		t.Helper()
-		sc := bufio.NewScanner(strings.NewReader(tests))
-		name := ""
-		for sc.Scan() {
-			line := strings.TrimSpace(sc.Text())
-			if line == "" {
-				name = ""
-				continue
-			}
-			if line[0] == '#' {
-				name = cmp.Or(name, strings.TrimSpace(line[1:]))
-				continue
-			}
-			s := sc.Text()
-			g := ""
-			for sc.Scan() {
-				line = strings.TrimSpace(sc.Text())
-				if line == "" || line[0] == '#' {
-					break
-				}
-				g += sc.Text() + "\n"
-			}
-			if !yield(testCase{name, s, g}) {
-				return
-			}
-			name = strings.TrimSpace(strings.TrimPrefix(line, "#"))
-		}
-		if err := sc.Err(); err != nil {
-			t.Fatalf("error reading tests: %v", err)
-		}
-	}
-}
--- a/grammar/internal/diff/diff.go
+++ b/grammar/internal/diff/diff.go
@@ -1,261 +0,0 @@
-// Copyright 2022 The Go Authors. All rights reserved.
-// Use of this source code is governed by a BSD-style
-// license that can be found in the LICENSE file.
-
-package diff
-
-import (
-	"bytes"
-	"fmt"
-	"sort"
-	"strings"
-)
-
-// A pair is a pair of values tracked for both the x and y side of a diff.
-// It is typically a pair of line indexes.
-type pair struct{ x, y int }
-
-// Diff returns an anchored diff of the two texts old and new
-// in the “unified diff” format. If old and new are identical,
-// Diff returns a nil slice (no output).
-//
-// Unix diff implementations typically look for a diff with
-// the smallest number of lines inserted and removed,
-// which can in the worst case take time quadratic in the
-// number of lines in the texts. As a result, many implementations
-// either can be made to run for a long time or cut off the search
-// after a predetermined amount of work.
-//
-// In contrast, this implementation looks for a diff with the
-// smallest number of “unique” lines inserted and removed,
-// where unique means a line that appears just once in both old and new.
-// We call this an “anchored diff” because the unique lines anchor
-// the chosen matching regions. An anchored diff is usually clearer
-// than a standard diff, because the algorithm does not try to
-// reuse unrelated blank lines or closing braces.
-// The algorithm also guarantees to run in O(n log n) time
-// instead of the standard O(n²) time.
-//
-// Some systems call this approach a “patience diff,” named for
-// the “patience sorting” algorithm, itself named for a solitaire card game.
-// We avoid that name for two reasons. First, the name has been used
-// for a few different variants of the algorithm, so it is imprecise.
-// Second, the name is frequently interpreted as meaning that you have
-// to wait longer (to be patient) for the diff, meaning that it is a slower algorithm,
-// when in fact the algorithm is faster than the standard one.
-func Diff(oldName string, old []byte, newName string, new []byte) []byte {
-	if bytes.Equal(old, new) {
-		return nil
-	}
-	x := lines(old)
-	y := lines(new)
-
-	// Print diff header.
-	var out bytes.Buffer
-	fmt.Fprintf(&out, "diff %s %s\n", oldName, newName)
-	fmt.Fprintf(&out, "--- %s\n", oldName)
-	fmt.Fprintf(&out, "+++ %s\n", newName)
-
-	// Loop over matches to consider,
-	// expanding each match to include surrounding lines,
-	// and then printing diff chunks.
-	// To avoid setup/teardown cases outside the loop,
-	// tgs returns a leading {0,0} and trailing {len(x), len(y)} pair
-	// in the sequence of matches.
-	var (
-		done  pair     // printed up to x[:done.x] and y[:done.y]
-		chunk pair     // start lines of current chunk
-		count pair     // number of lines from each side in current chunk
-		ctext []string // lines for current chunk
-	)
-	for _, m := range tgs(x, y) {
-		if m.x < done.x {
-			// Already handled scanning forward from earlier match.
-			continue
-		}
-
-		// Expand matching lines as far as possible,
-		// establishing that x[start.x:end.x] == y[start.y:end.y].
-		// Note that on the first (or last) iteration we may (or definitely do)
-		// have an empty match: start.x==end.x and start.y==end.y.
-		start := m
-		for start.x > done.x && start.y > done.y && x[start.x-1] == y[start.y-1] {
-			start.x--
-			start.y--
-		}
-		end := m
-		for end.x < len(x) && end.y < len(y) && x[end.x] == y[end.y] {
-			end.x++
-			end.y++
-		}
-
-		// Emit the mismatched lines before start into this chunk.
-		// (No effect on first sentinel iteration, when start = {0,0}.)
-		for _, s := range x[done.x:start.x] {
-			ctext = append(ctext, "-"+s)
-			count.x++
-		}
-		for _, s := range y[done.y:start.y] {
-			ctext = append(ctext, "+"+s)
-			count.y++
-		}
-
-		// If we're not at EOF and have too few common lines,
-		// the chunk includes all the common lines and continues.
-		const C = 3 // number of context lines
-		if (end.x < len(x) || end.y < len(y)) &&
-			(end.x-start.x < C || (len(ctext) > 0 && end.x-start.x < 2*C)) {
-			for _, s := range x[start.x:end.x] {
-				ctext = append(ctext, " "+s)
-				count.x++
-				count.y++
-			}
-			done = end
-			continue
-		}
-
-		// End chunk with common lines for context.
-		if len(ctext) > 0 {
-			n := end.x - start.x
-			if n > C {
-				n = C
-			}
-			for _, s := range x[start.x : start.x+n] {
-				ctext = append(ctext, " "+s)
-				count.x++
-				count.y++
-			}
-			done = pair{start.x + n, start.y + n}
-
-			// Format and emit chunk.
-			// Convert line numbers to 1-indexed.
-			// Special case: empty file shows up as 0,0 not 1,0.
-			if count.x > 0 {
-				chunk.x++
-			}
-			if count.y > 0 {
-				chunk.y++
-			}
-			fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", chunk.x, count.x, chunk.y, count.y)
-			for _, s := range ctext {
-				out.WriteString(s)
-			}
-			count.x = 0
-			count.y = 0
-			ctext = ctext[:0]
-		}
-
-		// If we reached EOF, we're done.
-		if end.x >= len(x) && end.y >= len(y) {
-			break
-		}
-
-		// Otherwise start a new chunk.
-		chunk = pair{end.x - C, end.y - C}
-		for _, s := range x[chunk.x:end.x] {
-			ctext = append(ctext, " "+s)
-			count.x++
-			count.y++
-		}
-		done = end
-	}
-
-	return out.Bytes()
-}
-
-// lines returns the lines in the file x, including newlines.
-// If the file does not end in a newline, one is supplied
-// along with a warning about the missing newline.
-func lines(x []byte) []string {
-	l := strings.SplitAfter(string(x), "\n")
-	if l[len(l)-1] == "" {
-		l = l[:len(l)-1]
-	} else {
-		// Treat last line as having a message about the missing newline attached,
-		// using the same text as BSD/GNU diff (including the leading backslash).
-		l[len(l)-1] += "\n\\ No newline at end of file\n"
-	}
-	return l
-}
-
-// tgs returns the pairs of indexes of the longest common subsequence
-// of unique lines in x and y, where a unique line is one that appears
-// once in x and once in y.
-//
-// The longest common subsequence algorithm is as described in
-// Thomas G. Szymanski, “A Special Case of the Maximal Common
-// Subsequence Problem,” Princeton TR #170 (January 1975),
-// available at https://research.swtch.com/tgs170.pdf.
-func tgs(x, y []string) []pair {
-	// Count the number of times each string appears in a and b.
-	// We only care about 0, 1, many, counted as 0, -1, -2
-	// for the x side and 0, -4, -8 for the y side.
-	// Using negative numbers now lets us distinguish positive line numbers later.
-	m := make(map[string]int)
-	for _, s := range x {
-		if c := m[s]; c > -2 {
-			m[s] = c - 1
-		}
-	}
-	for _, s := range y {
-		if c := m[s]; c > -8 {
-			m[s] = c - 4
-		}
-	}
-
-	// Now unique strings can be identified by m[s] = -1+-4.
-	//
-	// Gather the indexes of those strings in x and y, building:
-	//	xi[i] = increasing indexes of unique strings in x.
-	//	yi[i] = increasing indexes of unique strings in y.
-	//	inv[i] = index j such that x[xi[i]] = y[yi[j]].
-	var xi, yi, inv []int
-	for i, s := range y {
-		if m[s] == -1+-4 {
-			m[s] = len(yi)
-			yi = append(yi, i)
-		}
-	}
-	for i, s := range x {
-		if j, ok := m[s]; ok && j >= 0 {
-			xi = append(xi, i)
-			inv = append(inv, j)
-		}
-	}
-
-	// Apply Algorithm A from Szymanski's paper.
-	// In those terms, A = J = inv and B = [0, n).
-	// We add sentinel pairs {0,0}, and {len(x),len(y)}
-	// to the returned sequence, to help the processing loop.
-	J := inv
-	n := len(xi)
-	T := make([]int, n)
-	L := make([]int, n)
-	for i := range T {
-		T[i] = n + 1
-	}
-	for i := range n {
-		k := sort.Search(n, func(k int) bool {
-			return T[k] >= J[i]
-		})
-		T[k] = J[i]
-		L[i] = k + 1
-	}
-	k := 0
-	for _, v := range L {
-		if k < v {
-			k = v
-		}
-	}
-	seq := make([]pair, 2+k)
-	seq[1+k] = pair{len(x), len(y)} // sentinel at end
-	lastj := n
-	for i := n - 1; i >= 0; i-- {
-		if L[i] == k && J[i] < lastj {
-			seq[k] = pair{xi[i], yi[J[i]]}
-			k--
-		}
-	}
-	seq[0] = pair{0, 0} // sentinel at start
-	return seq
-}
--- a/grammar/internal/diff/diff_test.go
+++ b/grammar/internal/diff/diff_test.go
@@ -1,44 +0,0 @@
-// Copyright 2022 The Go Authors. All rights reserved.
-// Use of this source code is governed by a BSD-style
-// license that can be found in the LICENSE file.
-
-package diff
-
-import (
-	"bytes"
-	"path/filepath"
-	"testing"
-
-	"golang.org/x/tools/txtar"
-)
-
-func clean(text []byte) []byte {
-	text = bytes.ReplaceAll(text, []byte("$\n"), []byte("\n"))
-	text = bytes.TrimSuffix(text, []byte("^D\n"))
-	return text
-}
-
-func Test(t *testing.T) {
-	files, _ := filepath.Glob("testdata/*.txt")
-	if len(files) == 0 {
-		t.Fatalf("no testdata")
-	}
-
-	for _, file := range files {
-		t.Run(filepath.Base(file), func(t *testing.T) {
-			a, err := txtar.ParseFile(file)
-			if err != nil {
-				t.Fatal(err)
-			}
-			if len(a.Files) != 3 || a.Files[2].Name != "diff" {
-				t.Fatalf("%s: want three files, third named \"diff\"", file)
-			}
-			diffs := Diff(a.Files[0].Name, clean(a.Files[0].Data), a.Files[1].Name, clean(a.Files[1].Data))
-			want := clean(a.Files[2].Data)
-			if !bytes.Equal(diffs, want) {
-				t.Fatalf("%s: have:\n%s\nwant:\n%s\n%s", file,
-					diffs, want, Diff("have", diffs, "want", want))
-			}
-		})
-	}
-}
--- a/grammar/internal/diff/testdata/allnew.txt
+++ b/grammar/internal/diff/testdata/allnew.txt
@@ -1,13 +0,0 @@
-- old --
-- new --
-a
-b
-c
-- diff --
-diff old new
--- old
-+++ new
-@@ -0,0 +1,3 @@
-+a
-+b
-+c
--- a/grammar/internal/diff/testdata/allold.txt
+++ b/grammar/internal/diff/testdata/allold.txt
@@ -1,13 +0,0 @@
-- old --
-a
-b
-c
-- new --
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,3 +0,0 @@
-a
-b
-c
--- a/grammar/internal/diff/testdata/basic.txt
+++ b/grammar/internal/diff/testdata/basic.txt
@@ -1,35 +0,0 @@
-Example from Hunt and McIlroy, “An Algorithm for Differential File Comparison.”
-https://www.cs.dartmouth.edu/~doug/diff.pdf
-
-- old --
-a
-b
-c
-d
-e
-f
-g
-- new --
-w
-a
-b
-x
-y
-z
-e
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,7 +1,7 @@
-+w
- a
- b
-c
-d
-+x
-+y
-+z
- e
-f
-g
--- a/grammar/internal/diff/testdata/dups.txt
+++ b/grammar/internal/diff/testdata/dups.txt
@@ -1,40 +0,0 @@
-- old --
-a
-
-b
-
-c
-
-d
-
-e
-
-f
-- new --
-a
-
-B
-
-C
-
-d
-
-e
-
-f
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,8 +1,8 @@
- a
- $
-b
-
-c
-+B
-+
-+C
- $
- d
- $
--- a/grammar/internal/diff/testdata/end.txt
+++ b/grammar/internal/diff/testdata/end.txt
@@ -1,38 +0,0 @@
-- old --
-1
-2
-3
-4
-5
-6
-7
-eight
-nine
-ten
-eleven
-- new --
-1
-2
-3
-4
-5
-6
-7
-8
-9
-10
-- diff --
-diff old new
--- old
-+++ new
-@@ -5,7 +5,6 @@
- 5
- 6
- 7
-eight
-nine
-ten
-eleven
-+8
-+9
-+10
--- a/grammar/internal/diff/testdata/eof.txt
+++ b/grammar/internal/diff/testdata/eof.txt
@@ -1,9 +0,0 @@
-- old --
-a
-b
-c^D
-- new --
-a
-b
-c^D
-- diff --
--- a/grammar/internal/diff/testdata/eof1.txt
+++ b/grammar/internal/diff/testdata/eof1.txt
@@ -1,18 +0,0 @@
-- old --
-a
-b
-c
-- new --
-a
-b
-c^D
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,3 +1,3 @@
- a
- b
-c
-+c
-\ No newline at end of file
--- a/grammar/internal/diff/testdata/eof2.txt
+++ b/grammar/internal/diff/testdata/eof2.txt
@@ -1,18 +0,0 @@
-- old --
-a
-b
-c^D
-- new --
-a
-b
-c
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,3 +1,3 @@
- a
- b
-c
-\ No newline at end of file
-+c
--- a/grammar/internal/diff/testdata/long.txt
+++ b/grammar/internal/diff/testdata/long.txt
@@ -1,62 +0,0 @@
-- old --
-1
-2
-3
-4
-5
-6
-7
-8
-9
-10
-11
-12
-13
-14
-14½
-15
-16
-17
-18
-19
-20
-- new --
-1
-2
-3
-4
-5
-6
-8
-9
-10
-11
-12
-13
-14
-17
-18
-19
-20
-- diff --
-diff old new
--- old
-+++ new
-@@ -4,7 +4,6 @@
- 4
- 5
- 6
-7
- 8
- 9
- 10
-@@ -12,9 +11,6 @@
- 12
- 13
- 14
-14½
-15
-16
- 17
- 18
- 19
--- a/grammar/internal/diff/testdata/same.txt
+++ b/grammar/internal/diff/testdata/same.txt
@@ -1,5 +0,0 @@
-- old --
-hello world
-- new --
-hello world
-- diff --
--- a/grammar/internal/diff/testdata/start.txt
+++ b/grammar/internal/diff/testdata/start.txt
@@ -1,34 +0,0 @@
-- old --
-e
-pi
-4
-5
-6
-7
-8
-9
-10
-- new --
-1
-2
-3
-4
-5
-6
-7
-8
-9
-10
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,5 +1,6 @@
-e
-pi
-+1
-+2
-+3
- 4
- 5
- 6
--- a/grammar/internal/diff/testdata/triv.txt
+++ b/grammar/internal/diff/testdata/triv.txt
@@ -1,40 +0,0 @@
-Another example from Hunt and McIlroy,
-“An Algorithm for Differential File Comparison.”
-https://www.cs.dartmouth.edu/~doug/diff.pdf
-
-Anchored diff gives up on finding anything,
-since there are no unique lines.
-
-- old --
-a
-b
-c
-a
-b
-b
-a
-- new --
-c
-a
-b
-a
-b
-c
-- diff --
-diff old new
--- old
-+++ new
-@@ -1,7 +1,6 @@
-a
-b
-c
-a
-b
-b
-a
-+c
-+a
-+b
-+a
-+b
-+c
--- a/grammar/jsonschema/decode.go
+++ b/grammar/jsonschema/decode.go
@@ -1,171 +0,0 @@
-package jsonschema
-
-import (
-	"bytes"
-	"encoding/json"
-	"errors"
-)
-
-// Schema holds a JSON schema.
-type Schema struct {
-	// Name is the name of the property. For the parent/root property, this
-	// is "root". For child properties, this is the name of the property.
-	Name string `json:"-"`
-
-	// Type is the type of the property.
-	//
-	// TODO: Union types (e.g. make this a []string).
-	Type string
-
-	// PrefixItems is a list of schemas for each item in a tuple. By
-	// default, the tuple is "closed." unless Items is set to true or a
-	// valid Schema.
-	PrefixItems []*Schema
-
-	// Items is the schema for each item in a list.
-	//
-	// If it is missing, or its JSON value is "null" or "false", it is nil.
-	// If the JSON value is "true", it is set to the empty Schema. If the
-	// JSON value is an object, it will be decoded as a Schema.
-	Items *Schema
-
-	// MinItems specifies the minimum number of items allowed in a list.
-	MinItems int
-
-	// MaxItems specifies the maximum number of items allowed in a list.
-	MaxItems int
-
-	// Properties is the schema for each property of an object.
-	Properties []*Schema
-
-	// Format is the format of the property. This is used to validate the
-	// property against a specific format.
-	//
-	// It is the callers responsibility to validate the property against
-	// the format.
-	Format string
-
-	// Minimum specifies the minimum value for numeric properties.
-	Minimum float64
-
-	// Maximum specifies the maximum value for numeric properties.
-	Maximum float64
-
-	// Enum is a list of valid values for the property.
-	Enum []json.RawMessage
-}
-
-func (s *Schema) UnmarshalJSON(data []byte) error {
-	type S Schema
-	w := struct {
-		Properties props
-		Items      items
-		*S
-	}{
-		S: (*S)(s),
-	}
-	if err := json.Unmarshal(data, &w); err != nil {
-		return err
-	}
-	if w.Items.set {
-		s.Items = &w.Items.Schema
-	}
-	s.Properties = w.Properties
-	return nil
-}
-
-type items struct {
-	Schema
-	set bool
-}
-
-func (s *items) UnmarshalJSON(data []byte) error {
-	switch b := data[0]; b {
-	case 't':
-		*s = items{set: true}
-	case '{':
-		type I items
-		if err := json.Unmarshal(data, (*I)(s)); err != nil {
-			return err
-		}
-		s.set = true
-	case 'n', 'f':
-	default:
-		return errors.New("invalid Items")
-	}
-	return nil
-}
-
-// EffectiveType returns the effective type of the schema. If the Type field is
-// not empty, it is returned; otherwise:
-//
-//   - If the schema has both Properties and Items, it returns an empty string.
-//   - If the schema has Properties, it returns "object".
-//   - If the schema has Items, it returns "array".
-//   - If the schema has neither Properties nor Items, it returns "value".
-//
-// The returned string is never empty.
-func (d *Schema) EffectiveType() string {
-	if d.Type == "" {
-		if len(d.Properties) > 0 {
-			return "object"
-		}
-		if len(d.PrefixItems) > 0 || d.Items != nil {
-			return "array"
-		}
-		return "value"
-	}
-	return d.Type
-}
-
-// props is an ordered list of properties. The order of the properties
-// is the order in which they were defined in the schema.
-type props []*Schema
-
-var _ json.Unmarshaler = (*props)(nil)
-
-func (v *props) UnmarshalJSON(data []byte) error {
-	if len(data) == 0 {
-		return nil
-	}
-	if data[0] != '{' {
-		return errors.New("expected object")
-	}
-
-	d := json.NewDecoder(bytes.NewReader(data))
-
-	// TODO(bmizerany): Consider DisallowUnknownFields. Currently, we, like
-	// llama.cpp, ignore unknown fields, which could be lead to unexpected
-	// behavior for clients of this package, since they may not be aware
-	// that "additionalFields", "itemsPrefix", etc, are being ignored.
-	//
-	// For now, just do what llama.cpp does.
-
-	t, err := d.Token()
-	if err != nil {
-		return err
-	}
-	if t != json.Delim('{') {
-		return errors.New("expected object")
-	}
-	for d.More() {
-		// Use the first token (map key) as the property name, then
-		// decode the rest of the object fields into a Schema and
-		// append.
-		t, err := d.Token()
-		if err != nil {
-			return err
-		}
-		if t == json.Delim('}') {
-			return nil
-		}
-		s := &Schema{
-			Name: t.(string),
-		}
-		if err := d.Decode(s); err != nil {
-			return err
-		}
-		*v = append(*v, s)
-	}
-	return nil
-}
--- a/grammar/jsonschema/decode_test.go
+++ b/grammar/jsonschema/decode_test.go
@@ -1,104 +0,0 @@
-package jsonschema
-
-import (
-	"encoding/json"
-	"reflect"
-	"strings"
-	"testing"
-
-	"github.com/google/go-cmp/cmp"
-)
-
-const testSchemaBasic = `
-{
-  "properties": {
-    "tupleClosedEmpty":   { "prefixItems": [] },
-    "tupleClosedMissing": { "prefixItems": [{}] },
-    "tupleClosedNull":    { "prefixItems": [{}], "items": null },
-    "tupleClosedFalse":   { "prefixItems": [{}], "items": false },
-    "tupleOpenTrue":      { "prefixItems": [{}], "items": true },
-    "tupleOpenEmpty":     { "prefixItems": [{}], "items": {} },
-    "tupleOpenTyped":     { "prefixItems": [{}], "items": {"type": "boolean"} },
-    "tupleOpenMax":       { "prefixItems": [{}], "items": true, "maxItems": 3},
-
-    "array": { "items": {"type": "number"} },
-
-    "null": { "type": "null" },
-    "string": { "type": "string" },
-    "boolean": { "type": "boolean" }
-  }
-}
-`
-
-func TestSchemaUnmarshal(t *testing.T) {
-	var got *Schema
-	if err := json.Unmarshal([]byte(testSchemaBasic), &got); err != nil {
-		t.Fatalf("Unmarshal: %v", err)
-	}
-	want := &Schema{
-		Properties: []*Schema{
-			{Name: "tupleClosedEmpty", PrefixItems: []*Schema{}, Items: nil},
-			{Name: "tupleClosedMissing", PrefixItems: []*Schema{{}}, Items: nil},
-			{Name: "tupleClosedNull", PrefixItems: []*Schema{{}}, Items: nil},
-			{Name: "tupleClosedFalse", PrefixItems: []*Schema{{}}, Items: nil},
-
-			{Name: "tupleOpenTrue", PrefixItems: []*Schema{{}}, Items: &Schema{}},
-			{Name: "tupleOpenEmpty", PrefixItems: []*Schema{{}}, Items: &Schema{}},
-			{Name: "tupleOpenTyped", PrefixItems: []*Schema{{}}, Items: &Schema{Type: "boolean"}},
-			{Name: "tupleOpenMax", PrefixItems: []*Schema{{}}, Items: &Schema{}, MaxItems: 3},
-
-			{Name: "array", Items: &Schema{Type: "number"}},
-
-			{Name: "null", Type: "null"},
-			{Name: "string", Type: "string"},
-			{Name: "boolean", Type: "boolean"},
-		},
-	}
-
-	if diff := cmp.Diff(want, got); diff != "" {
-		t.Errorf("(-want, +got)\n%s", diff)
-	}
-}
-
-func TestEffectiveType(t *testing.T) {
-	const schema = `
-		{"properties": {
-			"o": {"type": "object"},
-			"a": {"type": "array"},
-			"n": {"type": "number"},
-			"s": {"type": "string"},
-			"z": {"type": "null"},
-			"b": {"type": "boolean"},
-
-			"t0": {"prefixItems": [{}], "items": {"type": "number"}},
-			"t1": {"items": {"type": "number"}, "maxItems": 3},
-
-			"v": {"maxItems": 3}
-		}}
-	`
-
-	var s *Schema
-	if err := json.Unmarshal([]byte(schema), &s); err != nil {
-		t.Fatalf("json.Unmarshal: %v", err)
-	}
-
-	var got []string
-	for _, p := range s.Properties {
-		got = append(got, p.EffectiveType())
-	}
-
-	want := strings.Fields(`
-		object
-		array
-		number
-		string
-		null
-		boolean
-		array
-		array
-		value
-	`)
-	if !reflect.DeepEqual(want, got) {
-		t.Errorf("\ngot:\n\t%v\nwant:\n\t%v", got, want)
-	}
-}
--- a/grammar/testdata/schemas.txt
+++ b/grammar/testdata/schemas.txt
@@ -1,76 +0,0 @@
-# This file holds tests for JSON schema to EBNF grammar conversions.
-#
-# The format is a JSON schema, followed by the expected EBNF grammar. Each test
-# MAY be preceded by a comment that describes the test (e.g. the test name), followed by
-# the JSON schema and the expected EBNF grammar. If no comment is present, the test
-# name the tests number in the file (e.g. "#0", "#1", etc.)
-#
-# Blank lines signify the end or start of a new test. Comments can be added
-# anywhere in the file, but they must be preceded by a '#' character and start at
-# the beginning of the line.
-
-# default
-{}
-root ::= value;
-
-{"properties": {}}
-root ::= value;
-
-# array
-{"properties": {"a": {"type": "array", "items": {"type": "string"}}}}
-root_0_tuple_0 ::= string;
-root_0         ::= "[" ( root_0_tuple_0 )* "]";
-root           ::= "{" "a" ":" root_0 "}";
-
-# array with nested array
-{"type": "array", "items": {"type": "array", "items": {"type": "string"}}}
-root_tuple_0_tuple_0 ::= string;
-root_tuple_0         ::= "[" ( root_tuple_0_tuple_0 )* "]";
-root                 ::= "[" ( root_tuple_0 )* "]";
-
-# object
-{"properties": {"e": {}}}
-root_0 ::= value;
-root   ::= "{" "e" ":" root_0 "}";
-
-# object with nested object
-{"properties": {"o": {"type": "object", "properties": {"e": {}}}}}
-root_0_0 ::= value;
-root_0   ::= "{" "e" ":" root_0_0 "}";
-root     ::= "{" "o" ":" root_0 "}";
-
-# boolean
-{"type": "boolean"}
-root ::= boolean;
-
-# number
-{"properties": {"n": {"type": "number", "minimum": 123, "maximum": 4567}}}
-root_0 ::= number;
-root   ::= "{" "n" ":" root_0 "}";
-
-# string
-{"type": "string"}
-root ::= string;
-
-# string with enum
-{"type": "string", "enum": ["a", "b", "c"]}
-root ::= ( "\"a\"" "|" "\"b\"" "|" "\"c\"" );
-
-# spaces in key
-{"properties": {"a b": {}}}
-root_0 ::= value;
-root   ::= "{" "a b" ":" root_0 "}";
-
-# issue7978
-{ "type": "object", "properties": { "steps": { "type": "array", "items": { "type": "object", "properties": { "explanation": { "type": "string" }, "output": { "type": "string" } }, "required": [ "explanation", "output" ], "additionalProperties": false } }, "final_answer": { "type": "string" } }, "required": [ "steps", "final_answer" ], "additionalProperties": false }
-root_0_tuple_0_0 ::= string;
-root_0_tuple_0_1 ::= string;
-root_0_tuple_0   ::= "{" "explanation" ":" root_0_tuple_0_0 "," "output" ":" root_0_tuple_0_1 "}";
-root_0           ::= "[" ( root_0_tuple_0 )* "]";
-root_1           ::= string;
-root             ::= "{" "steps" ":" root_0 "," "final_answer" ":" root_1 "}";
-
-# !! # special characters in key
-# !! {"properties": {"a!b": {}}}
-# !! !invalid character '!' in key
-# !! 
--- a/integration/llm_image_test.go
+++ b/integration/llm_image_test.go
@@ -66,35 +66,6 @@ func TestIntegrationMllama(t *testing.T) {
 	DoGenerate(ctx, t, client, req, []string{resp}, 240*time.Second, 30*time.Second)
 }

-func TestIntegrationSplitBatch(t *testing.T) {
-	image, err := base64.StdEncoding.DecodeString(imageEncoding)
-	require.NoError(t, err)
-	req := api.GenerateRequest{
-		Model: "gemma3:4b",
-		// Fill up a chunk of the batch so the image will partially spill over into the next one
-		System: "Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed aliquet, justo in malesuada lobortis, odio ligula volutpat quam, quis faucibus ipsum magna quis sapien. Aliquam in venenatis diam, eu viverra magna. Phasellus imperdiet hendrerit volutpat. Vivamus sem ex, facilisis placerat felis non, dictum elementum est. Phasellus aliquam imperdiet lacus, eget placerat ligula sodales vel. Pellentesque nec auctor mi. Curabitur arcu nisi, faucibus eget nunc id, viverra interdum mi. Curabitur ornare ipsum ex, ac euismod ex aliquam in. Vestibulum id magna at purus accumsan fermentum. Proin scelerisque posuere nunc quis interdum. Maecenas sed mollis nisl. Etiam vitae ipsum interdum, placerat est quis, tincidunt velit. Nullam tempor nibh non lorem volutpat efficitur. Cras laoreet diam imperdiet ipsum auctor bibendum. Suspendisse ultrices urna sed metus sagittis suscipit. Quisque ullamcorper aliquam nibh ut mollis. Aenean dapibus mauris pharetra, venenatis elit ac, hendrerit odio. Cras vestibulum erat tempor, lobortis justo eu, lobortis ipsum. Nam laoreet dapibus sem. Proin vel diam ultrices, elementum ante et, ornare lectus. Proin eu accumsan nisl. Praesent ac ex vitae ipsum vulputate tristique facilisis sit amet lacus. Nullam faucibus magna a pellentesque pretium. Nunc lacinia ullamcorper sollicitudin. Donec vitae accumsan turpis, sed porttitor est. Donec porttitor mi vitae augue faucibus, vel mollis diam tincidunt.",
-		Prompt: "what does the text in this image say?",
-		Stream: &stream,
-		Options: map[string]interface{}{
-			"seed":        42,
-			"temperature": 0.0,
-		},
-		Images: []api.ImageData{
-			image,
-		},
-	}
-
-	// Note: sometimes it returns "the ollamas" sometimes "the ollams"
-	resp := "the ollam"
-	ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-	require.NoError(t, PullIfMissing(ctx, client, req.Model))
-	// llava models on CPU can be quite slow to start,
-	DoGenerate(ctx, t, client, req, []string{resp}, 120*time.Second, 30*time.Second)
-}
-
 const imageEncoding = `iVBORw0KGgoAAAANSUhEUgAAANIAAAB4CAYAAACHHqzKAAAAAXNSR0IArs4c6QAAAIRlWElmTU0AKgAAAAgABQESAAMAAAABAAEAAAEaAAUAAAABAAAASgEb
 AAUAAAABAAAAUgEoAAMAAAABAAIAAIdpAAQAAAABAAAAWgAAAAAAAABIAAAAAQAAAEgAAAABAAOgAQADAAAAAQABAACgAgAEAAAAAQAAANKgAwAEAAAAAQAA
 AHgAAAAAXdsepgAAAAlwSFlzAAALEwAACxMBAJqcGAAAAVlpVFh0WE1MOmNvbS5hZG9iZS54bXAAAAAAADx4OnhtcG1ldGEgeG1sbnM6eD0iYWRvYmU6bnM6
--- a/kvcache/cache.go
+++ b/kvcache/cache.go
@@ -43,13 +43,8 @@ type Cache interface {

 	// ** cache management **

-	// Init sets up runtime parameters.
-	// backend: Used to allocate cache data storage and execute management operations (such as defrag)
-	// dtype: The data type for storing cache entries
-	// maxSequences: The maximum number of sequences stored in the cache - across all batches
-	// capacity: The number of cache entries to store, per sequence
-	// maxBatch: The maximum number of tokens that can occur in a single batch
-	Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity, maxBatch int)
+	// Init sets up runtime parameters
+	Init(backend ml.Backend, dtype ml.DType, capacity int32)

 	// Close closes the cache and frees resources associated with it
 	Close()
@@ -57,7 +52,7 @@ type Cache interface {
 	// StartForward is called before the start of the model's forward pass.
 	// For each token in the coming batch, there must be a corresponding
 	// entry in positions and seqs.
-	StartForward(ctx ml.Context, batch input.Batch) error
+	StartForward(ctx ml.Context, opts input.Options) error

 	// CopyPrefix copies tokens in the range [0, len) from srcSeq to dstSeq
 	CopyPrefix(srcSeq, dstSeq int, len int32)
--- a/kvcache/causal.go
+++ b/kvcache/causal.go
@@ -20,6 +20,7 @@ type shiftFn func(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, e
 // The mask is of shape history size, batch size
 type Causal struct {
 	DType      ml.DType
+	Capacity   int32
 	windowSize int32

 	opts CausalOptions
@@ -97,7 +98,7 @@ func NewSWACache(windowSize int32, shift shiftFn) *Causal {
 	}
 }

-func (c *Causal) Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity, maxBatch int) {
+func (c *Causal) Init(backend ml.Backend, dtype ml.DType, capacity int32) {
 	if c.config == nil {
 		var config ml.CacheConfig
 		if cc, ok := backend.(ml.BackendCacheConfig); ok {
@@ -118,16 +119,9 @@ func (c *Causal) Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity
 		c.config.MaskDType = ml.DTypeF32
 	}

-	var cacheSize int
-	if c.windowSize == math.MaxInt32 || capacity < int(c.windowSize)+maxBatch {
-		cacheSize = maxSequences * capacity
-	} else {
-		cacheSize = maxSequences * (int(c.windowSize) + maxBatch)
-	}
-	cacheSize = roundUp(cacheSize, c.config.CachePadding)
-	c.cells = make([]cacheCell, cacheSize)
-
 	c.DType = dtype
+	c.Capacity = int32(roundUp(int(capacity), c.config.CachePadding))
+	c.cells = make([]cacheCell, c.Capacity)
 	c.cellRanges = make(map[int]cellRange)
 	c.backend = backend
 }
@@ -146,14 +140,12 @@ func (c *Causal) Close() {
 	}
 }

-func (c *Causal) StartForward(ctx ml.Context, batch input.Batch) error {
-	c.curBatchSize = len(batch.Positions)
-	c.curSequences = batch.Sequences
-	c.curPositions = batch.Positions
+func (c *Causal) StartForward(ctx ml.Context, opts input.Options) error {
+	c.curBatchSize = len(opts.Positions)
+	c.curSequences = opts.Sequences
+	c.curPositions = opts.Positions
 	c.opts.Except = nil

-	c.updateSlidingWindow()
-
 	var err error
 	c.curLoc, err = c.findStartLoc()
 	if errors.Is(err, ErrKvCacheFull) {
@@ -165,8 +157,8 @@ func (c *Causal) StartForward(ctx ml.Context, batch input.Batch) error {
 	}

 	c.curCellRange = newRange()
-	for i, pos := range batch.Positions {
-		seq := batch.Sequences[i]
+	for i, pos := range opts.Positions {
+		seq := opts.Sequences[i]

 		c.cells[c.curLoc+i] = cacheCell{pos: pos, sequences: []int{seq}}

@@ -218,51 +210,7 @@ func (c *Causal) findStartLoc() (int, error) {
 		}
 	}

-	return 0, fmt.Errorf("%w (length: %v)", ErrKvCacheFull, len(c.cells))
-}
-
-func (c *Causal) updateSlidingWindow() {
-	if c.windowSize == math.MaxInt32 {
-		return
-	}
-
-	// create a map of unique sequences to the lowest position in that sequence
-	lowestPos := make(map[int]int32)
-	for i := range c.curPositions {
-		seq := c.curSequences[i]
-
-		pos, ok := lowestPos[seq]
-		if !ok {
-			pos = c.curPositions[i]
-		} else if c.curPositions[i] < pos {
-			pos = c.curPositions[i]
-		}
-
-		lowestPos[seq] = pos
-	}
-
-	// delete any entries that are beyond the window of the oldest position in the sequence
-	for seq, pos := range lowestPos {
-		oldRange, ok := c.cellRanges[seq]
-		if !ok {
-			continue
-		}
-
-		newRange := newRange()
-
-		for i := oldRange.min; i <= oldRange.max; i++ {
-			if slices.Contains(c.cells[i].sequences, seq) {
-				if c.cells[i].pos < pos-c.windowSize {
-					c.cells[i].sequences = slices.DeleteFunc(c.cells[i].sequences, func(s int) bool { return s == seq })
-				} else {
-					newRange.min = min(newRange.min, i)
-					newRange.max = max(newRange.max, i)
-				}
-			}
-		}
-
-		c.cellRanges[seq] = newRange
-	}
+	return 0, fmt.Errorf("%w (length: %v)", ErrKvCacheFull, c.Capacity)
 }

 func roundDown(length, pad int) int {
@@ -317,7 +265,7 @@ func (c *Causal) buildMask(ctx ml.Context) (ml.Tensor, error) {
 	return maskTensor, nil
 }

-func (c *Causal) moveCells(ctx ml.Context, src, dst, length int) {
+func (c *Causal) moveCells(ctx ml.Context, src, dst, len int) {
 	for i, key := range c.keys {
 		if key == nil {
 			continue
@@ -327,8 +275,8 @@ func (c *Causal) moveCells(ctx ml.Context, src, dst, length int) {
 		numKVHeads := key.Dim(1)
 		rowSize := key.Stride(2)

-		kSrcView := key.View(ctx, rowSize*src, kHeadDim*numKVHeads*length)
-		kDstView := key.View(ctx, rowSize*dst, kHeadDim*numKVHeads*length)
+		kSrcView := key.View(ctx, rowSize*src, kHeadDim*numKVHeads*len)
+		kDstView := key.View(ctx, rowSize*dst, kHeadDim*numKVHeads*len)

 		value := c.values[i]
 		var vSrcView, vDstView ml.Tensor
@@ -336,14 +284,14 @@ func (c *Causal) moveCells(ctx ml.Context, src, dst, length int) {
 			vHeadDim := value.Dim(1)
 			elemSize := value.Stride(0)

-			vSrcView = value.View(ctx, elemSize*src, length, len(c.cells)*elemSize, vHeadDim*numKVHeads)
-			vDstView = value.View(ctx, elemSize*dst, length, len(c.cells)*elemSize, vHeadDim*numKVHeads)
+			vSrcView = value.View(ctx, elemSize*src, len, int(c.Capacity)*elemSize, vHeadDim*numKVHeads)
+			vDstView = value.View(ctx, elemSize*dst, len, int(c.Capacity)*elemSize, vHeadDim*numKVHeads)
 		} else {
 			vHeadDim := value.Dim(0)
 			rowSize := value.Stride(2)

-			vSrcView = value.View(ctx, rowSize*src, vHeadDim*numKVHeads*length)
-			vDstView = value.View(ctx, rowSize*dst, vHeadDim*numKVHeads*length)
+			vSrcView = value.View(ctx, rowSize*src, vHeadDim*numKVHeads*len)
+			vDstView = value.View(ctx, rowSize*dst, vHeadDim*numKVHeads*len)
 		}

 		ctx.Forward(
@@ -373,8 +321,7 @@ func (c *Causal) defrag() {
 	ctx := c.backend.NewContext()

 	// For every move, 6 tensors are required per layer (2 views and a
-	// copy for each of k and v). We also need to refer to the original
-	// k and v cache tensors - once per layer, not per move.
+	// copy for each of k and v).
 	layers := 0
 	for _, key := range c.keys {
 		if key == nil {
@@ -383,7 +330,7 @@ func (c *Causal) defrag() {
 		layers++
 	}

-	maxMoves := (ctx.MaxGraphNodes() - 2*layers) / (6 * layers)
+	maxMoves := ctx.MaxGraphNodes() / (6 * layers)
 	moves := 0

 	var pendingSrc, pendingDst, pendingLen int
@@ -532,14 +479,14 @@ func (c *Causal) Put(ctx ml.Context, key, value ml.Tensor) {
 	}

 	if _, ok := c.keys[c.curLayer]; !ok {
-		c.keys[c.curLayer] = c.ctxs[c.curLayer].Zeros(c.DType, kHeadDim, numKVHeads, len(c.cells))
+		c.keys[c.curLayer] = c.ctxs[c.curLayer].Zeros(c.DType, kHeadDim, numKVHeads, int(c.Capacity))
 	}

 	if _, ok := c.values[c.curLayer]; !ok {
 		if c.config.PermutedV {
-			c.values[c.curLayer] = c.ctxs[c.curLayer].Zeros(c.DType, len(c.cells), vHeadDim, numKVHeads)
+			c.values[c.curLayer] = c.ctxs[c.curLayer].Zeros(c.DType, int(c.Capacity), vHeadDim, numKVHeads)
 		} else {
-			c.values[c.curLayer] = c.ctxs[c.curLayer].Zeros(c.DType, vHeadDim, numKVHeads, len(c.cells))
+			c.values[c.curLayer] = c.ctxs[c.curLayer].Zeros(c.DType, vHeadDim, numKVHeads, int(c.Capacity))
 		}
 	}

@@ -550,7 +497,7 @@ func (c *Causal) Put(ctx ml.Context, key, value ml.Tensor) {
 		elemSize := c.values[c.curLayer].Stride(0)

 		value = value.Permute(ctx, 1, 2, 0, 3)
-		ctx.Forward(value.Copy(ctx, c.values[c.curLayer].View(ctx, elemSize*c.curLoc, batchSize, len(c.cells)*elemSize, vHeadDim*numKVHeads)))
+		ctx.Forward(value.Copy(ctx, c.values[c.curLayer].View(ctx, elemSize*c.curLoc, batchSize, int(c.Capacity)*elemSize, vHeadDim*numKVHeads)))
 	} else {
 		rowSize := c.values[c.curLayer].Stride(2)

--- a/kvcache/causal_test.go
+++ b/kvcache/causal_test.go
@@ -25,7 +25,7 @@ func TestStore(t *testing.T) {
 	cache := NewCausalCache(nil)
 	defer cache.Close()

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
+	cache.Init(backend, ml.DTypeF16, 16)

 	tests := []testCase{
 		{
@@ -58,11 +58,11 @@ func TestSWA(t *testing.T) {
 	cache := NewSWACache(1, nil)
 	defer cache.Close()

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
+	cache.Init(backend, ml.DTypeF32, 16)

 	tests := []testCase{
 		{
-			name:          "FirstBatch",
+			name:          "SlidingWindow",
 			in:            []float32{1, 2, 3, 4},
 			inShape:       []int{1, 1, 4},
 			seqs:          []int{0, 0, 0, 0},
@@ -71,16 +71,6 @@ func TestSWA(t *testing.T) {
 			expectedShape: []int{1, 1, 4},
 			expectedMask:  []float32{0, float32(math.Inf(-1)), float32(math.Inf(-1)), float32(math.Inf(-1)), 0, 0, float32(math.Inf(-1)), float32(math.Inf(-1)), float32(math.Inf(-1)), 0, 0, float32(math.Inf(-1)), float32(math.Inf(-1)), float32(math.Inf(-1)), 0, 0},
 		},
-		{
-			name:          "SecondBatch",
-			in:            []float32{5, 6},
-			inShape:       []int{1, 1, 2},
-			seqs:          []int{0, 0},
-			pos:           []int32{4, 5},
-			expected:      []float32{5, 6, 3, 4},
-			expectedShape: []int{1, 1, 4},
-			expectedMask:  []float32{0, float32(math.Inf(-1)), float32(math.Inf(-1)), 0, 0, 0, float32(math.Inf(-1)), float32(math.Inf(-1))},
-		},
 	}

 	testCache(t, backend, cache, tests)
@@ -91,7 +81,7 @@ func TestSequences(t *testing.T) {
 	cache := NewCausalCache(nil)
 	defer cache.Close()

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
+	cache.Init(backend, ml.DTypeF16, 16)

 	tests := []testCase{
 		{
@@ -126,7 +116,7 @@ func TestRemove(t *testing.T) {
 	})
 	defer cache.Close()

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
+	cache.Init(backend, ml.DTypeF16, 16)

 	tests := []testCase{
 		{
@@ -191,7 +181,7 @@ func TestDefrag(t *testing.T) {
 	})
 	defer cache.Close()

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
+	cache.Init(backend, ml.DTypeF16, 16)

 	tests := []testCase{
 		{
@@ -239,7 +229,7 @@ func TestCopy(t *testing.T) {
 	cache := NewCausalCache(func(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, error) { return key, nil })
 	defer cache.Close()

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
+	cache.Init(backend, ml.DTypeF16, 16)

 	tests := []testCase{
 		{
@@ -280,7 +270,7 @@ func testCache(t *testing.T, backend ml.Backend, cache Cache, tests []testCase)
 			context := backend.NewContext()
 			defer context.Close()

-			err := cache.StartForward(context, input.Batch{Positions: test.pos, Sequences: test.seqs})
+			err := cache.StartForward(context, input.Options{Positions: test.pos, Sequences: test.seqs})
 			if err != nil {
 				panic(err)
 			}
--- a/kvcache/encoder.go
+++ b/kvcache/encoder.go
@@ -49,7 +49,7 @@ func NewEncoderCache() *EncoderCache {
 	}
 }

-func (c *EncoderCache) Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity, maxBatch int) {
+func (c *EncoderCache) Init(backend ml.Backend, dtype ml.DType, capacity int32) {
 	if c.config == nil {
 		var config ml.CacheConfig
 		if cc, ok := backend.(ml.BackendCacheConfig); ok {
@@ -58,10 +58,6 @@ func (c *EncoderCache) Init(backend ml.Backend, dtype ml.DType, maxSequences, ca
 		c.config = &config
 	}

-	if maxSequences > 1 {
-		panic(fmt.Errorf("encoder cache does not support multiple sequences; requested: %v", maxSequences))
-	}
-
 	if c.config.CachePadding != 0 && c.config.CachePadding != 1 {
 		panic(fmt.Errorf("encoder cache is unable to enforce requested CachePadding (%v)", c.config.CachePadding))
 	}
@@ -83,10 +79,10 @@ func (c *EncoderCache) Close() {
 	}
 }

-func (c *EncoderCache) StartForward(ctx ml.Context, batch input.Batch) error {
+func (c *EncoderCache) StartForward(ctx ml.Context, opts input.Options) error {
 	// We work with the most recent image
-	if len(batch.Multimodal) > 0 {
-		c.curPos = batch.Positions[batch.Multimodal[len(batch.Multimodal)-1].Index]
+	if len(opts.Multimodal) > 0 {
+		c.curPos = opts.Positions[opts.Multimodal[len(opts.Multimodal)-1].Index]
 	}

 	return nil
--- a/kvcache/wrapper.go
+++ b/kvcache/wrapper.go
@@ -23,9 +23,9 @@ func NewWrapperCache(caches ...Cache) *WrapperCache {
 	}
 }

-func (c *WrapperCache) Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity, maxBatch int) {
+func (c *WrapperCache) Init(backend ml.Backend, dtype ml.DType, capacity int32) {
 	for _, cache := range c.caches {
-		cache.Init(backend, dtype, maxSequences, capacity, maxBatch)
+		cache.Init(backend, dtype, capacity)
 	}
 }

@@ -41,14 +41,14 @@ func (c *WrapperCache) Close() {
 	}
 }

-func (c *WrapperCache) StartForward(ctx ml.Context, batch input.Batch) error {
+func (c *WrapperCache) StartForward(ctx ml.Context, opts input.Options) error {
 	for i, cache := range c.caches {
-		err := cache.StartForward(ctx, batch)
+		err := cache.StartForward(ctx, opts)
 		if err != nil {
 			// unwind on error - Remove with endIndex set to math.MaxInt32 does not fail
 			for j := i - 1; j >= 0; j-- {
-				for k := range batch.Positions {
-					_ = c.caches[j].Remove(batch.Sequences[k], batch.Positions[k], math.MaxInt32)
+				for k := range opts.Positions {
+					_ = c.caches[j].Remove(opts.Sequences[k], opts.Positions[k], math.MaxInt32)
 				}
 			}
 			return err
--- a/llama/llama.cpp/src/llama-arch.cpp
+++ b/llama/llama.cpp/src/llama-arch.cpp
@@ -37,7 +37,6 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
    { LLM_ARCH_MINICPM3,         "minicpm3"         },
    { LLM_ARCH_GEMMA,            "gemma"            },
    { LLM_ARCH_GEMMA2,           "gemma2"           },
-    { LLM_ARCH_GEMMA3,           "gemma3"           },
    { LLM_ARCH_STARCODER2,       "starcoder2"       },
    { LLM_ARCH_MAMBA,            "mamba"            },
    { LLM_ARCH_XVERSE,           "xverse"           },
@@ -805,24 +804,6 @@ static const std::map<llm_arch, std::map<llm_tensor, const char *>> LLM_TENSOR_N
            { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
        },
    },
-    {
-        LLM_ARCH_GEMMA3,
-        {
-            { LLM_TENSOR_TOKEN_EMBD,      "token_embd" },
-            { LLM_TENSOR_OUTPUT_NORM,     "output_norm" },
-            { LLM_TENSOR_ATTN_NORM,       "blk.%d.attn_norm" },
-            { LLM_TENSOR_ATTN_Q,          "blk.%d.attn_q" },
-            { LLM_TENSOR_ATTN_K,          "blk.%d.attn_k" },
-            { LLM_TENSOR_ATTN_V,          "blk.%d.attn_v" },
-            { LLM_TENSOR_ATTN_OUT,        "blk.%d.attn_output" },
-            { LLM_TENSOR_ATTN_POST_NORM,  "blk.%d.post_attention_norm" },
-            { LLM_TENSOR_FFN_NORM,        "blk.%d.ffn_norm" },
-            { LLM_TENSOR_FFN_GATE,        "blk.%d.ffn_gate" },
-            { LLM_TENSOR_FFN_DOWN,        "blk.%d.ffn_down" },
-            { LLM_TENSOR_FFN_UP,          "blk.%d.ffn_up" },
-            { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
-        },
-    },
    {
        LLM_ARCH_STARCODER2,
        {
--- a/llama/llama.cpp/src/llama-arch.h
+++ b/llama/llama.cpp/src/llama-arch.h
@@ -41,7 +41,6 @@ enum llm_arch {
    LLM_ARCH_MINICPM3,
    LLM_ARCH_GEMMA,
    LLM_ARCH_GEMMA2,
-    LLM_ARCH_GEMMA3,
    LLM_ARCH_STARCODER2,
    LLM_ARCH_MAMBA,
    LLM_ARCH_XVERSE,
--- a/llama/llama.cpp/src/llama-model.cpp
+++ b/llama/llama.cpp/src/llama-model.cpp
@@ -878,9 +878,6 @@ void llama_model::load_hparams(llama_model_loader & ml) {
                    default: type = LLM_TYPE_UNKNOWN;
               }
            } break;
-        case LLM_ARCH_GEMMA3:
-            {
-            } break;
        case LLM_ARCH_STARCODER2:
            {
                ml.get_key(LLM_KV_ATTENTION_LAYERNORM_EPS, hparams.f_norm_eps);
@@ -2540,9 +2537,6 @@ bool llama_model::load_tensors(llama_model_loader & ml) {
                        layer.ffn_post_norm = create_tensor(tn(LLM_TENSOR_FFN_POST_NORM, "weight", i), {n_embd}, 0);
                    }
                } break;
-            case LLM_ARCH_GEMMA3:
-                {
-                } break;
            case LLM_ARCH_STARCODER2:
                {
                    tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
@@ -4035,7 +4029,6 @@ enum llama_rope_type llama_model_rope_type(const struct llama_model * model) {
        case LLM_ARCH_PHIMOE:
        case LLM_ARCH_GEMMA:
        case LLM_ARCH_GEMMA2:
-        case LLM_ARCH_GEMMA3:
        case LLM_ARCH_STARCODER2:
        case LLM_ARCH_OPENELM:
        case LLM_ARCH_GPTNEOX:
--- a/llama/llama.cpp/src/llama-quant.cpp
+++ b/llama/llama.cpp/src/llama-quant.cpp
@@ -737,15 +737,6 @@ static void llama_model_quantize_impl(const std::string & fname_inp, const std::
        // This used to be a regex, but <regex> has an extreme cost to compile times.
        bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?

-        // don't quantize vision stuff
-        quantize &= name.find("v.blk.") == std::string::npos;
-
-        quantize &= name.find("mm.mm_input_projection.weight") == std::string::npos;
-        quantize &= name.find("mm.mm_soft_emb_norm.weight") == std::string::npos;
-        quantize &= name.find("v.patch_embedding.weight") == std::string::npos;
-        quantize &= name.find("v.position_embedding.weight") == std::string::npos;
-        quantize &= name.find("v.post_layernorm.weight") == std::string::npos;
-
        // quantize only 2D and 3D tensors (experts)
        quantize &= (ggml_n_dims(tensor) >= 2);

--- a/llama/patches/0021-gemma3-quantization.patch
+++ b/llama/patches/0021-gemma3-quantization.patch
@@ -1,113 +0,0 @@
-From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
-From: Patrick Devine <patrick@infrahq.com>
-Date: Fri, 14 Mar 2025 16:33:23 -0700
-Subject: [PATCH] gemma3 quantization
-
---
- src/llama-arch.cpp  | 19 +++++++++++++++++++
- src/llama-arch.h    |  1 +
- src/llama-model.cpp |  7 +++++++
- src/llama-quant.cpp |  9 +++++++++
- 4 files changed, 36 insertions(+)
-
-diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
-index b6f20286..b443fcd3 100644
--- a/src/llama-arch.cpp
-+++ b/src/llama-arch.cpp
-@@ -37,6 +37,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
-     { LLM_ARCH_MINICPM3,         "minicpm3"         },
-     { LLM_ARCH_GEMMA,            "gemma"            },
-     { LLM_ARCH_GEMMA2,           "gemma2"           },
-+    { LLM_ARCH_GEMMA3,           "gemma3"           },
-     { LLM_ARCH_STARCODER2,       "starcoder2"       },
-     { LLM_ARCH_MAMBA,            "mamba"            },
-     { LLM_ARCH_XVERSE,           "xverse"           },
-@@ -804,6 +805,24 @@ static const std::map<llm_arch, std::map<llm_tensor, const char *>> LLM_TENSOR_N
-             { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
-         },
-     },
-+    {
-+        LLM_ARCH_GEMMA3,
-+        {
-+            { LLM_TENSOR_TOKEN_EMBD,      "token_embd" },
-+            { LLM_TENSOR_OUTPUT_NORM,     "output_norm" },
-+            { LLM_TENSOR_ATTN_NORM,       "blk.%d.attn_norm" },
-+            { LLM_TENSOR_ATTN_Q,          "blk.%d.attn_q" },
-+            { LLM_TENSOR_ATTN_K,          "blk.%d.attn_k" },
-+            { LLM_TENSOR_ATTN_V,          "blk.%d.attn_v" },
-+            { LLM_TENSOR_ATTN_OUT,        "blk.%d.attn_output" },
-+            { LLM_TENSOR_ATTN_POST_NORM,  "blk.%d.post_attention_norm" },
-+            { LLM_TENSOR_FFN_NORM,        "blk.%d.ffn_norm" },
-+            { LLM_TENSOR_FFN_GATE,        "blk.%d.ffn_gate" },
-+            { LLM_TENSOR_FFN_DOWN,        "blk.%d.ffn_down" },
-+            { LLM_TENSOR_FFN_UP,          "blk.%d.ffn_up" },
-+            { LLM_TENSOR_FFN_POST_NORM,   "blk.%d.post_ffw_norm" },
-+        },
-+    },
-     {
-         LLM_ARCH_STARCODER2,
-         {
-diff --git a/src/llama-arch.h b/src/llama-arch.h
-index ec742224..aad92a5d 100644
--- a/src/llama-arch.h
-+++ b/src/llama-arch.h
-@@ -41,6 +41,7 @@ enum llm_arch {
-     LLM_ARCH_MINICPM3,
-     LLM_ARCH_GEMMA,
-     LLM_ARCH_GEMMA2,
-+    LLM_ARCH_GEMMA3,
-     LLM_ARCH_STARCODER2,
-     LLM_ARCH_MAMBA,
-     LLM_ARCH_XVERSE,
-diff --git a/src/llama-model.cpp b/src/llama-model.cpp
-index ab1a07d1..70183041 100644
--- a/src/llama-model.cpp
-+++ b/src/llama-model.cpp
-@@ -878,6 +878,9 @@ void llama_model::load_hparams(llama_model_loader & ml) {
-                     default: type = LLM_TYPE_UNKNOWN;
-                }
-             } break;
-+        case LLM_ARCH_GEMMA3:
-+            {
-+            } break;
-         case LLM_ARCH_STARCODER2:
-             {
-                 ml.get_key(LLM_KV_ATTENTION_LAYERNORM_EPS, hparams.f_norm_eps);
-@@ -2537,6 +2540,9 @@ bool llama_model::load_tensors(llama_model_loader & ml) {
-                         layer.ffn_post_norm = create_tensor(tn(LLM_TENSOR_FFN_POST_NORM, "weight", i), {n_embd}, 0);
-                     }
-                 } break;
-+            case LLM_ARCH_GEMMA3:
-+                {
-+                } break;
-             case LLM_ARCH_STARCODER2:
-                 {
-                     tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
-@@ -4029,6 +4035,7 @@ enum llama_rope_type llama_model_rope_type(const struct llama_model * model) {
-         case LLM_ARCH_PHIMOE:
-         case LLM_ARCH_GEMMA:
-         case LLM_ARCH_GEMMA2:
-+        case LLM_ARCH_GEMMA3:
-         case LLM_ARCH_STARCODER2:
-         case LLM_ARCH_OPENELM:
-         case LLM_ARCH_GPTNEOX:
-diff --git a/src/llama-quant.cpp b/src/llama-quant.cpp
-index 6eb1da08..d2f3a510 100644
--- a/src/llama-quant.cpp
-+++ b/src/llama-quant.cpp
-@@ -737,6 +737,15 @@ static void llama_model_quantize_impl(const std::string & fname_inp, const std::
-         // This used to be a regex, but <regex> has an extreme cost to compile times.
-         bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?
- 
-+        // don't quantize vision stuff
-+        quantize &= name.find("v.blk.") == std::string::npos;
-+
-+        quantize &= name.find("mm.mm_input_projection.weight") == std::string::npos;
-+        quantize &= name.find("mm.mm_soft_emb_norm.weight") == std::string::npos;
-+        quantize &= name.find("v.patch_embedding.weight") == std::string::npos;
-+        quantize &= name.find("v.position_embedding.weight") == std::string::npos;
-+        quantize &= name.find("v.post_layernorm.weight") == std::string::npos;
-+
-         // quantize only 2D and 3D tensors (experts)
-         quantize &= (ggml_n_dims(tensor) >= 2);
- 
--- a/llm/server.go
+++ b/llm/server.go
@@ -29,7 +29,6 @@ import (
 	"github.com/ollama/ollama/envconfig"
 	"github.com/ollama/ollama/format"
 	"github.com/ollama/ollama/fs/ggml"
-	"github.com/ollama/ollama/grammar"
 	"github.com/ollama/ollama/llama"
 	"github.com/ollama/ollama/model"
 )
@@ -403,7 +402,7 @@ func NewLlamaServer(gpus discover.GpuInfoList, modelPath string, f *ggml.GGML, a
 			s.cmd.Env = append(s.cmd.Env, visibleDevicesEnv+"="+visibleDevicesEnvVal)
 		}

-		slog.Info("starting llama server", "cmd", s.cmd)
+		slog.Info("starting llama server", "cmd", s.cmd.String())
 		if envconfig.Debug() {
 			filteredEnv := []string{}
 			for _, ev := range s.cmd.Env {
@@ -471,7 +470,7 @@ const ( // iota is reset to 0
 	ServerStatusError
 )

-func (s ServerStatus) String() string {
+func (s ServerStatus) ToString() string {
 	switch s {
 	case ServerStatusReady:
 		return "llm server ready"
@@ -486,9 +485,12 @@ func (s ServerStatus) String() string {
 	}
 }

-type ServerStatusResponse struct {
-	Status   ServerStatus `json:"status"`
-	Progress float32      `json:"progress"`
+type ServerStatusResp struct {
+	Status          string  `json:"status"`
+	SlotsIdle       int     `json:"slots_idle"`
+	SlotsProcessing int     `json:"slots_processing"`
+	Error           string  `json:"error"`
+	Progress        float32 `json:"progress"`
 }

 func (s *llmServer) getServerStatus(ctx context.Context) (ServerStatus, error) {
@@ -500,7 +502,7 @@ func (s *llmServer) getServerStatus(ctx context.Context) (ServerStatus, error) {
 		}
 		if s.cmd.ProcessState.ExitCode() == -1 {
 			// Most likely a signal killed it, log some more details to try to help troubleshoot
-			slog.Warn("llama runner process no longer running", "sys", s.cmd.ProcessState.Sys(), "string", s.cmd.ProcessState)
+			slog.Warn("llama runner process no longer running", "sys", s.cmd.ProcessState.Sys(), "string", s.cmd.ProcessState.String())
 		}
 		return ServerStatusError, fmt.Errorf("llama runner process no longer running: %d %s", s.cmd.ProcessState.ExitCode(), msg)
 	}
@@ -525,19 +527,21 @@ func (s *llmServer) getServerStatus(ctx context.Context) (ServerStatus, error) {
 		return ServerStatusError, fmt.Errorf("read health request: %w", err)
 	}

-	var ssr ServerStatusResponse
-	if err := json.Unmarshal(body, &ssr); err != nil {
+	var status ServerStatusResp
+	if err := json.Unmarshal(body, &status); err != nil {
 		return ServerStatusError, fmt.Errorf("health unmarshal encode response: %w", err)
 	}

-	switch ssr.Status {
-	case ServerStatusLoadingModel:
-		s.loadProgress = ssr.Progress
-		return ssr.Status, nil
-	case ServerStatusReady, ServerStatusNoSlotsAvailable:
-		return ssr.Status, nil
+	switch status.Status {
+	case "ok":
+		return ServerStatusReady, nil
+	case "no slot available":
+		return ServerStatusNoSlotsAvailable, nil
+	case "loading model":
+		s.loadProgress = status.Progress
+		return ServerStatusLoadingModel, nil
 	default:
-		return ssr.Status, fmt.Errorf("server error: %+v", ssr)
+		return ServerStatusError, fmt.Errorf("server error: %+v", status)
 	}
 }

@@ -612,7 +616,7 @@ func (s *llmServer) WaitUntilRunning(ctx context.Context) error {
 		status, _ := s.getServerStatus(ctx)
 		if lastStatus != status && status != ServerStatusReady {
 			// Only log on status changes
-			slog.Info("waiting for server to become available", "status", status)
+			slog.Info("waiting for server to become available", "status", status.ToString())
 		}
 		switch status {
 		case ServerStatusReady:
@@ -626,7 +630,7 @@ func (s *llmServer) WaitUntilRunning(ctx context.Context) error {
 				slog.Debug(fmt.Sprintf("model load progress %0.2f", s.loadProgress))
 				stallTimer = time.Now().Add(stallDuration)
 			} else if !fullyLoaded && int(s.loadProgress*100.0) >= 100 {
-				slog.Debug("model load completed, waiting for server to become available", "status", status)
+				slog.Debug("model load completed, waiting for server to become available", "status", status.ToString())
 				stallTimer = time.Now().Add(stallDuration)
 				fullyLoaded = true
 			}
@@ -667,26 +671,63 @@ type ImageData struct {
 	AspectRatioID int    `json:"aspect_ratio_id"`
 }

+type completion struct {
+	Content      string `json:"content"`
+	Model        string `json:"model"`
+	Prompt       string `json:"prompt"`
+	Stop         bool   `json:"stop"`
+	StoppedLimit bool   `json:"stopped_limit"`
+
+	Timings struct {
+		PredictedN  int     `json:"predicted_n"`
+		PredictedMS float64 `json:"predicted_ms"`
+		PromptN     int     `json:"prompt_n"`
+		PromptMS    float64 `json:"prompt_ms"`
+	}
+}
+
 type CompletionRequest struct {
 	Prompt  string
 	Format  json.RawMessage
 	Images  []ImageData
 	Options *api.Options
-
-	Grammar string // set before sending the request to the subprocess
 }

 type CompletionResponse struct {
-	Content            string        `json:"content"`
-	DoneReason         string        `json:"done_reason"`
-	Done               bool          `json:"done"`
-	PromptEvalCount    int           `json:"prompt_eval_count"`
-	PromptEvalDuration time.Duration `json:"prompt_eval_duration"`
-	EvalCount          int           `json:"eval_count"`
-	EvalDuration       time.Duration `json:"eval_duration"`
+	Content            string
+	DoneReason         string
+	Done               bool
+	PromptEvalCount    int
+	PromptEvalDuration time.Duration
+	EvalCount          int
+	EvalDuration       time.Duration
 }

 func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn func(CompletionResponse)) error {
+	request := map[string]any{
+		"prompt":            req.Prompt,
+		"stream":            true,
+		"n_predict":         req.Options.NumPredict,
+		"n_keep":            req.Options.NumKeep,
+		"main_gpu":          req.Options.MainGPU,
+		"temperature":       req.Options.Temperature,
+		"top_k":             req.Options.TopK,
+		"top_p":             req.Options.TopP,
+		"min_p":             req.Options.MinP,
+		"typical_p":         req.Options.TypicalP,
+		"repeat_last_n":     req.Options.RepeatLastN,
+		"repeat_penalty":    req.Options.RepeatPenalty,
+		"presence_penalty":  req.Options.PresencePenalty,
+		"frequency_penalty": req.Options.FrequencyPenalty,
+		"mirostat":          req.Options.Mirostat,
+		"mirostat_tau":      req.Options.MirostatTau,
+		"mirostat_eta":      req.Options.MirostatEta,
+		"seed":              req.Options.Seed,
+		"stop":              req.Options.Stop,
+		"image_data":        req.Images,
+		"cache_prompt":      true,
+	}
+
 	if len(req.Format) > 0 {
 		switch string(req.Format) {
 		case `null`, `""`:
@@ -694,31 +735,21 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 			// these as "not set".
 			break
 		case `"json"`:
-			req.Grammar = grammarJSON
+			request["grammar"] = grammarJSON
 		default:
 			if req.Format[0] != '{' {
 				return fmt.Errorf("invalid format: %q; expected \"json\" or a valid JSON Schema object", req.Format)
 			}

 			// User provided a JSON schema
-			g, err := grammar.FromSchema(nil, req.Format)
-			if err != nil {
-				return fmt.Errorf("invalid JSON schema in format: %w", err)
+			g := llama.SchemaToGrammar(req.Format)
+			if g == nil {
+				return fmt.Errorf("invalid JSON schema in format")
 			}
-			req.Grammar = string(g)
+			request["grammar"] = string(g)
 		}
 	}

-	if req.Options == nil {
-		opts := api.DefaultOptions()
-		req.Options = &opts
-	}
-
-	if req.Options == nil {
-		opts := api.DefaultOptions()
-		req.Options = &opts
-	}
-
 	if err := s.sem.Acquire(ctx, 1); err != nil {
 		if errors.Is(err, context.Canceled) {
 			slog.Info("aborting completion request due to client closing the connection")
@@ -733,12 +764,13 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 	if req.Options.NumPredict < 0 || req.Options.NumPredict > 10*s.options.NumCtx {
 		req.Options.NumPredict = 10 * s.options.NumCtx
 	}
+
 	// Make sure the server is ready
 	status, err := s.getServerStatusRetry(ctx)
 	if err != nil {
 		return err
 	} else if status != ServerStatusReady {
-		return fmt.Errorf("unexpected server status: %s", status)
+		return fmt.Errorf("unexpected server status: %s", status.ToString())
 	}

 	// Handling JSON marshaling with special characters unescaped.
@@ -746,7 +778,7 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 	enc := json.NewEncoder(buffer)
 	enc.SetEscapeHTML(false)

-	if err := enc.Encode(req); err != nil {
+	if err := enc.Encode(request); err != nil {
 		return fmt.Errorf("failed to marshal data: %v", err)
 	}

@@ -797,7 +829,7 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 				evt = line
 			}

-			var c CompletionResponse
+			var c completion
 			if err := json.Unmarshal(evt, &c); err != nil {
 				return fmt.Errorf("error unmarshalling llm prediction response: %v", err)
 			}
@@ -821,8 +853,20 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 				})
 			}

-			if c.Done {
-				fn(c)
+			if c.Stop {
+				doneReason := "stop"
+				if c.StoppedLimit {
+					doneReason = "length"
+				}
+
+				fn(CompletionResponse{
+					Done:               true,
+					DoneReason:         doneReason,
+					PromptEvalCount:    c.Timings.PromptN,
+					PromptEvalDuration: parseDurationMs(c.Timings.PromptMS),
+					EvalCount:          c.Timings.PredictedN,
+					EvalDuration:       parseDurationMs(c.Timings.PredictedMS),
+				})
 				return nil
 			}
 		}
@@ -870,7 +914,7 @@ func (s *llmServer) Embedding(ctx context.Context, input string) ([]float32, err
 	if err != nil {
 		return nil, err
 	} else if status != ServerStatusReady {
-		return nil, fmt.Errorf("unexpected server status: %s", status)
+		return nil, fmt.Errorf("unexpected server status: %s", status.ToString())
 	}

 	data, err := json.Marshal(EmbeddingRequest{Content: input})
@@ -1015,3 +1059,12 @@ func (s *llmServer) EstimatedVRAMByGPU(gpuID string) uint64 {
 	}
 	return 0
 }
+
+func parseDurationMs(ms float64) time.Duration {
+	dur, err := time.ParseDuration(fmt.Sprintf("%fms", ms))
+	if err != nil {
+		panic(err)
+	}
+
+	return dur
+}
--- a/ml/backend.go
+++ b/ml/backend.go
@@ -2,7 +2,6 @@ package ml

 import (
 	"bytes"
-	"context"
 	"encoding/binary"
 	"fmt"
 	"os"
@@ -61,10 +60,6 @@ type CacheConfig struct {

 // BackendParams controls how the backend loads and executes models
 type BackendParams struct {
-	// Progress is a callback function that allows reporting percentage completion
-	// of model loading
-	Progress func(float32)
-
 	// NumThreads sets the number of threads to use if running on the CPU
 	NumThreads int

@@ -81,9 +76,9 @@ type BackendParams struct {
 	FlashAttention bool
 }

-var backends = make(map[string]func(context.Context, *os.File, BackendParams) (Backend, error))
+var backends = make(map[string]func(*os.File, BackendParams) (Backend, error))

-func RegisterBackend(name string, f func(context.Context, *os.File, BackendParams) (Backend, error)) {
+func RegisterBackend(name string, f func(*os.File, BackendParams) (Backend, error)) {
 	if _, ok := backends[name]; ok {
 		panic("backend: backend already registered")
 	}
@@ -91,9 +86,9 @@ func RegisterBackend(name string, f func(context.Context, *os.File, BackendParam
 	backends[name] = f
 }

-func NewBackend(ctx context.Context, f *os.File, params BackendParams) (Backend, error) {
+func NewBackend(f *os.File, params BackendParams) (Backend, error) {
 	if backend, ok := backends["ggml"]; ok {
-		return backend(ctx, f, params)
+		return backend(f, params)
 	}

 	return nil, fmt.Errorf("unsupported backend")
--- a/ml/backend/ggml/ggml.go
+++ b/ml/backend/ggml/ggml.go
@@ -9,17 +9,15 @@ package ggml
 import "C"

 import (
-	"context"
+	"errors"
 	"fmt"
 	"io"
 	"log/slog"
 	"maps"
 	"os"
-	"runtime"
 	"slices"
 	"strconv"
 	"strings"
-	"sync/atomic"
 	"unicode"
 	"unsafe"

@@ -60,7 +58,7 @@ type Backend struct {
 	maxGraphNodes int
 }

-func New(ctx context.Context, r *os.File, params ml.BackendParams) (ml.Backend, error) {
+func New(r *os.File, params ml.BackendParams) (ml.Backend, error) {
 	meta, n, err := fs.Decode(r, -1)
 	if err != nil {
 		return nil, err
@@ -299,16 +297,12 @@ func New(ctx context.Context, r *os.File, params ml.BackendParams) (ml.Backend,
 		}
 	}

-	var doneBytes atomic.Uint64
-	totalBytes := uint64(n) - meta.Tensors().Offset
-
-	g, ctx := errgroup.WithContext(ctx)
-	g.SetLimit(runtime.GOMAXPROCS(0))
+	// concurrently read in tensor data. uses a section reader which is safe for concurrent reads
+	sr := io.NewSectionReader(r, int64(meta.Tensors().Offset), n-int64(meta.Tensors().Offset))
+	var g errgroup.Group
 	for _, t := range meta.Tensors().Items() {
-		g.Go(func() error {
-			tts := make([]*C.struct_ggml_tensor, max(1, len(targets[t.Name])))
-			for i := range tts {
-				target := targets[t.Name][i]
+		for _, target := range targets[t.Name] {
+			g.Go(func() error {
 				if target == "" {
 					target = t.Name
 				}
@@ -318,44 +312,23 @@ func New(ctx context.Context, r *os.File, params ml.BackendParams) (ml.Backend,
 					return fmt.Errorf("unassigned tensor: %s", t.Name)
 				}

-				tts[i] = tt
-			}
-
-			sr := io.NewSectionReader(r, int64(meta.Tensors().Offset+t.Offset), int64(t.Size()))
-			bts := make([]byte, 128*format.KibiByte)
-
-			var s uint64
-			for s < t.Size() {
-				n, err := io.ReadFull(sr, bts[:min(len(bts), int(t.Size()-s))])
+				bts := make([]byte, t.Size())
+				n, err := io.ReadFull(io.NewSectionReader(sr, int64(t.Offset), int64(t.Size())), bts)
 				if err != nil {
 					return err
 				}

-				for _, tt := range tts {
-					C.ggml_backend_tensor_set(tt, unsafe.Pointer(&bts[0]), C.size_t(s), C.size_t(n))
+				if n != len(bts) {
+					return errors.New("short read")
 				}

-				s += uint64(n)
-
-				if params.Progress != nil {
-					done := doneBytes.Add(uint64(n))
-					params.Progress(float32(done) / float32(totalBytes))
-				}
-			}
-
-			return nil
-		})
+				C.ggml_backend_tensor_set(tt, unsafe.Pointer(&bts[0]), 0, C.size_t(t.Size()))
+				return nil
+			})
+		}
 	}

-	// start a goroutine to cancel the errgroup if the parent context is done
-	go func() {
-		<-ctx.Done()
-		g.Go(func() error {
-			return ctx.Err()
-		})
-	}()
-
-	if err := g.Wait(); err != nil {
+	if g.Wait() != nil {
 		return nil, err
 	}

@@ -398,7 +371,7 @@ func New(ctx context.Context, r *os.File, params ml.BackendParams) (ml.Backend,
 			(*C.ggml_backend_buffer_type_t)(unsafe.Pointer(&schedBufts[0])),
 			C.int(len(schedBackends)),
 			C.size_t(maxGraphNodes),
-			C._Bool(len(gpus) > 1 && slices.Contains(gpus, output.d)),
+			true,
 		),
 		input:  deviceBufferTypes[input.d],
 		output: deviceBufferTypes[output.d],
--- a/model/input/input.go
+++ b/model/input/input.go
@@ -1,7 +1,5 @@
 package input

-import "github.com/ollama/ollama/ml"
-
 // Input represents one token in the input stream
 type Input struct {
 	// Token is a single element of text.
@@ -17,12 +15,6 @@ type Input struct {
 	// stored in Multimodal, used for caching and comparing
 	// equality.
 	MultimodalHash uint64
-
-	// SameBatch forces the following number of tokens to be processed
-	// in a single batch, breaking and extending batches as needed.
-	// Useful for things like images that must be processed in one
-	// shot.
-	SameBatch int
 }

 // MultimodalIndex is a multimodal element (such as an image)
@@ -35,24 +27,11 @@ type MultimodalIndex struct {
 	Multimodal any
 }

-// Batch contains the inputs for a model forward pass
-type Batch struct {
-	// Inputs is the input tokens, including placeholders for multimodal inputs.
-	Inputs ml.Tensor
-
-	// Multimodal is a set of multimodal embeddings previously created by
-	// EncodeMultimodal, along with an index into Inputs. Unused for text-only
-	// models or for batches without multimodal elements.
+// Options contains the inputs for a model forward pass
+type Options struct {
+	Inputs     []int32
 	Multimodal []MultimodalIndex
-
-	// Positions is the position for each Input, relative to its sequence. Equal
-	// in length to Inputs.
-	Positions []int32
-
-	// Sequences is the sequence for each Input. Equal in length to Inputs.
-	Sequences []int
-
-	// Outputs are the set of indicies into Inputs for which output data should
-	// be returned.
-	Outputs []int32
+	Positions  []int32
+	Sequences  []int
+	Outputs    []int32
 }
--- a/model/model.go
+++ b/model/model.go
@@ -1,7 +1,6 @@
 package model

 import (
-	"context"
 	"errors"
 	"fmt"
 	_ "image/jpeg"
@@ -27,7 +26,7 @@ var ErrNoVisionModel = errors.New("this model is missing data required for image

 // Model implements a specific model architecture, defining the forward pass and any model-specific configuration
 type Model interface {
-	Forward(ml.Context, input.Batch) (ml.Tensor, error)
+	Forward(ml.Context, input.Options) (ml.Tensor, error)

 	Backend() ml.Backend
 	Config() config
@@ -61,7 +60,7 @@ type MultimodalProcessor interface {
 	// This function is also responsible for updating MultimodalHash for any Multimodal
 	// that is modified to ensure that there is a unique hash value that accurately
 	// represents the contents.
-	PostTokenize([]input.Input) ([]input.Input, error)
+	PostTokenize(ml.Context, []input.Input) ([]input.Input, error)
 }

 // Base implements the common fields and methods for all models
@@ -95,14 +94,14 @@ func Register(name string, f func(ml.Config) (Model, error)) {
 }

 // New initializes a new model instance with the provided configuration based on the metadata in the model file
-func New(ctx context.Context, modelPath string, params ml.BackendParams) (Model, error) {
+func New(modelPath string, params ml.BackendParams) (Model, error) {
 	r, err := os.Open(modelPath)
 	if err != nil {
 		return nil, err
 	}
 	defer r.Close()

-	b, err := ml.NewBackend(ctx, r, params)
+	b, err := ml.NewBackend(r, params)
 	if err != nil {
 		return nil, err
 	}
@@ -281,30 +280,24 @@ func canNil(t reflect.Type) bool {
 		t.Kind() == reflect.Slice
 }

-func Forward(ctx ml.Context, m Model, inputs []int32, batch input.Batch) (ml.Tensor, error) {
-	if len(batch.Positions) != len(batch.Sequences) {
-		return nil, fmt.Errorf("length of positions (%v) must match length of seqs (%v)", len(batch.Positions), len(batch.Sequences))
+func Forward(ctx ml.Context, m Model, opts input.Options) (ml.Tensor, error) {
+	if len(opts.Positions) != len(opts.Sequences) {
+		return nil, fmt.Errorf("length of positions (%v) must match length of seqs (%v)", len(opts.Positions), len(opts.Sequences))
 	}

-	if len(batch.Positions) < 1 {
+	if len(opts.Positions) < 1 {
 		return nil, errors.New("batch size cannot be less than 1")
 	}

-	var err error
-	batch.Inputs, err = ctx.Input().FromIntSlice(inputs, len(inputs))
-	if err != nil {
-		return nil, err
-	}
-
 	cache := m.Config().Cache
 	if cache != nil {
-		err := cache.StartForward(ctx, batch)
+		err := cache.StartForward(ctx, opts)
 		if err != nil {
 			return nil, err
 		}
 	}

-	t, err := m.Forward(ctx, batch)
+	t, err := m.Forward(ctx, opts)
 	if err != nil {
 		return nil, err
 	}
--- a/model/model_test.go
+++ b/model/model_test.go
@@ -163,7 +163,7 @@ func TestGetTextProcessor(t *testing.T) {

 type notTextProcessorModel struct{}

-func (notTextProcessorModel) Forward(ml.Context, input.Batch) (ml.Tensor, error) {
+func (notTextProcessorModel) Forward(ml.Context, input.Options) (ml.Tensor, error) {
 	panic("unimplemented")
 }

--- a/model/models/gemma2/model.go
+++ b/model/models/gemma2/model.go
@@ -168,18 +168,23 @@ func (l *Layer) Forward(ctx ml.Context, hiddenState, positionIDs, outputs ml.Ten
 	return hiddenState.Add(ctx, residual)
 }

-func (m *Model) Forward(ctx ml.Context, batch input.Batch) (ml.Tensor, error) {
-	positions, err := ctx.Input().FromIntSlice(batch.Positions, len(batch.Positions))
+func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
+	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
 	if err != nil {
 		return nil, err
 	}

-	outputs, err := ctx.Input().FromIntSlice(batch.Outputs, len(batch.Outputs))
+	positions, err := ctx.Input().FromIntSlice(opts.Positions, len(opts.Positions))
 	if err != nil {
 		return nil, err
 	}

-	hiddenState := m.TokenEmbedding.Forward(ctx, batch.Inputs)
+	outputs, err := ctx.Output().FromIntSlice(opts.Outputs, len(opts.Outputs))
+	if err != nil {
+		return nil, err
+	}
+
+	hiddenState := m.TokenEmbedding.Forward(ctx, inputs)
 	hiddenState = hiddenState.Scale(ctx, math.Sqrt(float64(m.Options.hiddenSize)))

 	if len(m.Layers) == gemma27BLayerCount {
@@ -206,7 +211,8 @@ func (m *Model) Forward(ctx ml.Context, batch input.Batch) (ml.Tensor, error) {
 	// final logit softcap
 	hiddenState = hiddenState.Scale(ctx, 1.0/float64(m.Options.finalLogitSoftcap))
 	hiddenState = hiddenState.Tanh(ctx)
-	return hiddenState.Scale(ctx, float64(m.Options.finalLogitSoftcap)), nil
+	hiddenState = hiddenState.Scale(ctx, float64(m.Options.finalLogitSoftcap))
+	return hiddenState.Rows(ctx, outputs), nil
 }

 func init() {
--- a/model/models/gemma3/model.go
+++ b/model/models/gemma3/model.go
@@ -2,9 +2,10 @@ package gemma3

 import (
 	"bytes"
+	"encoding/binary"
+	"hash/fnv"
 	"image"
 	"math"
-	"slices"

 	"github.com/ollama/ollama/kvcache"
 	"github.com/ollama/ollama/ml"
@@ -111,23 +112,36 @@ func (m *Model) EncodeMultimodal(ctx ml.Context, multimodalData []byte) (any, er
 	return visionOutputs, nil
 }

-func (m *Model) PostTokenize(inputs []input.Input) ([]input.Input, error) {
+type imageToken struct {
+	embedding ml.Tensor
+	index     int
+}
+
+func (m *Model) PostTokenize(ctx ml.Context, inputs []input.Input) ([]input.Input, error) {
 	var result []input.Input
+	fnvHash := fnv.New64a()

 	for _, inp := range inputs {
 		if inp.Multimodal == nil {
 			result = append(result, inp)
 		} else {
+			imageInputs := []input.Input{
+				{Token: 108},    // "\n\n"
+				{Token: 255999}, // "<start_of_image>""
+			}
+			result = append(result, imageInputs...)
+
+			// add image embeddings
 			inputMultimodal := inp.Multimodal.(ml.Tensor)

-			result = append(result,
-				input.Input{Token: 108, SameBatch: inputMultimodal.Dim(1) + 3},               // "\n\n"
-				input.Input{Token: 255999},                                                   // "<start_of_image>""
-				input.Input{Multimodal: inputMultimodal, MultimodalHash: inp.MultimodalHash}, // image data is on the first placeholder
-			)
+			for i := range inputMultimodal.Dim(1) {
+				fnvHash.Reset()
+				binary.Write(fnvHash, binary.NativeEndian, inp.MultimodalHash)
+				fnvHash.Write([]byte{byte(i)})

-			// add image token placeholders
-			result = append(result, slices.Repeat([]input.Input{{Token: 0}}, inputMultimodal.Dim(1)-1)...)
+				imageToken := imageToken{embedding: inputMultimodal, index: i}
+				result = append(result, input.Input{Multimodal: imageToken, MultimodalHash: fnvHash.Sum64()})
+			}

 			result = append(result,
 				input.Input{Token: 256000}, // <end_of_image>
@@ -139,18 +153,23 @@ func (m *Model) PostTokenize(inputs []input.Input) ([]input.Input, error) {
 	return result, nil
 }

-func (m *Model) Forward(ctx ml.Context, batch input.Batch) (ml.Tensor, error) {
-	positions, err := ctx.Input().FromIntSlice(batch.Positions, len(batch.Positions))
+func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
+	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
 	if err != nil {
 		return nil, err
 	}

-	outputs, err := ctx.Input().FromIntSlice(batch.Outputs, len(batch.Outputs))
+	positions, err := ctx.Input().FromIntSlice(opts.Positions, len(opts.Positions))
 	if err != nil {
 		return nil, err
 	}

-	return m.TextModel.Forward(ctx, batch.Inputs, positions, outputs, batch, m.Cache), nil
+	outputs, err := ctx.Output().FromIntSlice(opts.Outputs, len(opts.Outputs))
+	if err != nil {
+		return nil, err
+	}
+
+	return m.TextModel.Forward(ctx, inputs, positions, outputs, opts, m.Cache), nil
 }

 func init() {
--- a/model/models/gemma3/model_text.go
+++ b/model/models/gemma3/model_text.go
@@ -171,20 +171,53 @@ func (l *TextLayer) Forward(ctx ml.Context, layer int, hiddenState, positionIDs,
 	return hiddenState.Add(ctx, residual)
 }

-func (m *TextModel) Forward(ctx ml.Context, inputs, positions, outputs ml.Tensor, batch input.Batch, cache kvcache.Cache) ml.Tensor {
+func setImageEmbeddings(ctx ml.Context, hiddenState ml.Tensor, multimodal []input.MultimodalIndex) []int {
+	var embedding ml.Tensor
+	var src, dst, length int
+	var except []int
+
+	for _, image := range multimodal {
+		imageToken := image.Multimodal.(imageToken)
+		imageSrc := imageToken.index
+		imageDst := image.Index
+
+		if embedding == nil {
+			embedding = imageToken.embedding
+			src = imageSrc
+			dst = imageDst
+			length = 1
+		} else if embedding == imageToken.embedding && imageSrc+1 == src && imageDst+1 == dst {
+			src = imageSrc
+			dst = imageDst
+			length++
+		} else if embedding == imageToken.embedding && src+length == imageSrc && dst+length == imageDst {
+			length++
+		} else {
+			visionOutputs := embedding.View(ctx, src*embedding.Stride(1), length*embedding.Dim(0))
+			ctx.Forward(visionOutputs.Copy(ctx, hiddenState.View(ctx, dst*hiddenState.Stride(1), length*hiddenState.Dim(0))))
+
+			embedding = imageToken.embedding
+			src = imageSrc
+			dst = imageDst
+			length = 1
+		}
+
+		except = append(except, imageDst)
+	}
+
+	if embedding != nil {
+		visionOutputs := embedding.View(ctx, src*embedding.Stride(1), length*embedding.Dim(0))
+		ctx.Forward(visionOutputs.Copy(ctx, hiddenState.View(ctx, dst*hiddenState.Stride(1), length*hiddenState.Dim(0))))
+	}
+
+	return except
+}
+
+func (m *TextModel) Forward(ctx ml.Context, inputs, positions, outputs ml.Tensor, opts input.Options, cache kvcache.Cache) ml.Tensor {
 	hiddenState := m.TokenEmbedding.Forward(ctx, inputs)
 	hiddenState = hiddenState.Scale(ctx, math.Sqrt(float64(m.TextOptions.hiddenSize)))

-	// set image embeddings
-	var except []int
-	for _, image := range batch.Multimodal {
-		visionOutputs := image.Multimodal.(ml.Tensor)
-		ctx.Forward(visionOutputs.Copy(ctx, hiddenState.View(ctx, image.Index*hiddenState.Stride(1), visionOutputs.Dim(0)*visionOutputs.Dim(1))))
-
-		for i := range visionOutputs.Dim(1) {
-			except = append(except, image.Index+i)
-		}
-	}
+	except := setImageEmbeddings(ctx, hiddenState, opts.Multimodal)

 	for i, layer := range m.Layers {
 		// gemma alternates between the sliding window (local) and causal (global)
--- a/model/models/llama/model.go
+++ b/model/models/llama/model.go
@@ -13,9 +13,9 @@ import (
 )

 type Options struct {
-	hiddenSize, numHeads, numKVHeads int
-	eps, ropeBase, ropeScale         float32
-	ropeDim                          uint32
+	hiddenSize, numHeads, numKVHeads, headDim int
+	eps, ropeBase, ropeScale                  float32
+	ropeDim                                   uint32
 }

 type Model struct {
@@ -37,6 +37,8 @@ func New(c ml.Config) (model.Model, error) {

 	m := Model{
 		BytePairEncoding: model.NewBytePairEncoding(
+			// TODO: need to set this in the conversion for mistral:
+			// tokenizer.ggml.pretokenizer = [^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]*[\p{Ll}\p{Lm}\p{Lo}\p{M}]+|[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+[\p{Ll}\p{Lm}\p{Lo}\p{M}]*|\p{N}| ?[^\s\p{L}\p{N}]+[\r\n/]*|\s*[\r\n]+|\s+(?!\S)|\s+
 			c.String("tokenizer.ggml.pretokenizer", `(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+`),
 			&model.Vocabulary{
 				Values: c.Strings("tokenizer.ggml.tokens"),
@@ -53,6 +55,7 @@ func New(c ml.Config) (model.Model, error) {
 			hiddenSize: int(c.Uint("embedding_length")),
 			numHeads:   int(c.Uint("attention.head_count")),
 			numKVHeads: int(c.Uint("attention.head_count_kv")),
+			headDim:    int(c.Uint("attention.key_length")),
 			eps:        c.Float("attention.layer_norm_rms_epsilon"),
 			ropeBase:   c.Float("rope.freq_base"),
 			ropeScale:  c.Float("rope.freq_scale", 1),
@@ -75,24 +78,36 @@ type SelfAttention struct {

 func (sa *SelfAttention) Forward(ctx ml.Context, hiddenState, positionIDs ml.Tensor, cache kvcache.Cache, opts *Options) ml.Tensor {
 	batchSize := hiddenState.Dim(1)
-	headDim := opts.hiddenSize / opts.numHeads
 	ropeType := uint32(0)
+	// Get head dimension - use explicit value if available, otherwise calculate
+	headDim := opts.headDim
+	if headDim == 0 {
+		headDim = opts.hiddenSize / opts.numHeads
+	}

+	// Query projection and reshape
 	q := sa.Query.Forward(ctx, hiddenState)
 	q = q.Reshape(ctx, headDim, opts.numHeads, batchSize)
 	q = q.RoPE(ctx, positionIDs, sa.RopeFactors, opts.ropeDim, ropeType, opts.ropeBase, opts.ropeScale)

+	// Key projection and reshape
 	k := sa.Key.Forward(ctx, hiddenState)
 	k = k.Reshape(ctx, headDim, opts.numKVHeads, batchSize)
 	k = k.RoPE(ctx, positionIDs, sa.RopeFactors, opts.ropeDim, ropeType, opts.ropeBase, opts.ropeScale)

+	// Value projection and reshape
 	v := sa.Value.Forward(ctx, hiddenState)
 	v = v.Reshape(ctx, headDim, opts.numKVHeads, batchSize)

+	// Attention computation
 	scaleFactor := 1.0 / math.Sqrt(float64(headDim))
 	kqv := nn.Attention(ctx, q, k, v, scaleFactor, cache)
-	kqv = kqv.Reshape(ctx, opts.hiddenSize, batchSize)

+	// Reshape attention output for final projection
+	outputDim := headDim * opts.numHeads
+	kqv = kqv.Reshape(ctx, outputDim, batchSize)
+
+	// Apply output projection
 	return sa.Output.Forward(ctx, kqv)
 }

@@ -139,18 +154,23 @@ func (l *Layer) Forward(ctx ml.Context, hiddenState, positionIDs, outputs ml.Ten
 	return hiddenState.Add(ctx, residual)
 }

-func (m *Model) Forward(ctx ml.Context, batch input.Batch) (ml.Tensor, error) {
-	positions, err := ctx.Input().FromIntSlice(batch.Positions, len(batch.Positions))
+func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
+	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
 	if err != nil {
 		return nil, err
 	}

-	outputs, err := ctx.Input().FromIntSlice(batch.Outputs, len(batch.Outputs))
+	positions, err := ctx.Input().FromIntSlice(opts.Positions, len(opts.Positions))
 	if err != nil {
 		return nil, err
 	}

-	hiddenState := m.TokenEmbedding.Forward(ctx, batch.Inputs)
+	outputs, err := ctx.Output().FromIntSlice(opts.Outputs, len(opts.Outputs))
+	if err != nil {
+		return nil, err
+	}
+
+	hiddenState := m.TokenEmbedding.Forward(ctx, inputs)

 	for i, layer := range m.Layers {
 		m.Cache.SetLayer(i)
--- a/model/models/mllama/model.go
+++ b/model/models/mllama/model.go
@@ -106,17 +106,17 @@ func (m *Model) EncodeMultimodal(ctx ml.Context, multimodalData []byte) (any, er
 	return m.Projector.Forward(ctx, crossAttentionStates), nil
 }

-func (m *Model) PostTokenize(inputs []input.Input) ([]input.Input, error) {
+func (m *Model) PostTokenize(ctx ml.Context, inputs []input.Input) ([]input.Input, error) {
 	var images []input.Input
 	fnvHash := fnv.New64a()

 	for i := range inputs {
 		if inputs[i].Multimodal == nil {
 			if len(images) > 0 {
-				inputs[i].Multimodal = []ml.Tensor{images[0].Multimodal.(ml.Tensor)}
+				inputs[i].Multimodal = images[0].Multimodal
 				inputs[i].MultimodalHash = images[0].MultimodalHash
 				for j := 1; j < len(images); j++ {
-					inputs[i].Multimodal = append(inputs[i].Multimodal.([]ml.Tensor), images[0].Multimodal.(ml.Tensor))
+					inputs[i].Multimodal = inputs[i].Multimodal.(ml.Tensor).Concat(ctx, images[j].Multimodal.(ml.Tensor), 3)
 					fnvHash.Reset()
 					binary.Write(fnvHash, binary.NativeEndian, inputs[i].MultimodalHash)
 					binary.Write(fnvHash, binary.NativeEndian, inputs[j].MultimodalHash)
@@ -135,27 +135,29 @@ func (m *Model) PostTokenize(inputs []input.Input) ([]input.Input, error) {
 	return inputs, nil
 }

-func (m *Model) Forward(ctx ml.Context, batch input.Batch) (ml.Tensor, error) {
+func (m *Model) Forward(ctx ml.Context, opts input.Options) (ml.Tensor, error) {
 	var crossAttentionStates ml.Tensor
-	if len(batch.Multimodal) > 0 {
-		images := batch.Multimodal[len(batch.Multimodal)-1].Multimodal.([]ml.Tensor)
-		if len(images) > 0 {
-			crossAttentionStates = images[len(images)-1]
-		}
+	if len(opts.Multimodal) > 0 {
+		crossAttentionStates = opts.Multimodal[len(opts.Multimodal)-1].Multimodal.(ml.Tensor)
 	}

-	positions, err := ctx.Input().FromIntSlice(batch.Positions, len(batch.Positions))
+	inputs, err := ctx.Input().FromIntSlice(opts.Inputs, len(opts.Inputs))
 	if err != nil {
 		return nil, err
 	}

-	outputs, err := ctx.Input().FromIntSlice(batch.Outputs, len(batch.Outputs))
+	positions, err := ctx.Input().FromIntSlice(opts.Positions, len(opts.Positions))
+	if err != nil {
+		return nil, err
+	}
+
+	outputs, err := ctx.Output().FromIntSlice(opts.Outputs, len(opts.Outputs))
 	if err != nil {
 		return nil, err
 	}

 	// TODO: attention mask, cross attention mask
-	return m.TextModel.Forward(ctx, batch.Inputs, positions, outputs, nil, crossAttentionStates, nil, m.Cache.(*kvcache.WrapperCache)), nil
+	return m.TextModel.Forward(ctx, inputs, positions, outputs, nil, crossAttentionStates, nil, m.Cache.(*kvcache.WrapperCache)), nil
 }

 func init() {
--- a/model/process_text.go
+++ b/model/process_text.go
@@ -32,8 +32,6 @@ type TextProcessor interface {
 	Encode(s string, addSpecial bool) ([]int32, error)
 	Decode([]int32) (string, error)
 	Is(int32, Special) bool
-
-	Vocab() *Vocabulary
 }

 type Vocabulary struct {
--- a/model/process_text_spm.go
+++ b/model/process_text_spm.go
@@ -49,10 +49,6 @@ func NewSentencePieceModel(pre string, vocab *Vocabulary) SentencePieceModel {
 	}
 }

-func (spm SentencePieceModel) Vocab() *Vocabulary {
-	return spm.vocab
-}
-
 func (spm SentencePieceModel) Is(id int32, special Special) bool {
 	return spm.vocab.Is(id, special)
 }
--- a/model/process_text_test.go
+++ b/model/process_text_test.go
@@ -209,6 +209,326 @@ func TestLlama(t *testing.T) {
 	})
 }

+// tekken loads the Tekken tokenizer for testing
+func tekken(t testing.TB) TextProcessor {
+	t.Helper()
+
+	// Load tokenizer config from mistral-small
+	tokenizerConfigPath := filepath.Join("testdata", "mistral-small", "tokenizer_config.json")
+	configFile, err := os.Open(tokenizerConfigPath)
+	if err != nil {
+		t.Fatal(err)
+	}
+	defer configFile.Close()
+
+	var config struct {
+		AddBosToken bool `json:"add_bos_token"`
+		AddEosToken bool `json:"add_eos_token"`
+		BosToken    struct {
+			Content string `json:"content"`
+		} `json:"bos_token"`
+		EosToken struct {
+			Content string `json:"content"`
+		} `json:"eos_token"`
+	}
+	if err := json.NewDecoder(configFile).Decode(&config); err != nil {
+		t.Fatal(err)
+	}
+
+	// Load tokenizer.json which contains the vocabulary and other settings
+	tokenizerJsonPath := filepath.Join("testdata", "mistral-small", "tokenizer.json")
+	tokenizerFile, err := os.Open(tokenizerJsonPath)
+	if err != nil {
+		t.Fatal(err)
+	}
+	defer tokenizerFile.Close()
+
+	var tokenizerData struct {
+		Model struct {
+			Type   string           `json:"type"`
+			Vocab  map[string]int32 `json:"vocab"`
+			Merges []string         `json:"merges"`
+		} `json:"model"`
+		AddedTokens []struct {
+			Id      int32  `json:"id"`
+			Content string `json:"content"`
+			Special bool   `json:"special"`
+		} `json:"added_tokens"`
+		PreTokenizer struct {
+			Type          string `json:"type"`
+			Pretokenizers []struct {
+				Type    string `json:"type"`
+				Pattern struct {
+					String string `json:"String"`
+				} `json:"pattern"`
+				Behavior string `json:"behavior"`
+			} `json:"pretokenizers"`
+		} `json:"pre_tokenizer"`
+	}
+	if err := json.NewDecoder(tokenizerFile).Decode(&tokenizerData); err != nil {
+		t.Fatal(err)
+	}
+
+	// Extract the pattern from pre_tokenizer if available
+	var pattern string
+	if tokenizerData.PreTokenizer.Type == "Sequence" && len(tokenizerData.PreTokenizer.Pretokenizers) > 0 {
+		pattern = tokenizerData.PreTokenizer.Pretokenizers[0].Pattern.String
+	}
+
+	// Combine regular vocab and added tokens
+	vocab := tokenizerData.Model.Vocab
+
+	// Add special tokens from added_tokens
+	for _, token := range tokenizerData.AddedTokens {
+		vocab[token.Content] = token.Id
+	}
+
+	// Create vocabulary arrays
+	maxId := int32(-1)
+	for _, id := range vocab {
+		if id > maxId {
+			maxId = id
+		}
+	}
+
+	vocabSize := int(maxId + 1)
+	types := make([]uint32, vocabSize)
+	tokens := make([]string, vocabSize)
+	scores := make([]float32, vocabSize)
+
+	for token, id := range vocab {
+		tokens[id] = token
+		types[id] = TOKEN_TYPE_NORMAL
+
+		// Assign appropriate token types for special tokens
+		if token == "<s>" {
+			types[id] = TOKEN_TYPE_CONTROL
+		} else if token == "</s>" {
+			types[id] = TOKEN_TYPE_CONTROL
+		} else if token == "[INST]" || token == "[/INST]" {
+			types[id] = TOKEN_TYPE_CONTROL
+		}
+	}
+
+	// In Tekken, we don't need to load merges separately as they're part of the model
+	var merges []string
+
+	// Create vocabulary object
+	vocabObj := &Vocabulary{
+		Values: tokens,
+		Types:  types,
+		Scores: scores,
+		Merges: merges,
+		BOS:    vocab[config.BosToken.Content],
+		EOS:    vocab[config.EosToken.Content],
+		AddBOS: config.AddBosToken,
+		AddEOS: config.AddEosToken,
+	}
+
+	// Use pattern from tokenizer.json if available
+	if pattern != "" {
+		// Ensure pattern has proper escaping for Go regexp
+		pattern = strings.ReplaceAll(pattern, "p{", "\\p{")
+		return NewBytePairEncoding(pattern, vocabObj)
+	}
+
+	// Fallback pattern if not found
+	return NewBytePairEncoding(
+		`\p{L}+|\p{N}+|[^\s\p{L}\p{N}]+|\s+`,
+		vocabObj,
+	)
+}
+
+func TestTekken(t *testing.T) {
+	// Skip if the test data isn't available
+	if _, err := os.Stat(filepath.Join("testdata", "mistral-small")); os.IsNotExist(err) {
+		t.Skip("Mistral-small test data not available")
+	}
+
+	tokenizer := tekken(t)
+
+	t.Run("whitespace_handling", func(t *testing.T) {
+		t.Parallel()
+
+		// The key difference from SentencePiece is that Tekken doesn't prepend whitespace
+		cases := []struct {
+			input    string
+			expected string
+		}{
+			{" hello", " hello"},
+			{"hello ", "hello "},
+			{"hello world", "hello world"},
+			{" hello world ", " hello world "},
+		}
+
+		for _, tc := range cases {
+			ids, err := tokenizer.Encode(tc.input, false)
+			if err != nil {
+				t.Errorf("Failed to encode %q: %v", tc.input, err)
+				continue
+			}
+
+			decoded, err := tokenizer.Decode(ids)
+			if err != nil {
+				t.Errorf("Failed to decode tokens for %q: %v", tc.input, err)
+				continue
+			}
+
+			if decoded != tc.expected {
+				t.Errorf("Whitespace handling: got %q, want %q", decoded, tc.expected)
+			}
+		}
+	})
+
+	t.Run("chat_templates", func(t *testing.T) {
+		t.Parallel()
+
+		// Test the Tekken chat template format which doesn't have spaces after special tokens
+		templates := []struct {
+			input       string
+			expectSpace bool // whether we expect a space after special tokens
+		}{
+			{"<s>[INST]user message[/INST]", false},
+			{"<s>[INST] user message[/INST]", true},
+			{"<s>[INST]user message [/INST]", true},
+		}
+
+		for _, tc := range templates {
+			ids, err := tokenizer.Encode(tc.input, false)
+			if err != nil {
+				t.Errorf("Failed to encode %q: %v", tc.input, err)
+				continue
+			}
+
+			decoded, err := tokenizer.Decode(ids)
+			if err != nil {
+				t.Errorf("Failed to decode tokens for %q: %v", tc.input, err)
+				continue
+			}
+
+			// Check if there's a space after special tokens
+			hasSpaceAfterINST := strings.Contains(decoded, "[INST] ")
+
+			if hasSpaceAfterINST != tc.expectSpace {
+				t.Errorf("Chat template space handling: got space=%v, want space=%v for %q",
+					hasSpaceAfterINST, tc.expectSpace, tc.input)
+			}
+		}
+	})
+
+	t.Run("special_tokens", func(t *testing.T) {
+		t.Parallel()
+
+		// Test how Tekken handles special tokens
+		cases := []struct {
+			input    string
+			expected []string // We'll check if these tokens are in the decoded output
+		}{
+			{"<s>[INST]hello[/INST]", []string{"<s>", "[INST]", "hello", "[/INST]"}},
+			{"[INST]hello[/INST]</s>", []string{"[INST]", "hello", "[/INST]", "</s>"}},
+			{"<s>[INST]hello[/INST]</s>[INST]again[/INST]", []string{"<s>", "[INST]", "hello", "[/INST]", "</s>", "[INST]", "again", "[/INST]"}},
+		}
+
+		for _, tc := range cases {
+			ids, err := tokenizer.Encode(tc.input, false)
+			if err != nil {
+				t.Errorf("Failed to encode %q: %v", tc.input, err)
+				continue
+			}
+
+			decoded, err := tokenizer.Decode(ids)
+			if err != nil {
+				t.Errorf("Failed to decode tokens for %q: %v", tc.input, err)
+				continue
+			}
+
+			for _, expected := range tc.expected {
+				if !strings.Contains(decoded, expected) {
+					t.Errorf("Special token handling: %q missing in decoded output %q", expected, decoded)
+				}
+			}
+		}
+	})
+
+	t.Run("vocabulary_coverage", func(t *testing.T) {
+		t.Parallel()
+
+		// Tekken has a larger vocabulary, so test coverage of various token types
+		samples := []string{
+			"Hello world!",
+			"This is a test of the Tekken tokenizer.",
+			"It has a considerably larger vocabulary size.",
+			"Special characters: !@#$%^&*()",
+			"Numbers: 1234567890",
+			"Multiple languages: こんにちは 你好 안녕하세요",
+			"Code snippets: def function(): return True",
+		}
+
+		for _, sample := range samples {
+			ids, err := tokenizer.Encode(sample, false)
+			if err != nil {
+				t.Errorf("Failed to encode %q: %v", sample, err)
+				continue
+			}
+
+			decoded, err := tokenizer.Decode(ids)
+			if err != nil {
+				t.Errorf("Failed to decode tokens for %q: %v", sample, err)
+				continue
+			}
+
+			if decoded != sample {
+				t.Errorf("Vocabulary coverage: got %q, want %q", decoded, sample)
+			}
+		}
+	})
+
+	t.Run("splitting_behavior", func(t *testing.T) {
+		t.Parallel()
+
+		// Test the splitting behavior which might differ from SentencePiece
+		cases := map[string][]string{
+			"Hello World!": {"Hello", " World", "!"},
+			"user message": {"user", " message"},
+			"[INST]hello":  {"[INST]", "hello"},
+			"hello[/INST]": {"hello", "[/INST]"},
+		}
+
+		for s, want := range cases {
+			got := slices.Collect(tokenizer.(*BytePairEncoding).split(s))
+			if diff := cmp.Diff(want, got); diff != "" {
+				t.Errorf("Splitting behavior no match (-want +got):\n%s", diff)
+			}
+		}
+	})
+
+	t.Run("full_chat_sequence", func(t *testing.T) {
+		t.Parallel()
+
+		// Test a complete chat sequence with Tekken's format
+		chatSequence := "<s>[INST]user message[/INST]assistant message</s>[INST]new user message[/INST]"
+
+		ids, err := tokenizer.Encode(chatSequence, false)
+		if err != nil {
+			t.Fatalf("Failed to encode chat sequence: %v", err)
+		}
+
+		decoded, err := tokenizer.Decode(ids)
+		if err != nil {
+			t.Fatalf("Failed to decode chat sequence tokens: %v", err)
+		}
+
+		// In Tekken, the whitespace shouldn't be added after special tokens
+		if strings.Contains(decoded, "[INST] ") {
+			t.Errorf("Tekken chat sequence has unexpected space after [INST]: %q", decoded)
+		}
+
+		if strings.Contains(decoded, "[/INST] ") {
+			t.Errorf("Tekken chat sequence has unexpected space after [/INST]: %q", decoded)
+		}
+	})
+}
+
 func BenchmarkBytePairEncoding(b *testing.B) {
 	tokenizer := llama(b)
 	bts, err := os.ReadFile(filepath.Join("testdata", "war-and-peace.txt"))
--- a/runner/llamarunner/runner.go
+++ b/runner/llamarunner/runner.go
@@ -24,7 +24,6 @@ import (

 	"github.com/ollama/ollama/api"
 	"github.com/ollama/ollama/llama"
-	"github.com/ollama/ollama/llm"
 	"github.com/ollama/ollama/runner/common"
 )

@@ -100,7 +99,7 @@ type NewSequenceParams struct {
 	embedding      bool
 }

-func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSequenceParams) (*Sequence, error) {
+func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequenceParams) (*Sequence, error) {
 	s.ready.Wait()

 	startTime := time.Now()
@@ -164,7 +163,7 @@ func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSe
 // inputs processes the prompt and images into a list of inputs
 // by splitting the prompt on [img-<n>] tags, tokenizing text and
 // generating image embeddings for each image
-func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input, error) {
+func (s *Server) inputs(prompt string, images []ImageData) ([]input, error) {
 	var inputs []input
 	var parts []string
 	var matches [][]string
@@ -230,7 +229,7 @@ type Server struct {
 	image *ImageContext

 	// status for external health reporting - loading, ready to serve, etc.
-	status llm.ServerStatus
+	status ServerStatus

 	// current progress on loading the model
 	progress float32
@@ -542,18 +541,75 @@ func (s *Server) processBatch(tokenBatch *llama.Batch, embedBatch *llama.Batch)
 	return nil
 }

+// TODO (jmorganca): use structs from the api package to avoid duplication
+// this way the api acts as a proxy instead of using a different api for the
+// runner
+type Options struct {
+	api.Runner
+
+	NumKeep          int      `json:"n_keep"`
+	Seed             int      `json:"seed"`
+	NumPredict       int      `json:"n_predict"`
+	TopK             int      `json:"top_k"`
+	TopP             float32  `json:"top_p"`
+	MinP             float32  `json:"min_p"`
+	TypicalP         float32  `json:"typical_p"`
+	RepeatLastN      int      `json:"repeat_last_n"`
+	Temperature      float32  `json:"temperature"`
+	RepeatPenalty    float32  `json:"repeat_penalty"`
+	PresencePenalty  float32  `json:"presence_penalty"`
+	FrequencyPenalty float32  `json:"frequency_penalty"`
+	Mirostat         int      `json:"mirostat"`
+	MirostatTau      float32  `json:"mirostat_tau"`
+	MirostatEta      float32  `json:"mirostat_eta"`
+	Stop             []string `json:"stop"`
+}
+
+type ImageData struct {
+	Data          []byte `json:"data"`
+	ID            int    `json:"id"`
+	AspectRatioID int    `json:"aspect_ratio_id"`
+}
+
+type CompletionRequest struct {
+	Prompt      string      `json:"prompt"`
+	Images      []ImageData `json:"image_data"`
+	Grammar     string      `json:"grammar"`
+	CachePrompt bool        `json:"cache_prompt"`
+
+	Options
+}
+
+type Timings struct {
+	PredictedN  int     `json:"predicted_n"`
+	PredictedMS float64 `json:"predicted_ms"`
+	PromptN     int     `json:"prompt_n"`
+	PromptMS    float64 `json:"prompt_ms"`
+}
+
+type CompletionResponse struct {
+	Content string `json:"content"`
+	Stop    bool   `json:"stop"`
+
+	Model        string  `json:"model,omitempty"`
+	Prompt       string  `json:"prompt,omitempty"`
+	StoppedLimit bool    `json:"stopped_limit,omitempty"`
+	PredictedN   int     `json:"predicted_n,omitempty"`
+	PredictedMS  float64 `json:"predicted_ms,omitempty"`
+	PromptN      int     `json:"prompt_n,omitempty"`
+	PromptMS     float64 `json:"prompt_ms,omitempty"`
+
+	Timings Timings `json:"timings"`
+}
+
 func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
-	var req llm.CompletionRequest
+	var req CompletionRequest
+	req.Options = Options(api.DefaultOptions())
 	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
 		http.Error(w, "Bad request", http.StatusBadRequest)
 		return
 	}

-	if req.Options == nil {
-		opts := api.DefaultOptions()
-		req.Options = &opts
-	}
-
 	// Set the headers to indicate streaming
 	w.Header().Set("Content-Type", "application/json")
 	w.Header().Set("Transfer-Encoding", "chunked")
@@ -564,28 +620,26 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 		return
 	}

-	// Extract options from the CompletionRequest
-	samplingParams := llama.SamplingParams{
-		TopK:           req.Options.TopK,
-		TopP:           req.Options.TopP,
-		MinP:           req.Options.MinP,
-		TypicalP:       req.Options.TypicalP,
-		Temp:           req.Options.Temperature,
-		RepeatLastN:    req.Options.RepeatLastN,
-		PenaltyRepeat:  req.Options.RepeatPenalty,
-		PenaltyFreq:    req.Options.FrequencyPenalty,
-		PenaltyPresent: req.Options.PresencePenalty,
-		Mirostat:       req.Options.Mirostat,
-		MirostatTau:    req.Options.MirostatTau,
-		MirostatEta:    req.Options.MirostatEta,
-		Seed:           uint32(req.Options.Seed),
-		Grammar:        req.Grammar,
-	}
+	var samplingParams llama.SamplingParams
+	samplingParams.TopK = req.TopK
+	samplingParams.TopP = req.TopP
+	samplingParams.MinP = req.MinP
+	samplingParams.TypicalP = req.TypicalP
+	samplingParams.Temp = req.Temperature
+	samplingParams.RepeatLastN = req.RepeatLastN
+	samplingParams.PenaltyRepeat = req.RepeatPenalty
+	samplingParams.PenaltyFreq = req.FrequencyPenalty
+	samplingParams.PenaltyPresent = req.PresencePenalty
+	samplingParams.Mirostat = req.Mirostat
+	samplingParams.MirostatTau = req.MirostatTau
+	samplingParams.MirostatEta = req.MirostatEta
+	samplingParams.Seed = uint32(req.Seed)
+	samplingParams.Grammar = req.Grammar

 	seq, err := s.NewSequence(req.Prompt, req.Images, NewSequenceParams{
-		numPredict:     req.Options.NumPredict,
-		stop:           req.Options.Stop,
-		numKeep:        req.Options.NumKeep,
+		numPredict:     req.NumPredict,
+		stop:           req.Stop,
+		numKeep:        req.NumKeep,
 		samplingParams: &samplingParams,
 		embedding:      false,
 	})
@@ -608,7 +662,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	found := false
 	for i, sq := range s.seqs {
 		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, true)
+			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
 			if err != nil {
 				s.mu.Unlock()
 				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
@@ -637,7 +691,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 			return
 		case content, ok := <-seq.responses:
 			if ok {
-				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
+				if err := json.NewEncoder(w).Encode(&CompletionResponse{
 					Content: content,
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -648,17 +702,15 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 				flusher.Flush()
 			} else {
 				// Send the final response
-				doneReason := "stop"
-				if seq.doneReason == "limit" {
-					doneReason = "length"
-				}
-				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
-					Done:               true,
-					DoneReason:         doneReason,
-					PromptEvalCount:    seq.numPromptInputs,
-					PromptEvalDuration: seq.startGenerationTime.Sub(seq.startProcessingTime),
-					EvalCount:          seq.numDecoded,
-					EvalDuration:       time.Since(seq.startGenerationTime),
+				if err := json.NewEncoder(w).Encode(&CompletionResponse{
+					Stop:         true,
+					StoppedLimit: seq.doneReason == "limit",
+					Timings: Timings{
+						PromptN:     seq.numPromptInputs,
+						PromptMS:    float64(seq.startGenerationTime.Sub(seq.startProcessingTime).Milliseconds()),
+						PredictedN:  seq.numDecoded,
+						PredictedMS: float64(time.Since(seq.startGenerationTime).Milliseconds()),
+					},
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode final response: %v", err), http.StatusInternalServerError)
 				}
@@ -669,8 +721,17 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	}
 }

+type EmbeddingRequest struct {
+	Content     string `json:"content"`
+	CachePrompt bool   `json:"cache_prompt"`
+}
+
+type EmbeddingResponse struct {
+	Embedding []float32 `json:"embedding"`
+}
+
 func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {
-	var req llm.EmbeddingRequest
+	var req EmbeddingRequest
 	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
 		http.Error(w, fmt.Sprintf("bad request: %s", err), http.StatusBadRequest)
 		return
@@ -700,7 +761,7 @@ func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {
 	found := false
 	for i, sq := range s.seqs {
 		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, false)
+			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
 			if err != nil {
 				s.mu.Unlock()
 				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
@@ -721,17 +782,41 @@ func (s *Server) embeddings(w http.ResponseWriter, r *http.Request) {

 	embedding := <-seq.embedding

-	if err := json.NewEncoder(w).Encode(&llm.EmbeddingResponse{
+	if err := json.NewEncoder(w).Encode(&EmbeddingResponse{
 		Embedding: embedding,
 	}); err != nil {
 		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
 	}
 }

+type HealthResponse struct {
+	Status   string  `json:"status"`
+	Progress float32 `json:"progress"`
+}
+
+type ServerStatus int
+
+const (
+	ServerStatusReady ServerStatus = iota
+	ServerStatusLoadingModel
+	ServerStatusError
+)
+
+func (s ServerStatus) ToString() string {
+	switch s {
+	case ServerStatusReady:
+		return "ok"
+	case ServerStatusLoadingModel:
+		return "loading model"
+	default:
+		return "server error"
+	}
+}
+
 func (s *Server) health(w http.ResponseWriter, r *http.Request) {
 	w.Header().Set("Content-Type", "application/json")
-	if err := json.NewEncoder(w).Encode(&llm.ServerStatusResponse{
-		Status:   s.status,
+	if err := json.NewEncoder(w).Encode(&HealthResponse{
+		Status:   s.status.ToString(),
 		Progress: s.progress,
 	}); err != nil {
 		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -794,7 +879,7 @@ func (s *Server) loadModel(
 		panic(err)
 	}

-	s.status = llm.ServerStatusReady
+	s.status = ServerStatusReady
 	s.ready.Done()
 }

@@ -852,7 +937,7 @@ func Execute(args []string) error {
 		parallel:  *parallel,
 		seqs:      make([]*Sequence, *parallel),
 		seqsSem:   semaphore.NewWeighted(int64(*parallel)),
-		status:    llm.ServerStatusLoadingModel,
+		status:    ServerStatusLoadingModel,
 	}

 	var tensorSplitFloats []float32
--- a/runner/ollamarunner/cache.go
+++ b/runner/ollamarunner/cache.go
@@ -31,10 +31,8 @@ type InputCache struct {
 	cache kvcache.Cache
 }

-func NewInputCache(model model.Model, kvCacheType string, kvSize int32, numSlots int, batchSize int, multiUserCache bool) (*InputCache, error) {
-	numCtx := kvSize / int32(numSlots)
-
-	if numCtx < 1 {
+func NewInputCache(model model.Model, kvCacheType string, kvSize int32, numSlots int, multiUserCache bool) (*InputCache, error) {
+	if kvSize/int32(numSlots) < 1 {
 		return nil, fmt.Errorf("must have at least one kv cache entry per parallel sequence (kv: %v parallel: %v)", kvSize, numSlots)
 	}

@@ -46,11 +44,11 @@ func NewInputCache(model model.Model, kvCacheType string, kvSize int32, numSlots

 	cache := model.Config().Cache
 	if cache != nil {
-		cache.Init(model.Backend(), kvCacheTypeFromStr(kvCacheType), numSlots, int(numCtx), batchSize)
+		cache.Init(model.Backend(), kvCacheTypeFromStr(kvCacheType), kvSize)
 	}

 	return &InputCache{
-		numCtx:         numCtx,
+		numCtx:         kvSize / int32(numSlots),
 		enabled:        cache != nil,
 		slots:          slots,
 		multiUserCache: multiUserCache,
@@ -91,7 +89,7 @@ type InputCacheSlot struct {
 	lastUsed time.Time
 }

-func (c *InputCache) LoadCacheSlot(prompt []input.Input) (*InputCacheSlot, []input.Input, error) {
+func (c *InputCache) LoadCacheSlot(prompt []input.Input, cachePrompt bool) (*InputCacheSlot, []input.Input, error) {
 	var slot *InputCacheSlot
 	var numPast int32
 	var err error
@@ -109,6 +107,10 @@ func (c *InputCache) LoadCacheSlot(prompt []input.Input) (*InputCacheSlot, []inp
 		return nil, nil, err
 	}

+	if !cachePrompt {
+		numPast = 0
+	}
+
 	slot.InUse = true
 	slot.lastUsed = time.Now()

--- a/runner/ollamarunner/cache_test.go
+++ b/runner/ollamarunner/cache_test.go
@@ -297,131 +297,3 @@ func TestShiftDiscard(t *testing.T) {
 		})
 	}
 }
-
-func TestLoadCacheSlot(t *testing.T) {
-	tests := []struct {
-		name           string
-		cache          InputCache
-		prompt         []input.Input
-		wantErr        bool
-		expectedSlotId int
-		expectedPrompt int // expected length of remaining prompt
-	}{
-		{
-			name: "Basic cache hit - single user",
-			cache: InputCache{
-				multiUserCache: false,
-				slots: []InputCacheSlot{
-					{
-						Id:       0,
-						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
-						InUse:    false,
-						lastUsed: time.Now().Add(-time.Second),
-					},
-					{
-						Id:       1,
-						Inputs:   []input.Input{},
-						InUse:    false,
-						lastUsed: time.Now().Add(-2 * time.Second),
-					},
-				},
-			},
-			prompt:         []input.Input{{Token: 1}, {Token: 2}, {Token: 3}},
-			wantErr:        false,
-			expectedSlotId: 0,
-			expectedPrompt: 1, // Only token 3 remains
-		},
-		{
-			name: "Basic cache hit - multi user",
-			cache: InputCache{
-				multiUserCache: true,
-				slots: []InputCacheSlot{
-					{
-						Id:       0,
-						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
-						InUse:    false,
-						lastUsed: time.Now().Add(-time.Second),
-					},
-					{
-						Id:       1,
-						Inputs:   []input.Input{},
-						InUse:    false,
-						lastUsed: time.Now().Add(-2 * time.Second),
-					},
-				},
-			},
-			prompt:         []input.Input{{Token: 1}, {Token: 2}, {Token: 3}},
-			wantErr:        false,
-			expectedSlotId: 0,
-			expectedPrompt: 1, // Only token 3 remains
-		},
-		{
-			name: "Exact match - leave one input",
-			cache: InputCache{
-				multiUserCache: false,
-				slots: []InputCacheSlot{
-					{
-						Id:       0,
-						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
-						InUse:    false,
-						lastUsed: time.Now().Add(-time.Second),
-					},
-				},
-			},
-			prompt:         []input.Input{{Token: 1}, {Token: 2}},
-			wantErr:        false,
-			expectedSlotId: 0,
-			expectedPrompt: 1, // Should leave 1 token for sampling
-		},
-		{
-			name: "No available slots",
-			cache: InputCache{
-				multiUserCache: false,
-				slots: []InputCacheSlot{
-					{
-						Id:       0,
-						Inputs:   []input.Input{{Token: 1}, {Token: 2}},
-						InUse:    true,
-						lastUsed: time.Now().Add(-time.Second),
-					},
-				},
-			},
-			prompt:         []input.Input{{Token: 1}, {Token: 2}, {Token: 3}},
-			wantErr:        true,
-			expectedSlotId: -1,
-			expectedPrompt: -1,
-		},
-	}
-
-	for _, tt := range tests {
-		t.Run(tt.name, func(t *testing.T) {
-			slot, remainingPrompt, err := tt.cache.LoadCacheSlot(tt.prompt)
-
-			// Check error state
-			if (err != nil) != tt.wantErr {
-				t.Errorf("LoadCacheSlot() error = %v, wantErr %v", err, tt.wantErr)
-				return
-			}
-
-			if tt.wantErr {
-				return // Skip further checks if we expected an error
-			}
-
-			// Verify slot ID
-			if slot.Id != tt.expectedSlotId {
-				t.Errorf("LoadCacheSlot() slot ID = %v, expected %v", slot.Id, tt.expectedSlotId)
-			}
-
-			// Verify slot is now marked in use
-			if !slot.InUse {
-				t.Errorf("LoadCacheSlot() slot not marked InUse")
-			}
-
-			// Verify remaining prompt length
-			if len(remainingPrompt) != tt.expectedPrompt {
-				t.Errorf("LoadCacheSlot() remaining prompt length = %v, expected %v",
-					len(remainingPrompt), tt.expectedPrompt)
-			}
-		})
-	}
-}
--- a/runner/ollamarunner/runner.go
+++ b/runner/ollamarunner/runner.go
@@ -24,7 +24,6 @@ import (
 	"golang.org/x/sync/semaphore"

 	"github.com/ollama/ollama/api"
-	"github.com/ollama/ollama/llm"
 	"github.com/ollama/ollama/ml"
 	"github.com/ollama/ollama/model"
 	"github.com/ollama/ollama/model/input"
@@ -34,14 +33,10 @@ import (
 	_ "github.com/ollama/ollama/model/models"
 )

-type contextList struct {
-	list []ml.Context
-}
-
 type Sequence struct {
-	// ctxs are used for allocating tensors that last the lifetime of the sequence, such as
+	// ctx for allocating tensors that last the lifetime of the sequence, such as
 	// multimodal embeddings
-	ctxs *contextList
+	ctx ml.Context

 	// batch index
 	iBatch int
@@ -99,12 +94,13 @@ type NewSequenceParams struct {
 	embedding  bool
 }

-func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSequenceParams) (*Sequence, error) {
+func (s *Server) NewSequence(prompt string, images []ImageData, params NewSequenceParams) (*Sequence, error) {
 	s.ready.Wait()

 	startTime := time.Now()
+	ctx := s.model.Backend().NewContext()

-	inputs, ctxs, err := s.inputs(prompt, images)
+	inputs, err := s.inputs(ctx, prompt, images)
 	if err != nil {
 		return nil, fmt.Errorf("failed to process inputs: %w", err)
 	} else if len(inputs) == 0 {
@@ -115,9 +111,6 @@ func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSe
 		params.numKeep = int32(len(inputs))
 	}

-	// TODO(jessegross): We should ensure that we always leave minBatch of context space to shift,
-	// otherwise we might truncate or split the batch against the model's wishes
-
 	// Ensure that at least 1 input can be discarded during shift
 	params.numKeep = min(params.numKeep, s.cache.numCtx-1)

@@ -133,7 +126,7 @@ func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSe
 	// TODO(jessegross): Ingest cached history for grammar

 	return &Sequence{
-		ctxs:                ctxs,
+		ctx:                 ctx,
 		inputs:              inputs,
 		numPromptInputs:     len(inputs),
 		startProcessingTime: startTime,
@@ -152,7 +145,7 @@ func (s *Server) NewSequence(prompt string, images []llm.ImageData, params NewSe
 // inputs processes the prompt and images into a list of inputs
 // by splitting the prompt on [img-<n>] tags, tokenizing text and
 // decoding images
-func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input.Input, *contextList, error) {
+func (s *Server) inputs(ctx ml.Context, prompt string, images []ImageData) ([]input.Input, error) {
 	var inputs []input.Input
 	var parts []string
 	var matches [][]string
@@ -167,19 +160,12 @@ func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input.Input, *
 		parts = []string{prompt}
 	}

-	var contexts contextList
-	runtime.AddCleanup(&contexts, func(ctxs []ml.Context) {
-		for _, ctx := range ctxs {
-			ctx.Close()
-		}
-	}, contexts.list)
-
 	postTokenize := false
 	for i, part := range parts {
 		// text - tokenize
 		tokens, err := s.model.(model.TextProcessor).Encode(part, i == 0)
 		if err != nil {
-			return nil, nil, err
+			return nil, err
 		}

 		for _, t := range tokens {
@@ -199,14 +185,12 @@ func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input.Input, *
 			}

 			if imageIndex < 0 {
-				return nil, nil, fmt.Errorf("invalid image index: %d", n)
+				return nil, fmt.Errorf("invalid image index: %d", n)
 			}

-			ctx := s.model.Backend().NewContext()
-			contexts.list = append(contexts.list, ctx)
 			imageEmbeddings, err := multimodalProcessor.EncodeMultimodal(ctx, images[imageIndex].Data)
 			if err != nil {
-				return nil, nil, err
+				return nil, err
 			}

 			s.multimodalHash.Reset()
@@ -220,13 +204,13 @@ func (s *Server) inputs(prompt string, images []llm.ImageData) ([]input.Input, *

 	if visionModel && postTokenize {
 		var err error
-		inputs, err = multimodalProcessor.PostTokenize(inputs)
+		inputs, err = multimodalProcessor.PostTokenize(ctx, inputs)
 		if err != nil {
-			return nil, nil, err
+			return nil, err
 		}
 	}

-	return inputs, &contexts, nil
+	return inputs, nil
 }

 type Server struct {
@@ -238,7 +222,7 @@ type Server struct {
 	model model.Model

 	// status for external health reporting - loading, ready to serve, etc.
-	status llm.ServerStatus
+	status ServerStatus

 	// current progress on loading the model
 	progress float32
@@ -321,6 +305,7 @@ func (s *Server) removeSequence(seqIndex int, reason string) {
 	close(seq.responses)
 	close(seq.embedding)
 	seq.cache.InUse = false
+	seq.ctx.Close()
 	s.seqs[seqIndex] = nil
 	s.seqsSem.Release(1)
 }
@@ -348,8 +333,7 @@ func (s *Server) processBatch() error {
 	}
 	defer s.mu.Unlock()

-	var batchInputs []int32
-	var batch input.Batch
+	var options input.Options

 	for i, seq := range s.seqs {
 		if seq == nil {
@@ -367,46 +351,33 @@ func (s *Server) processBatch() error {
 			seq.cache.Inputs = []input.Input{}
 		}

-		batchSize := s.batchSize
-
 		for j, inp := range seq.inputs {
-			// If we are required to put following inputs into a single batch then extend the
-			// batch size. Since we are only extending the size the minimum amount possible, this
-			// will cause a break if we have pending inputs.
-			minBatch := 1 + inp.SameBatch
-			if minBatch > batchSize {
-				batchSize = minBatch
+			if int32(len(seq.cache.Inputs)+len(seq.pendingInputs)+1) > s.cache.numCtx {
+				if len(seq.pendingInputs) == 0 {
+					err := s.cache.ShiftCacheSlot(seq.cache, seq.numKeep)
+					if err != nil {
+						return err
+					}
+				} else {
+					break
+				}
 			}

-			if len(seq.pendingInputs)+minBatch > batchSize {
+			if j >= s.batchSize {
 				break
 			}

-			// If the sum of our working set (already processed tokens, tokens we added to this
-			// batch, required following tokens) exceeds the context size, then trigger a shift
-			// now so we don't have to do one later when we can't break the batch.
-			if int32(len(seq.cache.Inputs)+len(seq.pendingInputs)+minBatch) > s.cache.numCtx {
-				if len(seq.pendingInputs) != 0 {
-					break
-				}
-
-				err := s.cache.ShiftCacheSlot(seq.cache, seq.numKeep)
-				if err != nil {
-					return err
-				}
-			}
-
-			batchInputs = append(batchInputs, inp.Token)
+			options.Inputs = append(options.Inputs, inp.Token)
 			if inp.Multimodal != nil {
-				batch.Multimodal = append(batch.Multimodal, input.MultimodalIndex{Index: len(batchInputs) - 1, Multimodal: inp.Multimodal})
+				options.Multimodal = append(options.Multimodal, input.MultimodalIndex{Index: len(options.Inputs) - 1, Multimodal: inp.Multimodal})
 			}

-			batch.Positions = append(batch.Positions, int32(len(seq.cache.Inputs)+len(seq.pendingInputs)))
-			batch.Sequences = append(batch.Sequences, seq.cache.Id)
+			options.Positions = append(options.Positions, int32(len(seq.cache.Inputs)+len(seq.pendingInputs)))
+			options.Sequences = append(options.Sequences, seq.cache.Id)

-			seq.iBatch = len(batch.Outputs)
+			seq.iBatch = len(options.Outputs)
 			if j+1 == len(seq.inputs) {
-				batch.Outputs = append(batch.Outputs, int32(len(batchInputs)-1))
+				options.Outputs = append(options.Outputs, int32(len(options.Inputs)-1))
 			}
 			seq.pendingInputs = append(seq.pendingInputs, inp)
 		}
@@ -414,14 +385,14 @@ func (s *Server) processBatch() error {
 		seq.inputs = seq.inputs[len(seq.pendingInputs):]
 	}

-	if len(batchInputs) == 0 {
+	if len(options.Inputs) == 0 {
 		return nil
 	}

 	ctx := s.model.Backend().NewContext()
 	defer ctx.Close()

-	modelOutput, err := model.Forward(ctx, s.model, batchInputs, batch)
+	modelOutput, err := model.Forward(ctx, s.model, options)
 	if err != nil {
 		return fmt.Errorf("failed to decode batch: %w", err)
 	}
@@ -461,7 +432,7 @@ func (s *Server) processBatch() error {
 		}

 		// sample a token
-		vocabSize := len(logits) / len(batch.Outputs)
+		vocabSize := len(logits) / len(options.Outputs)

 		token, err := seq.sampler.Sample(logits[seq.iBatch*vocabSize : (seq.iBatch+1)*vocabSize])
 		if err != nil {
@@ -530,18 +501,75 @@ func (s *Server) processBatch() error {
 	return nil
 }

+// TODO (jmorganca): use structs from the api package to avoid duplication
+// this way the api acts as a proxy instead of using a different api for the
+// runner
+type Options struct {
+	api.Runner
+
+	NumKeep          int      `json:"n_keep"`
+	Seed             int      `json:"seed"`
+	NumPredict       int      `json:"n_predict"`
+	TopK             int      `json:"top_k"`
+	TopP             float32  `json:"top_p"`
+	MinP             float32  `json:"min_p"`
+	TypicalP         float32  `json:"typical_p"`
+	RepeatLastN      int      `json:"repeat_last_n"`
+	Temperature      float32  `json:"temperature"`
+	RepeatPenalty    float32  `json:"repeat_penalty"`
+	PresencePenalty  float32  `json:"presence_penalty"`
+	FrequencyPenalty float32  `json:"frequency_penalty"`
+	Mirostat         int      `json:"mirostat"`
+	MirostatTau      float32  `json:"mirostat_tau"`
+	MirostatEta      float32  `json:"mirostat_eta"`
+	Stop             []string `json:"stop"`
+}
+
+type ImageData struct {
+	Data          []byte `json:"data"`
+	ID            int    `json:"id"`
+	AspectRatioID int    `json:"aspect_ratio_id"`
+}
+
+type CompletionRequest struct {
+	Prompt      string      `json:"prompt"`
+	Images      []ImageData `json:"image_data"`
+	Grammar     string      `json:"grammar"`
+	CachePrompt bool        `json:"cache_prompt"`
+
+	Options
+}
+
+type Timings struct {
+	PredictedN  int     `json:"predicted_n"`
+	PredictedMS float64 `json:"predicted_ms"`
+	PromptN     int     `json:"prompt_n"`
+	PromptMS    float64 `json:"prompt_ms"`
+}
+
+type CompletionResponse struct {
+	Content string `json:"content"`
+	Stop    bool   `json:"stop"`
+
+	Model        string  `json:"model,omitempty"`
+	Prompt       string  `json:"prompt,omitempty"`
+	StoppedLimit bool    `json:"stopped_limit,omitempty"`
+	PredictedN   int     `json:"predicted_n,omitempty"`
+	PredictedMS  float64 `json:"predicted_ms,omitempty"`
+	PromptN      int     `json:"prompt_n,omitempty"`
+	PromptMS     float64 `json:"prompt_ms,omitempty"`
+
+	Timings Timings `json:"timings"`
+}
+
 func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
-	var req llm.CompletionRequest
+	var req CompletionRequest
+	req.Options = Options(api.DefaultOptions())
 	if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
 		http.Error(w, "Bad request", http.StatusBadRequest)
 		return
 	}

-	if req.Options == nil {
-		opts := api.DefaultOptions()
-		req.Options = &opts
-	}
-
 	// Set the headers to indicate streaming
 	w.Header().Set("Content-Type", "application/json")
 	w.Header().Set("Transfer-Encoding", "chunked")
@@ -563,18 +591,18 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	}

 	sampler := sample.NewSampler(
-		req.Options.Temperature,
-		req.Options.TopK,
-		req.Options.TopP,
-		req.Options.MinP,
-		req.Options.Seed,
+		req.Temperature,
+		req.TopK,
+		req.TopP,
+		req.MinP,
+		req.Seed,
 		grammar,
 	)

 	seq, err := s.NewSequence(req.Prompt, req.Images, NewSequenceParams{
-		numPredict: req.Options.NumPredict,
-		stop:       req.Options.Stop,
-		numKeep:    int32(req.Options.NumKeep),
+		numPredict: req.NumPredict,
+		stop:       req.Stop,
+		numKeep:    int32(req.NumKeep),
 		sampler:    sampler,
 		embedding:  false,
 	})
@@ -597,7 +625,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	found := false
 	for i, sq := range s.seqs {
 		if sq == nil {
-			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs)
+			seq.cache, seq.inputs, err = s.cache.LoadCacheSlot(seq.inputs, req.CachePrompt)
 			if err != nil {
 				s.mu.Unlock()
 				http.Error(w, fmt.Sprintf("Failed to load cache: %v", err), http.StatusInternalServerError)
@@ -624,7 +652,7 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 			return
 		case content, ok := <-seq.responses:
 			if ok {
-				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
+				if err := json.NewEncoder(w).Encode(&CompletionResponse{
 					Content: content,
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -635,17 +663,15 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 				flusher.Flush()
 			} else {
 				// Send the final response
-				doneReason := "stop"
-				if seq.doneReason == "limit" {
-					doneReason = "length"
-				}
-				if err := json.NewEncoder(w).Encode(&llm.CompletionResponse{
-					Done:               true,
-					DoneReason:         doneReason,
-					PromptEvalCount:    seq.numPromptInputs,
-					PromptEvalDuration: seq.startGenerationTime.Sub(seq.startProcessingTime),
-					EvalCount:          seq.numPredicted,
-					EvalDuration:       time.Since(seq.startGenerationTime),
+				if err := json.NewEncoder(w).Encode(&CompletionResponse{
+					Stop:         true,
+					StoppedLimit: seq.doneReason == "limit",
+					Timings: Timings{
+						PromptN:     seq.numPromptInputs,
+						PromptMS:    float64(seq.startGenerationTime.Sub(seq.startProcessingTime).Milliseconds()),
+						PredictedN:  seq.numPredicted,
+						PredictedMS: float64(time.Since(seq.startGenerationTime).Milliseconds()),
+					},
 				}); err != nil {
 					http.Error(w, fmt.Sprintf("failed to encode final response: %v", err), http.StatusInternalServerError)
 				}
@@ -656,10 +682,43 @@ func (s *Server) completion(w http.ResponseWriter, r *http.Request) {
 	}
 }

+type EmbeddingRequest struct {
+	Content     string `json:"content"`
+	CachePrompt bool   `json:"cache_prompt"`
+}
+
+type EmbeddingResponse struct {
+	Embedding []float32 `json:"embedding"`
+}
+
+type HealthResponse struct {
+	Status   string  `json:"status"`
+	Progress float32 `json:"progress"`
+}
+
+type ServerStatus int
+
+const (
+	ServerStatusReady ServerStatus = iota
+	ServerStatusLoadingModel
+	ServerStatusError
+)
+
+func (s ServerStatus) ToString() string {
+	switch s {
+	case ServerStatusReady:
+		return "ok"
+	case ServerStatusLoadingModel:
+		return "loading model"
+	default:
+		return "server error"
+	}
+}
+
 func (s *Server) health(w http.ResponseWriter, r *http.Request) {
 	w.Header().Set("Content-Type", "application/json")
-	if err := json.NewEncoder(w).Encode(&llm.ServerStatusResponse{
-		Status:   s.status,
+	if err := json.NewEncoder(w).Encode(&HealthResponse{
+		Status:   s.status.ToString(),
 		Progress: s.progress,
 	}); err != nil {
 		http.Error(w, fmt.Sprintf("failed to encode response: %v", err), http.StatusInternalServerError)
@@ -678,7 +737,6 @@ func (m *multiLPath) String() string {
 }

 func (s *Server) loadModel(
-	ctx context.Context,
 	mpath string,
 	params ml.BackendParams,
 	lpath multiLPath,
@@ -688,7 +746,7 @@ func (s *Server) loadModel(
 	multiUserCache bool,
 ) {
 	var err error
-	s.model, err = model.New(ctx, mpath, params)
+	s.model, err = model.New(mpath, params)
 	if err != nil {
 		panic(err)
 	}
@@ -700,7 +758,7 @@ func (s *Server) loadModel(
 		panic("loras are not yet implemented")
 	}

-	s.cache, err = NewInputCache(s.model, kvCacheType, int32(kvSize), parallel, s.batchSize, multiUserCache)
+	s.cache, err = NewInputCache(s.model, kvCacheType, int32(kvSize), parallel, multiUserCache)
 	if err != nil {
 		panic(err)
 	}
@@ -714,7 +772,7 @@ func (s *Server) loadModel(
 	s.seqs = make([]*Sequence, s.parallel)
 	s.seqsSem = semaphore.NewWeighted(int64(s.parallel))

-	s.status = llm.ServerStatusReady
+	s.status = ServerStatusReady
 	s.ready.Done()
 }

@@ -766,7 +824,7 @@ func Execute(args []string) error {

 	server := &Server{
 		batchSize: *batchSize,
-		status:    llm.ServerStatusLoadingModel,
+		status:    ServerStatusLoadingModel,
 	}

 	// TODO(jessegross): Parameters that need to be implemented:
@@ -784,9 +842,6 @@ func Execute(args []string) error {
 	}

 	params := ml.BackendParams{
-		Progress: func(progress float32) {
-			server.progress = progress
-		},
 		NumThreads:     *threads,
 		NumGPULayers:   *numGPULayers,
 		MainGPU:        *mainGPU,
@@ -795,13 +850,13 @@ func Execute(args []string) error {
 	}

 	server.ready.Add(1)
-	ctx, cancel := context.WithCancel(context.Background())
-	defer cancel()
-
-	go server.loadModel(ctx, *mpath, params, lpaths, *parallel, *kvCacheType, *kvSize, *multiUserCache)
+	go server.loadModel(*mpath, params, lpaths, *parallel, *kvCacheType, *kvSize, *multiUserCache)

 	server.cond = sync.NewCond(&server.mu)

+	ctx, cancel := context.WithCancel(context.Background())
+	defer cancel()
+
 	go server.run(ctx)

 	addr := "127.0.0.1:" + strconv.Itoa(*port)
--- a/sample/samplers.go
+++ b/sample/samplers.go
@@ -26,10 +26,6 @@ type Sampler struct {
 }

 func (s *Sampler) Sample(logits []float32) (int32, error) {
-	if len(logits) == 0 {
-		return -1, errors.New("sample: no logits provided to sample")
-	}
-
 	tokens := make([]token, len(logits))
 	for i := range logits {
 		tokens[i].id = int32(i)
@@ -91,13 +87,19 @@ func (s *Sampler) sample(tokens []token) (token, error) {
 	// topK also sorts the tokens in descending order of logits
 	tokens = topK(tokens, s.topK)

-	// scale and normalize the tokens in place
-	temperature(tokens, s.temperature)
-	softmax(tokens)
+	tokens = temperature(tokens, s.temperature)
+	tokens = softmax(tokens)

 	tokens = topP(tokens, s.topP)
 	tokens = minP(tokens, s.minP)

+	// TODO: this should fall back to greedy sampling
+	// or topP, topK values etc should be such that
+	// there are always tokens to sample from
+	if len(tokens) == 0 {
+		return token{}, errors.New("no tokens to sample from")
+	}
+
 	var r float32
 	if s.rng != nil {
 		r = s.rng.Float32()
@@ -120,9 +122,6 @@ func (s *Sampler) sample(tokens []token) (token, error) {
 		return 1
 	})

-	if math.IsNaN(float64(sum)) {
-		return token{}, errors.New("sample: logits sum to NaN, check model output")
-	}
 	return tokens[idx], nil
 }

--- a/sample/samplers_test.go
+++ b/sample/samplers_test.go
@@ -1,7 +1,6 @@
 package sample

 import (
-	"math"
 	"math/rand/v2"
 	"testing"
 )
@@ -30,29 +29,6 @@ func TestWeighted(t *testing.T) {
 	if want != got {
 		t.Errorf("index mismatch: want %d, got %d", want, got)
 	}
-
-	// Test very high p
-	logits = []float32{1.0, 0.9999999999999999, 0.5, 0.1}
-	// Use extremely small topP to filter out all tokens
-	sampler = NewSampler(1.0, 0, 1e-10, 0, 0, nil)
-	got, err = sampler.Sample(logits)
-	if err != nil {
-		t.Error(err)
-		return
-	}
-	// Should get the token with the highest logit
-	want = int32(0)
-	if want != got {
-		t.Errorf("index mismatch: want %d, got %d", want, got)
-	}
-
-	logits = []float32{float32(math.NaN()), float32(math.NaN()), float32(math.NaN())}
-	sampler = NewSampler(1, 0, 0.95, 0.05, 0, nil)
-	got, err = sampler.Sample(logits)
-	if err == nil {
-		t.Errorf("expected error, got %d", got)
-		return
-	}
 }

 func BenchmarkSample(b *testing.B) {
--- a/sample/state_machine.go
+++ b/sample/state_machine.go
@@ -1,176 +0,0 @@
-package sample
-
-import (
-	"bytes"
-	"strings"
-
-	"github.com/ollama/ollama/model"
-)
-
-type Node struct {
-	TransitionEdges map[rune]*Node
-}
-
-type Graph struct {
-	proc        model.TextProcessor
-	decodedToks []string
-	curNode     *Node
-	grammar     []byte
-	rules       map[string]string
-}
-
-// baseRules is the set of rules that are used to parse the grammar
-// JSON grammar from RFC 7159
-var baseRules = map[string]string{
-	"object":  "\"{\" (kv (\",\" kv)*)? \"}\"",
-	"array":   "\"[\" (value (\",\" value)*)? \"]\"",
-	"string":  "\"\\\"\" char* \"\\\"\"",
-	"number":  "\"-\"? integer frac? exp?",
-	"kv":      "string \":\" value",
-	"integer": "\"0\" | [1-9] [0-9]*",
-	"frac":    "\".\" [0-9]+",
-	"exp":     "(\"e\" | \"E\") (\"+\" | \"-\") [0-9]+",
-	"escape":  "[\"/\" | \"b\" | \"f\" | \"n\" | \"r\" | \"t\" | unicode]",
-	"char":    "[^\"\\\\] | escape",
-	"space":   "(\" \" | \"\\t\" | \"\\n\" | \"\\r\")*",
-	"hex":     "[0-9] | [a-f] | [A-F]",
-	"boolean": "\"true\" | \"false\"",
-	"value":   "object | array | string | number | boolean | \"null\"",
-	"null":    "\"null\"",
-}
-
-func (g *Graph) BuildGraph(node *Node) error {
-	vocab := g.proc.Vocab()
-	decodedToks := make([]string, len(vocab.Values))
-	for i := range vocab.Values {
-		token, err := g.proc.Decode([]int32{int32(i)})
-		if err != nil {
-			return err
-		}
-		decodedToks[i] = token
-	}
-
-	g.decodedToks = decodedToks
-	g.rules = baseRules
-	g.rootPrefixes()
-	rootNode := &Node{
-		TransitionEdges: make(map[rune]*Node),
-	}
-	g.parseRule(g.rules["root"], rootNode)
-
-	return nil
-}
-
-// rootPrefixes extracts all root prefixes from the grammar
-// and parses the grammar string to extract root prefixes
-func (g *Graph) rootPrefixes() {
-	lines := bytes.Split(g.grammar, []byte("\n"))
-	for _, line := range lines {
-		line = bytes.TrimSpace(line)
-		if len(line) == 0 || bytes.HasPrefix(line, []byte("#")) {
-			continue
-		}
-
-		parts := bytes.SplitN(line, []byte("::="), 2)
-		if len(parts) != 2 {
-			continue
-		}
-
-		ruleName := string(bytes.TrimSpace(parts[0]))
-		if strings.HasPrefix(ruleName, "root") {
-			g.rules[ruleName] = string(bytes.TrimSpace(parts[1]))
-		}
-	}
-}
-
-// parseRule parses a grammar rule and returns a Node
-func (g *Graph) parseRule(rule string, curNode *Node) *Node {
-	/*
-		Here are the special characters in BNF grammar and their functions:
-		::= - Definition operator, means "is defined as"
-		| - Alternation, means "or"
-		* - Zero or more repetitions of preceding element
-		+ - One or more repetitions
-		? - Optional (zero or one occurrence)
-		[] - Character class, matches any single character within brackets
-		[^] - Negated character class, matches any character NOT listed
-		() - Grouping of elements
-		- - Range operator in character classes (e.g., [a-z])
-		"" - Literal string match
-	*/
-
-	// Split rule into tokens by whitespace
-	tokens := strings.Fields(rule)
-	if len(tokens) == 0 {
-		return &Node{
-			TransitionEdges: make(map[rune]*Node),
-		}
-	}
-
-	// Handle integer rule
-	if strings.Contains(rule, "[0-9]+") {
-		// Create node for first digit 1-9
-		firstDigitNode := &Node{
-			TransitionEdges: make(map[rune]*Node),
-		}
-		for r := '1'; r <= '9'; r++ {
-			curNode.TransitionEdges[r] = firstDigitNode
-		}
-
-		// Create node for subsequent digits 0-9
-		zeroToNineNode := &Node{
-			TransitionEdges: make(map[rune]*Node),
-		}
-		for r := '0'; r <= '9'; r++ {
-			// Loop back to same node for * operator
-			zeroToNineNode.TransitionEdges[r] = zeroToNineNode
-		}
-
-		// Connect first digit to subsequent digits
-		firstDigitNode.TransitionEdges = zeroToNineNode.TransitionEdges
-
-		// Also handle the "0" case
-		if strings.Contains(rule, "\"0\"") {
-			zeroNode := &Node{
-				TransitionEdges: make(map[rune]*Node),
-			}
-			curNode.TransitionEdges['0'] = zeroNode
-		}
-
-		return curNode
-	}
-
-	// recursive case
-	// grammar options
-	// TODO: handle left recursion
-	if strings.Contains(rule, "|") {
-		parts := strings.Split(rule, "|")
-		savedNode := curNode
-		for _, part := range parts {
-			// TODO: add correct transitions
-			g.parseRule(part, savedNode)
-		}
-	}
-
-	for _, token := range tokens {
-		if strings.HasPrefix(token, "\"") && strings.HasSuffix(token, "\"") {
-			token = strings.Trim(token, "\"")
-
-			for _, r := range token {
-				newNode := &Node{
-					TransitionEdges: make(map[rune]*Node),
-				}
-				curNode.TransitionEdges[r] = newNode
-				curNode = newNode
-			}
-			// strNode := &Node{
-			// 	TransitionEdges: make(map[rune]*Node),
-			// }
-
-			// TODO: length constraint
-			// to self
-		}
-	}
-
-	return curNode
-}
--- a/sample/structured_outputs.go
+++ b/sample/structured_outputs.go
@@ -1,3 +0,0 @@
-package sample
-
-type StructuredOutput struct{}
--- a/sample/structured_outputs_test.go
+++ b/sample/structured_outputs_test.go
@@ -1,194 +0,0 @@
-package sample
-
-import (
-	"testing"
-
-	"github.com/ollama/ollama/model"
-)
-
-func TestBuildGraph(t *testing.T) {
-	tests := []struct {
-		name    string
-		grammar []byte
-		wantErr bool
-	}{
-		{
-			name:    "empty grammar",
-			grammar: []byte{},
-			wantErr: false,
-		},
-		{
-			name: "valid grammar",
-			grammar: []byte(`root ::= value
-value ::= string | number`),
-			wantErr: false,
-		},
-	}
-
-	for _, tt := range tests {
-		t.Run(tt.name, func(t *testing.T) {
-			g := &Graph{
-				proc:    &mockProcessor{},
-				grammar: tt.grammar,
-				rules:   make(map[string]string),
-			}
-
-			node := &Node{
-				TransitionEdges: make(map[rune]*Node),
-			}
-
-			err := g.BuildGraph(node)
-			if (err != nil) != tt.wantErr {
-				t.Errorf("BuildGraph() error = %v, wantErr %v", err, tt.wantErr)
-			}
-
-			if !tt.wantErr {
-				if len(g.decodedToks) == 0 {
-					t.Error("Expected decoded tokens, got none")
-				}
-				if len(g.rules) == 0 {
-					t.Error("Expected rules to be populated")
-				}
-			}
-		})
-	}
-}
-
-func TestRootPrefixes(t *testing.T) {
-	tests := []struct {
-		name     string
-		grammar  []byte
-		expected map[string]string
-	}{
-		{
-			name:     "empty grammar",
-			grammar:  []byte{},
-			expected: map[string]string{},
-		},
-		{
-			name: "grammar with root prefix",
-			grammar: []byte(`root ::= value
-root_string ::= string`),
-			expected: map[string]string{
-				"root":        "value",
-				"root_string": "string",
-			},
-		},
-		{
-			name: "grammar with comments and empty lines",
-			grammar: []byte(`# comment
-root ::= value
-
-# another comment
-root_number ::= number`),
-			expected: map[string]string{
-				"root":        "value",
-				"root_number": "number",
-			},
-		},
-	}
-
-	for _, tt := range tests {
-		t.Run(tt.name, func(t *testing.T) {
-			g := &Graph{
-				grammar: tt.grammar,
-				rules:   make(map[string]string),
-			}
-
-			g.rootPrefixes()
-
-			for k, v := range tt.expected {
-				if actual, ok := g.rules[k]; !ok || actual != v {
-					t.Errorf("Expected rule %s = %s, got %s", k, v, actual)
-				}
-			}
-		})
-	}
-}
-
-func TestParseRule(t *testing.T) {
-	tests := []struct {
-		name     string
-		rule     string
-		expected string
-	}{
-		{
-			name:     "empty rule",
-			rule:     "",
-			expected: "",
-		},
-		{
-			name:     "simple string",
-			rule:     "root ::= \"test_string\"",
-			expected: "test_string",
-		},
-		{
-			name:     "simple string",
-			rule:     "root ::= \"test_string\" | \"test_string2\"",
-			expected: "test_stringtest_string2",
-		},
-		{
-			name: "integer",
-			rule: "root ::= [0-9]+",
-			// TODO: this is infinite acutally
-			expected: "0123456789",
-		},
-		// TODO: handle left recursion
-		// {
-		// 	name:     "left recursion",
-		// 	rule:     "root ::= root \"test_string\"",
-		// 	expected: "test_string",
-		// },
-	}
-
-	for _, tt := range tests {
-		t.Run(tt.name, func(t *testing.T) {
-			g := &Graph{
-				rules: make(map[string]string),
-			}
-
-			rootNode := &Node{
-				TransitionEdges: make(map[rune]*Node),
-			}
-			curNode := rootNode
-			g.parseRule(tt.rule, curNode)
-			sb := ""
-			for {
-				if len(curNode.TransitionEdges) == 0 {
-					break
-				}
-
-				for r, n := range curNode.TransitionEdges {
-					sb += string(r)
-					curNode = n
-				}
-				t.Logf("sb: %s", sb)
-			}
-
-			if sb != tt.expected {
-				t.Errorf("Expected %s, got %s", tt.expected, sb)
-			}
-		})
-	}
-}
-
-// mockProcessor implements the TextProcessor interface for testing
-type mockProcessor struct{}
-
-func (m *mockProcessor) Decode(tokens []int32) (string, error) {
-	return "test", nil
-}
-
-func (m *mockProcessor) Vocab() *model.Vocabulary {
-	return &model.Vocabulary{
-		Values: []string{"test1", "test2"},
-	}
-}
-
-func (m *mockProcessor) Encode(s string, addSpecial bool) ([]int32, error) {
-	return []int32{0, 1}, nil
-}
-
-func (m *mockProcessor) Is(token int32, special model.Special) bool {
-	return false
-}
--- a/sample/transforms.go
+++ b/sample/transforms.go
@@ -26,16 +26,17 @@ func (h *tokenHeap) Pop() any {
 }

 // temperature applies scaling to the logits
-func temperature(ts []token, temp float32) {
+func temperature(ts []token, temp float32) []token {
 	// Ensure temperature clipping near 0 to avoid numerical instability
 	temp = max(temp, 1e-7)
 	for i := range ts {
 		ts[i].value = ts[i].value / temp
 	}
+	return ts
 }

 // softmax applies normalization to the logits
-func softmax(ts []token) {
+func softmax(ts []token) []token {
 	// Find max logit for numerical stability
 	maxLogit := float32(math.Inf(-1))
 	for _, t := range ts {
@@ -55,6 +56,8 @@ func softmax(ts []token) {
 	for i := range ts {
 		ts[i].value /= sum
 	}
+
+	return ts
 }

 // topK limits the number of tokens considered to the k highest logits
@@ -96,7 +99,6 @@ func topK(ts []token, k int) []token {
 }

 // topP limits tokens to those with cumulative probability p
-// requires ts to be sorted in descending order of probabilities
 func topP(ts []token, p float32) []token {
 	if p == 1.0 {
 		return ts
@@ -107,24 +109,37 @@ func topP(ts []token, p float32) []token {
 	for i, t := range ts {
 		sum += t.value
 		if sum > float32(p) {
-			return ts[:i+1]
+			ts = ts[:i+1]
+			return ts
 		}
 	}

 	return ts
 }

-// minP filters tokens with probabilities >= p * max_prob
-// requires ts to be sorted in descending order of probabilities
+// minP limits tokens to those with cumulative probability p
 func minP(ts []token, p float32) []token {
-	maxProb := ts[0].value
+	if p == 1.0 {
+		return ts
+	}

-	threshold := maxProb * p
-
-	for i, t := range ts {
-		if t.value < threshold {
-			return ts[:i]
+	maxProb := float32(math.Inf(-1))
+	for _, token := range ts {
+		if token.value > maxProb {
+			maxProb = token.value
 		}
 	}
+
+	threshold := maxProb * float32(p)
+
+	// Filter tokens in-place
+	validTokens := ts[:0]
+	for i, token := range ts {
+		if token.value >= threshold {
+			validTokens = append(validTokens, ts[i])
+		}
+	}
+
+	ts = validTokens
 	return ts
 }
--- a/sample/transforms_test.go
+++ b/sample/transforms_test.go
@@ -34,22 +34,17 @@ func compareLogits(t *testing.T, name string, want []float32, got []token) {

 func TestTemperature(t *testing.T) {
 	input := []float32{1.0, 4.0, -2.0, 0.0}
-	tokens := toTokens(input)
-	temperature(tokens, 0.5)
+	got := temperature(toTokens(input), 0.5)
 	want := []float32{2.0, 8.0, -4.0, 0.0}
-	compareLogits(t, "temperature(0.5)", want, tokens)
+	compareLogits(t, "temperature(0.5)", want, got)

-	input = []float32{1.0, 4.0, -2.0, 0.0}
-	tokens = toTokens(input)
-	temperature(tokens, 1.0)
+	got = temperature(toTokens(input), 1.0)
 	want = []float32{1.0, 4.0, -2.0, 0.0}
-	compareLogits(t, "temperature(1)", want, tokens)
+	compareLogits(t, "temperature(1)", want, got)

-	input = []float32{1.0, 4.0, -2.0, 0.0}
-	tokens = toTokens(input)
-	temperature(tokens, 0.0)
+	got = temperature(toTokens(input), 0.0)
 	want = []float32{1e7, 4e7, -2e7, 0.0}
-	compareLogits(t, "temperature(0)", want, tokens)
+	compareLogits(t, "temperature(0)", want, got)
 }

 func TestSoftmax(t *testing.T) {
@@ -95,17 +90,16 @@ func TestSoftmax(t *testing.T) {

 	for _, tt := range tests {
 		t.Run(tt.name, func(t *testing.T) {
-			tokens := toTokens(tt.input)
-			softmax(tokens)
+			got := softmax(toTokens(tt.input))

 			if tt.expected != nil {
-				compareLogits(t, tt.name, tt.expected, tokens)
+				compareLogits(t, tt.name, tt.expected, got)
 				return
 			}

 			// Check probabilities sum to 1
 			var sum float32
-			for _, token := range tokens {
+			for _, token := range got {
 				sum += token.value
 				if token.value < 0 || token.value > 1 {
 					t.Errorf("probability out of range [0,1]: got %f", token.value)
@@ -120,44 +114,38 @@ func TestSoftmax(t *testing.T) {

 func TestTopK(t *testing.T) {
 	input := []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
-	tokens := toTokens(input)
-	tokens = topK(tokens, 5)
-	if len(tokens) != 5 {
-		t.Errorf("topK(5): wrong length: want 5, got %d", len(tokens))
+
+	// Test k=5
+	got := topK(toTokens(input), 5)
+	if len(got) != 5 {
+		t.Errorf("topK(5): wrong length: want 5, got %d", len(got))
 	}
+	// Should keep highest 3 values in descending order
 	want := []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154}
-	compareLogits(t, "topK(3)", want, tokens)
+	compareLogits(t, "topK(3)", want, got)

-	tokens = toTokens(input)
-	tokens = topK(tokens, 20)
-	if len(tokens) != len(input) {
-		t.Errorf("topK(20): wrong length: want %d, got %d", len(input), len(tokens))
+	got = topK(toTokens(input), 20)
+	if len(got) != len(input) {
+		t.Errorf("topK(20): wrong length: want %d, got %d", len(input), len(got))
 	}

+	// Test k=-1
 	input = []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
 	want = []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154, 0.043722924, 0.036774673, 0.026986899, 0.01681367, 0.0046718004, 0.00412893, 0.0030491839}
-	tokens = toTokens(input)
-	tokens = topK(tokens, -1)
-	if len(tokens) != len(input) {
-		t.Errorf("topK(-1): wrong length: want %d, got %d", len(input), len(tokens))
+	got = topK(toTokens(input), -1)
+	if len(got) != len(input) {
+		t.Errorf("topK(-1): wrong length: want %d, got %d", len(input), len(got))
 	}
-	compareLogits(t, "topK(-1)", want, tokens)
+	compareLogits(t, "topK(-1)", want, got)

+	// Test k=0
 	input = []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
 	want = []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154, 0.043722924, 0.036774673, 0.026986899, 0.01681367, 0.0046718004, 0.00412893, 0.0030491839}
-	tokens = toTokens(input)
-	tokens = topK(tokens, 0)
-	if len(tokens) != len(input) {
-		t.Errorf("topK(-1): wrong length: want %d, got %d", len(input), len(tokens))
-	}
-	compareLogits(t, "topK(-1)", want, tokens)
-
-	input = []float32{-1e7, -2e7, -3e7, -4e7}
-	tokens = toTokens(input)
-	tokens = topK(tokens, 1)
-	if len(tokens) < 1 {
-		t.Error("topK should keep at least one token")
+	got = topK(toTokens(input), 0)
+	if len(got) != len(input) {
+		t.Errorf("topK(-1): wrong length: want %d, got %d", len(input), len(got))
 	}
+	compareLogits(t, "topK(-1)", want, got)
 }

 func TestTopP(t *testing.T) {
@@ -165,134 +153,50 @@ func TestTopP(t *testing.T) {
 	tokens := toTokens(input)

 	// First apply temperature and softmax to get probabilities
-	softmax(tokens)
+	tokens = softmax(tokens)
 	tokens = topK(tokens, 20)

-	// Test with very high p value
-	got := topP(tokens, 1.0)
-
-	// Should keep all tokens since p is 1
-	if len(got) != len(input) {
-		t.Errorf("topP(1.0): should keep all tokens, got %d, want %d", len(got), len(input))
-	}
-
-	// Test with normal p value
-	got = topP(tokens, 0.95)
+	// Then apply topP
+	got := topP(tokens, 0.95)

+	// Should keep tokens until cumsum > 0.95
 	if len(got) > 3 {
-		t.Errorf("topP(0.95): kept too many tokens: got %d", len(tokens))
-		t.Logf("got: %v", got)
-	}
-
-	// Test edge case - ensure at least one token remains
-	input = []float32{-1e6, -1e6, -1e7}
-	tokens = toTokens(input)
-	tokens = topK(tokens, 20)
-	softmax(tokens)
-	got = topP(tokens, 0.0)
-	if len(got) < 1 {
-		t.Error("topP should keep at least one token")
-	}
-
-	// Test with zero p value
-	got = topP(tokens, 0.0)
-
-	// Should keep only the highest probability token
-	if len(got) != 1 {
-		t.Errorf("topP(0.0): should keep only one token, got %d", len(got))
-		t.Logf("got: %v", got)
-	}
-
-	tokens = toTokens(input)
-	tokens = topK(tokens, 20)
-	softmax(tokens)
-	got = topP(tokens, 1e-10)
-	if len(got) == 0 {
-		t.Errorf("topP(1e-10): should keep at least one token, got %d", len(got))
+		t.Errorf("topP(0.95): kept too many tokens: got %d", len(got))
 		t.Logf("got: %v", got)
 	}
 }

 func TestMinP(t *testing.T) {
-	input := []float32{-2, 0, -1, -3, 2, 1, 4, 3}
+	input := []float32{-3, -2, -1, 0, 1, 2, 4, 3}
 	tokens := toTokens(input)

 	// First apply temperature and softmax
-	tokens = topK(tokens, 20)
-	softmax(tokens)
+	tokens = softmax(tokens)

-	tokens = minP(tokens, 1.0)
-
-	if len(tokens) != 1 {
-		t.Errorf("minP(1.0): should keep all tokens, got %d, want %d", len(tokens), len(tokens))
-	}
-
-	// Test with normal p value
-	tokens = toTokens(input) // Reset tokens
-	tokens = topK(tokens, 20)
-	softmax(tokens)
-	tokens = minP(tokens, 0.2)
+	// Then apply minP
+	got := minP(tokens, 0.2)

 	// Should keep tokens with prob >= 0.2 * max_prob
-	if len(tokens) > 3 {
-		t.Errorf("minP(0.2): kept too many tokens: got %d", len(tokens))
-		t.Logf("got: %v", tokens)
+	if len(got) > 3 {
+		t.Errorf("minP(0.2): kept too many tokens: got %d", len(got))
 	}
+}
+
+func TestSortLogits(t *testing.T) {
+	input := []float32{0.026986899, 0.043722924, 0.036774673, 0.27755088, 0.0046718004, 0.08582123, 0.20409796, 0.00412893, 0.15720603, 0.045046154, 0.0030491839, 0.01681367}
+	tokens := toTokens(input)

-	// Test with zero p value
-	tokens = toTokens(input) // Reset tokens
 	tokens = topK(tokens, 20)
-	softmax(tokens)
-	tokens = minP(tokens, 0.0)

-	// Should keep only the highest probability token
-	if len(tokens) != len(input) {
-		t.Errorf("minP(0.0): should keep only one token, got %d", len(tokens))
-		t.Logf("got: %v", tokens)
-	}
-
-	// Test with single token
-	tokens = toTokens(input[:1])
-	tokens = topK(tokens, 20)
-	softmax(tokens)
-	tokens = minP(tokens, 0.1)
-
-	// Should keep only the highest probability token
-	if len(tokens) != 1 {
-		t.Errorf("minP(0.1): should return single token, got %d", len(tokens))
-		t.Logf("got: %v", tokens)
-	}
-
-	input = []float32{1e-10, 1e-10, 1e-10}
-	tokens = toTokens(input)
-	softmax(tokens)
-	tokens = minP(tokens, 1.0)
-	if len(tokens) < 1 {
-		t.Error("minP should keep at least one token even with extreme probabilities")
-		got := minP(tokens, 1.0)
-
-		if len(got) != 1 {
-			t.Errorf("minP(1.0): should keep all tokens, got %d, want %d", len(got), len(tokens))
-		}
-
-		// Test with normal p value
-		got = minP(tokens, 0.2)
-
-		// Should keep tokens with prob >= 0.2 * max_prob
-		if len(got) > 3 {
-			t.Errorf("minP(0.2): kept too many tokens: got %d", len(got))
-			t.Logf("got: %v", got)
-		}
-
-		// Test with zero p value
-		got = minP(tokens, 0.0)
-
-		// Should keep only the highest probability token
-		if len(got) != len(tokens) {
-			t.Errorf("minP(0.0): should keep only one token, got %d", len(got))
-			t.Logf("got: %v", got)
+	for i := 1; i < len(tokens); i++ {
+		if tokens[i].value > tokens[i-1].value {
+			t.Errorf("sortLogits: tokens not sorted in descending order at index %d: %f > %f",
+				i, tokens[i].value, tokens[i-1].value)
 		}
 	}
+
+	want := []float32{0.27755088, 0.20409796, 0.15720603, 0.08582123, 0.045046154, 0.043722924, 0.036774673, 0.026986899, 0.01681367, 0.0046718004, 0.00412893, 0.0030491839}
+	compareLogits(t, "sortLogits", want, tokens)
 }

 func BenchmarkTransforms(b *testing.B) {
@@ -327,7 +231,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			tokens = topK(tokensCopy, 10)
+			topK(tokensCopy, 10)
 		}
 	})

@@ -335,7 +239,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			tokens = topP(tokensCopy, 0.9)
+			topP(tokensCopy, 0.9)
 		}
 	})

@@ -343,7 +247,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			tokens = minP(tokensCopy, 0.2)
+			minP(tokensCopy, 0.2)
 		}
 	})

@@ -351,7 +255,7 @@ func BenchmarkTransforms(b *testing.B) {
 		b.ResetTimer()
 		for b.Loop() {
 			copy(tokensCopy, tokens)
-			tokens = topK(tokensCopy, 200000)
+			topK(tokensCopy, 200000)
 		}
 	})
 }
--- a/scripts/build_darwin.sh
+++ b/scripts/build_darwin.sh
@@ -8,7 +8,7 @@ usage() {
    exit 1
 }

-export VERSION=${VERSION:-$(git describe --tags --first-parent --abbrev=7 --long --dirty --always | sed -e "s/^v//g")}
+export VERSION=${VERSION:-$(git describe --tags --dirty)}
 export GOFLAGS="'-ldflags=-w -s \"-X=github.com/ollama/ollama/version.Version=${VERSION#v}\" \"-X=github.com/ollama/ollama/server.mode=release\"'"
 export CGO_CPPFLAGS='-mmacosx-version-min=11.3'

--- a/server/internal/client/ollama/registry.go
+++ b/server/internal/client/ollama/registry.go
@@ -25,7 +25,6 @@ import (
 	"os"
 	"path/filepath"
 	"runtime"
-	"runtime/debug"
 	"slices"
 	"strconv"
 	"strings"
@@ -37,6 +36,7 @@ import (
 	"golang.org/x/sync/errgroup"

 	"github.com/ollama/ollama/server/internal/cache/blob"
+	"github.com/ollama/ollama/server/internal/internal/backoff"
 	"github.com/ollama/ollama/server/internal/internal/names"

 	_ "embed"
@@ -59,11 +59,6 @@ var (
 	// ErrCached is passed to [Trace.PushUpdate] when a layer already
 	// exists. It is a non-fatal error and is never returned by [Registry.Push].
 	ErrCached = errors.New("cached")
-
-	// ErrIncomplete is returned by [Registry.Pull] when a model pull was
-	// incomplete due to one or more layer download failures. Users that
-	// want specific errors should use [WithTrace].
-	ErrIncomplete = errors.New("incomplete")
 )

 // Defaults
@@ -217,6 +212,12 @@ type Registry struct {
 	// request. If zero, [DefaultChunkingThreshold] is used.
 	ChunkingThreshold int64

+	// MaxChunkSize is the maximum size of a chunk to download. If zero,
+	// the default is [DefaultMaxChunkSize].
+	//
+	// It is only used when a layer is larger than [MaxChunkingThreshold].
+	MaxChunkSize int64
+
 	// Mask, if set, is the name used to convert non-fully qualified names
 	// to fully qualified names. If empty, [DefaultMask] is used.
 	Mask string
@@ -258,7 +259,6 @@ func DefaultRegistry() (*Registry, error) {
 	}

 	var rc Registry
-	rc.UserAgent = UserAgent()
 	rc.Key, err = ssh.ParseRawPrivateKey(keyPEM)
 	if err != nil {
 		return nil, err
@@ -274,27 +274,6 @@ func DefaultRegistry() (*Registry, error) {
 	return &rc, nil
 }

-func UserAgent() string {
-	buildinfo, _ := debug.ReadBuildInfo()
-
-	version := buildinfo.Main.Version
-	if version == "(devel)" {
-		// When using `go run .` the version is "(devel)". This is seen
-		// as an invalid version by ollama.com and so it defaults to
-		// "needs upgrade" for some requests, such as pulls. These
-		// checks can be skipped by using the special version "v0.0.0",
-		// so we set it to that here.
-		version = "v0.0.0"
-	}
-
-	return fmt.Sprintf("ollama/%s (%s %s) Go/%s",
-		version,
-		runtime.GOARCH,
-		runtime.GOOS,
-		runtime.Version(),
-	)
-}
-
 func (r *Registry) maxStreams() int {
 	return cmp.Or(r.MaxStreams, runtime.GOMAXPROCS(0))
 }
@@ -434,14 +413,13 @@ func canRetry(err error) bool {
 //
 // It always calls update with a nil error.
 type trackingReader struct {
-	l      *Layer
-	r      io.Reader
-	update func(l *Layer, n int64, err error)
+	r io.Reader
+	n *atomic.Int64
 }

 func (r *trackingReader) Read(p []byte) (n int, err error) {
 	n, err = r.r.Read(p)
-	r.update(r.l, int64(n), nil)
+	r.n.Add(int64(n))
 	return
 }

@@ -457,11 +435,6 @@ func (r *Registry) Pull(ctx context.Context, name string) error {
 	if err != nil {
 		return err
 	}
-
-	// TODO(bmizerany): decide if this should be considered valid. Maybe
-	// server-side we special case '{}' to have some special meaning? Maybe
-	// "archiving" a tag (which is how we reason about it in the registry
-	// already, just with a different twist).
 	if len(m.Layers) == 0 {
 		return fmt.Errorf("%w: no layers", ErrManifestInvalid)
 	}
@@ -471,7 +444,11 @@ func (r *Registry) Pull(ctx context.Context, name string) error {
 		return err
 	}

-	// TODO(bmizerany): work to remove the need to do this
+	exists := func(l *Layer) bool {
+		info, err := c.Get(l.Digest)
+		return err == nil && info.Size == l.Size
+	}
+
 	layers := m.Layers
 	if m.Config != nil && m.Config.Digest.IsValid() {
 		layers = append(layers, m.Config)
@@ -479,97 +456,99 @@ func (r *Registry) Pull(ctx context.Context, name string) error {

 	// Send initial layer trace events to allow clients to have an
 	// understanding of work to be done before work starts.
-	var expected int64
 	t := traceFromContext(ctx)
-	for _, l := range layers {
+	skip := make([]bool, len(layers))
+	for i, l := range layers {
 		t.update(l, 0, nil)
-		expected += l.Size
+		if exists(l) {
+			skip[i] = true
+			t.update(l, l.Size, ErrCached)
+		}
 	}

-	var received atomic.Int64
-	var g errgroup.Group
+	g, ctx := errgroup.WithContext(ctx)
 	g.SetLimit(r.maxStreams())
-	for _, l := range layers {
-		info, err := c.Get(l.Digest)
-		if err == nil && info.Size == l.Size {
-			received.Add(l.Size)
-			t.update(l, l.Size, ErrCached)
+	for i, l := range layers {
+		if skip[i] {
 			continue
 		}

-		var wg sync.WaitGroup
 		chunked, err := c.Chunked(l.Digest, l.Size)
 		if err != nil {
 			t.update(l, 0, err)
 			continue
 		}
+		defer chunked.Close()

+		var progress atomic.Int64
 		for cs, err := range r.chunksums(ctx, name, l) {
 			if err != nil {
-				// Chunksum stream interrupted. Note in trace
-				// log and let in-flight downloads complete.
-				// This will naturally trigger ErrIncomplete
-				// since received < expected bytes.
-				t.update(l, 0, err)
+				t.update(l, progress.Load(), err)
 				break
 			}

-			wg.Add(1)
 			g.Go(func() (err error) {
-				defer func() {
-					if err == nil {
-						received.Add(cs.Chunk.Size())
-					} else {
-						err = fmt.Errorf("error downloading %s: %w", cs.Digest.Short(), err)
+				defer func() { t.update(l, progress.Load(), err) }()
+
+				for _, err := range backoff.Loop(ctx, 3*time.Second) {
+					if err != nil {
+						return err
 					}
-					wg.Done()
-				}()
+					err := func() error {
+						req, err := http.NewRequestWithContext(ctx, "GET", cs.URL, nil)
+						if err != nil {
+							return err
+						}
+						req.Header.Set("Range", fmt.Sprintf("bytes=%d-%d", cs.Chunk.Start, cs.Chunk.End))
+						res, err := sendRequest(r.client(), req)
+						if err != nil {
+							return err
+						}
+						defer res.Body.Close()

-				req, err := http.NewRequestWithContext(ctx, "GET", cs.URL, nil)
-				if err != nil {
-					return err
-				}
-				req.Header.Set("Range", fmt.Sprintf("bytes=%d-%d", cs.Chunk.Start, cs.Chunk.End))
-				res, err := sendRequest(r.client(), req)
-				if err != nil {
-					return err
-				}
-				defer res.Body.Close()
+						// Count bytes towards
+						// progress, as they arrive, so
+						// that our bytes piggyback
+						// other chunk updates on
+						// completion.
+						//
+						// This tactic is enough to
+						// show "smooth" progress given
+						// the current CLI client. In
+						// the near future, the server
+						// should report download rate
+						// since it knows better than
+						// a client that is measuring
+						// rate based on wall-clock
+						// time-since-last-update.
+						body := &trackingReader{r: res.Body, n: &progress}

-				body := &trackingReader{l: l, r: res.Body, update: t.update}
-				return chunked.Put(cs.Chunk, cs.Digest, body)
+						err = chunked.Put(cs.Chunk, cs.Digest, body)
+						if err != nil {
+							return err
+						}
+
+						return nil
+					}()
+					if !canRetry(err) {
+						return err
+					}
+				}
+				return nil
 			})
 		}
-
-		// Close writer immediately after downloads finish, not at Pull
-		// exit. Using defer would keep file descriptors open until all
-		// layers complete, potentially exhausting system limits with
-		// many layers.
-		//
-		// The WaitGroup tracks when all chunks finish downloading,
-		// allowing precise writer closure in a background goroutine.
-		// Each layer briefly uses one extra goroutine while at most
-		// maxStreams()-1 chunks download in parallel.
-		//
-		// This caps file descriptors at maxStreams() instead of
-		// growing with layer count.
-		g.Go(func() error {
-			wg.Wait()
-			chunked.Close()
-			return nil
-		})
 	}
 	if err := g.Wait(); err != nil {
 		return err
 	}
-	if received.Load() != expected {
-		return fmt.Errorf("%w: received %d/%d", ErrIncomplete, received.Load(), expected)
-	}

+	// store the manifest blob
 	md := blob.DigestFromBytes(m.Data)
 	if err := blob.PutBytes(c, md, m.Data); err != nil {
 		return err
 	}
+
+	// commit the manifest with a link
 	return c.Link(m.Name, md)
 }

--- a/server/internal/client/ollama/registry_test.go
+++ b/server/internal/client/ollama/registry_test.go
@@ -17,7 +17,6 @@ import (
 	"reflect"
 	"slices"
 	"strings"
-	"sync"
 	"testing"
 	"time"

@@ -25,28 +24,6 @@ import (
 	"github.com/ollama/ollama/server/internal/testutil"
 )

-func ExampleRegistry_cancelOnFirstError() {
-	ctx, cancel := context.WithCancel(context.Background())
-	defer cancel()
-
-	ctx = WithTrace(ctx, &Trace{
-		Update: func(l *Layer, n int64, err error) {
-			if err != nil {
-				// Discontinue pulling layers if there is an
-				// error instead of continuing to pull more
-				// data.
-				cancel()
-			}
-		},
-	})
-
-	var r Registry
-	if err := r.Pull(ctx, "model"); err != nil {
-		// panic for demo purposes
-		panic(err)
-	}
-}
-
 func TestManifestMarshalJSON(t *testing.T) {
 	// All manifests should contain an "empty" config object.
 	var m Manifest
@@ -79,21 +56,21 @@ func (rr recordRoundTripper) RoundTrip(req *http.Request) (*http.Response, error

 // newClient constructs a cache with predefined manifests for testing. The manifests are:
 //
-//	empty:         no data
-//	zero:          no layers
-//	single:        one layer with the contents "exists"
-//	multiple:      two layers with the contents "exists" and "here"
-//	notfound:      a layer that does not exist in the cache
-//	null:          one null layer (e.g. [null])
-//	sizemismatch:  one valid layer, and one with a size mismatch (file size is less than the reported size)
-//	invalid:       a layer with invalid JSON data
+//	empty: no data
+//	zero: no layers
+//	single: one layer with the contents "exists"
+//	multiple: two layers with the contents "exists" and "here"
+//	notfound: a layer that does not exist in the cache
+//	null: one null layer (e.g. [null])
+//	sizemismatch: one valid layer, and one with a size mismatch (file size is less than the reported size)
+//	invalid: a layer with invalid JSON data
 //
 // Tests that want to ensure the client does not communicate with the upstream
 // registry should pass a nil handler, which will cause a panic if
 // communication is attempted.
 //
 // To simulate a network error, pass a handler that returns a 499 status code.
-func newClient(t *testing.T, upstreamRegistry http.HandlerFunc) (*Registry, *blob.DiskCache) {
+func newClient(t *testing.T, h http.HandlerFunc) (*Registry, *blob.DiskCache) {
 	t.Helper()

 	c, err := blob.Open(t.TempDir())
@@ -111,7 +88,7 @@ func newClient(t *testing.T, upstreamRegistry http.HandlerFunc) (*Registry, *blo
 	r := &Registry{
 		Cache: c,
 		HTTPClient: &http.Client{
-			Transport: recordRoundTripper(upstreamRegistry),
+			Transport: recordRoundTripper(h),
 		},
 	}

@@ -790,79 +767,3 @@ func TestUnlink(t *testing.T) {
 		}
 	})
 }
-
-func TestPullChunksums(t *testing.T) {
-	check := testutil.Checker(t)
-
-	content := "hello"
-	var chunksums string
-	contentDigest := func() blob.Digest {
-		return blob.DigestFromBytes(content)
-	}
-	rc, c := newClient(t, func(w http.ResponseWriter, r *http.Request) {
-		switch {
-		case strings.Contains(r.URL.Path, "/manifests/latest"):
-			fmt.Fprintf(w, `{"layers":[{"digest":%q,"size":%d}]}`, contentDigest(), len(content))
-		case strings.HasSuffix(r.URL.Path, "/chunksums/"+contentDigest().String()):
-			loc := fmt.Sprintf("http://blob.store/v2/library/test/blobs/%s", contentDigest())
-			w.Header().Set("Content-Location", loc)
-			io.WriteString(w, chunksums)
-		case strings.Contains(r.URL.Path, "/blobs/"+contentDigest().String()):
-			http.ServeContent(w, r, contentDigest().String(), time.Time{}, strings.NewReader(content))
-		default:
-			t.Errorf("unexpected request: %v", r)
-			http.NotFound(w, r)
-		}
-	})
-
-	rc.MaxStreams = 1        // prevent concurrent chunk downloads
-	rc.ChunkingThreshold = 1 // for all blobs to be chunked
-
-	var mu sync.Mutex
-	var reads []int64
-	ctx := WithTrace(t.Context(), &Trace{
-		Update: func(l *Layer, n int64, err error) {
-			t.Logf("Update: %v %d %v", l, n, err)
-			mu.Lock()
-			reads = append(reads, n)
-			mu.Unlock()
-		},
-	})
-
-	chunksums = fmt.Sprintf("%s 0-2\n%s 3-4\n",
-		blob.DigestFromBytes("hel"),
-		blob.DigestFromBytes("lo"),
-	)
-	err := rc.Pull(ctx, "test")
-	check(err)
-	wantReads := []int64{
-		0, // initial signaling of layer pull starting
-		3, // first chunk read
-		2, // second chunk read
-	}
-	if !slices.Equal(reads, wantReads) {
-		t.Errorf("reads = %v; want %v", reads, wantReads)
-	}
-
-	mw, err := rc.Resolve(t.Context(), "test")
-	check(err)
-	mg, err := rc.ResolveLocal("test")
-	check(err)
-	if !reflect.DeepEqual(mw, mg) {
-		t.Errorf("mw = %v; mg = %v", mw, mg)
-	}
-	for i := range mg.Layers {
-		_, err = c.Get(mg.Layers[i].Digest)
-		if err != nil {
-			t.Errorf("Get(%v): %v", mg.Layers[i].Digest, err)
-		}
-	}
-
-	// missing chunks
-	content = "llama"
-	chunksums = fmt.Sprintf("%s 0-1\n", blob.DigestFromBytes("ll"))
-	err = rc.Pull(ctx, "missingchunks")
-	if err == nil {
-		t.Error("expected error because of missing chunks")
-	}
-}
--- a/server/internal/registry/server.go
+++ b/server/internal/registry/server.go
@@ -200,7 +200,7 @@ type params struct {
 	//
 	// Unfortunately, this API was designed to be a bit awkward. Stream is
 	// defined to default to true if not present, so we need a way to check
-	// if the client decisively set it to false. So, we use a pointer to a
+	// if the client decisively it to false. So, we use a pointer to a
 	// bool. Gross.
 	//
 	// Use [stream()] to get the correct value for this field.
@@ -280,17 +280,17 @@ func (s *Local) handlePull(w http.ResponseWriter, r *http.Request) error {
 	progress := make(map[*ollama.Layer]int64)

 	progressCopy := make(map[*ollama.Layer]int64, len(progress))
-	flushProgress := func() {
+	pushUpdate := func() {
 		defer maybeFlush()

-		// TODO(bmizerany): Flushing every layer in one update doesn't
-		// scale well. We could flush only the modified layers or track
-		// the full download. Needs further consideration, though it's
-		// fine for now.
+		// TODO(bmizerany): This scales poorly with more layers due to
+		// needing to flush out them all in one big update. We _could_
+		// just flush on the changed ones, or just track the whole
+		// download. Needs more thought. This is fine for now.
 		mu.Lock()
 		maps.Copy(progressCopy, progress)
 		mu.Unlock()
-		for l, n := range progressCopy {
+		for l, n := range progress {
 			enc.Encode(progressUpdateJSON{
 				Digest:    l.Digest,
 				Total:     l.Size,
@@ -298,26 +298,19 @@ func (s *Local) handlePull(w http.ResponseWriter, r *http.Request) error {
 			})
 		}
 	}
-	defer flushProgress()

-	t := time.NewTicker(1000 * time.Hour) // "unstarted" timer
+	t := time.NewTicker(time.Hour) // "unstarted" timer
 	start := sync.OnceFunc(func() {
-		flushProgress() // flush initial state
+		pushUpdate()
 		t.Reset(100 * time.Millisecond)
 	})
 	ctx := ollama.WithTrace(r.Context(), &ollama.Trace{
 		Update: func(l *ollama.Layer, n int64, err error) {
 			if n > 0 {
-				// Block flushing progress updates until every
-				// layer is accounted for. Clients depend on a
-				// complete model size to calculate progress
-				// correctly; if they use an incomplete total,
-				// progress indicators would erratically jump
-				// as new layers are registered.
-				start()
+				start() // flush initial state
 			}
 			mu.Lock()
-			progress[l] += n
+			progress[l] = n
 			mu.Unlock()
 		},
 	})
@@ -330,9 +323,9 @@ func (s *Local) handlePull(w http.ResponseWriter, r *http.Request) error {
 	for {
 		select {
 		case <-t.C:
-			flushProgress()
+			pushUpdate()
 		case err := <-done:
-			flushProgress()
+			pushUpdate()
 			if err != nil {
 				var status string
 				if errors.Is(err, ollama.ErrModelNotFound) {
--- a/server/model.go
+++ b/server/model.go
@@ -82,7 +82,7 @@ func detectChatTemplate(layers []*layerGGML) ([]*layerGGML, error) {
 	for _, layer := range layers {
 		if s := layer.GGML.KV().ChatTemplate(); s != "" {
 			if t, err := template.Named(s); err != nil {
-				slog.Debug("template detection", "error", err, "template", s)
+				slog.Debug("template detection", "error", err)
 			} else {
 				layer, err := NewLayer(t.Reader(), "application/vnd.ollama.image.template")
 				if err != nil {
--- a/server/prompt.go
+++ b/server/prompt.go
@@ -26,6 +26,7 @@ func chatPrompt(ctx context.Context, m *Model, tokenize tokenizeFunc, opts *api.
 	var system []api.Message

 	isMllama := checkMllamaModelFamily(m)
+	isGemma3 := checkGemma3ModelFamily(m)

 	var imageNumTokens int
 	// TODO: Ideally we would compute this from the projector metadata but some pieces are implementation dependent
@@ -40,7 +41,7 @@ func chatPrompt(ctx context.Context, m *Model, tokenize tokenizeFunc, opts *api.
 	n := len(msgs) - 1
 	// in reverse, find all messages that fit into context window
 	for i := n; i >= 0; i-- {
-		if isMllama && len(msgs[i].Images) > 1 {
+		if (isMllama || isGemma3) && len(msgs[i].Images) > 1 {
 			return "", nil, errTooManyImages
 		}

@@ -157,3 +158,12 @@ func checkMllamaModelFamily(m *Model) bool {
 	}
 	return false
 }
+
+func checkGemma3ModelFamily(m *Model) bool {
+	for _, arch := range m.Config.ModelFamilies {
+		if arch == "gemma3" {
+			return true
+		}
+	}
+	return false
+}
--- a/template/gemma3-instruct.gotmpl
+++ b/template/gemma3-instruct.gotmpl
@@ -1,13 +0,0 @@
-{{- range $i, $_ := .Messages }}
-{{- $last := eq (len (slice $.Messages $i)) 1 }}
-{{- if eq .Role "user" }}<start_of_turn>user
-{{- if and (eq $i 1) $.System }}
-{{ $.System }}
-{{ end }}
-{{ .Content }}<end_of_turn>
-{{ else if eq .Role "assistant" }}<start_of_turn>model
-{{ .Content }}<end_of_turn>
-{{ end }}
-{{- if $last }}<start_of_turn>model
-{{ end }}
-{{- end }}
--- a/template/gemma3-instruct.json
+++ b/template/gemma3-instruct.json
@@ -1,6 +0,0 @@
-{
-  "stop": [
-    "<end_of_turn>"
-  ],
-  "temperature": 0.1
-}
--- a/template/index.json
+++ b/template/index.json
@@ -87,10 +87,6 @@
    "template": "{{ bos_token }}{% if messages[0]['role'] == 'system' %}{{ raise_exception('System role not supported') }}{% endif %}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% if (message['role'] == 'assistant') %}{% set role = 'model' %}{% else %}{% set role = message['role'] %}{% endif %}{{ '<start_of_turn>' + role + '\n' + message['content'] | trim + '<end_of_turn>\n' }}{% endfor %}{% if add_generation_prompt %}{{'<start_of_turn>model\n'}}{% endif %}",
    "name": "gemma-instruct"
  },
-  {
-    "template": "{{ bos_token }}\n{%- if messages[0]['role'] == 'system' -%}\n    {%- if messages[0]['content'] is string -%}\n        {%- set first_user_prefix = messages[0]['content'] + '\n\n' -%}\n    {%- else -%}\n        {%- set first_user_prefix = messages[0]['content'][0]['text'] + '\n\n' -%}\n    {%- endif -%}\n    {%- set loop_messages = messages[1:] -%}\n{%- else -%}\n    {%- set first_user_prefix = \"\" -%}\n    {%- set loop_messages = messages -%}\n{%- endif -%}\n{%- for message in loop_messages -%}\n    {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%}\n        {{ raise_exception(\"Conversation roles must alternate user/assistant/user/assistant/...\") }}\n    {%- endif -%}\n    {%- if (message['role'] == 'assistant') -%}\n        {%- set role = \"model\" -%}\n    {%- else -%}\n        {%- set role = message['role'] -%}\n    {%- endif -%}\n    {{ '<start_of_turn>' + role + '\n' + (first_user_prefix if loop.first else \"\") }}\n    {%- if message['content'] is string -%}\n        {{ message['content'] | trim }}\n    {%- elif message['content'] is iterable -%}\n        {%- for item in message['content'] -%}\n            {%- if item['type'] == 'image' -%}\n                {{ '<start_of_image>' }}\n            {%- elif item['type'] == 'text' -%}\n                {{ item['text'] | trim }}\n            {%- endif -%}\n        {%- endfor -%}\n    {%- else -%}\n        {{ raise_exception(\"Invalid content type\") }}\n    {%- endif -%}\n    {{ '<end_of_turn>\n' }}\n{%- endfor -%}\n{%- if add_generation_prompt -%}\n    {{'<start_of_turn>model\n'}}\n{%- endif -%}\n",
-    "name": "gemma3-instruct"
-  },
  {
    "template": "{% set loop_messages = messages %}{% for message in loop_messages %}{% set content = '<|start_header_id|>' + message['role'] + '<|end_header_id|>\n\n'+ message['content'] | trim + '<|eot_id|>' %}{% if loop.index0 == 0 %}{% set content = bos_token + content %}{% endif %}{{ content }}{% endfor %}{% if add_generation_prompt %}{{ '<|start_header_id|>assistant<|end_header_id|>\n\n' }}{% endif %}",
    "name": "llama3-instruct"
--- a/template/testdata/gemma3-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/gemma3-instruct.gotmpl/system-user-assistant-user
@@ -1,10 +0,0 @@
-<start_of_turn>user
-You are a helpful assistant.
-
-Hello, how are you?<end_of_turn>
-<start_of_turn>model
-I'm doing great. How can I help you today?<end_of_turn>
-<start_of_turn>user
-I'd like to show off how chat templating works!<end_of_turn>
-<start_of_turn>model
-
--- a/template/testdata/gemma3-instruct.gotmpl/user
+++ b/template/testdata/gemma3-instruct.gotmpl/user
@@ -1,4 +0,0 @@
-<start_of_turn>user
-Hello, how are you?<end_of_turn>
-<start_of_turn>model
-
--- a/template/testdata/gemma3-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/gemma3-instruct.gotmpl/user-assistant-user
@@ -1,8 +0,0 @@
-<start_of_turn>user
-Hello, how are you?<end_of_turn>
-<start_of_turn>model
-I'm doing great. How can I help you today?<end_of_turn>
-<start_of_turn>user
-I'd like to show off how chat templating works!<end_of_turn>
-<start_of_turn>model
-