wip

prototyping
grammar: introduce new grammar package
2025-03-25 16:45:27 -07:00 · 2025-03-25 15:00:14 -07:00 · 2025-03-24 11:58:06 -07:00
392 changed files with 47459 additions and 63115 deletions
--- a/.github/workflows/release.yaml
+++ b/.github/workflows/release.yaml
@@ -432,22 +432,6 @@ jobs:
          docker buildx imagetools inspect ollama/ollama:${{ steps.metadata.outputs.version }}
        working-directory: ${{ runner.temp }}

-  # Trigger downstream release process
-  trigger:
-    runs-on: ubuntu-latest
-    environment: release
-    needs: [darwin-build, windows-build, windows-depends]
-    steps:
-      - name: Trigger downstream release process
-        run: |
-          curl -L \
-            -X POST \
-            -H "Accept: application/vnd.github+json" \
-            -H "Authorization: Bearer ${{ secrets.RELEASE_TOKEN }}" \
-            -H "X-GitHub-Api-Version: 2022-11-28" \
-            https://api.github.com/repos/ollama/${{ vars.RELEASE_REPO }}/dispatches \
-            -d "{\"event_type\": \"trigger-workflow\", \"client_payload\": {\"run_id\": \"${GITHUB_RUN_ID}\", \"version\": \"${GITHUB_REF_NAME#v}\"}}"
-
  # Aggregate all the assets and ship a release
  release:
    needs: [darwin-sign, windows-sign, linux-build]
--- a/.github/workflows/test.yaml
+++ b/.github/workflows/test.yaml
@@ -237,5 +237,5 @@ jobs:
      - uses: actions/checkout@v4
      - name: Verify patches apply cleanly and do not change files
        run: |
-          make -f Makefile.sync clean checkout apply-patches sync
-          git diff --compact-summary --exit-code
+          make -f Makefile.sync clean sync
+          git diff --compact-summary --exit-code
--- a/.golangci.yaml
+++ b/.golangci.yaml
@@ -19,8 +19,8 @@ linters:
    - nolintlint
    - nosprintfhostport
    - staticcheck
+    - tenv
    - unconvert
-    - usetesting
    - wastedassign
    - whitespace
  disable:
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -24,7 +24,6 @@ set(GGML_LLAMAFILE ON)
 set(GGML_CUDA_PEER_MAX_BATCH_SIZE 128)
 set(GGML_CUDA_GRAPHS ON)
 set(GGML_CUDA_FA ON)
-set(GGML_CUDA_COMPRESSION_MODE default)

 if((CMAKE_OSX_ARCHITECTURES AND NOT CMAKE_OSX_ARCHITECTURES MATCHES "arm64")
    OR (NOT CMAKE_OSX_ARCHITECTURES AND NOT CMAKE_SYSTEM_PROCESSOR MATCHES "arm|aarch64|ARM64|ARMv[0-9]+"))
@@ -87,9 +86,9 @@ if(CMAKE_CUDA_COMPILER)
    )
 endif()

-set(WINDOWS_AMDGPU_TARGETS_EXCLUDE_REGEX "^gfx(906|908|90a|1200|1201):xnack[+-]$"
+set(WINDOWS_AMDGPU_TARGETS_EXCLUDE_REGEX "^gfx(906|908|90a):xnack[+-]$"
    CACHE STRING
-    "Regular expression describing AMDGPU_TARGETS not supported on Windows. Override to force building these targets. Default \"^gfx(906|908|90a|1200|1201):xnack[+-]$\"."
+    "Regular expression describing AMDGPU_TARGETS not supported on Windows. Override to force building these targets. Default \"^gfx(906|908|90a):xnack[+-]$\"."
 )

 check_language(HIP)
@@ -98,7 +97,7 @@ if(CMAKE_HIP_COMPILER)

    find_package(hip REQUIRED)
    if(NOT AMDGPU_TARGETS)
-        list(FILTER AMDGPU_TARGETS INCLUDE REGEX "^gfx(900|94[012]|101[02]|1030|110[012]|120[01])$")
+        list(FILTER AMDGPU_TARGETS INCLUDE REGEX "^gfx(900|94[012]|101[02]|1030|110[012])$")
    elseif(WIN32 AND WINDOWS_AMDGPU_TARGETS_EXCLUDE_REGEX)
        list(FILTER AMDGPU_TARGETS EXCLUDE REGEX ${WINDOWS_AMDGPU_TARGETS_EXCLUDE_REGEX})
    endif()
--- a/CMakePresets.json
+++ b/CMakePresets.json
@@ -21,16 +21,14 @@
      "name": "CUDA 11",
      "inherits": [ "CUDA" ],
      "cacheVariables": {
-        "CMAKE_CUDA_ARCHITECTURES": "50;52;53;60;61;70;75;80;86",
-        "CMAKE_CUDA_FLAGS": "-Wno-deprecated-gpu-targets"
+        "CMAKE_CUDA_ARCHITECTURES": "50;52;53;60;61;70;75;80;86"
      }
    },
    {
      "name": "CUDA 12",
      "inherits": [ "CUDA" ],
      "cacheVariables": {
-        "CMAKE_CUDA_ARCHITECTURES": "50;60;61;70;75;80;86;87;89;90;90a;120",
-        "CMAKE_CUDA_FLAGS": "-Wno-deprecated-gpu-targets"
+        "CMAKE_CUDA_ARCHITECTURES": "50;60;61;70;75;80;86;87;89;90;90a;120"
      }
    },
    {
@@ -58,7 +56,7 @@
      "name": "ROCm 6",
      "inherits": [ "ROCm" ],
      "cacheVariables": {
-        "AMDGPU_TARGETS": "gfx900;gfx940;gfx941;gfx942;gfx1010;gfx1012;gfx1030;gfx1100;gfx1101;gfx1102;gfx1151;gfx1200;gfx1201;gfx906:xnack-;gfx908:xnack-;gfx90a:xnack+;gfx90a:xnack-"
+        "AMDGPU_TARGETS": "gfx900;gfx940;gfx941;gfx942;gfx1010;gfx1012;gfx1030;gfx1100;gfx1101;gfx1102;gfx1151;gfx906:xnack-;gfx908:xnack-;gfx90a:xnack+;gfx90a:xnack-"
      }
    }
  ],
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -51,7 +51,7 @@ see if the change were accepted.

 The title should look like:

-    <package>: <short description>
+   <package>: <short description>

 The package is the most affected Go package. If the change does not affect Go
 code, then use the directory name instead. Changes to a single well-known
--- a/4
+++ b/4
@@ -104,8 +104,8 @@ COPY --from=cuda-12 dist/lib/ollama/cuda_v12 /lib/ollama/cuda_v12
 FROM --platform=linux/arm64 scratch AS arm64
 COPY --from=cuda-11 dist/lib/ollama/cuda_v11 /lib/ollama/cuda_v11
 COPY --from=cuda-12 dist/lib/ollama/cuda_v12 /lib/ollama/cuda_v12
-COPY --from=jetpack-5 dist/lib/ollama/cuda_v11 /lib/ollama/cuda_jetpack5
-COPY --from=jetpack-6 dist/lib/ollama/cuda_v12 /lib/ollama/cuda_jetpack6
+COPY --from=jetpack-5 dist/lib/ollama/cuda_v11 lib/ollama/cuda_jetpack5
+COPY --from=jetpack-6 dist/lib/ollama/cuda_v12 lib/ollama/cuda_jetpack6

 FROM scratch AS rocm
 COPY --from=rocm-6 dist/lib/ollama/rocm /lib/ollama/rocm
--- a/Makefile.sync
+++ b/Makefile.sync
@@ -1,6 +1,6 @@
 UPSTREAM=https://github.com/ggerganov/llama.cpp.git
 WORKDIR=llama/vendor
-FETCH_HEAD=de4c07f93783a1a96456a44dc16b9db538ee1618
+FETCH_HEAD=d7cfe1ffe0f435d0048a6058d529daf76e072d9c

 .PHONY: help
 help:
@@ -15,30 +15,27 @@ help:
 	@echo "    make -f $(lastword $(MAKEFILE_LIST)) clean sync"

 .PHONY: sync
-sync: llama/build-info.cpp ml/backend/ggml/ggml/src/ggml-metal/ggml-metal-embed.metal
+sync: llama/build-info.cpp llama/llama.cpp ml/backend/ggml/ggml apply-patches

-llama/build-info.cpp: llama/build-info.cpp.in llama/llama.cpp
-	sed -e 's|@FETCH_HEAD@|$(FETCH_HEAD)|' <$< >$@
-
-ml/backend/ggml/ggml/src/ggml-metal/ggml-metal-embed.metal: ml/backend/ggml/ggml
-	go generate ./$(@D)
+.PHONY: llama/build-info.cpp
+llama/build-info.cpp: llama/build-info.cpp.in
+	sed -e 's|@FETCH_HEAD@|$(FETCH_HEAD)|' $< > $@

 .PHONY: llama/llama.cpp
-llama/llama.cpp: llama/vendor/
+llama/llama.cpp: llama/vendor/ apply-patches
 	rsync -arvzc -f "merge $@/.rsync-filter" $< $@

-.PHONY: ml/backend/ggml/ggml
-ml/backend/ggml/ggml: llama/vendor/ggml/
+.PHONY: ml/backend/ggml/ggml apply-patches
+ml/backend/ggml/ggml: llama/vendor/ggml/ apply-patches
 	rsync -arvzc -f "merge $@/.rsync-filter" $< $@

 PATCHES=$(wildcard llama/patches/*.patch)
-PATCHED=$(join $(dir $(PATCHES)), $(addsuffix ed, $(addprefix ., $(notdir $(PATCHES)))))

 .PHONY: apply-patches
 .NOTPARALLEL:
-apply-patches: $(PATCHED)
+apply-patches: $(addsuffix ed, $(PATCHES))

-llama/patches/.%.patched: llama/patches/%.patch
+%.patched: %.patch
 	@if git -c user.name=nobody -c 'user.email=<>' -C $(WORKDIR) am -3 $(realpath $<); then touch $@; else git -C $(WORKDIR) am --abort; exit 1; fi

 .PHONY: checkout
@@ -60,4 +57,4 @@ format-patches: llama/patches

 .PHONE: clean
 clean: checkout
-	$(RM) llama/patches/.*.patched
+	$(RM) $(addsuffix ed, $(PATCHES))
--- a/README.md
+++ b/README.md
@@ -61,8 +61,6 @@ Here are some example models that can be downloaded:
 | QwQ                | 32B        | 20GB  | `ollama run qwq`                 |
 | DeepSeek-R1        | 7B         | 4.7GB | `ollama run deepseek-r1`         |
 | DeepSeek-R1        | 671B       | 404GB | `ollama run deepseek-r1:671b`    |
-| Llama 4            | 109B       | 67GB  | `ollama run llama4:scout`        |
-| Llama 4            | 400B       | 245GB | `ollama run llama4:maverick`     |
 | Llama 3.3          | 70B        | 43GB  | `ollama run llama3.3`            |
 | Llama 3.2          | 3B         | 2.0GB | `ollama run llama3.2`            |
 | Llama 3.2          | 1B         | 1.3GB | `ollama run llama3.2:1b`         |
@@ -79,7 +77,7 @@ Here are some example models that can be downloaded:
 | Code Llama         | 7B         | 3.8GB | `ollama run codellama`           |
 | Llama 2 Uncensored | 7B         | 3.8GB | `ollama run llama2-uncensored`   |
 | LLaVA              | 7B         | 4.5GB | `ollama run llava`               |
-| Granite-3.3         | 8B         | 4.9GB | `ollama run granite3.3`          |
+| Granite-3.2         | 8B         | 4.9GB | `ollama run granite3.2`          |

 > [!NOTE]
 > You should have at least 8 GB of RAM available to run the 7B models, 16 GB to run the 13B models, and 32 GB to run the 33B models.
@@ -287,13 +285,12 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Bionic GPT](https://github.com/bionic-gpt/bionic-gpt)
 - [HTML UI](https://github.com/rtcfirefly/ollama-ui)
 - [Saddle](https://github.com/jikkuatwork/saddle)
- [TagSpaces](https://www.tagspaces.org) (A platform for file-based apps, [utilizing Ollama](https://docs.tagspaces.org/ai/) for the generation of tags and descriptions)
 - [Chatbot UI](https://github.com/ivanfioravanti/chatbot-ollama)
 - [Chatbot UI v2](https://github.com/mckaywrigley/chatbot-ui)
 - [Typescript UI](https://github.com/ollama-interface/Ollama-Gui?tab=readme-ov-file)
 - [Minimalistic React UI for Ollama Models](https://github.com/richawo/minimal-llm-ui)
 - [Ollamac](https://github.com/kevinhermawan/Ollamac)
- [big-AGI](https://github.com/enricoros/big-AGI)
+- [big-AGI](https://github.com/enricoros/big-AGI/blob/main/docs/config-local-ollama.md)
 - [Cheshire Cat assistant framework](https://github.com/cheshire-cat-ai/core)
 - [Amica](https://github.com/semperai/amica)
 - [chatd](https://github.com/BruceMacD/chatd)
@@ -314,8 +311,6 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Ollama Basic Chat: Uses HyperDiv Reactive UI](https://github.com/rapidarchitect/ollama_basic_chat)
 - [Ollama-chats RPG](https://github.com/drazdra/ollama-chats)
 - [IntelliBar](https://intellibar.app/) (AI-powered assistant for macOS)
- [Jirapt](https://github.com/AliAhmedNada/jirapt) (Jira Integration to generate issues, tasks, epics)
- [ojira](https://github.com/AliAhmedNada/ojira) (Jira chrome plugin to easily generate descriptions for tasks)
 - [QA-Pilot](https://github.com/reid41/QA-Pilot) (Interactive chat tool that can leverage Ollama models for rapid understanding and navigation of GitHub code repositories)
 - [ChatOllama](https://github.com/sugarforever/chat-ollama) (Open Source Chatbot based on Ollama with Knowledge Bases)
 - [CRAG Ollama Chat](https://github.com/Nagi-ovo/CRAG-Ollama-Chat) (Simple Web Search with Corrective RAG)
@@ -329,14 +324,13 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [RWKV-Runner](https://github.com/josStorer/RWKV-Runner) (RWKV offline LLM deployment tool, also usable as a client for ChatGPT and Ollama)
 - [Ollama Grid Search](https://github.com/dezoito/ollama-grid-search) (app to evaluate and compare models)
 - [Olpaka](https://github.com/Otacon/olpaka) (User-friendly Flutter Web App for Ollama)
- [Casibase](https://casibase.org) (An open source AI knowledge base and dialogue system combining the latest RAG, SSO, ollama support, and multiple large language models.)
 - [OllamaSpring](https://github.com/CrazyNeil/OllamaSpring) (Ollama Client for macOS)
 - [LLocal.in](https://github.com/kartikm7/llocal) (Easy to use Electron Desktop Client for Ollama)
 - [Shinkai Desktop](https://github.com/dcSpark/shinkai-apps) (Two click install Local AI using Ollama + Files + RAG)
- [AiLama](https://github.com/zeyoyt/ailama) (A Discord User App that allows you to interact with Ollama anywhere in Discord)
+- [AiLama](https://github.com/zeyoyt/ailama) (A Discord User App that allows you to interact with Ollama anywhere in discord )
 - [Ollama with Google Mesop](https://github.com/rapidarchitect/ollama_mesop/) (Mesop Chat Client implementation with Ollama)
 - [R2R](https://github.com/SciPhi-AI/R2R) (Open-source RAG engine)
- [Ollama-Kis](https://github.com/elearningshow/ollama-kis) (A simple easy-to-use GUI with sample custom LLM for Drivers Education)
+- [Ollama-Kis](https://github.com/elearningshow/ollama-kis) (A simple easy to use GUI with sample custom LLM for Drivers Education)
 - [OpenGPA](https://opengpa.org) (Open-source offline-first Enterprise Agentic Application)
 - [Painting Droid](https://github.com/mateuszmigas/painting-droid) (Painting app with AI integrations)
 - [Kerlig AI](https://www.kerlig.com/) (AI writing assistant for macOS)
@@ -345,16 +339,16 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [LLMStack](https://github.com/trypromptly/LLMStack) (No-code multi-agent framework to build LLM agents and workflows)
 - [BoltAI for Mac](https://boltai.com) (AI Chat Client for Mac)
 - [Harbor](https://github.com/av/harbor) (Containerized LLM Toolkit with Ollama as default backend)
- [PyGPT](https://github.com/szczyglis-dev/py-gpt) (AI desktop assistant for Linux, Windows, and Mac)
- [Alpaca](https://github.com/Jeffser/Alpaca) (An Ollama client application for Linux and macOS made with GTK4 and Adwaita)
+- [PyGPT](https://github.com/szczyglis-dev/py-gpt) (AI desktop assistant for Linux, Windows and Mac)
+- [Alpaca](https://github.com/Jeffser/Alpaca) (An Ollama client application for linux and macos made with GTK4 and Adwaita)
 - [AutoGPT](https://github.com/Significant-Gravitas/AutoGPT/blob/master/docs/content/platform/ollama.md) (AutoGPT Ollama integration)
 - [Go-CREW](https://www.jonathanhecl.com/go-crew/) (Powerful Offline RAG in Golang)
 - [PartCAD](https://github.com/openvmp/partcad/) (CAD model generation with OpenSCAD and CadQuery)
- [Ollama4j Web UI](https://github.com/ollama4j/ollama4j-web-ui) - Java-based Web UI for Ollama built with Vaadin, Spring Boot, and Ollama4j
+- [Ollama4j Web UI](https://github.com/ollama4j/ollama4j-web-ui) - Java-based Web UI for Ollama built with Vaadin, Spring Boot and Ollama4j
 - [PyOllaMx](https://github.com/kspviswa/pyOllaMx) - macOS application capable of chatting with both Ollama and Apple MLX models.
- [Cline](https://github.com/cline/cline) - Formerly known as Claude Dev is a VSCode extension for multi-file/whole-repo coding
+- [Claude Dev](https://github.com/saoudrizwan/claude-dev) - VSCode extension for multi-file/whole-repo coding
 - [Cherry Studio](https://github.com/kangfenmao/cherry-studio) (Desktop client with Ollama support)
- [ConfiChat](https://github.com/1runeberg/confichat) (Lightweight, standalone, multi-platform, and privacy-focused LLM chat interface with optional encryption)
+- [ConfiChat](https://github.com/1runeberg/confichat) (Lightweight, standalone, multi-platform, and privacy focused LLM chat interface with optional encryption)
 - [Archyve](https://github.com/nickthecook/archyve) (RAG-enabling document library)
 - [crewAI with Mesop](https://github.com/rapidarchitect/ollama-crew-mesop) (Mesop Web Interface to run crewAI with Ollama)
 - [Tkinter-based client](https://github.com/chyok/ollama-gui) (Python tkinter-based Client for Ollama)
@@ -372,7 +366,7 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [DualMind](https://github.com/tcsenpai/dualmind) (Experimental app allowing two models to talk to each other in the terminal or in a web interface)
 - [ollamarama-matrix](https://github.com/h1ddenpr0cess20/ollamarama-matrix) (Ollama chatbot for the Matrix chat protocol)
 - [ollama-chat-app](https://github.com/anan1213095357/ollama-chat-app) (Flutter-based chat app)
- [Perfect Memory AI](https://www.perfectmemory.ai/) (Productivity AI assists personalized by what you have seen on your screen, heard, and said in the meetings)
+- [Perfect Memory AI](https://www.perfectmemory.ai/) (Productivity AI assists personalized by what you have seen on your screen, heard and said in the meetings)
 - [Hexabot](https://github.com/hexastack/hexabot) (A conversational AI builder)
 - [Reddit Rate](https://github.com/rapidarchitect/reddit_analyzer) (Search and Rate Reddit topics with a weighted summation)
 - [OpenTalkGpt](https://github.com/adarshM84/OpenTalkGpt) (Chrome Extension to manage open-source models supported by Ollama, create custom models, and chat with models from a user-friendly UI)
@@ -390,7 +384,7 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [ChibiChat](https://github.com/CosmicEventHorizon/ChibiChat) (Kotlin-based Android app to chat with Ollama and Koboldcpp API endpoints)
 - [LocalLLM](https://github.com/qusaismael/localllm) (Minimal Web-App to run ollama models on it with a GUI)
 - [Ollamazing](https://github.com/buiducnhat/ollamazing) (Web extension to run Ollama models)
- [OpenDeepResearcher-via-searxng](https://github.com/benhaotang/OpenDeepResearcher-via-searxng) (A Deep Research equivalent endpoint with Ollama support for running locally)
+- [OpenDeepResearcher-via-searxng](https://github.com/benhaotang/OpenDeepResearcher-via-searxng) (A Deep Research equivent endpoint with Ollama support for running locally)
 - [AntSK](https://github.com/AIDotNet/AntSK) (Out-of-the-box & Adaptable RAG Chatbot)
 - [MaxKB](https://github.com/1Panel-dev/MaxKB/) (Ready-to-use & flexible RAG Chatbot)
 - [yla](https://github.com/danielekp/yla) (Web interface to freely interact with your customized models)
@@ -398,13 +392,8 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [1Panel](https://github.com/1Panel-dev/1Panel/) (Web-based Linux Server Management Tool)
 - [AstrBot](https://github.com/Soulter/AstrBot/) (User-friendly LLM-based multi-platform chatbot with a WebUI, supporting RAG, LLM agents, and plugins integration)
 - [Reins](https://github.com/ibrahimcetin/reins) (Easily tweak parameters, customize system prompts per chat, and enhance your AI experiments with reasoning model support.)
- [Flufy](https://github.com/Aharon-Bensadoun/Flufy) (A beautiful chat interface for interacting with Ollama's API. Built with React, TypeScript, and Material-UI.)
 - [Ellama](https://github.com/zeozeozeo/ellama) (Friendly native app to chat with an Ollama instance)
 - [screenpipe](https://github.com/mediar-ai/screenpipe) Build agents powered by your screen history
- [Ollamb](https://github.com/hengkysteen/ollamb) (Simple yet rich in features, cross-platform built with Flutter and designed for Ollama. Try the [web demo](https://hengkysteen.github.io/demo/ollamb/).)
- [Writeopia](https://github.com/Writeopia/Writeopia) (Text editor with integration with Ollama)
- [AppFlowy](https://github.com/AppFlowy-IO/AppFlowy) (AI collaborative workspace with Ollama, cross-platform and self-hostable)
- [Lumina](https://github.com/cushydigit/lumina.git) (A lightweight, minimal React.js frontend for interacting with Ollama servers)

 ### Cloud

@@ -444,10 +433,7 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [SwollamaCLI](https://github.com/marcusziade/Swollama) bundled with the Swollama Swift package. [Demo](https://github.com/marcusziade/Swollama?tab=readme-ov-file#cli-usage)
 - [aichat](https://github.com/sigoden/aichat) All-in-one LLM CLI tool featuring Shell Assistant, Chat-REPL, RAG, AI tools & agents, with access to OpenAI, Claude, Gemini, Ollama, Groq, and more.
 - [PowershAI](https://github.com/rrg92/powershai) PowerShell module that brings AI to terminal on Windows, including support for Ollama
- [DeepShell](https://github.com/Abyss-c0re/deepshell) Your self-hosted AI assistant. Interactive Shell, Files and Folders analysis.
 - [orbiton](https://github.com/xyproto/orbiton) Configuration-free text editor and IDE with support for tab completion with Ollama.
- [orca-cli](https://github.com/molbal/orca-cli) Ollama Registry CLI Application - Browse, pull, and download models from Ollama Registry in your terminal.
- [GGUF-to-Ollama](https://github.com/jonathanhecl/gguf-to-ollama) - Importing GGUF to Ollama made easy (multiplatform)

 ### Apple Vision Pro

@@ -474,7 +460,7 @@ See the [API documentation](./docs/api.md) for all endpoints.

 ### Libraries

- [LangChain](https://python.langchain.com/docs/integrations/chat/ollama/) and [LangChain.js](https://js.langchain.com/docs/integrations/chat/ollama/) with [example](https://js.langchain.com/docs/tutorials/local_rag/)
+- [LangChain](https://python.langchain.com/docs/integrations/llms/ollama) and [LangChain.js](https://js.langchain.com/docs/integrations/chat/ollama/) with [example](https://js.langchain.com/docs/tutorials/local_rag/)
 - [Firebase Genkit](https://firebase.google.com/docs/genkit/plugins/ollama)
 - [crewAI](https://github.com/crewAIInc/crewAI)
 - [Yacana](https://remembersoftwares.github.io/yacana/) (User-friendly multi-agent framework for brainstorming and executing predetermined flows with built-in tool integration)
@@ -521,21 +507,20 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Swollama for Swift](https://github.com/marcusziade/Swollama) with [DocC](https://marcusziade.github.io/Swollama/documentation/swollama/)
 - [GoLamify](https://github.com/prasad89/golamify)
 - [Ollama for Haskell](https://github.com/tusharad/ollama-haskell)
- [multi-llm-ts](https://github.com/nbonamy/multi-llm-ts) (A Typescript/JavaScript library allowing access to different LLM in a unified API)
+- [multi-llm-ts](https://github.com/nbonamy/multi-llm-ts) (A Typescript/JavaScript library allowing access to different LLM in unified API)
 - [LlmTornado](https://github.com/lofcz/llmtornado) (C# library providing a unified interface for major FOSS & Commercial inference APIs)
 - [Ollama for Zig](https://github.com/dravenk/ollama-zig)
 - [Abso](https://github.com/lunary-ai/abso) (OpenAI-compatible TypeScript SDK for any LLM provider)
 - [Nichey](https://github.com/goodreasonai/nichey) is a Python package for generating custom wikis for your research topic
 - [Ollama for D](https://github.com/kassane/ollama-d)
- [OllamaPlusPlus](https://github.com/HardCodeDev777/OllamaPlusPlus) (Very simple C++ library for Ollama)

 ### Mobile

- [SwiftChat](https://github.com/aws-samples/swift-chat) (Lightning-fast Cross-platform AI chat app with native UI for Android, iOS, and iPad)
+- [SwiftChat](https://github.com/aws-samples/swift-chat) (Lightning-fast Cross-platform AI chat app with native UI for Android, iOS and iPad)
 - [Enchanted](https://github.com/AugustDev/enchanted)
 - [Maid](https://github.com/Mobile-Artificial-Intelligence/maid)
 - [Ollama App](https://github.com/JHubi1/ollama-app) (Modern and easy-to-use multi-platform client for Ollama)
- [ConfiChat](https://github.com/1runeberg/confichat) (Lightweight, standalone, multi-platform, and privacy-focused LLM chat interface with optional encryption)
+- [ConfiChat](https://github.com/1runeberg/confichat) (Lightweight, standalone, multi-platform, and privacy focused LLM chat interface with optional encryption)
 - [Ollama Android Chat](https://github.com/sunshine0523/OllamaServer) (No need for Termux, start the Ollama service with one click on an Android device)
 - [Reins](https://github.com/ibrahimcetin/reins) (Easily tweak parameters, customize system prompts per chat, and enhance your AI experiments with reasoning model support.)

@@ -559,7 +544,7 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Obsidian Local GPT plugin](https://github.com/pfrankov/obsidian-local-gpt)
 - [Open Interpreter](https://docs.openinterpreter.com/language-model-setup/local-models/ollama)
 - [Llama Coder](https://github.com/ex3ndr/llama-coder) (Copilot alternative using Ollama)
- [Ollama Copilot](https://github.com/bernardo-bruning/ollama-copilot) (Proxy that allows you to use Ollama as a copilot like GitHub Copilot)
+- [Ollama Copilot](https://github.com/bernardo-bruning/ollama-copilot) (Proxy that allows you to use ollama as a copilot like Github copilot)
 - [twinny](https://github.com/rjmacarthy/twinny) (Copilot and Copilot chat alternative using Ollama)
 - [Wingman-AI](https://github.com/RussellCanfield/wingman-ai) (Copilot code and chat alternative using Ollama and Hugging Face)
 - [Page Assist](https://github.com/n4ze3m/page-assist) (Chrome Extension)
@@ -569,8 +554,8 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Discord-Ollama Chat Bot](https://github.com/kevinthedang/discord-ollama) (Generalized TypeScript Discord Bot w/ Tuning Documentation)
 - [ChatGPTBox: All in one browser extension](https://github.com/josStorer/chatGPTBox) with [Integrating Tutorial](https://github.com/josStorer/chatGPTBox/issues/616#issuecomment-1975186467)
 - [Discord AI chat/moderation bot](https://github.com/rapmd73/Companion) Chat/moderation bot written in python. Uses Ollama to create personalities.
- [Headless Ollama](https://github.com/nischalj10/headless-ollama) (Scripts to automatically install ollama client & models on any OS for apps that depend on ollama server)
- [Terraform AWS Ollama & Open WebUI](https://github.com/xuyangbocn/terraform-aws-self-host-llm) (A Terraform module to deploy on AWS a ready-to-use Ollama service, together with its front-end Open WebUI service.)
+- [Headless Ollama](https://github.com/nischalj10/headless-ollama) (Scripts to automatically install ollama client & models on any OS for apps that depends on ollama server)
+- [Terraform AWS Ollama & Open WebUI](https://github.com/xuyangbocn/terraform-aws-self-host-llm) (A Terraform module to deploy on AWS a ready-to-use Ollama service, together with its front end Open WebUI service.)
 - [node-red-contrib-ollama](https://github.com/jakubburkiewicz/node-red-contrib-ollama)
 - [Local AI Helper](https://github.com/ivostoykov/localAI) (Chrome and Firefox extensions that enable interactions with the active tab and customisable API endpoints. Includes secure storage for user prompts.)
 - [vnc-lm](https://github.com/jake83741/vnc-lm) (Discord bot for messaging with LLMs through Ollama and LiteLLM. Seamlessly move between local and flagship models.)
@@ -584,7 +569,6 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [Simple-Discord-AI](https://github.com/zyphixor/simple-discord-ai)
 - [LLM Telegram Bot](https://github.com/innightwolfsleep/llm_telegram_bot) (telegram bot, primary for RP. Oobabooga-like buttons, [A1111](https://github.com/AUTOMATIC1111/stable-diffusion-webui) API integration e.t.c)
 - [mcp-llm](https://github.com/sammcj/mcp-llm) (MCP Server to allow LLMs to call other LLMs)
- [UnityCodeLama](https://github.com/HardCodeDev777/UnityCodeLama) (Unity Edtior tool to analyze scripts via Ollama)

 ### Supported backends

--- a/api/client_test.go
+++ b/api/client_test.go
@@ -1,6 +1,7 @@
 package api

 import (
+	"context"
 	"encoding/json"
 	"fmt"
 	"net/http"
@@ -136,7 +137,7 @@ func TestClientStream(t *testing.T) {
 			client := NewClient(&url.URL{Scheme: "http", Host: ts.Listener.Addr().String()}, http.DefaultClient)

 			var receivedChunks []ChatResponse
-			err := client.stream(t.Context(), http.MethodPost, "/v1/chat", nil, func(chunk []byte) error {
+			err := client.stream(context.Background(), http.MethodPost, "/v1/chat", nil, func(chunk []byte) error {
 				var resp ChatResponse
 				if err := json.Unmarshal(chunk, &resp); err != nil {
 					return fmt.Errorf("failed to unmarshal chunk: %w", err)
@@ -222,7 +223,7 @@ func TestClientDo(t *testing.T) {
 				ID      string `json:"id"`
 				Success bool   `json:"success"`
 			}
-			err := client.do(t.Context(), http.MethodPost, "/v1/messages", nil, &resp)
+			err := client.do(context.Background(), http.MethodPost, "/v1/messages", nil, &resp)

 			if tc.wantErr != "" {
 				if err == nil {
--- a/api/types.go
+++ b/api/types.go
@@ -12,7 +12,6 @@ import (
 	"time"

 	"github.com/ollama/ollama/envconfig"
-	"github.com/ollama/ollama/types/model"
 )

 // StatusError is an error with an HTTP status code and message.
@@ -76,13 +75,13 @@ type GenerateRequest struct {
 	// this request.
 	KeepAlive *Duration `json:"keep_alive,omitempty"`

-	// Images is an optional list of raw image bytes accompanying this
+	// Images is an optional list of base64-encoded images accompanying this
 	// request, for multimodal models.
 	Images []ImageData `json:"images,omitempty"`

 	// Options lists model-specific options. For example, temperature can be
 	// set through this field, if the model supports it.
-	Options map[string]any `json:"options"`
+	Options map[string]interface{} `json:"options"`
 }

 // ChatRequest describes a request sent by [Client.Chat].
@@ -107,7 +106,7 @@ type ChatRequest struct {
 	Tools `json:"tools,omitempty"`

 	// Options lists model-specific options.
-	Options map[string]any `json:"options"`
+	Options map[string]interface{} `json:"options"`
 }

 type Tools []Tool
@@ -163,65 +162,19 @@ func (t *ToolCallFunctionArguments) String() string {

 type Tool struct {
 	Type     string       `json:"type"`
-	Items    any          `json:"items,omitempty"`
 	Function ToolFunction `json:"function"`
 }

-// PropertyType can be either a string or an array of strings
-type PropertyType []string
-
-// UnmarshalJSON implements the json.Unmarshaler interface
-func (pt *PropertyType) UnmarshalJSON(data []byte) error {
-	// Try to unmarshal as a string first
-	var s string
-	if err := json.Unmarshal(data, &s); err == nil {
-		*pt = []string{s}
-		return nil
-	}
-
-	// If that fails, try to unmarshal as an array of strings
-	var a []string
-	if err := json.Unmarshal(data, &a); err != nil {
-		return err
-	}
-	*pt = a
-	return nil
-}
-
-// MarshalJSON implements the json.Marshaler interface
-func (pt PropertyType) MarshalJSON() ([]byte, error) {
-	if len(pt) == 1 {
-		// If there's only one type, marshal as a string
-		return json.Marshal(pt[0])
-	}
-	// Otherwise marshal as an array
-	return json.Marshal([]string(pt))
-}
-
-// String returns a string representation of the PropertyType
-func (pt PropertyType) String() string {
-	if len(pt) == 0 {
-		return ""
-	}
-	if len(pt) == 1 {
-		return pt[0]
-	}
-	return fmt.Sprintf("%v", []string(pt))
-}
-
 type ToolFunction struct {
 	Name        string `json:"name"`
 	Description string `json:"description"`
 	Parameters  struct {
 		Type       string   `json:"type"`
-		Defs       any      `json:"$defs,omitempty"`
-		Items      any      `json:"items,omitempty"`
 		Required   []string `json:"required"`
 		Properties map[string]struct {
-			Type        PropertyType `json:"type"`
-			Items       any          `json:"items,omitempty"`
-			Description string       `json:"description"`
-			Enum        []any        `json:"enum,omitempty"`
+			Type        string   `json:"type"`
+			Description string   `json:"description"`
+			Enum        []string `json:"enum,omitempty"`
 		} `json:"properties"`
 	} `json:"parameters"`
 }
@@ -271,6 +224,9 @@ type Options struct {
 	RepeatPenalty    float32  `json:"repeat_penalty,omitempty"`
 	PresencePenalty  float32  `json:"presence_penalty,omitempty"`
 	FrequencyPenalty float32  `json:"frequency_penalty,omitempty"`
+	Mirostat         int      `json:"mirostat,omitempty"`
+	MirostatTau      float32  `json:"mirostat_tau,omitempty"`
+	MirostatEta      float32  `json:"mirostat_eta,omitempty"`
 	Stop             []string `json:"stop,omitempty"`
 }

@@ -280,7 +236,12 @@ type Runner struct {
 	NumBatch  int   `json:"num_batch,omitempty"`
 	NumGPU    int   `json:"num_gpu,omitempty"`
 	MainGPU   int   `json:"main_gpu,omitempty"`
+	LowVRAM   bool  `json:"low_vram,omitempty"`
+	F16KV     bool  `json:"f16_kv,omitempty"` // Deprecated: This option is ignored
+	LogitsAll bool  `json:"logits_all,omitempty"`
+	VocabOnly bool  `json:"vocab_only,omitempty"`
 	UseMMap   *bool `json:"use_mmap,omitempty"`
+	UseMLock  bool  `json:"use_mlock,omitempty"`
 	NumThread int   `json:"num_thread,omitempty"`
 }

@@ -299,7 +260,7 @@ type EmbedRequest struct {
 	Truncate *bool `json:"truncate,omitempty"`

 	// Options lists model-specific options.
-	Options map[string]any `json:"options"`
+	Options map[string]interface{} `json:"options"`
 }

 // EmbedResponse is the response from [Client.Embed].
@@ -325,7 +286,7 @@ type EmbeddingRequest struct {
 	KeepAlive *Duration `json:"keep_alive,omitempty"`

 	// Options lists model-specific options.
-	Options map[string]any `json:"options"`
+	Options map[string]interface{} `json:"options"`
 }

 // EmbeddingResponse is the response from [Client.Embeddings].
@@ -371,7 +332,7 @@ type ShowRequest struct {
 	Template string `json:"template"`
 	Verbose  bool   `json:"verbose"`

-	Options map[string]any `json:"options"`
+	Options map[string]interface{} `json:"options"`

 	// Deprecated: set the model name with Model instead
 	Name string `json:"name"`
@@ -379,18 +340,17 @@ type ShowRequest struct {

 // ShowResponse is the response returned from [Client.Show].
 type ShowResponse struct {
-	License       string             `json:"license,omitempty"`
-	Modelfile     string             `json:"modelfile,omitempty"`
-	Parameters    string             `json:"parameters,omitempty"`
-	Template      string             `json:"template,omitempty"`
-	System        string             `json:"system,omitempty"`
-	Details       ModelDetails       `json:"details,omitempty"`
-	Messages      []Message          `json:"messages,omitempty"`
-	ModelInfo     map[string]any     `json:"model_info,omitempty"`
-	ProjectorInfo map[string]any     `json:"projector_info,omitempty"`
-	Tensors       []Tensor           `json:"tensors,omitempty"`
-	Capabilities  []model.Capability `json:"capabilities,omitempty"`
-	ModifiedAt    time.Time          `json:"modified_at,omitempty"`
+	License       string         `json:"license,omitempty"`
+	Modelfile     string         `json:"modelfile,omitempty"`
+	Parameters    string         `json:"parameters,omitempty"`
+	Template      string         `json:"template,omitempty"`
+	System        string         `json:"system,omitempty"`
+	Details       ModelDetails   `json:"details,omitempty"`
+	Messages      []Message      `json:"messages,omitempty"`
+	ModelInfo     map[string]any `json:"model_info,omitempty"`
+	ProjectorInfo map[string]any `json:"projector_info,omitempty"`
+	Tensors       []Tensor       `json:"tensors,omitempty"`
+	ModifiedAt    time.Time      `json:"modified_at,omitempty"`
 }

 // CopyRequest is the request passed to [Client.Copy].
@@ -463,6 +423,13 @@ type ProcessModelResponse struct {
 	SizeVRAM  int64        `json:"size_vram"`
 }

+type RetrieveModelResponse struct {
+	Id      string `json:"id"`
+	Object  string `json:"object"`
+	Created int64  `json:"created"`
+	OwnedBy string `json:"owned_by"`
+}
+
 type TokenResponse struct {
 	Token string `json:"token"`
 }
@@ -536,7 +503,7 @@ func (m *Metrics) Summary() {
 	}
 }

-func (opts *Options) FromMap(m map[string]any) error {
+func (opts *Options) FromMap(m map[string]interface{}) error {
 	valueOpts := reflect.ValueOf(opts).Elem() // names of the fields in the options struct
 	typeOpts := reflect.TypeOf(opts).Elem()   // types of the fields in the options struct

@@ -593,12 +560,12 @@ func (opts *Options) FromMap(m map[string]any) error {
 				}
 				field.SetString(val)
 			case reflect.Slice:
-				// JSON unmarshals to []any, not []string
-				val, ok := val.([]any)
+				// JSON unmarshals to []interface{}, not []string
+				val, ok := val.([]interface{})
 				if !ok {
 					return fmt.Errorf("option %q must be of type array", key)
 				}
-				// convert []any to []string
+				// convert []interface{} to []string
 				slice := make([]string, len(val))
 				for i, item := range val {
 					str, ok := item.(string)
@@ -645,6 +612,9 @@ func DefaultOptions() Options {
 		RepeatPenalty:    1.1,
 		PresencePenalty:  0.0,
 		FrequencyPenalty: 0.0,
+		Mirostat:         0,
+		MirostatTau:      5.0,
+		MirostatEta:      0.1,
 		Seed:             -1,

 		Runner: Runner{
@@ -653,6 +623,8 @@ func DefaultOptions() Options {
 			NumBatch:  512,
 			NumGPU:    -1, // -1 here indicates that NumGPU should be set dynamically
 			NumThread: 0,  // let the runtime decide
+			LowVRAM:   false,
+			UseMLock:  false,
 			UseMMap:   nil,
 		},
 	}
@@ -700,7 +672,7 @@ func (d *Duration) UnmarshalJSON(b []byte) (err error) {
 }

 // FormatParams converts specified parameter options to their correct types
-func FormatParams(params map[string][]string) (map[string]any, error) {
+func FormatParams(params map[string][]string) (map[string]interface{}, error) {
 	opts := Options{}
 	valueOpts := reflect.ValueOf(&opts).Elem() // names of the fields in the options struct
 	typeOpts := reflect.TypeOf(opts)           // types of the fields in the options struct
@@ -714,7 +686,7 @@ func FormatParams(params map[string][]string) (map[string]any, error) {
 		}
 	}

-	out := make(map[string]any)
+	out := make(map[string]interface{})
 	// iterate params and set values based on json struct tags
 	for key, vals := range params {
 		if opt, ok := jsonOpts[key]; !ok {
--- a/api/types_test.go
+++ b/api/types_test.go
@@ -134,7 +134,7 @@ func TestUseMmapParsingFromJSON(t *testing.T) {

 	for _, test := range tests {
 		t.Run(test.name, func(t *testing.T) {
-			var oMap map[string]any
+			var oMap map[string]interface{}
 			err := json.Unmarshal([]byte(test.req), &oMap)
 			require.NoError(t, err)
 			opts := DefaultOptions()
@@ -231,144 +231,3 @@ func TestMessage_UnmarshalJSON(t *testing.T) {
 		}
 	}
 }
-
-func TestToolFunction_UnmarshalJSON(t *testing.T) {
-	tests := []struct {
-		name    string
-		input   string
-		wantErr string
-	}{
-		{
-			name: "valid enum with same types",
-			input: `{
-				"name": "test",
-				"description": "test function",
-				"parameters": {
-					"type": "object",
-					"required": ["test"],
-					"properties": {
-						"test": {
-							"type": "string",
-							"description": "test prop",
-							"enum": ["a", "b", "c"]
-						}
-					}
-				}
-			}`,
-			wantErr: "",
-		},
-		{
-			name: "empty enum array",
-			input: `{
-				"name": "test",
-				"description": "test function",
-				"parameters": {
-					"type": "object",
-					"required": ["test"],
-					"properties": {
-						"test": {
-							"type": "string",
-							"description": "test prop",
-							"enum": []
-						}
-					}
-				}
-			}`,
-			wantErr: "",
-		},
-	}
-
-	for _, tt := range tests {
-		t.Run(tt.name, func(t *testing.T) {
-			var tf ToolFunction
-			err := json.Unmarshal([]byte(tt.input), &tf)
-
-			if tt.wantErr != "" {
-				require.Error(t, err)
-				assert.Contains(t, err.Error(), tt.wantErr)
-			} else {
-				require.NoError(t, err)
-			}
-		})
-	}
-}
-
-func TestPropertyType_UnmarshalJSON(t *testing.T) {
-	tests := []struct {
-		name     string
-		input    string
-		expected PropertyType
-	}{
-		{
-			name:     "string type",
-			input:    `"string"`,
-			expected: PropertyType{"string"},
-		},
-		{
-			name:     "array of types",
-			input:    `["string", "number"]`,
-			expected: PropertyType{"string", "number"},
-		},
-		{
-			name:     "array with single type",
-			input:    `["string"]`,
-			expected: PropertyType{"string"},
-		},
-	}
-
-	for _, test := range tests {
-		t.Run(test.name, func(t *testing.T) {
-			var pt PropertyType
-			if err := json.Unmarshal([]byte(test.input), &pt); err != nil {
-				t.Errorf("Unexpected error: %v", err)
-			}
-
-			if len(pt) != len(test.expected) {
-				t.Errorf("Length mismatch: got %v, expected %v", len(pt), len(test.expected))
-			}
-
-			for i, v := range pt {
-				if v != test.expected[i] {
-					t.Errorf("Value mismatch at index %d: got %v, expected %v", i, v, test.expected[i])
-				}
-			}
-		})
-	}
-}
-
-func TestPropertyType_MarshalJSON(t *testing.T) {
-	tests := []struct {
-		name     string
-		input    PropertyType
-		expected string
-	}{
-		{
-			name:     "single type",
-			input:    PropertyType{"string"},
-			expected: `"string"`,
-		},
-		{
-			name:     "multiple types",
-			input:    PropertyType{"string", "number"},
-			expected: `["string","number"]`,
-		},
-		{
-			name:     "empty type",
-			input:    PropertyType{},
-			expected: `[]`,
-		},
-	}
-
-	for _, test := range tests {
-		t.Run(test.name, func(t *testing.T) {
-			data, err := json.Marshal(test.input)
-			if err != nil {
-				t.Errorf("Unexpected error: %v", err)
-			}
-
-			if string(data) != test.expected {
-				t.Errorf("Marshaled data mismatch: got %v, expected %v", string(data), test.expected)
-			}
-		})
-	}
-}
--- a/app/lifecycle/logging.go
+++ b/app/lifecycle/logging.go
@@ -4,14 +4,20 @@ import (
 	"fmt"
 	"log/slog"
 	"os"
+	"path/filepath"
 	"strconv"
 	"strings"

 	"github.com/ollama/ollama/envconfig"
-	"github.com/ollama/ollama/logutil"
 )

 func InitLogging() {
+	level := slog.LevelInfo
+
+	if envconfig.Debug() {
+		level = slog.LevelDebug
+	}
+
 	var logFile *os.File
 	var err error
 	// Detect if we're a GUI app on windows, and if not, send logs to console
@@ -27,8 +33,20 @@ func InitLogging() {
 			return
 		}
 	}
+	handler := slog.NewTextHandler(logFile, &slog.HandlerOptions{
+		Level:     level,
+		AddSource: true,
+		ReplaceAttr: func(_ []string, attr slog.Attr) slog.Attr {
+			if attr.Key == slog.SourceKey {
+				source := attr.Value.Any().(*slog.Source)
+				source.File = filepath.Base(source.File)
+			}
+			return attr
+		},
+	})
+
+	slog.SetDefault(slog.New(handler))

-	slog.SetDefault(logutil.NewLogger(logFile, envconfig.LogLevel()))
 	slog.Info("ollama app started")
 }

--- a/benchmark/server_benchmark_test.go
+++ b/benchmark/server_benchmark_test.go
@@ -78,7 +78,7 @@ func BenchmarkColdStart(b *testing.B) {

 	for _, tt := range tests {
 		b.Run(fmt.Sprintf("%s/cold/%s", m, tt.name), func(b *testing.B) {
-			ctx := b.Context()
+			ctx := context.Background()

 			// Set number of tokens as our throughput metric
 			b.SetBytes(int64(tt.maxTokens))
@@ -92,7 +92,7 @@ func BenchmarkColdStart(b *testing.B) {
 				req := &api.GenerateRequest{
 					Model:   m,
 					Prompt:  tt.prompt,
-					Options: map[string]any{"num_predict": tt.maxTokens, "temperature": 0.1},
+					Options: map[string]interface{}{"num_predict": tt.maxTokens, "temperature": 0.1},
 				}

 				runGenerateBenchmark(b, ctx, client, req)
@@ -113,7 +113,7 @@ func BenchmarkWarmStart(b *testing.B) {

 	for _, tt := range tests {
 		b.Run(fmt.Sprintf("%s/warm/%s", m, tt.name), func(b *testing.B) {
-			ctx := b.Context()
+			ctx := context.Background()

 			// Pre-warm the model
 			warmup(client, m, tt.prompt, b)
@@ -140,7 +140,7 @@ func setup(b *testing.B) *api.Client {
 	if err != nil {
 		b.Fatal(err)
 	}
-	if _, err := client.Show(b.Context(), &api.ShowRequest{Model: modelName(b)}); err != nil {
+	if _, err := client.Show(context.Background(), &api.ShowRequest{Model: modelName(b)}); err != nil {
 		b.Fatalf("Model unavailable: %v", err)
 	}

@@ -155,7 +155,7 @@ func warmup(client *api.Client, model string, prompt string, b *testing.B) {
 			&api.GenerateRequest{
 				Model:   model,
 				Prompt:  prompt,
-				Options: map[string]any{"num_predict": 50, "temperature": 0.1},
+				Options: map[string]interface{}{"num_predict": 50, "temperature": 0.1},
 			},
 			func(api.GenerateResponse) error { return nil },
 		)
--- a/cmd/cmd.go
+++ b/cmd/cmd.go
@@ -18,7 +18,6 @@ import (
 	"os/signal"
 	"path/filepath"
 	"runtime"
-	"slices"
 	"sort"
 	"strconv"
 	"strings"
@@ -31,7 +30,6 @@ import (
 	"github.com/olekukonko/tablewriter"
 	"github.com/spf13/cobra"
 	"golang.org/x/crypto/ssh"
-	"golang.org/x/sync/errgroup"
 	"golang.org/x/term"

 	"github.com/ollama/ollama/api"
@@ -42,7 +40,6 @@ import (
 	"github.com/ollama/ollama/runner"
 	"github.com/ollama/ollama/server"
 	"github.com/ollama/ollama/types/model"
-	"github.com/ollama/ollama/types/syncmap"
 	"github.com/ollama/ollama/version"
 )

@@ -108,7 +105,7 @@ func CreateHandler(cmd *cobra.Command, args []string) error {
 	}
 	spinner.Stop()

-	req.Model = args[0]
+	req.Name = args[0]
 	quantize, _ := cmd.Flags().GetString("quantize")
 	if quantize != "" {
 		req.Quantize = quantize
@@ -119,54 +116,34 @@ func CreateHandler(cmd *cobra.Command, args []string) error {
 		return err
 	}

-	var g errgroup.Group
-	g.SetLimit(max(runtime.GOMAXPROCS(0)-1, 1))
-
-	files := syncmap.NewSyncMap[string, string]()
-	for f, digest := range req.Files {
-		g.Go(func() error {
+	if len(req.Files) > 0 {
+		fileMap := map[string]string{}
+		for f, digest := range req.Files {
 			if _, err := createBlob(cmd, client, f, digest, p); err != nil {
 				return err
 			}
-
-			// TODO: this is incorrect since the file might be in a subdirectory
-			//       instead this should take the path relative to the model directory
-			//       but the current implementation does not allow this
-			files.Store(filepath.Base(f), digest)
-			return nil
-		})
+			fileMap[filepath.Base(f)] = digest
+		}
+		req.Files = fileMap
 	}

-	adapters := syncmap.NewSyncMap[string, string]()
-	for f, digest := range req.Adapters {
-		g.Go(func() error {
+	if len(req.Adapters) > 0 {
+		fileMap := map[string]string{}
+		for f, digest := range req.Adapters {
 			if _, err := createBlob(cmd, client, f, digest, p); err != nil {
 				return err
 			}
-
-			// TODO: same here
-			adapters.Store(filepath.Base(f), digest)
-			return nil
-		})
+			fileMap[filepath.Base(f)] = digest
+		}
+		req.Adapters = fileMap
 	}

-	if err := g.Wait(); err != nil {
-		return err
-	}
-
-	req.Files = files.Items()
-	req.Adapters = adapters.Items()
-
 	bars := make(map[string]*progress.Bar)
 	fn := func(resp api.ProgressResponse) error {
 		if resp.Digest != "" {
 			bar, ok := bars[resp.Digest]
 			if !ok {
-				msg := resp.Status
-				if msg == "" {
-					msg = fmt.Sprintf("pulling %s...", resp.Digest[7:19])
-				}
-				bar = progress.NewBar(msg, resp.Total, resp.Completed)
+				bar = progress.NewBar(fmt.Sprintf("pulling %s...", resp.Digest[7:19]), resp.Total, resp.Completed)
 				bars[resp.Digest] = bar
 				p.Add(resp.Digest, bar)
 			}
@@ -235,7 +212,7 @@ func createBlob(cmd *cobra.Command, client *api.Client, path string, digest stri
 		}
 	}()

-	if err := client.CreateBlob(cmd.Context(), digest, io.TeeReader(bin, &pw)); err != nil {
+	if err = client.CreateBlob(cmd.Context(), digest, io.TeeReader(bin, &pw)); err != nil {
 		return "", err
 	}
 	return digest, nil
@@ -290,7 +267,7 @@ func RunHandler(cmd *cobra.Command, args []string) error {
 	opts := runOptions{
 		Model:    args[0],
 		WordWrap: os.Getenv("TERM") == "xterm-256color",
-		Options:  map[string]any{},
+		Options:  map[string]interface{}{},
 	}

 	format, err := cmd.Flags().GetString("format")
@@ -362,11 +339,6 @@ func RunHandler(cmd *cobra.Command, args []string) error {
 		return err
 	}

-	opts.MultiModal = slices.Contains(info.Capabilities, model.CapabilityVision)
-
-	// TODO: remove the projector info and vision info checks below,
-	// these are left in for backwards compatibility with older servers
-	// that don't have the capabilities field in the model info
 	if len(info.ProjectorInfo) != 0 {
 		opts.MultiModal = true
 	}
@@ -697,15 +669,6 @@ func showInfo(resp *api.ShowResponse, verbose bool, w io.Writer) error {
 		return
 	})

-	if len(resp.Capabilities) > 0 {
-		tableRender("Capabilities", func() (rows [][]string) {
-			for _, capability := range resp.Capabilities {
-				rows = append(rows, []string{"", capability.String()})
-			}
-			return
-		})
-	}
-
 	if resp.ProjectorInfo != nil {
 		tableRender("Projector", func() (rows [][]string) {
 			arch := resp.ProjectorInfo["general.architecture"].(string)
@@ -830,38 +793,13 @@ func PullHandler(cmd *cobra.Command, args []string) error {

 	fn := func(resp api.ProgressResponse) error {
 		if resp.Digest != "" {
-			if resp.Completed == 0 {
-				// This is the initial status update for the
-				// layer, which the server sends before
-				// beginning the download, for clients to
-				// compute total size and prepare for
-				// downloads, if needed.
-				//
-				// Skipping this here to avoid showing a 0%
-				// progress bar, which *should* clue the user
-				// into the fact that many things are being
-				// downloaded and that the current active
-				// download is not that last. However, in rare
-				// cases it seems to be triggering to some, and
-				// it isn't worth explaining, so just ignore
-				// and regress to the old UI that keeps giving
-				// you the "But wait, there is more!" after
-				// each "100% done" bar, which is "better."
-				return nil
-			}
-
 			if spinner != nil {
 				spinner.Stop()
 			}

 			bar, ok := bars[resp.Digest]
 			if !ok {
-				name, isDigest := strings.CutPrefix(resp.Digest, "sha256:")
-				name = strings.TrimSpace(name)
-				if isDigest {
-					name = name[:min(12, len(name))]
-				}
-				bar = progress.NewBar(fmt.Sprintf("pulling %s:", name), resp.Total, resp.Completed)
+				bar = progress.NewBar(fmt.Sprintf("pulling %s...", resp.Digest[7:19]), resp.Total, resp.Completed)
 				bars[resp.Digest] = bar
 				p.Add(resp.Digest, bar)
 			}
@@ -881,7 +819,11 @@ func PullHandler(cmd *cobra.Command, args []string) error {
 	}

 	request := api.PullRequest{Name: args[0], Insecure: insecure}
-	return client.Pull(cmd.Context(), &request, fn)
+	if err := client.Pull(cmd.Context(), &request, fn); err != nil {
+		return err
+	}
+
+	return nil
 }

 type generateContextKey string
@@ -895,7 +837,7 @@ type runOptions struct {
 	Format      string
 	System      string
 	Images      []api.ImageData
-	Options     map[string]any
+	Options     map[string]interface{}
 	MultiModal  bool
 	KeepAlive   *api.Duration
 }
@@ -1424,6 +1366,7 @@ func NewCLI() *cobra.Command {
 				envVars["OLLAMA_NOPRUNE"],
 				envVars["OLLAMA_ORIGINS"],
 				envVars["OLLAMA_SCHED_SPREAD"],
+				envVars["OLLAMA_TMPDIR"],
 				envVars["OLLAMA_FLASH_ATTENTION"],
 				envVars["OLLAMA_KV_CACHE_TYPE"],
 				envVars["OLLAMA_LLM_LIBRARY"],
--- a/cmd/cmd_test.go
+++ b/cmd/cmd_test.go
@@ -2,6 +2,7 @@ package cmd

 import (
 	"bytes"
+	"context"
 	"encoding/json"
 	"io"
 	"net/http"
@@ -15,7 +16,6 @@ import (
 	"github.com/spf13/cobra"

 	"github.com/ollama/ollama/api"
-	"github.com/ollama/ollama/types/model"
 )

 func TestShowInfo(t *testing.T) {
@@ -260,34 +260,6 @@ Weigh anchor!
 			t.Errorf("unexpected output (-want +got):\n%s", diff)
 		}
 	})
-
-	t.Run("capabilities", func(t *testing.T) {
-		var b bytes.Buffer
-		if err := showInfo(&api.ShowResponse{
-			Details: api.ModelDetails{
-				Family:            "test",
-				ParameterSize:     "7B",
-				QuantizationLevel: "FP16",
-			},
-			Capabilities: []model.Capability{model.CapabilityVision, model.CapabilityTools},
-		}, false, &b); err != nil {
-			t.Fatal(err)
-		}
-
-		expect := "  Model\n" +
-			"    architecture    test    \n" +
-			"    parameters      7B      \n" +
-			"    quantization    FP16    \n" +
-			"\n" +
-			"  Capabilities\n" +
-			"    vision    \n" +
-			"    tools     \n" +
-			"\n"
-
-		if diff := cmp.Diff(expect, b.String()); diff != "" {
-			t.Errorf("unexpected output (-want +got):\n%s", diff)
-		}
-	})
 }

 func TestDeleteHandler(t *testing.T) {
@@ -336,7 +308,7 @@ func TestDeleteHandler(t *testing.T) {
 	t.Cleanup(mockServer.Close)

 	cmd := &cobra.Command{}
-	cmd.SetContext(t.Context())
+	cmd.SetContext(context.TODO())
 	if err := DeleteHandler(cmd, []string{"test-model"}); err != nil {
 		t.Fatalf("DeleteHandler failed: %v", err)
 	}
@@ -398,6 +370,11 @@ func TestGetModelfileName(t *testing.T) {
 			var expectedFilename string

 			if tt.fileExists {
+				tempDir, err := os.MkdirTemp("", "modelfiledir")
+				defer os.RemoveAll(tempDir)
+				if err != nil {
+					t.Fatalf("temp modelfile dir creation failed: %v", err)
+				}
 				var fn string
 				if tt.modelfileName != "" {
 					fn = tt.modelfileName
@@ -405,11 +382,10 @@ func TestGetModelfileName(t *testing.T) {
 					fn = "Modelfile"
 				}

-				tempFile, err := os.CreateTemp(t.TempDir(), fn)
+				tempFile, err := os.CreateTemp(tempDir, fn)
 				if err != nil {
 					t.Fatalf("temp modelfile creation failed: %v", err)
 				}
-				defer tempFile.Close()

 				expectedFilename = tempFile.Name()
 				err = cmd.Flags().Set("file", expectedFilename)
@@ -524,7 +500,7 @@ func TestPushHandler(t *testing.T) {

 			cmd := &cobra.Command{}
 			cmd.Flags().Bool("insecure", false, "")
-			cmd.SetContext(t.Context())
+			cmd.SetContext(context.TODO())

 			// Redirect stderr to capture progress output
 			oldStderr := os.Stderr
@@ -629,7 +605,7 @@ func TestListHandler(t *testing.T) {
 			t.Setenv("OLLAMA_HOST", mockServer.URL)

 			cmd := &cobra.Command{}
-			cmd.SetContext(t.Context())
+			cmd.SetContext(context.TODO())

 			// Capture stdout
 			oldStdout := os.Stdout
@@ -684,7 +660,7 @@ func TestCreateHandler(t *testing.T) {
 						return
 					}

-					if req.Model != "test-model" {
+					if req.Name != "test-model" {
 						t.Errorf("expected model name 'test-model', got %s", req.Name)
 					}

@@ -724,7 +700,7 @@ func TestCreateHandler(t *testing.T) {
 			}))
 			t.Setenv("OLLAMA_HOST", mockServer.URL)
 			t.Cleanup(mockServer.Close)
-			tempFile, err := os.CreateTemp(t.TempDir(), "modelfile")
+			tempFile, err := os.CreateTemp("", "modelfile")
 			if err != nil {
 				t.Fatal(err)
 			}
@@ -744,7 +720,7 @@ func TestCreateHandler(t *testing.T) {
 			}

 			cmd.Flags().Bool("insecure", false, "")
-			cmd.SetContext(t.Context())
+			cmd.SetContext(context.TODO())

 			// Redirect stderr to capture progress output
 			oldStderr := os.Stderr
--- a/cmd/interactive.go
+++ b/cmd/interactive.go
@@ -44,7 +44,7 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 		fmt.Fprintln(os.Stderr, "Use \"\"\" to begin a multi-line message.")

 		if opts.MultiModal {
-			fmt.Fprintf(os.Stderr, "Use %s to include .jpg, .png, or .webp images.\n", filepath.FromSlash("/path/to/file"))
+			fmt.Fprintf(os.Stderr, "Use %s to include .jpg or .png images.\n", filepath.FromSlash("/path/to/file"))
 		}

 		fmt.Fprintln(os.Stderr, "")
@@ -503,7 +503,6 @@ func normalizeFilePath(fp string) string {
 		"\\\\", "\\", // Escaped backslash
 		"\\*", "*", // Escaped asterisk
 		"\\?", "?", // Escaped question mark
-		"\\~", "~", // Escaped tilde
 	).Replace(fp)
 }

@@ -511,7 +510,7 @@ func extractFileNames(input string) []string {
 	// Regex to match file paths starting with optional drive letter, / ./ \ or .\ and include escaped or unescaped spaces (\ or %20)
 	// and followed by more characters and a file extension
 	// This will capture non filename strings, but we'll check for file existence to remove mismatches
-	regexPattern := `(?:[a-zA-Z]:)?(?:\./|/|\\)[\S\\ ]+?\.(?i:jpg|jpeg|png|webp)\b`
+	regexPattern := `(?:[a-zA-Z]:)?(?:\./|/|\\)[\S\\ ]+?\.(?i:jpg|jpeg|png)\b`
 	re := regexp.MustCompile(regexPattern)

 	return re.FindAllString(input, -1)
@@ -531,8 +530,6 @@ func extractFileData(input string) (string, []api.ImageData, error) {
 			return "", imgs, err
 		}
 		fmt.Fprintf(os.Stderr, "Added image '%s'\n", nfp)
-		input = strings.ReplaceAll(input, "'"+nfp+"'", "")
-		input = strings.ReplaceAll(input, "'"+fp+"'", "")
 		input = strings.ReplaceAll(input, fp, "")
 		imgs = append(imgs, data)
 	}
@@ -553,7 +550,7 @@ func getImageData(filePath string) ([]byte, error) {
 	}

 	contentType := http.DetectContentType(buf)
-	allowedTypes := []string{"image/jpeg", "image/jpg", "image/png", "image/webp"}
+	allowedTypes := []string{"image/jpeg", "image/jpg", "image/png"}
 	if !slices.Contains(allowedTypes, contentType) {
 		return nil, fmt.Errorf("invalid image type: %s", contentType)
 	}
--- a/cmd/interactive_test.go
+++ b/cmd/interactive_test.go
@@ -1,8 +1,6 @@
 package cmd

 import (
-	"os"
-	"path/filepath"
 	"testing"

 	"github.com/stretchr/testify/assert"
@@ -12,17 +10,14 @@ func TestExtractFilenames(t *testing.T) {
 	// Unix style paths
 	input := ` some preamble 
 ./relative\ path/one.png inbetween1 ./not a valid two.jpg inbetween2 ./1.svg
-/unescaped space /three.jpeg inbetween3 /valid\ path/dir/four.png "./quoted with spaces/five.JPG
-/unescaped space /six.webp inbetween6 /valid\ path/dir/seven.WEBP`
+/unescaped space /three.jpeg inbetween3 /valid\ path/dir/four.png "./quoted with spaces/five.JPG`
 	res := extractFileNames(input)
-	assert.Len(t, res, 7)
+	assert.Len(t, res, 5)
 	assert.Contains(t, res[0], "one.png")
 	assert.Contains(t, res[1], "two.jpg")
 	assert.Contains(t, res[2], "three.jpeg")
 	assert.Contains(t, res[3], "four.png")
 	assert.Contains(t, res[4], "five.JPG")
-	assert.Contains(t, res[5], "six.webp")
-	assert.Contains(t, res[6], "seven.WEBP")
 	assert.NotContains(t, res[4], '"')
 	assert.NotContains(t, res, "inbetween1")
 	assert.NotContains(t, res, "./1.svg")
@@ -33,12 +28,10 @@ func TestExtractFilenames(t *testing.T) {
 /absolute/nospace/three.jpeg inbetween3 /absolute/with space/four.png inbetween4
 ./relative\ path/five.JPG inbetween5 "./relative with/spaces/six.png inbetween6
 d:\path with\spaces\seven.JPEG inbetween7 c:\users\jdoe\eight.png inbetween8 
- d:\program files\someplace\nine.png inbetween9 "E:\program files\someplace\ten.PNG
-c:/users/jdoe/eleven.webp inbetween11 c:/program files/someplace/twelve.WebP inbetween12
-d:\path with\spaces\thirteen.WEBP some ending
+ d:\program files\someplace\nine.png inbetween9 "E:\program files\someplace\ten.PNG some ending
 `
 	res = extractFileNames(input)
-	assert.Len(t, res, 13)
+	assert.Len(t, res, 10)
 	assert.NotContains(t, res, "inbetween2")
 	assert.Contains(t, res[0], "one.png")
 	assert.Contains(t, res[0], "c:")
@@ -56,31 +49,4 @@ d:\path with\spaces\thirteen.WEBP some ending
 	assert.Contains(t, res[8], "d:")
 	assert.Contains(t, res[9], "ten.PNG")
 	assert.Contains(t, res[9], "E:")
-	assert.Contains(t, res[10], "eleven.webp")
-	assert.Contains(t, res[10], "c:")
-	assert.Contains(t, res[11], "twelve.WebP")
-	assert.Contains(t, res[11], "c:")
-	assert.Contains(t, res[12], "thirteen.WEBP")
-	assert.Contains(t, res[12], "d:")
-}
-
-// Ensure that file paths wrapped in single quotes are removed with the quotes.
-func TestExtractFileDataRemovesQuotedFilepath(t *testing.T) {
-	dir := t.TempDir()
-	fp := filepath.Join(dir, "img.jpg")
-	data := make([]byte, 600)
-	copy(data, []byte{
-		0xff, 0xd8, 0xff, 0xe0, 0x00, 0x10, 'J', 'F', 'I', 'F',
-		0x00, 0x01, 0x01, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
-		0xff, 0xd9,
-	})
-	if err := os.WriteFile(fp, data, 0o600); err != nil {
-		t.Fatalf("failed to write test image: %v", err)
-	}
-
-	input := "before '" + fp + "' after"
-	cleaned, imgs, err := extractFileData(input)
-	assert.NoError(t, err)
-	assert.Len(t, imgs, 1)
-	assert.Equal(t, cleaned, "before  after")
 }
--- a/convert/convert.go
+++ b/convert/convert.go
@@ -1,26 +1,25 @@
 package convert

 import (
-	"cmp"
 	"encoding/json"
 	"errors"
 	"fmt"
+	"io"
 	"io/fs"
 	"log/slog"
-	"os"
-	"slices"
 	"strings"

 	"github.com/ollama/ollama/fs/ggml"
 )

 type ModelParameters struct {
-	Architectures []string `json:"architectures"`
-	VocabSize     uint32   `json:"vocab_size"`
+	Architectures []string       `json:"architectures"`
+	VocabSize     uint32         `json:"vocab_size"`
+	TextModel     TextParameters `json:"text_config"`
+}

-	TextModel struct {
-		VocabSize uint32 `json:"vocab_size"`
-	} `json:"text_config"`
+type TextParameters struct {
+	VocabSize uint32 `json:"vocab_size"`
 }

 type AdapterParameters struct {
@@ -85,17 +84,27 @@ func (ModelParameters) specialTokenTypes() []string {
 	}
 }

+func (ModelParameters) writeFile(ws io.WriteSeeker, kv ggml.KV, ts []ggml.Tensor) error {
+	return ggml.WriteGGUF(ws, kv, ts)
+}
+
+func (AdapterParameters) writeFile(ws io.WriteSeeker, kv ggml.KV, ts []ggml.Tensor) error {
+	return ggml.WriteGGUF(ws, kv, ts)
+}
+
 type ModelConverter interface {
 	// KV maps parameters to LLM key-values
 	KV(*Tokenizer) ggml.KV
 	// Tensors maps input tensors to LLM tensors. Model specific modifications can be done here.
-	Tensors([]Tensor) []*ggml.Tensor
+	Tensors([]Tensor) []ggml.Tensor
 	// Replacements returns a list of string pairs to replace in tensor names.
 	// See [strings.Replacer](https://pkg.go.dev/strings#Replacer) for details
 	Replacements() []string

 	// specialTokenTypes returns any special token types the model uses
 	specialTokenTypes() []string
+	// writeFile writes the model to the provided io.WriteSeeker
+	writeFile(io.WriteSeeker, ggml.KV, []ggml.Tensor) error
 }

 type moreParser interface {
@@ -106,13 +115,15 @@ type AdapterConverter interface {
 	// KV maps parameters to LLM key-values
 	KV(ggml.KV) ggml.KV
 	// Tensors maps input tensors to LLM tensors. Adapter specific modifications can be done here.
-	Tensors([]Tensor) []*ggml.Tensor
+	Tensors([]Tensor) []ggml.Tensor
 	// Replacements returns a list of string pairs to replace in tensor names.
 	// See [strings.Replacer](https://pkg.go.dev/strings#Replacer) for details
 	Replacements() []string
+
+	writeFile(io.WriteSeeker, ggml.KV, []ggml.Tensor) error
 }

-func ConvertAdapter(fsys fs.FS, f *os.File, baseKV ggml.KV) error {
+func ConvertAdapter(fsys fs.FS, ws io.WriteSeeker, baseKV ggml.KV) error {
 	bts, err := fs.ReadFile(fsys, "adapter_config.json")
 	if err != nil {
 		return err
@@ -147,14 +158,14 @@ func ConvertAdapter(fsys fs.FS, f *os.File, baseKV ggml.KV) error {
 		return err
 	}

-	return writeFile(f, conv.KV(baseKV), conv.Tensors(ts))
+	return conv.writeFile(ws, conv.KV(baseKV), conv.Tensors(ts))
 }

 // Convert writes an Ollama compatible model to the provided io.WriteSeeker based on configurations
 // and files it finds in the input path.
 // Supported input model formats include safetensors.
 // Supported input tokenizers files include tokenizer.json (preferred) and tokenizer.model.
-func ConvertModel(fsys fs.FS, f *os.File) error {
+func ConvertModel(fsys fs.FS, ws io.WriteSeeker) error {
 	bts, err := fs.ReadFile(fsys, "config.json")
 	if err != nil {
 		return err
@@ -171,14 +182,8 @@ func ConvertModel(fsys fs.FS, f *os.File) error {

 	var conv ModelConverter
 	switch p.Architectures[0] {
-	case "LlamaForCausalLM":
+	case "LlamaForCausalLM", "MistralForCausalLM":
 		conv = &llamaModel{}
-	case "MllamaForConditionalGeneration":
-		conv = &mllamaModel{}
-	case "Llama4ForConditionalGeneration":
-		conv = &llama4Model{}
-	case "Mistral3ForConditionalGeneration":
-		conv = &mistral3Model{}
 	case "MixtralForCausalLM":
 		conv = &mixtralModel{}
 	case "GemmaForCausalLM":
@@ -191,8 +196,6 @@ func ConvertModel(fsys fs.FS, f *os.File) error {
 		conv = &phi3Model{}
 	case "Qwen2ForCausalLM":
 		conv = &qwen2Model{}
-	case "Qwen2_5_VLForConditionalGeneration":
-		conv = &qwen25VLModel{}
 	case "BertModel":
 		conv = &bertModel{}
 	case "CohereForCausalLM":
@@ -216,22 +219,24 @@ func ConvertModel(fsys fs.FS, f *os.File) error {
 		return err
 	}

-	vocabSize := int(cmp.Or(p.VocabSize, p.TextModel.VocabSize))
+	vocabSize := int(p.VocabSize)
+	if vocabSize == 0 {
+		tVocabSize := int(p.TextModel.VocabSize)
+		vocabSize = tVocabSize
+	}

 	switch {
 	case vocabSize == 0:
-		slog.Debug("vocabulary size was not explicitly set by the model", "default size", len(t.Vocabulary.Tokens))
+		slog.Warn("vocabulary size was not explicitly set by the model", "default size", len(t.Vocabulary.Tokens))
 	case vocabSize > len(t.Vocabulary.Tokens):
-		slog.Debug("vocabulary is smaller than expected, padding with dummy tokens", "expect", vocabSize, "actual", len(t.Vocabulary.Tokens))
+		slog.Warn("vocabulary is smaller than expected, padding with dummy tokens", "expect", vocabSize, "actual", len(t.Vocabulary.Tokens))
 		for i := range vocabSize - len(t.Vocabulary.Tokens) {
 			t.Vocabulary.Tokens = append(t.Vocabulary.Tokens, fmt.Sprintf("[PAD%d]", i))
 			t.Vocabulary.Scores = append(t.Vocabulary.Scores, -1)
 			t.Vocabulary.Types = append(t.Vocabulary.Types, tokenTypeUserDefined)
 		}
 	case vocabSize < len(t.Vocabulary.Tokens):
-		slog.Debug("vocabulary is larger than expected", "want", vocabSize, "got", len(t.Vocabulary.Tokens))
-		p.VocabSize = uint32(len(t.Vocabulary.Tokens))
-		p.TextModel.VocabSize = uint32(len(t.Vocabulary.Tokens))
+		return fmt.Errorf("vocabulary is larger than expected '%d' instead of '%d'", len(t.Vocabulary.Tokens), vocabSize)
 	default:
 		slog.Debug("vocabulary", "size", len(t.Vocabulary.Tokens))
 	}
@@ -241,13 +246,5 @@ func ConvertModel(fsys fs.FS, f *os.File) error {
 		return err
 	}

-	return writeFile(f, conv.KV(t), conv.Tensors(ts))
-}
-
-func writeFile(f *os.File, kv ggml.KV, ts []*ggml.Tensor) error {
-	for i := range ts {
-		ts[i].Shape = slices.Clone(ts[i].Shape)
-		slices.Reverse(ts[i].Shape)
-	}
-	return ggml.WriteGGUF(f, kv, ts)
+	return conv.writeFile(ws, conv.KV(t), conv.Tensors(ts))
 }
--- a/convert/convert_bert.go
+++ b/convert/convert_bert.go
@@ -132,8 +132,8 @@ func (p *bertModel) KV(t *Tokenizer) ggml.KV {
 	return kv
 }

-func (p *bertModel) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (p *bertModel) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor
 	for _, t := range ts {
 		if slices.Contains([]string{
 			"embeddings.position_ids",
@@ -143,7 +143,7 @@ func (p *bertModel) Tensors(ts []Tensor) []*ggml.Tensor {
 			continue
 		}

-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
--- a/convert/convert_commandr.go
+++ b/convert/convert_commandr.go
@@ -43,10 +43,10 @@ func (p *commandrModel) KV(t *Tokenizer) ggml.KV {
 	return kv
 }

-func (p *commandrModel) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (p *commandrModel) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor
 	for _, t := range ts {
-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
--- a/convert/convert_gemma.go
+++ b/convert/convert_gemma.go
@@ -42,14 +42,14 @@ func (p *gemmaModel) KV(t *Tokenizer) ggml.KV {
 	return kv
 }

-func (p *gemmaModel) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (p *gemmaModel) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor
 	for _, t := range ts {
 		if !strings.HasPrefix(t.Name(), "v.") && strings.HasSuffix(t.Name(), "_norm.weight") {
 			t.SetRepacker(p.addOne)
 		}

-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
--- a/convert/convert_gemma2_adapter.go
+++ b/convert/convert_gemma2_adapter.go
@@ -21,8 +21,8 @@ func (p *gemma2Adapter) KV(baseKV ggml.KV) ggml.KV {
 	return kv
 }

-func (p *gemma2Adapter) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (p *gemma2Adapter) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor
 	for _, t := range ts {
 		shape := t.Shape()
 		if (strings.HasSuffix(t.Name(), "weight.lora_a") && shape[0] > shape[1]) ||
@@ -31,7 +31,7 @@ func (p *gemma2Adapter) Tensors(ts []Tensor) []*ggml.Tensor {
 			t.SetRepacker(p.repack)
 		}

-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
--- a/convert/convert_llama.go
+++ b/convert/convert_llama.go
@@ -28,12 +28,12 @@ type llamaModel struct {
 	NumKeyValueHeads      uint32  `json:"num_key_value_heads"`
 	RopeTheta             float32 `json:"rope_theta"`
 	RopeScaling           struct {
-		Type                          string  `json:"type"`
-		RopeType                      string  `json:"rope_type"`
-		Factor                        float32 `json:"factor"`
-		LowFrequencyFactor            float32 `json:"low_freq_factor"`
-		HighFrequencyFactor           float32 `json:"high_freq_factor"`
-		OriginalMaxPositionEmbeddings uint32  `json:"original_max_position_embeddings"`
+		Type                            string  `json:"type"`
+		RopeType                        string  `json:"rope_type"`
+		Factor                          float32 `json:"factor"`
+		LowFrequencyFactor              float32 `json:"low_freq_factor"`
+		HighFrequencyFactor             float32 `json:"high_freq_factor"`
+		OriginalMaxPositionalEmbeddings uint32  `json:"original_max_positional_embeddings"`

 		factors ropeFactor
 	} `json:"rope_scaling"`
@@ -42,8 +42,6 @@ type llamaModel struct {
 	LayerNormEpsilon float32 `json:"layer_norm_epsilon"`
 	NormEpsilon      float32 `json:"norm_epsilon"`
 	HeadDim          uint32  `json:"head_dim"`
-
-	skipRepack bool
 }

 var _ ModelConverter = (*llamaModel)(nil)
@@ -72,10 +70,6 @@ func (p *llamaModel) KV(t *Tokenizer) ggml.KV {
 		kv["llama.rope.dimension_count"] = p.HiddenSize / headCount
 	}

-	if p.HeadDim > 0 {
-		kv["llama.attention.head_dim"] = p.HeadDim
-	}
-
 	if p.RopeTheta > 0 {
 		kv["llama.rope.freq_base"] = p.RopeTheta
 	}
@@ -90,7 +84,7 @@ func (p *llamaModel) KV(t *Tokenizer) ggml.KV {
 			factorLow := cmp.Or(p.RopeScaling.LowFrequencyFactor, 1.0)
 			factorHigh := cmp.Or(p.RopeScaling.HighFrequencyFactor, 4.0)

-			original := cmp.Or(p.RopeScaling.OriginalMaxPositionEmbeddings, 8192)
+			original := cmp.Or(p.RopeScaling.OriginalMaxPositionalEmbeddings, 8192)
 			lambdaLow := float32(original) / factorLow
 			lambdaHigh := float32(original) / factorHigh

@@ -126,11 +120,11 @@ func (p *llamaModel) KV(t *Tokenizer) ggml.KV {
 	return kv
 }

-func (p *llamaModel) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (p *llamaModel) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor

 	if p.RopeScaling.factors != nil {
-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     "rope_freqs.weight",
 			Kind:     0,
 			Shape:    []uint64{uint64(len(p.RopeScaling.factors))},
@@ -139,13 +133,12 @@ func (p *llamaModel) Tensors(ts []Tensor) []*ggml.Tensor {
 	}

 	for _, t := range ts {
-		if strings.HasSuffix(t.Name(), "attn_q.weight") || strings.HasSuffix(t.Name(), "attn_k.weight") {
-			if !p.skipRepack {
-				t.SetRepacker(p.repack)
-			}
+		if strings.HasSuffix(t.Name(), "attn_q.weight") ||
+			strings.HasSuffix(t.Name(), "attn_k.weight") {
+			t.SetRepacker(p.repack)
 		}

-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
--- a/convert/convert_llama4.go
+++ b/convert/convert_llama4.go
@@ -1,169 +0,0 @@
-package convert
-
-import (
-	"slices"
-	"strings"
-
-	"github.com/pdevine/tensor"
-	"github.com/pdevine/tensor/native"
-
-	"github.com/ollama/ollama/fs/ggml"
-)
-
-type llama4Model struct {
-	ModelParameters
-	TextModel struct {
-		llamaModel
-		NumExpertsPerToken     uint32 `json:"num_experts_per_tok"`
-		NumLocalExperts        uint32 `json:"num_local_experts"`
-		InterleaveMOELayerStep uint32 `json:"interleave_moe_layer_step"`
-		UseQKNorm              bool   `json:"use_qk_norm"`
-		IntermediateSizeMLP    uint32 `json:"intermediate_size_mlp"`
-		AttentionChunkSize     uint32 `json:"attention_chunk_size"`
-	} `json:"text_config"`
-	VisionModel struct {
-		NumHiddenLayers   uint32  `json:"num_hidden_layers"`
-		HiddenSize        uint32  `json:"hidden_size"`
-		IntermediateSize  uint32  `json:"intermediate_size"`
-		NumAttentionHeads uint32  `json:"num_attention_heads"`
-		ImageSize         uint32  `json:"image_size"`
-		PatchSize         uint32  `json:"patch_size"`
-		RopeTheta         float32 `json:"rope_theta"`
-		NormEpsilon       float32 `json:"norm_eps"`
-		PixelShuffleRatio float32 `json:"pixel_shuffle_ratio"`
-	} `json:"vision_config"`
-}
-
-// KV implements ModelConverter.
-func (p *llama4Model) KV(t *Tokenizer) ggml.KV {
-	kv := p.ModelParameters.KV(t)
-	kv["general.architecture"] = "llama4"
-
-	for k, v := range p.TextModel.KV(t) {
-		if strings.HasPrefix(k, "llama.") {
-			kv[strings.ReplaceAll(k, "llama.", "llama4.")] = v
-		}
-	}
-
-	kv["llama4.feed_forward_length"] = p.TextModel.IntermediateSizeMLP
-	kv["llama4.expert_feed_forward_length"] = p.TextModel.IntermediateSize
-
-	kv["llama4.expert_count"] = p.TextModel.NumLocalExperts
-	kv["llama4.expert_used_count"] = p.TextModel.NumExpertsPerToken
-	kv["llama4.interleave_moe_layer_step"] = p.TextModel.InterleaveMOELayerStep
-	kv["llama4.use_qk_norm"] = p.TextModel.UseQKNorm
-	kv["llama4.attention.chunk_size"] = p.TextModel.AttentionChunkSize
-
-	kv["llama4.vision.block_count"] = p.VisionModel.NumHiddenLayers
-	kv["llama4.vision.embedding_length"] = p.VisionModel.HiddenSize
-	kv["llama4.vision.feed_forward_length"] = p.VisionModel.IntermediateSize
-	kv["llama4.vision.attention.head_count"] = p.VisionModel.NumAttentionHeads
-	kv["llama4.vision.image_size"] = p.VisionModel.ImageSize
-	kv["llama4.vision.patch_size"] = p.VisionModel.PatchSize
-	kv["llama4.vision.rope.freq_base"] = p.VisionModel.RopeTheta
-	kv["llama4.vision.layer_norm_epsilon"] = p.VisionModel.NormEpsilon
-	kv["llama4.vision.pixel_shuffle_ratio"] = p.VisionModel.PixelShuffleRatio
-	return kv
-}
-
-// Replacements implements ModelConverter.
-func (p *llama4Model) Replacements() []string {
-	return append(
-		p.TextModel.Replacements(),
-		"language_model.", "",
-		"vision_model", "v",
-		"multi_modal_projector", "mm",
-		"feed_forward.down_proj", "ffn_down",
-		"feed_forward.up_proj", "ffn_up",
-		"feed_forward.gate_proj", "ffn_gate",
-		"feed_forward.", "ffn_",
-		"shared_expert.down_proj", "down_shexp",
-		"shared_expert.gate_proj", "gate_shexp",
-		"shared_expert.up_proj", "up_shexp",
-		"experts.down_proj", "down_exps.weight",
-		"experts.gate_up_proj", "gate_up_exps.weight",
-		"router", "gate_inp",
-		"patch_embedding.linear", "patch_embedding",
-	)
-}
-
-// Tensors implements ModelConverter.
-func (p *llama4Model) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
-
-	var textTensors []Tensor
-	for _, t := range ts {
-		if strings.HasPrefix(t.Name(), "v.") || strings.HasPrefix(t.Name(), "mm.") {
-			out = append(out, &ggml.Tensor{
-				Name:     t.Name(),
-				Kind:     t.Kind(),
-				Shape:    t.Shape(),
-				WriterTo: t,
-			})
-		} else if strings.Contains(t.Name(), "ffn_gate_up_exps") {
-			// gate and up projectors are fused
-			// dims[1], dims[2] must be swapped
-			// [experts, hidden_size, intermediate_size * 2] --> [experts, intermediate_size, hidden_size]
-			halfDim := int(t.Shape()[2]) / 2
-
-			newShape := slices.Clone(t.Shape())
-			newShape[1], newShape[2] = newShape[2]/2, newShape[1]
-			for i, name := range []string{"ffn_gate_exps", "ffn_up_exps"} {
-				// clone tensor since we need separate repackers
-				tt := t.Clone()
-				tt.SetRepacker(p.repack(nil, nil, tensor.S(i*halfDim, (i+1)*halfDim)))
-				out = append(out, &ggml.Tensor{
-					Name:     strings.ReplaceAll(tt.Name(), "ffn_gate_up_exps", name),
-					Kind:     tt.Kind(),
-					Shape:    newShape,
-					WriterTo: tt,
-				})
-			}
-		} else if strings.Contains(t.Name(), "ffn_down_exps") {
-			// dims[1], dims[2] must be swapped
-			// [experts, intermediate_size, hidden_size] --> [experts, hidden_size, intermediate_size]
-			t.SetRepacker(p.repack())
-			newShape := slices.Clone(t.Shape())
-			newShape[1], newShape[2] = newShape[2], newShape[1]
-			out = append(out, &ggml.Tensor{
-				Name:     t.Name(),
-				Kind:     t.Kind(),
-				Shape:    newShape,
-				WriterTo: t,
-			})
-		} else {
-			textTensors = append(textTensors, t)
-		}
-	}
-
-	p.TextModel.skipRepack = true
-	out = append(out, p.TextModel.Tensors(textTensors)...)
-	return out
-}
-
-func (p *llama4Model) repack(slice ...tensor.Slice) Repacker {
-	return func(name string, data []float32, shape []uint64) ([]float32, error) {
-		dims := make([]int, len(shape))
-		for i, dim := range shape {
-			dims[i] = int(dim)
-		}
-
-		var t tensor.Tensor = tensor.New(tensor.WithShape(dims...), tensor.WithBacking(data))
-		t, err := t.Slice(slice...)
-		if err != nil {
-			return nil, err
-		}
-
-		if err := t.T(0, 2, 1); err != nil {
-			return nil, err
-		}
-
-		t = tensor.Materialize(t)
-		// flatten tensor so it can be return as a vector
-		if err := t.Reshape(t.Shape().TotalSize()); err != nil {
-			return nil, err
-		}
-
-		return native.VectorF32(t.(*tensor.Dense))
-	}
-}
--- a/convert/convert_llama_adapter.go
+++ b/convert/convert_llama_adapter.go
@@ -29,8 +29,8 @@ func (p *llamaAdapter) KV(baseKV ggml.KV) ggml.KV {
 	return kv
 }

-func (p *llamaAdapter) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (p *llamaAdapter) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor
 	for _, t := range ts {
 		shape := t.Shape()
 		if (strings.HasSuffix(t.Name(), "weight.lora_a") && shape[0] > shape[1]) ||
@@ -41,7 +41,7 @@ func (p *llamaAdapter) Tensors(ts []Tensor) []*ggml.Tensor {
 			t.SetRepacker(p.repack)
 		}

-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    shape,
--- a/convert/convert_mistral.go
+++ b/convert/convert_mistral.go
@@ -1,190 +0,0 @@
-package convert
-
-import (
-	"cmp"
-	"fmt"
-	"strings"
-
-	"github.com/pdevine/tensor"
-	"github.com/pdevine/tensor/native"
-
-	"github.com/ollama/ollama/fs/ggml"
-)
-
-type mistral3Model struct {
-	ModelParameters
-	ImageTokenIndex    uint32 `json:"image_token_index"`
-	SpatialMergeSize   uint32 `json:"spatial_merge_size"`
-	VisionFeatureLayer int32  `json:"vision_feature_layer"`
-	TextModel          struct {
-		NumHiddenLayers       uint32  `json:"num_hidden_layers"`
-		MaxPositionEmbeddings uint32  `json:"max_position_embeddings"`
-		HiddenSize            uint32  `json:"hidden_size"`
-		IntermediateSize      uint32  `json:"intermediate_size"`
-		NumAttentionHeads     uint32  `json:"num_attention_heads"`
-		NumKeyValueHeads      uint32  `json:"num_key_value_heads"`
-		RopeTheta             float32 `json:"rope_theta"`
-		RMSNormEPS            float32 `json:"rms_norm_eps"`
-		HeadDim               uint32  `json:"head_dim"`
-		SlidingWindow         *uint32 `json:"sliding_window"`
-		HiddenAct             string  `json:"hidden_act"`
-		VocabSize             uint32  `json:"vocab_size"`
-	} `json:"text_config"`
-	VisionModel struct {
-		NumAttentionHeads uint32  `json:"num_attention_heads"`
-		NumHiddenLayers   uint32  `json:"num_hidden_layers"`
-		HiddenSize        uint32  `json:"hidden_size"`
-		IntermediateSize  uint32  `json:"intermediate_size"`
-		ImageSize         uint32  `json:"image_size"`
-		NumChannels       uint32  `json:"num_channels"`
-		PatchSize         uint32  `json:"patch_size"`
-		HeadDim           uint32  `json:"head_dim"`
-		HiddenAct         string  `json:"hidden_act"`
-		RopeTheta         float32 `json:"rope_theta"`
-	} `json:"vision_config"`
-	MultiModalProjectorBias bool   `json:"multimodal_projector_bias"`
-	ProjectorHiddenAct      string `json:"projector_hidden_act"`
-}
-
-func (p *mistral3Model) KV(t *Tokenizer) ggml.KV {
-	kv := p.ModelParameters.KV(t)
-	kv["general.architecture"] = "mistral3"
-	kv["mistral3.vocab_size"] = p.TextModel.VocabSize
-
-	// Text configuration
-	kv["mistral3.block_count"] = p.TextModel.NumHiddenLayers
-	kv["mistral3.context_length"] = p.TextModel.MaxPositionEmbeddings
-	kv["mistral3.embedding_length"] = p.TextModel.HiddenSize
-	kv["mistral3.feed_forward_length"] = p.TextModel.IntermediateSize
-	kv["mistral3.attention.head_count"] = p.TextModel.NumAttentionHeads
-	kv["mistral3.attention.head_count_kv"] = p.TextModel.NumKeyValueHeads
-	kv["mistral3.attention.layer_norm_rms_epsilon"] = p.TextModel.RMSNormEPS
-	kv["mistral3.attention.key_length"] = p.TextModel.HeadDim
-	kv["mistral3.attention.value_length"] = p.TextModel.HeadDim
-	kv["mistral3.rope.dimension_count"] = p.TextModel.HiddenSize / p.TextModel.NumHiddenLayers
-	kv["mistral3.rope.freq_base"] = p.TextModel.RopeTheta
-
-	// Vision configuration
-	kv["mistral3.vision.block_count"] = p.VisionModel.NumHiddenLayers
-	kv["mistral3.vision.embedding_length"] = p.VisionModel.HiddenSize
-	kv["mistral3.vision.feed_forward_length"] = p.VisionModel.IntermediateSize
-	kv["mistral3.vision.attention.head_count"] = p.VisionModel.NumAttentionHeads
-	kv["mistral3.vision.attention.key_length"] = p.VisionModel.HeadDim
-	kv["mistral3.vision.image_size"] = p.VisionModel.ImageSize
-	kv["mistral3.vision.patch_size"] = p.VisionModel.PatchSize
-	kv["mistral3.vision.num_channels"] = p.VisionModel.NumChannels
-	// kv["mistral3.vision.attention.layer_norm_epsilon"] = 1e-05 // Default value
-	kv["mistral3.vision.rope.freq_base"] = p.VisionModel.RopeTheta
-
-	// Multimodal configuration
-	kv["mistral3.image_token_index"] = p.ImageTokenIndex
-	kv["mistral3.spatial_merge_size"] = p.SpatialMergeSize
-
-	kv["mistral3.mm.projector_bias"] = p.MultiModalProjectorBias
-
-	if p.ProjectorHiddenAct != "" {
-		kv["mistral3.mm.projector_hidden_act"] = p.ProjectorHiddenAct
-	}
-
-	return kv
-}
-
-func (p *mistral3Model) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
-
-	for _, t := range ts {
-		if !strings.HasPrefix(t.Name(), "v.") {
-			if strings.HasSuffix(t.Name(), ".attn_q.weight") ||
-				strings.HasSuffix(t.Name(), ".attn_k.weight") {
-				t.SetRepacker(p.repack)
-			}
-		}
-
-		out = append(out, &ggml.Tensor{
-			Name:     t.Name(),
-			Kind:     t.Kind(),
-			Shape:    t.Shape(),
-			WriterTo: t,
-		})
-	}
-
-	return out
-}
-
-func (p *mistral3Model) Replacements() []string {
-	return []string{
-		"language_model.model.norm", "output_norm",
-		"language_model.model.", "",
-		"language_model.", "",
-		"layers", "blk",
-		"transformer.layers", "blk",
-		"vision_tower", "v",
-		"ln_pre", "encoder_norm",
-		"input_layernorm", "attn_norm",
-		"post_attention_layernorm", "ffn_norm",
-		"embed_tokens", "token_embd",
-		"self_attn.q_proj", "attn_q",
-		"self_attn.k_proj", "attn_k",
-		"self_attn.v_proj", "attn_v",
-		"self_attn.o_proj", "attn_output",
-		"mlp.down_proj", "ffn_down",
-		"mlp.gate_proj", "ffn_gate",
-		"mlp.up_proj", "ffn_up",
-		"attention.q_proj", "attn_q",
-		"attention.k_proj", "attn_k",
-		"attention.v_proj", "attn_v",
-		"attention.o_proj", "attn_output",
-		"attention_norm", "attn_norm",
-		"feed_forward.gate_proj", "ffn_gate",
-		"feed_forward.down_proj", "ffn_down",
-		"feed_forward.up_proj", "ffn_up",
-		"multi_modal_projector", "mm",
-		"ffn_norm", "ffn_norm",
-		"lm_head", "output",
-	}
-}
-
-func (p *mistral3Model) repack(name string, data []float32, shape []uint64) ([]float32, error) {
-	var dims []int
-	for _, dim := range shape {
-		dims = append(dims, int(dim))
-	}
-
-	var heads uint32
-	if strings.HasSuffix(name, ".attn_q.weight") {
-		heads = p.TextModel.NumAttentionHeads
-	} else if strings.HasSuffix(name, ".attn_k.weight") {
-		heads = cmp.Or(p.TextModel.NumKeyValueHeads, p.TextModel.NumAttentionHeads)
-	} else {
-		return nil, fmt.Errorf("unknown tensor for repack: %s", name)
-	}
-
-	n := tensor.New(tensor.WithShape(dims...), tensor.WithBacking(data))
-	if err := n.Reshape(append([]int{int(heads), 2, dims[0] / int(heads) / 2}, dims[1:]...)...); err != nil {
-		return nil, err
-	}
-
-	if err := n.T(0, 2, 1, 3); err != nil {
-		return nil, err
-	}
-
-	if err := n.Reshape(dims...); err != nil {
-		return nil, err
-	}
-
-	if err := n.Transpose(); err != nil {
-		return nil, err
-	}
-
-	ts, err := native.SelectF32(n, 1)
-	if err != nil {
-		return nil, err
-	}
-
-	var f32s []float32
-	for _, t := range ts {
-		f32s = append(f32s, t...)
-	}
-
-	return f32s, nil
-}
--- a/convert/convert_mixtral.go
+++ b/convert/convert_mixtral.go
@@ -29,7 +29,7 @@ func (p *mixtralModel) KV(t *Tokenizer) ggml.KV {
 	return kv
 }

-func (p *mixtralModel) Tensors(ts []Tensor) []*ggml.Tensor {
+func (p *mixtralModel) Tensors(ts []Tensor) []ggml.Tensor {
 	oldnew := []string{
 		"model.layers", "blk",
 		"w1", "ffn_gate_exps",
@@ -56,10 +56,10 @@ func (p *mixtralModel) Tensors(ts []Tensor) []*ggml.Tensor {
 		return true
 	})

-	var out []*ggml.Tensor
+	var out []ggml.Tensor
 	for n, e := range experts {
 		// TODO(mxyng): sanity check experts
-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     n,
 			Kind:     e[0].Kind(),
 			Shape:    append([]uint64{uint64(len(e))}, e[0].Shape()...),
--- a/convert/convert_mllama.go
+++ b/convert/convert_mllama.go
@@ -1,160 +0,0 @@
-package convert
-
-import (
-	"strings"
-
-	"github.com/ollama/ollama/fs/ggml"
-	"github.com/pdevine/tensor"
-	"github.com/pdevine/tensor/native"
-)
-
-type mllamaModel struct {
-	ModelParameters
-	TextModel struct {
-		llamaModel
-
-		CrossAttentionLayers []int32 `json:"cross_attention_layers"`
-	} `json:"text_config"`
-	VisionModel struct {
-		NumHiddenLayers           uint32  `json:"num_hidden_layers"`
-		NumGlobalLayers           uint32  `json:"num_global_layers"`
-		IntermediateLayersIndices []int32 `json:"intermediate_layers_indices"`
-
-		HiddenSize       uint32 `json:"hidden_size"`
-		IntermediateSize uint32 `json:"intermediate_size"`
-
-		AttentionHeads uint32 `json:"attention_heads"`
-
-		ImageSize   uint32  `json:"image_size"`
-		PatchSize   uint32  `json:"patch_size"`
-		NumChannels uint32  `json:"num_channels"`
-		MaxNumTiles uint32  `json:"max_num_tiles"`
-		NormEpsilon float32 `json:"norm_eps"`
-		RopeTheta   float32 `json:"rope.freq_base"`
-	} `json:"vision_config"`
-}
-
-func (m *mllamaModel) KV(t *Tokenizer) ggml.KV {
-	kv := m.ModelParameters.KV(t)
-	kv["general.architecture"] = "mllama"
-
-	for k, v := range m.TextModel.KV(t) {
-		if strings.HasPrefix(k, "llama.") {
-			kv[strings.ReplaceAll(k, "llama.", "mllama.")] = v
-		}
-	}
-
-	kv["mllama.attention.cross_attention_layers"] = m.TextModel.CrossAttentionLayers
-
-	kv["mllama.vision.block_count"] = m.VisionModel.NumHiddenLayers
-	kv["mllama.vision.global.block_count"] = m.VisionModel.NumGlobalLayers
-	kv["mllama.vision.intermediate_layers_indices"] = m.VisionModel.IntermediateLayersIndices
-
-	kv["mllama.vision.embedding_length"] = m.VisionModel.HiddenSize
-	kv["mllama.vision.feed_forward_length"] = m.VisionModel.IntermediateSize
-
-	kv["mllama.vision.attention.head_count"] = m.VisionModel.AttentionHeads
-	kv["mllama.vision.attention.layer_norm_epsilon"] = m.VisionModel.NormEpsilon
-
-	kv["mllama.vision.image_size"] = m.VisionModel.ImageSize
-	kv["mllama.vision.patch_size"] = m.VisionModel.PatchSize
-	kv["mllama.vision.max_num_tiles"] = m.VisionModel.MaxNumTiles
-	kv["mllama.vision.num_channels"] = m.VisionModel.NumChannels
-
-	return kv
-}
-
-func (m *mllamaModel) Replacements() []string {
-	return append(
-		m.TextModel.Replacements(),
-		"language_model.", "",
-		"gate_attn", "attn_gate",
-		"gate_ffn", "ffn_gate",
-		"cross_attn.", "cross_attn_",
-		"vision_model", "v",
-		"class_embedding", "class_embd",
-		"patch_embedding", "patch_embd",
-		"gated_positional_embedding.tile_embedding", "tile_position_embd",
-		"gated_positional_embedding.embedding", "position_embd.weight",
-		"gated_positional_embedding", "position_embd",
-		"embedding.weight", "weight",
-		"pre_tile_positional_embedding", "pre_tile_position_embd",
-		"post_tile_positional_embedding", "post_tile_position_embd",
-		"layernorm_pre", "pre_ln",
-		"layernorm_post", "post_ln",
-		"global_transformer.layers", "global.blk",
-		"transformer.layers", "blk",
-		"mlp.fc1", "ffn_up",
-		"mlp.fc2", "ffn_down",
-		"multi_modal_projector", "mm.0",
-	)
-}
-
-func (m *mllamaModel) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
-	var text []Tensor
-	for _, t := range ts {
-		if t.Name() == "v.position_embd.gate" {
-			for _, name := range []string{"v.position_embd.gate", "v.tile_position_embd.gate"} {
-				tt := t.Clone()
-				tt.SetRepacker(m.repack(name))
-				out = append(out, &ggml.Tensor{
-					Name:     name,
-					Kind:     t.Kind(),
-					Shape:    t.Shape(),
-					WriterTo: tt,
-				})
-			}
-		} else if t.Name() == "v.pre_tile_position_embd.gate" || t.Name() == "v.post_tile_position_embd.gate" {
-			t.SetRepacker(m.repack(t.Name()))
-			out = append(out, &ggml.Tensor{
-				Name:     t.Name(),
-				Kind:     t.Kind(),
-				Shape:    t.Shape(),
-				WriterTo: t,
-			})
-		} else if strings.HasPrefix(t.Name(), "v.") || strings.HasPrefix(t.Name(), "mm.") {
-			out = append(out, &ggml.Tensor{
-				Name:     t.Name(),
-				Kind:     t.Kind(),
-				Shape:    t.Shape(),
-				WriterTo: t,
-			})
-		} else {
-			text = append(text, t)
-		}
-	}
-
-	return append(out, m.TextModel.Tensors(text)...)
-}
-
-func (m *mllamaModel) repack(name string) Repacker {
-	return func(_ string, data []float32, shape []uint64) (_ []float32, err error) {
-		dims := make([]int, len(shape))
-		for i, dim := range shape {
-			dims[i] = int(dim)
-		}
-
-		var t tensor.Tensor = tensor.New(tensor.WithShape(dims...), tensor.WithBacking(data))
-
-		t, err = tensor.Tanh(t)
-		if err != nil {
-			return nil, err
-		}
-
-		if name == "v.position_embd.gate" {
-			t, err = tensor.Sub(float32(1), t)
-			if err != nil {
-				return nil, err
-			}
-		}
-
-		t = tensor.Materialize(t)
-		// flatten tensor so it can be return as a vector
-		if err := t.Reshape(t.Shape().TotalSize()); err != nil {
-			return nil, err
-		}
-
-		return native.VectorF32(t.(*tensor.Dense))
-	}
-}
--- a/convert/convert_phi3.go
+++ b/convert/convert_phi3.go
@@ -68,19 +68,19 @@ func (p *phi3Model) KV(t *Tokenizer) ggml.KV {
 	return kv
 }

-func (p *phi3Model) Tensors(ts []Tensor) []*ggml.Tensor {
+func (p *phi3Model) Tensors(ts []Tensor) []ggml.Tensor {
 	var addRopeFactors sync.Once

-	out := make([]*ggml.Tensor, 0, len(ts)+2)
+	out := make([]ggml.Tensor, 0, len(ts)+2)
 	for _, t := range ts {
 		if strings.HasPrefix(t.Name(), "blk.0.") {
 			addRopeFactors.Do(func() {
-				out = append(out, &ggml.Tensor{
+				out = append(out, ggml.Tensor{
 					Name:     "rope_factors_long.weight",
 					Kind:     0,
 					Shape:    []uint64{uint64(len(p.RopeScaling.LongFactor))},
 					WriterTo: p.RopeScaling.LongFactor,
-				}, &ggml.Tensor{
+				}, ggml.Tensor{
 					Name:     "rope_factors_short.weight",
 					Kind:     0,
 					Shape:    []uint64{uint64(len(p.RopeScaling.ShortFactor))},
@@ -89,7 +89,7 @@ func (p *phi3Model) Tensors(ts []Tensor) []*ggml.Tensor {
 			})
 		}

-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
@@ -118,5 +118,6 @@ func (p *phi3Model) Replacements() []string {
 type ropeFactor []float32

 func (r ropeFactor) WriteTo(w io.Writer) (int64, error) {
-	return 0, binary.Write(w, binary.LittleEndian, r)
+	err := binary.Write(w, binary.LittleEndian, r)
+	return 0, err
 }
--- a/convert/convert_qwen2.go
+++ b/convert/convert_qwen2.go
@@ -15,7 +15,6 @@ type qwen2Model struct {
 		Type                          string     `json:"type"`
 		Factor                        ropeFactor `json:"factor"`
 		OriginalMaxPositionEmbeddings uint32     `json:"original_max_position_embeddings"`
-		MropeSection                  []int32    `json:"mrope_section"`
 	} `json:"rope_scaling"`
 	RMSNormEPS float32 `json:"rms_norm_eps"`
 }
@@ -40,18 +39,16 @@ func (q *qwen2Model) KV(t *Tokenizer) ggml.KV {
 	case "yarn":
 		kv["qwen2.rope.scaling.type"] = q.RopeScaling.Type
 		kv["qwen2.rope.scaling.factor"] = q.RopeScaling.Factor
-	case "mrope", "default":
-		kv["qwen2.rope.mrope_section"] = q.RopeScaling.MropeSection
 	default:
 		panic("unknown rope scaling type")
 	}
 	return kv
 }

-func (q *qwen2Model) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
+func (q *qwen2Model) Tensors(ts []Tensor) []ggml.Tensor {
+	var out []ggml.Tensor
 	for _, t := range ts {
-		out = append(out, &ggml.Tensor{
+		out = append(out, ggml.Tensor{
 			Name:     t.Name(),
 			Kind:     t.Kind(),
 			Shape:    t.Shape(),
--- a/convert/convert_qwen25vl.go
+++ b/convert/convert_qwen25vl.go
@@ -1,102 +0,0 @@
-package convert
-
-import (
-	"cmp"
-	"slices"
-	"strings"
-
-	"github.com/ollama/ollama/fs/ggml"
-)
-
-type qwen25VLModel struct {
-	qwen2Model
-
-	VisionModel struct {
-		Depth               uint32  `json:"depth"`
-		HiddenSize          uint32  `json:"hidden_size"`
-		NumHeads            uint32  `json:"num_heads"`
-		InChannels          uint32  `json:"in_chans"`
-		PatchSize           uint32  `json:"patch_size"`
-		SpatialMergeSize    uint32  `json:"spatial_merge_size"`
-		SpatialPatchSize    uint32  `json:"spatial_patch_size"`
-		WindowSize          uint32  `json:"window_size"`
-		RMSNormEps          float32 `json:"layer_norm_epsilon"`
-		RopeTheta           float32 `json:"rope_theta"`
-		FullAttentionBlocks []int32 `json:"fullatt_block_indexes"`
-		TemporalPatchSize   uint32  `json:"temporal_patch_size"`
-	} `json:"vision_config"`
-}
-
-var _ ModelConverter = (*qwen25VLModel)(nil)
-
-func (q *qwen25VLModel) KV(t *Tokenizer) ggml.KV {
-	kv := q.ModelParameters.KV(t)
-	kv["general.architecture"] = "qwen25vl"
-
-	for k, v := range q.qwen2Model.KV(t) {
-		if strings.HasPrefix(k, "qwen2.") {
-			kv[strings.Replace(k, "qwen2.", "qwen25vl.", 1)] = v
-		}
-	}
-
-	if q.VisionModel.FullAttentionBlocks == nil {
-		kv["qwen25vl.vision.fullatt_block_indexes"] = []int32{7, 15, 23, 31}
-	}
-
-	kv["qwen25vl.vision.block_count"] = cmp.Or(q.VisionModel.Depth, 32)
-	kv["qwen25vl.vision.embedding_length"] = q.VisionModel.HiddenSize
-	kv["qwen25vl.vision.attention.head_count"] = cmp.Or(q.VisionModel.NumHeads, 16)
-	kv["qwen25vl.vision.num_channels"] = q.VisionModel.InChannels
-	kv["qwen25vl.vision.patch_size"] = cmp.Or(q.VisionModel.PatchSize, 14)
-	kv["qwen25vl.vision.spatial_merge_size"] = cmp.Or(q.VisionModel.SpatialMergeSize, 2)
-	kv["qwen25vl.vision.spatial_patch_size"] = q.VisionModel.SpatialPatchSize
-	kv["qwen25vl.vision.window_size"] = cmp.Or(q.VisionModel.WindowSize, 112)
-	kv["qwen25vl.vision.attention.layer_norm_epsilon"] = cmp.Or(q.VisionModel.RMSNormEps, 1e-6)
-	kv["qwen25vl.vision.rope.freq_base"] = cmp.Or(q.VisionModel.RopeTheta, 1e4)
-	kv["qwen25vl.vision.fullatt_block_indexes"] = q.VisionModel.FullAttentionBlocks
-	kv["qwen25vl.vision.temporal_patch_size"] = cmp.Or(q.VisionModel.TemporalPatchSize, 2)
-
-	return kv
-}
-
-func (q *qwen25VLModel) Tensors(ts []Tensor) []*ggml.Tensor {
-	var out []*ggml.Tensor
-
-	for _, t := range ts {
-		if strings.Contains(t.Name(), "patch_embed.proj") {
-			for t := range splitDim(t, 2,
-				strings.NewReplacer("patch_embed.proj", "patch_embd_0"),
-				strings.NewReplacer("patch_embed.proj", "patch_embd_1"),
-			) {
-				t.Shape = slices.DeleteFunc(t.Shape, func(i uint64) bool { return i == 1 })
-				out = append(out, t)
-			}
-		} else if strings.Contains(t.Name(), "attn.qkv") {
-			out = append(out, slices.Collect(splitDim(t, 0,
-				strings.NewReplacer("attn.qkv", "attn_q"),
-				strings.NewReplacer("attn.qkv", "attn_k"),
-				strings.NewReplacer("attn.qkv", "attn_v"),
-			))...)
-		} else {
-			out = append(out, &ggml.Tensor{
-				Name:     t.Name(),
-				Kind:     t.Kind(),
-				Shape:    t.Shape(),
-				WriterTo: t,
-			})
-		}
-	}
-
-	return out
-}
-
-func (p *qwen25VLModel) Replacements() []string {
-	return append(
-		p.qwen2Model.Replacements(),
-		"visual", "v",
-		"blocks", "blk",
-		"attn.proj", "attn_out",
-		"norm1", "ln1",
-		"norm2", "ln2",
-	)
-}
--- a/convert/convert_test.go
+++ b/convert/convert_test.go
@@ -11,6 +11,7 @@ import (
 	"io"
 	"io/fs"
 	"log/slog"
+	"math"
 	"os"
 	"path/filepath"
 	"slices"
@@ -47,7 +48,7 @@ func convertFull(t *testing.T, fsys fs.FS) (*os.File, ggml.KV, ggml.Tensors) {
 	}
 	t.Cleanup(func() { r.Close() })

-	m, _, err := ggml.Decode(r, -1)
+	m, _, err := ggml.Decode(r, math.MaxInt)
 	if err != nil {
 		t.Fatal(err)
 	}
@@ -130,7 +131,6 @@ func TestConvertModel(t *testing.T) {
 			if err != nil {
 				t.Fatal(err)
 			}
-			defer expectFile.Close()

 			var expect map[string]string
 			if err := json.NewDecoder(expectFile).Decode(&expect); err != nil {
@@ -332,7 +332,7 @@ func TestConvertAdapter(t *testing.T) {
 			}
 			defer r.Close()

-			m, _, err := ggml.Decode(r, -1)
+			m, _, err := ggml.Decode(r, math.MaxInt)
 			if err != nil {
 				t.Fatal(err)
 			}
--- a/convert/fs.go
+++ b/convert/fs.go
@@ -0,0 +1,58 @@
+package convert
+
+import (
+	"archive/zip"
+	"errors"
+	"io"
+	"io/fs"
+	"os"
+	"path/filepath"
+)
+
+type ZipReader struct {
+	r *zip.Reader
+	p string
+
+	// limit is the maximum size of a file that can be read directly
+	// from the zip archive. Files larger than this size will be extracted
+	limit int64
+}
+
+func NewZipReader(r *zip.Reader, p string, limit int64) fs.FS {
+	return &ZipReader{r, p, limit}
+}
+
+func (z *ZipReader) Open(name string) (fs.File, error) {
+	r, err := z.r.Open(name)
+	if err != nil {
+		return nil, err
+	}
+	defer r.Close()
+
+	if fi, err := r.Stat(); err != nil {
+		return nil, err
+	} else if fi.Size() < z.limit {
+		return r, nil
+	}
+
+	if !filepath.IsLocal(name) {
+		return nil, zip.ErrInsecurePath
+	}
+
+	n := filepath.Join(z.p, name)
+	if _, err := os.Stat(n); errors.Is(err, os.ErrNotExist) {
+		w, err := os.Create(n)
+		if err != nil {
+			return nil, err
+		}
+		defer w.Close()
+
+		if _, err := io.Copy(w, r); err != nil {
+			return nil, err
+		}
+	} else if err != nil {
+		return nil, err
+	}
+
+	return os.Open(n)
+}
--- a/convert/reader.go
+++ b/convert/reader.go
@@ -11,15 +11,14 @@ type Tensor interface {
 	Name() string
 	Shape() []uint64
 	Kind() uint32
-	SetRepacker(Repacker)
+	SetRepacker(repacker)
 	WriteTo(io.Writer) (int64, error)
-	Clone() Tensor
 }

 type tensorBase struct {
-	name     string
-	shape    []uint64
-	repacker Repacker
+	name  string
+	shape []uint64
+	repacker
 }

 func (t tensorBase) Name() string {
@@ -37,11 +36,7 @@ const (

 func (t tensorBase) Kind() uint32 {
 	if strings.HasSuffix(t.name, ".ffn_gate_inp.weight") ||
-		t.name == "token_types.weight" ||
-		t.name == "v.positional_embedding_vlm" ||
-		t.name == "v.tile_position_embd.weight" ||
-		t.name == "v.pre_tile_position_embd.weight" ||
-		t.name == "v.post_tile_position_embd.weight" {
+		t.name == "token_types.weight" {
 		// these tensors are always F32
 		return 0
 	}
@@ -56,18 +51,21 @@ func (t tensorBase) Kind() uint32 {
 	}
 }

-func (t *tensorBase) SetRepacker(fn Repacker) {
+func (t *tensorBase) SetRepacker(fn repacker) {
 	t.repacker = fn
 }

-type Repacker func(string, []float32, []uint64) ([]float32, error)
+type repacker func(string, []float32, []uint64) ([]float32, error)

 func parseTensors(fsys fs.FS, replacer *strings.Replacer) ([]Tensor, error) {
 	patterns := []struct {
 		Pattern string
 		Func    func(fs.FS, *strings.Replacer, ...string) ([]Tensor, error)
 	}{
-		{"*.safetensors", parseSafetensors},
+		{"model-*-of-*.safetensors", parseSafetensors},
+		{"model.safetensors", parseSafetensors},
+		{"adapters.safetensors", parseSafetensors},
+		{"adapter_model.safetensors", parseSafetensors},
 		{"pytorch_model-*-of-*.bin", parseTorch},
 		{"pytorch_model.bin", parseTorch},
 		{"consolidated.*.pth", parseTorch},
--- a/convert/reader_safetensors.go
+++ b/convert/reader_safetensors.go
@@ -94,21 +94,6 @@ type safetensor struct {
 	*tensorBase
 }

-func (st safetensor) Clone() Tensor {
-	return &safetensor{
-		fs:     st.fs,
-		path:   st.path,
-		dtype:  st.dtype,
-		offset: st.offset,
-		size:   st.size,
-		tensorBase: &tensorBase{
-			name:     st.name,
-			repacker: st.repacker,
-			shape:    slices.Clone(st.shape),
-		},
-	}
-}
-
 func (st safetensor) WriteTo(w io.Writer) (int64, error) {
 	f, err := st.fs.Open(st.path)
 	if err != nil {
--- a/convert/reader_torch.go
+++ b/convert/reader_torch.go
@@ -43,17 +43,6 @@ type torch struct {
 	*tensorBase
 }

-func (t torch) Clone() Tensor {
-	return torch{
-		storage: t.storage,
-		tensorBase: &tensorBase{
-			name:     t.name,
-			shape:    t.shape,
-			repacker: t.repacker,
-		},
-	}
-}
-
 func (pt torch) WriteTo(w io.Writer) (int64, error) {
 	return 0, nil
 }
--- a/convert/sentencepiece/sentencepiece_model.pb.go
+++ b/convert/sentencepiece/sentencepiece_model.pb.go
@@ -1360,7 +1360,7 @@ func file_sentencepiece_model_proto_rawDescGZIP() []byte {

 var file_sentencepiece_model_proto_enumTypes = make([]protoimpl.EnumInfo, 2)
 var file_sentencepiece_model_proto_msgTypes = make([]protoimpl.MessageInfo, 6)
-var file_sentencepiece_model_proto_goTypes = []any{
+var file_sentencepiece_model_proto_goTypes = []interface{}{
 	(TrainerSpec_ModelType)(0),         // 0: sentencepiece.TrainerSpec.ModelType
 	(ModelProto_SentencePiece_Type)(0), // 1: sentencepiece.ModelProto.SentencePiece.Type
 	(*TrainerSpec)(nil),                // 2: sentencepiece.TrainerSpec
@@ -1392,7 +1392,7 @@ func file_sentencepiece_model_proto_init() {
 		return
 	}
 	if !protoimpl.UnsafeEnabled {
-		file_sentencepiece_model_proto_msgTypes[0].Exporter = func(v any, i int) any {
+		file_sentencepiece_model_proto_msgTypes[0].Exporter = func(v interface{}, i int) interface{} {
 			switch v := v.(*TrainerSpec); i {
 			case 0:
 				return &v.state
@@ -1406,7 +1406,7 @@ func file_sentencepiece_model_proto_init() {
 				return nil
 			}
 		}
-		file_sentencepiece_model_proto_msgTypes[1].Exporter = func(v any, i int) any {
+		file_sentencepiece_model_proto_msgTypes[1].Exporter = func(v interface{}, i int) interface{} {
 			switch v := v.(*NormalizerSpec); i {
 			case 0:
 				return &v.state
@@ -1420,7 +1420,7 @@ func file_sentencepiece_model_proto_init() {
 				return nil
 			}
 		}
-		file_sentencepiece_model_proto_msgTypes[2].Exporter = func(v any, i int) any {
+		file_sentencepiece_model_proto_msgTypes[2].Exporter = func(v interface{}, i int) interface{} {
 			switch v := v.(*SelfTestData); i {
 			case 0:
 				return &v.state
@@ -1434,7 +1434,7 @@ func file_sentencepiece_model_proto_init() {
 				return nil
 			}
 		}
-		file_sentencepiece_model_proto_msgTypes[3].Exporter = func(v any, i int) any {
+		file_sentencepiece_model_proto_msgTypes[3].Exporter = func(v interface{}, i int) interface{} {
 			switch v := v.(*ModelProto); i {
 			case 0:
 				return &v.state
@@ -1448,7 +1448,7 @@ func file_sentencepiece_model_proto_init() {
 				return nil
 			}
 		}
-		file_sentencepiece_model_proto_msgTypes[4].Exporter = func(v any, i int) any {
+		file_sentencepiece_model_proto_msgTypes[4].Exporter = func(v interface{}, i int) interface{} {
 			switch v := v.(*SelfTestData_Sample); i {
 			case 0:
 				return &v.state
@@ -1460,7 +1460,7 @@ func file_sentencepiece_model_proto_init() {
 				return nil
 			}
 		}
-		file_sentencepiece_model_proto_msgTypes[5].Exporter = func(v any, i int) any {
+		file_sentencepiece_model_proto_msgTypes[5].Exporter = func(v interface{}, i int) interface{} {
 			switch v := v.(*ModelProto_SentencePiece); i {
 			case 0:
 				return &v.state
--- a/convert/tensor.go
+++ b/convert/tensor.go
@@ -1,56 +0,0 @@
-package convert
-
-import (
-	"iter"
-	"slices"
-	"strings"
-
-	"github.com/ollama/ollama/fs/ggml"
-	"github.com/pdevine/tensor"
-	"github.com/pdevine/tensor/native"
-)
-
-// splitDim splits a tensor along a specified dimension into multiple tensors. The dimension
-// is split evenly based on the number of replacers provided.
-func splitDim(t Tensor, dim int, replacers ...*strings.Replacer) iter.Seq[*ggml.Tensor] {
-	return func(yield func(*ggml.Tensor) bool) {
-		for i, replacer := range replacers {
-			shape := slices.Clone(t.Shape())
-			shape[dim] = shape[dim] / uint64(len(replacers))
-
-			slice := slices.Repeat([]tensor.Slice{nil}, len(shape))
-			slice[dim] = tensor.S(i*int(shape[dim]), (i+1)*int(shape[dim]))
-
-			tt := t.Clone()
-			tt.SetRepacker(func(_ string, data []float32, shape []uint64) ([]float32, error) {
-				dims := make([]int, len(shape))
-				for i := range shape {
-					dims[i] = int(shape[i])
-				}
-
-				var t tensor.Tensor = tensor.New(tensor.WithShape(dims...), tensor.WithBacking(data))
-				t, err := t.Slice(slice...)
-				if err != nil {
-					return nil, err
-				}
-
-				t = tensor.Materialize(t)
-				// flatten tensor so it can be written as a vector
-				if err := t.Reshape(t.Shape().TotalSize()); err != nil {
-					return nil, err
-				}
-
-				return native.VectorF32(t.(*tensor.Dense))
-			})
-
-			if !yield(&ggml.Tensor{
-				Name:     replacer.Replace(t.Name()),
-				Kind:     t.Kind(),
-				Shape:    shape,
-				WriterTo: tt,
-			}) {
-				break
-			}
-		}
-	}
-}
--- a/discover/cpu_common.go
+++ b/discover/cpu_common.go
@@ -12,7 +12,7 @@ func IsNUMA() bool {
 		// numa support in llama.cpp is linux only
 		return false
 	}
-	ids := map[string]any{}
+	ids := map[string]interface{}{}
 	packageIds, _ := filepath.Glob("/sys/devices/system/cpu/cpu*/topology/physical_package_id")
 	for _, packageId := range packageIds {
 		id, err := os.ReadFile(packageId)
--- a/discover/gpu.go
+++ b/discover/gpu.go
@@ -670,7 +670,7 @@ func loadOneapiMgmt(oneapiLibPaths []string) (int, *C.oneapi_handle_t, string, e
 }

 func getVerboseState() C.uint16_t {
-	if envconfig.LogLevel() < slog.LevelInfo {
+	if envconfig.Debug() {
 		return C.uint16_t(1)
 	}
 	return C.uint16_t(0)
--- a/discover/gpu_info.h
+++ b/discover/gpu_info.h
@@ -27,14 +27,12 @@

 #endif

-#ifndef LOG
 #define LOG(verbose, ...) \
  do { \
    if (verbose) { \
      fprintf(stderr, __VA_ARGS__); \
    } \
  } while (0)
-#endif

 #ifdef __cplusplus
 extern "C" {
--- a/discover/gpu_info_cudart.c
+++ b/discover/gpu_info_cudart.c
@@ -1,7 +1,6 @@
 #ifndef __APPLE__  // TODO - maybe consider nvidia support on intel macs?

 #include <string.h>
-#include <inttypes.h>
 #include "gpu_info_cudart.h"

 void cudart_init(char *cudart_lib_path, cudart_init_resp_t *resp) {
@@ -59,7 +58,7 @@ void cudart_init(char *cudart_lib_path, cudart_init_resp_t *resp) {
    LOG(resp->ch.verbose, "cudaSetDevice err: %d\n", ret);
    UNLOAD_LIBRARY(resp->ch.handle);
    resp->ch.handle = NULL;
-    if (ret == CUDART_ERROR_INSUFFICIENT_DRIVER) {
+    if (ret == CUDA_ERROR_INSUFFICIENT_DRIVER) {
      resp->err = strdup("your nvidia driver is too old or missing.  If you have a CUDA GPU please upgrade to run ollama");
      return;
    }
@@ -169,9 +168,9 @@ void cudart_bootstrap(cudart_handle_t h, int i, mem_info_t *resp) {
  resp->free = memInfo.free;
  resp->used = memInfo.used;

-  LOG(h.verbose, "[%s] CUDA totalMem %" PRId64 "\n", resp->gpu_id, resp->total);
-  LOG(h.verbose, "[%s] CUDA freeMem %" PRId64 "\n", resp->gpu_id, resp->free);
-  LOG(h.verbose, "[%s] CUDA usedMem %" PRId64 "\n", resp->gpu_id, resp->used);
+  LOG(h.verbose, "[%s] CUDA totalMem %lu\n", resp->gpu_id, resp->total);
+  LOG(h.verbose, "[%s] CUDA freeMem %lu\n", resp->gpu_id, resp->free);
+  LOG(h.verbose, "[%s] CUDA usedMem %lu\n", resp->gpu_id, resp->used);
  LOG(h.verbose, "[%s] Compute Capability %d.%d\n", resp->gpu_id, resp->major, resp->minor);
 }

@@ -181,4 +180,4 @@ void cudart_release(cudart_handle_t h) {
  h.handle = NULL;
 }

-#endif  // __APPLE__
+#endif  // __APPLE__
--- a/discover/gpu_info_nvcuda.c
+++ b/discover/gpu_info_nvcuda.c
@@ -1,7 +1,6 @@
 #ifndef __APPLE__  // TODO - maybe consider nvidia support on intel macs?

 #include <string.h>
-#include <inttypes.h>
 #include "gpu_info_nvcuda.h"

 void nvcuda_init(char *nvcuda_lib_path, nvcuda_init_resp_t *resp) {
@@ -194,8 +193,8 @@ void nvcuda_bootstrap(nvcuda_handle_t h, int i, mem_info_t *resp) {
  resp->total = memInfo.total;
  resp->free = memInfo.free;

-  LOG(h.verbose, "[%s] CUDA totalMem %" PRId64 "mb\n", resp->gpu_id, resp->total / 1024 / 1024);
-  LOG(h.verbose, "[%s] CUDA freeMem %" PRId64 "mb\n", resp->gpu_id, resp->free / 1024 / 1024);
+  LOG(h.verbose, "[%s] CUDA totalMem %lu mb\n", resp->gpu_id, resp->total / 1024 / 1024);
+  LOG(h.verbose, "[%s] CUDA freeMem %lu mb\n", resp->gpu_id, resp->free / 1024 / 1024);
  LOG(h.verbose, "[%s] Compute Capability %d.%d\n", resp->gpu_id, resp->major, resp->minor);

  
@@ -248,4 +247,4 @@ void nvcuda_release(nvcuda_handle_t h) {
  h.handle = NULL;
 }

-#endif  // __APPLE__
+#endif  // __APPLE__
--- a/discover/gpu_linux.go
+++ b/discover/gpu_linux.go
@@ -111,7 +111,6 @@ func GetCPUDetails() ([]CPU, error) {
 	if err != nil {
 		return nil, err
 	}
-	defer file.Close()
 	return linuxCPUDetails(file)
 }

@@ -169,11 +168,13 @@ func linuxCPUDetails(file io.Reader) ([]CPU, error) {
 	for id, s := range socketByID {
 		s.CoreCount = len(coreBySocket[id])
 		s.ThreadCount = 0
+		for _, tc := range threadsByCoreBySocket[id] {
+			s.ThreadCount += tc
+		}

 		// This only works if HT is enabled, consider a more reliable model, maybe cache size comparisons?
 		efficiencyCoreCount := 0
 		for _, threads := range threadsByCoreBySocket[id] {
-			s.ThreadCount += threads
 			if threads == 1 {
 				efficiencyCoreCount++
 			}
--- a/docs/api.md
+++ b/docs/api.md
@@ -19,7 +19,7 @@

 ### Model names

-Model names follow a `model:tag` format, where `model` can have an optional namespace such as `example/model`. Some examples are `orca-mini:3b-q8_0` and `llama3:70b`. The tag is optional and, if not provided, will default to `latest`. The tag is used to identify a specific version.
+Model names follow a `model:tag` format, where `model` can have an optional namespace such as `example/model`. Some examples are `orca-mini:3b-q4_1` and `llama3:70b`. The tag is optional and, if not provided, will default to `latest`. The tag is used to identify a specific version.

 ### Durations

@@ -173,7 +173,7 @@ curl http://localhost:11434/api/generate -d '{

 ##### Response

-```json5
+```json
 {
  "model": "codellama:code",
  "created_at": "2024-07-22T20:47:51.147561Z",
@@ -394,6 +394,9 @@ curl http://localhost:11434/api/generate -d '{
    "repeat_penalty": 1.2,
    "presence_penalty": 1.5,
    "frequency_penalty": 1.0,
+    "mirostat": 1,
+    "mirostat_tau": 0.8,
+    "mirostat_eta": 0.6,
    "penalize_newline": true,
    "stop": ["\n", "user:"],
    "numa": false,
@@ -401,7 +404,10 @@ curl http://localhost:11434/api/generate -d '{
    "num_batch": 2,
    "num_gpu": 1,
    "main_gpu": 0,
+    "low_vram": false,
+    "vocab_only": false,
    "use_mmap": true,
+    "use_mlock": false,
    "num_thread": 8
  }
 }'
@@ -952,8 +958,19 @@ If you are creating a model from a safetensors directory or from a GGUF file, yo

 | Type | Recommended |
 | --- | :-: |
+| q2_K | |
+| q3_K_L | |
+| q3_K_M | |
+| q3_K_S | |
+| q4_0 | |
+| q4_1 | |
 | q4_K_M | * |
 | q4_K_S | |
+| q5_0 | |
+| q5_1 | |
+| q5_K_M | |
+| q5_K_S | |
+| q6_K | |
 | q8_0 | * |

 ### Examples
@@ -998,8 +1015,8 @@ Quantize a non-quantized model.

 ```shell
 curl http://localhost:11434/api/create -d '{
-  "model": "llama3.2:quantized",
-  "from": "llama3.2:3b-instruct-fp16",
+  "model": "llama3.1:quantized",
+  "from": "llama3.1:8b-instruct-fp16",
  "quantize": "q4_K_M"
 }'
 ```
@@ -1009,14 +1026,12 @@ curl http://localhost:11434/api/create -d '{
 A stream of JSON objects is returned:

 ```json
-{"status":"quantizing F16 model to Q4_K_M","digest":"0","total":6433687776,"completed":12302}
-{"status":"quantizing F16 model to Q4_K_M","digest":"0","total":6433687776,"completed":6433687552}
-{"status":"verifying conversion"}
-{"status":"creating new layer sha256:fb7f4f211b89c6c4928ff4ddb73db9f9c0cfca3e000c3e40d6cf27ddc6ca72eb"}
-{"status":"using existing layer sha256:966de95ca8a62200913e3f8bfbf84c8494536f1b94b49166851e76644e966396"}
-{"status":"using existing layer sha256:fcc5a6bec9daf9b561a68827b67ab6088e1dba9d1fa2a50d7bbcc8384e0a265d"}
-{"status":"using existing layer sha256:a70ff7e570d97baaf4e62ac6e6ad9975e04caa6d900d3742d37698494479e0cd"}
+{"status":"quantizing F16 model to Q4_K_M"}
+{"status":"creating new layer sha256:667b0c1932bc6ffc593ed1d03f895bf2dc8dc6df21db3042284a6f4416b06a29"}
+{"status":"using existing layer sha256:11ce4ee3e170f6adebac9a991c22e22ab3f8530e154ee669954c4bc73061c258"}
+{"status":"using existing layer sha256:0ba8f0e314b4264dfd19df045cde9d4c394a52474bf92ed6a3de22a4ca31a177"}
 {"status":"using existing layer sha256:56bb8bd477a519ffa694fc449c2413c6f0e1d3b1c88fa7e3c9d88d3ae49d4dcb"}
+{"status":"creating new layer sha256:455f34728c9b5dd3376378bfb809ee166c145b0b4c1f1a6feca069055066ef9a"}
 {"status":"writing manifest"}
 {"status":"success"}
 ```
@@ -1154,37 +1169,29 @@ A single JSON object will be returned.
 {
  "models": [
    {
-      "name": "deepseek-r1:latest",
-      "model": "deepseek-r1:latest",
-      "modified_at": "2025-05-10T08:06:48.639712648-07:00",
-      "size": 4683075271,
-      "digest": "0a8c266910232fd3291e71e5ba1e058cc5af9d411192cf88b6d30e92b6e73163",
+      "name": "codellama:13b",
+      "modified_at": "2023-11-04T14:56:49.277302595-07:00",
+      "size": 7365960935,
+      "digest": "9f438cb9cd581fc025612d27f7c1a6669ff83a8bb0ed86c94fcf4c5440555697",
      "details": {
-        "parent_model": "",
        "format": "gguf",
-        "family": "qwen2",
-        "families": [
-          "qwen2"
-        ],
-        "parameter_size": "7.6B",
-        "quantization_level": "Q4_K_M"
+        "family": "llama",
+        "families": null,
+        "parameter_size": "13B",
+        "quantization_level": "Q4_0"
      }
    },
    {
-      "name": "llama3.2:latest",
-      "model": "llama3.2:latest",
-      "modified_at": "2025-05-04T17:37:44.706015396-07:00",
-      "size": 2019393189,
-      "digest": "a80c4f17acd55265feec403c7aef86be0c25983ab279d83f3bcd3abbcb5b8b72",
+      "name": "llama3:latest",
+      "modified_at": "2023-12-07T09:32:18.757212583-08:00",
+      "size": 3825819519,
+      "digest": "fe938a131f40e6f6d40083c9f0f430a515233eb2edaa6d72eb85c50d64f2300e",
      "details": {
-        "parent_model": "",
        "format": "gguf",
        "family": "llama",
-        "families": [
-          "llama"
-        ],
-        "parameter_size": "3.2B",
-        "quantization_level": "Q4_K_M"
+        "families": null,
+        "parameter_size": "7B",
+        "quantization_level": "Q4_0"
      }
    }
  ]
@@ -1210,13 +1217,13 @@ Show information about a model including details, modelfile, template, parameter

 ```shell
 curl http://localhost:11434/api/show -d '{
-  "model": "llava"
+  "model": "llama3.2"
 }'
 ```

 #### Response

-```json5
+```json
 {
  "modelfile": "# Modelfile generated by \"ollama show\"\n# To build a new Modelfile based on this one, replace the FROM line with:\n# FROM llava:latest\n\nFROM /Users/matt/.ollama/models/blobs/sha256:200765e1283640ffbd013184bf496e261032fa75b99498a9613be4e94d63ad52\nTEMPLATE \"\"\"{{ .System }}\nUSER: {{ .Prompt }}\nASSISTANT: \"\"\"\nPARAMETER num_ctx 4096\nPARAMETER stop \"\u003c/s\u003e\"\nPARAMETER stop \"USER:\"\nPARAMETER stop \"ASSISTANT:\"",
  "parameters": "num_keep                       24\nstop                           \"<|start_header_id|>\"\nstop                           \"<|end_header_id|>\"\nstop                           \"<|eot_id|>\"",
@@ -1253,11 +1260,7 @@ curl http://localhost:11434/api/show -d '{
    "tokenizer.ggml.pre": "llama-bpe",
    "tokenizer.ggml.token_type": [],        // populates if `verbose=true`
    "tokenizer.ggml.tokens": []             // populates if `verbose=true`
-  },
-  "capabilities": [
-    "completion",
-    "vision"
-  ],
+  }
 }
 ```

--- a/docs/faq.md
+++ b/docs/faq.md
@@ -20,13 +20,7 @@ Please refer to the [GPU docs](./gpu.md).

 ## How can I specify the context window size?

-By default, Ollama uses a context window size of 4096 tokens. 
-
-This can be overridden with the `OLLAMA_CONTEXT_LENGTH` environment variable. For example, to set the default context window to 8K, use: 
-
-```shell
-OLLAMA_CONTEXT_LENGTH=8192 ollama serve
-```
+By default, Ollama uses a context window size of 2048 tokens. This can be overridden with the `OLLAMA_CONTEXT_LENGTH` environment variable. For example, to set the default context length to 8K, use: `OLLAMA_CONTEXT_LENGTH=8192 ollama serve`.

 To change this when using `ollama run`, use `/set parameter`:

--- a/docs/modelfile.md
+++ b/docs/modelfile.md
@@ -150,6 +150,9 @@ PARAMETER <parameter> <parametervalue>

 | Parameter      | Description                                                                                                                                                                                                                                             | Value Type | Example Usage        |
 | -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | -------------------- |
+| mirostat       | Enable Mirostat sampling for controlling perplexity. (default: 0, 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0)                                                                                                                                         | int        | mirostat 0           |
+| mirostat_eta   | Influences how quickly the algorithm responds to feedback from the generated text. A lower learning rate will result in slower adjustments, while a higher learning rate will make the algorithm more responsive. (Default: 0.1)                        | float      | mirostat_eta 0.1     |
+| mirostat_tau   | Controls the balance between coherence and diversity of the output. A lower value will result in more focused and coherent text. (Default: 5.0)                                                                                                         | float      | mirostat_tau 5.0     |
 | num_ctx        | Sets the size of the context window used to generate the next token. (Default: 2048)                                                                                                                                                                    | int        | num_ctx 4096         |
 | repeat_last_n  | Sets how far back for the model to look back to prevent repetition. (Default: 64, 0 = disabled, -1 = num_ctx)                                                                                                                                           | int        | repeat_last_n 64     |
 | repeat_penalty | Sets how strongly to penalize repetitions. A higher value (e.g., 1.5) will penalize repetitions more strongly, while a lower value (e.g., 0.9) will be more lenient. (Default: 1.1)                                                                     | float      | repeat_penalty 1.1   |
--- a/docs/template.md
+++ b/docs/template.md
@@ -12,7 +12,7 @@ A basic Go template consists of three main parts:

 Here's an example of a simple chat template:

-```go
+```gotmpl
 {{- range .Messages }}
 {{ .Role }}: {{ .Content }}
 {{- end }}
@@ -162,6 +162,6 @@ CodeLlama [7B](https://ollama.com/library/codellama:7b-code) and [13B](https://o

 Codestral [22B](https://ollama.com/library/codestral:22b) supports fill-in-middle.

-```go
+```gotmpl
 [SUFFIX]{{ .Suffix }}[PREFIX] {{ .Prompt }}
 ```
--- a/docs/troubleshooting.md
+++ b/docs/troubleshooting.md
@@ -9,7 +9,7 @@ cat ~/.ollama/logs/server.log
 On **Linux** systems with systemd, the logs can be found with this command:

 ```shell
-journalctl -u ollama --no-pager --follow --pager-end 
+journalctl -u ollama --no-pager
 ```

 When you run Ollama in a **container**, the logs go to stdout/stderr in the container:
@@ -26,6 +26,7 @@ When you run Ollama on **Windows**, there are a few different locations. You can
 - `explorer %LOCALAPPDATA%\Ollama` to view logs.  The most recent server logs will be in `server.log` and older logs will be in `server-#.log` 
 - `explorer %LOCALAPPDATA%\Programs\Ollama` to browse the binaries (The installer adds this to your user PATH)
 - `explorer %HOMEPATH%\.ollama` to browse where models and configuration is stored
+- `explorer %TEMP%` where temporary executable files are stored in one or more `ollama*` directories

 To enable additional debug logging to help troubleshoot problems, first **Quit the running app from the tray menu** then in a powershell terminal

@@ -68,6 +69,10 @@ If you run into problems on Linux and want to install an older version, or you'd
 curl -fsSL https://ollama.com/install.sh | OLLAMA_VERSION=0.5.7 sh
 ```

+## Linux tmp noexec 
+
+If your system is configured with the "noexec" flag where Ollama stores its temporary executable files, you can specify an alternate location by setting OLLAMA_TMPDIR to a location writable by the user ollama runs as. For example OLLAMA_TMPDIR=/usr/share/ollama/
+
 ## Linux docker

 If Ollama initially works on the GPU in a docker container, but then switches to running on CPU after some period of time with errors in the server log reporting GPU discovery failures, this can be resolved by disabling systemd cgroup management in Docker.  Edit `/etc/docker/daemon.json` on the host and add `"exec-opts": ["native.cgroupdriver=cgroupfs"]` to the docker configuration.
--- a/docs/windows.md
+++ b/docs/windows.md
@@ -62,6 +62,7 @@ the explorer window by hitting `<Ctrl>+R` and type in:
    - *upgrade.log* contains log output for upgrades
 - `explorer %LOCALAPPDATA%\Programs\Ollama` contains the binaries (The installer adds this to your user PATH)
 - `explorer %HOMEPATH%\.ollama` contains models and configuration
+- `explorer %TEMP%` contains temporary executable files in one or more `ollama*` directories

 ## Uninstall

--- a/envconfig/config.go
+++ b/envconfig/config.go
@@ -149,22 +149,9 @@ func Bool(k string) func() bool {
 	}
 }

-// LogLevel returns the log level for the application.
-// Values are 0 or false INFO (Default), 1 or true DEBUG, 2 TRACE
-func LogLevel() slog.Level {
-	level := slog.LevelInfo
-	if s := Var("OLLAMA_DEBUG"); s != "" {
-		if b, _ := strconv.ParseBool(s); b {
-			level = slog.LevelDebug
-		} else if i, _ := strconv.ParseInt(s, 10, 64); i != 0 {
-			level = slog.Level(i * -4)
-		}
-	}
-
-	return level
-}
-
 var (
+	// Debug enabled additional debug information.
+	Debug = Bool("OLLAMA_DEBUG")
 	// FlashAttention enables the experimental flash attention feature.
 	FlashAttention = Bool("OLLAMA_FLASH_ATTENTION")
 	// KvCacheType is the quantization type for the K/V cache.
@@ -182,7 +169,7 @@ var (
 	// Enable the new Ollama engine
 	NewEngine = Bool("OLLAMA_NEW_ENGINE")
 	// ContextLength sets the default context length
-	ContextLength = Uint("OLLAMA_CONTEXT_LENGTH", 4096)
+	ContextLength = Uint("OLLAMA_CONTEXT_LENGTH", 2048)
 )

 func String(s string) func() string {
@@ -222,6 +209,8 @@ var (
 	MaxRunners = Uint("OLLAMA_MAX_LOADED_MODELS", 0)
 	// MaxQueue sets the maximum number of queued requests. MaxQueue can be configured via the OLLAMA_MAX_QUEUE environment variable.
 	MaxQueue = Uint("OLLAMA_MAX_QUEUE", 512)
+	// MaxVRAM sets a maximum VRAM override in bytes. MaxVRAM can be configured via the OLLAMA_MAX_VRAM environment variable.
+	MaxVRAM = Uint("OLLAMA_MAX_VRAM", 0)
 )

 func Uint64(key string, defaultValue uint64) func() uint64 {
@@ -249,7 +238,7 @@ type EnvVar struct {

 func AsMap() map[string]EnvVar {
 	ret := map[string]EnvVar{
-		"OLLAMA_DEBUG":             {"OLLAMA_DEBUG", LogLevel(), "Show additional debug information (e.g. OLLAMA_DEBUG=1)"},
+		"OLLAMA_DEBUG":             {"OLLAMA_DEBUG", Debug(), "Show additional debug information (e.g. OLLAMA_DEBUG=1)"},
 		"OLLAMA_FLASH_ATTENTION":   {"OLLAMA_FLASH_ATTENTION", FlashAttention(), "Enabled flash attention"},
 		"OLLAMA_KV_CACHE_TYPE":     {"OLLAMA_KV_CACHE_TYPE", KvCacheType(), "Quantization type for the K/V cache (default: f16)"},
 		"OLLAMA_GPU_OVERHEAD":      {"OLLAMA_GPU_OVERHEAD", GpuOverhead(), "Reserve a portion of VRAM per GPU (bytes)"},
@@ -266,7 +255,7 @@ func AsMap() map[string]EnvVar {
 		"OLLAMA_ORIGINS":           {"OLLAMA_ORIGINS", AllowedOrigins(), "A comma separated list of allowed origins"},
 		"OLLAMA_SCHED_SPREAD":      {"OLLAMA_SCHED_SPREAD", SchedSpread(), "Always schedule model across all GPUs"},
 		"OLLAMA_MULTIUSER_CACHE":   {"OLLAMA_MULTIUSER_CACHE", MultiUserCache(), "Optimize prompt caching for multi-user scenarios"},
-		"OLLAMA_CONTEXT_LENGTH":    {"OLLAMA_CONTEXT_LENGTH", ContextLength(), "Context length to use unless otherwise specified (default: 4096)"},
+		"OLLAMA_CONTEXT_LENGTH":    {"OLLAMA_CONTEXT_LENGTH", ContextLength(), "Context length to use unless otherwise specified (default: 2048)"},
 		"OLLAMA_NEW_ENGINE":        {"OLLAMA_NEW_ENGINE", NewEngine(), "Enable the new Ollama engine"},

 		// Informational
--- a/envconfig/config_test.go
+++ b/envconfig/config_test.go
@@ -1,13 +1,11 @@
 package envconfig

 import (
-	"log/slog"
 	"math"
 	"testing"
 	"time"

 	"github.com/google/go-cmp/cmp"
-	"github.com/ollama/ollama/logutil"
 )

 func TestHost(t *testing.T) {
@@ -281,8 +279,8 @@ func TestVar(t *testing.T) {

 func TestContextLength(t *testing.T) {
 	cases := map[string]uint{
-		"":     4096,
-		"2048": 2048,
+		"":     2048,
+		"4096": 4096,
 	}

 	for k, v := range cases {
@@ -294,34 +292,3 @@ func TestContextLength(t *testing.T) {
 		})
 	}
 }
-
-func TestLogLevel(t *testing.T) {
-	cases := map[string]slog.Level{
-		// Default to INFO
-		"":      slog.LevelInfo,
-		"false": slog.LevelInfo,
-		"f":     slog.LevelInfo,
-		"0":     slog.LevelInfo,
-
-		// True values enable Debug
-		"true": slog.LevelDebug,
-		"t":    slog.LevelDebug,
-
-		// Positive values increase verbosity
-		"1": slog.LevelDebug,
-		"2": logutil.LevelTrace,
-
-		// Negative values decrease verbosity
-		"-1": slog.LevelWarn,
-		"-2": slog.LevelError,
-	}
-
-	for k, v := range cases {
-		t.Run(k, func(t *testing.T) {
-			t.Setenv("OLLAMA_DEBUG", k)
-			if i := LogLevel(); i != v {
-				t.Errorf("%s: expected %d, got %d", k, v, i)
-			}
-		})
-	}
-}
--- a/format/time_test.go
+++ b/format/time_test.go
@@ -5,7 +5,7 @@ import (
 	"time"
 )

-func assertEqual(t *testing.T, a any, b any) {
+func assertEqual(t *testing.T, a interface{}, b interface{}) {
 	if a != b {
 		t.Errorf("Assert failed, expected %v, got %v", b, a)
 	}
--- a/fs/config.go
+++ b/fs/config.go
@@ -1,13 +0,0 @@
-package fs
-
-type Config interface {
-	Architecture() string
-	String(string, ...string) string
-	Uint(string, ...uint32) uint32
-	Float(string, ...float32) float32
-	Bool(string, ...bool) bool
-
-	Strings(string, ...[]string) []string
-	Ints(string, ...[]int32) []int32
-	Floats(string, ...[]float32) []float32
-}
--- a/fs/ggml/ggml.go
+++ b/fs/ggml/ggml.go
@@ -6,7 +6,6 @@ import (
 	"fmt"
 	"io"
 	"log/slog"
-	"math"
 	"slices"
 	"strings"

@@ -34,15 +33,15 @@ func (kv KV) Kind() string {
 }

 func (kv KV) ParameterCount() uint64 {
-	return keyValue(kv, "general.parameter_count", uint64(0))
+	return keyValue[uint64](kv, "general.parameter_count")
 }

-func (kv KV) FileType() FileType {
+func (kv KV) FileType() fileType {
 	if t := kv.Uint("general.file_type"); t > 0 {
-		return FileType(t)
+		return fileType(t)
 	}

-	return FileTypeUnknown
+	return fileTypeUnknown
 }

 func (kv KV) BlockCount() uint64 {
@@ -106,44 +105,39 @@ func (kv KV) Bool(key string, defaultValue ...bool) bool {
 }

 func (kv KV) Strings(key string, defaultValue ...[]string) []string {
-	return keyValue(kv, key, &array[string]{values: append(defaultValue, []string(nil))[0]}).values
-}
+	r := keyValue(kv, key, &array{})
+	s := make([]string, r.size)
+	for i := range r.size {
+		s[i] = r.values[i].(string)
+	}

-func (kv KV) Ints(key string, defaultValue ...[]int32) []int32 {
-	return keyValue(kv, key, &array[int32]{values: append(defaultValue, []int32(nil))[0]}).values
+	return s
 }

 func (kv KV) Uints(key string, defaultValue ...[]uint32) []uint32 {
-	return keyValue(kv, key, &array[uint32]{values: append(defaultValue, []uint32(nil))[0]}).values
+	r := keyValue(kv, key, &array{})
+	s := make([]uint32, r.size)
+	for i := range r.size {
+		s[i] = uint32(r.values[i].(int32))
+	}
+
+	return s
 }

 func (kv KV) Floats(key string, defaultValue ...[]float32) []float32 {
-	return keyValue(kv, key, &array[float32]{values: append(defaultValue, []float32(nil))[0]}).values
+	r := keyValue(kv, key, &array{})
+	s := make([]float32, r.size)
+	for i := range r.size {
+		s[i] = float32(r.values[i].(float32))
+	}
+	return s
 }

 func (kv KV) OllamaEngineRequired() bool {
-	return slices.Contains([]string{
-		"gemma3",
-		"mistral3",
-		"llama4",
-		"mllama",
-		"qwen25vl",
-	}, kv.Architecture())
+	return kv.Architecture() == "gemma3"
 }

-type valueTypes interface {
-	uint8 | int8 | uint16 | int16 |
-		uint32 | int32 | uint64 | int64 |
-		string | float32 | float64 | bool
-}
-
-type arrayValueTypes interface {
-	*array[uint8] | *array[int8] | *array[uint16] | *array[int16] |
-		*array[uint32] | *array[int32] | *array[uint64] | *array[int64] |
-		*array[string] | *array[float32] | *array[float64] | *array[bool]
-}
-
-func keyValue[T valueTypes | arrayValueTypes](kv KV, key string, defaultValue ...T) T {
+func keyValue[T string | uint32 | uint64 | float32 | *array | bool](kv KV, key string, defaultValue ...T) T {
 	if !strings.HasPrefix(key, "tokenizer.") && !strings.HasPrefix(key, "general.") {
 		key = kv.Architecture() + "." + key
 	}
@@ -152,7 +146,7 @@ func keyValue[T valueTypes | arrayValueTypes](kv KV, key string, defaultValue ..
 		return val.(T)
 	}

-	slog.Debug("key not found", "key", key, "default", defaultValue[0])
+	slog.Warn("key not found", "key", key, "default", defaultValue[0])
 	return defaultValue[0]
 }

@@ -229,11 +223,7 @@ func (t Tensor) block() (n int) {
 }

 func (t Tensor) blockSize() uint64 {
-	return (TensorType)(t.Kind).BlockSize()
-}
-
-func (t TensorType) BlockSize() uint64 {
-	switch t {
+	switch t.Kind {
 	case
 		0,  // F32
 		1,  // F16
@@ -259,77 +249,73 @@ func (t TensorType) BlockSize() uint64 {
 }

 func (t Tensor) typeSize() uint64 {
-	return TensorType(t.Kind).TypeSize()
-}
+	blockSize := t.blockSize()

-func (t TensorType) TypeSize() uint64 {
-	blockSize := t.BlockSize()
-
-	switch t {
-	case TensorTypeF32:
+	switch t.Kind {
+	case 0: // FP32
 		return 4
-	case TensorTypeF16:
+	case 1: // FP16
 		return 2
-	case TensorTypeQ4_0:
+	case 2: // Q4_0
 		return 2 + blockSize/2
-	case TensorTypeQ4_1:
+	case 3: // Q4_1
 		return 2 + 2 + blockSize/2
-	case TensorTypeQ5_0:
+	case 6: // Q5_0
 		return 2 + 4 + blockSize/2
-	case TensorTypeQ5_1:
+	case 7: // Q5_1
 		return 2 + 2 + 4 + blockSize/2
-	case TensorTypeQ8_0:
+	case 8: // Q8_0
 		return 2 + blockSize
-	case TensorTypeQ8_1:
+	case 9: // Q8_1
 		return 2 + 2 + blockSize
-	case TensorTypeQ2_K:
+	case 10: // Q2_K
 		return blockSize/16 + blockSize/4 + 2 + 2
-	case TensorTypeQ3_K:
+	case 11: // Q3_K
 		return blockSize/8 + blockSize/4 + 12 + 2
-	case TensorTypeQ4_K:
+	case 12: // Q4_K
 		return 2 + 2 + 12 + blockSize/2
-	case TensorTypeQ5_K:
+	case 13: // Q5_K
 		return 2 + 2 + 12 + blockSize/8 + blockSize/2
-	case TensorTypeQ6_K:
+	case 14: // Q6_K
 		return blockSize/2 + blockSize/4 + blockSize/16 + 2
-	case TensorTypeQ8_K:
+	case 15: // Q8_K
 		return 4 + blockSize + 2*blockSize/16
-	case tensorTypeIQ2_XXS:
+	case 16: // IQ2_XXS
 		return 2 + 2*blockSize/8
-	case tensorTypeIQ2_XS:
+	case 17: // IQ2_XS
 		return 2 + 2*blockSize/8 + blockSize/32
-	case tensorTypeIQ3_XXS:
+	case 18: // IQ3_XXS
 		return 2 + blockSize/4 + blockSize/8
-	case tensorTypeIQ1_S:
+	case 19: // IQ1_S
 		return 2 + blockSize/8 + blockSize/16
-	case tensorTypeIQ4_NL:
+	case 20: // IQ4_NL
 		return 2 + blockSize/2
-	case tensorTypeIQ3_S:
+	case 21: // IQ3_S
 		return 2 + blockSize/4 + blockSize/8 + blockSize/32 + 4
-	case tensorTypeIQ2_S:
+	case 22: // IQ2_S
 		return 2 + blockSize/4 + blockSize/16
-	case tensorTypeIQ4_XS:
+	case 23: // IQ4_XS
 		return 2 + 2 + blockSize/2 + blockSize/64
-	case TensorTypeI8:
+	case 24: // I8
 		return 1
-	case TensorTypeI16:
+	case 25: // I16
 		return 2
-	case TensorTypeI32:
+	case 26: // I32
 		return 4
-	case TensorTypeI64:
+	case 27: // I64
 		return 8
-	case TensorTypeF64:
+	case 28: // F64
 		return 8
-	case tensorTypeIQ1_M:
+	case 29: // IQ1_M
 		return blockSize/8 + blockSize/16 + blockSize/32
-	case TensorTypeBF16:
+	case 30: // BF16
 		return 2
 	default:
 		return 0
 	}
 }

-func (t Tensor) Elements() uint64 {
+func (t Tensor) parameters() uint64 {
 	var count uint64 = 1
 	for _, n := range t.Shape {
 		count *= n
@@ -338,11 +324,11 @@ func (t Tensor) Elements() uint64 {
 }

 func (t Tensor) Size() uint64 {
-	return t.Elements() * t.typeSize() / t.blockSize()
+	return t.parameters() * t.typeSize() / t.blockSize()
 }

 func (t Tensor) Type() string {
-	return TensorType(t.Kind).String()
+	return fileType(t.Kind).String()
 }

 type container interface {
@@ -386,8 +372,13 @@ func DetectContentType(b []byte) string {
 // Decode decodes a GGML model from the given reader.
 //
 // It collects array values for arrays with a size less than or equal to
-// maxArraySize. If the maxArraySize is negative, all arrays are collected.
+// maxArraySize. If maxArraySize is 0, the default value of 1024 is used. If
+// the maxArraySize is negative, all arrays are collected.
 func Decode(rs io.ReadSeeker, maxArraySize int) (*GGML, int64, error) {
+	if maxArraySize == 0 {
+		maxArraySize = 1024
+	}
+
 	rs = bufioutil.NewBufferedSeeker(rs, 32<<10)

 	var magic uint32
@@ -422,11 +413,11 @@ func Decode(rs io.ReadSeeker, maxArraySize int) (*GGML, int64, error) {
 	}, offset, nil
 }

-func (f GGML) GraphSize(context, batch uint64, numParallel int, kvCacheType string) (kv []uint64, partialOffload, fullOffload uint64) {
+func (f GGML) GraphSize(context, batch uint64, kvCacheType string) (kv, partialOffload, fullOffload uint64) {
 	embedding := f.KV().EmbeddingLength()
 	heads := f.KV().HeadCount()
 	headsKV := f.KV().HeadCountKV()
-	vocab := uint64(f.KV()["tokenizer.ggml.tokens"].(*array[string]).size)
+	vocab := uint64(f.KV()["tokenizer.ggml.tokens"].(*array).size)

 	embeddingHeads := f.KV().EmbeddingHeadCount()
 	embeddingHeadsK := f.KV().EmbeddingHeadCountK()
@@ -435,13 +426,10 @@ func (f GGML) GraphSize(context, batch uint64, numParallel int, kvCacheType stri
 	layers := f.Tensors().GroupLayers()

 	bytesPerElement := kvCacheBytesPerElement(kvCacheType)
-	kv = make([]uint64, f.KV().BlockCount())
-	for i := range kv {
-		kv[i] = uint64(float64(context*(embeddingHeadsK+embeddingHeadsV)*headsKV) * bytesPerElement)
-	}
+	kv = uint64(float64(context*f.KV().BlockCount()*(embeddingHeadsK+embeddingHeadsV)*headsKV) * bytesPerElement)

 	switch f.KV().Architecture() {
-	case "llama", "llama4":
+	case "llama":
 		fullOffload = max(
 			4*batch*(1+4*embedding+context*(1+heads)),
 			4*batch*(embedding+vocab),
@@ -455,7 +443,7 @@ func (f GGML) GraphSize(context, batch uint64, numParallel int, kvCacheType stri

 		if ffnGateExpsWeight, ok := layers["blk.0"]["ffn_gate_exps.weight"]; ok {
 			// mixtral 8x22b
-			ff := uint64(f.KV().Uint("feed_forward_length"))
+			ff := uint64(f.KV()["llama.feed_forward_length"].(uint32))
 			partialOffload = max(
 				3*ffnGateExpsWeight.Size()+4*batch*(2*ff+headsKV+embedding+context+embeddingHeads*headsKV),
 				4*(context*batch*heads+context*embeddingHeads*headsKV+batch*1024+embeddingHeads*headsKV*batch),
@@ -472,14 +460,16 @@ func (f GGML) GraphSize(context, batch uint64, numParallel int, kvCacheType stri
 	case "mllama":
 		var visionTokens, tiles uint64 = 1601, 4

-		crossAttentionLayers := f.KV().Ints("attention.cross_attention_layers")
-		for i := range kv {
-			if slices.Contains(crossAttentionLayers, int32(i)) {
-				kv[i] = headsKV * (embeddingHeadsK + embeddingHeadsV) *
-					4 * // sizeof(float32)
-					visionTokens *
-					tiles
-			}
+		if crossAttentionLayers, ok := f.KV()["mllama.attention.cross_attention_layers"].(*array); ok {
+			kv = headsKV *
+				(embeddingHeadsK + embeddingHeadsV) * // one for K, one for V
+				(2* // sizeof(float16)
+					(f.KV().BlockCount()-uint64(crossAttentionLayers.size))* // num non-cross attention layers
+					context +
+					4* // sizeof(float32)
+						uint64(crossAttentionLayers.size)* // num cross attention layers
+						visionTokens*
+						tiles)
 		}

 		fullOffload = max(
@@ -491,7 +481,7 @@ func (f GGML) GraphSize(context, batch uint64, numParallel int, kvCacheType stri
 		var ropeFreqsCount uint64
 		if ropeFreqs, ok := f.Tensors().GroupLayers()["rope_freqs"]; ok {
 			if ropeFreqsWeights, ok := ropeFreqs["weights"]; ok {
-				ropeFreqsCount = ropeFreqsWeights.Elements()
+				ropeFreqsCount = ropeFreqsWeights.parameters()
 			}
 		}

@@ -515,20 +505,6 @@ func (f GGML) GraphSize(context, batch uint64, numParallel int, kvCacheType stri
 				4*embeddingHeadsK*context*8+
 				embedding*embeddingHeadsK*heads*9/16,
 		)
-
-		// Gemma2 also has sliding window attention but we only have an optimized implementation in the Ollama
-		// engine. Gemma3 always uses the Ollama engine.
-		if f.KV().Architecture() == "gemma3" {
-			const gemma3GlobalCacheCount = 6
-			slidingWindow := (uint64(numParallel) * uint64(f.KV().Uint("attention.sliding_window"))) + batch
-			for i := range kv {
-				// Every 6th layer is a global layer, which is the full context size that has already been set. The other
-				// layers are the smaller local (sliding) layers.
-				if (i+1)%gemma3GlobalCacheCount != 0 {
-					kv[i] = uint64(float64(slidingWindow*(embeddingHeadsK+embeddingHeadsV)*headsKV) * bytesPerElement)
-				}
-			}
-		}
 	case "command-r":
 		fullOffload = max(
 			4*batch*(embedding+vocab),
@@ -647,36 +623,10 @@ func (llm GGML) VisionGraphSize() (weights, graphSize uint64) {
 			embeddingLength*numPatches*maxNumTiles +
 			9*embeddingLength*numPaddedPatches*maxNumTiles +
 			numPaddedPatches*maxNumTiles*numPaddedPatches*maxNumTiles*headCount)
-	case "gemma3", "mistral3":
+	case "gemma3":
 		graphSize = 4 * (imageSize*imageSize*numChannels +
 			embeddingLength*patchSize +
 			numPatches*numPatches*headCount)
-	case "qwen25vl":
-		maxPixels := uint64(llm.KV().Uint("vision.max_pixels", 28*28*1280))
-		mergeSize := uint64(llm.KV().Uint("vision.spatial_merge_size", 2))
-		temporalPatchSize := uint64(2)
-
-		// Calculate max possible patches based on max_pixels
-		maxHeight := uint64(math.Sqrt(float64(maxPixels)))
-		maxWidth := maxPixels / maxHeight
-		maxGridHeight := maxHeight / patchSize
-		maxGridWidth := maxWidth / patchSize
-		// Account for merged patches (2x2 grid)
-		numPatches := (maxGridHeight * maxGridWidth) / (mergeSize * mergeSize)
-
-		// Calculate graph size based on typical operations in ProcessImage and createPatches
-		graphSize = 4 * (maxPixels*numChannels + // Original image storage
-			// Normalized pixels
-			maxPixels*numChannels +
-			// Patches storage (numPatches * channels * temporalPatchSize * patchSize^2)
-			numPatches*numChannels*temporalPatchSize*patchSize*patchSize +
-			// Self-attention calculations (similar to other architectures)
-			numPatches*numPatches*headCount +
-			// Additional buffer for processing
-			embeddingLength*numPatches)
-	case "llama4":
-		// vision graph is computed independently in the same schedule
-		// and is negligible compared to the worst case text graph
 	}

 	return weights, graphSize
--- a/fs/ggml/ggml_test.go
+++ b/fs/ggml/ggml_test.go
@@ -2,7 +2,6 @@ package ggml

 import (
 	"maps"
-	"math"
 	"slices"
 	"strconv"
 	"strings"
@@ -211,61 +210,3 @@ func TestTensorTypes(t *testing.T) {
 		})
 	}
 }
-
-func TestKeyValue(t *testing.T) {
-	kv := KV{
-		"general.architecture": "test",
-		"test.strings":         &array[string]{size: 3, values: []string{"a", "b", "c"}},
-		"test.float32s":        &array[float32]{size: 3, values: []float32{1.0, 2.0, 3.0}},
-		"test.int32s":          &array[int32]{size: 3, values: []int32{1, 2, 3}},
-		"test.uint32s":         &array[uint32]{size: 3, values: []uint32{1, 2, 3}},
-	}
-
-	if diff := cmp.Diff(kv.Strings("strings"), []string{"a", "b", "c"}); diff != "" {
-		t.Errorf("unexpected strings (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Strings("nonexistent.strings"), []string(nil)); diff != "" {
-		t.Errorf("unexpected strings (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Strings("default.strings", []string{"ollama"}), []string{"ollama"}); diff != "" {
-		t.Errorf("unexpected strings (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Floats("float32s"), []float32{1.0, 2.0, 3.0}); diff != "" {
-		t.Errorf("unexpected float32s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Floats("nonexistent.float32s"), []float32(nil)); diff != "" {
-		t.Errorf("unexpected float32s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Floats("default.float32s", []float32{math.MaxFloat32}), []float32{math.MaxFloat32}); diff != "" {
-		t.Errorf("unexpected float32s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Ints("int32s"), []int32{1, 2, 3}); diff != "" {
-		t.Errorf("unexpected int8s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Ints("nonexistent.int32s"), []int32(nil)); diff != "" {
-		t.Errorf("unexpected int8s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Ints("default.int32s", []int32{math.MaxInt32}), []int32{math.MaxInt32}); diff != "" {
-		t.Errorf("unexpected int8s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Uints("uint32s"), []uint32{1, 2, 3}); diff != "" {
-		t.Errorf("unexpected uint8s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Uints("nonexistent.uint32s"), []uint32(nil)); diff != "" {
-		t.Errorf("unexpected uint8s (-got +want):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(kv.Uints("default.uint32s", []uint32{math.MaxUint32}), []uint32{math.MaxUint32}); diff != "" {
-		t.Errorf("unexpected uint8s (-got +want):\n%s", diff)
-	}
-}
--- a/fs/ggml/gguf.go
+++ b/fs/ggml/gguf.go
@@ -9,12 +9,8 @@ import (
 	"io"
 	"log/slog"
 	"maps"
-	"os"
-	"runtime"
 	"slices"
 	"strings"
-
-	"golang.org/x/sync/errgroup"
 )

 type containerGGUF struct {
@@ -40,6 +36,10 @@ type containerGGUF struct {
 	maxArraySize int
 }

+func (c *containerGGUF) canCollectArray(size int) bool {
+	return c.maxArraySize < 0 || size <= c.maxArraySize
+}
+
 func (c *containerGGUF) Name() string {
 	return "gguf"
 }
@@ -229,13 +229,16 @@ func (llm *gguf) Decode(rs io.ReadSeeker) error {
 		}

 		llm.tensors = append(llm.tensors, &tensor)
-		llm.parameters += tensor.Elements()
+		llm.parameters += tensor.parameters()
 	}

 	// patch KV with parameter count
 	llm.kv["general.parameter_count"] = llm.parameters

-	alignment := llm.kv.Uint("general.alignment", 32)
+	alignment, ok := llm.kv["general.alignment"].(uint32)
+	if !ok {
+		alignment = 32
+	}

 	offset, err := rs.Seek(0, io.SeekCurrent)
 	if err != nil {
@@ -295,23 +298,6 @@ func readGGUFV1String(llm *gguf, r io.Reader) (string, error) {
 	return b.String(), nil
 }

-func readGGUFV1StringsData(llm *gguf, r io.Reader, a *array[string]) (any, error) {
-	for i := range a.size {
-		if a.values != nil {
-			e, err := readGGUFV1String(llm, r)
-			if err != nil {
-				return nil, err
-			}
-
-			a.values[i] = e
-		} else {
-			discardGGUFString(llm, r)
-		}
-	}
-
-	return a, nil
-}
-
 func discardGGUFString(llm *gguf, r io.Reader) error {
 	buf := llm.scratch[:8]
 	_, err := io.ReadFull(r, buf)
@@ -369,44 +355,78 @@ func writeGGUFString(w io.Writer, s string) error {
 	return err
 }

-func readGGUFStringsData(llm *gguf, r io.Reader, a *array[string]) (any, error) {
-	for i := range a.size {
-		if a.values != nil {
-			e, err := readGGUFString(llm, r)
-			if err != nil {
-				return nil, err
-			}
+type array struct {
+	size   int
+	values []any
+}

+func (a *array) MarshalJSON() ([]byte, error) {
+	return json.Marshal(a.values)
+}
+
+func readGGUFV1Array(llm *gguf, r io.Reader) (*array, error) {
+	t, err := readGGUF[uint32](llm, r)
+	if err != nil {
+		return nil, err
+	}
+
+	n, err := readGGUF[uint32](llm, r)
+	if err != nil {
+		return nil, err
+	}
+
+	a := &array{size: int(n)}
+	if llm.canCollectArray(int(n)) {
+		a.values = make([]any, 0, int(n))
+	}
+
+	for i := range n {
+		var e any
+		switch t {
+		case ggufTypeUint8:
+			e, err = readGGUF[uint8](llm, r)
+		case ggufTypeInt8:
+			e, err = readGGUF[int8](llm, r)
+		case ggufTypeUint16:
+			e, err = readGGUF[uint16](llm, r)
+		case ggufTypeInt16:
+			e, err = readGGUF[int16](llm, r)
+		case ggufTypeUint32:
+			e, err = readGGUF[uint32](llm, r)
+		case ggufTypeInt32:
+			e, err = readGGUF[int32](llm, r)
+		case ggufTypeUint64:
+			e, err = readGGUF[uint64](llm, r)
+		case ggufTypeInt64:
+			e, err = readGGUF[int64](llm, r)
+		case ggufTypeFloat32:
+			e, err = readGGUF[float32](llm, r)
+		case ggufTypeFloat64:
+			e, err = readGGUF[float64](llm, r)
+		case ggufTypeBool:
+			e, err = readGGUF[bool](llm, r)
+		case ggufTypeString:
+			e, err = readGGUFV1String(llm, r)
+		default:
+			return nil, fmt.Errorf("invalid array type: %d", t)
+		}
+		if err != nil {
+			return nil, err
+		}
+
+		if a.values != nil {
 			a.values[i] = e
-		} else {
-			discardGGUFString(llm, r)
 		}
 	}

 	return a, nil
 }

-type array[T any] struct {
-	// size is the actual size of the array
-	size int
-
-	// values is the array of values. this is nil if the array is larger than configured maxSize
-	values []T
-}
-
-func (a *array[T]) MarshalJSON() ([]byte, error) {
-	return json.Marshal(a.values)
-}
-
-func newArray[T any](size, maxSize int) *array[T] {
-	a := array[T]{size: size}
-	if maxSize < 0 || size <= maxSize {
-		a.values = make([]T, size)
+func readGGUFArray(llm *gguf, r io.Reader) (*array, error) {
+	if llm.Version == 1 {
+		return readGGUFV1Array(llm, r)
 	}
-	return &a
-}

-func readGGUFArray(llm *gguf, r io.Reader) (any, error) {
 	t, err := readGGUF[uint32](llm, r)
 	if err != nil {
 		return nil, err
@@ -417,55 +437,45 @@ func readGGUFArray(llm *gguf, r io.Reader) (any, error) {
 		return nil, err
 	}

-	switch t {
-	case ggufTypeUint8:
-		a := newArray[uint8](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeInt8:
-		a := newArray[int8](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeUint16:
-		a := newArray[uint16](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeInt16:
-		a := newArray[int16](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeUint32:
-		a := newArray[uint32](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeInt32:
-		a := newArray[int32](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeUint64:
-		a := newArray[uint64](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeInt64:
-		a := newArray[int64](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeFloat32:
-		a := newArray[float32](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeFloat64:
-		a := newArray[float64](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeBool:
-		a := newArray[bool](int(n), llm.maxArraySize)
-		return readGGUFArrayData(llm, r, a)
-	case ggufTypeString:
-		a := newArray[string](int(n), llm.maxArraySize)
-		if llm.Version == 1 {
-			return readGGUFV1StringsData(llm, r, a)
-		}
-
-		return readGGUFStringsData(llm, r, a)
-	default:
-		return nil, fmt.Errorf("invalid array type: %d", t)
+	a := &array{size: int(n)}
+	if llm.canCollectArray(int(n)) {
+		a.values = make([]any, int(n))
 	}
-}

-func readGGUFArrayData[T any](llm *gguf, r io.Reader, a *array[T]) (any, error) {
-	for i := range a.size {
-		e, err := readGGUF[T](llm, r)
+	for i := range n {
+		var e any
+		switch t {
+		case ggufTypeUint8:
+			e, err = readGGUF[uint8](llm, r)
+		case ggufTypeInt8:
+			e, err = readGGUF[int8](llm, r)
+		case ggufTypeUint16:
+			e, err = readGGUF[uint16](llm, r)
+		case ggufTypeInt16:
+			e, err = readGGUF[int16](llm, r)
+		case ggufTypeUint32:
+			e, err = readGGUF[uint32](llm, r)
+		case ggufTypeInt32:
+			e, err = readGGUF[int32](llm, r)
+		case ggufTypeUint64:
+			e, err = readGGUF[uint64](llm, r)
+		case ggufTypeInt64:
+			e, err = readGGUF[int64](llm, r)
+		case ggufTypeFloat32:
+			e, err = readGGUF[float32](llm, r)
+		case ggufTypeFloat64:
+			e, err = readGGUF[float64](llm, r)
+		case ggufTypeBool:
+			e, err = readGGUF[bool](llm, r)
+		case ggufTypeString:
+			if a.values != nil {
+				e, err = readGGUFString(llm, r)
+			} else {
+				err = discardGGUFString(llm, r)
+			}
+		default:
+			return nil, fmt.Errorf("invalid array type: %d", t)
+		}
 		if err != nil {
 			return nil, err
 		}
@@ -492,38 +502,23 @@ func writeGGUFArray[S ~[]E, E any](w io.Writer, t uint32, s S) error {
 		return err
 	}

-	if t == ggufTypeString {
-		for _, e := range any(s).([]string) {
-			if err := binary.Write(w, binary.LittleEndian, uint64(len(e))); err != nil {
-				return err
-			}
-
-			if err := binary.Write(w, binary.LittleEndian, []byte(e)); err != nil {
-				return err
-			}
-		}
-		return nil
-	}
-
 	return binary.Write(w, binary.LittleEndian, s)
 }

-func WriteGGUF(f *os.File, kv KV, ts []*Tensor) error {
-	alignment := kv.Uint("general.alignment", 32)
-
-	if err := binary.Write(f, binary.LittleEndian, []byte("GGUF")); err != nil {
+func WriteGGUF(ws io.WriteSeeker, kv KV, ts []Tensor) error {
+	if err := binary.Write(ws, binary.LittleEndian, []byte("GGUF")); err != nil {
 		return err
 	}

-	if err := binary.Write(f, binary.LittleEndian, uint32(3)); err != nil {
+	if err := binary.Write(ws, binary.LittleEndian, uint32(3)); err != nil {
 		return err
 	}

-	if err := binary.Write(f, binary.LittleEndian, uint64(len(ts))); err != nil {
+	if err := binary.Write(ws, binary.LittleEndian, uint64(len(ts))); err != nil {
 		return err
 	}

-	if err := binary.Write(f, binary.LittleEndian, uint64(len(kv))); err != nil {
+	if err := binary.Write(ws, binary.LittleEndian, uint64(len(kv))); err != nil {
 		return err
 	}

@@ -531,12 +526,12 @@ func WriteGGUF(f *os.File, kv KV, ts []*Tensor) error {
 	slices.Sort(keys)

 	for _, key := range keys {
-		if err := ggufWriteKV(f, key, kv[key]); err != nil {
+		if err := ggufWriteKV(ws, key, kv[key]); err != nil {
 			return err
 		}
 	}

-	slices.SortStableFunc(ts, func(a, b *Tensor) int {
+	slices.SortStableFunc(ts, func(a, b Tensor) int {
 		if i, j := a.block(), b.block(); i < 0 && j > 0 {
 			return 1
 		} else if i > 0 && j < 0 {
@@ -547,34 +542,22 @@ func WriteGGUF(f *os.File, kv KV, ts []*Tensor) error {
 	})

 	var s uint64
-	for i := range ts {
-		ts[i].Offset = s
-		if err := ggufWriteTensorInfo(f, ts[i]); err != nil {
+	for _, t := range ts {
+		t.Offset = s
+		if err := ggufWriteTensorInfo(ws, t); err != nil {
 			return err
 		}
-		s += ts[i].Size()
-		s += uint64(ggufPadding(int64(s), int64(alignment)))
+		s += t.Size()
 	}

-	offset, err := f.Seek(0, io.SeekCurrent)
-	if err != nil {
-		return err
-	}
-	offset += ggufPadding(offset, int64(alignment))
-
-	var g errgroup.Group
-	g.SetLimit(runtime.GOMAXPROCS(0))
-	// TODO consider reducing if tensors size * gomaxprocs is larger than free memory
+	var alignment int64 = 32
 	for _, t := range ts {
-		t := t
-		w := io.NewOffsetWriter(f, offset+int64(t.Offset))
-		g.Go(func() error {
-			_, err := t.WriteTo(w)
+		if err := ggufWriteTensor(ws, t, alignment); err != nil {
 			return err
-		})
+		}
 	}

-	return g.Wait()
+	return nil
 }

 func ggufWriteKV(ws io.WriteSeeker, k string, v any) error {
@@ -589,10 +572,8 @@ func ggufWriteKV(ws io.WriteSeeker, k string, v any) error {

 	var err error
 	switch v := v.(type) {
-	case uint32, FileType:
+	case uint32:
 		err = writeGGUF(ws, ggufTypeUint32, v)
-	case uint64:
-		err = writeGGUF(ws, ggufTypeUint64, v)
 	case float32:
 		err = writeGGUF(ws, ggufTypeFloat32, v)
 	case bool:
@@ -601,20 +582,32 @@ func ggufWriteKV(ws io.WriteSeeker, k string, v any) error {
 		err = writeGGUFString(ws, v)
 	case []int32:
 		err = writeGGUFArray(ws, ggufTypeInt32, v)
-	case *array[int32]:
-		err = writeGGUFArray(ws, ggufTypeInt32, v.values)
 	case []uint32:
 		err = writeGGUFArray(ws, ggufTypeUint32, v)
-	case *array[uint32]:
-		err = writeGGUFArray(ws, ggufTypeUint32, v.values)
 	case []float32:
 		err = writeGGUFArray(ws, ggufTypeFloat32, v)
-	case *array[float32]:
-		err = writeGGUFArray(ws, ggufTypeFloat32, v.values)
 	case []string:
-		err = writeGGUFArray(ws, ggufTypeString, v)
-	case *array[string]:
-		err = writeGGUFArray(ws, ggufTypeString, v.values)
+		if err := binary.Write(ws, binary.LittleEndian, ggufTypeArray); err != nil {
+			return err
+		}
+
+		if err := binary.Write(ws, binary.LittleEndian, ggufTypeString); err != nil {
+			return err
+		}
+
+		if err := binary.Write(ws, binary.LittleEndian, uint64(len(v))); err != nil {
+			return err
+		}
+
+		for _, e := range v {
+			if err := binary.Write(ws, binary.LittleEndian, uint64(len(e))); err != nil {
+				return err
+			}
+
+			if err := binary.Write(ws, binary.LittleEndian, []byte(e)); err != nil {
+				return err
+			}
+		}
 	default:
 		return fmt.Errorf("improper type for '%s'", k)
 	}
@@ -622,7 +615,7 @@ func ggufWriteKV(ws io.WriteSeeker, k string, v any) error {
 	return err
 }

-func ggufWriteTensorInfo(ws io.WriteSeeker, t *Tensor) error {
+func ggufWriteTensorInfo(ws io.WriteSeeker, t Tensor) error {
 	slog.Debug(t.Name, "kind", t.Kind, "shape", t.Shape, "offset", t.Offset)
 	if err := binary.Write(ws, binary.LittleEndian, uint64(len(t.Name))); err != nil {
 		return err
@@ -636,8 +629,8 @@ func ggufWriteTensorInfo(ws io.WriteSeeker, t *Tensor) error {
 		return err
 	}

-	for _, n := range t.Shape {
-		if err := binary.Write(ws, binary.LittleEndian, n); err != nil {
+	for i := range len(t.Shape) {
+		if err := binary.Write(ws, binary.LittleEndian, t.Shape[len(t.Shape)-i-1]); err != nil {
 			return err
 		}
 	}
@@ -649,6 +642,20 @@ func ggufWriteTensorInfo(ws io.WriteSeeker, t *Tensor) error {
 	return binary.Write(ws, binary.LittleEndian, t.Offset)
 }

+func ggufWriteTensor(ws io.WriteSeeker, t Tensor, alignment int64) error {
+	offset, err := ws.Seek(0, io.SeekCurrent)
+	if err != nil {
+		return err
+	}
+
+	if err := binary.Write(ws, binary.LittleEndian, bytes.Repeat([]byte{0}, int(ggufPadding(offset, alignment)))); err != nil {
+		return err
+	}
+
+	_, err = t.WriteTo(ws)
+	return err
+}
+
 func ggufPadding(offset, align int64) int64 {
 	return (align - offset%align) % align
 }
--- a/fs/ggml/gguf_test.go
+++ b/fs/ggml/gguf_test.go
@@ -1,63 +0,0 @@
-package ggml
-
-import (
-	"bytes"
-	"os"
-	"slices"
-	"testing"
-
-	"github.com/google/go-cmp/cmp"
-)
-
-func TestWriteGGUF(t *testing.T) {
-	w, err := os.CreateTemp(t.TempDir(), "*.bin")
-	if err != nil {
-		t.Fatal(err)
-	}
-	defer w.Close()
-
-	if err := WriteGGUF(w, KV{
-		"general.alignment": uint32(16),
-	}, []*Tensor{
-		{Name: "test.0", Shape: []uint64{2, 3}, WriterTo: bytes.NewBuffer(slices.Repeat([]byte{0}, 2*3*4))},
-		{Name: "test.1", Shape: []uint64{2, 3}, WriterTo: bytes.NewBuffer(slices.Repeat([]byte{0}, 2*3*4))},
-		{Name: "test.2", Shape: []uint64{2, 3}, WriterTo: bytes.NewBuffer(slices.Repeat([]byte{0}, 2*3*4))},
-		{Name: "test.3", Shape: []uint64{2, 3}, WriterTo: bytes.NewBuffer(slices.Repeat([]byte{0}, 2*3*4))},
-		{Name: "test.4", Shape: []uint64{2, 3}, WriterTo: bytes.NewBuffer(slices.Repeat([]byte{0}, 2*3*4))},
-		{Name: "test.5", Shape: []uint64{2, 3}, WriterTo: bytes.NewBuffer(slices.Repeat([]byte{0}, 2*3*4))},
-	}); err != nil {
-		t.Fatal(err)
-	}
-
-	r, err := os.Open(w.Name())
-	if err != nil {
-		t.Fatal(err)
-	}
-	defer r.Close()
-
-	ff, _, err := Decode(r, 0)
-	if err != nil {
-		t.Fatal(err)
-	}
-
-	if diff := cmp.Diff(ff.KV(), KV{
-		"general.alignment":       uint32(16),
-		"general.parameter_count": uint64(36),
-	}); diff != "" {
-		t.Errorf("Mismatch (-want +got):\n%s", diff)
-	}
-
-	if diff := cmp.Diff(ff.Tensors(), Tensors{
-		Offset: 336,
-		items: []*Tensor{
-			{Name: "test.0", Offset: 0, Shape: []uint64{2, 3}},
-			{Name: "test.1", Offset: 32, Shape: []uint64{2, 3}},
-			{Name: "test.2", Offset: 64, Shape: []uint64{2, 3}},
-			{Name: "test.3", Offset: 96, Shape: []uint64{2, 3}},
-			{Name: "test.4", Offset: 128, Shape: []uint64{2, 3}},
-			{Name: "test.5", Offset: 160, Shape: []uint64{2, 3}},
-		},
-	}, cmp.AllowUnexported(Tensors{})); diff != "" {
-		t.Errorf("Mismatch (-want +got):\n%s", diff)
-	}
-}
--- a/fs/ggml/type.go
+++ b/fs/ggml/type.go
@@ -1,31 +1,26 @@
 package ggml

-import (
-	"fmt"
-	"log/slog"
-	"strings"
-)
+import "fmt"

-// FileType is the Go equivalent to llama_ftype used for gguf file typing
-type FileType uint32
+type fileType uint32

 const (
-	FileTypeF32 FileType = iota
-	FileTypeF16
+	fileTypeF32 fileType = iota
+	fileTypeF16
 	fileTypeQ4_0
 	fileTypeQ4_1
-	fileTypeQ4_1_F16 // unused by GGML
-	fileTypeQ4_2     // unused by GGML
-	fileTypeQ4_3     // unused by GGML
-	FileTypeQ8_0
+	fileTypeQ4_1_F16
+	fileTypeQ4_2 // unused
+	fileTypeQ4_3 // unused
+	fileTypeQ8_0
 	fileTypeQ5_0
 	fileTypeQ5_1
 	fileTypeQ2_K
 	fileTypeQ3_K_S
 	fileTypeQ3_K_M
 	fileTypeQ3_K_L
-	FileTypeQ4_K_S
-	FileTypeQ4_K_M
+	fileTypeQ4_K_S
+	fileTypeQ4_K_M
 	fileTypeQ5_K_S
 	fileTypeQ5_K_M
 	fileTypeQ6_K
@@ -42,62 +37,93 @@ const (
 	fileTypeIQ2_M
 	fileTypeIQ4_XS
 	fileTypeIQ1_M
-	FileTypeBF16
-	fileTypeQ4_0_4_4 // unused by GGML
-	fileTypeQ4_0_4_8 // unused by GGML
-	fileTypeQ4_0_8_8 // unused by GGML
-	fileTypeTQ1_0
-	fileTypeTQ2_0
+	fileTypeBF16

-	FileTypeUnknown = 1024
+	fileTypeUnknown
 )

-// ParseFileType parses the provided GGUF file type
-// Only Ollama supported types are considered valid
-func ParseFileType(s string) (FileType, error) {
+func ParseFileType(s string) (fileType, error) {
 	switch s {
 	case "F32":
-		return FileTypeF32, nil
+		return fileTypeF32, nil
 	case "F16":
-		return FileTypeF16, nil
+		return fileTypeF16, nil
+	case "Q4_0":
+		return fileTypeQ4_0, nil
+	case "Q4_1":
+		return fileTypeQ4_1, nil
+	case "Q4_1_F16":
+		return fileTypeQ4_1_F16, nil
 	case "Q8_0":
-		return FileTypeQ8_0, nil
+		return fileTypeQ8_0, nil
+	case "Q5_0":
+		return fileTypeQ5_0, nil
+	case "Q5_1":
+		return fileTypeQ5_1, nil
+	case "Q2_K":
+		return fileTypeQ2_K, nil
+	case "Q3_K_S":
+		return fileTypeQ3_K_S, nil
+	case "Q3_K_M":
+		return fileTypeQ3_K_M, nil
+	case "Q3_K_L":
+		return fileTypeQ3_K_L, nil
 	case "Q4_K_S":
-		return FileTypeQ4_K_S, nil
-	case "Q4_K_M", "Q4_K":
-		return FileTypeQ4_K_M, nil
+		return fileTypeQ4_K_S, nil
+	case "Q4_K_M":
+		return fileTypeQ4_K_M, nil
+	case "Q5_K_S":
+		return fileTypeQ5_K_S, nil
+	case "Q5_K_M":
+		return fileTypeQ5_K_M, nil
+	case "Q6_K":
+		return fileTypeQ6_K, nil
+	case "IQ2_XXS":
+		return fileTypeIQ2_XXS, nil
+	case "IQ2_XS":
+		return fileTypeIQ2_XS, nil
+	case "Q2_K_S":
+		return fileTypeQ2_K_S, nil
+	case "IQ3_XS":
+		return fileTypeIQ3_XS, nil
+	case "IQ3_XXS":
+		return fileTypeIQ3_XXS, nil
+	case "IQ1_S":
+		return fileTypeIQ1_S, nil
+	case "IQ4_NL":
+		return fileTypeIQ4_NL, nil
+	case "IQ3_S":
+		return fileTypeIQ3_S, nil
+	case "IQ3_M":
+		return fileTypeIQ3_M, nil
+	case "IQ2_S":
+		return fileTypeIQ2_S, nil
+	case "IQ2_M":
+		return fileTypeIQ2_M, nil
+	case "IQ4_XS":
+		return fileTypeIQ4_XS, nil
+	case "IQ1_M":
+		return fileTypeIQ1_M, nil
 	case "BF16":
-		return FileTypeBF16, nil
+		return fileTypeBF16, nil
 	default:
-		supportedFileTypes := []FileType{
-			FileTypeF32,
-			FileTypeF16,
-			FileTypeQ4_K_S,
-			FileTypeQ4_K_M,
-			FileTypeQ8_0,
-			// fsggml.FileTypeBF16, // TODO
-		}
-		strs := make([]string, len(supportedFileTypes))
-		for i := range supportedFileTypes {
-			strs[i] = supportedFileTypes[i].String()
-		}
-
-		return FileTypeUnknown, fmt.Errorf("unsupported quantization type %s - supported types are %s", s, strings.Join(strs, ", "))
+		return fileTypeUnknown, fmt.Errorf("unknown fileType: %s", s)
 	}
 }

-func (t FileType) String() string {
-	// Note: this routine will return a broader set of file types for existing models
+func (t fileType) String() string {
 	switch t {
-	case FileTypeF32:
+	case fileTypeF32:
 		return "F32"
-	case FileTypeF16:
+	case fileTypeF16:
 		return "F16"
 	case fileTypeQ4_0:
 		return "Q4_0"
 	case fileTypeQ4_1:
 		return "Q4_1"
-	case FileTypeQ8_0:
+	case fileTypeQ4_1_F16:
+		return "Q4_1_F16"
+	case fileTypeQ8_0:
 		return "Q8_0"
 	case fileTypeQ5_0:
 		return "Q5_0"
@@ -111,9 +137,9 @@ func (t FileType) String() string {
 		return "Q3_K_M"
 	case fileTypeQ3_K_L:
 		return "Q3_K_L"
-	case FileTypeQ4_K_S:
+	case fileTypeQ4_K_S:
 		return "Q4_K_S"
-	case FileTypeQ4_K_M:
+	case fileTypeQ4_K_M:
 		return "Q4_K_M"
 	case fileTypeQ5_K_S:
 		return "Q5_K_S"
@@ -121,198 +147,39 @@ func (t FileType) String() string {
 		return "Q5_K_M"
 	case fileTypeQ6_K:
 		return "Q6_K"
+	case fileTypeIQ2_XXS:
+		return "IQ2_XXS"
+	case fileTypeIQ2_XS:
+		return "IQ2_XS"
 	case fileTypeQ2_K_S:
 		return "Q2_K_S"
-	case FileTypeBF16:
+	case fileTypeIQ3_XS:
+		return "IQ3_XS"
+	case fileTypeIQ3_XXS:
+		return "IQ3_XXS"
+	case fileTypeIQ1_S:
+		return "IQ1_S"
+	case fileTypeIQ4_NL:
+		return "IQ4_NL"
+	case fileTypeIQ3_S:
+		return "IQ3_S"
+	case fileTypeIQ3_M:
+		return "IQ3_M"
+	case fileTypeIQ2_S:
+		return "IQ2_S"
+	case fileTypeIQ4_XS:
+		return "IQ4_XS"
+	case fileTypeIQ2_M:
+		return "IQ2_M"
+	case fileTypeIQ1_M:
+		return "IQ1_M"
+	case fileTypeBF16:
 		return "BF16"
 	default:
 		return "unknown"
 	}
 }

-func (t FileType) Value() uint32 {
+func (t fileType) Value() uint32 {
 	return uint32(t)
 }
-
-func (ftype FileType) ToTensorType() TensorType {
-	switch ftype {
-	case FileTypeF32:
-		return TensorTypeF32
-	case FileTypeF16:
-		return TensorTypeF16
-	case fileTypeQ4_0:
-		return TensorTypeQ4_0
-	case fileTypeQ4_1:
-		return TensorTypeQ4_1
-	case FileTypeQ8_0:
-		return TensorTypeQ8_0
-	case fileTypeQ5_0:
-		return TensorTypeQ5_0
-	case fileTypeQ5_1:
-		return TensorTypeQ5_1
-	case fileTypeQ2_K:
-		return TensorTypeQ2_K
-	case fileTypeQ3_K_S:
-		return TensorTypeQ3_K
-	case fileTypeQ3_K_M:
-		return TensorTypeQ3_K
-	case fileTypeQ3_K_L:
-		return TensorTypeQ3_K
-	case FileTypeQ4_K_S:
-		return TensorTypeQ4_K
-	case FileTypeQ4_K_M:
-		return TensorTypeQ4_K
-	case fileTypeQ5_K_S:
-		return TensorTypeQ5_K
-	case fileTypeQ5_K_M:
-		return TensorTypeQ5_K
-	case fileTypeQ6_K:
-		return TensorTypeQ6_K
-	case fileTypeQ2_K_S:
-		return TensorTypeQ2_K
-	case FileTypeBF16:
-		return TensorTypeBF16
-	default:
-		slog.Warn("unsupported file type", "type", ftype)
-		return 0 // F32
-	}
-}
-
-// TensorType is equivalent to ggml_type for individual tensor types
-// Note: these are not the same as FileType
-type TensorType uint32
-
-const (
-	TensorTypeF32 TensorType = iota
-	TensorTypeF16
-	TensorTypeQ4_0
-	TensorTypeQ4_1
-	tensorTypeQ4_2 // unused by GGML
-	tensorTypeQ4_3 // unused by GGML
-	TensorTypeQ5_0
-	TensorTypeQ5_1
-	TensorTypeQ8_0
-	TensorTypeQ8_1
-	TensorTypeQ2_K
-	TensorTypeQ3_K
-	TensorTypeQ4_K
-	TensorTypeQ5_K
-	TensorTypeQ6_K
-	TensorTypeQ8_K
-	tensorTypeIQ2_XXS // not supported by ollama
-	tensorTypeIQ2_XS  // not supported by ollama
-	tensorTypeIQ3_XXS // not supported by ollama
-	tensorTypeIQ1_S   // not supported by ollama
-	tensorTypeIQ4_NL  // not supported by ollama
-	tensorTypeIQ3_S   // not supported by ollama
-	tensorTypeIQ2_S   // not supported by ollama
-	tensorTypeIQ4_XS  // not supported by ollama
-	TensorTypeI8
-	TensorTypeI16
-	TensorTypeI32
-	TensorTypeI64
-	TensorTypeF64
-	tensorTypeIQ1_M // not supported by ollama
-	TensorTypeBF16
-	tensorTypeQ4_0_4_4   // unused by GGML
-	tensorTypeQ4_0_4_8   // unused by GGML
-	tensorTypeQ4_0_8_8   // unused by GGML
-	tensorTypeTQ1_0      // not supported by ollama
-	tensorTypeTQ2_0      // not supported by ollama
-	tensorTypeIQ4_NL_4_4 // unused by GGML
-	tensorTypeIQ4_NL_4_8 // unused by GGML
-	tensorTypeIQ4_NL_8_8 // unused by GGML
-)
-
-// ParseFileType parses the provided GGUF file type
-// Only Ollama supported types are considered valid
-func ParseTensorType(s string) (TensorType, error) {
-	switch s {
-	case "F32":
-		return TensorTypeF32, nil
-	case "F16":
-		return TensorTypeF16, nil
-	case "Q4_0":
-		return TensorTypeQ4_0, nil
-	case "Q4_1":
-		return TensorTypeQ4_1, nil
-	case "Q5_0":
-		return TensorTypeQ5_0, nil
-	case "Q5_1":
-		return TensorTypeQ5_1, nil
-	case "Q8_0":
-		return TensorTypeQ8_0, nil
-	case "Q8_1":
-		return TensorTypeQ8_1, nil
-	case "Q2_K":
-		return TensorTypeQ2_K, nil
-	case "Q3_K":
-		return TensorTypeQ3_K, nil
-	case "Q4_K":
-		return TensorTypeQ4_K, nil
-	case "Q5_K":
-		return TensorTypeQ5_K, nil
-	case "Q6_K":
-		return TensorTypeQ6_K, nil
-	case "Q8_K":
-		return TensorTypeQ8_K, nil
-	case "F64":
-		return TensorTypeF64, nil
-	case "BF16":
-		return TensorTypeBF16, nil
-	default:
-		return 0, fmt.Errorf("unsupported quantization type %s", s)
-	}
-}
-
-func (t TensorType) IsQuantized() bool {
-	switch t {
-	case TensorTypeF32, TensorTypeF16, TensorTypeBF16:
-		return false
-	default:
-		return true
-	}
-}
-
-func (t TensorType) RowSize(ne uint64) uint64 {
-	return t.TypeSize() * ne / t.BlockSize()
-}
-
-func (t TensorType) String() string {
-	switch t {
-	case TensorTypeF32:
-		return "F32"
-	case TensorTypeF16:
-		return "F16"
-	case TensorTypeQ4_0:
-		return "Q4_0"
-	case TensorTypeQ4_1:
-		return "Q4_1"
-	case TensorTypeQ5_0:
-		return "Q5_0"
-	case TensorTypeQ5_1:
-		return "Q5_1"
-	case TensorTypeQ8_0:
-		return "Q8_0"
-	case TensorTypeQ8_1:
-		return "Q8_1"
-	case TensorTypeQ2_K:
-		return "Q2_K"
-	case TensorTypeQ3_K:
-		return "Q3_K"
-	case TensorTypeQ4_K:
-		return "Q4_K"
-	case TensorTypeQ5_K:
-		return "Q5_K"
-	case TensorTypeQ6_K:
-		return "Q6_K"
-	case TensorTypeQ8_K:
-		return "Q8_K"
-	case TensorTypeF64:
-		return "F64"
-	case TensorTypeBF16:
-		return "BF16"
-	default:
-		return "unknown"
-	}
-}
--- a/go.mod
+++ b/go.mod
@@ -11,7 +11,7 @@ require (
 	github.com/spf13/cobra v1.7.0
 	github.com/stretchr/testify v1.9.0
 	github.com/x448/float16 v0.8.4
-	golang.org/x/sync v0.12.0
+	golang.org/x/sync v0.11.0
 )

 require (
@@ -70,12 +70,12 @@ require (
 	github.com/twitchyliquid64/golang-asm v0.15.1 // indirect
 	github.com/ugorji/go/codec v1.2.12 // indirect
 	golang.org/x/arch v0.8.0 // indirect
-	golang.org/x/crypto v0.36.0
+	golang.org/x/crypto v0.33.0
 	golang.org/x/exp v0.0.0-20250218142911-aa4b98e5adaa
-	golang.org/x/net v0.38.0 // indirect
-	golang.org/x/sys v0.31.0
-	golang.org/x/term v0.30.0
-	golang.org/x/text v0.23.0
+	golang.org/x/net v0.35.0 // indirect
+	golang.org/x/sys v0.30.0
+	golang.org/x/term v0.29.0
+	golang.org/x/text v0.22.0
 	google.golang.org/protobuf v1.34.1
 	gopkg.in/yaml.v3 v3.0.1 // indirect
 )
--- a/go.sum
+++ b/go.sum
@@ -214,8 +214,8 @@ golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACk
 golang.org/x/crypto v0.0.0-20190510104115-cbcb75029529/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
 golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
 golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto=
-golang.org/x/crypto v0.36.0 h1:AnAEvhDddvBdpY+uR+MyHmuZzzNqXSe/GvuDeob5L34=
-golang.org/x/crypto v0.36.0/go.mod h1:Y4J0ReaxCR1IMaabaSMugxJES1EpwhBHhv2bDHklZvc=
+golang.org/x/crypto v0.33.0 h1:IOBPskki6Lysi0lo9qQvbxiQ+FvsCC/YWOecCHAixus=
+golang.org/x/crypto v0.33.0/go.mod h1:bVdXmD7IV/4GdElGPozy6U7lWdRXA4qyRVGJV57uQ5M=
 golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
 golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
 golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
@@ -257,8 +257,8 @@ golang.org/x/net v0.0.0-20200822124328-c89045814202/go.mod h1:/O7V0waA8r7cgGh81R
 golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU=
 golang.org/x/net v0.0.0-20210405180319-a5a99cb37ef4/go.mod h1:p54w0d4576C0XHj96bSt6lcn1PtDYWL6XObtHCRCNQM=
 golang.org/x/net v0.0.0-20210614182718-04defd469f4e/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y=
-golang.org/x/net v0.38.0 h1:vRMAPTMaeGqVhG5QyLJHqNDwecKTomGeqbnfZyKlBI8=
-golang.org/x/net v0.38.0/go.mod h1:ivrbrMbzFq5J41QOQh0siUuly180yBYtLp+CKbEaFx8=
+golang.org/x/net v0.35.0 h1:T5GQRQb2y08kTAByq9L4/bz8cipCdA8FbRTXewonqY8=
+golang.org/x/net v0.35.0/go.mod h1:EglIi67kWsHKlRzzVMUD93VMSWGFOMSZgxFjparz1Qk=
 golang.org/x/oauth2 v0.0.0-20180821212333-d2e6202438be/go.mod h1:N/0e6XlmueqKjAGxoOufVs8QHGRruUQn6yWY3a++T0U=
 golang.org/x/oauth2 v0.0.0-20200107190931-bf48bf16ab8d/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
 golang.org/x/sync v0.0.0-20180314180146-1d60e4601c6f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
@@ -268,8 +268,8 @@ golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJ
 golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
 golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
 golang.org/x/sync v0.0.0-20210220032951-036812b2e83c/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
-golang.org/x/sync v0.12.0 h1:MHc5BpPuC30uJk597Ri8TV3CNZcTLu6B6z4lJy+g6Jw=
-golang.org/x/sync v0.12.0/go.mod h1:1dzgHSNfp02xaA81J2MS99Qcpr2w7fw1gpm99rleRqA=
+golang.org/x/sync v0.11.0 h1:GGz8+XQP4FvTTrjZPzNKTMFtSXH80RAzG+5ghFPgK9w=
+golang.org/x/sync v0.11.0/go.mod h1:Czt+wKu1gCyEFDUtn0jG5QVvpJ6rzVqr5aXyt9drQfk=
 golang.org/x/sys v0.0.0-20180830151530-49385e6e1522/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
 golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
 golang.org/x/sys v0.0.0-20190312061237-fead79001313/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
@@ -285,17 +285,17 @@ golang.org/x/sys v0.0.0-20210510120138-977fb7262007/go.mod h1:oPkhp1MJrh7nUepCBc
 golang.org/x/sys v0.0.0-20210630005230-0f9fa26af87c/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
-golang.org/x/sys v0.31.0 h1:ioabZlmFYtWhL+TRYpcnNlLwhyxaM9kWTDEmfnprqik=
-golang.org/x/sys v0.31.0/go.mod h1:BJP2sWEmIv4KK5OTEluFJCKSidICx8ciO85XgH3Ak8k=
+golang.org/x/sys v0.30.0 h1:QjkSwP/36a20jFYWkSue1YwXzLmsV5Gfq7Eiy72C1uc=
+golang.org/x/sys v0.30.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
 golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
-golang.org/x/term v0.30.0 h1:PQ39fJZ+mfadBm0y5WlL4vlM7Sx1Hgf13sMIY2+QS9Y=
-golang.org/x/term v0.30.0/go.mod h1:NYYFdzHoI5wRh/h5tDMdMqCqPJZEuNqVR5xJLd/n67g=
+golang.org/x/term v0.29.0 h1:L6pJp37ocefwRRtYPKSWOWzOtWSxVajvz2ldH/xi3iU=
+golang.org/x/term v0.29.0/go.mod h1:6bl4lRlvVuDgSf3179VpIxBF0o10JUpXWOnI7nErv7s=
 golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
 golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
 golang.org/x/text v0.3.5/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
 golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
-golang.org/x/text v0.23.0 h1:D71I7dUrlY+VX0gQShAThNGHFxZ13dGLBHQLVl1mJlY=
-golang.org/x/text v0.23.0/go.mod h1:/BLNzu4aZCJ1+kcD0DNRotWKage4q2rGVAg4o22unh4=
+golang.org/x/text v0.22.0 h1:bofq7m3/HAFvbF51jz3Q9wLg3jkvSPuiZu/pD1XwgtM=
+golang.org/x/text v0.22.0/go.mod h1:YRoo4H8PVmsu+E3Ou7cqLVH8oXWIHVoX0jqUWALQhfY=
 golang.org/x/tools v0.0.0-20180525024113-a5b4c53f6e8b/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
 golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
 golang.org/x/tools v0.0.0-20190114222345-bf090417da8b/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
--- a/grammar/bench_test.go
+++ b/grammar/bench_test.go
@@ -0,0 +1,22 @@
+//go:build go1.24
+
+package grammar
+
+import "testing"
+
+func BenchmarkFromSchema(b *testing.B) {
+	for tt := range testCases(b) {
+		b.Run("", func(b *testing.B) {
+			s := []byte(tt.schema)
+
+			b.ReportAllocs()
+			for b.Loop() {
+				_, err := FromSchema(nil, s)
+				if err != nil {
+					b.Fatalf("GrammarFromSchema: %v", err)
+				}
+			}
+		})
+		return
+	}
+}
--- a/grammar/grammar.go
+++ b/grammar/grammar.go
@@ -0,0 +1,227 @@
+package grammar
+
+import (
+	"bytes"
+	"encoding/json"
+	"fmt"
+	"iter"
+	"strconv"
+
+	"github.com/ollama/ollama/grammar/jsonschema"
+)
+
+const jsonTerms = `
+# Unicode
+#
+# Unicode characters can be specified directly in the grammar, for example
+# hiragana ::= [ぁ-ゟ], or with escapes: 8-bit (\xXX), 16-bit (\uXXXX) or 32-bit
+# (\UXXXXXXXX).
+unicode ::= \x{hex}{2} | \u{hex}{4} | \U{hex}{8}
+
+# JSON grammar from RFC 7159
+null    ::= "null"
+object  ::= "{" (kv ("," kv)*)? "}"
+array   ::= "[" (value ("," value)*)? "]"
+kv      ::= string ":" value
+integer ::= "0" | [1-9] [0-9]*
+number  ::= "-"? integer frac? exp?
+frac    ::= "." [0-9]+
+exp     ::= ("e" | "E") ("+" | "-") [0-9]+
+string  ::= "\"" char* "\""
+escape  ::= ["/" | "b" | "f" | "n" | "r" | "t" | unicode]
+char    ::= [^"\\] | escape
+space   ::= (" " | "\t" | "\n" | "\r")*
+hex     ::= [0-9] | [a-f] | [A-F]
+boolean ::= "true" | "false"
+value   ::= object | array | string | number | boolean | "null"
+
+# User-defined
+`
+
+// FromSchema generates a grammar from a JSON schema.
+func FromSchema(buf []byte, jsonSchema []byte) ([]byte, error) {
+	var s *jsonschema.Schema
+	if err := json.Unmarshal(jsonSchema, &s); err != nil {
+		return nil, err
+	}
+
+	var g builder
+
+	// "root" is the only rule that is guaranteed to exist, so we start
+	// with its length for padding, and then adjust it as we go.
+	g.pad = len("root")
+	for id := range dependencies("root", s) {
+		g.pad = max(g.pad, len(id))
+	}
+
+	g.b.WriteString(jsonTerms)
+
+	ids := make(map[*jsonschema.Schema]string)
+	for id, s := range dependencies("root", s) {
+		ids[s] = id
+		g.define(id)
+		if err := fromSchema(&g, ids, s); err != nil {
+			return nil, err
+		}
+	}
+	g.define("root")
+	if err := fromSchema(&g, ids, s); err != nil {
+		return nil, err
+	}
+	g.define("") // finalize the last rule
+	return g.b.Bytes(), nil
+}
+
+func fromSchema(g *builder, ids map[*jsonschema.Schema]string, s *jsonschema.Schema) error {
+	switch typ := s.EffectiveType(); typ {
+	case "array":
+		if len(s.PrefixItems) == 0 && s.Items == nil {
+			g.u("array")
+		} else {
+			g.q("[")
+			for i, s := range s.PrefixItems {
+				if i > 0 {
+					g.q(",")
+				}
+				g.u(ids[s])
+			}
+			if s.Items != nil {
+				g.u("(")
+				if len(s.PrefixItems) > 0 {
+					g.q(",")
+				}
+				g.u(ids[s.Items])
+				g.u(")*")
+			}
+			g.q("]")
+		}
+	case "object":
+		if len(s.Properties) == 0 {
+			g.u("object")
+		} else {
+			g.q("{")
+			for i, p := range s.Properties {
+				name := ids[p]
+				if i > 0 {
+					g.q(",")
+				}
+				g.q(p.Name)
+				g.q(":")
+				g.u(name)
+			}
+			g.q("}")
+		}
+	case "number":
+		buildConstrainedNumber(g, s)
+	case "string":
+		if len(s.Enum) == 0 {
+			g.u("string")
+		} else {
+			g.u("(")
+			for i, e := range s.Enum {
+				if i > 0 {
+					g.q("|")
+				}
+				g.q(string(e))
+			}
+			g.u(")")
+		}
+	case "boolean", "value", "null", "integer":
+		g.u(typ)
+	default:
+		return fmt.Errorf("%s: unsupported type %q", s.Name, typ)
+	}
+	return nil
+}
+
+// dependencies returns a sequence of all child dependencies of the schema in
+// post-order.
+//
+// The first value is the id/pointer to the dependency, and the second value
+// is the schema.
+func dependencies(id string, s *jsonschema.Schema) iter.Seq2[string, *jsonschema.Schema] {
+	return func(yield func(string, *jsonschema.Schema) bool) {
+		for i, p := range s.Properties {
+			id := fmt.Sprintf("%s_%d", id, i)
+			for did, d := range dependencies(id, p) {
+				if !yield(did, d) {
+					return
+				}
+			}
+			if !yield(id, p) {
+				return
+			}
+		}
+		for i, p := range s.PrefixItems {
+			id := fmt.Sprintf("tuple_%d", i)
+			for did, d := range dependencies(id, p) {
+				id := fmt.Sprintf("%s_%s", id, did)
+				if !yield(id, d) {
+					return
+				}
+			}
+			if !yield(id, p) {
+				return
+			}
+		}
+		if s.Items != nil {
+			id := fmt.Sprintf("%s_tuple_%d", id, len(s.PrefixItems))
+			for did, d := range dependencies(id, s.Items) {
+				if !yield(did, d) {
+					return
+				}
+			}
+			if !yield(id, s.Items) {
+				return
+			}
+		}
+	}
+}
+
+type builder struct {
+	b     bytes.Buffer
+	pad   int
+	rules int
+	items int
+}
+
+// define terminates the current rule, if any, and then either starts a new
+// rule or does nothing else if the name is empty.
+func (b *builder) define(name string) {
+	if b.rules > 0 {
+		b.b.WriteString(";\n")
+	}
+	if name == "" {
+		return
+	}
+	fmt.Fprintf(&b.b, "% -*s", b.pad, name)
+	b.b.WriteString(" ::=")
+	b.rules++
+	b.items = 0
+}
+
+// quote appends a terminal to the current rule.
+func (b *builder) q(s string) {
+	if b.items > 0 {
+		b.b.WriteString(" ")
+	}
+	b.b.WriteString(" ")
+	b.b.WriteString(strconv.Quote(s))
+}
+
+// u appends a non-terminal to the current rule.
+func (b *builder) u(s string) {
+	if b.items > 0 {
+		b.b.WriteString(" ")
+	}
+	b.b.WriteString(" ")
+	b.b.WriteString(s)
+}
+
+func buildConstrainedNumber(b *builder, s *jsonschema.Schema) {
+	if s.Minimum == 0 && s.Maximum == 0 {
+		b.u("TODO")
+	} else {
+		b.u("number")
+	}
+}
--- a/grammar/grammar_test.go
+++ b/grammar/grammar_test.go
@@ -0,0 +1,75 @@
+package grammar
+
+import (
+	"bufio"
+	"cmp"
+	"iter"
+	"strings"
+	"testing"
+
+	_ "embed"
+
+	"github.com/ollama/ollama/grammar/internal/diff"
+)
+
+func TestFromSchema(t *testing.T) {
+	for tt := range testCases(t) {
+		t.Run(tt.name, func(t *testing.T) {
+			g, err := FromSchema(nil, []byte(tt.schema))
+			if err != nil {
+				t.Fatalf("FromSchema: %v", err)
+			}
+			got := string(g)
+			got = strings.TrimPrefix(got, jsonTerms)
+			if got != tt.want {
+				t.Logf("schema:\n%s", tt.schema)
+				t.Fatal(string(diff.Diff("got", []byte(got), "want", []byte(tt.want))))
+			}
+		})
+	}
+}
+
+type testCase struct {
+	name   string
+	schema string
+	want   string
+}
+
+//go:embed testdata/schemas.txt
+var tests string
+
+func testCases(t testing.TB) iter.Seq[testCase] {
+	t.Helper()
+	return func(yield func(testCase) bool) {
+		t.Helper()
+		sc := bufio.NewScanner(strings.NewReader(tests))
+		name := ""
+		for sc.Scan() {
+			line := strings.TrimSpace(sc.Text())
+			if line == "" {
+				name = ""
+				continue
+			}
+			if line[0] == '#' {
+				name = cmp.Or(name, strings.TrimSpace(line[1:]))
+				continue
+			}
+			s := sc.Text()
+			g := ""
+			for sc.Scan() {
+				line = strings.TrimSpace(sc.Text())
+				if line == "" || line[0] == '#' {
+					break
+				}
+				g += sc.Text() + "\n"
+			}
+			if !yield(testCase{name, s, g}) {
+				return
+			}
+			name = strings.TrimSpace(strings.TrimPrefix(line, "#"))
+		}
+		if err := sc.Err(); err != nil {
+			t.Fatalf("error reading tests: %v", err)
+		}
+	}
+}
--- a/grammar/internal/diff/diff.go
+++ b/grammar/internal/diff/diff.go
@@ -0,0 +1,261 @@
+// Copyright 2022 The Go Authors. All rights reserved.
+// Use of this source code is governed by a BSD-style
+// license that can be found in the LICENSE file.
+
+package diff
+
+import (
+	"bytes"
+	"fmt"
+	"sort"
+	"strings"
+)
+
+// A pair is a pair of values tracked for both the x and y side of a diff.
+// It is typically a pair of line indexes.
+type pair struct{ x, y int }
+
+// Diff returns an anchored diff of the two texts old and new
+// in the “unified diff” format. If old and new are identical,
+// Diff returns a nil slice (no output).
+//
+// Unix diff implementations typically look for a diff with
+// the smallest number of lines inserted and removed,
+// which can in the worst case take time quadratic in the
+// number of lines in the texts. As a result, many implementations
+// either can be made to run for a long time or cut off the search
+// after a predetermined amount of work.
+//
+// In contrast, this implementation looks for a diff with the
+// smallest number of “unique” lines inserted and removed,
+// where unique means a line that appears just once in both old and new.
+// We call this an “anchored diff” because the unique lines anchor
+// the chosen matching regions. An anchored diff is usually clearer
+// than a standard diff, because the algorithm does not try to
+// reuse unrelated blank lines or closing braces.
+// The algorithm also guarantees to run in O(n log n) time
+// instead of the standard O(n²) time.
+//
+// Some systems call this approach a “patience diff,” named for
+// the “patience sorting” algorithm, itself named for a solitaire card game.
+// We avoid that name for two reasons. First, the name has been used
+// for a few different variants of the algorithm, so it is imprecise.
+// Second, the name is frequently interpreted as meaning that you have
+// to wait longer (to be patient) for the diff, meaning that it is a slower algorithm,
+// when in fact the algorithm is faster than the standard one.
+func Diff(oldName string, old []byte, newName string, new []byte) []byte {
+	if bytes.Equal(old, new) {
+		return nil
+	}
+	x := lines(old)
+	y := lines(new)
+
+	// Print diff header.
+	var out bytes.Buffer
+	fmt.Fprintf(&out, "diff %s %s\n", oldName, newName)
+	fmt.Fprintf(&out, "--- %s\n", oldName)
+	fmt.Fprintf(&out, "+++ %s\n", newName)
+
+	// Loop over matches to consider,
+	// expanding each match to include surrounding lines,
+	// and then printing diff chunks.
+	// To avoid setup/teardown cases outside the loop,
+	// tgs returns a leading {0,0} and trailing {len(x), len(y)} pair
+	// in the sequence of matches.
+	var (
+		done  pair     // printed up to x[:done.x] and y[:done.y]
+		chunk pair     // start lines of current chunk
+		count pair     // number of lines from each side in current chunk
+		ctext []string // lines for current chunk
+	)
+	for _, m := range tgs(x, y) {
+		if m.x < done.x {
+			// Already handled scanning forward from earlier match.
+			continue
+		}
+
+		// Expand matching lines as far as possible,
+		// establishing that x[start.x:end.x] == y[start.y:end.y].
+		// Note that on the first (or last) iteration we may (or definitely do)
+		// have an empty match: start.x==end.x and start.y==end.y.
+		start := m
+		for start.x > done.x && start.y > done.y && x[start.x-1] == y[start.y-1] {
+			start.x--
+			start.y--
+		}
+		end := m
+		for end.x < len(x) && end.y < len(y) && x[end.x] == y[end.y] {
+			end.x++
+			end.y++
+		}
+
+		// Emit the mismatched lines before start into this chunk.
+		// (No effect on first sentinel iteration, when start = {0,0}.)
+		for _, s := range x[done.x:start.x] {
+			ctext = append(ctext, "-"+s)
+			count.x++
+		}
+		for _, s := range y[done.y:start.y] {
+			ctext = append(ctext, "+"+s)
+			count.y++
+		}
+
+		// If we're not at EOF and have too few common lines,
+		// the chunk includes all the common lines and continues.
+		const C = 3 // number of context lines
+		if (end.x < len(x) || end.y < len(y)) &&
+			(end.x-start.x < C || (len(ctext) > 0 && end.x-start.x < 2*C)) {
+			for _, s := range x[start.x:end.x] {
+				ctext = append(ctext, " "+s)
+				count.x++
+				count.y++
+			}
+			done = end
+			continue
+		}
+
+		// End chunk with common lines for context.
+		if len(ctext) > 0 {
+			n := end.x - start.x
+			if n > C {
+				n = C
+			}
+			for _, s := range x[start.x : start.x+n] {
+				ctext = append(ctext, " "+s)
+				count.x++
+				count.y++
+			}
+			done = pair{start.x + n, start.y + n}
+
+			// Format and emit chunk.
+			// Convert line numbers to 1-indexed.
+			// Special case: empty file shows up as 0,0 not 1,0.
+			if count.x > 0 {
+				chunk.x++
+			}
+			if count.y > 0 {
+				chunk.y++
+			}
+			fmt.Fprintf(&out, "@@ -%d,%d +%d,%d @@\n", chunk.x, count.x, chunk.y, count.y)
+			for _, s := range ctext {
+				out.WriteString(s)
+			}
+			count.x = 0
+			count.y = 0
+			ctext = ctext[:0]
+		}
+
+		// If we reached EOF, we're done.
+		if end.x >= len(x) && end.y >= len(y) {
+			break
+		}
+
+		// Otherwise start a new chunk.
+		chunk = pair{end.x - C, end.y - C}
+		for _, s := range x[chunk.x:end.x] {
+			ctext = append(ctext, " "+s)
+			count.x++
+			count.y++
+		}
+		done = end
+	}
+
+	return out.Bytes()
+}
+
+// lines returns the lines in the file x, including newlines.
+// If the file does not end in a newline, one is supplied
+// along with a warning about the missing newline.
+func lines(x []byte) []string {
+	l := strings.SplitAfter(string(x), "\n")
+	if l[len(l)-1] == "" {
+		l = l[:len(l)-1]
+	} else {
+		// Treat last line as having a message about the missing newline attached,
+		// using the same text as BSD/GNU diff (including the leading backslash).
+		l[len(l)-1] += "\n\\ No newline at end of file\n"
+	}
+	return l
+}
+
+// tgs returns the pairs of indexes of the longest common subsequence
+// of unique lines in x and y, where a unique line is one that appears
+// once in x and once in y.
+//
+// The longest common subsequence algorithm is as described in
+// Thomas G. Szymanski, “A Special Case of the Maximal Common
+// Subsequence Problem,” Princeton TR #170 (January 1975),
+// available at https://research.swtch.com/tgs170.pdf.
+func tgs(x, y []string) []pair {
+	// Count the number of times each string appears in a and b.
+	// We only care about 0, 1, many, counted as 0, -1, -2
+	// for the x side and 0, -4, -8 for the y side.
+	// Using negative numbers now lets us distinguish positive line numbers later.
+	m := make(map[string]int)
+	for _, s := range x {
+		if c := m[s]; c > -2 {
+			m[s] = c - 1
+		}
+	}
+	for _, s := range y {
+		if c := m[s]; c > -8 {
+			m[s] = c - 4
+		}
+	}
+
+	// Now unique strings can be identified by m[s] = -1+-4.
+	//
+	// Gather the indexes of those strings in x and y, building:
+	//	xi[i] = increasing indexes of unique strings in x.
+	//	yi[i] = increasing indexes of unique strings in y.
+	//	inv[i] = index j such that x[xi[i]] = y[yi[j]].
+	var xi, yi, inv []int
+	for i, s := range y {
+		if m[s] == -1+-4 {
+			m[s] = len(yi)
+			yi = append(yi, i)
+		}
+	}
+	for i, s := range x {
+		if j, ok := m[s]; ok && j >= 0 {
+			xi = append(xi, i)
+			inv = append(inv, j)
+		}
+	}
+
+	// Apply Algorithm A from Szymanski's paper.
+	// In those terms, A = J = inv and B = [0, n).
+	// We add sentinel pairs {0,0}, and {len(x),len(y)}
+	// to the returned sequence, to help the processing loop.
+	J := inv
+	n := len(xi)
+	T := make([]int, n)
+	L := make([]int, n)
+	for i := range T {
+		T[i] = n + 1
+	}
+	for i := range n {
+		k := sort.Search(n, func(k int) bool {
+			return T[k] >= J[i]
+		})
+		T[k] = J[i]
+		L[i] = k + 1
+	}
+	k := 0
+	for _, v := range L {
+		if k < v {
+			k = v
+		}
+	}
+	seq := make([]pair, 2+k)
+	seq[1+k] = pair{len(x), len(y)} // sentinel at end
+	lastj := n
+	for i := n - 1; i >= 0; i-- {
+		if L[i] == k && J[i] < lastj {
+			seq[k] = pair{xi[i], yi[J[i]]}
+			k--
+		}
+	}
+	seq[0] = pair{0, 0} // sentinel at start
+	return seq
+}
--- a/grammar/internal/diff/diff_test.go
+++ b/grammar/internal/diff/diff_test.go
@@ -0,0 +1,44 @@
+// Copyright 2022 The Go Authors. All rights reserved.
+// Use of this source code is governed by a BSD-style
+// license that can be found in the LICENSE file.
+
+package diff
+
+import (
+	"bytes"
+	"path/filepath"
+	"testing"
+
+	"golang.org/x/tools/txtar"
+)
+
+func clean(text []byte) []byte {
+	text = bytes.ReplaceAll(text, []byte("$\n"), []byte("\n"))
+	text = bytes.TrimSuffix(text, []byte("^D\n"))
+	return text
+}
+
+func Test(t *testing.T) {
+	files, _ := filepath.Glob("testdata/*.txt")
+	if len(files) == 0 {
+		t.Fatalf("no testdata")
+	}
+
+	for _, file := range files {
+		t.Run(filepath.Base(file), func(t *testing.T) {
+			a, err := txtar.ParseFile(file)
+			if err != nil {
+				t.Fatal(err)
+			}
+			if len(a.Files) != 3 || a.Files[2].Name != "diff" {
+				t.Fatalf("%s: want three files, third named \"diff\"", file)
+			}
+			diffs := Diff(a.Files[0].Name, clean(a.Files[0].Data), a.Files[1].Name, clean(a.Files[1].Data))
+			want := clean(a.Files[2].Data)
+			if !bytes.Equal(diffs, want) {
+				t.Fatalf("%s: have:\n%s\nwant:\n%s\n%s", file,
+					diffs, want, Diff("have", diffs, "want", want))
+			}
+		})
+	}
+}
--- a/grammar/internal/diff/testdata/allnew.txt
+++ b/grammar/internal/diff/testdata/allnew.txt
@@ -0,0 +1,13 @@
+-- old --
+-- new --
+a
+b
+c
+-- diff --
+diff old new
+--- old
+++ new
+@@ -0,0 +1,3 @@
+a
+b
+c
--- a/grammar/internal/diff/testdata/allold.txt
+++ b/grammar/internal/diff/testdata/allold.txt
@@ -0,0 +1,13 @@
+-- old --
+a
+b
+c
+-- new --
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,3 +0,0 @@
+-a
+-b
+-c
--- a/grammar/internal/diff/testdata/basic.txt
+++ b/grammar/internal/diff/testdata/basic.txt
@@ -0,0 +1,35 @@
+Example from Hunt and McIlroy, “An Algorithm for Differential File Comparison.”
+https://www.cs.dartmouth.edu/~doug/diff.pdf
+
+-- old --
+a
+b
+c
+d
+e
+f
+g
+-- new --
+w
+a
+b
+x
+y
+z
+e
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,7 +1,7 @@
+w
+ a
+ b
+-c
+-d
+x
+y
+z
+ e
+-f
+-g
--- a/grammar/internal/diff/testdata/dups.txt
+++ b/grammar/internal/diff/testdata/dups.txt
@@ -0,0 +1,40 @@
+-- old --
+a
+
+b
+
+c
+
+d
+
+e
+
+f
+-- new --
+a
+
+B
+
+C
+
+d
+
+e
+
+f
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,8 +1,8 @@
+ a
+ $
+-b
+-
+-c
+B
+
+C
+ $
+ d
+ $
--- a/grammar/internal/diff/testdata/end.txt
+++ b/grammar/internal/diff/testdata/end.txt
@@ -0,0 +1,38 @@
+-- old --
+1
+2
+3
+4
+5
+6
+7
+eight
+nine
+ten
+eleven
+-- new --
+1
+2
+3
+4
+5
+6
+7
+8
+9
+10
+-- diff --
+diff old new
+--- old
+++ new
+@@ -5,7 +5,6 @@
+ 5
+ 6
+ 7
+-eight
+-nine
+-ten
+-eleven
+8
+9
+10
--- a/grammar/internal/diff/testdata/eof.txt
+++ b/grammar/internal/diff/testdata/eof.txt
@@ -0,0 +1,9 @@
+-- old --
+a
+b
+c^D
+-- new --
+a
+b
+c^D
+-- diff --
--- a/grammar/internal/diff/testdata/eof1.txt
+++ b/grammar/internal/diff/testdata/eof1.txt
@@ -0,0 +1,18 @@
+-- old --
+a
+b
+c
+-- new --
+a
+b
+c^D
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,3 +1,3 @@
+ a
+ b
+-c
+c
+\ No newline at end of file
--- a/grammar/internal/diff/testdata/eof2.txt
+++ b/grammar/internal/diff/testdata/eof2.txt
@@ -0,0 +1,18 @@
+-- old --
+a
+b
+c^D
+-- new --
+a
+b
+c
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,3 +1,3 @@
+ a
+ b
+-c
+\ No newline at end of file
+c
--- a/grammar/internal/diff/testdata/long.txt
+++ b/grammar/internal/diff/testdata/long.txt
@@ -0,0 +1,62 @@
+-- old --
+1
+2
+3
+4
+5
+6
+7
+8
+9
+10
+11
+12
+13
+14
+14½
+15
+16
+17
+18
+19
+20
+-- new --
+1
+2
+3
+4
+5
+6
+8
+9
+10
+11
+12
+13
+14
+17
+18
+19
+20
+-- diff --
+diff old new
+--- old
+++ new
+@@ -4,7 +4,6 @@
+ 4
+ 5
+ 6
+-7
+ 8
+ 9
+ 10
+@@ -12,9 +11,6 @@
+ 12
+ 13
+ 14
+-14½
+-15
+-16
+ 17
+ 18
+ 19
--- a/grammar/internal/diff/testdata/same.txt
+++ b/grammar/internal/diff/testdata/same.txt
@@ -0,0 +1,5 @@
+-- old --
+hello world
+-- new --
+hello world
+-- diff --
--- a/grammar/internal/diff/testdata/start.txt
+++ b/grammar/internal/diff/testdata/start.txt
@@ -0,0 +1,34 @@
+-- old --
+e
+pi
+4
+5
+6
+7
+8
+9
+10
+-- new --
+1
+2
+3
+4
+5
+6
+7
+8
+9
+10
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,5 +1,6 @@
+-e
+-pi
+1
+2
+3
+ 4
+ 5
+ 6
--- a/grammar/internal/diff/testdata/triv.txt
+++ b/grammar/internal/diff/testdata/triv.txt
@@ -0,0 +1,40 @@
+Another example from Hunt and McIlroy,
+“An Algorithm for Differential File Comparison.”
+https://www.cs.dartmouth.edu/~doug/diff.pdf
+
+Anchored diff gives up on finding anything,
+since there are no unique lines.
+
+-- old --
+a
+b
+c
+a
+b
+b
+a
+-- new --
+c
+a
+b
+a
+b
+c
+-- diff --
+diff old new
+--- old
+++ new
+@@ -1,7 +1,6 @@
+-a
+-b
+-c
+-a
+-b
+-b
+-a
+c
+a
+b
+a
+b
+c
--- a/grammar/jsonschema/decode.go
+++ b/grammar/jsonschema/decode.go
@@ -0,0 +1,171 @@
+package jsonschema
+
+import (
+	"bytes"
+	"encoding/json"
+	"errors"
+)
+
+// Schema holds a JSON schema.
+type Schema struct {
+	// Name is the name of the property. For the parent/root property, this
+	// is "root". For child properties, this is the name of the property.
+	Name string `json:"-"`
+
+	// Type is the type of the property.
+	//
+	// TODO: Union types (e.g. make this a []string).
+	Type string
+
+	// PrefixItems is a list of schemas for each item in a tuple. By
+	// default, the tuple is "closed." unless Items is set to true or a
+	// valid Schema.
+	PrefixItems []*Schema
+
+	// Items is the schema for each item in a list.
+	//
+	// If it is missing, or its JSON value is "null" or "false", it is nil.
+	// If the JSON value is "true", it is set to the empty Schema. If the
+	// JSON value is an object, it will be decoded as a Schema.
+	Items *Schema
+
+	// MinItems specifies the minimum number of items allowed in a list.
+	MinItems int
+
+	// MaxItems specifies the maximum number of items allowed in a list.
+	MaxItems int
+
+	// Properties is the schema for each property of an object.
+	Properties []*Schema
+
+	// Format is the format of the property. This is used to validate the
+	// property against a specific format.
+	//
+	// It is the callers responsibility to validate the property against
+	// the format.
+	Format string
+
+	// Minimum specifies the minimum value for numeric properties.
+	Minimum float64
+
+	// Maximum specifies the maximum value for numeric properties.
+	Maximum float64
+
+	// Enum is a list of valid values for the property.
+	Enum []json.RawMessage
+}
+
+func (s *Schema) UnmarshalJSON(data []byte) error {
+	type S Schema
+	w := struct {
+		Properties props
+		Items      items
+		*S
+	}{
+		S: (*S)(s),
+	}
+	if err := json.Unmarshal(data, &w); err != nil {
+		return err
+	}
+	if w.Items.set {
+		s.Items = &w.Items.Schema
+	}
+	s.Properties = w.Properties
+	return nil
+}
+
+type items struct {
+	Schema
+	set bool
+}
+
+func (s *items) UnmarshalJSON(data []byte) error {
+	switch b := data[0]; b {
+	case 't':
+		*s = items{set: true}
+	case '{':
+		type I items
+		if err := json.Unmarshal(data, (*I)(s)); err != nil {
+			return err
+		}
+		s.set = true
+	case 'n', 'f':
+	default:
+		return errors.New("invalid Items")
+	}
+	return nil
+}
+
+// EffectiveType returns the effective type of the schema. If the Type field is
+// not empty, it is returned; otherwise:
+//
+//   - If the schema has both Properties and Items, it returns an empty string.
+//   - If the schema has Properties, it returns "object".
+//   - If the schema has Items, it returns "array".
+//   - If the schema has neither Properties nor Items, it returns "value".
+//
+// The returned string is never empty.
+func (d *Schema) EffectiveType() string {
+	if d.Type == "" {
+		if len(d.Properties) > 0 {
+			return "object"
+		}
+		if len(d.PrefixItems) > 0 || d.Items != nil {
+			return "array"
+		}
+		return "value"
+	}
+	return d.Type
+}
+
+// props is an ordered list of properties. The order of the properties
+// is the order in which they were defined in the schema.
+type props []*Schema
+
+var _ json.Unmarshaler = (*props)(nil)
+
+func (v *props) UnmarshalJSON(data []byte) error {
+	if len(data) == 0 {
+		return nil
+	}
+	if data[0] != '{' {
+		return errors.New("expected object")
+	}
+
+	d := json.NewDecoder(bytes.NewReader(data))
+
+	// TODO(bmizerany): Consider DisallowUnknownFields. Currently, we, like
+	// llama.cpp, ignore unknown fields, which could be lead to unexpected
+	// behavior for clients of this package, since they may not be aware
+	// that "additionalFields", "itemsPrefix", etc, are being ignored.
+	//
+	// For now, just do what llama.cpp does.
+
+	t, err := d.Token()
+	if err != nil {
+		return err
+	}
+	if t != json.Delim('{') {
+		return errors.New("expected object")
+	}
+	for d.More() {
+		// Use the first token (map key) as the property name, then
+		// decode the rest of the object fields into a Schema and
+		// append.
+		t, err := d.Token()
+		if err != nil {
+			return err
+		}
+		if t == json.Delim('}') {
+			return nil
+		}
+		s := &Schema{
+			Name: t.(string),
+		}
+		if err := d.Decode(s); err != nil {
+			return err
+		}
+		*v = append(*v, s)
+	}
+	return nil
+}
--- a/grammar/jsonschema/decode_test.go
+++ b/grammar/jsonschema/decode_test.go
@@ -0,0 +1,104 @@
+package jsonschema
+
+import (
+	"encoding/json"
+	"reflect"
+	"strings"
+	"testing"
+
+	"github.com/google/go-cmp/cmp"
+)
+
+const testSchemaBasic = `
+{
+  "properties": {
+    "tupleClosedEmpty":   { "prefixItems": [] },
+    "tupleClosedMissing": { "prefixItems": [{}] },
+    "tupleClosedNull":    { "prefixItems": [{}], "items": null },
+    "tupleClosedFalse":   { "prefixItems": [{}], "items": false },
+    "tupleOpenTrue":      { "prefixItems": [{}], "items": true },
+    "tupleOpenEmpty":     { "prefixItems": [{}], "items": {} },
+    "tupleOpenTyped":     { "prefixItems": [{}], "items": {"type": "boolean"} },
+    "tupleOpenMax":       { "prefixItems": [{}], "items": true, "maxItems": 3},
+
+    "array": { "items": {"type": "number"} },
+
+    "null": { "type": "null" },
+    "string": { "type": "string" },
+    "boolean": { "type": "boolean" }
+  }
+}
+`
+
+func TestSchemaUnmarshal(t *testing.T) {
+	var got *Schema
+	if err := json.Unmarshal([]byte(testSchemaBasic), &got); err != nil {
+		t.Fatalf("Unmarshal: %v", err)
+	}
+	want := &Schema{
+		Properties: []*Schema{
+			{Name: "tupleClosedEmpty", PrefixItems: []*Schema{}, Items: nil},
+			{Name: "tupleClosedMissing", PrefixItems: []*Schema{{}}, Items: nil},
+			{Name: "tupleClosedNull", PrefixItems: []*Schema{{}}, Items: nil},
+			{Name: "tupleClosedFalse", PrefixItems: []*Schema{{}}, Items: nil},
+
+			{Name: "tupleOpenTrue", PrefixItems: []*Schema{{}}, Items: &Schema{}},
+			{Name: "tupleOpenEmpty", PrefixItems: []*Schema{{}}, Items: &Schema{}},
+			{Name: "tupleOpenTyped", PrefixItems: []*Schema{{}}, Items: &Schema{Type: "boolean"}},
+			{Name: "tupleOpenMax", PrefixItems: []*Schema{{}}, Items: &Schema{}, MaxItems: 3},
+
+			{Name: "array", Items: &Schema{Type: "number"}},
+
+			{Name: "null", Type: "null"},
+			{Name: "string", Type: "string"},
+			{Name: "boolean", Type: "boolean"},
+		},
+	}
+
+	if diff := cmp.Diff(want, got); diff != "" {
+		t.Errorf("(-want, +got)\n%s", diff)
+	}
+}
+
+func TestEffectiveType(t *testing.T) {
+	const schema = `
+		{"properties": {
+			"o": {"type": "object"},
+			"a": {"type": "array"},
+			"n": {"type": "number"},
+			"s": {"type": "string"},
+			"z": {"type": "null"},
+			"b": {"type": "boolean"},
+
+			"t0": {"prefixItems": [{}], "items": {"type": "number"}},
+			"t1": {"items": {"type": "number"}, "maxItems": 3},
+
+			"v": {"maxItems": 3}
+		}}
+	`
+
+	var s *Schema
+	if err := json.Unmarshal([]byte(schema), &s); err != nil {
+		t.Fatalf("json.Unmarshal: %v", err)
+	}
+
+	var got []string
+	for _, p := range s.Properties {
+		got = append(got, p.EffectiveType())
+	}
+
+	want := strings.Fields(`
+		object
+		array
+		number
+		string
+		null
+		boolean
+		array
+		array
+		value
+	`)
+	if !reflect.DeepEqual(want, got) {
+		t.Errorf("\ngot:\n\t%v\nwant:\n\t%v", got, want)
+	}
+}
--- a/grammar/testdata/schemas.txt
+++ b/grammar/testdata/schemas.txt
@@ -0,0 +1,76 @@
+# This file holds tests for JSON schema to EBNF grammar conversions.
+#
+# The format is a JSON schema, followed by the expected EBNF grammar. Each test
+# MAY be preceded by a comment that describes the test (e.g. the test name), followed by
+# the JSON schema and the expected EBNF grammar. If no comment is present, the test
+# name the tests number in the file (e.g. "#0", "#1", etc.)
+#
+# Blank lines signify the end or start of a new test. Comments can be added
+# anywhere in the file, but they must be preceded by a '#' character and start at
+# the beginning of the line.
+
+# default
+{}
+root ::= value;
+
+{"properties": {}}
+root ::= value;
+
+# array
+{"properties": {"a": {"type": "array", "items": {"type": "string"}}}}
+root_0_tuple_0 ::= string;
+root_0         ::= "[" ( root_0_tuple_0 )* "]";
+root           ::= "{" "a" ":" root_0 "}";
+
+# array with nested array
+{"type": "array", "items": {"type": "array", "items": {"type": "string"}}}
+root_tuple_0_tuple_0 ::= string;
+root_tuple_0         ::= "[" ( root_tuple_0_tuple_0 )* "]";
+root                 ::= "[" ( root_tuple_0 )* "]";
+
+# object
+{"properties": {"e": {}}}
+root_0 ::= value;
+root   ::= "{" "e" ":" root_0 "}";
+
+# object with nested object
+{"properties": {"o": {"type": "object", "properties": {"e": {}}}}}
+root_0_0 ::= value;
+root_0   ::= "{" "e" ":" root_0_0 "}";
+root     ::= "{" "o" ":" root_0 "}";
+
+# boolean
+{"type": "boolean"}
+root ::= boolean;
+
+# number
+{"properties": {"n": {"type": "number", "minimum": 123, "maximum": 4567}}}
+root_0 ::= number;
+root   ::= "{" "n" ":" root_0 "}";
+
+# string
+{"type": "string"}
+root ::= string;
+
+# string with enum
+{"type": "string", "enum": ["a", "b", "c"]}
+root ::= ( "\"a\"" "|" "\"b\"" "|" "\"c\"" );
+
+# spaces in key
+{"properties": {"a b": {}}}
+root_0 ::= value;
+root   ::= "{" "a b" ":" root_0 "}";
+
+# issue7978
+{ "type": "object", "properties": { "steps": { "type": "array", "items": { "type": "object", "properties": { "explanation": { "type": "string" }, "output": { "type": "string" } }, "required": [ "explanation", "output" ], "additionalProperties": false } }, "final_answer": { "type": "string" } }, "required": [ "steps", "final_answer" ], "additionalProperties": false }
+root_0_tuple_0_0 ::= string;
+root_0_tuple_0_1 ::= string;
+root_0_tuple_0   ::= "{" "explanation" ":" root_0_tuple_0_0 "," "output" ":" root_0_tuple_0_1 "}";
+root_0           ::= "[" ( root_0_tuple_0 )* "]";
+root_1           ::= string;
+root             ::= "{" "steps" ":" root_0 "," "final_answer" ":" root_1 "}";
+
+# !! # special characters in key
+# !! {"properties": {"a!b": {}}}
+# !! !invalid character '!' in key
+# !! 
--- a/integration/api_test.go
+++ b/integration/api_test.go
@@ -1,412 +0,0 @@
-//go:build integration
-
-package integration
-
-import (
-	"bytes"
-	"context"
-	"fmt"
-	"math/rand"
-	"strings"
-	"testing"
-	"time"
-
-	"github.com/ollama/ollama/api"
-)
-
-func TestAPIGenerate(t *testing.T) {
-	initialTimeout := 60 * time.Second
-	streamTimeout := 30 * time.Second
-	ctx, cancel := context.WithTimeout(context.Background(), 1*time.Minute)
-	defer cancel()
-	// Set up the test data
-	req := api.GenerateRequest{
-		Model:  smol,
-		Prompt: "why is the sky blue? be brief",
-		Options: map[string]interface{}{
-			"temperature": 0,
-			"seed":        123,
-		},
-	}
-	anyResp := []string{"rayleigh", "scattering"}
-
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-	if err := PullIfMissing(ctx, client, req.Model); err != nil {
-		t.Fatalf("pull failed %s", err)
-	}
-
-	tests := []struct {
-		name   string
-		stream bool
-	}{
-		{
-			name:   "stream",
-			stream: true,
-		},
-		{
-			name:   "no_stream",
-			stream: false,
-		},
-	}
-
-	for _, test := range tests {
-		t.Run(test.name, func(t *testing.T) {
-			stallTimer := time.NewTimer(initialTimeout)
-			var buf bytes.Buffer
-			fn := func(response api.GenerateResponse) error {
-				// Fields that must always be present
-				if response.Model == "" {
-					t.Errorf("response missing model: %#v", response)
-				}
-				if response.Done {
-					// Required fields for final updates:
-					if response.DoneReason == "" && *req.Stream {
-						// TODO - is the lack of done reason on non-stream a bug?
-						t.Errorf("final response missing done_reason: %#v", response)
-					}
-					if response.Metrics.TotalDuration == 0 {
-						t.Errorf("final response missing total_duration: %#v", response)
-					}
-					if response.Metrics.LoadDuration == 0 {
-						t.Errorf("final response missing load_duration: %#v", response)
-					}
-					if response.Metrics.PromptEvalDuration == 0 {
-						t.Errorf("final response missing prompt_eval_duration: %#v", response)
-					}
-					if response.Metrics.EvalCount == 0 {
-						t.Errorf("final response missing eval_count: %#v", response)
-					}
-					if response.Metrics.EvalDuration == 0 {
-						t.Errorf("final response missing eval_duration: %#v", response)
-					}
-					if len(response.Context) == 0 {
-						t.Errorf("final response missing context: %#v", response)
-					}
-
-					// Note: caching can result in no prompt eval count, so this can't be verified reliably
-					// if response.Metrics.PromptEvalCount == 0 {
-					// 	t.Errorf("final response missing prompt_eval_count: %#v", response)
-					// }
-
-				} // else incremental response, nothing to check right now...
-				buf.Write([]byte(response.Response))
-				if !stallTimer.Reset(streamTimeout) {
-					return fmt.Errorf("stall was detected while streaming response, aborting")
-				}
-				return nil
-			}
-
-			done := make(chan int)
-			var genErr error
-			go func() {
-				req.Stream = &test.stream
-				req.Options["seed"] = rand.Int() // bust cache for prompt eval results
-				genErr = client.Generate(ctx, &req, fn)
-				done <- 0
-			}()
-
-			select {
-			case <-stallTimer.C:
-				if buf.Len() == 0 {
-					t.Errorf("generate never started.  Timed out after :%s", initialTimeout.String())
-				} else {
-					t.Errorf("generate stalled.  Response so far:%s", buf.String())
-				}
-			case <-done:
-				if genErr != nil {
-					t.Fatalf("failed with %s request prompt %s ", req.Model, req.Prompt)
-				}
-				// Verify the response contains the expected data
-				response := buf.String()
-				atLeastOne := false
-				for _, resp := range anyResp {
-					if strings.Contains(strings.ToLower(response), resp) {
-						atLeastOne = true
-						break
-					}
-				}
-				if !atLeastOne {
-					t.Errorf("none of %v found in %s", anyResp, response)
-				}
-			case <-ctx.Done():
-				t.Error("outer test context done while waiting for generate")
-			}
-		})
-	}
-
-	// Validate PS while we're at it...
-	resp, err := client.ListRunning(ctx)
-	if err != nil {
-		t.Fatalf("list models API error: %s", err)
-	}
-	if resp == nil || len(resp.Models) == 0 {
-		t.Fatalf("list models API returned empty list while model should still be loaded")
-	}
-	// Find the model we just loaded and verify some attributes
-	found := false
-	for _, model := range resp.Models {
-		if strings.Contains(model.Name, req.Model) {
-			found = true
-			if model.Model == "" {
-				t.Errorf("model field omitted: %#v", model)
-			}
-			if model.Size == 0 {
-				t.Errorf("size omitted: %#v", model)
-			}
-			if model.Digest == "" {
-				t.Errorf("digest omitted: %#v", model)
-			}
-			verifyModelDetails(t, model.Details)
-			var nilTime time.Time
-			if model.ExpiresAt == nilTime {
-				t.Errorf("expires_at omitted: %#v", model)
-			}
-			// SizeVRAM could be zero.
-		}
-	}
-	if !found {
-		t.Errorf("unable to locate running model: %#v", resp)
-	}
-}
-
-func TestAPIChat(t *testing.T) {
-	initialTimeout := 60 * time.Second
-	streamTimeout := 30 * time.Second
-	ctx, cancel := context.WithTimeout(context.Background(), 1*time.Minute)
-	defer cancel()
-	// Set up the test data
-	req := api.ChatRequest{
-		Model: smol,
-		Messages: []api.Message{
-			{
-				Role:    "user",
-				Content: "why is the sky blue?  be brief",
-			},
-		},
-		Options: map[string]interface{}{
-			"temperature": 0,
-			"seed":        123,
-		},
-	}
-	anyResp := []string{"rayleigh", "scattering"}
-
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-	if err := PullIfMissing(ctx, client, req.Model); err != nil {
-		t.Fatalf("pull failed %s", err)
-	}
-
-	tests := []struct {
-		name   string
-		stream bool
-	}{
-		{
-			name:   "stream",
-			stream: true,
-		},
-		{
-			name:   "no_stream",
-			stream: false,
-		},
-	}
-
-	for _, test := range tests {
-		t.Run(test.name, func(t *testing.T) {
-			stallTimer := time.NewTimer(initialTimeout)
-			var buf bytes.Buffer
-			fn := func(response api.ChatResponse) error {
-				// Fields that must always be present
-				if response.Model == "" {
-					t.Errorf("response missing model: %#v", response)
-				}
-				if response.Done {
-					// Required fields for final updates:
-					var nilTime time.Time
-					if response.CreatedAt == nilTime {
-						t.Errorf("final response missing total_duration: %#v", response)
-					}
-					if response.DoneReason == "" {
-						t.Errorf("final response missing done_reason: %#v", response)
-					}
-					if response.Metrics.TotalDuration == 0 {
-						t.Errorf("final response missing total_duration: %#v", response)
-					}
-					if response.Metrics.LoadDuration == 0 {
-						t.Errorf("final response missing load_duration: %#v", response)
-					}
-					if response.Metrics.PromptEvalDuration == 0 {
-						t.Errorf("final response missing prompt_eval_duration: %#v", response)
-					}
-					if response.Metrics.EvalCount == 0 {
-						t.Errorf("final response missing eval_count: %#v", response)
-					}
-					if response.Metrics.EvalDuration == 0 {
-						t.Errorf("final response missing eval_duration: %#v", response)
-					}
-
-					if response.Metrics.PromptEvalCount == 0 {
-						t.Errorf("final response missing prompt_eval_count: %#v", response)
-					}
-				} // else incremental response, nothing to check right now...
-				buf.Write([]byte(response.Message.Content))
-				if !stallTimer.Reset(streamTimeout) {
-					return fmt.Errorf("stall was detected while streaming response, aborting")
-				}
-				return nil
-			}
-
-			done := make(chan int)
-			var genErr error
-			go func() {
-				req.Stream = &test.stream
-				req.Options["seed"] = rand.Int() // bust cache for prompt eval results
-				genErr = client.Chat(ctx, &req, fn)
-				done <- 0
-			}()
-
-			select {
-			case <-stallTimer.C:
-				if buf.Len() == 0 {
-					t.Errorf("chat never started.  Timed out after :%s", initialTimeout.String())
-				} else {
-					t.Errorf("chat stalled.  Response so far:%s", buf.String())
-				}
-			case <-done:
-				if genErr != nil {
-					t.Fatalf("failed with %s request prompt %v", req.Model, req.Messages)
-				}
-				// Verify the response contains the expected data
-				response := buf.String()
-				atLeastOne := false
-				for _, resp := range anyResp {
-					if strings.Contains(strings.ToLower(response), resp) {
-						atLeastOne = true
-						break
-					}
-				}
-				if !atLeastOne {
-					t.Errorf("none of %v found in %s", anyResp, response)
-				}
-			case <-ctx.Done():
-				t.Error("outer test context done while waiting for chat")
-			}
-		})
-	}
-}
-
-func TestAPIListModels(t *testing.T) {
-	ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-
-	// Make sure we have at least one model so an empty list can be considered a failure
-	if err := PullIfMissing(ctx, client, smol); err != nil {
-		t.Fatalf("pull failed %s", err)
-	}
-
-	resp, err := client.List(ctx)
-	if err != nil {
-		t.Fatalf("unable to list models: %s", err)
-	}
-	if len(resp.Models) == 0 {
-		t.Fatalf("list should not be empty")
-	}
-	model := resp.Models[0]
-	if model.Name == "" {
-		t.Errorf("first model name empty: %#v", model)
-	}
-	var nilTime time.Time
-	if model.ModifiedAt == nilTime {
-		t.Errorf("first model modified_at empty: %#v", model)
-	}
-	if model.Size == 0 {
-		t.Errorf("first model size empty: %#v", model)
-	}
-	if model.Digest == "" {
-		t.Errorf("first model digest empty: %#v", model)
-	}
-	verifyModelDetails(t, model.Details)
-}
-
-func verifyModelDetails(t *testing.T, details api.ModelDetails) {
-	if details.Format == "" {
-		t.Errorf("first model details.format empty: %#v", details)
-	}
-	if details.Family == "" {
-		t.Errorf("first model details.family empty: %#v", details)
-	}
-	if details.ParameterSize == "" {
-		t.Errorf("first model details.parameter_size empty: %#v", details)
-	}
-	if details.QuantizationLevel == "" {
-		t.Errorf("first model details.quantization_level empty: %#v", details)
-	}
-}
-
-func TestAPIShowModel(t *testing.T) {
-	modelName := "llama3.2"
-	ctx, cancel := context.WithTimeout(context.Background(), 1*time.Minute)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-
-	if err := PullIfMissing(ctx, client, modelName); err != nil {
-		t.Fatalf("pull failed %s", err)
-	}
-	resp, err := client.Show(ctx, &api.ShowRequest{Name: modelName})
-	if err != nil {
-		t.Fatalf("unable to show model: %s", err)
-	}
-	if resp.License == "" {
-		t.Errorf("%s missing license: %#v", modelName, resp)
-	}
-	if resp.Modelfile == "" {
-		t.Errorf("%s missing modelfile: %#v", modelName, resp)
-	}
-	if resp.Parameters == "" {
-		t.Errorf("%s missing parameters: %#v", modelName, resp)
-	}
-	if resp.Template == "" {
-		t.Errorf("%s missing template: %#v", modelName, resp)
-	}
-	// llama3 omits system
-	verifyModelDetails(t, resp.Details)
-	// llama3 ommits messages
-	if len(resp.ModelInfo) == 0 {
-		t.Errorf("%s missing model_info: %#v", modelName, resp)
-	}
-	// llama3 omits projectors
-	var nilTime time.Time
-	if resp.ModifiedAt == nilTime {
-		t.Errorf("%s missing modified_at: %#v", modelName, resp)
-	}
-}
-
-func TestAPIEmbeddings(t *testing.T) {
-	ctx, cancel := context.WithTimeout(context.Background(), 1*time.Minute)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-	req := api.EmbeddingRequest{
-		Model:  "orca-mini",
-		Prompt: "why is the sky blue?",
-		Options: map[string]interface{}{
-			"temperature": 0,
-			"seed":        123,
-		},
-	}
-
-	if err := PullIfMissing(ctx, client, req.Model); err != nil {
-		t.Fatalf("pull failed %s", err)
-	}
-
-	resp, err := client.Embeddings(ctx, &req)
-	if err != nil {
-		t.Fatalf("embeddings call failed %s", err)
-	}
-	if len(resp.Embedding) == 0 {
-		t.Errorf("zero length embedding response")
-	}
-}
--- a/integration/basic_test.go
+++ b/integration/basic_test.go
@@ -14,15 +14,15 @@ import (
 	"github.com/stretchr/testify/require"
 )

-func TestBlueSky(t *testing.T) {
+func TestOrcaMiniBlueSky(t *testing.T) {
 	ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
 	defer cancel()
 	// Set up the test data
 	req := api.GenerateRequest{
-		Model:  smol,
+		Model:  "orca-mini",
 		Prompt: "why is the sky blue?",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"temperature": 0,
 			"seed":        123,
 		},
@@ -31,7 +31,6 @@ func TestBlueSky(t *testing.T) {
 }

 func TestUnicode(t *testing.T) {
-	skipUnderMinVRAM(t, 6)
 	ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute)
 	defer cancel()
 	// Set up the test data
@@ -40,7 +39,7 @@ func TestUnicode(t *testing.T) {
 		Model:  "deepseek-coder-v2:16b-lite-instruct-q2_K",
 		Prompt: "天空为什么是蓝色的?",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"temperature": 0,
 			"seed":        123,
 			// Workaround deepseek context shifting bug
@@ -62,7 +61,7 @@ func TestExtendedUnicodeOutput(t *testing.T) {
 		Model:  "gemma2:2b",
 		Prompt: "Output some smily face emoji",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"temperature": 0,
 			"seed":        123,
 		},
@@ -94,10 +93,10 @@ func TestUnicodeModelDir(t *testing.T) {
 	defer cancel()

 	req := api.GenerateRequest{
-		Model:  smol,
+		Model:  "orca-mini",
 		Prompt: "why is the sky blue?",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"temperature": 0,
 			"seed":        123,
 		},
--- a/integration/concurrency_test.go
+++ b/integration/concurrency_test.go
@@ -21,11 +21,11 @@ func TestMultiModelConcurrency(t *testing.T) {
 	var (
 		req = [2]api.GenerateRequest{
 			{
-				Model:     "llama3.2:1b",
+				Model:     "orca-mini",
 				Prompt:    "why is the ocean blue?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
@@ -34,7 +34,7 @@ func TestMultiModelConcurrency(t *testing.T) {
 				Prompt:    "what is the origin of the us thanksgiving holiday?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
@@ -67,7 +67,7 @@ func TestMultiModelConcurrency(t *testing.T) {
 	wg.Wait()
 }

-func TestIntegrationConcurrentPredict(t *testing.T) {
+func TestIntegrationConcurrentPredictOrcaMini(t *testing.T) {
 	req, resp := GenerateRequests()
 	reqLimit := len(req)
 	iterLimit := 5
@@ -117,9 +117,6 @@ func TestMultiModelStress(t *testing.T) {
 	if err != nil {
 		t.Fatal(err)
 	}
-	if maxVram < 2*format.GibiByte {
-		t.Skip("VRAM less than 2G, skipping model stress tests")
-	}

 	type model struct {
 		name string
@@ -128,8 +125,8 @@ func TestMultiModelStress(t *testing.T) {

 	smallModels := []model{
 		{
-			name: "llama3.2:1b",
-			size: 2876 * format.MebiByte,
+			name: "orca-mini",
+			size: 2992 * format.MebiByte,
 		},
 		{
 			name: "phi",
--- a/integration/context_test.go
+++ b/integration/context_test.go
@@ -23,7 +23,7 @@ func TestLongInputContext(t *testing.T) {
 		Model:  "llama2",
 		Prompt: "Oh, don’t speak to me of Austria. Perhaps I don’t understand things, but Austria never has wished, and does not wish, for war. She is betraying us! Russia alone must save Europe. Our gracious sovereign recognizes his high vocation and will be true to it. That is the one thing I have faith in! Our good and wonderful sovereign has to perform the noblest role on earth, and he is so virtuous and noble that God will not forsake him. He will fulfill his vocation and crush the hydra of revolution, which has become more terrible than ever in the person of this murderer and villain! We alone must avenge the blood of the just one.... Whom, I ask you, can we rely on?... England with her commercial spirit will not and cannot understand the Emperor Alexander’s loftiness of soul. She has refused to evacuate Malta. She wanted to find, and still seeks, some secret motive in our actions. What answer did Novosíltsev get? None. The English have not understood and cannot understand the self-abnegation of our Emperor who wants nothing for himself, but only desires the good of mankind. And what have they promised? Nothing! And what little they have promised they will not perform! Prussia has always declared that Buonaparte is invincible, and that all Europe is powerless before him.... And I don’t believe a word that Hardenburg says, or Haugwitz either. This famous Prussian neutrality is just a trap. I have faith only in God and the lofty destiny of our adored monarch. He will save Europe! What country is this referring to?",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"temperature": 0,
 			"seed":        123,
 			"num_ctx":     128,
@@ -50,7 +50,7 @@ func TestContextExhaustion(t *testing.T) {
 		Model:  "llama2",
 		Prompt: "Write me a story with a ton of emojis?",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"temperature": 0,
 			"seed":        123,
 			"num_ctx":     128,
--- a/integration/embed_test.go
+++ b/integration/embed_test.go
@@ -34,15 +34,13 @@ func cosineSimilarity[V float32 | float64](v1, v2 []V) V {
 func TestAllMiniLMEmbeddings(t *testing.T) {
 	ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
 	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()

 	req := api.EmbeddingRequest{
 		Model:  "all-minilm",
 		Prompt: "why is the sky blue?",
 	}

-	res, err := embeddingTestHelper(ctx, client, t, req)
+	res, err := embeddingTestHelper(ctx, t, req)

 	if err != nil {
 		t.Fatalf("error: %v", err)
@@ -64,15 +62,13 @@ func TestAllMiniLMEmbeddings(t *testing.T) {
 func TestAllMiniLMEmbed(t *testing.T) {
 	ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
 	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()

 	req := api.EmbedRequest{
 		Model: "all-minilm",
 		Input: "why is the sky blue?",
 	}

-	res, err := embedTestHelper(ctx, client, t, req)
+	res, err := embedTestHelper(ctx, t, req)

 	if err != nil {
 		t.Fatalf("error: %v", err)
@@ -102,15 +98,13 @@ func TestAllMiniLMEmbed(t *testing.T) {
 func TestAllMiniLMBatchEmbed(t *testing.T) {
 	ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
 	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()

 	req := api.EmbedRequest{
 		Model: "all-minilm",
 		Input: []string{"why is the sky blue?", "why is the grass green?"},
 	}

-	res, err := embedTestHelper(ctx, client, t, req)
+	res, err := embedTestHelper(ctx, t, req)

 	if err != nil {
 		t.Fatalf("error: %v", err)
@@ -150,8 +144,6 @@ func TestAllMiniLMBatchEmbed(t *testing.T) {
 func TestAllMiniLMEmbedTruncate(t *testing.T) {
 	ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
 	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()

 	truncTrue, truncFalse := true, false

@@ -190,7 +182,7 @@ func TestAllMiniLMEmbedTruncate(t *testing.T) {
 	res := make(map[string]*api.EmbedResponse)

 	for _, req := range reqs {
-		response, err := embedTestHelper(ctx, client, t, req.Request)
+		response, err := embedTestHelper(ctx, t, req.Request)
 		if err != nil {
 			t.Fatalf("error: %v", err)
 		}
@@ -206,7 +198,7 @@ func TestAllMiniLMEmbedTruncate(t *testing.T) {
 	}

 	// check that truncate set to false returns an error if context length is exceeded
-	_, err := embedTestHelper(ctx, client, t, api.EmbedRequest{
+	_, err := embedTestHelper(ctx, t, api.EmbedRequest{
 		Model:    "all-minilm",
 		Input:    "why is the sky blue?",
 		Truncate: &truncFalse,
@@ -218,7 +210,9 @@ func TestAllMiniLMEmbedTruncate(t *testing.T) {
 	}
 }

-func embeddingTestHelper(ctx context.Context, client *api.Client, t *testing.T, req api.EmbeddingRequest) (*api.EmbeddingResponse, error) {
+func embeddingTestHelper(ctx context.Context, t *testing.T, req api.EmbeddingRequest) (*api.EmbeddingResponse, error) {
+	client, _, cleanup := InitServerConnection(ctx, t)
+	defer cleanup()
 	if err := PullIfMissing(ctx, client, req.Model); err != nil {
 		t.Fatalf("failed to pull model %s: %v", req.Model, err)
 	}
@@ -232,7 +226,9 @@ func embeddingTestHelper(ctx context.Context, client *api.Client, t *testing.T,
 	return response, nil
 }

-func embedTestHelper(ctx context.Context, client *api.Client, t *testing.T, req api.EmbedRequest) (*api.EmbedResponse, error) {
+func embedTestHelper(ctx context.Context, t *testing.T, req api.EmbedRequest) (*api.EmbedResponse, error) {
+	client, _, cleanup := InitServerConnection(ctx, t)
+	defer cleanup()
 	if err := PullIfMissing(ctx, client, req.Model); err != nil {
 		t.Fatalf("failed to pull model %s: %v", req.Model, err)
 	}
--- a/integration/llm_image_test.go
+++ b/integration/llm_image_test.go
@@ -12,51 +12,58 @@ import (
 	"github.com/stretchr/testify/require"
 )

-func TestVisionModels(t *testing.T) {
-	skipUnderMinVRAM(t, 6)
-	type testCase struct {
-		model string
-	}
-	testCases := []testCase{
-		{
-			model: "llava:7b",
+func TestIntegrationLlava(t *testing.T) {
+	image, err := base64.StdEncoding.DecodeString(imageEncoding)
+	require.NoError(t, err)
+	req := api.GenerateRequest{
+		Model:  "llava:7b",
+		Prompt: "what does the text in this image say?",
+		Stream: &stream,
+		Options: map[string]interface{}{
+			"seed":        42,
+			"temperature": 0.0,
 		},
-		{
-			model: "llama3.2-vision",
-		},
-		{
-			model: "gemma3",
+		Images: []api.ImageData{
+			image,
 		},
 	}

-	for _, v := range testCases {
-		t.Run(v.model, func(t *testing.T) {
-			image, err := base64.StdEncoding.DecodeString(imageEncoding)
-			require.NoError(t, err)
-			req := api.GenerateRequest{
-				Model:  v.model,
-				Prompt: "what does the text in this image say?",
-				Stream: &stream,
-				Options: map[string]any{
-					"seed":        42,
-					"temperature": 0.0,
-				},
-				Images: []api.ImageData{
-					image,
-				},
-			}
-			ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
-			defer cancel()
-			client, _, cleanup := InitServerConnection(ctx, t)
+	// Note: sometimes it returns "the ollamas" sometimes "the ollams"
+	resp := "the ollam"
+	ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute)
+	defer cancel()
+	client, _, cleanup := InitServerConnection(ctx, t)
+	defer cleanup()
+	require.NoError(t, PullIfMissing(ctx, client, req.Model))
+	// llava models on CPU can be quite slow to start,
+	DoGenerate(ctx, t, client, req, []string{resp}, 120*time.Second, 30*time.Second)
+}

-			// Note: sometimes it returns "the ollamas" sometimes "the ollams"
-			resp := "the ollam"
-			defer cleanup()
-			require.NoError(t, PullIfMissing(ctx, client, req.Model))
-			// llava models on CPU can be quite slow to start
-			DoGenerate(ctx, t, client, req, []string{resp}, 240*time.Second, 30*time.Second)
-		})
+func TestIntegrationMllama(t *testing.T) {
+	image, err := base64.StdEncoding.DecodeString(imageEncoding)
+	require.NoError(t, err)
+	req := api.GenerateRequest{
+		// TODO fix up once we publish the final image
+		Model:  "x/llama3.2-vision",
+		Prompt: "what does the text in this image say?",
+		Stream: &stream,
+		Options: map[string]interface{}{
+			"seed":        42,
+			"temperature": 0.0,
+		},
+		Images: []api.ImageData{
+			image,
+		},
 	}
+
+	resp := "the ollamas"
+	ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
+	defer cancel()
+	client, _, cleanup := InitServerConnection(ctx, t)
+	defer cleanup()
+	require.NoError(t, PullIfMissing(ctx, client, req.Model))
+	// mllama models on CPU can be quite slow to start,
+	DoGenerate(ctx, t, client, req, []string{resp}, 240*time.Second, 30*time.Second)
 }

 func TestIntegrationSplitBatch(t *testing.T) {
@@ -68,7 +75,7 @@ func TestIntegrationSplitBatch(t *testing.T) {
 		System: "Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed aliquet, justo in malesuada lobortis, odio ligula volutpat quam, quis faucibus ipsum magna quis sapien. Aliquam in venenatis diam, eu viverra magna. Phasellus imperdiet hendrerit volutpat. Vivamus sem ex, facilisis placerat felis non, dictum elementum est. Phasellus aliquam imperdiet lacus, eget placerat ligula sodales vel. Pellentesque nec auctor mi. Curabitur arcu nisi, faucibus eget nunc id, viverra interdum mi. Curabitur ornare ipsum ex, ac euismod ex aliquam in. Vestibulum id magna at purus accumsan fermentum. Proin scelerisque posuere nunc quis interdum. Maecenas sed mollis nisl. Etiam vitae ipsum interdum, placerat est quis, tincidunt velit. Nullam tempor nibh non lorem volutpat efficitur. Cras laoreet diam imperdiet ipsum auctor bibendum. Suspendisse ultrices urna sed metus sagittis suscipit. Quisque ullamcorper aliquam nibh ut mollis. Aenean dapibus mauris pharetra, venenatis elit ac, hendrerit odio. Cras vestibulum erat tempor, lobortis justo eu, lobortis ipsum. Nam laoreet dapibus sem. Proin vel diam ultrices, elementum ante et, ornare lectus. Proin eu accumsan nisl. Praesent ac ex vitae ipsum vulputate tristique facilisis sit amet lacus. Nullam faucibus magna a pellentesque pretium. Nunc lacinia ullamcorper sollicitudin. Donec vitae accumsan turpis, sed porttitor est. Donec porttitor mi vitae augue faucibus, vel mollis diam tincidunt.",
 		Prompt: "what does the text in this image say?",
 		Stream: &stream,
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"seed":        42,
 			"temperature": 0.0,
 		},
--- a/integration/llm_test.go
+++ b/integration/llm_test.go
@@ -17,30 +17,30 @@ var (
 	stream = false
 	req    = [2]api.GenerateRequest{
 		{
-			Model:  smol,
+			Model:  "orca-mini",
 			Prompt: "why is the ocean blue?",
 			Stream: &stream,
-			Options: map[string]any{
+			Options: map[string]interface{}{
 				"seed":        42,
 				"temperature": 0.0,
 			},
 		}, {
-			Model:  smol,
+			Model:  "orca-mini",
 			Prompt: "what is the origin of the us thanksgiving holiday?",
 			Stream: &stream,
-			Options: map[string]any{
+			Options: map[string]interface{}{
 				"seed":        42,
 				"temperature": 0.0,
 			},
 		},
 	}
 	resp = [2][]string{
-		{"sunlight", "scattering", "interact"},
+		{"sunlight"},
 		{"england", "english", "massachusetts", "pilgrims"},
 	}
 )

-func TestIntegrationSimple(t *testing.T) {
+func TestIntegrationSimpleOrcaMini(t *testing.T) {
 	ctx, cancel := context.WithTimeout(context.Background(), time.Second*120)
 	defer cancel()
 	GenerateTestHelper(ctx, t, req[0], resp[0])
--- a/integration/max_queue_test.go
+++ b/integration/max_queue_test.go
@@ -30,9 +30,9 @@ func TestMaxQueue(t *testing.T) {
 	t.Setenv("OLLAMA_MAX_QUEUE", strconv.Itoa(threadCount))

 	req := api.GenerateRequest{
-		Model:  smol,
+		Model:  "orca-mini",
 		Prompt: "write a long historical fiction story about christopher columbus.  use at least 10 facts from his actual journey",
-		Options: map[string]any{
+		Options: map[string]interface{}{
 			"seed":        42,
 			"temperature": 0.0,
 		},
@@ -52,8 +52,8 @@ func TestMaxQueue(t *testing.T) {
 	embedCtx := ctx

 	var genwg sync.WaitGroup
-	genwg.Add(1)
 	go func() {
+		genwg.Add(1)
 		defer genwg.Done()
 		slog.Info("Starting generate request")
 		DoGenerate(ctx, t, client, req, resp, 45*time.Second, 5*time.Second)
@@ -61,7 +61,7 @@ func TestMaxQueue(t *testing.T) {
 	}()

 	// Give the generate a chance to get started before we start hammering on embed requests
-	time.Sleep(10 * time.Millisecond)
+	time.Sleep(5 * time.Millisecond)

 	threadCount += 10 // Add a few extra to ensure we push the queue past its limit
 	busyCount := 0
@@ -71,8 +71,8 @@ func TestMaxQueue(t *testing.T) {
 	counterMu := sync.Mutex{}
 	var embedwg sync.WaitGroup
 	for i := 0; i < threadCount; i++ {
-		embedwg.Add(1)
 		go func(i int) {
+			embedwg.Add(1)
 			defer embedwg.Done()
 			slog.Info("embed started", "id", i)
 			embedReq := api.EmbeddingRequest{
--- a/integration/model_arch_test.go
+++ b/integration/model_arch_test.go
@@ -1,184 +0,0 @@
-//go:build integration && models
-
-package integration
-
-import (
-	"context"
-	"encoding/json"
-	"fmt"
-	"io/ioutil"
-	"log/slog"
-	"os"
-	"path/filepath"
-	"strconv"
-	"strings"
-	"testing"
-	"time"
-
-	"github.com/ollama/ollama/api"
-	"github.com/ollama/ollama/format"
-)
-
-var (
-	started    = time.Now()
-	chatModels = []string{
-		"granite3-moe:latest",
-		"granite-code:latest",
-		"nemotron-mini:latest",
-		"command-r:latest",
-		"gemma2:latest",
-		"gemma:latest",
-		"internlm2:latest",
-		"phi3.5:latest",
-		"phi3:latest",
-		// "phi:latest", // flaky, sometimes generates no response on first query
-		"stablelm2:latest", // Predictions are off, crashes on small VRAM GPUs
-		"falcon:latest",
-		"falcon2:latest",
-		"minicpm-v:latest",
-		"mistral:latest",
-		"orca-mini:latest",
-		"llama2:latest",
-		"llama3.1:latest",
-		"llama3.2:latest",
-		"llama3.2-vision:latest",
-		"qwen2.5-coder:latest",
-		"qwen:latest",
-		"solar-pro:latest",
-	}
-)
-
-func TestModelsGenerate(t *testing.T) {
-	softTimeout, hardTimeout := getTimeouts(t)
-	slog.Info("Setting timeouts", "soft", softTimeout, "hard", hardTimeout)
-	ctx, cancel := context.WithTimeout(context.Background(), hardTimeout)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-
-	// TODO use info API eventually
-	var maxVram uint64
-	var err error
-	if s := os.Getenv("OLLAMA_MAX_VRAM"); s != "" {
-		maxVram, err = strconv.ParseUint(s, 10, 64)
-		if err != nil {
-			t.Fatalf("invalid  OLLAMA_MAX_VRAM %v", err)
-		}
-	} else {
-		slog.Warn("No VRAM info available, testing all models, so larger ones might timeout...")
-	}
-
-	for _, model := range chatModels {
-		t.Run(model, func(t *testing.T) {
-			if time.Now().Sub(started) > softTimeout {
-				t.Skip("skipping remaining tests to avoid excessive runtime")
-			}
-			if err := PullIfMissing(ctx, client, model); err != nil {
-				t.Fatalf("pull failed %s", err)
-			}
-			if maxVram > 0 {
-				resp, err := client.List(ctx)
-				if err != nil {
-					t.Fatalf("list models failed %v", err)
-				}
-				for _, m := range resp.Models {
-					if m.Name == model && float32(m.Size)*1.2 > float32(maxVram) {
-						t.Skipf("model %s is too large for available VRAM: %s > %s", model, format.HumanBytes(m.Size), format.HumanBytes(int64(maxVram)))
-					}
-				}
-			}
-			// TODO - fiddle with context size
-			req := api.GenerateRequest{
-				Model:  model,
-				Prompt: "why is the sky blue?",
-				Options: map[string]interface{}{
-					"temperature": 0,
-					"seed":        123,
-				},
-			}
-			anyResp := []string{"rayleigh", "scattering", "atmosphere", "nitrogen", "oxygen"}
-			DoGenerate(ctx, t, client, req, anyResp, 120*time.Second, 30*time.Second)
-		})
-	}
-}
-
-func TestModelsEmbed(t *testing.T) {
-	softTimeout, hardTimeout := getTimeouts(t)
-	ctx, cancel := context.WithTimeout(context.Background(), hardTimeout)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-
-	// TODO use info API eventually
-	var maxVram uint64
-	var err error
-	if s := os.Getenv("OLLAMA_MAX_VRAM"); s != "" {
-		maxVram, err = strconv.ParseUint(s, 10, 64)
-		if err != nil {
-			t.Fatalf("invalid  OLLAMA_MAX_VRAM %v", err)
-		}
-	} else {
-		slog.Warn("No VRAM info available, testing all models, so larger ones might timeout...")
-	}
-
-	data, err := ioutil.ReadFile(filepath.Join("testdata", "embed.json"))
-	if err != nil {
-		t.Fatalf("failed to open test data file: %s", err)
-	}
-	testCase := map[string][]float64{}
-	err = json.Unmarshal(data, &testCase)
-	if err != nil {
-		t.Fatalf("failed to load test data: %s", err)
-	}
-	for model, expected := range testCase {
-
-		t.Run(model, func(t *testing.T) {
-			if time.Now().Sub(started) > softTimeout {
-				t.Skip("skipping remaining tests to avoid excessive runtime")
-			}
-			if err := PullIfMissing(ctx, client, model); err != nil {
-				t.Fatalf("pull failed %s", err)
-			}
-			if maxVram > 0 {
-				resp, err := client.List(ctx)
-				if err != nil {
-					t.Fatalf("list models failed %v", err)
-				}
-				for _, m := range resp.Models {
-					if m.Name == model && float32(m.Size)*1.2 > float32(maxVram) {
-						t.Skipf("model %s is too large for available VRAM: %s > %s", model, format.HumanBytes(m.Size), format.HumanBytes(int64(maxVram)))
-					}
-				}
-			}
-			req := api.EmbeddingRequest{
-				Model:  model,
-				Prompt: "why is the sky blue?",
-				Options: map[string]interface{}{
-					"temperature": 0,
-					"seed":        123,
-				},
-			}
-			resp, err := client.Embeddings(ctx, &req)
-			if err != nil {
-				t.Fatalf("embeddings call failed %s", err)
-			}
-			if len(resp.Embedding) == 0 {
-				t.Errorf("zero length embedding response")
-			}
-			if len(expected) != len(resp.Embedding) {
-				expStr := make([]string, len(resp.Embedding))
-				for i, v := range resp.Embedding {
-					expStr[i] = fmt.Sprintf("%0.6f", v)
-				}
-				// When adding new models, use this output to populate the testdata/embed.json
-				fmt.Printf("expected\n%s\n", strings.Join(expStr, ", "))
-				t.Fatalf("expected %d, got %d", len(expected), len(resp.Embedding))
-			}
-			sim := cosineSimilarity(resp.Embedding, expected)
-			if sim < 0.99 {
-				t.Fatalf("expected %v, got %v (similarity: %f)", expected[0:5], resp.Embedding[0:5], sim)
-			}
-		})
-	}
-
-}
--- a/integration/quantization_test.go
+++ b/integration/quantization_test.go
@@ -1,130 +0,0 @@
-//go:build integration && models
-
-package integration
-
-import (
-	"bytes"
-	"context"
-	"fmt"
-	"log/slog"
-	"strings"
-	"testing"
-	"time"
-
-	"github.com/ollama/ollama/api"
-)
-
-func TestQuantization(t *testing.T) {
-	sourceModels := []string{
-		"qwen2.5:0.5b-instruct-fp16",
-	}
-	quantizations := []string{
-		"Q8_0",
-		"Q4_K_S",
-		"Q4_K_M",
-		"Q4_K",
-	}
-	softTimeout, hardTimeout := getTimeouts(t)
-	started := time.Now()
-	slog.Info("Setting timeouts", "soft", softTimeout, "hard", hardTimeout)
-	ctx, cancel := context.WithTimeout(context.Background(), hardTimeout)
-	defer cancel()
-	client, _, cleanup := InitServerConnection(ctx, t)
-	defer cleanup()
-
-	for _, base := range sourceModels {
-		if err := PullIfMissing(ctx, client, base); err != nil {
-			t.Fatalf("pull failed %s", err)
-		}
-		for _, quant := range quantizations {
-			newName := fmt.Sprintf("%s__%s", base, quant)
-			t.Run(newName, func(t *testing.T) {
-				if time.Now().Sub(started) > softTimeout {
-					t.Skip("skipping remaining tests to avoid excessive runtime")
-				}
-				req := &api.CreateRequest{
-					Model:        newName,
-					Quantization: quant,
-					From:         base,
-				}
-				fn := func(resp api.ProgressResponse) error {
-					// fmt.Print(".")
-					return nil
-				}
-				t.Logf("quantizing: %s -> %s", base, quant)
-				if err := client.Create(ctx, req, fn); err != nil {
-					t.Fatalf("create failed %s", err)
-				}
-				defer func() {
-					req := &api.DeleteRequest{
-						Model: newName,
-					}
-					t.Logf("deleting: %s -> %s", base, quant)
-					if err := client.Delete(ctx, req); err != nil {
-						t.Logf("failed to clean up %s: %s", req.Model, err)
-					}
-				}()
-				// Check metadata on the model
-				resp, err := client.Show(ctx, &api.ShowRequest{Name: newName})
-				if err != nil {
-					t.Fatalf("unable to show model: %s", err)
-				}
-				if !strings.Contains(resp.Details.QuantizationLevel, quant) {
-					t.Fatalf("unexpected quantization for %s:\ngot: %s", newName, resp.Details.QuantizationLevel)
-				}
-
-				stream := true
-				genReq := api.GenerateRequest{
-					Model:     newName,
-					Prompt:    "why is the sky blue?",
-					KeepAlive: &api.Duration{Duration: 3 * time.Second},
-					Options: map[string]any{
-						"seed":        42,
-						"temperature": 0.0,
-					},
-					Stream: &stream,
-				}
-				t.Logf("verifying: %s -> %s", base, quant)
-
-				// Some smaller quantizations can cause models to have poor quality
-				// or get stuck in repetition loops, so we stop as soon as we have any matches
-				anyResp := []string{"rayleigh", "scattering", "day", "sun", "moon", "color", "nitrogen", "oxygen"}
-				reqCtx, reqCancel := context.WithCancel(ctx)
-				atLeastOne := false
-				var buf bytes.Buffer
-				genfn := func(response api.GenerateResponse) error {
-					buf.Write([]byte(response.Response))
-					fullResp := strings.ToLower(buf.String())
-					for _, resp := range anyResp {
-						if strings.Contains(fullResp, resp) {
-							atLeastOne = true
-							t.Log(fullResp)
-							reqCancel()
-							break
-						}
-					}
-					return nil
-				}
-
-				done := make(chan int)
-				var genErr error
-				go func() {
-					genErr = client.Generate(reqCtx, &genReq, genfn)
-					done <- 0
-				}()
-
-				select {
-				case <-done:
-					if genErr != nil && !atLeastOne {
-						t.Fatalf("failed with %s request prompt %s ", genReq.Model, genReq.Prompt)
-					}
-				case <-ctx.Done():
-					t.Error("outer test context done while waiting for generate")
-				}
-
-				t.Logf("passed")
-
-			})
-		}
-	}
-}
--- a/integration/testdata/embed.json
+++ b/integration/testdata/embed.json
--- a/integration/utils_test.go
+++ b/integration/utils_test.go
@@ -24,14 +24,9 @@ import (

 	"github.com/ollama/ollama/api"
 	"github.com/ollama/ollama/app/lifecycle"
-	"github.com/ollama/ollama/format"
 	"github.com/stretchr/testify/require"
 )

-const (
-	smol = "llama3.2:1b"
-)
-
 func Init() {
 	lifecycle.InitLogging()
 }
@@ -145,7 +140,7 @@ func PullIfMissing(ctx context.Context, client *api.Client, modelName string) er

 	showCtx, cancel := context.WithDeadlineCause(
 		ctx,
-		time.Now().Add(20*time.Second),
+		time.Now().Add(10*time.Second),
 		fmt.Errorf("show for existing model %s took too long", modelName),
 	)
 	defer cancel()
@@ -162,7 +157,7 @@ func PullIfMissing(ctx context.Context, client *api.Client, modelName string) er
 	}
 	slog.Info("model missing", "model", modelName)

-	stallDuration := 60 * time.Second // This includes checksum verification, which can take a while on larger models, and slower systems
+	stallDuration := 30 * time.Second // This includes checksum verification, which can take a while on larger models
 	stallTimer := time.NewTimer(stallDuration)
 	fn := func(resp api.ProgressResponse) error {
 		// fmt.Print(".")
@@ -217,7 +212,6 @@ func InitServerConnection(ctx context.Context, t *testing.T) (*api.Client, strin
 					slog.Error("failed to open server log", "logfile", lifecycle.ServerLogFile, "error", err)
 					return
 				}
-				defer fp.Close()
 				data, err := io.ReadAll(fp)
 				if err != nil {
 					slog.Error("failed to read server log", "logfile", lifecycle.ServerLogFile, "error", err)
@@ -289,51 +283,51 @@ func DoGenerate(ctx context.Context, t *testing.T, client *api.Client, genReq ap
 }

 // Generate a set of requests
-// By default each request uses llama3.2 as the model
+// By default each request uses orca-mini as the model
 func GenerateRequests() ([]api.GenerateRequest, [][]string) {
 	return []api.GenerateRequest{
 			{
-				Model:     smol,
+				Model:     "orca-mini",
 				Prompt:    "why is the ocean blue?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
 			}, {
-				Model:     smol,
+				Model:     "orca-mini",
 				Prompt:    "why is the color of dirt brown?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
 			}, {
-				Model:     smol,
+				Model:     "orca-mini",
 				Prompt:    "what is the origin of the us thanksgiving holiday?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
 			}, {
-				Model:     smol,
+				Model:     "orca-mini",
 				Prompt:    "what is the origin of independence day?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
 			}, {
-				Model:     smol,
+				Model:     "orca-mini",
 				Prompt:    "what is the composition of air?",
 				Stream:    &stream,
 				KeepAlive: &api.Duration{Duration: 10 * time.Second},
-				Options: map[string]any{
+				Options: map[string]interface{}{
 					"seed":        42,
 					"temperature": 0.0,
 				},
@@ -347,26 +341,3 @@ func GenerateRequests() ([]api.GenerateRequest, [][]string) {
 			{"nitrogen", "oxygen", "carbon", "dioxide"},
 		}
 }
-
-func skipUnderMinVRAM(t *testing.T, gb uint64) {
-	// TODO use info API in the future
-	if s := os.Getenv("OLLAMA_MAX_VRAM"); s != "" {
-		maxVram, err := strconv.ParseUint(s, 10, 64)
-		require.NoError(t, err)
-		// Don't hammer on small VRAM cards...
-		if maxVram < gb*format.GibiByte {
-			t.Skip("skipping with small VRAM to avoid timeouts")
-		}
-	}
-}
-
-func getTimeouts(t *testing.T) (soft time.Duration, hard time.Duration) {
-	deadline, hasDeadline := t.Deadline()
-	if !hasDeadline {
-		return 8 * time.Minute, 10 * time.Minute
-	} else if deadline.Compare(time.Now().Add(2*time.Minute)) <= 0 {
-		t.Skip("too little time")
-		return time.Duration(0), time.Duration(0)
-	}
-	return -time.Since(deadline.Add(-2 * time.Minute)), -time.Since(deadline.Add(-20 * time.Second))
-}
--- a/kvcache/cache.go
+++ b/kvcache/cache.go
@@ -56,18 +56,12 @@ type Cache interface {

 	// StartForward is called before the start of the model's forward pass.
 	// For each token in the coming batch, there must be a corresponding
-	// entry in positions and seqs. reserve is to preallocate memory
-	// without actually storing data in the cache.
-	StartForward(ctx ml.Context, batch input.Batch, reserve bool) error
+	// entry in positions and seqs.
+	StartForward(ctx ml.Context, batch input.Batch) error

 	// CopyPrefix copies tokens in the range [0, len) from srcSeq to dstSeq
 	CopyPrefix(srcSeq, dstSeq int, len int32)

-	// CanResume returns true if the cache can continue with the next token at
-	// the given position and sequence. Assumes that the caller has already
-	// verified the contents of the cache.
-	CanResume(seq int, pos int32) bool
-
 	// Remove deletes tokens in the range [beginIndex, endIndex) from seq. Set
 	// endIndex to math.MaxInt32 to remove everything starting at beginIndex.
 	//
--- a/kvcache/causal.go
+++ b/kvcache/causal.go
@@ -21,7 +21,6 @@ type shiftFn func(ctx ml.Context, layer int, key, shift ml.Tensor) (ml.Tensor, e
 type Causal struct {
 	DType      ml.DType
 	windowSize int32
-	chunkSize  int32

 	opts CausalOptions

@@ -98,17 +97,6 @@ func NewSWACache(windowSize int32, shift shiftFn) *Causal {
 	}
 }

-func NewChunkedAttentionCache(chunkSize int32, shift shiftFn) *Causal {
-	return &Causal{
-		windowSize: math.MaxInt32,
-		chunkSize:  chunkSize,
-		shiftFn:    shift,
-		ctxs:       make(map[int]ml.Context),
-		keys:       make(map[int]ml.Tensor),
-		values:     make(map[int]ml.Tensor),
-	}
-}
-
 func (c *Causal) Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity, maxBatch int) {
 	if c.config == nil {
 		var config ml.CacheConfig
@@ -131,10 +119,10 @@ func (c *Causal) Init(backend ml.Backend, dtype ml.DType, maxSequences, capacity
 	}

 	var cacheSize int
-	if c.windowSize == math.MaxInt32 || capacity < int(c.windowSize) {
+	if c.windowSize == math.MaxInt32 || capacity < int(c.windowSize)+maxBatch {
 		cacheSize = maxSequences * capacity
 	} else {
-		cacheSize = (maxSequences * int(c.windowSize)) + maxBatch
+		cacheSize = maxSequences * (int(c.windowSize) + maxBatch)
 	}
 	cacheSize = roundUp(cacheSize, c.config.CachePadding)
 	c.cells = make([]cacheCell, cacheSize)
@@ -158,60 +146,51 @@ func (c *Causal) Close() {
 	}
 }

-func (c *Causal) StartForward(ctx ml.Context, batch input.Batch, reserve bool) error {
+func (c *Causal) StartForward(ctx ml.Context, batch input.Batch) error {
 	c.curBatchSize = len(batch.Positions)
 	c.curSequences = batch.Sequences
 	c.curPositions = batch.Positions
 	c.opts.Except = nil

-	if !reserve {
-		c.updateSlidingWindow()
-
-		var err error
-		c.curLoc, err = c.findStartLoc()
-		if errors.Is(err, ErrKvCacheFull) {
-			c.defrag()
-			c.curLoc, err = c.findStartLoc()
-		}
-		if err != nil {
-			return err
-		}
-
-		c.curCellRange = newRange()
-		for i, pos := range batch.Positions {
-			seq := batch.Sequences[i]
-
-			c.cells[c.curLoc+i] = cacheCell{pos: pos, sequences: []int{seq}}
-
-			seqRange, ok := c.cellRanges[seq]
-			if !ok {
-				seqRange = newRange()
-			}
-
-			if c.curLoc+i > seqRange.max {
-				seqRange.max = c.curLoc + i
-			}
-			if seqRange.max > c.curCellRange.max {
-				c.curCellRange.max = seqRange.max
-			}
-
-			if c.curLoc+i < seqRange.min {
-				seqRange.min = c.curLoc + i
-			}
-			if seqRange.min < c.curCellRange.min {
-				c.curCellRange.min = seqRange.min
-			}
-			c.cellRanges[seq] = seqRange
-		}
-	} else {
-		// If we are reserving memory, don't update any of the cache metadata but set the size
-		// to the worst case.
-		c.curLoc = 0
-		c.curCellRange.min = 0
-		c.curCellRange.max = len(c.cells) - 1
-	}
+	c.updateSlidingWindow()

 	var err error
+	c.curLoc, err = c.findStartLoc()
+	if errors.Is(err, ErrKvCacheFull) {
+		c.defrag()
+		c.curLoc, err = c.findStartLoc()
+	}
+	if err != nil {
+		return err
+	}
+
+	c.curCellRange = newRange()
+	for i, pos := range batch.Positions {
+		seq := batch.Sequences[i]
+
+		c.cells[c.curLoc+i] = cacheCell{pos: pos, sequences: []int{seq}}
+
+		seqRange, ok := c.cellRanges[seq]
+		if !ok {
+			seqRange = newRange()
+		}
+
+		if c.curLoc+i > seqRange.max {
+			seqRange.max = c.curLoc + i
+		}
+		if seqRange.max > c.curCellRange.max {
+			c.curCellRange.max = seqRange.max
+		}
+
+		if c.curLoc+i < seqRange.min {
+			seqRange.min = c.curLoc + i
+		}
+		if seqRange.min < c.curCellRange.min {
+			c.curCellRange.min = seqRange.min
+		}
+		c.cellRanges[seq] = seqRange
+	}
+
 	c.curMask, err = c.buildMask(ctx)

 	return err
@@ -239,7 +218,7 @@ func (c *Causal) findStartLoc() (int, error) {
 		}
 	}

-	return 0, fmt.Errorf("%w (cache: %v batch: %v)", ErrKvCacheFull, len(c.cells), c.curBatchSize)
+	return 0, fmt.Errorf("%w (length: %v)", ErrKvCacheFull, len(c.cells))
 }

 func (c *Causal) updateSlidingWindow() {
@@ -312,7 +291,6 @@ func (c *Causal) buildMask(ctx ml.Context) (ml.Tensor, error) {
 		for j := c.curCellRange.min; j <= c.curCellRange.max; j++ {
 			if !slices.Contains(c.cells[j].sequences, c.curSequences[i]) ||
 				(enabled && c.cells[j].pos > c.curPositions[i]) ||
-				c.chunkSize > 0 && c.cells[j].pos < c.curPositions[i]-c.curPositions[i]%c.chunkSize ||
 				c.cells[j].pos < c.curPositions[i]-c.windowSize {
 				mask[i*length+(j-c.curCellRange.min)] = float32(math.Inf(-1))
 			}
@@ -603,35 +581,6 @@ func (c *Causal) CopyPrefix(srcSeq, dstSeq int, len int32) {
 	c.cellRanges[dstSeq] = seqRange
 }

-func (c *Causal) CanResume(seq int, pos int32) bool {
-	if c.windowSize == math.MaxInt32 {
-		return true
-	}
-
-	seqRange, ok := c.cellRanges[seq]
-	if !ok {
-		return false
-	}
-
-	// for sliding window, check that the window of the new sequence is contained in
-	// the window of what we are storing
-	var last int32 = -1
-	for i := seqRange.min; i <= seqRange.max; i++ {
-		if slices.Contains(c.cells[i].sequences, seq) {
-			last = max(last, c.cells[i].pos)
-		}
-	}
-
-	if last == -1 {
-		return false
-	}
-
-	lastWindowStart := max(0, last-c.windowSize)
-	posWindowStart := max(0, pos-c.windowSize)
-
-	return posWindowStart >= lastWindowStart
-}
-
 func (c *Causal) shift(seq int, beginIndex, offset int32) error {
 	if c.shiftFn == nil {
 		return ErrNotSupported
@@ -686,12 +635,6 @@ func (c *Causal) shift(seq int, beginIndex, offset int32) error {
 }

 func (c *Causal) Remove(seq int, beginIndex, endIndex int32) error {
-	// TODO(jessegross): We should check to see if removing the middle of the sequence will
-	// cause the sliding window to encompass tokens that we no longer have. If so, then we
-	// should return an error, which will trigger the runner to evaluate the full history and
-	// rebuild the window. However, if we have multimodal inputs in our history, this reuse
-	// results in use after free, so we don't do it for now.
-
 	var offset int32
 	if endIndex != math.MaxInt32 {
 		offset = beginIndex - endIndex
@@ -706,7 +649,8 @@ func (c *Causal) Remove(seq int, beginIndex, endIndex int32) error {
 			} else {
 				if c.cells[i].pos >= endIndex {
 					if slices.ContainsFunc(c.cells[i].sequences, func(s int) bool { return s != seq }) {
-						return errors.New("shifting cells shared by multiple sequences not supported")
+						// TODO(jessegross): Need to be careful about data shared between sequences
+						return errors.New("shifting on cells shared by multiple sequences not yet implemented")
 					}

 					c.cells[i].pos += offset
--- a/kvcache/causal_test.go
+++ b/kvcache/causal_test.go
@@ -86,64 +86,6 @@ func TestSWA(t *testing.T) {
 	testCache(t, backend, cache, tests)
 }

-func TestChunkedAttention(t *testing.T) {
-	cache := NewChunkedAttentionCache(2, nil)
-	defer cache.Close()
-
-	var b testBackend
-	cache.Init(&b, ml.DTypeF16, 1, 16, 16)
-
-	x := float32(math.Inf(-1))
-
-	testCache(
-		t, &b, cache,
-		[]testCase{
-			{
-				name:          "FirstBatch",
-				in:            []float32{1, 2, 3, 4},
-				inShape:       []int{1, 1, 4},
-				seqs:          []int{0, 0, 0, 0},
-				pos:           []int32{0, 1, 2, 3},
-				expected:      []float32{1, 2, 3, 4},
-				expectedShape: []int{1, 1, 4},
-				expectedMask: []float32{
-					0, x, x, x,
-					0, 0, x, x,
-					x, x, 0, x,
-					x, x, 0, 0,
-				},
-			},
-			{
-				name:          "SecondBatch",
-				in:            []float32{5, 6, 7},
-				inShape:       []int{1, 1, 3},
-				seqs:          []int{0, 0, 0},
-				pos:           []int32{4, 5, 6},
-				expected:      []float32{1, 2, 3, 4, 5, 6, 7},
-				expectedShape: []int{1, 1, 7},
-				expectedMask: []float32{
-					x, x, x, x, 0, x, x,
-					x, x, x, x, 0, 0, x,
-					x, x, x, x, x, x, 0,
-				},
-			},
-			{
-				name:          "ThirdBatch",
-				in:            []float32{8, 9},
-				inShape:       []int{1, 1, 2},
-				seqs:          []int{0, 0},
-				pos:           []int32{7, 8},
-				expected:      []float32{1, 2, 3, 4, 5, 6, 7, 8, 9},
-				expectedShape: []int{1, 1, 9},
-				expectedMask: []float32{
-					x, x, x, x, x, x, 0, 0, x,
-					x, x, x, x, x, x, x, x, 0,
-				},
-			},
-		},
-	)
-}
-
 func TestSequences(t *testing.T) {
 	backend := &testBackend{}
 	cache := NewCausalCache(nil)
@@ -338,7 +280,7 @@ func testCache(t *testing.T, backend ml.Backend, cache Cache, tests []testCase)
 			context := backend.NewContext()
 			defer context.Close()

-			err := cache.StartForward(context, input.Batch{Positions: test.pos, Sequences: test.seqs}, false)
+			err := cache.StartForward(context, input.Batch{Positions: test.pos, Sequences: test.seqs})
 			if err != nil {
 				panic(err)
 			}
@@ -351,94 +293,21 @@ func testCache(t *testing.T, backend ml.Backend, cache Cache, tests []testCase)

 			context.Forward(out, mask).Compute(out, mask)

-			if !slices.Equal(out.Floats(), test.expected) {
-				t.Errorf("TestCache: have %v; want %v", out.Floats(), test.expected)
-			}
-
-			if !slices.Equal(out.Shape(), test.expectedShape) {
-				t.Errorf("TestCache: has shape %v; want %v", out.Shape(), test.expectedShape)
-			}
-
-			if !slices.Equal(mask.Floats(), test.expectedMask) {
-				t.Errorf("TestCache: have mask: have %v want %v", mask.Floats(), test.expectedMask)
+			if !slices.Equal(out.Floats(), test.expected) || !slices.Equal(out.Shape(), test.expectedShape) || !slices.Equal(mask.Floats(), test.expectedMask) {
+				t.Errorf("TestCache: have %v (shape %v); want %v (shape %v); mask: have %v (shape %v) want %v", out.Floats(), out.Shape(), test.expected, test.expectedShape, mask.Floats(), mask.Shape(), test.expectedMask)
 			}
 		})
 	}
 }

-func TestCanResume(t *testing.T) {
-	backend := &testBackend{}
-	windowSize := int32(4)
-	cache := NewSWACache(windowSize, nil)
-	defer cache.Close()
+type testBackend struct{}

-	cache.Init(backend, ml.DTypeF16, 1, 16, 16)
-
-	context := backend.NewContext()
-	defer context.Close()
-
-	err := cache.StartForward(context, input.Batch{
-		Positions: []int32{0, 1, 2, 3},
-		Sequences: []int{0, 0, 0, 0},
-	}, false)
-	if err != nil {
-		t.Fatalf("StartForward failed: %v", err)
-	}
-
-	cache.SetLayer(0)
-	tensor, _ := context.FromFloatSlice([]float32{1, 2, 3, 4}, 1, 1, 4)
-	cache.Put(context, tensor, tensor)
-
-	// with window size 4, nothing has slid out of the window yet
-	if !cache.CanResume(0, 0) {
-		t.Errorf("CanResume(0, 0) = false, want true (within window)")
-	}
-	if !cache.CanResume(0, 1) {
-		t.Errorf("CanResume(0, 1) = false, want true (within window)")
-	}
-	if !cache.CanResume(0, 2) {
-		t.Errorf("CanResume(0, 2) = false, want true (within window)")
-	}
-	if !cache.CanResume(0, 3) {
-		t.Errorf("CanResume(0, 3) = false, want true (latest position)")
-	}
-
-	// shift window by adding position 4
-	err = cache.StartForward(context, input.Batch{
-		Positions: []int32{4, 5},
-		Sequences: []int{0, 0},
-	}, false)
-	if err != nil {
-		t.Fatalf("StartForward failed: %v", err)
-	}
-
-	cache.SetLayer(0)
-	tensor, _ = context.FromFloatSlice([]float32{5, 6}, 1, 1, 2)
-	cache.Put(context, tensor, tensor)
-
-	// only the latest position has overlapping windows
-	if cache.CanResume(0, 0) {
-		t.Errorf("after shift: CanResume(0, 0) = true, want false (outside window)")
-	}
-	if cache.CanResume(0, 1) {
-		t.Errorf("after shift: CanResume(0, 1) = true, want false (outside window)")
-	}
-	if cache.CanResume(0, 2) {
-		t.Errorf("after shift: CanResume(0, 2) = true, want false (outside window)")
-	}
-	if cache.CanResume(0, 3) {
-		t.Errorf("after shift: CanResume(0, 3) = true, want false (outside window)")
-	}
-	if cache.CanResume(0, 4) {
-		t.Errorf("after shift: CanResume(0, 4) = true, want false (outside window)")
-	}
-	if !cache.CanResume(0, 5) {
-		t.Errorf("after shift: CanResume(0, 5) = false, want true (latest position)")
-	}
+func (b *testBackend) Config() ml.Config {
+	panic("not implemented")
 }

-type testBackend struct {
-	ml.Backend
+func (b *testBackend) Get(name string) ml.Tensor {
+	panic("not implemented")
 }

 func (b *testBackend) NewContext() ml.Context {
@@ -449,10 +318,12 @@ func (b *testBackend) NewContextSize(int) ml.Context {
 	return &testContext{}
 }

-type testContext struct {
-	ml.Context
+func (b *testBackend) SystemInfo() string {
+	return "not implemented"
 }

+type testContext struct{}
+
 func (c *testContext) Empty(dtype ml.DType, shape ...int) ml.Tensor {
 	total := 0

@@ -490,26 +361,14 @@ func (c *testContext) FromIntSlice(s []int32, shape ...int) (ml.Tensor, error) {
 	return out, nil
 }

-func (c *testContext) Arange(start, stop, step float32, dtype ml.DType) ml.Tensor {
-	s := make([]float32, 0, int((stop-start)/step))
-	for i := start; i < stop; i += step {
-		s = append(s, i)
-	}
-
-	out, _ := c.FromFloatSlice(s, len(s))
-	out.(*testTensor).dtype = dtype
-	return out
-}
-
 func (c *testContext) Input() ml.Context    { return c }
+func (c *testContext) Output() ml.Context   { return c }
 func (c *testContext) Layer(int) ml.Context { return c }

 func (c *testContext) Forward(...ml.Tensor) ml.Context { return c }

 func (c *testContext) Compute(...ml.Tensor) {}

-func (c *testContext) Reserve() error { return nil }
-
 func (c *testContext) MaxGraphNodes() int {
 	return 10
 }
@@ -517,8 +376,6 @@ func (c *testContext) MaxGraphNodes() int {
 func (c *testContext) Close() {}

 type testTensor struct {
-	ml.Tensor
-
 	dtype       ml.DType
 	elementSize int
 	data        []float32
@@ -546,20 +403,16 @@ func (t *testTensor) DType() ml.DType {
 	return t.dtype
 }

+func (t *testTensor) Bytes() []byte {
+	panic("not implemented")
+}
+
 func (t *testTensor) Floats() []float32 {
 	out := make([]float32, len(t.data))
 	copy(out, t.data)
 	return out
 }

-func (t *testTensor) Neg(ctx ml.Context) ml.Tensor {
-	out := ctx.Empty(t.DType(), t.Shape()...).(*testTensor)
-	for i := range out.data {
-		out.data[i] = -t.data[i]
-	}
-	return out
-}
-
 func (t *testTensor) Add(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
 	out := ctx.Empty(t.DType(), t.Shape()...).(*testTensor)

@@ -570,6 +423,66 @@ func (t *testTensor) Add(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
 	return out
 }

+func (t *testTensor) Mul(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Mulmat(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) MulmatFullPrec(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Softmax(ctx ml.Context) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) LayerNorm(ctx ml.Context, weight, bias ml.Tensor, eps float32) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) RMSNorm(ctx ml.Context, weight ml.Tensor, eps float32) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Scale(ctx ml.Context, s float64) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) AvgPool1D(ctx ml.Context, k, s, p int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) AvgPool2D(ctx ml.Context, k, s int, p float32) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Conv2D(ctx ml.Context, weight ml.Tensor, s0, s1, p0, p1, d0, d1 int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) RoPE(ctx ml.Context, positionIDs, ropeFactors ml.Tensor, dim, ropeType uint32, base, scale float32) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Tanh(ctx ml.Context) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) GELU(ctx ml.Context) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) SILU(ctx ml.Context) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Reshape(ctx ml.Context, shape ...int) ml.Tensor {
+	panic("not implemented")
+}
+
 func (t *testTensor) View(ctx ml.Context, offset int, shape ...int) ml.Tensor {
 	offset /= t.elementSize

@@ -592,6 +505,38 @@ func (t *testTensor) View(ctx ml.Context, offset int, shape ...int) ml.Tensor {
 	return view
 }

+func (t *testTensor) Permute(ctx ml.Context, shape ...int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Contiguous(ctx ml.Context) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Set(ctx ml.Context, t2 ml.Tensor, offset int, strides ...int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Pad(ctx ml.Context, shape ...int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Unpad(ctx ml.Context, shape ...int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Stack(ctx ml.Context, dim int, s ...ml.Tensor) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Concat(ctx ml.Context, t2 ml.Tensor, dim int) ml.Tensor {
+	panic("not implemented")
+}
+
+func (t *testTensor) Rows(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
+	panic("not implemented")
+}
+
 func (t *testTensor) Copy(ctx ml.Context, t2 ml.Tensor) ml.Tensor {
 	copy(t2.(*testTensor).data, t.data)
 	return nil
--- a/kvcache/encoder.go
+++ b/kvcache/encoder.go
@@ -27,11 +27,6 @@ type EncoderCache struct {
 	// anything will be stored)
 	curPos int32

-	// curReserve indicates that this forward pass is only for
-	// memory reservation and we should not update our metadata
-	// based on it.
-	curReserve bool
-
 	// ** cache metadata **

 	// was something stored in the cache?
@@ -88,14 +83,12 @@ func (c *EncoderCache) Close() {
 	}
 }

-func (c *EncoderCache) StartForward(ctx ml.Context, batch input.Batch, reserve bool) error {
+func (c *EncoderCache) StartForward(ctx ml.Context, batch input.Batch) error {
 	// We work with the most recent image
 	if len(batch.Multimodal) > 0 {
 		c.curPos = batch.Positions[batch.Multimodal[len(batch.Multimodal)-1].Index]
 	}

-	c.curReserve = reserve
-
 	return nil
 }

@@ -112,10 +105,8 @@ func (c *EncoderCache) Get(ctx ml.Context) (ml.Tensor, ml.Tensor, ml.Tensor) {
 }

 func (c *EncoderCache) Put(ctx ml.Context, key, value ml.Tensor) {
-	if !c.curReserve {
-		c.encoderPos = c.curPos
-		c.encoderCached = true
-	}
+	c.encoderPos = c.curPos
+	c.encoderCached = true

 	if c.config.PermutedV {
 		value = value.Permute(ctx, 1, 2, 0, 3)
@@ -143,10 +134,6 @@ func (c *EncoderCache) CopyPrefix(srcSeq, dstSeq int, len int32) {
 	panic("encoder cache does not support multiple sequences")
 }

-func (c *EncoderCache) CanResume(seq int, pos int32) bool {
-	return true
-}
-
 func (c *EncoderCache) Remove(seq int, beginIndex, endIndex int32) error {
 	if c.encoderPos >= beginIndex && c.encoderPos < endIndex {
 		c.encoderCached = false
--- a/kvcache/wrapper.go
+++ b/kvcache/wrapper.go
@@ -41,9 +41,9 @@ func (c *WrapperCache) Close() {
 	}
 }

-func (c *WrapperCache) StartForward(ctx ml.Context, batch input.Batch, reserve bool) error {
+func (c *WrapperCache) StartForward(ctx ml.Context, batch input.Batch) error {
 	for i, cache := range c.caches {
-		err := cache.StartForward(ctx, batch, reserve)
+		err := cache.StartForward(ctx, batch)
 		if err != nil {
 			// unwind on error - Remove with endIndex set to math.MaxInt32 does not fail
 			for j := i - 1; j >= 0; j-- {
@@ -87,16 +87,6 @@ func (c *WrapperCache) CopyPrefix(srcSeq, dstSeq int, len int32) {
 	}
 }

-func (c *WrapperCache) CanResume(seq int, pos int32) bool {
-	for _, cache := range c.caches {
-		if !cache.CanResume(seq, pos) {
-			return false
-		}
-	}
-
-	return true
-}
-
 func (c *WrapperCache) Remove(seq int, beginIndex, endIndex int32) error {
 	// If the one of these fails, the caller is supposed to retry with endIndex set to math.MaxInt32, which should not fail
 	for _, cache := range c.caches {
--- a/llama/build-info.cpp
+++ b/llama/build-info.cpp
@@ -1,4 +1,4 @@
 int LLAMA_BUILD_NUMBER = 0;
-char const *LLAMA_COMMIT = "de4c07f93783a1a96456a44dc16b9db538ee1618";
+char const *LLAMA_COMMIT = "d7cfe1ffe0f435d0048a6058d529daf76e072d9c";
 char const *LLAMA_COMPILER = "";
 char const *LLAMA_BUILD_TARGET = "";
--- a/Show More
+++ b/Show More