Merge https://github.com/ollama/ollama

Support image input for OpenAI chat compatibility (#5208 )
* OpenAI v1 models * Refactor Writers * Add Test Co-Authored-By: Attila Kerekes * Credit Co-Author Co-Authored-By: Attila Kerekes <439392+keriati@users.noreply.github.com> * Empty List Testing * Use Namespace for Ownedby * Update Test * Add back envconfig * v1/models docs * Use ModelName Parser * Test Names * Remove Docs * Clean Up * Test name Co-authored-by: Jeffrey Morgan <jmorganca@gmail.com> * Add Middleware for Chat and List * Testing Cleanup * Test with Fatal * Add functionality to chat test * Support image input for OpenAI chat * Decoding * Fix message processing logic * openai vision test * type errors * clean up * redundant check * merge conflicts * merge conflicts * merge conflicts * flattening and smaller image * add test * support python and js SDKs and mandate prefixing * clean up --------- Co-authored-by: Attila Kerekes <439392+keriati@users.noreply.github.com> Co-authored-by: Jeffrey Morgan <jmorganca@gmail.com>
2024-07-14 16:51:20 +05:30 · 2024-07-13 22:07:45 -07:00 · 2024-07-13 20:56:24 -07:00 · 2024-07-13 15:08:00 -07:00 · 2024-07-13 09:25:31 -07:00 · 2024-07-13 09:20:05 -07:00
109 changed files with 1730 additions and 1090 deletions
--- a/.github/workflows/release.yaml
+++ b/.github/workflows/release.yaml
@ -147,7 +147,7 @@ jobs:
        run: |
          $ErrorActionPreference = "Stop"
          write-host "downloading AMD HIP Installer"
-          Invoke-WebRequest -Uri "https://download.amd.com/developer/eula/rocm-hub/AMD-Software-PRO-Edition-23.Q4-WinSvr2022-For-HIP.exe" -OutFile "${env:RUNNER_TEMP}\rocm-install.exe"
+          Invoke-WebRequest -Uri "https://download.amd.com/developer/eula/rocm-hub/AMD-Software-PRO-Edition-24.Q3-WinSvr2022-For-HIP.exe" -OutFile "${env:RUNNER_TEMP}\rocm-install.exe"
          write-host "Installing AMD HIP"
          Start-Process "${env:RUNNER_TEMP}\rocm-install.exe" -ArgumentList '-install' -NoNewWindow -Wait
          write-host "Completed AMD HIP"
@ -304,10 +304,6 @@ jobs:
          write-host "Installing plugin"
          & "${env:RUNNER_TEMP}\plugin\*\kmscng.msi" /quiet
          write-host "plugin installed"
      - name: remove unwanted mingw dll.a files
        run: |
          Remove-Item "C:\mingw64\x86_64-w64-mingw32\lib\libpthread.dll.a"
          Remove-Item "C:\mingw64\x86_64-w64-mingw32\lib\libwinpthread.dll.a"
      - uses: actions/setup-go@v5
        with:
          go-version-file: go.mod
--- a/.github/workflows/test.yaml
+++ b/.github/workflows/test.yaml
@ -169,7 +169,7 @@ jobs:
        run: |
          $ErrorActionPreference = "Stop"
          write-host "downloading AMD HIP Installer"
-          Invoke-WebRequest -Uri "https://download.amd.com/developer/eula/rocm-hub/AMD-Software-PRO-Edition-23.Q4-WinSvr2022-For-HIP.exe" -OutFile "${env:RUNNER_TEMP}\rocm-install.exe"
+          Invoke-WebRequest -Uri "https://download.amd.com/developer/eula/rocm-hub/AMD-Software-PRO-Edition-24.Q3-WinSvr2022-For-HIP.exe" -OutFile "${env:RUNNER_TEMP}\rocm-install.exe"
          write-host "Installing AMD HIP"
          Start-Process "${env:RUNNER_TEMP}\rocm-install.exe" -ArgumentList '-install' -NoNewWindow -Wait
          write-host "Completed AMD HIP"
--- a/README.md
+++ b/README.md
@ -293,6 +293,7 @@ See the [API documentation](./docs/api.md) for all endpoints.
 - [OllamaSpring](https://github.com/CrazyNeil/OllamaSpring) (Ollama Client for macOS)
 - [LLocal.in](https://github.com/kartikm7/llocal) (Easy to use Electron Desktop Client for Ollama)
 - [Ollama with Google Mesop](https://github.com/rapidarchitect/ollama_mesop/) (Mesop Chat Client implementation with Ollama)
 - [Kerlig AI](https://www.kerlig.com/) (AI writing assistant for macOS)
 ### Terminal
--- a/api/types.go
+++ b/api/types.go
@ -221,6 +221,8 @@ type DeleteRequest struct {
 type ShowRequest struct {
 	Model    string `json:"model"`
 	System   string `json:"system"`
 	// Template is deprecated
 	Template string `json:"template"`
 	Verbose  bool   `json:"verbose"`
--- a/app/ollama.iss
+++ b/app/ollama.iss
@ -127,6 +127,10 @@ Type: filesandordirs; Name: "{%USERPROFILE}\.ollama\models"
 Type: filesandordirs; Name: "{%USERPROFILE}\.ollama\history"
 ; NOTE: if the user has a custom OLLAMA_MODELS it will be preserved
 [InstallDelete]
 Type: filesandordirs; Name: "{%TEMP}\ollama*"
 Type: filesandordirs; Name: "{%LOCALAPPDATA}\Programs\Ollama"
 [Messages]
 WizardReady=Ollama Windows Preview
 ReadyLabel1=%nLet's get you up and running with your own large language models.
--- a/cmd/cmd.go
+++ b/cmd/cmd.go
@ -843,7 +843,6 @@ type runOptions struct {
 	WordWrap    bool
 	Format      string
 	System      string
 	Template    string
 	Images      []api.ImageData
 	Options     map[string]interface{}
 	MultiModal  bool
@ -1037,7 +1036,6 @@ func generate(cmd *cobra.Command, opts runOptions) error {
 		Images:    opts.Images,
 		Format:    opts.Format,
 		System:    opts.System,
 		Template:  opts.Template,
 		Options:   opts.Options,
 		KeepAlive: opts.KeepAlive,
 	}
--- a/cmd/interactive.go
+++ b/cmd/interactive.go
@ -27,7 +27,6 @@ const (
 	MultilineNone MultilineState = iota
 	MultilinePrompt
 	MultilineSystem
 	MultilineTemplate
 )
 func loadModel(cmd *cobra.Command, opts *runOptions) error {
@ -94,7 +93,6 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 		fmt.Fprintln(os.Stderr, "Available Commands:")
 		fmt.Fprintln(os.Stderr, "  /set parameter ...     Set a parameter")
 		fmt.Fprintln(os.Stderr, "  /set system <string>   Set system message")
 		fmt.Fprintln(os.Stderr, "  /set template <string> Set prompt template")
 		fmt.Fprintln(os.Stderr, "  /set history           Enable history")
 		fmt.Fprintln(os.Stderr, "  /set nohistory         Disable history")
 		fmt.Fprintln(os.Stderr, "  /set wordwrap          Enable wordwrap")
@ -204,10 +202,6 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 				opts.Messages = append(opts.Messages, api.Message{Role: "system", Content: opts.System})
 				fmt.Println("Set system message.")
 				sb.Reset()
 			case MultilineTemplate:
 				opts.Template = sb.String()
 				fmt.Println("Set prompt template.")
 				sb.Reset()
 			}
 			multiline = MultilineNone
@ -326,17 +320,13 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 					}
 					fmt.Printf("Set parameter '%s' to '%s'\n", args[2], strings.Join(params, ", "))
 					opts.Options[args[2]] = fp[args[2]]
-				case "system", "template":
+				case "system":
 					if len(args) < 3 {
 						usageSet()
 						continue
 					}
-					if args[1] == "system" {
+					multiline = MultilineSystem
 						multiline = MultilineSystem
 					} else if args[1] == "template" {
 						multiline = MultilineTemplate
 					}
 					line := strings.Join(args[2:], " ")
 					line, ok := strings.CutPrefix(line, `"""`)
@ -356,23 +346,17 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 						continue
 					}
-					if args[1] == "system" {
+					opts.System = sb.String() // for display in modelfile
-						opts.System = sb.String() // for display in modelfile
+					newMessage := api.Message{Role: "system", Content: sb.String()}
-						newMessage := api.Message{Role: "system", Content: sb.String()}
+					// Check if the slice is not empty and the last message is from 'system'
-						// Check if the slice is not empty and the last message is from 'system'
+					if len(opts.Messages) > 0 && opts.Messages[len(opts.Messages)-1].Role == "system" {
-						if len(opts.Messages) > 0 && opts.Messages[len(opts.Messages)-1].Role == "system" {
+						// Replace the last message
-							// Replace the last message
+						opts.Messages[len(opts.Messages)-1] = newMessage
-							opts.Messages[len(opts.Messages)-1] = newMessage
+					} else {
-						} else {
+						opts.Messages = append(opts.Messages, newMessage)
 							opts.Messages = append(opts.Messages, newMessage)
 						}
 						fmt.Println("Set system message.")
 						sb.Reset()
 					} else if args[1] == "template" {
 						opts.Template = sb.String()
 						fmt.Println("Set prompt template.")
 						sb.Reset()
 					}
 					fmt.Println("Set system message.")
 					sb.Reset()
 					sb.Reset()
 					continue
@ -393,7 +377,6 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 				req := &api.ShowRequest{
 					Name:     opts.Model,
 					System:   opts.System,
 					Template: opts.Template,
 					Options:  opts.Options,
 				}
 				resp, err := client.Show(cmd.Context(), req)
@ -437,12 +420,9 @@ func generateInteractive(cmd *cobra.Command, opts runOptions) error {
 						fmt.Println("No system message was specified for this model.")
 					}
 				case "template":
-					switch {
+					if resp.Template != "" {
 					case opts.Template != "":
 						fmt.Println(opts.Template + "\n")
 					case resp.Template != "":
 						fmt.Println(resp.Template)
-					default:
+					} else {
 						fmt.Println("No prompt template was specified for this model.")
 					}
 				default:
@ -536,10 +516,6 @@ func buildModelfile(opts runOptions) string {
 		fmt.Fprintf(&mf, "SYSTEM \"\"\"%s\"\"\"\n", opts.System)
 	}
 	if opts.Template != "" {
 		fmt.Fprintf(&mf, "TEMPLATE \"\"\"%s\"\"\"\n", opts.Template)
 	}
 	keys := make([]string, 0)
 	for k := range opts.Options {
 		keys = append(keys, k)
--- a/cmd/interactive_test.go
+++ b/cmd/interactive_test.go
@ -59,7 +59,6 @@ func TestModelfileBuilder(t *testing.T) {
 	opts := runOptions{
 		Model:    "hork",
 		System:   "You are part horse and part shark, but all hork. Do horklike things",
 		Template: "This is a template.",
 		Messages: []api.Message{
 			{Role: "user", Content: "Hey there hork!"},
 			{Role: "assistant", Content: "Yes it is true, I am half horse, half shark."},
@ -75,7 +74,6 @@ func TestModelfileBuilder(t *testing.T) {
 	mf := buildModelfile(opts)
 	expectedModelfile := `FROM {{.Model}}
 SYSTEM """{{.System}}"""
 TEMPLATE """{{.Template}}"""
 PARAMETER penalize_newline false
 PARAMETER seed 42
 PARAMETER stop [hi there]
@ -97,7 +95,6 @@ MESSAGE assistant """Yes it is true, I am half horse, half shark."""
 	mf = buildModelfile(opts)
 	expectedModelfile = `FROM {{.ParentModel}}
 SYSTEM """{{.System}}"""
 TEMPLATE """{{.Template}}"""
 PARAMETER penalize_newline false
 PARAMETER seed 42
 PARAMETER stop [hi there]
--- a/docs/faq.md
+++ b/docs/faq.md
@ -272,4 +272,4 @@ The following server settings may be used to adjust how Ollama handles concurren
 - `OLLAMA_NUM_PARALLEL` - The maximum number of parallel requests each model will process at the same time.  The default will auto-select either 4 or 1 based on available memory.
 - `OLLAMA_MAX_QUEUE` - The maximum number of requests Ollama will queue when busy before rejecting additional requests. The default is 512
-Note: Windows with Radeon GPUs currently default to 1 model maximum due to limitations in ROCm v5.7 for available VRAM reporting.  Once ROCm v6 is available, Windows Radeon will follow the defaults above.  You may enable concurrent model loads on Radeon on Windows, but ensure you don't load more models than will fit into your GPUs VRAM.
+Note: Windows with Radeon GPUs currently default to 1 model maximum due to limitations in ROCm v5.7 for available VRAM reporting.  Once ROCm v6.2 is available, Windows Radeon will follow the defaults above.  You may enable concurrent model loads on Radeon on Windows, but ensure you don't load more models than will fit into your GPUs VRAM.
--- a/go.mod
+++ b/go.mod
@ -18,6 +18,7 @@ require (
 require (
 	github.com/agnivade/levenshtein v1.1.1
 	github.com/d4l3k/go-bfloat16 v0.0.0-20211005043715-690c3bdd05f1
 	github.com/google/go-cmp v0.6.0
 	github.com/mattn/go-runewidth v0.0.14
 	github.com/nlpodyssey/gopickle v0.3.0
 	github.com/pdevine/tensor v0.0.0-20240510204454-f88f4562727c
@ -71,7 +72,7 @@ require (
 	golang.org/x/net v0.25.0 // indirect
 	golang.org/x/sys v0.20.0
 	golang.org/x/term v0.20.0
-	golang.org/x/text v0.15.0 // indirect
+	golang.org/x/text v0.15.0
 	google.golang.org/protobuf v1.34.1
 	gopkg.in/yaml.v3 v3.0.1 // indirect
 )
--- a/gpu/amd_common.go
+++ b/gpu/amd_common.go
@ -49,9 +49,17 @@ func rocmGetVisibleDevicesEnv(gpuInfo []GpuInfo) (string, string) {
 }
 func commonAMDValidateLibDir() (string, error) {
-	// We try to favor system paths first, so that we can wire up the subprocess to use
+	// Favor our bundled version
-	// the system version.  Only use our bundled version if the system version doesn't work
+
-	// This gives users a more recovery options if versions have subtle problems at runtime
+	// Installer payload location if we're running the installed binary
 	exe, err := os.Executable()
 	if err == nil {
 		rocmTargetDir := filepath.Join(filepath.Dir(exe), "rocm")
 		if rocmLibUsable(rocmTargetDir) {
 			slog.Debug("detected ROCM next to ollama executable " + rocmTargetDir)
 			return rocmTargetDir, nil
 		}
 	}
 	// Prefer explicit HIP env var
 	hipPath := os.Getenv("HIP_PATH")
@ -87,14 +95,5 @@ func commonAMDValidateLibDir() (string, error) {
 		}
 	}
 	// Installer payload location if we're running the installed binary
 	exe, err := os.Executable()
 	if err == nil {
 		rocmTargetDir := filepath.Join(filepath.Dir(exe), "rocm")
 		if rocmLibUsable(rocmTargetDir) {
 			slog.Debug("detected ROCM next to ollama executable " + rocmTargetDir)
 			return rocmTargetDir, nil
 		}
 	}
 	return "", fmt.Errorf("no suitable rocm found, falling back to CPU")
 }
--- a/gpu/amd_hip_windows.go
+++ b/gpu/amd_hip_windows.go
@ -84,9 +84,8 @@ func (hl *HipLib) AMDDriverVersion() (driverMajor, driverMinor int, err error) {
 	}
 	slog.Debug("hipDriverGetVersion", "version", version)
-	// TODO - this isn't actually right, but the docs claim hipDriverGetVersion isn't accurate anyway...
+	driverMajor = version / 10000000
-	driverMajor = version / 1000
+	driverMinor = (version - (driverMajor * 10000000)) / 100000
 	driverMinor = (version - (driverMajor * 1000)) / 10
 	return driverMajor, driverMinor, nil
 }
--- a/gpu/amd_windows.go
+++ b/gpu/amd_windows.go
@ -22,8 +22,8 @@ const (
 var (
 	// Used to validate if the given ROCm lib is usable
-	ROCmLibGlobs          = []string{"hipblas.dll", "rocblas"}                 // TODO - probably include more coverage of files here...
+	ROCmLibGlobs          = []string{"hipblas.dll", "rocblas"}                 // This is not sufficient to discern v5 vs v6
-	RocmStandardLocations = []string{"C:\\Program Files\\AMD\\ROCm\\5.7\\bin"} // TODO glob?
+	RocmStandardLocations = []string{"C:\\Program Files\\AMD\\ROCm\\6.1\\bin"} // TODO glob?
 )
 func AMDGetGPUInfo() []RocmGPUInfo {
@ -35,12 +35,11 @@ func AMDGetGPUInfo() []RocmGPUInfo {
 	}
 	defer hl.Release()
-	// TODO - this reports incorrect version information, so omitting for now
+	driverMajor, driverMinor, err := hl.AMDDriverVersion()
-	// driverMajor, driverMinor, err := hl.AMDDriverVersion()
+	if err != nil {
-	// if err != nil {
+		// For now this is benign, but we may eventually need to fail compatibility checks
-	// 	// For now this is benign, but we may eventually need to fail compatibility checks
+		slog.Debug("error looking up amd driver version", "error", err)
-	// 	slog.Debug("error looking up amd driver version", "error", err)
+	}
 	// }
 	// Note: the HIP library automatically handles subsetting to any HIP_VISIBLE_DEVICES the user specified
 	count := hl.HipGetDeviceCount()
@ -132,10 +131,8 @@ func AMDGetGPUInfo() []RocmGPUInfo {
 				MinimumMemory:  rocmMinimumMemory,
 				Name:           name,
 				Compute:        gfx,
-
+				DriverMajor:    driverMajor,
-				// TODO - this information isn't accurate on windows, so don't report it until we find the right way to retrieve
+				DriverMinor:    driverMinor,
 				// DriverMajor:    driverMajor,
 				// DriverMinor:    driverMinor,
 			},
 			index: i,
 		}
--- a/gpu/gpu.go
+++ b/gpu/gpu.go
@ -274,6 +274,28 @@ func GetGPUInfo() GpuInfoList {
 				gpuInfo.DriverMajor = driverMajor
 				gpuInfo.DriverMinor = driverMinor
 				// query the management library as well so we can record any skew between the two
 				// which represents overhead on the GPU we must set aside on subsequent updates
 				if cHandles.nvml != nil {
 					C.nvml_get_free(*cHandles.nvml, C.int(gpuInfo.index), &memInfo.free, &memInfo.total, &memInfo.used)
 					if memInfo.err != nil {
 						slog.Warn("error looking up nvidia GPU memory", "error", C.GoString(memInfo.err))
 						C.free(unsafe.Pointer(memInfo.err))
 					} else {
 						if memInfo.free != 0 && uint64(memInfo.free) > gpuInfo.FreeMemory {
 							gpuInfo.OSOverhead = uint64(memInfo.free) - gpuInfo.FreeMemory
 							slog.Info("detected OS VRAM overhead",
 								"id", gpuInfo.ID,
 								"library", gpuInfo.Library,
 								"compute", gpuInfo.Compute,
 								"driver", fmt.Sprintf("%d.%d", gpuInfo.DriverMajor, gpuInfo.DriverMinor),
 								"name", gpuInfo.Name,
 								"overhead", format.HumanBytes2(gpuInfo.OSOverhead),
 							)
 						}
 					}
 				}
 				// TODO potentially sort on our own algorithm instead of what the underlying GPU library does...
 				cudaGPUs = append(cudaGPUs, gpuInfo)
 			}
@ -338,14 +360,17 @@ func GetGPUInfo() GpuInfoList {
 					"before",
 					"total", format.HumanBytes2(cpus[0].TotalMemory),
 					"free", format.HumanBytes2(cpus[0].FreeMemory),
 					"free_swap", format.HumanBytes2(cpus[0].FreeSwap),
 				),
 				slog.Group(
 					"now",
 					"total", format.HumanBytes2(mem.TotalMemory),
 					"free", format.HumanBytes2(mem.FreeMemory),
 					"free_swap", format.HumanBytes2(mem.FreeSwap),
 				),
 			)
 			cpus[0].FreeMemory = mem.FreeMemory
 			cpus[0].FreeSwap = mem.FreeSwap
 		}
 		var memInfo C.mem_info_t
@ -374,9 +399,14 @@ func GetGPUInfo() GpuInfoList {
 				slog.Warn("error looking up nvidia GPU memory")
 				continue
 			}
 			if cHandles.nvml != nil && gpu.OSOverhead > 0 {
 				// When using the management library update based on recorded overhead
 				memInfo.free -= C.uint64_t(gpu.OSOverhead)
 			}
 			slog.Debug("updating cuda memory data",
 				"gpu", gpu.ID,
 				"name", gpu.Name,
 				"overhead", format.HumanBytes2(gpu.OSOverhead),
 				slog.Group(
 					"before",
 					"total", format.HumanBytes2(gpu.TotalMemory),
--- a/gpu/gpu_darwin.go
+++ b/gpu/gpu_darwin.go
@ -56,7 +56,8 @@ func GetCPUInfo() GpuInfoList {
 func GetCPUMem() (memInfo, error) {
 	return memInfo{
 		TotalMemory: uint64(C.getPhysicalMemory()),
-		FreeMemory:  0,
+		FreeMemory:  uint64(C.getFreeMemory()),
 		// FreeSwap omitted as Darwin uses dynamic paging
 	}, nil
 }
--- a/gpu/gpu_info_darwin.h
+++ b/gpu/gpu_info_darwin.h
@ -2,3 +2,4 @@
 #include <stdint.h>
 uint64_t getRecommendedMaxVRAM();
 uint64_t getPhysicalMemory();
 uint64_t getFreeMemory();
--- a/gpu/gpu_info_darwin.m
+++ b/gpu/gpu_info_darwin.m
@ -1,4 +1,5 @@
-// go:build darwin
+#import <Foundation/Foundation.h>
 #import <mach/mach.h>
 #include "gpu_info_darwin.h"
 uint64_t getRecommendedMaxVRAM() {
@ -8,6 +9,27 @@ uint64_t getRecommendedMaxVRAM() {
  return result;
 }
 // getPhysicalMemory returns the total physical memory in bytes
 uint64_t getPhysicalMemory() {
-  return [[NSProcessInfo processInfo] physicalMemory];
+  return [NSProcessInfo processInfo].physicalMemory;
 }
 // getFreeMemory returns the total free memory in bytes, including inactive
 // memory that can be reclaimed by the system.
 uint64_t getFreeMemory() {
  mach_port_t host_port = mach_host_self();
  mach_msg_type_number_t host_size = sizeof(vm_statistics64_data_t) / sizeof(integer_t);
  vm_size_t pagesize;
  vm_statistics64_data_t vm_stat;
  host_page_size(host_port, &pagesize);
  if (host_statistics64(host_port, HOST_VM_INFO64, (host_info64_t)&vm_stat, &host_size) != KERN_SUCCESS) {
    return 0;
  }
  uint64_t free_memory = (uint64_t)vm_stat.free_count * pagesize;
  free_memory += (uint64_t)vm_stat.speculative_count * pagesize;
  free_memory += (uint64_t)vm_stat.inactive_count * pagesize;
  return free_memory;
 }
--- a/gpu/gpu_linux.go
+++ b/gpu/gpu_linux.go
@ -50,7 +50,7 @@ var OneapiMgmtName = "libze_intel_gpu.so"
 func GetCPUMem() (memInfo, error) {
 	var mem memInfo
-	var total, available, free, buffers, cached uint64
+	var total, available, free, buffers, cached, freeSwap uint64
 	f, err := os.Open("/proc/meminfo")
 	if err != nil {
 		return mem, err
@ -70,20 +70,21 @@ func GetCPUMem() (memInfo, error) {
 			_, err = fmt.Sscanf(line, "Buffers:%d", &buffers)
 		case strings.HasPrefix(line, "Cached:"):
 			_, err = fmt.Sscanf(line, "Cached:%d", &cached)
 		case strings.HasPrefix(line, "SwapFree:"):
 			_, err = fmt.Sscanf(line, "SwapFree:%d", &freeSwap)
 		default:
 			continue
 		}
 		if err != nil {
 			return mem, err
 		}
 		if total > 0 && available > 0 {
 			mem.TotalMemory = total * format.KibiByte
 			mem.FreeMemory = available * format.KibiByte
 			return mem, nil
 		}
 	}
 	mem.TotalMemory = total * format.KibiByte
-	mem.FreeMemory = (free + buffers + cached) * format.KibiByte
+	mem.FreeSwap = freeSwap * format.KibiByte
 	if available > 0 {
 		mem.FreeMemory = available * format.KibiByte
 	} else {
 		mem.FreeMemory = (free + buffers + cached) * format.KibiByte
 	}
 	return mem, nil
 }
--- a/gpu/gpu_windows.go
+++ b/gpu/gpu_windows.go
@ -51,5 +51,5 @@ func GetCPUMem() (memInfo, error) {
 	if r1 == 0 {
 		return memInfo{}, fmt.Errorf("GlobalMemoryStatusEx failed: %w", err)
 	}
-	return memInfo{TotalMemory: memStatus.TotalPhys, FreeMemory: memStatus.AvailPhys}, nil
+	return memInfo{TotalMemory: memStatus.TotalPhys, FreeMemory: memStatus.AvailPhys, FreeSwap: memStatus.AvailPageFile}, nil
 }
--- a/gpu/types.go
+++ b/gpu/types.go
@ -10,6 +10,7 @@ import (
 type memInfo struct {
 	TotalMemory uint64 `json:"total_memory,omitempty"`
 	FreeMemory  uint64 `json:"free_memory,omitempty"`
 	FreeSwap    uint64 `json:"free_swap,omitempty"`
 }
 // Beginning of an `ollama info` command
@ -52,7 +53,8 @@ type CPUInfo struct {
 type CudaGPUInfo struct {
 	GpuInfo
-	index int //nolint:unused,nolintlint
+	OSOverhead uint64 // Memory overhead between the driver library and management library
 	index      int    //nolint:unused,nolintlint
 }
 type CudaGPUInfoList []CudaGPUInfo
--- a/llm/ext_server/server.cpp
+++ b/llm/ext_server/server.cpp
@ -1413,7 +1413,7 @@ struct llama_server_context
            return get_slot(-1);
        }
-        LOG_INFO("slot with common prefix found", {{
+        LOG_DEBUG("slot with common prefix found", {{
            "slot_id", slot->id,
            "characters", longest
        }});
@ -1688,22 +1688,8 @@ struct llama_server_context
                    }
                    slot.params.n_keep = std::min(slot.n_ctx - 4, slot.params.n_keep);
                    char buf[256];
                    llama_model_meta_val_str(model, "general.architecture", buf, 256);
                    bool gemma2 = strcmp(buf, "gemma2") == 0;
                    int32_t truncate_at = slot.n_ctx;
                    // truncate at 2/3 of the context length for gemma2 models
                    // as they do not support context shifts (from the sliding window implementation).
                    // this way, prompts that almost fit the context length can still generate a full
                    // response without a sudden stop from hitting the context limit
                    if (gemma2) {
                        truncate_at = 2 * slot.n_ctx / 3;
                    }
                    // if input prompt is too big, truncate it, if group attention self-extend is disabled
-                    if (slot.ga_n == 1 && slot.n_prompt_tokens >= truncate_at)
+                    if (slot.ga_n == 1 && slot.n_prompt_tokens >= slot.n_ctx)
                    {
                        const int n_left = slot.n_ctx - slot.params.n_keep;
                        const int n_shift = n_left / 2;
@ -1731,19 +1717,6 @@ struct llama_server_context
                        GGML_ASSERT(slot.n_prompt_tokens < slot.n_ctx);
                    }
                    // Models with sliding window attention do not work with context shifts, so
                    // limit their prediction to the context length
                    if (gemma2) {
                        int32_t limit = slot.n_ctx - slot.n_prompt_tokens;
                        slot.n_predict = limit;
                        slot.params.n_predict = limit;
                        LOG_INFO("model does not support sliding window, limiting generation", {
                            {"n_ctx", slot.n_ctx},
                            {"n_prompt_tokens", slot.n_prompt_tokens},
                            {"n_predict", slot.n_predict}
                        });
                    }
                    if (!slot.params.cache_prompt)
                    {
                        llama_sampling_reset(slot.ctx_sampling);
--- a/llm/generate/gen_linux.sh
+++ b/llm/generate/gen_linux.sh
@ -77,7 +77,7 @@ if [ -z "${OLLAMA_SKIP_CPU_GENERATE}" ]; then
    if [ -n "${OLLAMA_CUSTOM_CPU_DEFS}" ]; then
        init_vars
        echo "OLLAMA_CUSTOM_CPU_DEFS=\"${OLLAMA_CUSTOM_CPU_DEFS}\""
-        CMAKE_DEFS="${OLLAMA_CUSTOM_CPU_DEFS} -DCMAKE_POSITION_INDEPENDENT_CODE=on ${CMAKE_DEFS}"
+        CMAKE_DEFS="${OLLAMA_CUSTOM_CPU_DEFS} -DBUILD_SHARED_LIBS=off -DCMAKE_POSITION_INDEPENDENT_CODE=on ${CMAKE_DEFS}"
        BUILD_DIR="../build/linux/${ARCH}/cpu"
        echo "Building custom CPU"
        build
@ -93,7 +93,7 @@ if [ -z "${OLLAMA_SKIP_CPU_GENERATE}" ]; then
        # -DGGML_AVX512_VBMI -- 2018 Intel Cannon Lake
        # -DGGML_AVX512_VNNI -- 2021 Intel Alder Lake
-        COMMON_CPU_DEFS="-DCMAKE_POSITION_INDEPENDENT_CODE=on -DGGML_NATIVE=off -DGGML_OPENMP=off"
+        COMMON_CPU_DEFS="-DBUILD_SHARED_LIBS=off -DCMAKE_POSITION_INDEPENDENT_CODE=on -DGGML_NATIVE=off -DGGML_OPENMP=off"
        if [ -z "${OLLAMA_CPU_TARGET}" -o "${OLLAMA_CPU_TARGET}" = "cpu" ]; then
            #
            # CPU first for the default library, set up as lowest common denominator for maximum compatibility (including Rosetta)
@ -178,7 +178,7 @@ if [ -z "${OLLAMA_SKIP_CUDA_GENERATE}" -a -d "${CUDA_LIB_DIR}" ]; then
        CMAKE_CUDA_DEFS="-DGGML_CUDA=on -DCMAKE_CUDA_ARCHITECTURES=${CMAKE_CUDA_ARCHITECTURES} ${OLLAMA_CUSTOM_CUDA_DEFS}"
        echo "Building custom CUDA GPU"
    else
-        CMAKE_CUDA_DEFS="-DGGML_CUDA=on -DCMAKE_CUDA_FLAGS=-t8 -DGGML_CUDA_FORCE_MMQ=on -DCMAKE_CUDA_ARCHITECTURES=${CMAKE_CUDA_ARCHITECTURES} -DCMAKE_LIBRARY_PATH=/usr/local/cuda/compat"
+        CMAKE_CUDA_DEFS="-DGGML_CUDA=on -DCMAKE_CUDA_FLAGS=-t8 -DCMAKE_CUDA_ARCHITECTURES=${CMAKE_CUDA_ARCHITECTURES}"
    fi
    CMAKE_DEFS="${COMMON_CMAKE_DEFS} ${CMAKE_DEFS} ${ARM64_DEFS} ${CMAKE_CUDA_DEFS}"
    BUILD_DIR="../build/linux/${ARCH}/cuda${CUDA_VARIANT}"
@ -254,7 +254,7 @@ if [ -z "${OLLAMA_SKIP_ROCM_GENERATE}" -a -d "${ROCM_PATH}" ]; then
        ROCM_VARIANT=_v$(ls ${ROCM_PATH}/lib/librocblas.so.*.*.????? | cut -f5 -d. || true)
    fi
    init_vars
-    CMAKE_DEFS="${COMMON_CMAKE_DEFS} ${CMAKE_DEFS} -DGGML_HIPBLAS=on -DCMAKE_C_COMPILER=$ROCM_PATH/llvm/bin/clang -DCMAKE_CXX_COMPILER=$ROCM_PATH/llvm/bin/clang++ -DAMDGPU_TARGETS=$(amdGPUs) -DGPU_TARGETS=$(amdGPUs)"
+    CMAKE_DEFS="${COMMON_CMAKE_DEFS} ${CMAKE_DEFS} -DGGML_HIPBLAS=on -DLLAMA_CUDA_NO_PEER_COPY=on -DCMAKE_C_COMPILER=$ROCM_PATH/llvm/bin/clang -DCMAKE_CXX_COMPILER=$ROCM_PATH/llvm/bin/clang++ -DAMDGPU_TARGETS=$(amdGPUs) -DGPU_TARGETS=$(amdGPUs)"
    # Users building from source can tune the exact flags we pass to cmake for configuring llama.cpp
    if [ -n "${OLLAMA_CUSTOM_ROCM_DEFS}" ]; then
        echo "OLLAMA_CUSTOM_ROCM_DEFS=\"${OLLAMA_CUSTOM_ROCM_DEFS}\""
--- a/llm/generate/gen_windows.ps1
+++ b/llm/generate/gen_windows.ps1
@ -6,18 +6,9 @@ function amdGPUs {
    if ($env:AMDGPU_TARGETS) {
        return $env:AMDGPU_TARGETS
    }
-    # TODO - load from some common data file for linux + windows build consistency
+    # Current supported rocblas list from ROCm v6.1.2 on windows
    $GPU_LIST = @(
        "gfx900"
        "gfx906:xnack-"
        "gfx908:xnack-"
        "gfx90a:xnack+"
        "gfx90a:xnack-"
        "gfx940"
        "gfx941"
        "gfx942"
        "gfx1010"
        "gfx1012"
        "gfx1030"
        "gfx1100"
        "gfx1101"
@ -366,6 +357,7 @@ function build_rocm() {
            "-DCMAKE_C_COMPILER=clang.exe",
            "-DCMAKE_CXX_COMPILER=clang++.exe",
            "-DGGML_HIPBLAS=on",
            "-DLLAMA_CUDA_NO_PEER_COPY=on",
            "-DHIP_PLATFORM=amd",
            "-DGGML_AVX=on",
            "-DGGML_AVX2=off",
@ -394,7 +386,6 @@ function build_rocm() {
        sign
        install
        # Assumes v5.7, may need adjustments for v6
        rm -ea 0 -recurse -force -path "${script:SRC_DIR}\dist\windows-${script:ARCH}\rocm\"
        md "${script:SRC_DIR}\dist\windows-${script:ARCH}\rocm\rocblas\library\" -ea 0 > $null
        cp "${env:HIP_PATH}\bin\hipblas.dll" "${script:SRC_DIR}\dist\windows-${script:ARCH}\rocm\"
--- a/llm/ggml.go
+++ b/llm/ggml.go
@ -424,6 +424,32 @@ func (llm GGML) GraphSize(context, batch uint64) (partialOffload, fullOffload ui
 			4*batch*(3*embedding+vocab)+embedding*vocab*105/128,
 			4*batch*(2*embedding+1+2*embeddingHeadsK*headsKV+context+context*headsKV)+4*embeddingHeadsK*context*headsKV+embedding*embeddingHeadsK*headsKV*9/16,
 		)
 	case "chatglm":
 		fullOffload = 4 * batch * (embedding + vocab)
 		partialOffload = 4*batch*(embedding+vocab) + embedding*vocab*105/128
 		if qkvBias, ok := layers["blk.0"]["attn_qkv.bias"]; ok {
 			fullOffload = max(
 				fullOffload,
 				4*batch*(2+
 					2*embedding+
 					context+
 					context*heads+
 					embeddingHeadsK*heads+
 					qkvBias.Shape[0]),
 			)
 			partialOffload = max(
 				partialOffload,
 				4*batch*(1+
 					2*embedding+
 					embeddingHeadsK*heads+
 					context+
 					context*heads)+
 					4*embeddingHeadsK*context+
 					4*context*embeddingHeadsK+
 					4*qkvBias.Shape[0],
 			)
 		}
 	}
 	return
--- a/llm/llama.cpp
+++ b/llm/llama.cpp
@ -1 +1 @@
-Subproject commit d7fd29fff16456ce9c3a23fd2d09a66256b05aff
+Subproject commit a8db2a9ce64cd4417f6a312ab61858f17f0f8584
--- a/llm/llm.go
+++ b/llm/llm.go
@ -1,12 +1,11 @@
 package llm
 // #cgo CFLAGS: -Illama.cpp -Illama.cpp/include -Illama.cpp/ggml/include
 // #cgo windows LDFLAGS: -static-libstdc++
 // #cgo LDFLAGS: -lllama -lggml -lstdc++ -lpthread
 // #cgo darwin,arm64 LDFLAGS: -L${SRCDIR}/build/darwin/arm64_static -L${SRCDIR}/build/darwin/arm64_static/src -L${SRCDIR}/build/darwin/arm64_static/ggml/src -framework Accelerate -framework Metal
 // #cgo darwin,amd64 LDFLAGS: -L${SRCDIR}/build/darwin/x86_64_static -L${SRCDIR}/build/darwin/x86_64_static/src -L${SRCDIR}/build/darwin/x86_64_static/ggml/src
-// #cgo windows,amd64 LDFLAGS: -L${SRCDIR}/build/windows/amd64_static -L${SRCDIR}/build/windows/amd64_static/src -L${SRCDIR}/build/windows/amd64_static/ggml/src
+// #cgo windows,amd64 LDFLAGS: -static-libstdc++ -static-libgcc -static -L${SRCDIR}/build/windows/amd64_static -L${SRCDIR}/build/windows/amd64_static/src -L${SRCDIR}/build/windows/amd64_static/ggml/src
-// #cgo windows,arm64 LDFLAGS: -L${SRCDIR}/build/windows/arm64_static -L${SRCDIR}/build/windows/arm64_static/src -L${SRCDIR}/build/windows/arm64_static/ggml/src
+// #cgo windows,arm64 LDFLAGS: -static-libstdc++ -static-libgcc -static -L${SRCDIR}/build/windows/arm64_static -L${SRCDIR}/build/windows/arm64_static/src -L${SRCDIR}/build/windows/arm64_static/ggml/src
 // #cgo linux,amd64 LDFLAGS: -L${SRCDIR}/build/linux/x86_64_static -L${SRCDIR}/build/linux/x86_64_static/src -L${SRCDIR}/build/linux/x86_64_static/ggml/src
 // #cgo linux,arm64 LDFLAGS: -L${SRCDIR}/build/linux/arm64_static -L${SRCDIR}/build/linux/arm64_static/src -L${SRCDIR}/build/linux/arm64_static/ggml/src
 // #include <stdlib.h>
@ -34,7 +33,7 @@ func Quantize(infile, outfile string, ftype fileType) error {
 	params.ftype = ftype.Value()
 	if rc := C.llama_model_quantize(cinfile, coutfile, &params); rc != 0 {
-		return fmt.Errorf("llama_model_quantize: %d", rc)
+		return fmt.Errorf("failed to quantize model. This model architecture may not be supported, or you may need to upgrade Ollama to the latest version")
 	}
 	return nil
--- a/llm/patches/05-default-pretokenizer.diff
+++ b/llm/patches/05-default-pretokenizer.diff
@ -1,11 +1,11 @@
 diff --git a/src/llama.cpp b/src/llama.cpp
-index 73f52435..2b81b4bd 100644
+index 2b9ace28..172640e2 100644
 --- a/src/llama.cpp
 +++ b/src/llama.cpp
-@@ -5092,16 +5092,7 @@ static void llm_load_vocab(
+@@ -5357,16 +5357,7 @@ static void llm_load_vocab(
         // for now, only BPE models have pre-tokenizers
         if (vocab.type == LLAMA_VOCAB_TYPE_BPE) {
             vocab.tokenizer_add_space_prefix = false;
             vocab.tokenizer_clean_spaces = true;
 -            if (tokenizer_pre.empty()) {
 -                LLAMA_LOG_WARN("%s: missing pre-tokenizer type, using: 'default'\n", __func__);
 -                LLAMA_LOG_WARN("%s:                                             \n", __func__);
@ -20,7 +20,7 @@ index 73f52435..2b81b4bd 100644
                 vocab.type_pre = LLAMA_VOCAB_PRE_TYPE_DEFAULT;
             } else if (
                     tokenizer_pre == "llama3"   ||
-@@ -5164,7 +5155,8 @@ static void llm_load_vocab(
+@@ -5439,7 +5430,8 @@ static void llm_load_vocab(
                 tokenizer_pre == "jais") {
                 vocab.type_pre = LLAMA_VOCAB_PRE_TYPE_JAIS;
             } else {
--- a/llm/server.go
+++ b/llm/server.go
@ -88,6 +88,7 @@ func NewLlamaServer(gpus gpu.GpuInfoList, model string, ggml *GGML, adapters, pr
 	var estimate MemoryEstimate
 	var systemTotalMemory uint64
 	var systemFreeMemory uint64
 	var systemSwapFreeMemory uint64
 	systemMemInfo, err := gpu.GetCPUMem()
 	if err != nil {
@ -95,7 +96,8 @@ func NewLlamaServer(gpus gpu.GpuInfoList, model string, ggml *GGML, adapters, pr
 	} else {
 		systemTotalMemory = systemMemInfo.TotalMemory
 		systemFreeMemory = systemMemInfo.FreeMemory
-		slog.Debug("system memory", "total", format.HumanBytes2(systemTotalMemory), "free", systemFreeMemory)
+		systemSwapFreeMemory = systemMemInfo.FreeSwap
 		slog.Debug("system memory", "total", format.HumanBytes2(systemTotalMemory), "free", format.HumanBytes2(systemFreeMemory), "free_swap", format.HumanBytes2(systemSwapFreeMemory))
 	}
 	// If the user wants zero GPU layers, reset the gpu list to be CPU/system ram info
@ -122,6 +124,16 @@ func NewLlamaServer(gpus gpu.GpuInfoList, model string, ggml *GGML, adapters, pr
 		}
 	}
 	// On linux, over-allocating CPU memory will almost always result in an error
 	if runtime.GOOS == "linux" {
 		systemMemoryRequired := estimate.TotalSize - estimate.VRAMSize
 		available := systemFreeMemory + systemSwapFreeMemory
 		if systemMemoryRequired > available {
 			slog.Warn("model request too large for system", "requested", format.HumanBytes2(systemMemoryRequired), "available", available, "total", format.HumanBytes2(systemTotalMemory), "free", format.HumanBytes2(systemFreeMemory), "swap", format.HumanBytes2(systemSwapFreeMemory))
 			return nil, fmt.Errorf("model requires more system memory (%s) than is available (%s)", format.HumanBytes2(systemMemoryRequired), format.HumanBytes2(available))
 		}
 	}
 	estimate.log()
 	// Loop through potential servers
@ -254,10 +266,6 @@ func NewLlamaServer(gpus gpu.GpuInfoList, model string, ggml *GGML, adapters, pr
 		params = append(params, "--tensor-split", estimate.TensorSplit)
 	}
 	if estimate.TensorSplit != "" {
 		params = append(params, "--tensor-split", estimate.TensorSplit)
 	}
 	for i := range len(servers) {
 		dir := availableServers[servers[i]]
 		if dir == "" {
@ -679,7 +687,7 @@ type CompletionRequest struct {
 	Prompt  string
 	Format  string
 	Images  []ImageData
-	Options api.Options
+	Options *api.Options
 }
 type CompletionResponse struct {
@ -699,10 +707,9 @@ func (s *llmServer) Completion(ctx context.Context, req CompletionRequest, fn fu
 	}
 	defer s.sem.Release(1)
-	// only allow maximum 10 "context shifts" to avoid infinite generation
+	// put an upper limit on num_predict to avoid the model running on forever
 	if req.Options.NumPredict < 0 || req.Options.NumPredict > 10*s.options.NumCtx {
 		req.Options.NumPredict = 10 * s.options.NumCtx
 		slog.Debug("setting token limit to 10x num_ctx", "num_ctx", s.options.NumCtx, "num_predict", req.Options.NumPredict)
 	}
 	request := map[string]any{
--- a/openai/openai.go
+++ b/openai/openai.go
@ -3,11 +3,13 @@ package openai
 import (
 	"bytes"
 	"encoding/base64"
 	"encoding/json"
 	"fmt"
 	"io"
 	"math/rand"
 	"net/http"
 	"strings"
 	"time"
 	"github.com/gin-gonic/gin"
@ -28,7 +30,7 @@ type ErrorResponse struct {
 type Message struct {
 	Role    string `json:"role"`
-	Content string `json:"content"`
+	Content any    `json:"content"`
 }
 type Choice struct {
@ -269,10 +271,66 @@ func toModel(r api.ShowResponse, m string) Model {
 	}
 }
-func fromChatRequest(r ChatCompletionRequest) api.ChatRequest {
+func fromChatRequest(r ChatCompletionRequest) (*api.ChatRequest, error) {
 	var messages []api.Message
 	for _, msg := range r.Messages {
-		messages = append(messages, api.Message{Role: msg.Role, Content: msg.Content})
+		switch content := msg.Content.(type) {
 		case string:
 			messages = append(messages, api.Message{Role: msg.Role, Content: content})
 		case []any:
 			message := api.Message{Role: msg.Role}
 			for _, c := range content {
 				data, ok := c.(map[string]any)
 				if !ok {
 					return nil, fmt.Errorf("invalid message format")
 				}
 				switch data["type"] {
 				case "text":
 					text, ok := data["text"].(string)
 					if !ok {
 						return nil, fmt.Errorf("invalid message format")
 					}
 					message.Content = text
 				case "image_url":
 					var url string
 					if urlMap, ok := data["image_url"].(map[string]any); ok {
 						if url, ok = urlMap["url"].(string); !ok {
 							return nil, fmt.Errorf("invalid message format")
 						}
 					} else {
 						if url, ok = data["image_url"].(string); !ok {
 							return nil, fmt.Errorf("invalid message format")
 						}
 					}
 					types := []string{"jpeg", "jpg", "png"}
 					valid := false
 					for _, t := range types {
 						prefix := "data:image/" + t + ";base64,"
 						if strings.HasPrefix(url, prefix) {
 							url = strings.TrimPrefix(url, prefix)
 							valid = true
 							break
 						}
 					}
 					if !valid {
 						return nil, fmt.Errorf("invalid image input")
 					}
 					img, err := base64.StdEncoding.DecodeString(url)
 					if err != nil {
 						return nil, fmt.Errorf("invalid message format")
 					}
 					message.Images = append(message.Images, img)
 				default:
 					return nil, fmt.Errorf("invalid message format")
 				}
 			}
 			messages = append(messages, message)
 		default:
 			return nil, fmt.Errorf("invalid message content type: %T", content)
 		}
 	}
 	options := make(map[string]interface{})
@ -323,13 +381,13 @@ func fromChatRequest(r ChatCompletionRequest) api.ChatRequest {
 		format = "json"
 	}
-	return api.ChatRequest{
+	return &api.ChatRequest{
 		Model:    r.Model,
 		Messages: messages,
 		Format:   format,
 		Options:  options,
 		Stream:   &r.Stream,
-	}
+	}, nil
 }
 func fromCompleteRequest(r CompletionRequest) (api.GenerateRequest, error) {
@ -338,12 +396,16 @@ func fromCompleteRequest(r CompletionRequest) (api.GenerateRequest, error) {
 	switch stop := r.Stop.(type) {
 	case string:
 		options["stop"] = []string{stop}
-	case []string:
+	case []any:
-		options["stop"] = stop
+		var stops []string
-	default:
+		for _, s := range stop {
-		if r.Stop != nil {
+			if str, ok := s.(string); ok {
-			return api.GenerateRequest{}, fmt.Errorf("invalid type for 'stop' field: %T", r.Stop)
+				stops = append(stops, str)
 			} else {
 				return api.GenerateRequest{}, fmt.Errorf("invalid type for 'stop' field: %T", s)
 			}
 		}
 		options["stop"] = stops
 	}
 	if r.MaxTokens != nil {
@ -652,7 +714,13 @@ func ChatMiddleware() gin.HandlerFunc {
 		}
 		var b bytes.Buffer
-		if err := json.NewEncoder(&b).Encode(fromChatRequest(req)); err != nil {
+
 		chatReq, err := fromChatRequest(req)
 		if err != nil {
 			c.AbortWithStatusJSON(http.StatusBadRequest, NewError(http.StatusBadRequest, err.Error()))
 		}
 		if err := json.NewEncoder(&b).Encode(chatReq); err != nil {
 			c.AbortWithStatusJSON(http.StatusInternalServerError, NewError(http.StatusInternalServerError, err.Error()))
 			return
 		}
--- a/openai/openai_test.go
+++ b/openai/openai_test.go
@ -2,8 +2,8 @@ package openai
 import (
 	"bytes"
 	"encoding/base64"
 	"encoding/json"
 	"fmt"
 	"io"
 	"net/http"
 	"net/http/httptest"
@ -16,7 +16,181 @@ import (
 	"github.com/stretchr/testify/assert"
 )
-func TestMiddleware(t *testing.T) {
+const prefix = `data:image/jpeg;base64,`
 const image = `iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=`
 const imageURL = prefix + image
 func TestMiddlewareRequests(t *testing.T) {
 	type testCase struct {
 		Name     string
 		Method   string
 		Path     string
 		Handler  func() gin.HandlerFunc
 		Setup    func(t *testing.T, req *http.Request)
 		Expected func(t *testing.T, req *http.Request)
 	}
 	var capturedRequest *http.Request
 	captureRequestMiddleware := func() gin.HandlerFunc {
 		return func(c *gin.Context) {
 			bodyBytes, _ := io.ReadAll(c.Request.Body)
 			c.Request.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 			capturedRequest = c.Request
 			c.Next()
 		}
 	}
 	testCases := []testCase{
 		{
 			Name:    "chat handler",
 			Method:  http.MethodPost,
 			Path:    "/api/chat",
 			Handler: ChatMiddleware,
 			Setup: func(t *testing.T, req *http.Request) {
 				body := ChatCompletionRequest{
 					Model:    "test-model",
 					Messages: []Message{{Role: "user", Content: "Hello"}},
 				}
 				bodyBytes, _ := json.Marshal(body)
 				req.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 				req.Header.Set("Content-Type", "application/json")
 			},
 			Expected: func(t *testing.T, req *http.Request) {
 				var chatReq api.ChatRequest
 				if err := json.NewDecoder(req.Body).Decode(&chatReq); err != nil {
 					t.Fatal(err)
 				}
 				if chatReq.Messages[0].Role != "user" {
 					t.Fatalf("expected 'user', got %s", chatReq.Messages[0].Role)
 				}
 				if chatReq.Messages[0].Content != "Hello" {
 					t.Fatalf("expected 'Hello', got %s", chatReq.Messages[0].Content)
 				}
 			},
 		},
 		{
 			Name:    "completions handler",
 			Method:  http.MethodPost,
 			Path:    "/api/generate",
 			Handler: CompletionsMiddleware,
 			Setup: func(t *testing.T, req *http.Request) {
 				temp := float32(0.8)
 				body := CompletionRequest{
 					Model:       "test-model",
 					Prompt:      "Hello",
 					Temperature: &temp,
 					Stop:        []string{"\n", "stop"},
 				}
 				bodyBytes, _ := json.Marshal(body)
 				req.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 				req.Header.Set("Content-Type", "application/json")
 			},
 			Expected: func(t *testing.T, req *http.Request) {
 				var genReq api.GenerateRequest
 				if err := json.NewDecoder(req.Body).Decode(&genReq); err != nil {
 					t.Fatal(err)
 				}
 				if genReq.Prompt != "Hello" {
 					t.Fatalf("expected 'Hello', got %s", genReq.Prompt)
 				}
 				if genReq.Options["temperature"] != 1.6 {
 					t.Fatalf("expected 1.6, got %f", genReq.Options["temperature"])
 				}
 				stopTokens, ok := genReq.Options["stop"].([]any)
 				if !ok {
 					t.Fatalf("expected stop tokens to be a list")
 				}
 				if stopTokens[0] != "\n" || stopTokens[1] != "stop" {
 					t.Fatalf("expected ['\\n', 'stop'], got %v", stopTokens)
 				}
 			},
 		},
 		{
 			Name:    "chat handler with image content",
 			Method:  http.MethodPost,
 			Path:    "/api/chat",
 			Handler: ChatMiddleware,
 			Setup: func(t *testing.T, req *http.Request) {
 				body := ChatCompletionRequest{
 					Model: "test-model",
 					Messages: []Message{
 						{
 							Role: "user", Content: []map[string]any{
 								{"type": "text", "text": "Hello"},
 								{"type": "image_url", "image_url": map[string]string{"url": imageURL}},
 							},
 						},
 					},
 				}
 				bodyBytes, _ := json.Marshal(body)
 				req.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 				req.Header.Set("Content-Type", "application/json")
 			},
 			Expected: func(t *testing.T, req *http.Request) {
 				var chatReq api.ChatRequest
 				if err := json.NewDecoder(req.Body).Decode(&chatReq); err != nil {
 					t.Fatal(err)
 				}
 				if chatReq.Messages[0].Role != "user" {
 					t.Fatalf("expected 'user', got %s", chatReq.Messages[0].Role)
 				}
 				if chatReq.Messages[0].Content != "Hello" {
 					t.Fatalf("expected 'Hello', got %s", chatReq.Messages[0].Content)
 				}
 				img, _ := base64.StdEncoding.DecodeString(imageURL[len(prefix):])
 				if !bytes.Equal(chatReq.Messages[0].Images[0], img) {
 					t.Fatalf("expected image encoding, got %s", chatReq.Messages[0].Images[0])
 				}
 			},
 		},
 	}
 	gin.SetMode(gin.TestMode)
 	router := gin.New()
 	endpoint := func(c *gin.Context) {
 		c.Status(http.StatusOK)
 	}
 	for _, tc := range testCases {
 		t.Run(tc.Name, func(t *testing.T) {
 			router = gin.New()
 			router.Use(captureRequestMiddleware())
 			router.Use(tc.Handler())
 			router.Handle(tc.Method, tc.Path, endpoint)
 			req, _ := http.NewRequest(tc.Method, tc.Path, nil)
 			if tc.Setup != nil {
 				tc.Setup(t, req)
 			}
 			resp := httptest.NewRecorder()
 			router.ServeHTTP(resp, req)
 			tc.Expected(t, capturedRequest)
 		})
 	}
 }
 func TestMiddlewareResponses(t *testing.T) {
 	type testCase struct {
 		Name     string
 		Method   string
@ -30,159 +204,7 @@ func TestMiddleware(t *testing.T) {
 	testCases := []testCase{
 		{
-			Name:     "chat handler",
+			Name:     "completions handler error forwarding",
 			Method:   http.MethodPost,
 			Path:     "/api/chat",
 			TestPath: "/api/chat",
 			Handler:  ChatMiddleware,
 			Endpoint: func(c *gin.Context) {
 				var chatReq api.ChatRequest
 				if err := c.ShouldBindJSON(&chatReq); err != nil {
 					c.JSON(http.StatusBadRequest, gin.H{"error": "invalid request"})
 					return
 				}
 				userMessage := chatReq.Messages[0].Content
 				var assistantMessage string
 				switch userMessage {
 				case "Hello":
 					assistantMessage = "Hello!"
 				default:
 					assistantMessage = "I'm not sure how to respond to that."
 				}
 				c.JSON(http.StatusOK, api.ChatResponse{
 					Message: api.Message{
 						Role:    "assistant",
 						Content: assistantMessage,
 					},
 				})
 			},
 			Setup: func(t *testing.T, req *http.Request) {
 				body := ChatCompletionRequest{
 					Model:    "test-model",
 					Messages: []Message{{Role: "user", Content: "Hello"}},
 				}
 				bodyBytes, _ := json.Marshal(body)
 				req.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 				req.Header.Set("Content-Type", "application/json")
 			},
 			Expected: func(t *testing.T, resp *httptest.ResponseRecorder) {
 				assert.Equal(t, http.StatusOK, resp.Code)
 				var chatResp ChatCompletion
 				if err := json.NewDecoder(resp.Body).Decode(&chatResp); err != nil {
 					t.Fatal(err)
 				}
 				if chatResp.Object != "chat.completion" {
 					t.Fatalf("expected chat.completion, got %s", chatResp.Object)
 				}
 				if chatResp.Choices[0].Message.Content != "Hello!" {
 					t.Fatalf("expected Hello!, got %s", chatResp.Choices[0].Message.Content)
 				}
 			},
 		},
 		{
 			Name:     "completions handler",
 			Method:   http.MethodPost,
 			Path:     "/api/generate",
 			TestPath: "/api/generate",
 			Handler:  CompletionsMiddleware,
 			Endpoint: func(c *gin.Context) {
 				c.JSON(http.StatusOK, api.GenerateResponse{
 					Response: "Hello!",
 				})
 			},
 			Setup: func(t *testing.T, req *http.Request) {
 				body := CompletionRequest{
 					Model:  "test-model",
 					Prompt: "Hello",
 				}
 				bodyBytes, _ := json.Marshal(body)
 				req.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 				req.Header.Set("Content-Type", "application/json")
 			},
 			Expected: func(t *testing.T, resp *httptest.ResponseRecorder) {
 				assert.Equal(t, http.StatusOK, resp.Code)
 				var completionResp Completion
 				if err := json.NewDecoder(resp.Body).Decode(&completionResp); err != nil {
 					t.Fatal(err)
 				}
 				if completionResp.Object != "text_completion" {
 					t.Fatalf("expected text_completion, got %s", completionResp.Object)
 				}
 				if completionResp.Choices[0].Text != "Hello!" {
 					t.Fatalf("expected Hello!, got %s", completionResp.Choices[0].Text)
 				}
 			},
 		},
 		{
 			Name:     "completions handler with params",
 			Method:   http.MethodPost,
 			Path:     "/api/generate",
 			TestPath: "/api/generate",
 			Handler:  CompletionsMiddleware,
 			Endpoint: func(c *gin.Context) {
 				var generateReq api.GenerateRequest
 				if err := c.ShouldBindJSON(&generateReq); err != nil {
 					c.JSON(http.StatusBadRequest, gin.H{"error": "invalid request"})
 					return
 				}
 				temperature := generateReq.Options["temperature"].(float64)
 				var assistantMessage string
 				switch temperature {
 				case 1.6:
 					assistantMessage = "Received temperature of 1.6"
 				default:
 					assistantMessage = fmt.Sprintf("Received temperature of %f", temperature)
 				}
 				c.JSON(http.StatusOK, api.GenerateResponse{
 					Response: assistantMessage,
 				})
 			},
 			Setup: func(t *testing.T, req *http.Request) {
 				temp := float32(0.8)
 				body := CompletionRequest{
 					Model:       "test-model",
 					Prompt:      "Hello",
 					Temperature: &temp,
 				}
 				bodyBytes, _ := json.Marshal(body)
 				req.Body = io.NopCloser(bytes.NewReader(bodyBytes))
 				req.Header.Set("Content-Type", "application/json")
 			},
 			Expected: func(t *testing.T, resp *httptest.ResponseRecorder) {
 				assert.Equal(t, http.StatusOK, resp.Code)
 				var completionResp Completion
 				if err := json.NewDecoder(resp.Body).Decode(&completionResp); err != nil {
 					t.Fatal(err)
 				}
 				if completionResp.Object != "text_completion" {
 					t.Fatalf("expected text_completion, got %s", completionResp.Object)
 				}
 				if completionResp.Choices[0].Text != "Received temperature of 1.6" {
 					t.Fatalf("expected Received temperature of 1.6, got %s", completionResp.Choices[0].Text)
 				}
 			},
 		},
 		{
 			Name:     "completions handler with error",
 			Method:   http.MethodPost,
 			Path:     "/api/generate",
 			TestPath: "/api/generate",
--- a/scripts/build_windows.ps1
+++ b/scripts/build_windows.ps1
@ -107,9 +107,12 @@ function gatherDependencies() {
    # TODO - this varies based on host build system and MSVC version - drive from dumpbin output
    # currently works for Win11 + MSVC 2019 + Cuda V11
-    cp "${env:VCToolsRedistDir}\x64\Microsoft.VC*.CRT\msvcp140.dll" "${script:DEPS_DIR}\ollama_runners\"
+    cp "${env:VCToolsRedistDir}\x64\Microsoft.VC*.CRT\msvcp140*.dll" "${script:DEPS_DIR}\ollama_runners\"
    cp "${env:VCToolsRedistDir}\x64\Microsoft.VC*.CRT\vcruntime140.dll" "${script:DEPS_DIR}\ollama_runners\"
    cp "${env:VCToolsRedistDir}\x64\Microsoft.VC*.CRT\vcruntime140_1.dll" "${script:DEPS_DIR}\ollama_runners\"
    foreach ($part in $("runtime", "stdio", "filesystem", "math", "convert", "heap", "string", "time", "locale", "environment")) {
        cp "$env:VCToolsRedistDir\..\..\..\Tools\Llvm\x64\bin\api-ms-win-crt-${part}*.dll" "${script:DEPS_DIR}\ollama_runners\"
    }
    cp "${script:SRC_DIR}\app\ollama_welcome.ps1" "${script:SRC_DIR}\dist\"
--- a/server/images.go
+++ b/server/images.go
@ -34,6 +34,8 @@ import (
 	"github.com/ollama/ollama/version"
 )
 var errCapabilityCompletion = errors.New("completion")
 type Capability string
 const CapabilityCompletion = Capability("completion")
@ -62,7 +64,10 @@ type Model struct {
 	Template *template.Template
 }
-func (m *Model) Has(caps ...Capability) bool {
+// CheckCapabilities checks if the model has the specified capabilities returning an error describing
 // any missing or unknown capabilities
 func (m *Model) CheckCapabilities(caps ...Capability) error {
 	var errs []error
 	for _, cap := range caps {
 		switch cap {
 		case CapabilityCompletion:
@ -81,15 +86,19 @@ func (m *Model) Has(caps ...Capability) bool {
 			}
 			if _, ok := ggml.KV()[fmt.Sprintf("%s.pooling_type", ggml.KV().Architecture())]; ok {
-				return false
+				errs = append(errs, errCapabilityCompletion)
 			}
 		default:
 			slog.Error("unknown capability", "capability", cap)
-			return false
+			return fmt.Errorf("unknown capability: %s", cap)
 		}
 	}
-	return true
+	if err := errors.Join(errs...); err != nil {
 		return fmt.Errorf("missing capabilities: %w", errors.Join(errs...))
 	}
 	return nil
 }
 func (m *Model) String() string {
--- a/server/prompt.go
+++ b/server/prompt.go
@ -1,217 +1,74 @@
 package server
 import (
-	"fmt"
+	"bytes"
 	"context"
 	"log/slog"
 	"strings"
 	"text/template/parse"
 	"github.com/ollama/ollama/api"
 	"github.com/ollama/ollama/llm"
 	"github.com/ollama/ollama/template"
 )
-// isResponseNode checks if the node contains .Response
+type tokenizeFunc func(context.Context, string) ([]int, error)
-func isResponseNode(node *parse.ActionNode) bool {
+
-	for _, cmd := range node.Pipe.Cmds {
+// chatPrompt accepts a list of messages and returns the prompt and images that should be used for the next chat turn.
-		for _, arg := range cmd.Args {
+// chatPrompt truncates any messages that exceed the context window of the model, making sure to always include 1) the
-			if fieldNode, ok := arg.(*parse.FieldNode); ok && len(fieldNode.Ident) > 0 {
+// latest message and 2) system messages
-				if fieldNode.Ident[0] == "Response" {
+func chatPrompt(ctx context.Context, m *Model, tokenize tokenizeFunc, opts *api.Options, msgs []api.Message) (prompt string, images []llm.ImageData, _ error) {
-					return true
+	var system []api.Message
-				}
+	// always include the last message
 	n := len(msgs) - 1
 	// in reverse, find all messages that fit into context window
 	for i := n - 1; i >= 0; i-- {
 		system = make([]api.Message, 0)
 		for j := range i {
 			if msgs[j].Role == "system" {
 				system = append(system, msgs[j])
 			}
 		}
 	}
 	return false
 }
-// formatTemplateForResponse formats the template AST to:
+		var b bytes.Buffer
-// 1. remove all nodes after the first .Response (if generate=true)
+		if err := m.Template.Execute(&b, template.Values{Messages: append(system, msgs[i:]...)}); err != nil {
-// 2. add a .Response node to the end if it doesn't exist
+			return "", nil, err
 // TODO(jmorganca): this should recursively cut the template before the first .Response
 func formatTemplateForResponse(tmpl *template.Template, generate bool) {
 	var found bool
 	for i, node := range tmpl.Tree.Root.Nodes {
 		if actionNode, ok := node.(*parse.ActionNode); ok {
 			if isResponseNode(actionNode) {
 				found = true
 				if generate {
 					tmpl.Tree.Root.Nodes = tmpl.Tree.Root.Nodes[:i+1]
 					break
 				}
 			}
 		}
 	}
-	if !found {
+		s, err := tokenize(ctx, b.String())
 		// add the response node if it doesn't exist
 		responseFieldNode := &parse.FieldNode{NodeType: parse.NodeField, Ident: []string{"Response"}}
 		responsePipeNode := &parse.PipeNode{NodeType: parse.NodePipe, Cmds: []*parse.CommandNode{{NodeType: parse.NodeCommand, Args: []parse.Node{responseFieldNode}}}}
 		responseActionNode := &parse.ActionNode{NodeType: parse.NodeAction, Pipe: responsePipeNode}
 		tmpl.Tree.Root.Nodes = append(tmpl.Tree.Root.Nodes, responseActionNode)
 	}
 }
 // Prompt renders a prompt from a template. If generate is set to true,
 // the response and parts of the template following it are not rendered
 func Prompt(tmpl *template.Template, system, prompt, response string, generate bool) (string, error) {
 	formatTemplateForResponse(tmpl, generate)
 	vars := map[string]any{
 		"System":   system,
 		"Prompt":   prompt,
 		"Response": response,
 	}
 	var sb strings.Builder
 	if err := tmpl.Execute(&sb, vars); err != nil {
 		return "", err
 	}
 	return sb.String(), nil
 }
 func countTokens(tmpl *template.Template, system string, prompt string, response string, encode func(string) ([]int, error)) (int, error) {
 	rendered, err := Prompt(tmpl, system, prompt, response, false)
 	if err != nil {
 		return 0, err
 	}
 	tokens, err := encode(rendered)
 	if err != nil {
 		slog.Error("failed to encode prompt", "err", err)
 		return 0, err
 	}
 	return len(tokens), err
 }
 // ChatPrompt builds up a prompt from a series of messages, truncating based on context window size
 func ChatPrompt(tmpl *template.Template, messages []api.Message, window int, encode func(string) ([]int, error)) (string, error) {
 	type prompt struct {
 		System   string
 		Prompt   string
 		Response string
 		images []int
 		tokens int
 	}
 	var p prompt
 	// iterate through messages to build up {system,user,response} prompts
 	var imgId int
 	var prompts []prompt
 	for _, msg := range messages {
 		switch strings.ToLower(msg.Role) {
 		case "system":
 			if p.System != "" || p.Prompt != "" || p.Response != "" {
 				prompts = append(prompts, p)
 				p = prompt{}
 			}
 			p.System = msg.Content
 		case "user":
 			if p.Prompt != "" || p.Response != "" {
 				prompts = append(prompts, p)
 				p = prompt{}
 			}
 			var sb strings.Builder
 			for range msg.Images {
 				fmt.Fprintf(&sb, "[img-%d] ", imgId)
 				p.images = append(p.images, imgId)
 				imgId += 1
 			}
 			sb.WriteString(msg.Content)
 			p.Prompt = sb.String()
 		case "assistant":
 			if p.Response != "" {
 				prompts = append(prompts, p)
 				p = prompt{}
 			}
 			p.Response = msg.Content
 		default:
 			return "", fmt.Errorf("invalid role: %s, role must be one of [system, user, assistant]", msg.Role)
 		}
 	}
 	// add final prompt
 	if p.System != "" || p.Prompt != "" || p.Response != "" {
 		prompts = append(prompts, p)
 	}
 	// calculate token lengths for each prompt, estimating 768 tokens per images
 	for i, p := range prompts {
 		tokens, err := countTokens(tmpl, p.System, p.Prompt, p.Response, encode)
 		if err != nil {
-			return "", err
+			return "", nil, err
 		}
-		prompts[i].tokens = tokens + len(prompts[i].images)*768
+		c := len(s)
-	}
+		if m.ProjectorPaths != nil {
-
+			for _, m := range msgs[i:] {
-	// truncate images and prompts starting from the beginning of the list
+				// images are represented as 768 sized embeddings
-	// until either one prompt remains or the total tokens fits the context window
+				// TODO: get embedding length from project metadata
-	// TODO (jmorganca): this doesn't account for the context window room required for the response
+				c += 768 * len(m.Images)
-	for {
+			}
 		var required int
 		for _, p := range prompts {
 			required += p.tokens
 		}
-		required += 1 // for bos token
+		if c > opts.NumCtx {
-
+			slog.Debug("truncating input messages which exceed context length", "truncated", len(msgs[i:]))
 		if required <= window {
 			slog.Debug("prompt now fits in context window", "required", required, "window", window)
 			break
 		} else {
 			n = i
 		}
 		prompt := &prompts[0]
 		if len(prompt.images) > 1 {
 			img := prompt.images[0]
 			slog.Debug("prompt longer than context window, removing image", "id", img, "required", required, "window", window)
 			prompt.images = prompt.images[1:]
 			prompt.Prompt = strings.Replace(prompt.Prompt, fmt.Sprintf(" [img-%d]", img), "", 1)
 			prompt.tokens -= 768
 			continue
 		}
 		if len(prompts) > 1 {
 			slog.Debug("required tokens longer than context window, removing first prompt", "prompt", prompts[0].tokens, "required", required, "window", window)
 			system := prompt.System
 			prompts = prompts[1:]
 			if system != "" && prompts[0].System == "" {
 				prompts[0].System = system
 				tokens, err := countTokens(tmpl, prompts[0].System, prompts[0].Prompt, prompts[0].Response, encode)
 				if err != nil {
 					return "", err
 				}
 				prompts[0].tokens = tokens + len(prompts[0].images)*768
 			}
 			continue
 		}
 		// stop truncating if there's only one prompt left
 		break
 	}
-	var sb strings.Builder
+	// truncate any messages that do not fit into the context window
-	for i, p := range prompts {
+	var b bytes.Buffer
-		// last prompt should leave the response unrendered (for completion)
+	if err := m.Template.Execute(&b, template.Values{Messages: append(system, msgs[n:]...)}); err != nil {
-		rendered, err := Prompt(tmpl, p.System, p.Prompt, p.Response, i == len(prompts)-1)
+		return "", nil, err
 		if err != nil {
 			return "", err
 		}
 		sb.WriteString(rendered)
 	}
-	return sb.String(), nil
+	for _, m := range msgs[n:] {
 		for _, i := range m.Images {
 			images = append(images, llm.ImageData{
 				ID:   len(images),
 				Data: i,
 			})
 		}
 	}
 	return b.String(), images, nil
 }
--- a/server/prompt_test.go
+++ b/server/prompt_test.go
@ -1,215 +1,222 @@
 package server
 import (
 	"bytes"
 	"context"
 	"strings"
 	"testing"
 	"github.com/google/go-cmp/cmp"
 	"github.com/ollama/ollama/api"
 	"github.com/ollama/ollama/template"
 )
-func TestPrompt(t *testing.T) {
+func tokenize(_ context.Context, s string) (tokens []int, err error) {
-	tests := []struct {
+	for range strings.Fields(s) {
-		name     string
+		tokens = append(tokens, len(tokens))
 		template string
 		system   string
 		prompt   string
 		response string
 		generate bool
 		want     string
 	}{
 		{
 			name:     "simple prompt",
 			template: "[INST] {{ .System }} {{ .Prompt }} [/INST]",
 			system:   "You are a Wizard.",
 			prompt:   "What are the potion ingredients?",
 			want:     "[INST] You are a Wizard. What are the potion ingredients? [/INST]",
 		},
 		{
 			name:     "implicit response",
 			template: "[INST] {{ .System }} {{ .Prompt }} [/INST]",
 			system:   "You are a Wizard.",
 			prompt:   "What are the potion ingredients?",
 			response: "I don't know.",
 			want:     "[INST] You are a Wizard. What are the potion ingredients? [/INST]I don't know.",
 		},
 		{
 			name:     "response",
 			template: "[INST] {{ .System }} {{ .Prompt }} [/INST] {{ .Response }}",
 			system:   "You are a Wizard.",
 			prompt:   "What are the potion ingredients?",
 			response: "I don't know.",
 			want:     "[INST] You are a Wizard. What are the potion ingredients? [/INST] I don't know.",
 		},
 		{
 			name:     "cut",
 			template: "<system>{{ .System }}</system><user>{{ .Prompt }}</user><assistant>{{ .Response }}</assistant>",
 			system:   "You are a Wizard.",
 			prompt:   "What are the potion ingredients?",
 			response: "I don't know.",
 			generate: true,
 			want:     "<system>You are a Wizard.</system><user>What are the potion ingredients?</user><assistant>I don't know.",
 		},
 		{
 			name:     "nocut",
 			template: "<system>{{ .System }}</system><user>{{ .Prompt }}</user><assistant>{{ .Response }}</assistant>",
 			system:   "You are a Wizard.",
 			prompt:   "What are the potion ingredients?",
 			response: "I don't know.",
 			want:     "<system>You are a Wizard.</system><user>What are the potion ingredients?</user><assistant>I don't know.</assistant>",
 		},
 	}
-	for _, tc := range tests {
+	return
 		t.Run(tc.name, func(t *testing.T) {
 			tmpl, err := template.Parse(tc.template)
 			if err != nil {
 				t.Fatal(err)
 			}
 			got, err := Prompt(tmpl, tc.system, tc.prompt, tc.response, tc.generate)
 			if err != nil {
 				t.Errorf("error = %v", err)
 			}
 			if got != tc.want {
 				t.Errorf("got = %v, want %v", got, tc.want)
 			}
 		})
 	}
 }
 func TestChatPrompt(t *testing.T) {
-	tests := []struct {
+	type expect struct {
-		name     string
+		prompt string
-		template string
+		images [][]byte
-		messages []api.Message
+	}
-		window   int
+
-		want     string
+	cases := []struct {
 		name  string
 		limit int
 		msgs  []api.Message
 		expect
 	}{
 		{
-			name:     "simple prompt",
+			name:  "messages",
-			template: "[INST] {{ .Prompt }} [/INST]",
+			limit: 64,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "user", Content: "Hello"},
+				{Role: "user", Content: "You're a test, Harry!"},
 				{Role: "assistant", Content: "I-I'm a what?"},
 				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager."},
 			},
 			expect: expect{
 				prompt: "You're a test, Harry! I-I'm a what? A test. And a thumping good one at that, I'd wager. ",
 			},
 			window: 1024,
 			want:   "[INST] Hello [/INST]",
 		},
 		{
-			name:     "with system message",
+			name:  "truncate messages",
-			template: "[INST] {{ if .System }}<<SYS>>{{ .System }}<</SYS>> {{ end }}{{ .Prompt }} [/INST]",
+			limit: 1,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry!"},
-				{Role: "user", Content: "Hello"},
+				{Role: "assistant", Content: "I-I'm a what?"},
 				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager."},
 			},
 			expect: expect{
 				prompt: "A test. And a thumping good one at that, I'd wager. ",
 			},
 			window: 1024,
 			want:   "[INST] <<SYS>>You are a Wizard.<</SYS>> Hello [/INST]",
 		},
 		{
-			name:     "with response",
+			name:  "truncate messages with image",
-			template: "[INST] {{ if .System }}<<SYS>>{{ .System }}<</SYS>> {{ end }}{{ .Prompt }} [/INST] {{ .Response }}",
+			limit: 64,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry!"},
-				{Role: "user", Content: "Hello"},
+				{Role: "assistant", Content: "I-I'm a what?"},
-				{Role: "assistant", Content: "I am?"},
+				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager.", Images: []api.ImageData{[]byte("something")}},
 			},
 			expect: expect{
 				prompt: "[img-0] A test. And a thumping good one at that, I'd wager. ",
 				images: [][]byte{
 					[]byte("something"),
 				},
 			},
 			window: 1024,
 			want:   "[INST] <<SYS>>You are a Wizard.<</SYS>> Hello [/INST] I am?",
 		},
 		{
-			name:     "with implicit response",
+			name:  "truncate messages with images",
-			template: "[INST] {{ if .System }}<<SYS>>{{ .System }}<</SYS>> {{ end }}{{ .Prompt }} [/INST]",
+			limit: 64,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry!", Images: []api.ImageData{[]byte("something")}},
-				{Role: "user", Content: "Hello"},
+				{Role: "assistant", Content: "I-I'm a what?"},
-				{Role: "assistant", Content: "I am?"},
+				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager.", Images: []api.ImageData{[]byte("somethingelse")}},
 			},
 			expect: expect{
 				prompt: "[img-0] A test. And a thumping good one at that, I'd wager. ",
 				images: [][]byte{
 					[]byte("somethingelse"),
 				},
 			},
 			window: 1024,
 			want:   "[INST] <<SYS>>You are a Wizard.<</SYS>> Hello [/INST]I am?",
 		},
 		{
-			name:     "with conversation",
+			name:  "messages with images",
-			template: "[INST] {{ if .System }}<<SYS>>{{ .System }}<</SYS>> {{ end }}{{ .Prompt }} [/INST] {{ .Response }} ",
+			limit: 2048,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry!", Images: []api.ImageData{[]byte("something")}},
-				{Role: "user", Content: "What are the potion ingredients?"},
+				{Role: "assistant", Content: "I-I'm a what?"},
-				{Role: "assistant", Content: "sugar"},
+				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager.", Images: []api.ImageData{[]byte("somethingelse")}},
-				{Role: "user", Content: "Anything else?"},
+			},
 			expect: expect{
 				prompt: "[img-0] You're a test, Harry! I-I'm a what? [img-1] A test. And a thumping good one at that, I'd wager. ",
 				images: [][]byte{
 					[]byte("something"),
 					[]byte("somethingelse"),
 				},
 			},
 			window: 1024,
 			want:   "[INST] <<SYS>>You are a Wizard.<</SYS>> What are the potion ingredients? [/INST] sugar [INST] Anything else? [/INST] ",
 		},
 		{
-			name:     "with truncation",
+			name:  "message with image tag",
-			template: "{{ .System }} {{ .Prompt }} {{ .Response }} ",
+			limit: 2048,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry! [img]", Images: []api.ImageData{[]byte("something")}},
-				{Role: "user", Content: "Hello"},
+				{Role: "assistant", Content: "I-I'm a what?"},
-				{Role: "assistant", Content: "I am?"},
+				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager.", Images: []api.ImageData{[]byte("somethingelse")}},
-				{Role: "user", Content: "Why is the sky blue?"},
+			},
-				{Role: "assistant", Content: "The sky is blue from rayleigh scattering"},
+			expect: expect{
 				prompt: "You're a test, Harry! [img-0] I-I'm a what? [img-1] A test. And a thumping good one at that, I'd wager. ",
 				images: [][]byte{
 					[]byte("something"),
 					[]byte("somethingelse"),
 				},
 			},
 			window: 10,
 			want:   "You are a Wizard. Why is the sky blue? The sky is blue from rayleigh scattering",
 		},
 		{
-			name:     "images",
+			name:  "messages with interleaved images",
-			template: "{{ .System }} {{ .Prompt }}",
+			limit: 2048,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry!"},
-				{Role: "user", Content: "Hello", Images: []api.ImageData{[]byte("base64")}},
+				{Role: "user", Images: []api.ImageData{[]byte("something")}},
 				{Role: "user", Images: []api.ImageData{[]byte("somethingelse")}},
 				{Role: "assistant", Content: "I-I'm a what?"},
 				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager."},
 			},
 			expect: expect{
 				prompt: "You're a test, Harry!\n\n[img-0]\n\n[img-1] I-I'm a what? A test. And a thumping good one at that, I'd wager. ",
 				images: [][]byte{
 					[]byte("something"),
 					[]byte("somethingelse"),
 				},
 			},
 			window: 1024,
 			want:   "You are a Wizard. [img-0] Hello",
 		},
 		{
-			name:     "images truncated",
+			name:  "truncate message with interleaved images",
-			template: "{{ .System }} {{ .Prompt }}",
+			limit: 1024,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "system", Content: "You are a Wizard."},
+				{Role: "user", Content: "You're a test, Harry!"},
-				{Role: "user", Content: "Hello", Images: []api.ImageData{[]byte("img1"), []byte("img2")}},
+				{Role: "user", Images: []api.ImageData{[]byte("something")}},
 				{Role: "user", Images: []api.ImageData{[]byte("somethingelse")}},
 				{Role: "assistant", Content: "I-I'm a what?"},
 				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager."},
 			},
 			expect: expect{
 				prompt: "[img-0] I-I'm a what? A test. And a thumping good one at that, I'd wager. ",
 				images: [][]byte{
 					[]byte("somethingelse"),
 				},
 			},
 			window: 1024,
 			want:   "You are a Wizard. [img-0] [img-1] Hello",
 		},
 		{
-			name:     "empty list",
+			name:  "message with system prompt",
-			template: "{{ .System }} {{ .Prompt }}",
+			limit: 2048,
-			messages: []api.Message{},
+			msgs: []api.Message{
-			window:   1024,
+				{Role: "system", Content: "You are the Test Who Lived."},
-			want:     "",
+				{Role: "user", Content: "You're a test, Harry!"},
 				{Role: "assistant", Content: "I-I'm a what?"},
 				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager."},
 			},
 			expect: expect{
 				prompt: "You are the Test Who Lived. You're a test, Harry! I-I'm a what? A test. And a thumping good one at that, I'd wager. ",
 			},
 		},
 		{
-			name:     "empty prompt",
+			name:  "out of order system",
-			template: "[INST] {{ if .System }}<<SYS>>{{ .System }}<</SYS>> {{ end }}{{ .Prompt }} [/INST] {{ .Response }} ",
+			limit: 2048,
-			messages: []api.Message{
+			msgs: []api.Message{
-				{Role: "user", Content: ""},
+				{Role: "user", Content: "You're a test, Harry!"},
 				{Role: "assistant", Content: "I-I'm a what?"},
 				{Role: "system", Content: "You are the Test Who Lived."},
 				{Role: "user", Content: "A test. And a thumping good one at that, I'd wager."},
 			},
 			expect: expect{
 				prompt: "You're a test, Harry! I-I'm a what? You are the Test Who Lived. A test. And a thumping good one at that, I'd wager. ",
 			},
 			window: 1024,
 			want:   "",
 		},
 	}
-	encode := func(s string) ([]int, error) {
+	tmpl, err := template.Parse(`
-		words := strings.Fields(s)
+{{- if .System }}{{ .System }} {{ end }}
-		return make([]int, len(words)), nil
+{{- if .Prompt }}{{ .Prompt }} {{ end }}
 {{- if .Response }}{{ .Response }} {{ end }}`)
 	if err != nil {
 		t.Fatal(err)
 	}
-	for _, tc := range tests {
+	for _, tt := range cases {
-		t.Run(tc.name, func(t *testing.T) {
+		t.Run(tt.name, func(t *testing.T) {
-			tmpl, err := template.Parse(tc.template)
+			model := Model{Template: tmpl, ProjectorPaths: []string{"vision"}}
 			opts := api.Options{Runner: api.Runner{NumCtx: tt.limit}}
 			prompt, images, err := chatPrompt(context.TODO(), &model, tokenize, &opts, tt.msgs)
 			if err != nil {
 				t.Fatal(err)
 			}
-			got, err := ChatPrompt(tmpl, tc.messages, tc.window, encode)
+			if tt.prompt != prompt {
-			if err != nil {
+				t.Errorf("expected %q, got %q", tt.prompt, prompt)
 				t.Errorf("error = %v", err)
 			}
-			if got != tc.want {
+			if diff := cmp.Diff(prompt, tt.prompt); diff != "" {
-				t.Errorf("got: %q, want: %q", got, tc.want)
+				t.Errorf("mismatch (-got +want):\n%s", diff)
 			}
 			if len(images) != len(tt.images) {
 				t.Fatalf("expected %d images, got %d", len(tt.images), len(images))
 			}
 			for i := range images {
 				if images[i].ID != i {
 					t.Errorf("expected ID %d, got %d", i, images[i].ID)
 				}
 				if !bytes.Equal(images[i].Data, tt.images[i]) {
 					t.Errorf("expected %q, got %q", tt.images[i], images[i])
 				}
 			}
 		})
 	}
--- a/server/routes.go
+++ b/server/routes.go
@ -1,13 +1,13 @@
 package server
 import (
 	"bytes"
 	"cmp"
 	"context"
 	"encoding/json"
 	"errors"
 	"fmt"
 	"io"
 	"io/fs"
 	"log/slog"
 	"net"
 	"net/http"
@ -54,6 +54,8 @@ func init() {
 	gin.SetMode(mode)
 }
 var errRequired = errors.New("is required")
 func modelOptions(model *Model, requestOpts map[string]interface{}) (api.Options, error) {
 	opts := api.DefaultOptions()
 	if err := opts.FromMap(model.Options); err != nil {
@ -67,242 +69,202 @@ func modelOptions(model *Model, requestOpts map[string]interface{}) (api.Options
 	return opts, nil
 }
-func isSupportedImageType(image []byte) bool {
+// scheduleRunner schedules a runner after validating inputs such as capabilities and model options.
-	contentType := http.DetectContentType(image)
+// It returns the allocated runner, model instance, and consolidated options if successful and error otherwise.
-	allowedTypes := []string{"image/jpeg", "image/jpg", "image/png"}
+func (s *Server) scheduleRunner(ctx context.Context, name string, caps []Capability, requestOpts map[string]any, keepAlive *api.Duration) (llm.LlamaServer, *Model, *api.Options, error) {
-	return slices.Contains(allowedTypes, contentType)
+	if name == "" {
 		return nil, nil, nil, fmt.Errorf("model %w", errRequired)
 	}
 	model, err := GetModel(name)
 	if err != nil {
 		return nil, nil, nil, err
 	}
 	if err := model.CheckCapabilities(caps...); err != nil {
 		return nil, nil, nil, fmt.Errorf("%s %w", name, err)
 	}
 	opts, err := modelOptions(model, requestOpts)
 	if err != nil {
 		return nil, nil, nil, err
 	}
 	runnerCh, errCh := s.sched.GetRunner(ctx, model, opts, keepAlive)
 	var runner *runnerRef
 	select {
 	case runner = <-runnerCh:
 	case err = <-errCh:
 		return nil, nil, nil, err
 	}
 	return runner.llama, model, &opts, nil
 }
 func (s *Server) GenerateHandler(c *gin.Context) {
 	checkpointStart := time.Now()
 	var req api.GenerateRequest
-	err := c.ShouldBindJSON(&req)
+	if err := c.ShouldBindJSON(&req); errors.Is(err, io.EOF) {
 	switch {
 	case errors.Is(err, io.EOF):
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "missing request body"})
 		return
-	case err != nil:
+	} else if err != nil {
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": err.Error()})
 		return
 	}
-	// validate the request
+	if req.Format != "" && req.Format != "json" {
-	switch {
+		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "format must be empty or \"json\""})
 	case req.Model == "":
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "model is required"})
 		return
-	case len(req.Format) > 0 && req.Format != "json":
+	} else if req.Raw && (req.Template != "" || req.System != "" || len(req.Context) > 0) {
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "format must be json"})
 		return
 	case req.Raw && (req.Template != "" || req.System != "" || len(req.Context) > 0):
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "raw mode does not support template, system, or context"})
 		return
 	}
-	for _, img := range req.Images {
+	caps := []Capability{CapabilityCompletion}
-		if !isSupportedImageType(img) {
+	r, m, opts, err := s.scheduleRunner(c.Request.Context(), req.Model, caps, req.Options, req.KeepAlive)
-			c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "unsupported image format"})
+	if errors.Is(err, errCapabilityCompletion) {
-			return
+		c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("%q does not support generate", req.Model)})
-		}
+		return
-	}
+	} else if err != nil {
-
+		handleScheduleError(c, req.Model, err)
 	model, err := GetModel(req.Model)
 	if err != nil {
 		var pErr *fs.PathError
 		if errors.As(err, &pErr) {
 			c.JSON(http.StatusNotFound, gin.H{"error": fmt.Sprintf("model '%s' not found, try pulling it first", req.Model)})
 			return
 		}
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
-	if !model.Has(CapabilityCompletion) {
+	checkpointLoaded := time.Now()
 		c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("%s does not support generate", req.Model)})
 		return
 	}
-	opts, err := modelOptions(model, req.Options)
+	if req.Prompt == "" {
 	if err != nil {
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
 	rCh, eCh := s.sched.GetRunner(c.Request.Context(), model, opts, req.KeepAlive)
 	var runner *runnerRef
 	select {
 	case runner = <-rCh:
 	case err = <-eCh:
 		handleErrorResponse(c, err)
 		return
 	}
 	// an empty request loads the model
 	// note: for a short while template was used in lieu
 	// of `raw` mode so we need to check for it too
 	if req.Prompt == "" && req.Template == "" && req.System == "" {
 		c.JSON(http.StatusOK, api.GenerateResponse{
 			CreatedAt:  time.Now().UTC(),
 			Model:      req.Model,
 			CreatedAt:  time.Now().UTC(),
 			Done:       true,
 			DoneReason: "load",
 		})
 		return
 	}
-	tmpl, err := template.Parse(req.Template)
+	images := make([]llm.ImageData, len(req.Images))
-	if err != nil {
+	for i := range req.Images {
-		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
+		images[i] = llm.ImageData{ID: i, Data: req.Images[i]}
 		return
 	}
-	checkpointLoaded := time.Now()
+	prompt := req.Prompt
-
+	if !req.Raw {
-	var prompt string
+		var msgs []api.Message
-	switch {
+		if req.System != "" {
-	case req.Raw:
+			msgs = append(msgs, api.Message{Role: "system", Content: req.System})
-		prompt = req.Prompt
+		} else if m.System != "" {
-	case req.Prompt != "":
+			msgs = append(msgs, api.Message{Role: "system", Content: m.System})
 		if req.Template == "" {
 			tmpl = model.Template
 		}
-		if req.System == "" {
+		for _, i := range images {
-			req.System = model.System
+			msgs = append(msgs, api.Message{Role: "user", Content: fmt.Sprintf("[img-%d]", i.ID)})
 		}
-		slog.Debug("generate handler", "prompt", req.Prompt)
+		msgs = append(msgs, api.Message{Role: "user", Content: req.Prompt})
 		slog.Debug("generate handler", "template", req.Template)
 		slog.Debug("generate handler", "system", req.System)
-		var sb strings.Builder
+		tmpl := m.Template
-		for i := range req.Images {
+		if req.Template != "" {
-			fmt.Fprintf(&sb, "[img-%d] ", i)
+			tmpl, err = template.Parse(req.Template)
 			if err != nil {
 				c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 				return
 			}
 		}
-		sb.WriteString(req.Prompt)
+		var b bytes.Buffer
 		p, err := Prompt(tmpl, req.System, sb.String(), "", true)
 		if err != nil {
 			c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 			return
 		}
 		sb.Reset()
 		if req.Context != nil {
-			prev, err := runner.llama.Detokenize(c.Request.Context(), req.Context)
+			s, err := r.Detokenize(c.Request.Context(), req.Context)
 			if err != nil {
 				c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 				return
 			}
-			sb.WriteString(prev)
+			b.WriteString(s)
 		}
-		sb.WriteString(p)
+		if err := tmpl.Execute(&b, template.Values{Messages: msgs}); err != nil {
 			c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 			return
 		}
-		prompt = sb.String()
+		prompt = b.String()
 	}
-	slog.Debug("generate handler", "prompt", prompt)
+	slog.Debug("generate request", "prompt", prompt, "images", images)
 	ch := make(chan any)
 	var generated strings.Builder
 	go func() {
 		// TODO (jmorganca): avoid building the response twice both here and below
 		var sb strings.Builder
 		defer close(ch)
-
+		if err := r.Completion(c.Request.Context(), llm.CompletionRequest{
-		fn := func(r llm.CompletionResponse) {
+			Prompt:  prompt,
-			// Build up the full response
+			Images:  images,
-			if _, err := generated.WriteString(r.Content); err != nil {
+			Format:  req.Format,
-				ch <- gin.H{"error": err.Error()}
+			Options: opts,
-				return
+		}, func(cr llm.CompletionResponse) {
-			}
+			res := api.GenerateResponse{
 			resp := api.GenerateResponse{
 				Model:      req.Model,
 				CreatedAt:  time.Now().UTC(),
-				Done:       r.Done,
+				Response:   cr.Content,
-				Response:   r.Content,
+				Done:       cr.Done,
-				DoneReason: r.DoneReason,
+				DoneReason: cr.DoneReason,
 				Metrics: api.Metrics{
-					PromptEvalCount:    r.PromptEvalCount,
+					PromptEvalCount:    cr.PromptEvalCount,
-					PromptEvalDuration: r.PromptEvalDuration,
+					PromptEvalDuration: cr.PromptEvalDuration,
-					EvalCount:          r.EvalCount,
+					EvalCount:          cr.EvalCount,
-					EvalDuration:       r.EvalDuration,
+					EvalDuration:       cr.EvalDuration,
 				},
 			}
-			if r.Done {
+			if _, err := sb.WriteString(cr.Content); err != nil {
-				resp.TotalDuration = time.Since(checkpointStart)
+				ch <- gin.H{"error": err.Error()}
-				resp.LoadDuration = checkpointLoaded.Sub(checkpointStart)
+			}
 			if cr.Done {
 				res.TotalDuration = time.Since(checkpointStart)
 				res.LoadDuration = checkpointLoaded.Sub(checkpointStart)
 				if !req.Raw {
-					p, err := Prompt(tmpl, req.System, req.Prompt, generated.String(), false)
+					tokens, err := r.Tokenize(c.Request.Context(), prompt+sb.String())
 					if err != nil {
 						c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 						return
 					}
 					// TODO (jmorganca): encode() should not strip special tokens
 					tokens, err := runner.llama.Tokenize(c.Request.Context(), p)
 					if err != nil {
 						ch <- gin.H{"error": err.Error()}
 						return
 					}
-
+					res.Context = append(req.Context, tokens...)
 					resp.Context = append(req.Context, tokens...)
 				}
 			}
-			ch <- resp
+			ch <- res
-		}
+		}); err != nil {
 		var images []llm.ImageData
 		for i := range req.Images {
 			images = append(images, llm.ImageData{
 				ID:   i,
 				Data: req.Images[i],
 			})
 		}
 		// Start prediction
 		req := llm.CompletionRequest{
 			Prompt:  prompt,
 			Format:  req.Format,
 			Images:  images,
 			Options: opts,
 		}
 		if err := runner.llama.Completion(c.Request.Context(), req, fn); err != nil {
 			ch <- gin.H{"error": err.Error()}
 		}
 	}()
 	if req.Stream != nil && !*req.Stream {
-		// Accumulate responses into the final response
+		var r api.GenerateResponse
 		var final api.GenerateResponse
 		var sb strings.Builder
-		for resp := range ch {
+		for rr := range ch {
-			switch r := resp.(type) {
+			switch t := rr.(type) {
 			case api.GenerateResponse:
-				sb.WriteString(r.Response)
+				sb.WriteString(t.Response)
-				final = r
+				r = t
 			case gin.H:
-				if errorMsg, ok := r["error"].(string); ok {
+				msg, ok := t["error"].(string)
-					c.JSON(http.StatusInternalServerError, gin.H{"error": errorMsg})
+				if !ok {
-					return
+					msg = "unexpected error format in response"
 				} else {
 					c.JSON(http.StatusInternalServerError, gin.H{"error": "unexpected error format in response"})
 					return
 				}
 				c.JSON(http.StatusInternalServerError, gin.H{"error": msg})
 				return
 			default:
-				c.JSON(http.StatusInternalServerError, gin.H{"error": "unexpected error"})
+				c.JSON(http.StatusInternalServerError, gin.H{"error": "unexpected response"})
 				return
 			}
 		}
-		final.Response = sb.String()
+		r.Response = sb.String()
-		c.JSON(http.StatusOK, final)
+		c.JSON(http.StatusOK, r)
 		return
 	}
@ -311,44 +273,17 @@ func (s *Server) GenerateHandler(c *gin.Context) {
 func (s *Server) EmbeddingsHandler(c *gin.Context) {
 	var req api.EmbeddingRequest
-	err := c.ShouldBindJSON(&req)
+	if err := c.ShouldBindJSON(&req); errors.Is(err, io.EOF) {
 	switch {
 	case errors.Is(err, io.EOF):
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "missing request body"})
 		return
-	case err != nil:
+	} else if err != nil {
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": err.Error()})
 		return
 	}
-	if req.Model == "" {
+	r, _, _, err := s.scheduleRunner(c.Request.Context(), req.Model, []Capability{}, req.Options, req.KeepAlive)
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "model is required"})
 		return
 	}
 	model, err := GetModel(req.Model)
 	if err != nil {
-		var pErr *fs.PathError
+		handleScheduleError(c, req.Model, err)
 		if errors.As(err, &pErr) {
 			c.JSON(http.StatusNotFound, gin.H{"error": fmt.Sprintf("model '%s' not found, try pulling it first", req.Model)})
 			return
 		}
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
 	opts, err := modelOptions(model, req.Options)
 	if err != nil {
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
 	rCh, eCh := s.sched.GetRunner(c.Request.Context(), model, opts, req.KeepAlive)
 	var runner *runnerRef
 	select {
 	case runner = <-rCh:
 	case err = <-eCh:
 		handleErrorResponse(c, err)
 		return
 	}
@ -358,17 +293,14 @@ func (s *Server) EmbeddingsHandler(c *gin.Context) {
 		return
 	}
-	embedding, err := runner.llama.Embedding(c.Request.Context(), req.Prompt)
+	embedding, err := r.Embedding(c.Request.Context(), req.Prompt)
 	if err != nil {
 		slog.Info(fmt.Sprintf("embedding generation failed: %v", err))
 		c.JSON(http.StatusInternalServerError, gin.H{"error": "failed to generate embedding"})
 		return
 	}
-	resp := api.EmbeddingResponse{
+	c.JSON(http.StatusOK, api.EmbeddingResponse{Embedding: embedding})
 		Embedding: embedding,
 	}
 	c.JSON(http.StatusOK, resp)
 }
 func (s *Server) PullModelHandler(c *gin.Context) {
@ -642,16 +574,9 @@ func GetModelInfo(req api.ShowRequest) (*api.ShowResponse, error) {
 		m.System = req.System
 	}
-	if req.Template != "" {
+	msgs := make([]api.Message, len(m.Messages))
-		m.Template, err = template.Parse(req.Template)
+	for i, msg := range m.Messages {
-		if err != nil {
+		msgs[i] = api.Message{Role: msg.Role, Content: msg.Content}
 			return nil, err
 		}
 	}
 	msgs := make([]api.Message, 0)
 	for _, msg := range m.Messages {
 		msgs = append(msgs, api.Message{Role: msg.Role, Content: msg.Content})
 	}
 	n := model.ParseName(req.Model)
@ -1214,132 +1139,63 @@ func (s *Server) ProcessHandler(c *gin.Context) {
 	c.JSON(http.StatusOK, api.ProcessResponse{Models: models})
 }
 // ChatPrompt builds up a prompt from a series of messages for the currently `loaded` model
 func chatPrompt(ctx context.Context, runner *runnerRef, template *template.Template, messages []api.Message, numCtx int) (string, error) {
 	encode := func(s string) ([]int, error) {
 		return runner.llama.Tokenize(ctx, s)
 	}
 	prompt, err := ChatPrompt(template, messages, numCtx, encode)
 	if err != nil {
 		return "", err
 	}
 	return prompt, nil
 }
 func (s *Server) ChatHandler(c *gin.Context) {
 	checkpointStart := time.Now()
 	var req api.ChatRequest
-	err := c.ShouldBindJSON(&req)
+	if err := c.ShouldBindJSON(&req); errors.Is(err, io.EOF) {
 	switch {
 	case errors.Is(err, io.EOF):
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "missing request body"})
 		return
-	case err != nil:
+	} else if err != nil {
 		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": err.Error()})
 		return
 	}
-	// validate the request
+	caps := []Capability{CapabilityCompletion}
-	switch {
+	r, m, opts, err := s.scheduleRunner(c.Request.Context(), req.Model, caps, req.Options, req.KeepAlive)
-	case req.Model == "":
+	if errors.Is(err, errCapabilityCompletion) {
-		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "model is required"})
+		c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("%q does not support chat", req.Model)})
 		return
-	case len(req.Format) > 0 && req.Format != "json":
+	} else if err != nil {
-		c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "format must be json"})
+		handleScheduleError(c, req.Model, err)
 		return
 	}
 	model, err := GetModel(req.Model)
 	if err != nil {
 		var pErr *fs.PathError
 		if errors.As(err, &pErr) {
 			c.JSON(http.StatusNotFound, gin.H{"error": fmt.Sprintf("model '%s' not found, try pulling it first", req.Model)})
 			return
 		}
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
 	if !model.Has(CapabilityCompletion) {
 		c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("%s does not support chat", req.Model)})
 		return
 	}
 	opts, err := modelOptions(model, req.Options)
 	if err != nil {
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
 	rCh, eCh := s.sched.GetRunner(c.Request.Context(), model, opts, req.KeepAlive)
 	var runner *runnerRef
 	select {
 	case runner = <-rCh:
 	case err = <-eCh:
 		handleErrorResponse(c, err)
 		return
 	}
 	checkpointLoaded := time.Now()
-	// if the first message is not a system message, then add the model's default system message
+	if len(req.Messages) == 0 {
-	if len(req.Messages) > 0 && req.Messages[0].Role != "system" {
+		c.JSON(http.StatusOK, api.ChatResponse{
 		req.Messages = append([]api.Message{
 			{
 				Role:    "system",
 				Content: model.System,
 			},
 		}, req.Messages...)
 	}
 	prompt, err := chatPrompt(c.Request.Context(), runner, model.Template, req.Messages, opts.NumCtx)
 	if err != nil {
 		c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
 		return
 	}
 	// an empty request loads the model
 	if len(req.Messages) == 0 || prompt == "" {
 		resp := api.ChatResponse{
 			CreatedAt:  time.Now().UTC(),
 			Model:      req.Model,
 			CreatedAt:  time.Now().UTC(),
 			Message:    api.Message{Role: "assistant"},
 			Done:       true,
 			DoneReason: "load",
-			Message:    api.Message{Role: "assistant"},
+		})
 		}
 		c.JSON(http.StatusOK, resp)
 		return
 	}
-	// only send images that are in the prompt
+	if req.Messages[0].Role != "system" {
-	var i int
+		req.Messages = append([]api.Message{{Role: "system", Content: m.System}}, req.Messages...)
 	var images []llm.ImageData
 	for _, m := range req.Messages {
 		for _, img := range m.Images {
 			if !isSupportedImageType(img) {
 				c.AbortWithStatusJSON(http.StatusBadRequest, gin.H{"error": "unsupported image format"})
 				return
 			}
 			if strings.Contains(prompt, fmt.Sprintf("[img-%d]", i)) {
 				images = append(images, llm.ImageData{Data: img, ID: i})
 			}
 			i += 1
 		}
 	}
-	slog.Debug("chat handler", "prompt", prompt, "images", len(images))
+	prompt, images, err := chatPrompt(c.Request.Context(), m, r.Tokenize, opts, req.Messages)
 	if err != nil {
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 		return
 	}
 	slog.Debug("chat request", "images", len(images), "prompt", prompt)
 	ch := make(chan any)
 	go func() {
 		defer close(ch)
-
+		if err := r.Completion(c.Request.Context(), llm.CompletionRequest{
-		fn := func(r llm.CompletionResponse) {
+			Prompt:  prompt,
-			resp := api.ChatResponse{
+			Images:  images,
 			Format:  req.Format,
 			Options: opts,
 		}, func(r llm.CompletionResponse) {
 			res := api.ChatResponse{
 				Model:      req.Model,
 				CreatedAt:  time.Now().UTC(),
 				Message:    api.Message{Role: "assistant", Content: r.Content},
@ -1354,62 +1210,57 @@ func (s *Server) ChatHandler(c *gin.Context) {
 			}
 			if r.Done {
-				resp.TotalDuration = time.Since(checkpointStart)
+				res.TotalDuration = time.Since(checkpointStart)
-				resp.LoadDuration = checkpointLoaded.Sub(checkpointStart)
+				res.LoadDuration = checkpointLoaded.Sub(checkpointStart)
 			}
-			ch <- resp
+			ch <- res
-		}
+		}); err != nil {
 		if err := runner.llama.Completion(c.Request.Context(), llm.CompletionRequest{
 			Prompt:  prompt,
 			Format:  req.Format,
 			Images:  images,
 			Options: opts,
 		}, fn); err != nil {
 			ch <- gin.H{"error": err.Error()}
 		}
 	}()
 	if req.Stream != nil && !*req.Stream {
-		// Accumulate responses into the final response
+		var r api.ChatResponse
 		var final api.ChatResponse
 		var sb strings.Builder
-		for resp := range ch {
+		for rr := range ch {
-			switch r := resp.(type) {
+			switch t := rr.(type) {
 			case api.ChatResponse:
-				sb.WriteString(r.Message.Content)
+				sb.WriteString(t.Message.Content)
-				final = r
+				r = t
 			case gin.H:
-				if errorMsg, ok := r["error"].(string); ok {
+				msg, ok := t["error"].(string)
-					c.JSON(http.StatusInternalServerError, gin.H{"error": errorMsg})
+				if !ok {
-					return
+					msg = "unexpected error format in response"
 				} else {
 					c.JSON(http.StatusInternalServerError, gin.H{"error": "unexpected error format in response"})
 					return
 				}
 				c.JSON(http.StatusInternalServerError, gin.H{"error": msg})
 				return
 			default:
-				c.JSON(http.StatusInternalServerError, gin.H{"error": "unexpected error"})
+				c.JSON(http.StatusInternalServerError, gin.H{"error": "unexpected response"})
 				return
 			}
 		}
-		final.Message = api.Message{Role: "assistant", Content: sb.String()}
+		r.Message.Content = sb.String()
-		c.JSON(http.StatusOK, final)
+		c.JSON(http.StatusOK, r)
 		return
 	}
 	streamResponse(c, ch)
 }
-func handleErrorResponse(c *gin.Context, err error) {
+func handleScheduleError(c *gin.Context, name string, err error) {
-	if errors.Is(err, context.Canceled) {
+	switch {
 	case errors.Is(err, errRequired):
 		c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
 	case errors.Is(err, context.Canceled):
 		c.JSON(499, gin.H{"error": "request canceled"})
-		return
+	case errors.Is(err, ErrMaxQueue):
 	}
 	if errors.Is(err, ErrMaxQueue) {
 		c.JSON(http.StatusServiceUnavailable, gin.H{"error": err.Error()})
-		return
+	case errors.Is(err, os.ErrNotExist):
 		c.JSON(http.StatusNotFound, gin.H{"error": fmt.Sprintf("model %q not found, try pulling it first", name)})
 	default:
 		c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 	}
 	c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
 }
--- a/server/routes_create_test.go
+++ b/server/routes_create_test.go
@ -545,9 +545,9 @@ func TestCreateDetectTemplate(t *testing.T) {
 		}
 		checkFileExists(t, filepath.Join(p, "blobs", "*"), []string{
 			filepath.Join(p, "blobs", "sha256-2f8e594e6f34b1b4d36a246628eeb3365ce442303d656f1fcc69e821722acea0"),
 			filepath.Join(p, "blobs", "sha256-542b217f179c7825eeb5bca3c77d2b75ed05bafbd3451d9188891a60a85337c6"),
 			filepath.Join(p, "blobs", "sha256-553c4a3f747b3d22a4946875f1cc8ed011c2930d83f864a0c7265f9ec0a20413"),
 			filepath.Join(p, "blobs", "sha256-c608dc615584cd20d9d830363dabf8a4783ae5d34245c3d8c115edb3bc7b28e4"),
 			filepath.Join(p, "blobs", "sha256-f836ee110db21567f826332e4cedd746c06d10664fd5a9ea3659e3683a944510"),
 		})
 	})
--- a/server/sched.go
+++ b/server/sched.go
@ -133,17 +133,8 @@ func (s *Scheduler) processPending(ctx context.Context) {
 				numParallel = 1
 				slog.Warn("multimodal models don't support parallel requests yet")
 			}
 			// Keep NumCtx and numParallel in sync
 			if numParallel > 1 {
 				pending.opts.NumCtx = pending.origNumCtx * numParallel
 			}
 			for {
 				cpus := s.getCpuFn()
 				var systemMem gpu.GpuInfo
 				if len(cpus) > 0 {
 					systemMem = cpus[0]
 				}
 				var runnerToExpire *runnerRef
 				s.loadedMu.Lock()
 				runner := s.loaded[pending.model.ModelPath]
@ -197,35 +188,15 @@ func (s *Scheduler) processPending(ctx context.Context) {
 						break
 					}
 					// Block attempting to load a model larger than system memory + GPU memory
 					estimate := llm.EstimateGPULayers(gpus, ggml, pending.model.ProjectorPaths, pending.opts)
 					maxSize := systemMem.FreeMemory
 					for _, gpu := range gpus {
 						if gpu.Library == "cpu" {
 							continue
 						}
 						if loadedCount == 0 {
 							// If no other models are loaded, set the limit based on what's available
 							maxSize += gpu.FreeMemory
 						} else {
 							// Other models could be unloaded, favor total memory for limit
 							maxSize += gpu.TotalMemory
 						}
 					}
 					if estimate.TotalSize > maxSize {
 						slog.Warn("model request too large for system", "requested", format.HumanBytes2(estimate.TotalSize), "system", format.HumanBytes2(maxSize))
 						pending.errCh <- fmt.Errorf("requested model (%s) is too large for this system (%s)", format.HumanBytes2(estimate.TotalSize), format.HumanBytes2(maxSize))
 						break
 					}
 					// Evaluate if the model will fit in the available system memory, or if we should unload a model first
 					if len(gpus) == 1 && gpus[0].Library == "cpu" {
 						// simplifying assumption of defaultParallel when in CPU mode
 						if numParallel <= 0 {
 							numParallel = defaultParallel
 							pending.opts.NumCtx = pending.origNumCtx * numParallel
 						}
 						pending.opts.NumCtx = pending.origNumCtx * numParallel
 						if loadedCount == 0 {
 							slog.Debug("cpu mode with first model, loading")
 							s.loadFn(pending, ggml, gpus, numParallel)
--- a/template/alpaca.gotmpl
+++ b/template/alpaca.gotmpl
@ -5,3 +5,4 @@
 {{ end }}### Response:
 {{ .Response }}
--- a/template/chatqa.gotmpl
+++ b/template/chatqa.gotmpl
@ -2,4 +2,5 @@
 {{ end }}{{ if .Prompt }}User: {{ .Prompt }}
-{{ end }}Assistant: <|begin_of_text|>{{ .Response }}
+{{ end }}Assistant: {{ .Response }}
--- a/template/codellama-70b-instruct.gotmpl
+++ b/template/codellama-70b-instruct.gotmpl
@ -1,8 +1,10 @@
-{{ if .System }} Source: system
+{{ if .System }}Source: system
- {{ .System }} <step>{{ end }} Source: user
+ {{ .System }} <step> {{ end }}Source: user
 {{ .Prompt }} <step> Source: assistant
 {{- if not .Response }}
 Destination: user
 {{- end }}
- {{ .Response }}<step>
+ {{ .Response }} <step> 
--- a/template/falcon-instruct.gotmpl
+++ b/template/falcon-instruct.gotmpl
@ -1,3 +1,5 @@
-{{ if .System }}{{ .System }}
+{{ if .System }}System: {{ .System }}
-{{ end }}{{ if .Prompt }}User: {{ .Prompt }}
+{{ end }}{{ if .Prompt }}User:
-{{ end }}Assistant: {{ .Response }}
+{{ .Prompt }}
 {{ end }}Falcon:
 {{ .Response }}
--- a/template/gemma-instruct.gotmpl
+++ b/template/gemma-instruct.gotmpl
@ -1,4 +1,5 @@
 <start_of_turn>user
-{{ if .System }}{{ .System }} {{ end }}{{ .Prompt }}<end_of_turn>
+{{ if .System }}{{ .System }}
 {{ end }}{{ .Prompt }}<end_of_turn>
 <start_of_turn>model
 {{ .Response }}<end_of_turn>
--- a/template/granite-instruct.gotmpl
+++ b/template/granite-instruct.gotmpl
@ -1,5 +1,4 @@
-{{ if .System }}
+{{ if .System }}System:
 System:
 {{ .System }}
 {{ end }}{{ if .Prompt }}Question:
@ -7,3 +6,4 @@ System:
 {{ end }}Answer:
 {{ .Response }}
--- a/template/llama2-chat.gotmpl
+++ b/template/llama2-chat.gotmpl
@ -1,3 +1,6 @@
-[INST] <<SYS>>{{ .System }}<</SYS>>
+[INST] <<SYS>>
 {{- if .System }}
 {{ .System }}
 {{ end }}<</SYS>>
-{{ .Prompt }} [/INST] {{ .Response }}
+{{ .Prompt }} [/INST] {{ .Response }}</s><s>
--- a/template/magicoder.gotmpl
+++ b/template/magicoder.gotmpl
@ -5,3 +5,4 @@
 {{ end }}@@ Response
 {{ .Response }}
--- a/template/mistral-instruct.gotmpl
+++ b/template/mistral-instruct.gotmpl
@ -1,6 +1,3 @@
-{{ if .System }}<|im_start|>system
+[INST] {{ if .System }}{{ .System }}
-{{ .System }}<|im_end|>
+
-{{ end }}{{ if .Prompt }}<|im_start|>user
+{{ end }}{{ .Prompt }}[/INST] {{ .Response }}</s>
 {{ .Prompt }}<|im_end|>
 {{ end }}<|im_start|>assistant
 {{ .Response }}<|im_end|>
--- a/template/openchat.gotmpl
+++ b/template/openchat.gotmpl
@ -1 +1 @@
-{{ .System }}<|end_of_turn|>GPT4 Correct User: {{ .Prompt }}<|end_of_turn|>GPT4 Correct Assistant: {{ .Response }}<|end_of_turn|>
+{{ if .System }}GPT4 Correct System: {{ .System }}<|end_of_turn|>{{ end }}GPT4 Correct User: {{ .Prompt }}<|end_of_turn|>GPT4 Correct Assistant: {{ .Response }}<|end_of_turn|>
--- a/template/solar-instruct.gotmpl
+++ b/template/solar-instruct.gotmpl
@ -5,4 +5,5 @@
 {{ .Prompt }}
 {{ end }}### Assistant:
-{{ .Response }}
+{{ .Response }}</s>
--- a/template/starcoder2-instruct.gotmpl
+++ b/template/starcoder2-instruct.gotmpl
@ -3,7 +3,6 @@
 {{ end }}{{ if .Prompt }}### Instruction
 {{ .Prompt }}
 {{ end }}### Response
 {{ .Response }}<|endoftext|>
--- a/template/template.go
+++ b/template/template.go
@ -5,6 +5,7 @@ import (
 	"embed"
 	"encoding/json"
 	"errors"
 	"fmt"
 	"io"
 	"math"
 	"slices"
@ -14,6 +15,7 @@ import (
 	"text/template/parse"
 	"github.com/agnivade/levenshtein"
 	"github.com/ollama/ollama/api"
 	"golang.org/x/exp/maps"
 )
@ -74,30 +76,59 @@ func Named(s string) (*named, error) {
 	return nil, errors.New("no matching template found")
 }
 var DefaultTemplate, _ = Parse("{{ .Prompt }}")
 type Template struct {
 	*template.Template
 	raw string
 }
 // response is a template node that can be added to templates that don't already have one
 var response = parse.ActionNode{
 	NodeType: parse.NodeAction,
 	Pipe: &parse.PipeNode{
 		NodeType: parse.NodePipe,
 		Cmds: []*parse.CommandNode{
 			{
 				NodeType: parse.NodeCommand,
 				Args: []parse.Node{
 					&parse.FieldNode{
 						NodeType: parse.NodeField,
 						Ident:    []string{"Response"},
 					},
 				},
 			},
 		},
 	},
 }
 func Parse(s string) (*Template, error) {
 	tmpl := template.New("").Option("missingkey=zero")
 	tmpl, err := tmpl.Parse(s)
 	if err != nil {
 		return nil, err
 	}
 	t := Template{Template: tmpl, raw: s}
 	if vars := t.Vars(); !slices.Contains(vars, "messages") && !slices.Contains(vars, "response") {
 		// touch up the template and append {{ .Response }}
 		tmpl.Tree.Root.Nodes = append(tmpl.Tree.Root.Nodes, &response)
 	}
 	return &t, nil
 }
 func (t *Template) String() string {
 	return t.raw
 }
 var DefaultTemplate, _ = Parse("{{ .Prompt }}")
 func Parse(s string) (*Template, error) {
 	t, err := template.New("").Option("missingkey=zero").Parse(s)
 	if err != nil {
 		return nil, err
 	}
 	return &Template{Template: t, raw: s}, nil
 }
 func (t *Template) Vars() []string {
 	var vars []string
-	for _, n := range t.Tree.Root.Nodes {
+	for _, tt := range t.Templates() {
-		vars = append(vars, parseNode(n)...)
+		for _, n := range tt.Root.Nodes {
 			vars = append(vars, parseNode(n)...)
 		}
 	}
 	set := make(map[string]struct{})
@ -110,6 +141,120 @@ func (t *Template) Vars() []string {
 	return vars
 }
 type Values struct {
 	Messages []api.Message
 	// forceLegacy is a flag used to test compatibility with legacy templates
 	forceLegacy bool
 }
 func (t *Template) Execute(w io.Writer, v Values) error {
 	system, messages := collate(v.Messages)
 	if !v.forceLegacy && slices.Contains(t.Vars(), "messages") {
 		return t.Template.Execute(w, map[string]any{
 			"System":   system,
 			"Messages": messages,
 		})
 	}
 	system = ""
 	var b bytes.Buffer
 	var prompt, response string
 	for _, m := range messages {
 		execute := func () error {
 			if err := t.Template.Execute(&b, map[string]any{
 				"System":   system,
 				"Prompt":   prompt,
 				"Response": response,
 			}); err != nil {
 				return err
 			}
 			system = ""
 			prompt = ""
 			response = ""
 			return nil
 		}
 		switch m.Role {
 		case "system":
 			if prompt != "" || response != "" {
 				if err := execute(); err != nil {
 					return err
 				}
 			}
 			system = m.Content
 		case "user":
 			if response != "" {
 				if err := execute(); err != nil {
 					return err
 				}
 			}
 			prompt = m.Content
 		case "assistant":
 			response = m.Content
 		}
 	}
 	var cut bool
 	nodes := deleteNode(t.Template.Root.Copy(), func(n parse.Node) bool {
 		switch t := n.(type) {
 		case *parse.ActionNode:
 		case *parse.FieldNode:
 			if slices.Contains(t.Ident, "Response") {
 				cut = true
 			}
 		}
 		return cut
 	})
 	tree := parse.Tree{Root: nodes.(*parse.ListNode)}
 	if err := template.Must(template.New("").AddParseTree("", &tree)).Execute(&b, map[string]any{
 		"System": system,
 		"Prompt": prompt,
 	}); err != nil {
 		return err
 	}
 	_, err := io.Copy(w, &b)
 	return err
 }
 // collate messages based on role. consecutive messages of the same role are merged
 // into a single message. collate also collects and returns all system messages.
 // collate mutates message content adding image tags ([img-%d]) as needed
 func collate(msgs []api.Message) (string, []*api.Message) {
 	var n int
 	var system []string
 	var collated []*api.Message
 	for i := range msgs {
 		msg := msgs[i]
 		for range msg.Images {
 			imageTag := fmt.Sprintf("[img-%d]", n)
 			if !strings.Contains(msg.Content, "[img]") {
 				msg.Content = strings.TrimSpace("[img] " + msg.Content)
 			}
 			msg.Content = strings.Replace(msg.Content, "[img]", imageTag, 1)
 			n++
 		}
 		if msg.Role == "system" {
 			system = append(system, msg.Content)
 		}
 		if len(collated) > 0 && collated[len(collated)-1].Role == msg.Role {
 			collated[len(collated)-1].Content += "\n\n" + msg.Content
 		} else {
 			collated = append(collated, &msg)
 		}
 	}
 	return strings.Join(system, "\n\n"), collated
 }
 func parseNode(n parse.Node) []string {
 	switch n := n.(type) {
 	case *parse.ActionNode:
@ -152,7 +297,78 @@ func parseNode(n parse.Node) []string {
 		return names
 	case *parse.FieldNode:
 		return n.Ident
 	case *parse.TemplateNode:
 		return parseNode(n.Pipe)
 	}
 	return nil
 }
 // deleteNode walks the node list and deletes nodes that match the predicate
 // this is currently to remove the {{ .Response }} node from templates
 func deleteNode(n parse.Node, fn func(parse.Node) bool) parse.Node {
 	var walk func(n parse.Node) parse.Node
 	walk = func(n parse.Node) parse.Node {
 		if fn(n) {
 			return nil
 		}
 		switch t := n.(type) {
 		case *parse.ListNode:
 			var nodes []parse.Node
 			for _, c := range t.Nodes {
 				if n := walk(c); n != nil {
 					nodes = append(nodes, n)
 				}
 			}
 			t.Nodes = nodes
 			return t
 		case *parse.IfNode:
 			t.BranchNode = *(walk(&t.BranchNode).(*parse.BranchNode))
 		case *parse.WithNode:
 			t.BranchNode = *(walk(&t.BranchNode).(*parse.BranchNode))
 		case *parse.RangeNode:
 			t.BranchNode = *(walk(&t.BranchNode).(*parse.BranchNode))
 		case *parse.BranchNode:
 			t.List = walk(t.List).(*parse.ListNode)
 			if t.ElseList != nil {
 				t.ElseList = walk(t.ElseList).(*parse.ListNode)
 			}
 		case *parse.ActionNode:
 			n := walk(t.Pipe)
 			if n == nil {
 				return nil
 			}
 			t.Pipe = n.(*parse.PipeNode)
 		case *parse.PipeNode:
 			var commands []*parse.CommandNode
 			for _, c := range t.Cmds {
 				var args []parse.Node
 				for _, a := range c.Args {
 					if n := walk(a); n != nil {
 						args = append(args, n)
 					}
 				}
 				if len(args) == 0 {
 					return nil
 				}
 				c.Args = args
 				commands = append(commands, c)
 			}
 			if len(commands) == 0 {
 				return nil
 			}
 			t.Cmds = commands
 		}
 		return n
 	}
 	return walk(n)
 }
--- a/template/template_test.go
+++ b/template/template_test.go
@ -8,9 +8,11 @@ import (
 	"os"
 	"path/filepath"
 	"slices"
 	"strings"
 	"testing"
 	"text/template"
 	"github.com/google/go-cmp/cmp"
 	"github.com/ollama/ollama/api"
 	"github.com/ollama/ollama/llm"
 )
@ -46,7 +48,7 @@ func TestNamed(t *testing.T) {
 					t.Fatal(err)
 				}
-				tmpl, err := template.New(s).Parse(b.String())
+				tmpl, err := Parse(b.String())
 				if err != nil {
 					t.Fatal(err)
 				}
@ -59,18 +61,125 @@ func TestNamed(t *testing.T) {
 	}
 }
 func TestTemplate(t *testing.T) {
 	cases := make(map[string][]api.Message)
 	for _, mm := range [][]api.Message{
 		{
 			{Role: "user", Content: "Hello, how are you?"},
 		},
 		{
 			{Role: "user", Content: "Hello, how are you?"},
 			{Role: "assistant", Content: "I'm doing great. How can I help you today?"},
 			{Role: "user", Content: "I'd like to show off how chat templating works!"},
 		},
 		{
 			{Role: "system", Content: "You are a helpful assistant."},
 			{Role: "user", Content: "Hello, how are you?"},
 			{Role: "assistant", Content: "I'm doing great. How can I help you today?"},
 			{Role: "user", Content: "I'd like to show off how chat templating works!"},
 		},
 	} {
 		var roles []string
 		for _, m := range mm {
 			roles = append(roles, m.Role)
 		}
 		cases[strings.Join(roles, "-")] = mm
 	}
 	matches, err := filepath.Glob("*.gotmpl")
 	if err != nil {
 		t.Fatal(err)
 	}
 	for _, match := range matches {
 		t.Run(match, func(t *testing.T) {
 			bts, err := os.ReadFile(match)
 			if err != nil {
 				t.Fatal(err)
 			}
 			tmpl, err := Parse(string(bts))
 			if err != nil {
 				t.Fatal(err)
 			}
 			for n, tt := range cases {
 				var actual bytes.Buffer
 				t.Run(n, func(t *testing.T) {
 					if err := tmpl.Execute(&actual, Values{Messages: tt}); err != nil {
 						t.Fatal(err)
 					}
 					expect, err := os.ReadFile(filepath.Join("testdata", match, n))
 					if err != nil {
 						t.Fatal(err)
 					}
 					bts := actual.Bytes()
 					if slices.Contains([]string{"chatqa.gotmpl", "llama2-chat.gotmpl", "mistral-instruct.gotmpl", "openchat.gotmpl", "vicuna.gotmpl"}, match) && bts[len(bts)-1] == ' ' {
 						t.Log("removing trailing space from output")
 						bts = bts[:len(bts)-1]
 					}
 					if diff := cmp.Diff(bts, expect); diff != "" {
 						t.Errorf("mismatch (-got +want):\n%s", diff)
 					}
 				})
 				t.Run("legacy", func(t *testing.T) {
 					t.Skip("legacy outputs are currently default outputs")
 					var legacy bytes.Buffer
 					if err := tmpl.Execute(&legacy, Values{Messages: tt, forceLegacy: true}); err != nil {
 						t.Fatal(err)
 					}
 					legacyBytes := legacy.Bytes()
 					if slices.Contains([]string{"chatqa.gotmpl", "openchat.gotmpl", "vicuna.gotmpl"}, match) && legacyBytes[len(legacyBytes)-1] == ' ' {
 						t.Log("removing trailing space from legacy output")
 						legacyBytes = legacyBytes[:len(legacyBytes)-1]
 					} else if slices.Contains([]string{"codellama-70b-instruct.gotmpl", "llama2-chat.gotmpl", "mistral-instruct.gotmpl"}, match) {
 						t.Skip("legacy outputs cannot be compared to messages outputs")
 					}
 					if diff := cmp.Diff(legacyBytes, actual.Bytes()); diff != "" {
 						t.Errorf("mismatch (-got +want):\n%s", diff)
 					}
 				})
 			}
 		})
 	}
 }
 func TestParse(t *testing.T) {
 	cases := []struct {
 		template string
 		vars     []string
 	}{
-		{"{{ .Prompt }}", []string{"prompt"}},
+		{"{{ .Prompt }}", []string{"prompt", "response"}},
-		{"{{ .System }} {{ .Prompt }}", []string{"prompt", "system"}},
+		{"{{ .System }} {{ .Prompt }}", []string{"prompt", "response", "system"}},
 		{"{{ .System }} {{ .Prompt }} {{ .Response }}", []string{"prompt", "response", "system"}},
-		{"{{ with .Tools }}{{ . }}{{ end }} {{ .System }} {{ .Prompt }}", []string{"prompt", "system", "tools"}},
+		{"{{ with .Tools }}{{ . }}{{ end }} {{ .System }} {{ .Prompt }}", []string{"prompt", "response", "system", "tools"}},
 		{"{{ range .Messages }}{{ .Role }} {{ .Content }}{{ end }}", []string{"content", "messages", "role"}},
-		{"{{ range .Messages }}{{ if eq .Role \"system\" }}SYSTEM: {{ .Content }}{{ else if eq .Role \"user\" }}USER: {{ .Content }}{{ else if eq .Role \"assistant\" }}ASSISTANT: {{ .Content }}{{ end }}{{ end }}", []string{"content", "messages", "role"}},
+		{`{{- range .Messages }}
-		{"{{ .Prompt }} {{ .Suffix }}", []string{"prompt", "suffix"}},
+{{- if eq .Role "system" }}SYSTEM:
 {{- else if eq .Role "user" }}USER:
 {{- else if eq .Role "assistant" }}ASSISTANT:
 {{- end }} {{ .Content }}
 {{- end }}`, []string{"content", "messages", "role"}},
 		{`{{- if .Messages }}
 {{- range .Messages }}<|im_start|>{{ .Role }}
 {{ .Content }}<|im_end|>
 {{ end }}<|im_start|>assistant
 {{ else -}}
 {{ if .System }}<|im_start|>system
 {{ .System }}<|im_end|>
 {{ end }}{{ if .Prompt }}<|im_start|>user
 {{ .Prompt }}<|im_end|>
 {{ end }}<|im_start|>assistant
 {{ .Response }}<|im_end|>
 {{- end -}}`, []string{"content", "messages", "prompt", "response", "role", "system"}},
 	}
 	for _, tt := range cases {
@ -80,9 +189,172 @@ func TestParse(t *testing.T) {
 				t.Fatal(err)
 			}
-			vars := tmpl.Vars()
+			if diff := cmp.Diff(tmpl.Vars(), tt.vars); diff != "" {
-			if !slices.Equal(tt.vars, vars) {
+				t.Errorf("mismatch (-got +want):\n%s", diff)
-				t.Errorf("expected %v, got %v", tt.vars, vars)
+			}
 		})
 	}
 }
 func TestExecuteWithMessages(t *testing.T) {
 	type template struct {
 		name     string
 		template string
 	}
 	cases := []struct {
 		name      string
 		templates []template
 		values    Values
 		expected  string
 	}{
 		{
 			"mistral",
 			[]template{
 				{"no response", `[INST] {{ if .System }}{{ .System }}
 {{ end }}{{ .Prompt }}[/INST] `},
 				{"response", `[INST] {{ if .System }}{{ .System }}
 {{ end }}{{ .Prompt }}[/INST] {{ .Response }}`},
 				{"messages", `[INST] {{ if .System }}{{ .System }}
 {{ end }}
 {{- range .Messages }}
 {{- if eq .Role "user" }}{{ .Content }}[/INST] {{ else if eq .Role "assistant" }}{{ .Content }}[INST] {{ end }}
 {{- end }}`},
 			},
 			Values{
 				Messages: []api.Message{
 					{Role: "user", Content: "Hello friend!"},
 					{Role: "assistant", Content: "Hello human!"},
 					{Role: "user", Content: "What is your name?"},
 				},
 			},
 			`[INST] Hello friend![/INST] Hello human![INST] What is your name?[/INST] `,
 		},
 		{
 			"mistral system",
 			[]template{
 				{"no response", `[INST] {{ if .System }}{{ .System }}
 {{ end }}{{ .Prompt }}[/INST] `},
 				{"response", `[INST] {{ if .System }}{{ .System }}
 {{ end }}{{ .Prompt }}[/INST] {{ .Response }}`},
 				{"messages", `[INST] {{ if .System }}{{ .System }}
 {{ end }}
 {{- range .Messages }}
 {{- if eq .Role "user" }}{{ .Content }}[/INST] {{ else if eq .Role "assistant" }}{{ .Content }}[INST] {{ end }}
 {{- end }}`},
 			},
 			Values{
 				Messages: []api.Message{
 					{Role: "system", Content: "You are a helpful assistant!"},
 					{Role: "user", Content: "Hello friend!"},
 					{Role: "assistant", Content: "Hello human!"},
 					{Role: "user", Content: "What is your name?"},
 				},
 			},
 			`[INST] You are a helpful assistant!
 Hello friend![/INST] Hello human![INST] What is your name?[/INST] `,
 		},
 		{
 			"chatml",
 			[]template{
 				// this does not have a "no response" test because it's impossible to render the same output
 				{"response", `{{ if .System }}<|im_start|>system
 {{ .System }}<|im_end|>
 {{ end }}{{ if .Prompt }}<|im_start|>user
 {{ .Prompt }}<|im_end|>
 {{ end }}<|im_start|>assistant
 {{ .Response }}<|im_end|>
 `},
 				{"messages", `
 {{- range $index, $_ := .Messages }}<|im_start|>{{ .Role }}
 {{ .Content }}<|im_end|>
 {{ end }}<|im_start|>assistant
 `},
 			},
 			Values{
 				Messages: []api.Message{
 					{Role: "system", Content: "You are a helpful assistant!"},
 					{Role: "user", Content: "Hello friend!"},
 					{Role: "assistant", Content: "Hello human!"},
 					{Role: "user", Content: "What is your name?"},
 				},
 			},
 			`<|im_start|>system
 You are a helpful assistant!<|im_end|>
 <|im_start|>user
 Hello friend!<|im_end|>
 <|im_start|>assistant
 Hello human!<|im_end|>
 <|im_start|>user
 What is your name?<|im_end|>
 <|im_start|>assistant
 `,
 		},
 		{
 			"moondream",
 			[]template{
 				// this does not have a "no response" test because it's impossible to render the same output
 				{"response", `{{ if .Prompt }}Question: {{ .Prompt }}
 {{ end }}Answer: {{ .Response }}
 `},
 				{"messages", `
 {{- range .Messages }}
 {{- if eq .Role "user" }}Question: {{ .Content }}
 {{ else if eq .Role "assistant" }}Answer: {{ .Content }}
 {{ end }}
 {{- end }}Answer: `},
 			},
 			Values{
 				Messages: []api.Message{
 					{Role: "user", Content: "What's in this image?", Images: []api.ImageData{[]byte("")}},
 					{Role: "assistant", Content: "It's a hot dog."},
 					{Role: "user", Content: "What's in _this_ image?"},
 					{Role: "user", Images: []api.ImageData{[]byte("")}},
 					{Role: "user", Content: "Is it a hot dog?"},
 				},
 			},
 			`Question: [img-0] What's in this image?
 Answer: It's a hot dog.
 Question: What's in _this_ image?
 [img-1]
 Is it a hot dog?
 Answer: `,
 		},
 	}
 	for _, tt := range cases {
 		t.Run(tt.name, func(t *testing.T) {
 			for _, ttt := range tt.templates {
 				t.Run(ttt.name, func(t *testing.T) {
 					tmpl, err := Parse(ttt.template)
 					if err != nil {
 						t.Fatal(err)
 					}
 					var b bytes.Buffer
 					if err := tmpl.Execute(&b, tt.values); err != nil {
 						t.Fatal(err)
 					}
 					if diff := cmp.Diff(b.String(), tt.expected); diff != "" {
 						t.Errorf("mismatch (-got +want):\n%s", diff)
 					}
 				})
 			}
 		})
 	}
--- a/template/testdata/alfred.gotmpl/system-user-assistant-user
+++ b/template/testdata/alfred.gotmpl/system-user-assistant-user
@ -0,0 +1 @@
 <start_system>You are a helpful assistant.<end_message><start_user>Hello, how are you?<end_message><start_assistant>I'm doing great. How can I help you today?<end_message><start_user>I'd like to show off how chat templating works!<end_message><start_assistant>
--- a/template/testdata/alfred.gotmpl/user
+++ b/template/testdata/alfred.gotmpl/user
@ -0,0 +1 @@
 <start_user>Hello, how are you?<end_message><start_assistant>
--- a/template/testdata/alfred.gotmpl/user-assistant-user
+++ b/template/testdata/alfred.gotmpl/user-assistant-user
@ -0,0 +1 @@
 <start_user>Hello, how are you?<end_message><start_assistant>I'm doing great. How can I help you today?<end_message><start_user>I'd like to show off how chat templating works!<end_message><start_assistant>
--- a/template/testdata/alpaca.gotmpl/system-user-assistant-user
+++ b/template/testdata/alpaca.gotmpl/system-user-assistant-user
@ -0,0 +1,12 @@
 You are a helpful assistant.
 ### Instruction:
 Hello, how are you?
 ### Response:
 I'm doing great. How can I help you today?
 ### Instruction:
 I'd like to show off how chat templating works!
 ### Response:
--- a/template/testdata/alpaca.gotmpl/user
+++ b/template/testdata/alpaca.gotmpl/user
@ -0,0 +1,4 @@
 ### Instruction:
 Hello, how are you?
 ### Response:
--- a/template/testdata/alpaca.gotmpl/user-assistant-user
+++ b/template/testdata/alpaca.gotmpl/user-assistant-user
@ -0,0 +1,10 @@
 ### Instruction:
 Hello, how are you?
 ### Response:
 I'm doing great. How can I help you today?
 ### Instruction:
 I'd like to show off how chat templating works!
 ### Response:
--- a/template/testdata/chatml.gotmpl/system-user-assistant-user
+++ b/template/testdata/chatml.gotmpl/system-user-assistant-user
@ -0,0 +1,9 @@
 <|im_start|>system
 You are a helpful assistant.<|im_end|>
 <|im_start|>user
 Hello, how are you?<|im_end|>
 <|im_start|>assistant
 I'm doing great. How can I help you today?<|im_end|>
 <|im_start|>user
 I'd like to show off how chat templating works!<|im_end|>
 <|im_start|>assistant
--- a/template/testdata/chatml.gotmpl/user
+++ b/template/testdata/chatml.gotmpl/user
@ -0,0 +1,3 @@
 <|im_start|>user
 Hello, how are you?<|im_end|>
 <|im_start|>assistant
--- a/template/testdata/chatml.gotmpl/user-assistant-user
+++ b/template/testdata/chatml.gotmpl/user-assistant-user
@ -0,0 +1,7 @@
 <|im_start|>user
 Hello, how are you?<|im_end|>
 <|im_start|>assistant
 I'm doing great. How can I help you today?<|im_end|>
 <|im_start|>user
 I'd like to show off how chat templating works!<|im_end|>
 <|im_start|>assistant
--- a/template/testdata/chatqa.gotmpl/system-user-assistant-user
+++ b/template/testdata/chatqa.gotmpl/system-user-assistant-user
@ -0,0 +1,9 @@
 System: You are a helpful assistant.
 User: Hello, how are you?
 Assistant: I'm doing great. How can I help you today?
 User: I'd like to show off how chat templating works!
 Assistant:
--- a/template/testdata/chatqa.gotmpl/user
+++ b/template/testdata/chatqa.gotmpl/user
@ -0,0 +1,3 @@
 User: Hello, how are you?
 Assistant:
--- a/template/testdata/chatqa.gotmpl/user-assistant-user
+++ b/template/testdata/chatqa.gotmpl/user-assistant-user
@ -0,0 +1,7 @@
 User: Hello, how are you?
 Assistant: I'm doing great. How can I help you today?
 User: I'd like to show off how chat templating works!
 Assistant:
--- a/template/testdata/codellama-70b-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/codellama-70b-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,12 @@
 Source: system
 You are a helpful assistant. <step> Source: user
 Hello, how are you? <step> Source: assistant
 I'm doing great. How can I help you today? <step> Source: user
 I'd like to show off how chat templating works! <step> Source: assistant
 Destination: user
--- a/template/testdata/codellama-70b-instruct.gotmpl/user
+++ b/template/testdata/codellama-70b-instruct.gotmpl/user
@ -0,0 +1,6 @@
 Source: user
 Hello, how are you? <step> Source: assistant
 Destination: user
--- a/template/testdata/codellama-70b-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/codellama-70b-instruct.gotmpl/user-assistant-user
@ -0,0 +1,10 @@
 Source: user
 Hello, how are you? <step> Source: assistant
 I'm doing great. How can I help you today? <step> Source: user
 I'd like to show off how chat templating works! <step> Source: assistant
 Destination: user
--- a/template/testdata/falcon-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/falcon-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,8 @@
 System: You are a helpful assistant.
 User:
 Hello, how are you?
 Falcon:
 I'm doing great. How can I help you today?
 User:
 I'd like to show off how chat templating works!
 Falcon:
--- a/template/testdata/falcon-instruct.gotmpl/user
+++ b/template/testdata/falcon-instruct.gotmpl/user
@ -0,0 +1,3 @@
 User:
 Hello, how are you?
 Falcon:
--- a/template/testdata/falcon-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/falcon-instruct.gotmpl/user-assistant-user
@ -0,0 +1,7 @@
 User:
 Hello, how are you?
 Falcon:
 I'm doing great. How can I help you today?
 User:
 I'd like to show off how chat templating works!
 Falcon:
--- a/template/testdata/gemma-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/gemma-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,8 @@
 <start_of_turn>user
 You are a helpful assistant.
 Hello, how are you?<end_of_turn>
 <start_of_turn>model
 I'm doing great. How can I help you today?<end_of_turn>
 <start_of_turn>user
 I'd like to show off how chat templating works!<end_of_turn>
 <start_of_turn>model
--- a/template/testdata/gemma-instruct.gotmpl/user
+++ b/template/testdata/gemma-instruct.gotmpl/user
@ -0,0 +1,3 @@
 <start_of_turn>user
 Hello, how are you?<end_of_turn>
 <start_of_turn>model
--- a/template/testdata/gemma-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/gemma-instruct.gotmpl/user-assistant-user
@ -0,0 +1,7 @@
 <start_of_turn>user
 Hello, how are you?<end_of_turn>
 <start_of_turn>model
 I'm doing great. How can I help you today?<end_of_turn>
 <start_of_turn>user
 I'd like to show off how chat templating works!<end_of_turn>
 <start_of_turn>model
--- a/template/testdata/granite-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/granite-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,13 @@
 System:
 You are a helpful assistant.
 Question:
 Hello, how are you?
 Answer:
 I'm doing great. How can I help you today?
 Question:
 I'd like to show off how chat templating works!
 Answer:
--- a/template/testdata/granite-instruct.gotmpl/user
+++ b/template/testdata/granite-instruct.gotmpl/user
@ -0,0 +1,4 @@
 Question:
 Hello, how are you?
 Answer:
--- a/template/testdata/granite-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/granite-instruct.gotmpl/user-assistant-user
@ -0,0 +1,10 @@
 Question:
 Hello, how are you?
 Answer:
 I'm doing great. How can I help you today?
 Question:
 I'd like to show off how chat templating works!
 Answer:
--- a/template/testdata/llama2-chat.gotmpl/system-user-assistant-user
+++ b/template/testdata/llama2-chat.gotmpl/system-user-assistant-user
@ -0,0 +1,7 @@
 [INST] <<SYS>>
 You are a helpful assistant.
 <</SYS>>
 Hello, how are you? [/INST] I'm doing great. How can I help you today?</s><s>[INST] <<SYS>><</SYS>>
 I'd like to show off how chat templating works! [/INST]
--- a/template/testdata/llama2-chat.gotmpl/user
+++ b/template/testdata/llama2-chat.gotmpl/user
@ -0,0 +1,3 @@
 [INST] <<SYS>><</SYS>>
 Hello, how are you? [/INST]
--- a/template/testdata/llama2-chat.gotmpl/user-assistant-user
+++ b/template/testdata/llama2-chat.gotmpl/user-assistant-user
@ -0,0 +1,5 @@
 [INST] <<SYS>><</SYS>>
 Hello, how are you? [/INST] I'm doing great. How can I help you today?</s><s>[INST] <<SYS>><</SYS>>
 I'd like to show off how chat templating works! [/INST]
--- a/template/testdata/llama3-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/llama3-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,10 @@
 <|start_header_id|>system<|end_header_id|>
 You are a helpful assistant.<|eot_id|><|start_header_id|>user<|end_header_id|>
 Hello, how are you?<|eot_id|><|start_header_id|>assistant<|end_header_id|>
 I'm doing great. How can I help you today?<|eot_id|><|start_header_id|>user<|end_header_id|>
 I'd like to show off how chat templating works!<|eot_id|><|start_header_id|>assistant<|end_header_id|>
--- a/template/testdata/llama3-instruct.gotmpl/user
+++ b/template/testdata/llama3-instruct.gotmpl/user
@ -0,0 +1,4 @@
 <|start_header_id|>user<|end_header_id|>
 Hello, how are you?<|eot_id|><|start_header_id|>assistant<|end_header_id|>
--- a/template/testdata/llama3-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/llama3-instruct.gotmpl/user-assistant-user
@ -0,0 +1,8 @@
 <|start_header_id|>user<|end_header_id|>
 Hello, how are you?<|eot_id|><|start_header_id|>assistant<|end_header_id|>
 I'm doing great. How can I help you today?<|eot_id|><|start_header_id|>user<|end_header_id|>
 I'd like to show off how chat templating works!<|eot_id|><|start_header_id|>assistant<|end_header_id|>
--- a/template/testdata/magicoder.gotmpl/system-user-assistant-user
+++ b/template/testdata/magicoder.gotmpl/system-user-assistant-user
@ -0,0 +1,12 @@
 You are a helpful assistant.
@@ Instruction
 Hello, how are you?
@@ Response
 I'm doing great. How can I help you today?
@@ Instruction
 I'd like to show off how chat templating works!
@@ Response
--- a/template/testdata/magicoder.gotmpl/user
+++ b/template/testdata/magicoder.gotmpl/user
@ -0,0 +1,4 @@
@@ Instruction
 Hello, how are you?
@@ Response
--- a/template/testdata/magicoder.gotmpl/user-assistant-user
+++ b/template/testdata/magicoder.gotmpl/user-assistant-user
@ -0,0 +1,10 @@
@@ Instruction
 Hello, how are you?
@@ Response
 I'm doing great. How can I help you today?
@@ Instruction
 I'd like to show off how chat templating works!
@@ Response
--- a/template/testdata/mistral-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/mistral-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,3 @@
 [INST] You are a helpful assistant.
 Hello, how are you?[/INST] I'm doing great. How can I help you today?</s>[INST] I'd like to show off how chat templating works![/INST]
--- a/template/testdata/mistral-instruct.gotmpl/user
+++ b/template/testdata/mistral-instruct.gotmpl/user
@ -0,0 +1 @@
 [INST] Hello, how are you?[/INST]
--- a/template/testdata/mistral-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/mistral-instruct.gotmpl/user-assistant-user
@ -0,0 +1 @@
 [INST] Hello, how are you?[/INST] I'm doing great. How can I help you today?</s>[INST] I'd like to show off how chat templating works![/INST]
--- a/template/testdata/openchat.gotmpl/system-user-assistant-user
+++ b/template/testdata/openchat.gotmpl/system-user-assistant-user
@ -0,0 +1 @@
 GPT4 Correct System: You are a helpful assistant.<|end_of_turn|>GPT4 Correct User: Hello, how are you?<|end_of_turn|>GPT4 Correct Assistant: I'm doing great. How can I help you today?<|end_of_turn|>GPT4 Correct User: I'd like to show off how chat templating works!<|end_of_turn|>GPT4 Correct Assistant:
--- a/template/testdata/openchat.gotmpl/user
+++ b/template/testdata/openchat.gotmpl/user
@ -0,0 +1 @@
 GPT4 Correct User: Hello, how are you?<|end_of_turn|>GPT4 Correct Assistant:
--- a/template/testdata/openchat.gotmpl/user-assistant-user
+++ b/template/testdata/openchat.gotmpl/user-assistant-user
@ -0,0 +1 @@
 GPT4 Correct User: Hello, how are you?<|end_of_turn|>GPT4 Correct Assistant: I'm doing great. How can I help you today?<|end_of_turn|>GPT4 Correct User: I'd like to show off how chat templating works!<|end_of_turn|>GPT4 Correct Assistant:
--- a/template/testdata/phi-3.gotmpl/system-user-assistant-user
+++ b/template/testdata/phi-3.gotmpl/system-user-assistant-user
@ -0,0 +1,9 @@
 <|system|>
 You are a helpful assistant.<|end|>
 <|user|>
 Hello, how are you?<|end|>
 <|assistant|>
 I'm doing great. How can I help you today?<|end|>
 <|user|>
 I'd like to show off how chat templating works!<|end|>
 <|assistant|>
--- a/template/testdata/phi-3.gotmpl/user
+++ b/template/testdata/phi-3.gotmpl/user
@ -0,0 +1,3 @@
 <|user|>
 Hello, how are you?<|end|>
 <|assistant|>
--- a/template/testdata/phi-3.gotmpl/user-assistant-user
+++ b/template/testdata/phi-3.gotmpl/user-assistant-user
@ -0,0 +1,7 @@
 <|user|>
 Hello, how are you?<|end|>
 <|assistant|>
 I'm doing great. How can I help you today?<|end|>
 <|user|>
 I'd like to show off how chat templating works!<|end|>
 <|assistant|>
--- a/template/testdata/solar-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/solar-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,13 @@
 ### System:
 You are a helpful assistant.
 ### User:
 Hello, how are you?
 ### Assistant:
 I'm doing great. How can I help you today?</s>
 ### User:
 I'd like to show off how chat templating works!
 ### Assistant:
--- a/template/testdata/solar-instruct.gotmpl/user
+++ b/template/testdata/solar-instruct.gotmpl/user
@ -0,0 +1,4 @@
 ### User:
 Hello, how are you?
 ### Assistant:
--- a/template/testdata/solar-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/solar-instruct.gotmpl/user-assistant-user
@ -0,0 +1,10 @@
 ### User:
 Hello, how are you?
 ### Assistant:
 I'm doing great. How can I help you today?</s>
 ### User:
 I'd like to show off how chat templating works!
 ### Assistant:
--- a/template/testdata/starcoder2-instruct.gotmpl/system-user-assistant-user
+++ b/template/testdata/starcoder2-instruct.gotmpl/system-user-assistant-user
@ -0,0 +1,12 @@
 You are a helpful assistant.
 ### Instruction
 Hello, how are you?
 ### Response
 I'm doing great. How can I help you today?<|endoftext|>
 ### Instruction
 I'd like to show off how chat templating works!
 ### Response
--- a/template/testdata/starcoder2-instruct.gotmpl/user
+++ b/template/testdata/starcoder2-instruct.gotmpl/user
@ -0,0 +1,4 @@
 ### Instruction
 Hello, how are you?
 ### Response
--- a/template/testdata/starcoder2-instruct.gotmpl/user-assistant-user
+++ b/template/testdata/starcoder2-instruct.gotmpl/user-assistant-user
@ -0,0 +1,10 @@
 ### Instruction
 Hello, how are you?
 ### Response
 I'm doing great. How can I help you today?<|endoftext|>
 ### Instruction
 I'd like to show off how chat templating works!
 ### Response
--- a/template/testdata/vicuna.gotmpl/system-user-assistant-user
+++ b/template/testdata/vicuna.gotmpl/system-user-assistant-user
@ -0,0 +1,6 @@
 You are a helpful assistant.
 USER: Hello, how are you?
 ASSISTANT: I'm doing great. How can I help you today?</s>
 USER: I'd like to show off how chat templating works!
 ASSISTANT:
--- a/Show more
+++ b/Show more
Author	SHA1	Message	Date
baalajimaestro	8c6402d194	Merge https://github.com/ollama/ollama	2024-07-14 16:51:20 +05:30
royjhan	e9f7f36029	Support image input for OpenAI chat compatibility (#5208 ) * OpenAI v1 models * Refactor Writers * Add Test Co-Authored-By: Attila Kerekes * Credit Co-Author Co-Authored-By: Attila Kerekes <439392+keriati@users.noreply.github.com> * Empty List Testing * Use Namespace for Ownedby * Update Test * Add back envconfig * v1/models docs * Use ModelName Parser * Test Names * Remove Docs * Clean Up * Test name Co-authored-by: Jeffrey Morgan <jmorganca@gmail.com> * Add Middleware for Chat and List * Testing Cleanup * Test with Fatal * Add functionality to chat test * Support image input for OpenAI chat * Decoding * Fix message processing logic * openai vision test * type errors * clean up * redundant check * merge conflicts * merge conflicts * merge conflicts * flattening and smaller image * add test * support python and js SDKs and mandate prefixing * clean up --------- Co-authored-by: Attila Kerekes <439392+keriati@users.noreply.github.com> Co-authored-by: Jeffrey Morgan <jmorganca@gmail.com>	2024-07-13 22:07:45 -07:00
Patrick Devine	057d31861e	remove template (#5655 )	2024-07-13 20:56:24 -07:00
jmorganca	f7ee012300	server: prepend system message in chat handler	2024-07-13 15:08:00 -07:00
Jeffrey Morgan	1ed0aa8fea	server: fix `context`, `load_duration` and `total_duration` fields (#5676 ) * server: fix `contet`, `load_duration` and `total_duration` fields * Update server/routes.go	2024-07-13 09:25:31 -07:00
Jeffrey Morgan	ef98803d63	llm: looser checks for minimum memory (#5677 )	2024-07-13 09:20:05 -07:00
Jarek	02fea420e5	Add Kerlig AI, an app for macOS (#5675 )	2024-07-13 08:33:46 -07:00
Michael Yang	22c5451fc2	fix system prompt (#5662 ) * fix system prompt * execute template when hitting previous roles * fix tests --------- Co-authored-by: jmorganca <jmorganca@gmail.com>	2024-07-12 21:04:44 -07:00
Patrick Devine	23ebbaa46e	Revert "remove template from tests" This reverts commit `9ac0a7a50b`.	2024-07-12 15:47:17 -07:00
Patrick Devine	9ac0a7a50b	remove template from tests	2024-07-12 15:41:31 -07:00
Michael Yang	e5c65a85df	Merge pull request #5653 from ollama/mxyng/collect-system template: preprocess message and collect system	2024-07-12 12:32:34 -07:00
Jeffrey Morgan	33627331a3	app: also clean up tempdir runners on install (#5646 )	2024-07-12 12:29:23 -07:00
Michael Yang	36c87c433b	template: preprocess message and collect system	2024-07-12 12:26:43 -07:00
Jeffrey Morgan	179737feb7	Clean up old files when installing on Windows (#5645 ) * app: always clean up install dir; force close applications * remove wildcard * revert `CloseApplications` * whitespace * update `LOCALAPPDATA` var	2024-07-11 22:53:46 -07:00
Michael Yang	47353f5ee4	Merge pull request #5639 from ollama/mxyng/unaggregated-system	2024-07-11 17:48:50 -07:00
Josh	10e768826c	fix: quant err message (#5616 )	2024-07-11 17:24:29 -07:00
Michael Yang	5056bb9c01	rename aggregate to contents	2024-07-11 17:00:26 -07:00
Jeffrey Morgan	c4cf8ad559	llm: avoid loading model if system memory is too small (#5637 ) * llm: avoid loading model if system memory is too small * update log * Instrument swap free space On linux and windows, expose how much swap space is available so we can take that into consideration when scheduling models * use `systemSwapFreeMemory` in check --------- Co-authored-by: Daniel Hiltgen <daniel@ollama.com>	2024-07-11 16:42:57 -07:00
Michael Yang	57ec6901eb	revert embedded templates to use prompt/response This reverts commit `19753c18c0`. for compat. messages will be added at a later date	2024-07-11 14:49:35 -07:00
Michael Yang	e64f9ebb44	do no automatically aggregate system messages	2024-07-11 14:49:35 -07:00
Jeffrey Morgan	791650ddef	sched: only error when over-allocating system memory (#5626 )	2024-07-11 00:53:12 -07:00
Jeffrey Morgan	efbf41ed81	llm: dont link cuda with compat libs (#5621 )	2024-07-10 20:01:52 -07:00
Michael Yang	cf15589851	Merge pull request #5620 from ollama/mxyng/templates update embedded templates	2024-07-10 17:16:24 -07:00
Michael Yang	19753c18c0	update embedded templates	2024-07-10 17:03:08 -07:00
Michael Yang	41be28096a	add system prompt to first legacy template	2024-07-10 17:03:08 -07:00
Michael Yang	37a570f962	Merge pull request #5612 from ollama/mxyng/mem chatglm graph	2024-07-10 14:18:33 -07:00
Michael Yang	5a739ff4cb	chatglm graph	2024-07-10 13:43:47 -07:00
Jeffrey Morgan	4e262eb2a8	remove `GGML_CUDA_FORCE_MMQ=on` from build (#5588 )	2024-07-10 13:17:13 -07:00
Daniel Hiltgen	4cfcbc328f	Merge pull request #5124 from dhiltgen/amd_windows Wire up windows AMD driver reporting	2024-07-10 12:50:23 -07:00
Daniel Hiltgen	79292ff3e0	Merge pull request #5555 from dhiltgen/msvc_deps Bundle missing CRT libraries	2024-07-10 12:50:02 -07:00
Daniel Hiltgen	8ea500441d	Merge pull request #5580 from dhiltgen/cuda_overhead Detect CUDA OS overhead	2024-07-10 12:47:31 -07:00
Daniel Hiltgen	b50c818623	Merge pull request #5607 from dhiltgen/win_rocm_v6 Bump ROCm on windows to 6.1.2	2024-07-10 12:47:10 -07:00
Daniel Hiltgen	b99e750b62	Merge pull request #5605 from dhiltgen/merge_glitch Remove duplicate merge glitch	2024-07-10 11:47:08 -07:00
Daniel Hiltgen	1f50356e8e	Bump ROCm on windows to 6.1.2 This also adjusts our algorithm to favor our bundled ROCm. I've confirmed VRAM reporting still doesn't work properly so we can't yet enable concurrency by default.	2024-07-10 11:01:22 -07:00
Daniel Hiltgen	22c81f62ec	Remove duplicate merge glitch	2024-07-10 09:01:33 -07:00
Daniel Hiltgen	2d1e3c3229	Merge pull request #5503 from dhiltgen/dual_rocm Workaround broken ROCm p2p copy	2024-07-09 15:44:16 -07:00
royjhan	4918fae535	OpenAI v1/completions: allow stop token list (#5551 ) * stop token parsing fix * add stop test	2024-07-09 14:01:26 -07:00
royjhan	0aff67877e	separate request tests (#5578 )	2024-07-09 13:48:31 -07:00
Daniel Hiltgen	f6f759fc5f	Detect CUDA OS Overhead This adds logic to detect skew between the driver and management library which can be attributed to OS overhead and records that so we can adjust subsequent management library free VRAM updates and avoid OOM scenarios.	2024-07-09 12:21:50 -07:00
Daniel Hiltgen	9544a57ee4	Merge pull request #5579 from dhiltgen/win_static_deps Statically link c++ and thread lib on windows	2024-07-09 12:21:13 -07:00
Daniel Hiltgen	b51e3b63ac	Statically link c++ and thread lib This makes sure we statically link the c++ and thread library on windows to avoid unnecessary runtime dependencies on non-standard DLLs	2024-07-09 11:34:30 -07:00
Michael Yang	6bbbc50f10	Merge pull request #5440 from ollama/mxyng/messages-templates update named templates	2024-07-09 09:36:32 -07:00
Michael Yang	9bbddc37a7	Merge pull request #5126 from ollama/mxyng/messages update message processing	2024-07-09 09:20:44 -07:00
Jeffrey Morgan	e4ff73297d	server: fix model reloads when setting `OLLAMA_NUM_PARALLEL` (#5560 ) * server: fix unneeded model reloads when setting `OLLAMA_NUM_PARALLEL` * remove whitespace change * undo some changes	2024-07-08 22:32:15 -07:00
Daniel Hiltgen	b44320db13	Bundle missing CRT libraries Some users are experienging runner startup errors due to not having these msvc redist libraries on their host	2024-07-08 18:24:21 -07:00
Daniel Hiltgen	0bacb30007	Workaround broken ROCm p2p copy Enable the build flag for llama.cpp to use CPU copy for multi-GPU scenarios.	2024-07-08 09:40:52 -07:00
Jeffrey Morgan	53da2c6965	llm: remove ambiguous comment when putting upper limit on predictions to avoid infinite generation (#5535 )	2024-07-07 14:32:05 -04:00
Jeffrey Morgan	d8def1ff94	llm: allow gemma 2 to context shift (#5534 )	2024-07-07 13:41:51 -04:00
Jeffrey Morgan	571dc61955	Update llama.cpp submodule to `a8db2a9c` (#5530 )	2024-07-07 13:03:09 -04:00
Jeffrey Morgan	0e09c380fc	llm: print caching notices in debug only (#5533 )	2024-07-07 12:38:04 -04:00
Jeffrey Morgan	0ee87615c7	sched: don't error if paging to disk on Windows and macOS (#5523 )	2024-07-06 22:01:52 -04:00
Jeffrey Morgan	f8241bfba3	gpu: report system free memory instead of 0 (#5521 )	2024-07-06 19:35:04 -04:00
Jeffrey Morgan	4607c70641	llm: add `-DBUILD_SHARED_LIBS=off` to common cpu cmake flags (#5520 )	2024-07-06 18:58:16 -04:00
jmorganca	c12f1c5b99	release: move mingw library cleanup to correct job	2024-07-06 16:12:29 -04:00
jmorganca	a08f20d910	release: remove unwanted mingw dll.a files	2024-07-06 15:21:15 -04:00
jmorganca	6cea036027	Revert "llm: only statically link libstdc++" This reverts commit `5796bfc401`.	2024-07-06 15:10:48 -04:00
Michael Yang	fb6cbc02fb	update named templates	2024-07-05 16:29:32 -07:00
Michael Yang	326363b3a7	no funcs	2024-07-05 13:17:25 -07:00
Michael Yang	ac7a842e55	fix model reloading ensure runtime model changes (template, system prompt, messages, options) are captured on model updates without needing to reload the server	2024-07-05 13:17:25 -07:00
Michael Yang	2c3fe1fd97	comments	2024-07-05 13:17:24 -07:00
Michael Yang	269ed6e6a2	update message processing	2024-07-05 13:16:58 -07:00
Daniel Hiltgen	784bf88b0d	Wire up windows AMD driver reporting This seems to be ROCm version, not actually driver version, but it may be useful for toggling logic for VRAM reporting in the future	2024-06-18 16:22:47 -07:00
		`@ -1 +1 @@`
			`Subproject commit d7fd29fff16456ce9c3a23fd2d09a66256b05aff`				`Subproject commit a8db2a9ce64cd4417f6a312ab61858f17f0f8584`
`@ -5,3 +5,4 @@`

	`{{ end }}### Response:`	`{{ end }}### Response:`
	`{{ .Response }}`	`{{ .Response }}`
`@ -2,4 +2,5 @@`

	`{{ end }}{{ if .Prompt }}User: {{ .Prompt }}`	`{{ end }}{{ if .Prompt }}User: {{ .Prompt }}`

	`{{ end }}Assistant: <\|begin_of_text\|>{{ .Response }}`	`{{ end }}Assistant: {{ .Response }}`
`@ -5,3 +5,4 @@`

	`{{ end }}@@ Response`	`{{ end }}@@ Response`
	`{{ .Response }}`	`{{ .Response }}`
		`@ -1 +1 @@`
			`{{ .System }}<\|end_of_turn\|>GPT4 Correct User: {{ .Prompt }}<\|end_of_turn\|>GPT4 Correct Assistant: {{ .Response }}<\|end_of_turn\|>`				`{{ if .System }}GPT4 Correct System: {{ .System }}<\|end_of_turn\|>{{ end }}GPT4 Correct User: {{ .Prompt }}<\|end_of_turn\|>GPT4 Correct Assistant: {{ .Response }}<\|end_of_turn\|>`
		`@ -0,0 +1 @@`
							`<start_system>You are a helpful assistant.<end_message><start_user>Hello, how are you?<end_message><start_assistant>I'm doing great. How can I help you today?<end_message><start_user>I'd like to show off how chat templating works!<end_message><start_assistant>`
		`@ -0,0 +1 @@`
							`<start_user>Hello, how are you?<end_message><start_assistant>`
		`@ -0,0 +1,3 @@`
							`[INST] <<SYS>><</SYS>>`

							`Hello, how are you? [/INST]`
		`@ -0,0 +1,4 @@`
							`<\|start_header_id\|>user<\|end_header_id\|>`

							`Hello, how are you?<\|eot_id\|><\|start_header_id\|>assistant<\|end_header_id\|>`