ollama/server/routes.go

468 lines
11 KiB
Go
Raw Normal View History

package server
import (
"context"
2023-07-06 10:40:11 -07:00
"encoding/json"
"errors"
"fmt"
"io"
"log"
"net"
"net/http"
"os"
2023-07-14 17:27:14 -07:00
"path/filepath"
2023-07-31 21:35:18 -04:00
"reflect"
2023-07-06 10:40:11 -07:00
"strings"
2023-07-18 11:59:42 -07:00
"sync"
2023-07-12 18:18:06 -07:00
"time"
2023-07-21 18:01:24 -07:00
"github.com/gin-contrib/cors"
"github.com/gin-gonic/gin"
"gonum.org/v1/gonum/mat"
2023-07-03 16:32:48 -04:00
"github.com/jmorganca/ollama/api"
2023-07-21 13:33:56 -07:00
"github.com/jmorganca/ollama/llm"
2023-08-04 18:56:40 -04:00
"github.com/jmorganca/ollama/vector"
)
2023-07-31 21:35:18 -04:00
var loaded struct {
2023-07-19 15:00:28 -07:00
mu sync.Mutex
2023-07-21 13:33:56 -07:00
llm llm.LLM
2023-08-04 18:56:40 -04:00
Embeddings []vector.Embedding
2023-07-19 15:00:28 -07:00
expireAt time.Time
expireTimer *time.Timer
2023-07-31 21:35:18 -04:00
digest string
options api.Options
2023-07-18 11:59:42 -07:00
}
2023-08-15 10:35:39 -03:00
var defaultSessionDuration = 5 * time.Minute
// load a model into memory if it is not already loaded, it is up to the caller to lock loaded.mu before calling this function
func load(model *Model, reqOpts map[string]interface{}, sessionDuration time.Duration) error {
2023-08-03 15:55:35 -04:00
opts := api.DefaultOptions()
if err := opts.FromMap(model.Options); err != nil {
log.Printf("could not load model options: %v", err)
return err
2023-08-03 15:55:35 -04:00
}
if err := opts.FromMap(reqOpts); err != nil {
2023-08-03 15:55:35 -04:00
log.Printf("could not merge model options: %v", err)
return err
2023-08-03 15:55:35 -04:00
}
if model.Digest != loaded.digest || !reflect.DeepEqual(loaded.options, opts) {
2023-07-31 21:35:18 -04:00
if loaded.llm != nil {
loaded.llm.Close()
loaded.llm = nil
loaded.digest = ""
2023-07-18 11:59:42 -07:00
}
2023-07-17 12:08:10 -07:00
2023-08-04 18:56:40 -04:00
if model.Embeddings != nil && len(model.Embeddings) > 0 {
opts.EmbeddingOnly = true // this is requried to generate embeddings, completions will still work
loaded.Embeddings = model.Embeddings
}
llmModel, err := llm.New(model.ModelPath, model.AdapterPaths, opts)
2023-07-18 11:59:42 -07:00
if err != nil {
return err
2023-07-18 11:59:42 -07:00
}
2023-07-21 13:33:56 -07:00
// set cache values before modifying opts
loaded.llm = llmModel
loaded.digest = model.Digest
loaded.options = opts
if opts.NumKeep < 0 {
2023-08-09 10:45:57 -04:00
promptWithSystem, err := model.Prompt(api.GenerateRequest{}, "")
if err != nil {
return err
}
2023-08-09 10:45:57 -04:00
promptNoSystem, err := model.Prompt(api.GenerateRequest{Context: []int{0}}, "")
if err != nil {
return err
}
2023-07-21 13:33:56 -07:00
tokensWithSystem := llmModel.Encode(promptWithSystem)
tokensNoSystem := llmModel.Encode(promptNoSystem)
2023-07-21 13:33:56 -07:00
opts.NumKeep = len(tokensWithSystem) - len(tokensNoSystem) + 1
2023-07-21 13:33:56 -07:00
llmModel.SetOptions(opts)
}
2023-07-19 15:00:28 -07:00
}
2023-07-31 21:35:18 -04:00
loaded.expireAt = time.Now().Add(sessionDuration)
2023-07-31 21:35:18 -04:00
if loaded.expireTimer == nil {
loaded.expireTimer = time.AfterFunc(sessionDuration, func() {
loaded.mu.Lock()
defer loaded.mu.Unlock()
2023-07-19 15:00:28 -07:00
2023-07-31 21:35:18 -04:00
if time.Now().Before(loaded.expireAt) {
2023-07-19 15:00:28 -07:00
return
}
2023-07-31 21:35:18 -04:00
if loaded.llm == nil {
2023-07-19 15:00:28 -07:00
return
}
2023-07-31 21:35:18 -04:00
loaded.llm.Close()
loaded.llm = nil
loaded.digest = ""
2023-07-19 15:00:28 -07:00
})
2023-07-06 10:40:11 -07:00
}
2023-07-31 21:35:18 -04:00
loaded.expireTimer.Reset(sessionDuration)
return nil
}
func GenerateHandler(c *gin.Context) {
loaded.mu.Lock()
defer loaded.mu.Unlock()
checkpointStart := time.Now()
var req api.GenerateRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
model, err := GetModel(req.Model)
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
2023-08-15 10:35:39 -03:00
sessionDuration := defaultSessionDuration // TODO: set this duration from the request if specified
if err := load(model, req.Options, sessionDuration); err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
2023-07-06 10:40:11 -07:00
2023-07-18 12:02:02 -07:00
checkpointLoaded := time.Now()
embedding := ""
if model.Embeddings != nil && len(model.Embeddings) > 0 {
promptEmbed, err := loaded.llm.Embedding(req.Prompt)
if err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
// TODO: set embed_top from specified parameters in modelfile
embed_top := 3
topK := vector.TopK(embed_top, mat.NewVecDense(len(promptEmbed), promptEmbed), loaded.Embeddings)
for _, e := range topK {
embedding = fmt.Sprintf("%s %s", embedding, e.Embedding.Data)
}
}
prompt, err := model.Prompt(req, embedding)
2023-07-11 14:57:17 -07:00
if err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
2023-07-04 00:47:00 -04:00
ch := make(chan any)
go func() {
defer close(ch)
2023-07-20 12:12:08 -07:00
fn := func(r api.GenerateResponse) {
2023-07-31 21:35:18 -04:00
loaded.expireAt = time.Now().Add(sessionDuration)
loaded.expireTimer.Reset(sessionDuration)
2023-07-19 15:00:28 -07:00
r.Model = req.Model
r.CreatedAt = time.Now().UTC()
if r.Done {
2023-07-18 12:02:02 -07:00
r.TotalDuration = time.Since(checkpointStart)
r.LoadDuration = checkpointLoaded.Sub(checkpointStart)
}
ch <- r
2023-07-20 12:12:08 -07:00
}
2023-07-31 21:35:18 -04:00
if err := loaded.llm.Predict(req.Context, prompt, fn); err != nil {
2023-07-20 12:12:08 -07:00
ch <- gin.H{"error": err.Error()}
}
}()
2023-07-11 14:57:17 -07:00
streamResponse(c, ch)
2023-07-11 11:54:22 -07:00
}
2023-07-06 10:40:11 -07:00
func EmbeddingHandler(c *gin.Context) {
loaded.mu.Lock()
defer loaded.mu.Unlock()
var req api.EmbeddingRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
model, err := GetModel(req.Model)
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
if err := load(model, req.Options, 5*time.Minute); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
if !loaded.options.EmbeddingOnly {
c.JSON(http.StatusBadRequest, gin.H{"error": "embedding option must be set to true"})
return
}
embedding, err := loaded.llm.Embedding(req.Prompt)
if err != nil {
log.Printf("embedding generation failed: %v", err)
c.JSON(http.StatusInternalServerError, gin.H{"error": "failed to generate embedding"})
return
}
resp := api.EmbeddingResponse{
Embedding: embedding,
}
c.JSON(http.StatusOK, resp)
}
2023-07-20 16:09:23 -07:00
func PullModelHandler(c *gin.Context) {
2023-07-11 11:54:22 -07:00
var req api.PullRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
ch := make(chan any)
go func() {
defer close(ch)
2023-07-18 18:51:30 -07:00
fn := func(r api.ProgressResponse) {
ch <- r
}
2023-07-18 18:51:30 -07:00
regOpts := &RegistryOptions{
Insecure: req.Insecure,
Username: req.Username,
Password: req.Password,
}
ctx, cancel := context.WithCancel(c.Request.Context())
defer cancel()
if err := PullModel(ctx, req.Name, regOpts, fn); err != nil {
2023-07-20 12:12:08 -07:00
ch <- gin.H{"error": err.Error()}
}
}()
streamResponse(c, ch)
}
2023-07-20 16:09:23 -07:00
func PushModelHandler(c *gin.Context) {
var req api.PushRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
2023-07-11 11:54:22 -07:00
return
}
2023-07-06 10:40:11 -07:00
ch := make(chan any)
go func() {
defer close(ch)
2023-07-18 18:51:30 -07:00
fn := func(r api.ProgressResponse) {
ch <- r
}
2023-07-18 18:51:30 -07:00
regOpts := &RegistryOptions{
Insecure: req.Insecure,
Username: req.Username,
Password: req.Password,
}
ctx := context.Background()
if err := PushModel(ctx, req.Name, regOpts, fn); err != nil {
2023-07-20 12:12:08 -07:00
ch <- gin.H{"error": err.Error()}
}
}()
streamResponse(c, ch)
}
2023-07-20 16:09:23 -07:00
func CreateModelHandler(c *gin.Context) {
var req api.CreateRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"message": err.Error()})
2023-07-12 19:07:15 -07:00
return
}
2023-07-11 11:54:22 -07:00
ch := make(chan any)
go func() {
defer close(ch)
fn := func(resp api.ProgressResponse) {
ch <- resp
}
ctx, cancel := context.WithCancel(c.Request.Context())
defer cancel()
if err := CreateModel(ctx, req.Name, req.Path, fn); err != nil {
2023-07-20 12:12:08 -07:00
ch <- gin.H{"error": err.Error()}
}
}()
2023-07-07 15:29:17 -07:00
streamResponse(c, ch)
2023-07-05 15:37:33 -04:00
}
2023-07-20 16:09:23 -07:00
func DeleteModelHandler(c *gin.Context) {
var req api.DeleteRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
if err := DeleteModel(req.Name); err != nil {
if os.IsNotExist(err) {
c.JSON(http.StatusNotFound, gin.H{"error": fmt.Sprintf("model '%s' not found", req.Name)})
} else {
2023-07-20 16:09:23 -07:00
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
}
return
}
2023-07-20 16:09:23 -07:00
}
func ListModelsHandler(c *gin.Context) {
2023-07-18 09:09:45 -07:00
var models []api.ListResponseModel
fp, err := GetManifestPath()
if err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
err = filepath.Walk(fp, func(path string, info os.FileInfo, err error) error {
if err != nil {
if errors.Is(err, os.ErrNotExist) {
log.Printf("manifest file does not exist: %s", fp)
return nil
}
2023-07-18 09:09:45 -07:00
return err
}
if !info.IsDir() {
fi, err := os.Stat(path)
if err != nil {
log.Printf("skipping file: %s", fp)
return nil
2023-07-18 09:09:45 -07:00
}
path := path[len(fp)+1:]
slashIndex := strings.LastIndex(path, "/")
if slashIndex == -1 {
return nil
}
tag := path[:slashIndex] + ":" + path[slashIndex+1:]
mp := ParseModelPath(tag)
manifest, err := GetManifest(mp)
if err != nil {
log.Printf("skipping file: %s", fp)
return nil
2023-07-18 09:09:45 -07:00
}
model := api.ListResponseModel{
Name: mp.GetShortTagname(),
Size: manifest.GetTotalSize(),
ModifiedAt: fi.ModTime(),
}
models = append(models, model)
}
return nil
})
if err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
2023-07-19 15:00:28 -07:00
c.JSON(http.StatusOK, api.ListResponse{Models: models})
2023-07-18 09:09:45 -07:00
}
2023-07-24 11:27:28 -04:00
func CopyModelHandler(c *gin.Context) {
var req api.CopyRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
if err := CopyModel(req.Source, req.Destination); err != nil {
if os.IsNotExist(err) {
c.JSON(http.StatusNotFound, gin.H{"error": fmt.Sprintf("model '%s' not found", req.Source)})
} else {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
}
return
}
}
2023-08-10 09:27:03 -07:00
func Serve(ln net.Listener, origins []string) error {
2023-07-21 18:01:24 -07:00
config := cors.DefaultConfig()
config.AllowWildcard = true
2023-08-10 09:27:03 -07:00
config.AllowOrigins = append(origins, []string{
2023-07-21 18:01:24 -07:00
"http://localhost",
"http://localhost:*",
"https://localhost",
"https://localhost:*",
"http://127.0.0.1",
"http://127.0.0.1:*",
"https://127.0.0.1",
"https://127.0.0.1:*",
"http://0.0.0.0",
"http://0.0.0.0:*",
"https://0.0.0.0",
"https://0.0.0.0:*",
2023-08-10 09:27:03 -07:00
}...)
2023-07-21 18:01:24 -07:00
2023-07-05 15:37:33 -04:00
r := gin.Default()
2023-07-21 18:01:24 -07:00
r.Use(cors.New(config))
2023-07-05 15:37:33 -04:00
2023-07-07 23:46:15 -04:00
r.GET("/", func(c *gin.Context) {
c.String(http.StatusOK, "Ollama is running")
})
2023-08-01 14:50:38 -04:00
r.HEAD("/", func(c *gin.Context) {
c.Status(http.StatusOK)
})
2023-07-07 23:46:15 -04:00
2023-07-20 16:09:23 -07:00
r.POST("/api/pull", PullModelHandler)
r.POST("/api/generate", GenerateHandler)
r.POST("/api/embeddings", EmbeddingHandler)
2023-07-20 16:09:23 -07:00
r.POST("/api/create", CreateModelHandler)
r.POST("/api/push", PushModelHandler)
2023-07-24 11:27:28 -04:00
r.POST("/api/copy", CopyModelHandler)
2023-07-20 16:09:23 -07:00
r.GET("/api/tags", ListModelsHandler)
r.DELETE("/api/delete", DeleteModelHandler)
log.Printf("Listening on %s", ln.Addr())
s := &http.Server{
Handler: r,
}
return s.Serve(ln)
}
2023-07-06 10:40:11 -07:00
func streamResponse(c *gin.Context, ch chan any) {
c.Header("Content-Type", "application/x-ndjson")
2023-07-11 11:54:22 -07:00
c.Stream(func(w io.Writer) bool {
val, ok := <-ch
if !ok {
return false
}
bts, err := json.Marshal(val)
if err != nil {
2023-07-31 16:46:37 -04:00
log.Printf("streamResponse: json.Marshal failed with %s", err)
2023-07-11 11:54:22 -07:00
return false
}
bts = append(bts, '\n')
if _, err := w.Write(bts); err != nil {
2023-07-31 16:46:37 -04:00
log.Printf("streamResponse: w.Write failed with %s", err)
2023-07-11 11:54:22 -07:00
return false
}
return true
})
}