Allow specifying stop conditions in Modelfile

2023-07-28 12:31:08 -04:00 · 2023-07-28 12:31:08 -04:00 · 6ed3ec0cb3
commit 6ed3ec0cb3
parent c75cafdb58 47bda0b860
4 changed files with 31 additions and 28 deletions
--- a/api/types.go
+++ b/api/types.go
@ -178,7 +178,7 @@ type Options struct {
 	MirostatTau      float32  `json:"mirostat_tau,omitempty"`
 	MirostatEta      float32  `json:"mirostat_eta,omitempty"`
 	PenalizeNewline  bool     `json:"penalize_newline,omitempty"`
-	StopConditions   []string `json:"stop_conditions,omitempty"`
+	Stop             []string `json:"stop,omitempty"`

 	NumThread int `json:"num_thread,omitempty"`
 }
--- a/docs/modelfile.md
+++ b/docs/modelfile.md
@ -98,20 +98,21 @@ PARAMETER <parameter> <parametervalue>

 ### Valid Parameters and Values

-| Parameter      | Description                                                                                                                                                                                                                                             | Value Type | Example Usage      |
-| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | ------------------ |
-| num_ctx        | Sets the size of the context window used to generate the next token. (Default: 2048)                                                                                                                                                                    | int        | num_ctx 4096       |
-| temperature    | The temperature of the model. Increasing the temperature will make the model answer more creatively. (Default: 0.8)                                                                                                                                     | float      | temperature 0.7    |
-| top_k          | Reduces the probability of generating nonsense. A higher value (e.g. 100) will give more diverse answers, while a lower value (e.g. 10) will be more conservative. (Default: 40)                                                                        | int        | top_k 40           |
-| top_p          | Works together with top-k. A higher value (e.g., 0.95) will lead to more diverse text, while a lower value (e.g., 0.5) will generate more focused and conservative text. (Default: 0.9)                                                                 | float      | top_p 0.9          |
-| num_gpu        | The number of GPUs to use. On macOS it defaults to 1 to enable metal support, 0 to disable.                                                                                                                                                             | int        | num_gpu 1          |
-| repeat_last_n  | Sets how far back for the model to look back to prevent repetition. (Default: 64, 0 = disabled, -1 = num_ctx)                                                                                                                                           | int        | repeat_last_n 64   |
-| repeat_penalty | Sets how strongly to penalize repetitions. A higher value (e.g., 1.5) will penalize repetitions more strongly, while a lower value (e.g., 0.9) will be more lenient. (Default: 1.1)                                                                     | float      | repeat_penalty 1.1 |
-| tfs_z          | Tail free sampling is used to reduce the impact of less probable tokens from the output. A higher value (e.g., 2.0) will reduce the impact more, while a value of 1.0 disables this setting. (default: 1)                                               | float      | tfs_z 1            |
-| mirostat       | Enable Mirostat sampling for controlling perplexity. (default: 0, 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0)                                                                                                                                         | int        | mirostat 0         |
-| mirostat_tau   | Controls the balance between coherence and diversity of the output. A lower value will result in more focused and coherent text. (Default: 5.0)                                                                                                         | float      | mirostat_tau 5.0   |
-| mirostat_eta   | Influences how quickly the algorithm responds to feedback from the generated text. A lower learning rate will result in slower adjustments, while a higher learning rate will make the algorithm more responsive. (Default: 0.1)                        | float      | mirostat_eta 0.1   |
-| num_thread     | Sets the number of threads to use during computation. By default, Ollama will detect this for optimal performance. It is recommended to set this value to the number of physical CPU cores your system has (as opposed to the logical number of cores). | int        | num_thread 8       |
+| Parameter      | Description                                                                                                                                                                                                                                             | Value Type | Example Usage        |
+| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | -------------------- |
+| mirostat       | Enable Mirostat sampling for controlling perplexity. (default: 0, 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0)                                                                                                                                         | int        | mirostat 0           |
+| mirostat_eta   | Influences how quickly the algorithm responds to feedback from the generated text. A lower learning rate will result in slower adjustments, while a higher learning rate will make the algorithm more responsive. (Default: 0.1)                        | float      | mirostat_eta 0.1     |
+| mirostat_tau   | Controls the balance between coherence and diversity of the output. A lower value will result in more focused and coherent text. (Default: 5.0)                                                                                                         | float      | mirostat_tau 5.0     |
+| num_ctx        | Sets the size of the context window used to generate the next token. (Default: 2048)                                                                                                                                                                    | int        | num_ctx 4096         |
+| num_gpu        | The number of GPUs to use. On macOS it defaults to 1 to enable metal support, 0 to disable.                                                                                                                                                             | int        | num_gpu 1            |
+| num_thread     | Sets the number of threads to use during computation. By default, Ollama will detect this for optimal performance. It is recommended to set this value to the number of physical CPU cores your system has (as opposed to the logical number of cores). | int        | num_thread 8         |
+| repeat_last_n  | Sets how far back for the model to look back to prevent repetition. (Default: 64, 0 = disabled, -1 = num_ctx)                                                                                                                                           | int        | repeat_last_n 64     |
+| repeat_penalty | Sets how strongly to penalize repetitions. A higher value (e.g., 1.5) will penalize repetitions more strongly, while a lower value (e.g., 0.9) will be more lenient. (Default: 1.1)                                                                     | float      | repeat_penalty 1.1   |
+| temperature    | The temperature of the model. Increasing the temperature will make the model answer more creatively. (Default: 0.8)                                                                                                                                     | float      | temperature 0.7      |
+| stop           | Sets the stop tokens to use.                                                                                                                                                                                                                            | string     | stop "AI assistant:" |
+| tfs_z          | Tail free sampling is used to reduce the impact of less probable tokens from the output. A higher value (e.g., 2.0) will reduce the impact more, while a value of 1.0 disables this setting. (default: 1)                                               | float      | tfs_z 1              |
+| top_k          | Reduces the probability of generating nonsense. A higher value (e.g. 100) will give more diverse answers, while a lower value (e.g. 10) will be more conservative. (Default: 40)                                                                        | int        | top_k 40             |
+| top_p          | Works together with top-k. A higher value (e.g., 0.95) will lead to more diverse text, while a lower value (e.g., 0.5) will generate more focused and conservative text. (Default: 0.9)                                                                 | float      | top_p 0.9            |

 ### TEMPLATE

--- a/llama/llama.go
+++ b/llama/llama.go
@ -246,7 +246,7 @@ func (llm *LLM) Predict(ctx []int, prompt string, fn func(api.GenerateResponse))
 }

 func (llm *LLM) checkStopConditions(b bytes.Buffer) error {
-	for _, stopCondition := range llm.StopConditions {
+	for _, stopCondition := range llm.Stop {
 		if stopCondition == b.String() {
 			return io.EOF
 		} else if strings.HasPrefix(stopCondition, b.String()) {
--- a/server/images.go
+++ b/server/images.go
@ -202,7 +202,7 @@ func CreateModel(name string, path string, fn func(resp api.ProgressResponse)) e
 	}

 	var layers []*LayerReader
-	params := make(map[string]string)
+	params := make(map[string][]string)

 	for _, c := range commands {
 		log.Printf("[%s] - %s\n", c.Name, c.Args)
@ -286,8 +286,8 @@ func CreateModel(name string, path string, fn func(resp api.ProgressResponse)) e
 			layer.MediaType = mediaType
 			layers = append(layers, layer)
 		default:
-			// runtime parameters
-			params[c.Name] = c.Args
+			// runtime parameters, build a list of args for each parameter to allow multiple values to be specified (ex: multiple stop tokens)
+			params[c.Name] = append(params[c.Name], c.Args)
 		}
 	}

@ -429,7 +429,7 @@ func GetLayerWithBufferFromLayer(layer *Layer) (*LayerReader, error) {
 	return newLayer, nil
 }

-func paramsToReader(params map[string]string) (io.ReadSeeker, error) {
+func paramsToReader(params map[string][]string) (io.ReadSeeker, error) {
 	opts := api.DefaultOptions()
 	typeOpts := reflect.TypeOf(opts)

@ -444,34 +444,36 @@ func paramsToReader(params map[string]string) (io.ReadSeeker, error) {

 	valueOpts := reflect.ValueOf(&opts).Elem()
 	// iterate params and set values based on json struct tags
-	for key, val := range params {
+	for key, vals := range params {
 		if opt, ok := jsonOpts[key]; ok {
 			field := valueOpts.FieldByName(opt.Name)
 			if field.IsValid() && field.CanSet() {
 				switch field.Kind() {
 				case reflect.Float32:
-					floatVal, err := strconv.ParseFloat(val, 32)
+					floatVal, err := strconv.ParseFloat(vals[0], 32)
 					if err != nil {
-						return nil, fmt.Errorf("invalid float value %s", val)
+						return nil, fmt.Errorf("invalid float value %s", vals)
 					}

 					field.SetFloat(floatVal)
 				case reflect.Int:
-					intVal, err := strconv.ParseInt(val, 10, 0)
+					intVal, err := strconv.ParseInt(vals[0], 10, 0)
 					if err != nil {
-						return nil, fmt.Errorf("invalid int value %s", val)
+						return nil, fmt.Errorf("invalid int value %s", vals)
 					}

 					field.SetInt(intVal)
 				case reflect.Bool:
-					boolVal, err := strconv.ParseBool(val)
+					boolVal, err := strconv.ParseBool(vals[0])
 					if err != nil {
-						return nil, fmt.Errorf("invalid bool value %s", val)
+						return nil, fmt.Errorf("invalid bool value %s", vals)
 					}

 					field.SetBool(boolVal)
 				case reflect.String:
-					field.SetString(val)
+					field.SetString(vals[0])
+				case reflect.Slice:
+					field.Set(reflect.ValueOf(vals))
 				default:
 					return nil, fmt.Errorf("unknown type %s for %s", field.Kind(), key)
 				}