feat: extend model configuration for llama.cpp (#536)

2025-05-28 14:35:00 +00:00 · 2023-06-07 21:46:19 +02:00 · 2023-06-07 21:46:19 +02:00 · 5abbb134d9
commit 5abbb134d9
parent 694dd4ad9e
6 changed files with 69 additions and 150 deletions
--- a/api/config.go
+++ b/api/config.go
@ -16,27 +16,33 @@ import (
 )

 type Config struct {
-	OpenAIRequest         `yaml:"parameters"`
-	Name                  string            `yaml:"name"`
-	StopWords             []string          `yaml:"stopwords"`
-	Cutstrings            []string          `yaml:"cutstrings"`
-	TrimSpace             []string          `yaml:"trimspace"`
-	ContextSize           int               `yaml:"context_size"`
-	F16                   bool              `yaml:"f16"`
-	Threads               int               `yaml:"threads"`
-	Debug                 bool              `yaml:"debug"`
-	Roles                 map[string]string `yaml:"roles"`
-	Embeddings            bool              `yaml:"embeddings"`
-	Backend               string            `yaml:"backend"`
-	TemplateConfig        TemplateConfig    `yaml:"template"`
-	MirostatETA           float64           `yaml:"mirostat_eta"`
-	MirostatTAU           float64           `yaml:"mirostat_tau"`
-	Mirostat              int               `yaml:"mirostat"`
-	NGPULayers            int               `yaml:"gpu_layers"`
-	ImageGenerationAssets string            `yaml:"asset_dir"`
+	OpenAIRequest  `yaml:"parameters"`
+	Name           string            `yaml:"name"`
+	StopWords      []string          `yaml:"stopwords"`
+	Cutstrings     []string          `yaml:"cutstrings"`
+	TrimSpace      []string          `yaml:"trimspace"`
+	ContextSize    int               `yaml:"context_size"`
+	F16            bool              `yaml:"f16"`
+	Threads        int               `yaml:"threads"`
+	Debug          bool              `yaml:"debug"`
+	Roles          map[string]string `yaml:"roles"`
+	Embeddings     bool              `yaml:"embeddings"`
+	Backend        string            `yaml:"backend"`
+	TemplateConfig TemplateConfig    `yaml:"template"`
+	MirostatETA    float64           `yaml:"mirostat_eta"`
+	MirostatTAU    float64           `yaml:"mirostat_tau"`
+	Mirostat       int               `yaml:"mirostat"`
+	NGPULayers     int               `yaml:"gpu_layers"`
+	MMap           bool              `yaml:"mmap"`
+	MMlock         bool              `yaml:"mmlock"`
+
+	TensorSplit           string `yaml:"tensor_split"`
+	MainGPU               string `yaml:"main_gpu"`
+	ImageGenerationAssets string `yaml:"asset_dir"`

 	PromptCachePath string `yaml:"prompt_cache_path"`
 	PromptCacheAll  bool   `yaml:"prompt_cache_all"`
+	PromptCacheRO   bool   `yaml:"prompt_cache_ro"`

 	PromptStrings, InputStrings []string
 	InputToken                  [][]int
@ -53,6 +59,12 @@ type ConfigMerger struct {
 	sync.Mutex
 }

+func defaultConfig(modelFile string) *Config {
+	return &Config{
+		OpenAIRequest: defaultRequest(modelFile),
+	}
+}
+
 func NewConfigMerger() *ConfigMerger {
 	return &ConfigMerger{
 		configs: make(map[string]Config),
@ -308,13 +320,11 @@ func readConfig(modelFile string, input *OpenAIRequest, cm *ConfigMerger, loader
 	var config *Config
 	cfg, exists := cm.GetConfig(modelFile)
 	if !exists {
-		config = &Config{
-			OpenAIRequest: defaultRequest(modelFile),
-			ContextSize:   ctx,
-			Threads:       threads,
-			F16:           f16,
-			Debug:         debug,
-		}
+		config = defaultConfig(modelFile)
+		config.ContextSize = ctx
+		config.Threads = threads
+		config.F16 = f16
+		config.Debug = debug
 	} else {
 		config = &cfg
 	}