Bump vLLM version + more options when loading models in vLLM (#1782)

* Bump vLLM version to 0.3.2 * Add vLLM model loading options * Remove transformers-exllama * Fix install exllama
2025-05-20 02:24:59 +00:00 · 2024-03-01 16:48:53 -05:00 · 2024-03-01 16:48:53 -05:00 · 939411300a
commit 939411300a
parent 1c312685aa
28 changed files with 736 additions and 641 deletions
--- a/core/config/backend_config.go
+++ b/core/config/backend_config.go
@ -118,16 +118,21 @@ type LLMConfig struct {
 	TrimSpace       []string `yaml:"trimspace"`
 	TrimSuffix      []string `yaml:"trimsuffix"`

-	ContextSize  int     `yaml:"context_size"`
-	NUMA         bool    `yaml:"numa"`
-	LoraAdapter  string  `yaml:"lora_adapter"`
-	LoraBase     string  `yaml:"lora_base"`
-	LoraScale    float32 `yaml:"lora_scale"`
-	NoMulMatQ    bool    `yaml:"no_mulmatq"`
-	DraftModel   string  `yaml:"draft_model"`
-	NDraft       int32   `yaml:"n_draft"`
-	Quantization string  `yaml:"quantization"`
-	MMProj       string  `yaml:"mmproj"`
+	ContextSize          int     `yaml:"context_size"`
+	NUMA                 bool    `yaml:"numa"`
+	LoraAdapter          string  `yaml:"lora_adapter"`
+	LoraBase             string  `yaml:"lora_base"`
+	LoraScale            float32 `yaml:"lora_scale"`
+	NoMulMatQ            bool    `yaml:"no_mulmatq"`
+	DraftModel           string  `yaml:"draft_model"`
+	NDraft               int32   `yaml:"n_draft"`
+	Quantization         string  `yaml:"quantization"`
+	GPUMemoryUtilization float32 `yaml:"gpu_memory_utilization"` // vLLM
+	TrustRemoteCode      bool    `yaml:"trust_remote_code"`      // vLLM
+	EnforceEager         bool    `yaml:"enforce_eager"`          // vLLM
+	SwapSpace            int     `yaml:"swap_space"`             // vLLM
+	MaxModelLen          int     `yaml:"max_model_len"`          // vLLM
+	MMProj               string  `yaml:"mmproj"`

 	RopeScaling string `yaml:"rope_scaling"`
 	ModelType   string `yaml:"type"`