feat: more embedded models, coqui fixes, add model usage and description (#1556)

* feat: add model descriptions and usage * remove default model gallery * models: add embeddings and tts * docs: update table * docs: updates * images: cleanup pip cache after install * images: always run apt-get clean * ux: improve gRPC connection errors * ux: improve some messages * fix: fix coqui when no AudioPath is passed by * embedded: add more models * Add usage * Reorder table
2025-05-27 22:15:00 +00:00 · 2024-01-08 00:37:02 +01:00 · 2024-01-08 00:37:02 +01:00 · e19d7226f8
commit e19d7226f8
parent 0843fe6c65
21 changed files with 216 additions and 45 deletions
--- a/embedded/models/all-minilm-l6-v2.yaml
+++ b/embedded/models/all-minilm-l6-v2.yaml
@ -0,0 +1,13 @@
+name: all-minilm-l6-v2
+backend: sentencetransformers
+embeddings: true
+parameters:
+  model: all-MiniLM-L6-v2
+
+usage: |
+    You can test this model with curl like this:
+
+    curl http://localhost:8080/embeddings -X POST -H "Content-Type: application/json" -d '{
+      "input": "Your text string goes here",
+      "model": "all-minilm-l6-v2"
+    }'
--- a/embedded/models/bark.yaml
+++ b/embedded/models/bark.yaml
@ -0,0 +1,8 @@
+usage: |
+    bark works without any configuration, to test it, you can run the following curl command:
+
+    curl http://localhost:8080/tts -H "Content-Type: application/json" -d '{         
+     "backend": "bark",
+     "input":"Hello, this is a test!"
+    }' | aplay
+# TODO: This is a placeholder until we manage to pre-load HF/Transformers models
--- a/embedded/models/bert-cpp.yaml
+++ b/embedded/models/bert-cpp.yaml
@ -0,0 +1,23 @@
+backend: bert-embeddings
+embeddings: true
+f16: true
+
+gpu_layers: 90
+mmap: true
+name: bert-cpp-minilm-v6
+
+parameters:
+  model: bert-MiniLM-L6-v2q4_0.bin
+
+download_files:
+- filename: "bert-MiniLM-L6-v2q4_0.bin"
+  sha256: "a5a174d8772c8a569faf9f3136c441f2c3855b5bf35ed32274294219533feaad"
+  uri: "https://huggingface.co/mudler/all-MiniLM-L6-v2/resolve/main/ggml-model-q4_0.bin"
+
+usage: |
+    You can test this model with curl like this:
+
+    curl http://localhost:8080/embeddings -X POST -H "Content-Type: application/json" -d '{
+      "input": "Your text string goes here",
+      "model": "bert-cpp-minilm-v6"
+    }'
--- a/embedded/models/coqui.yaml
+++ b/embedded/models/coqui.yaml
@ -0,0 +1,9 @@
+usage: |
+    coqui works without any configuration, to test it, you can run the following curl command:
+
+    curl http://localhost:8080/tts -H "Content-Type: application/json" -d '{         
+        "backend": "coqui",
+        "model": "tts_models/en/ljspeech/glow-tts",
+        "input":"Hello, this is a test!"
+        }'
+# TODO: This is a placeholder until we manage to pre-load HF/Transformers models
--- a/embedded/models/llava.yaml
+++ b/embedded/models/llava.yaml
@ -28,4 +28,9 @@ download_files:
 - filename: bakllava.gguf
  uri: huggingface://mys/ggml_bakllava-1/ggml-model-q4_k.gguf
 - filename: bakllava-mmproj.gguf
-  uri: huggingface://mys/ggml_bakllava-1/mmproj-model-f16.gguf
+  uri: huggingface://mys/ggml_bakllava-1/mmproj-model-f16.gguf
+
+usage: |
+    curl http://localhost:8080/v1/chat/completions -H "Content-Type: application/json" -d '{
+        "model": "llava",
+        "messages": [{"role": "user", "content": [{"type":"text", "text": "What is in the image?"}, {"type": "image_url", "image_url": {"url": "https://upload.wikimedia.org/wikipedia/commons/thumb/d/dd/Gfp-wisconsin-madison-the-nature-boardwalk.jpg/2560px-Gfp-wisconsin-madison-the-nature-boardwalk.jpg" }}], "temperature": 0.9}]}'
--- a/embedded/models/mistral-openorca.yaml
+++ b/embedded/models/mistral-openorca.yaml
@ -21,3 +21,9 @@ context_size: 4096
 f16: true
 stopwords:
 - <|im_end|>
+
+usage: |
+      curl http://localhost:8080/v1/chat/completions -H "Content-Type: application/json" -d '{
+          "model": "mistral-openorca",
+          "messages": [{"role": "user", "content": "How are you doing?", "temperature": 0.1}]
+      }'
--- a/embedded/models/rhasspy-voice-en-us-amy.yaml
+++ b/embedded/models/rhasspy-voice-en-us-amy.yaml
@ -0,0 +1,13 @@
+name: voice-en-us-amy-low
+download_files:
+  - filename: voice-en-us-amy-low.tar.gz
+    uri: https://github.com/rhasspy/piper/releases/download/v0.0.2/voice-en-us-amy-low.tar.gz
+
+
+usage: |
+    To test if this model works as expected, you can use the following curl command:
+
+    curl http://localhost:8080/tts -H "Content-Type: application/json" -d '{
+      "model":"en-us-amy-low.onnx",
+      "input": "Hi, this is a test."
+    }'
--- a/embedded/models/vall-e-x.yaml
+++ b/embedded/models/vall-e-x.yaml
@ -0,0 +1,8 @@
+usage: |
+    Vall-e-x works without any configuration, to test it, you can run the following curl command:
+
+    curl http://localhost:8080/tts -H "Content-Type: application/json" -d '{         
+     "backend": "vall-e-x",
+     "input":"Hello, this is a test!"
+    }' | aplay
+# TODO: This is a placeholder until we manage to pre-load HF/Transformers models
--- a/embedded/models/whisper-base.yaml
+++ b/embedded/models/whisper-base.yaml
@ -0,0 +1,18 @@
+name: whisper
+backend: whisper
+parameters:
+  model: ggml-whisper-base.bin
+
+usage: |
+    ## example audio file
+    wget --quiet --show-progress -O gb1.ogg https://upload.wikimedia.org/wikipedia/commons/1/1f/George_W_Bush_Columbia_FINAL.ogg
+
+    ## Send the example audio file to the transcriptions endpoint
+    curl http://localhost:8080/v1/audio/transcriptions \
+         -H "Content-Type: multipart/form-data" \
+         -F file="@$PWD/gb1.ogg" -F model="whisper"
+
+download_files:
+- filename: "ggml-whisper-base.bin"
+  sha256: "60ed5bc3dd14eea856493d334349b405782ddcaf0028d4b5df4088345fba2efe"
+  uri: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin"