feat: stream tokens usage (#4415)

* Use pb.Reply instead of []byte with Reply.GetMessage() in llama grpc to get the proper usage data in reply streaming mode at the last [DONE] frame * Fix 'hang' on empty message from the start Seems like that empty message marker trick was unnecessary --------- Co-authored-by: Ettore Di Giacinto <mudler@users.noreply.github.com>
2025-05-20 10:35:01 +00:00 · 2024-12-18 12:48:50 +04:00 · 2024-12-18 12:48:50 +04:00 · 2bc4b56a79
commit 2bc4b56a79
parent fc920cc58a
4 changed files with 17 additions and 9 deletions
--- a/core/backend/llm.go
+++ b/core/backend/llm.go
@ -117,8 +117,12 @@ func ModelInference(ctx context.Context, s string, messages []schema.Message, im
 			ss := ""

 			var partialRune []byte
-			err := inferenceModel.PredictStream(ctx, opts, func(chars []byte) {
-				partialRune = append(partialRune, chars...)
+			err := inferenceModel.PredictStream(ctx, opts, func(reply *proto.Reply) {
+				msg := reply.Message
+				partialRune = append(partialRune, msg...)
+
+				tokenUsage.Prompt = int(reply.PromptTokens)
+				tokenUsage.Completion = int(reply.Tokens)

 				for len(partialRune) > 0 {
 					r, size := utf8.DecodeRune(partialRune)
@ -132,6 +136,10 @@ func ModelInference(ctx context.Context, s string, messages []schema.Message, im

 					partialRune = partialRune[size:]
 				}
+
+				if len(msg) == 0 {
+					tokenCallback("", tokenUsage)
+				}
 			})
 			return LLMResponse{
 				Response: ss,