Stream loading state when swapping models (#371)

Swapping models can take a long time and leave a lot of silence while the model is loading. Rather than silently load the model in the background, this PR allows llama-swap to send status updates in the reasoning_content of a streaming chat response. Fixes: #366
2025-10-29 00:09:39 -07:00
parent f852689104
commit a89b803d4a
8 changed files with 375 additions and 51 deletions
--- a/proxy/config/config_posix_test.go
+++ b/proxy/config/config_posix_test.go
@@ -160,6 +160,8 @@ groups:
 		t.Fatalf("Failed to load config: %v", err)
 	}

+	modelLoadingState := false
+
 	expected := Config{
 		LogLevel:  "info",
 		StartPort: 5800,
@@ -171,36 +173,41 @@ groups:
 				Preload: []string{"model1", "model2"},
 			},
 		},
+		SendLoadingState: false,
 		Models: map[string]ModelConfig{
 			"model1": {
-				Cmd:           "path/to/cmd --arg1 one",
-				Proxy:         "http://localhost:8080",
-				Aliases:       []string{"m1", "model-one"},
-				Env:           []string{"VAR1=value1", "VAR2=value2"},
-				CheckEndpoint: "/health",
-				Name:          "Model 1",
-				Description:   "This is model 1",
+				Cmd:              "path/to/cmd --arg1 one",
+				Proxy:            "http://localhost:8080",
+				Aliases:          []string{"m1", "model-one"},
+				Env:              []string{"VAR1=value1", "VAR2=value2"},
+				CheckEndpoint:    "/health",
+				Name:             "Model 1",
+				Description:      "This is model 1",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model2": {
-				Cmd:           "path/to/server --arg1 one",
-				Proxy:         "http://localhost:8081",
-				Aliases:       []string{"m2"},
-				Env:           []string{},
-				CheckEndpoint: "/",
+				Cmd:              "path/to/server --arg1 one",
+				Proxy:            "http://localhost:8081",
+				Aliases:          []string{"m2"},
+				Env:              []string{},
+				CheckEndpoint:    "/",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model3": {
-				Cmd:           "path/to/cmd --arg1 one",
-				Proxy:         "http://localhost:8081",
-				Aliases:       []string{"mthree"},
-				Env:           []string{},
-				CheckEndpoint: "/",
+				Cmd:              "path/to/cmd --arg1 one",
+				Proxy:            "http://localhost:8081",
+				Aliases:          []string{"mthree"},
+				Env:              []string{},
+				CheckEndpoint:    "/",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model4": {
-				Cmd:           "path/to/cmd --arg1 one",
-				Proxy:         "http://localhost:8082",
-				CheckEndpoint: "/",
-				Aliases:       []string{},
-				Env:           []string{},
+				Cmd:              "path/to/cmd --arg1 one",
+				Proxy:            "http://localhost:8082",
+				CheckEndpoint:    "/",
+				Aliases:          []string{},
+				Env:              []string{},
+				SendLoadingState: &modelLoadingState,
 			},
 		},
 		HealthCheckTimeout: 15,