Stream loading state when swapping models (#371)

Swapping models can take a long time and leave a lot of silence while the model is loading. Rather than silently load the model in the background, this PR allows llama-swap to send status updates in the reasoning_content of a streaming chat response. Fixes: #366
2025-10-29 00:09:39 -07:00
parent f852689104
commit a89b803d4a
8 changed files with 375 additions and 51 deletions
--- a/proxy/config/config.go
+++ b/proxy/config/config.go
@@ -129,6 +129,9 @@ type Config struct {

 	// hooks, see: #209
 	Hooks HooksConfig `yaml:"hooks"`
+
+	// send loading state in reasoning
+	SendLoadingState bool `yaml:"sendLoadingState"`
 }

 func (c *Config) RealModelName(search string) (string, bool) {
@@ -350,6 +353,13 @@ func LoadConfigFromReader(r io.Reader) (Config, error) {
 			)
 		}

+		// if sendLoadingState is nil, set it to the global config value
+		// see #366
+		if modelConfig.SendLoadingState == nil {
+			v := config.SendLoadingState // copy it
+			modelConfig.SendLoadingState = &v
+		}
+
 		config.Models[modelId] = modelConfig
 	}

--- a/proxy/config/config_posix_test.go
+++ b/proxy/config/config_posix_test.go
@@ -160,6 +160,8 @@ groups:
 		t.Fatalf("Failed to load config: %v", err)
 	}

+	modelLoadingState := false
+
 	expected := Config{
 		LogLevel:  "info",
 		StartPort: 5800,
@@ -171,36 +173,41 @@ groups:
 				Preload: []string{"model1", "model2"},
 			},
 		},
+		SendLoadingState: false,
 		Models: map[string]ModelConfig{
 			"model1": {
-				Cmd:           "path/to/cmd --arg1 one",
-				Proxy:         "http://localhost:8080",
-				Aliases:       []string{"m1", "model-one"},
-				Env:           []string{"VAR1=value1", "VAR2=value2"},
-				CheckEndpoint: "/health",
-				Name:          "Model 1",
-				Description:   "This is model 1",
+				Cmd:              "path/to/cmd --arg1 one",
+				Proxy:            "http://localhost:8080",
+				Aliases:          []string{"m1", "model-one"},
+				Env:              []string{"VAR1=value1", "VAR2=value2"},
+				CheckEndpoint:    "/health",
+				Name:             "Model 1",
+				Description:      "This is model 1",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model2": {
-				Cmd:           "path/to/server --arg1 one",
-				Proxy:         "http://localhost:8081",
-				Aliases:       []string{"m2"},
-				Env:           []string{},
-				CheckEndpoint: "/",
+				Cmd:              "path/to/server --arg1 one",
+				Proxy:            "http://localhost:8081",
+				Aliases:          []string{"m2"},
+				Env:              []string{},
+				CheckEndpoint:    "/",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model3": {
-				Cmd:           "path/to/cmd --arg1 one",
-				Proxy:         "http://localhost:8081",
-				Aliases:       []string{"mthree"},
-				Env:           []string{},
-				CheckEndpoint: "/",
+				Cmd:              "path/to/cmd --arg1 one",
+				Proxy:            "http://localhost:8081",
+				Aliases:          []string{"mthree"},
+				Env:              []string{},
+				CheckEndpoint:    "/",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model4": {
-				Cmd:           "path/to/cmd --arg1 one",
-				Proxy:         "http://localhost:8082",
-				CheckEndpoint: "/",
-				Aliases:       []string{},
-				Env:           []string{},
+				Cmd:              "path/to/cmd --arg1 one",
+				Proxy:            "http://localhost:8082",
+				CheckEndpoint:    "/",
+				Aliases:          []string{},
+				Env:              []string{},
+				SendLoadingState: &modelLoadingState,
 			},
 		},
 		HealthCheckTimeout: 15,
--- a/proxy/config/config_windows_test.go
+++ b/proxy/config/config_windows_test.go
@@ -152,44 +152,51 @@ groups:
 		t.Fatalf("Failed to load config: %v", err)
 	}

+	modelLoadingState := false
+
 	expected := Config{
 		LogLevel:  "info",
 		StartPort: 5800,
 		Macros: MacroList{
 			{"svr-path", "path/to/server"},
 		},
+		SendLoadingState: false,
 		Models: map[string]ModelConfig{
 			"model1": {
-				Cmd:           "path/to/cmd --arg1 one",
-				CmdStop:       "taskkill /f /t /pid ${PID}",
-				Proxy:         "http://localhost:8080",
-				Aliases:       []string{"m1", "model-one"},
-				Env:           []string{"VAR1=value1", "VAR2=value2"},
-				CheckEndpoint: "/health",
+				Cmd:              "path/to/cmd --arg1 one",
+				CmdStop:          "taskkill /f /t /pid ${PID}",
+				Proxy:            "http://localhost:8080",
+				Aliases:          []string{"m1", "model-one"},
+				Env:              []string{"VAR1=value1", "VAR2=value2"},
+				CheckEndpoint:    "/health",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model2": {
-				Cmd:           "path/to/server --arg1 one",
-				CmdStop:       "taskkill /f /t /pid ${PID}",
-				Proxy:         "http://localhost:8081",
-				Aliases:       []string{"m2"},
-				Env:           []string{},
-				CheckEndpoint: "/",
+				Cmd:              "path/to/server --arg1 one",
+				CmdStop:          "taskkill /f /t /pid ${PID}",
+				Proxy:            "http://localhost:8081",
+				Aliases:          []string{"m2"},
+				Env:              []string{},
+				CheckEndpoint:    "/",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model3": {
-				Cmd:           "path/to/cmd --arg1 one",
-				CmdStop:       "taskkill /f /t /pid ${PID}",
-				Proxy:         "http://localhost:8081",
-				Aliases:       []string{"mthree"},
-				Env:           []string{},
-				CheckEndpoint: "/",
+				Cmd:              "path/to/cmd --arg1 one",
+				CmdStop:          "taskkill /f /t /pid ${PID}",
+				Proxy:            "http://localhost:8081",
+				Aliases:          []string{"mthree"},
+				Env:              []string{},
+				CheckEndpoint:    "/",
+				SendLoadingState: &modelLoadingState,
 			},
 			"model4": {
-				Cmd:           "path/to/cmd --arg1 one",
-				CmdStop:       "taskkill /f /t /pid ${PID}",
-				Proxy:         "http://localhost:8082",
-				CheckEndpoint: "/",
-				Aliases:       []string{},
-				Env:           []string{},
+				Cmd:              "path/to/cmd --arg1 one",
+				CmdStop:          "taskkill /f /t /pid ${PID}",
+				Proxy:            "http://localhost:8082",
+				CheckEndpoint:    "/",
+				Aliases:          []string{},
+				Env:              []string{},
+				SendLoadingState: &modelLoadingState,
 			},
 		},
 		HealthCheckTimeout: 15,
--- a/proxy/config/model_config.go
+++ b/proxy/config/model_config.go
@@ -35,6 +35,9 @@ type ModelConfig struct {
 	// Metadata: see #264
 	// Arbitrary metadata that can be exposed through the API
 	Metadata map[string]any `yaml:"metadata"`
+
+	// override global setting
+	SendLoadingState *bool `yaml:"sendLoadingState"`
 }

 func (m *ModelConfig) UnmarshalYAML(unmarshal func(interface{}) error) error {
--- a/proxy/config/model_config_test.go
+++ b/proxy/config/model_config_test.go
@@ -50,5 +50,25 @@ models:
 			}
 		})
 	}
-
+}
+
+func TestConfig_ModelSendLoadingState(t *testing.T) {
+	content := `
+sendLoadingState: true
+models:
+  model1:
+    cmd: path/to/cmd --port ${PORT}
+    sendLoadingState: false
+  model2:
+    cmd: path/to/cmd --port ${PORT}
+`
+	config, err := LoadConfigFromReader(strings.NewReader(content))
+	assert.NoError(t, err)
+	assert.True(t, config.SendLoadingState)
+	if assert.NotNil(t, config.Models["model1"].SendLoadingState) {
+		assert.False(t, *config.Models["model1"].SendLoadingState)
+	}
+	if assert.NotNil(t, config.Models["model2"].SendLoadingState) {
+		assert.True(t, *config.Models["model2"].SendLoadingState)
+	}
 }