Stream loading state when swapping models (#371)

Swapping models can take a long time and leave a lot of silence while the model is loading. Rather than silently load the model in the background, this PR allows llama-swap to send status updates in the reasoning_content of a streaming chat response. Fixes: #366
2025-10-29 00:09:39 -07:00
parent f852689104
commit a89b803d4a
8 changed files with 375 additions and 51 deletions
--- a/proxy/config/model_config_test.go
+++ b/proxy/config/model_config_test.go
@@ -50,5 +50,25 @@ models:
 			}
 		})
 	}
-
+}
+
+func TestConfig_ModelSendLoadingState(t *testing.T) {
+	content := `
+sendLoadingState: true
+models:
+  model1:
+    cmd: path/to/cmd --port ${PORT}
+    sendLoadingState: false
+  model2:
+    cmd: path/to/cmd --port ${PORT}
+`
+	config, err := LoadConfigFromReader(strings.NewReader(content))
+	assert.NoError(t, err)
+	assert.True(t, config.SendLoadingState)
+	if assert.NotNil(t, config.Models["model1"].SendLoadingState) {
+		assert.False(t, *config.Models["model1"].SendLoadingState)
+	}
+	if assert.NotNil(t, config.Models["model2"].SendLoadingState) {
+		assert.True(t, *config.Models["model2"].SendLoadingState)
+	}
 }