Stream loading state when swapping models (#371)

Swapping models can take a long time and leave a lot of silence while the model is loading. Rather than silently load the model in the background, this PR allows llama-swap to send status updates in the reasoning_content of a streaming chat response. Fixes: #366
2025-10-29 00:09:39 -07:00
parent f852689104
commit a89b803d4a
8 changed files with 375 additions and 51 deletions
--- a/config.example.yaml
+++ b/config.example.yaml
@@ -35,6 +35,14 @@ metricsMaxInMemory: 1000
 # - it is automatically incremented for every model that uses it
 startPort: 10001

+# sendLoadingState: inject loading status updates into the reasoning (thinking)
+# field
+# - optional, default: false
+# - when true, a stream of loading messages will be sent to the client in the
+#   reasoning field so chat UIs can show that loading is in progress.
+# - see #366 for more details
+sendLoadingState: true
+
 # macros: a dictionary of string substitutions
 # - optional, default: empty dictionary
 # - macros are reusable snippets
@@ -184,6 +192,10 @@ models:
    # - recommended to be omitted and the default used
    concurrencyLimit: 0

+    # sendLoadingState: overrides the global sendLoadingState setting for this model
+    # - optional, default: undefined (use global setting)
+    sendLoadingState: false
+
  # Unlisted model example:
  "qwen-unlisted":
    # unlisted: boolean, true or false