Improve handling of process that do not handle SIGTERM (#38 )

- Process TTL goroutine did not have a return after .Stop() - Improve logging - Add test TestProcess_LowTTLValue to measure SIGTERM error rate
Fix panic when requesting non-members of profiles
2025-01-20 14:39:52 -08:00 · 2025-01-16 12:06:38 -08:00 · 2025-01-13 22:37:30 -08:00 · 2025-01-12 19:48:35 -08:00
5 changed files with 96 additions and 7 deletions
@@ -5,10 +5,12 @@
 # Introduction
 llama-swap is a light weight, transparent proxy server that provides automatic model swapping to llama.cpp's server.

-Written in golang, it is very easy to install (single binary with no dependancies) and configure (single yaml file). Download a pre-built [release](https://github.com/mostlygeek/llama-swap/releases) or built it yourself from source with `make clean all`.
+Written in golang, it is very easy to install (single binary with no dependancies) and configure (single yaml file). 
+
+Download a pre-built [release](https://github.com/mostlygeek/llama-swap/releases) or build it yourself from source with `make clean all`.

 ## How does it work?
-When a request is made to an OpenAI compatible endpoints, lama-swap will extract the `model` value load the appropriate server configuration to serve it. If a server is already running it will stop it and start a new one. This is where the "swap" part comes in. The upstream server is automatically swapped to the correct one to serve the request.
+When a request is made to an OpenAI compatible endpoint, lama-swap will extract the `model` value and load the appropriate server configuration to serve it. If a server is already running it will stop it and start the correct one. This is where the "swap" part comes in. The upstream server is automatically swapped to the correct one to serve the request.

 In the most basic configuration llama-swap handles one model at a time. For more advanced use cases, the `profiles` feature can load multiple models at the same time. You have complete control over how your system resources are used.

@@ -26,7 +28,7 @@ Any OpenAI compatible server would work. llama-swap was originally designed for
  - `v1/chat/completions`
  - `v1/embeddings`
  - `v1/rerank`
-  - `v1/audio/speech`
+  - `v1/audio/speech` ([#36](https://github.com/mostlygeek/llama-swap/issues/36))
 - ✅ Multiple GPU support
 - ✅ Run multiple models at once with `profiles`
 - ✅ Remote log monitoring at `/log`
@@ -135,6 +135,7 @@ func (p *Process) start() error {
 				if time.Since(p.lastRequestHandled) > maxDuration {
 					fmt.Fprintf(p.logMonitor, "!!! Unloading model %s, TTL of %ds reached.\n", p.ID, p.config.UnloadAfter)
 					p.Stop()
+					return
 				}
 			}
 		}()
@@ -165,7 +166,6 @@ func (p *Process) Stop() {
 	// Pretty sure this stopping code needs some work for windows and
 	// will be a source of pain in the future.

-	p.cmd.Process.Signal(syscall.SIGTERM)
 	sigtermTimeout, cancel := context.WithTimeout(context.Background(), 5*time.Second)
 	defer cancel()

@@ -174,9 +174,11 @@ func (p *Process) Stop() {
 		sigtermNormal <- p.cmd.Wait()
 	}()

+	p.cmd.Process.Signal(syscall.SIGTERM)
+
 	select {
 	case <-sigtermTimeout.Done():
-		fmt.Fprintf(p.logMonitor, "!!! process for %s timed out waiting to stop\n", p.ID)
+		fmt.Fprintf(p.logMonitor, "XXX Process for %s timed out waiting to stop, sending SIGKILL to PID: %d\n", p.ID, p.cmd.Process.Pid)
 		p.cmd.Process.Kill()
 		p.cmd.Wait()
 	case err := <-sigtermNormal:
@@ -67,7 +67,6 @@ func TestProcess_BrokenModelConfig(t *testing.T) {
 	assert.Contains(t, w.Body.String(), "unable to start process")
 }

-// test that the process unloads after the TTL
 func TestProcess_UnloadAfterTTL(t *testing.T) {
 	if testing.Short() {
 		t.Skip("skipping long auto unload TTL test")
@@ -79,7 +78,7 @@ func TestProcess_UnloadAfterTTL(t *testing.T) {
 	config.UnloadAfter = 3 // seconds
 	assert.Equal(t, 3, config.UnloadAfter)

-	process := NewProcess("ttl", 2, config, NewLogMonitorWriter(io.Discard))
+	process := NewProcess("ttl_test", 2, config, NewLogMonitorWriter(io.Discard))
 	defer process.Stop()

 	// this should take 4 seconds
@@ -111,6 +110,33 @@ func TestProcess_UnloadAfterTTL(t *testing.T) {
 	assert.Equal(t, StateStopped, process.CurrentState())
 }

+func TestProcess_LowTTLValue(t *testing.T) {
+	if true { // change this code to run this ...
+		t.Skip("skipping test, edit process_test.go to run it ")
+	}
+
+	config := getTestSimpleResponderConfig("fast_ttl")
+	assert.Equal(t, 0, config.UnloadAfter)
+	config.UnloadAfter = 1 // second
+	assert.Equal(t, 1, config.UnloadAfter)
+
+	process := NewProcess("ttl", 2, config, NewLogMonitorWriter(os.Stdout))
+	defer process.Stop()
+
+	for i := 0; i < 100; i++ {
+		t.Logf("Waiting before sending request %d", i)
+		time.Sleep(1500 * time.Millisecond)
+
+		expected := fmt.Sprintf("echo=test_%d", i)
+		req := httptest.NewRequest("GET", fmt.Sprintf("/slow-respond?echo=%s&delay=50ms", expected), nil)
+		w := httptest.NewRecorder()
+		process.ProxyRequest(w, req)
+		assert.Equal(t, http.StatusOK, w.Code)
+		assert.Contains(t, w.Body.String(), expected)
+	}
+
+}
+
 // issue #19
 func TestProcess_HTTPRequestsHaveTimeToFinish(t *testing.T) {
 	if testing.Short() {
@@ -202,6 +202,21 @@ func (pm *ProxyManager) swapModel(requestedModel string) (*Process, error) {
 		return nil, fmt.Errorf("could not find modelID for %s", requestedModel)
 	}

+	// check if model is part of the profile
+	if profileName != "" {
+		found := false
+		for _, item := range pm.config.Profiles[profileName] {
+			if item == realModelName {
+				found = true
+				break
+			}
+		}
+
+		if !found {
+			return nil, fmt.Errorf("model %s part of profile %s", realModelName, profileName)
+		}
+	}
+
 	// exit early when already running, otherwise stop everything and swap
 	requestedProcessKey := ProcessKeyName(profileName, realModelName)

@@ -210,3 +210,47 @@ func TestProxyManager_ListModelsHandler(t *testing.T) {
 	// Ensure all expected models were returned
 	assert.Empty(t, expectedModels, "not all expected models were returned")
 }
+
+func TestProxyManager_ProfileNonMember(t *testing.T) {
+
+	model1 := "path1/model1"
+	model2 := "path2/model2"
+
+	profileMemberName := ProcessKeyName("test", model1)
+	profileNonMemberName := ProcessKeyName("test", model2)
+
+	config := &Config{
+		HealthCheckTimeout: 15,
+		Models: map[string]ModelConfig{
+			model1: getTestSimpleResponderConfig("model1"),
+			model2: getTestSimpleResponderConfig("model2"),
+		},
+		Profiles: map[string][]string{
+			"test": {model1},
+		},
+	}
+
+	proxy := New(config)
+	defer proxy.StopProcesses()
+
+	// actual member of profile
+	{
+		reqBody := fmt.Sprintf(`{"model":"%s"}`, profileMemberName)
+		req := httptest.NewRequest("POST", "/v1/chat/completions", bytes.NewBufferString(reqBody))
+		w := httptest.NewRecorder()
+
+		proxy.HandlerFunc(w, req)
+		assert.Equal(t, http.StatusOK, w.Code)
+		assert.Contains(t, w.Body.String(), "model1")
+	}
+
+	// actual model, but non-member will 404
+	{
+		reqBody := fmt.Sprintf(`{"model":"%s"}`, profileNonMemberName)
+		req := httptest.NewRequest("POST", "/v1/chat/completions", bytes.NewBufferString(reqBody))
+		w := httptest.NewRecorder()
+
+		proxy.HandlerFunc(w, req)
+		assert.Equal(t, http.StatusNotFound, w.Code)
+	}
+}
Author	SHA1	Message	Date
Benson Wong	2833517eef	Improve handling of process that do not handle SIGTERM (#38 ) - Process TTL goroutine did not have a return after .Stop() - Improve logging - Add test TestProcess_LowTTLValue to measure SIGTERM error rate	2025-01-20 14:39:52 -08:00
Benson Wong	abdc2bfdb3	Fix panic when requesting non-members of profiles A panic occurs when a request for an invalid profile:model pair is made. The edge case is that the profile exists and the model exists but they're not configured as a pair. This adds an additional check to make sure the profile:model pair is valid before attempting to swap the model.	2025-01-16 12:06:38 -08:00
Benson Wong	c3b834737f	Update README.md	2025-01-13 22:37:30 -08:00
Benson Wong	3c8e727b73	Update README.md	2025-01-12 19:48:35 -08:00