Make checkHealthTimeout Interruptable during startup (#102 )

interrupt and exit Process.start() early if the upstream process exits prematurely or unexpectedly.
Moderate security update for golang/x/net -> v0.38.0
2025-04-24 14:39:33 -07:00 · 2025-04-24 09:58:40 -07:00 · 2025-04-24 09:56:20 -07:00 · 2025-04-23 13:02:12 -07:00 · 2025-04-15 20:23:46 -07:00 · 2025-04-14 14:34:59 -07:00
14 changed files with 557 additions and 104 deletions
@@ -1,4 +1,4 @@
-![llama-swap header image](header.jpeg)
+![llama-swap header image](header2.png)
 ![GitHub Downloads (all assets, all releases)](https://img.shields.io/github/downloads/mostlygeek/llama-swap/total)
 ![GitHub Actions Workflow Status](https://img.shields.io/github/actions/workflow/status/mostlygeek/llama-swap/go-ci.yml)
 ![GitHub Repo stars](https://img.shields.io/github/stars/mostlygeek/llama-swap)
@@ -269,8 +269,4 @@ WantedBy=multi-user.target

 ## Star History

-<picture>
-   <source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date&theme=dark" />
-   <source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date" />
-   <img alt="Star History Chart" src="https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date" />
-</picture>
+[![Star History Chart](https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date)](https://www.star-history.com/#mostlygeek/llama-swap&Date)
@@ -3,7 +3,7 @@
 healthCheckTimeout: 90

 # valid log levels: debug, info (default), warn, error
-logLevel: info
+logLevel: debug

 models:
  "llama":
@@ -0,0 +1,153 @@
+# aider, QwQ, Qwen-Coder 2.5 and llama-swap
+
+This guide show how to use aider and llama-swap to get a 100% local coding co-pilot setup. The focus is on the trickest part which is configuring aider, llama-swap and llama-server to work together.
+
+## Here's what you you need:
+
+- aider - [installation docs](https://aider.chat/docs/install.html)
+- llama-server - [download latest release](https://github.com/ggml-org/llama.cpp/releases)
+- llama-swap - [download latest release](https://github.com/mostlygeek/llama-swap/releases)
+- [QwQ 32B](https://huggingface.co/bartowski/Qwen_QwQ-32B-GGUF) and [Qwen Coder 2.5 32B](https://huggingface.co/bartowski/Qwen2.5-Coder-32B-Instruct-GGUF) models
+- 24GB VRAM video card
+
+## Running aider
+
+The goal is getting this command line to work:
+
+```sh
+aider --architect \
+    --no-show-model-warnings \
+    --model openai/QwQ \
+    --editor-model openai/qwen-coder-32B \
+    --model-settings-file aider.model.settings.yml \
+    --openai-api-key "sk-na" \
+    --openai-api-base "http://10.0.1.24:8080/v1" \
+```
+
+Set `--openai-api-base` to the IP and port where your llama-swap is running.
+
+## Create an aider model settings file
+
+```yaml
+# aider.model.settings.yml
+
+#
+# !!! important: model names must match llama-swap configuration names !!!
+#
+
+- name: "openai/QwQ"
+  edit_format: diff
+  extra_params:
+    max_tokens: 16384
+    top_p: 0.95
+    top_k: 40
+    presence_penalty: 0.1
+    repetition_penalty: 1
+    num_ctx: 16384
+  use_temperature: 0.6
+  reasoning_tag: think
+  weak_model_name: "openai/qwen-coder-32B"
+  editor_model_name: "openai/qwen-coder-32B"
+
+- name: "openai/qwen-coder-32B"
+  edit_format: diff
+  extra_params:
+    max_tokens: 16384
+    top_p: 0.8
+    top_k: 20
+    repetition_penalty: 1.05
+  use_temperature: 0.6
+  reasoning_tag: think
+  editor_edit_format: editor-diff
+  editor_model_name: "openai/qwen-coder-32B"
+```
+
+## llama-swap configuration
+
+```yaml
+# config.yaml
+
+# The parameters are tweaked to fit model+context into 24GB VRAM GPUs
+models:
+  "qwen-coder-32B":
+    proxy: "http://127.0.0.1:8999"
+    cmd: >
+      /path/to/llama-server
+      --host 127.0.0.1 --port 8999 --flash-attn --slots
+      --ctx-size 16000
+      --cache-type-k q8_0 --cache-type-v q8_0
+       -ngl 99
+      --model /path/to/Qwen2.5-Coder-32B-Instruct-Q4_K_M.gguf
+
+  "QwQ":
+    proxy: "http://127.0.0.1:9503"
+    cmd: >
+      /path/to/llama-server
+      --host 127.0.0.1 --port 9503 --flash-attn --metrics--slots
+      --cache-type-k q8_0 --cache-type-v q8_0
+      --ctx-size 32000
+      --samplers "top_k;top_p;min_p;temperature;dry;typ_p;xtc"
+      --temp 0.6 --repeat-penalty 1.1 --dry-multiplier 0.5
+      --min-p 0.01 --top-k 40 --top-p 0.95
+      -ngl 99
+      --model /mnt/nvme/models/bartowski/Qwen_QwQ-32B-Q4_K_M.gguf
+```
+
+## Advanced, Dual GPU Configuration
+
+If you have _dual 24GB GPUs_ you can use llama-swap profiles to avoid swapping between QwQ and Qwen Coder.
+
+In llama-swap's configuration file:
+
+1. add a `profiles` section with `aider` as the profile name
+2. using the `env` field to specify the GPU IDs for each model
+
+```yaml
+# config.yaml
+
+# Add a profile for aider
+profiles:
+  aider:
+    - qwen-coder-32B
+    - QwQ
+
+models:
+  "qwen-coder-32B":
+    # manually set the GPU to run on
+    env:
+      - "CUDA_VISIBLE_DEVICES=0"
+    proxy: "http://127.0.0.1:8999"
+    cmd: /path/to/llama-server ...
+
+  "QwQ":
+    # manually set the GPU to run on
+    env:
+      - "CUDA_VISIBLE_DEVICES=1"
+    proxy: "http://127.0.0.1:9503"
+    cmd: /path/to/llama-server ...
+```
+
+Append the profile tag, `aider:`, to the model names in the model settings file
+
+```yaml
+# aider.model.settings.yml
+- name: "openai/aider:QwQ"
+  weak_model_name: "openai/aider:qwen-coder-32B-aider"
+  editor_model_name: "openai/aider:qwen-coder-32B-aider"
+
+- name: "openai/aider:qwen-coder-32B"
+  editor_model_name: "openai/aider:qwen-coder-32B-aider"
+```
+
+Run aider with:
+
+```sh
+$ aider --architect \
+    --no-show-model-warnings \
+    --model openai/aider:QwQ \
+    --editor-model openai/aider:qwen-coder-32B \
+    --config aider.conf.yml \
+    --model-settings-file aider.model.settings.yml
+    --openai-api-key "sk-na" \
+    --openai-api-base "http://10.0.1.24:8080/v1"
+```
@@ -0,0 +1,28 @@
+# this makes use of llama-swap's profile feature to
+# keep the architect and editor models in VRAM on different GPUs
+
+- name: "openai/aider:QwQ"
+  edit_format: diff
+  extra_params:
+    max_tokens: 16384
+    top_p: 0.95
+    top_k: 40
+    presence_penalty: 0.1
+    repetition_penalty: 1
+    num_ctx: 16384
+  use_temperature: 0.6
+  reasoning_tag: think
+  weak_model_name: "openai/aider:qwen-coder-32B"
+  editor_model_name: "openai/aider:qwen-coder-32B"
+
+- name: "openai/aider:qwen-coder-32B"
+  edit_format: diff
+  extra_params:
+    max_tokens: 16384
+    top_p: 0.8
+    top_k: 20
+    repetition_penalty: 1.05
+  use_temperature: 0.6
+  reasoning_tag: think
+  editor_edit_format: editor-diff
+  editor_model_name: "openai/aider:qwen-coder-32B"
@@ -0,0 +1,26 @@
+- name: "openai/QwQ"
+  edit_format: diff
+  extra_params:
+    max_tokens: 16384
+    top_p: 0.95
+    top_k: 40
+    presence_penalty: 0.1
+    repetition_penalty: 1
+    num_ctx: 16384
+  use_temperature: 0.6
+  reasoning_tag: think
+  weak_model_name: "openai/qwen-coder-32B"
+  editor_model_name: "openai/qwen-coder-32B"
+
+- name: "openai/qwen-coder-32B"
+  edit_format: diff
+  extra_params:
+    max_tokens: 16384
+    top_p: 0.8
+    top_k: 20
+    repetition_penalty: 1.05
+  use_temperature: 0.6
+  reasoning_tag: think
+  editor_edit_format: editor-diff
+  editor_model_name: "openai/qwen-coder-32B"
+
@@ -0,0 +1,49 @@
+healthCheckTimeout: 300
+logLevel: debug
+
+profiles:
+    aider:
+      - qwen-coder-32B
+      - QwQ
+
+models:
+  "qwen-coder-32B":
+    env:
+      - "CUDA_VISIBLE_DEVICES=0"
+    aliases:
+      - coder
+    proxy: "http://127.0.0.1:8999"
+
+    # set appropriate paths for your environment
+    cmd: >
+      /path/to/llama-server
+      --host 127.0.0.1 --port 8999 --flash-attn --slots
+      --ctx-size 16000
+      --ctx-size-draft 16000
+      --model /path/to/Qwen2.5-Coder-32B-Instruct-Q4_K_M.gguf
+      --model-draft /path/to/Qwen2.5-Coder-1.5B-Instruct-Q8_0.gguf
+      -ngl 99 -ngld 99
+      --draft-max 16 --draft-min 4 --draft-p-min 0.4
+      --cache-type-k q8_0 --cache-type-v q8_0
+  "QwQ":
+    env:
+      - "CUDA_VISIBLE_DEVICES=1"
+    proxy: "http://127.0.0.1:9503"
+
+    # set appropriate paths for your environment
+    cmd: >
+      /path/to/llama-server
+      --host 127.0.0.1 --port 9503
+      --flash-attn --metrics
+      --slots
+      --model /path/to/Qwen_QwQ-32B-Q4_K_M.gguf
+      --cache-type-k q8_0 --cache-type-v q8_0
+      --ctx-size 32000
+      --samplers "top_k;top_p;min_p;temperature;dry;typ_p;xtc"
+      --temp 0.6
+      --repeat-penalty 1.1
+      --dry-multiplier 0.5
+      --min-p 0.01
+      --top-k 40
+      --top-p 0.95
+      -ngl 99 -ngld 99
@@ -37,7 +37,7 @@ require (
 	github.com/ugorji/go/codec v1.2.12 // indirect
 	golang.org/x/arch v0.8.0 // indirect
 	golang.org/x/crypto v0.36.0 // indirect
-	golang.org/x/net v0.37.0 // indirect
+	golang.org/x/net v0.38.0 // indirect
 	golang.org/x/sys v0.31.0 // indirect
 	golang.org/x/text v0.23.0 // indirect
 	google.golang.org/protobuf v1.34.1 // indirect
@@ -86,6 +86,8 @@ golang.org/x/net v0.33.0 h1:74SYHlV8BIgHIFC/LrYkOGIwL19eTYXQ5wc6TBuO36I=
 golang.org/x/net v0.33.0/go.mod h1:HXLR5J+9DxmrqMwG9qjGCxZ+zKXxBru04zlTvWlWuN4=
 golang.org/x/net v0.37.0 h1:1zLorHbz+LYj7MQlSf1+2tPIIgibq2eL5xkrGk6f+2c=
 golang.org/x/net v0.37.0/go.mod h1:ivrbrMbzFq5J41QOQh0siUuly180yBYtLp+CKbEaFx8=
+golang.org/x/net v0.38.0 h1:vRMAPTMaeGqVhG5QyLJHqNDwecKTomGeqbnfZyKlBI8=
+golang.org/x/net v0.38.0/go.mod h1:ivrbrMbzFq5J41QOQh0siUuly180yBYtLp+CKbEaFx8=
 golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.20.0 h1:Od9JTbYCk261bKm4M/mw7AklTlFYIa0bIp9BgSm1S8Y=
@@ -12,32 +12,65 @@
            flex-direction: column;
            font-family: "Courier New", Courier, monospace;
        }
-        #log-controls {
-            margin: 0.5em;
+        .log-container {
            display: flex;
-            align-items: center;
-            justify-content: space-between; /* Spaces out elements evenly */
-        }
-        #log-controls input {
-            flex: 1;
-        }
-        #log-controls input:focus {
-           outline: none; /* Ensures no outline is shown when the input is focused */
-        }
-        #log-stream {
            flex: 1;
+            gap: 0.5em;
            margin: 0.5em;
+            min-height: 0;
+        }
+        .log-column {
+            display: flex;
+            flex-direction: column;
+            flex: 1;
+            min-width: 0;
+            transition: flex 0.3s ease;
+        }
+        .log-column.minimized {
+            flex: 0.1;
+            max-width: 50px;
+            border: 1px solid #777;
+            color: green;
+        }
+        .log-controls {
+            display: grid;
+            grid-template-columns: 1fr auto;
+            gap: 0.5em;
+            margin-bottom: 0.5em;
+        }
+        .log-controls input {
+            width: 100%;
+            padding: 4px;
+        }
+        .log-controls input:focus {
+           outline: none;
+        }
+        .log-stream {
+            flex: 1;
            padding: 1em;
            background: #f4f4f4;
            overflow-y: auto;
-            white-space: pre-wrap; /* Ensures line wrapping */
-            word-wrap: break-word; /* Ensures long words wrap */
+            white-space: pre-wrap;
+            word-wrap: break-word;
+            min-height: 0;
        }

        .regex-error {
            background-color: #ff0000 !important;
        }

+        /* Make headers clickable and show pointer cursor */
+        h2 {
+            cursor: pointer;
+            user-select: none;
+            margin: 0 0 0.5em 0;
+            padding: 0.5em;
+        }
+
+        h2:hover {
+            background-color: rgba(0, 0, 0, 0.05);
+        }
+
        /* Dark mode styles */
        @media (prefers-color-scheme: dark) {
            body {
@@ -45,101 +78,181 @@
                color: #fff;
            }

-            #log-stream {
+            .log-stream {
                background: #444;
                color: #fff;
            }

-            #log-controls input {
+            .log-controls input {
                background: #555;
                color: #fff;
                border: 1px solid #777;
            }

-            #log-controls button {
+            .log-controls button {
                background: #555;
                color: #fff;
                border: 1px solid #777;
            }
+
+            h2:hover {
+                background-color: rgba(255, 255, 255, 0.1);
+            }
+        }
+
+        /* Hide content when minimized */
+        .log-column.minimized .log-controls,
+        .log-column.minimized .log-stream {
+            display: none;
+        }
+
+        .log-column.minimized h2 {
+            writing-mode: vertical-rl;
+            text-orientation: mixed;
+            transform: rotate(180deg);
+            white-space: nowrap;
+            margin: auto;
        }
    </style>
 </head>
 <body>
-    <pre id="log-stream">Waiting for logs...</pre>
-    <div id="log-controls">
-        <input type="text" id="filter-input" placeholder="regex filter">
-        <button id="clear-button">clear</button>
+    <div class="log-container">
+        <div class="log-column">
+            <h2>Proxy Logs</h2>
+            <div class="log-controls">
+                <input type="text" id="proxy-filter-input" placeholder="proxy regex filter">
+                <button id="proxy-clear-button">clear</button>
+            </div>
+            <pre class="log-stream" id="proxy-log-stream">Waiting for proxy logs...</pre>
+        </div>
+        <div class="log-column minimized">
+            <h2>Upstream Logs</h2>
+            <div class="log-controls">
+                <input type="text" id="upstream-filter-input" placeholder="upstream regex filter">
+                <button id="upstream-clear-button">clear</button>
+            </div>
+            <pre class="log-stream" id="upstream-log-stream">Waiting for upstream logs...</pre>
+        </div>
    </div>
    <script>
-        const logStream = document.getElementById('log-stream');
-        const filterInput = document.getElementById('filter-input');
-        var logData = "";
-        let regexFilter = null;
+        class LogStream {
+            constructor(streamElement, filterInput, clearButton, endpoint) {
+                this.streamElement = streamElement;
+                this.filterInput = filterInput;
+                this.clearButton = clearButton;
+                this.endpoint = endpoint;
+                this.logData = "";
+                this.regexFilter = null;
+                this.eventSource = null;

-        function setupEventSource() {
-            if (typeof(EventSource) !== "undefined") {
-                const eventSource = new EventSource("/logs/streamSSE");
-
-                eventSource.onmessage = function(event) {
-                    logData += event.data;
-                    render()
-                };
-
-                eventSource.onerror = function(err) {
-                    logData = "EventSource failed: " + err.message;
-                };
-            } else {
-                logData = "SSE Not supported by this browser."
+                this.initialize();
            }
-        }

-        // poor-ai's react  ¯\_(ツ)_/¯
-        function render() {
-            if (regexFilter) {
-                const lines = logData.split('\n');
-                const filteredLines = lines.filter(line => {
-                    return regexFilter === null || regexFilter.test(line);
+            initialize() {
+                this.filterInput.addEventListener('input', () => this.updateFilter());
+                this.clearButton.addEventListener('click', () => {
+                    this.filterInput.value = "";
+                    this.regexFilter = null;
+                    this.render();
                });
-
-                if (filteredLines.length > 0) {
-                    logStream.textContent = filteredLines.join('\n') + '\n';
-                } else {
-                    logStream.textContent = "";
-                }
-            } else {
-                logStream.textContent = logData;
+                this.setupEventSource();
            }

-            logStream.scrollTop = logStream.scrollHeight;
-        }
+            setupEventSource() {
+                if (typeof(EventSource) === "undefined") {
+                    this.logData = "SSE Not supported by this browser.";
+                    this.render();
+                    return;
+                }
+
+                const connect = () => {
+                    this.eventSource = new EventSource(this.endpoint);
+
+                    this.eventSource.onmessage = (event) => {
+                        this.logData += event.data;
+                        this.render();
+                    };
+
+                    this.eventSource.onerror = (err) => {
+                        // Close the current connection
+                        this.eventSource.close();
+
+                        this.logData += "\nConnection lost. Retrying in 5 seconds...\n";
+                        this.render();
+
+                        // Attempt to reconnect after 5 seconds
+                        setTimeout(() => {
+                            this.logData += "Attempting to reconnect...\n";
+                            this.render();
+                            connect();
+                        }, 5000);
+                    };
+                };
+
+                // Initial connection
+                connect();
+            }
+
+            render() {
+                let content = this.logData;
+
+                if (this.regexFilter) {
+                    const lines = content.split('\n');
+                    const filteredLines = lines.filter(line => this.regexFilter.test(line));
+                    content = filteredLines.length > 0 ? filteredLines.join('\n') + '\n' : "";
+                }
+
+                this.streamElement.textContent = content;
+                this.streamElement.scrollTop = this.streamElement.scrollHeight;
+            }
+
+            updateFilter() {
+                const pattern = this.filterInput.value.trim();
+                this.filterInput.classList.remove('regex-error');
+
+                if (!pattern) {
+                    this.regexFilter = null;
+                    this.render();
+                    return;
+                }

-        function updateFilter() {
-            const pattern = filterInput.value.trim();
-            filterInput.classList.remove('regex-error');
-            if (pattern) {
                try {
-                    regexFilter = new RegExp(pattern);
+                    this.regexFilter = new RegExp(pattern);
                } catch (e) {
                    console.error("Invalid regex pattern:", e);
-                    regexFilter = null;
-                    filterInput.classList.add('regex-error');
-                    return
+                    this.regexFilter = null;
+                    this.filterInput.classList.add('regex-error');
+                    return;
                }
-            } else {
-                regexFilter = null;
-            }

-            render();
+                this.render();
+            }
        }

-        filterInput.addEventListener('input', updateFilter);
-        document.getElementById('clear-button').addEventListener('click', () => {
-            filterInput.value = "";
-            regexFilter = null;
-            render();
+        // Initialize both log streams
+        document.addEventListener('DOMContentLoaded', () => {
+            new LogStream(
+                document.getElementById('proxy-log-stream'),
+                document.getElementById('proxy-filter-input'),
+                document.getElementById('proxy-clear-button'),
+                "/logs/streamSSE/proxy"
+            );
+
+            new LogStream(
+                document.getElementById('upstream-log-stream'),
+                document.getElementById('upstream-filter-input'),
+                document.getElementById('upstream-clear-button'),
+                "/logs/streamSSE/upstream"
+            );
+
+            // Initialize clickable headers
+            document.querySelectorAll('h2').forEach(header => {
+                header.addEventListener('click', () => {
+                    const column = header.closest('.log-column');
+                    column.classList.toggle('minimized');
+                });
+            });
        });
-        setupEventSource();
-        updateFilter();
    </script>
 </body>
 </html>
@@ -34,6 +34,9 @@ type Process struct {
 	config ModelConfig
 	cmd    *exec.Cmd

+	// for p.cmd.Wait() select { ... }
+	cmdWaitChan chan error
+
 	processLogger *LogMonitor
 	proxyLogger   *LogMonitor

@@ -61,6 +64,7 @@ func NewProcess(ID string, healthCheckTimeout int, config ModelConfig, processLo
 		ID:                      ID,
 		config:                  config,
 		cmd:                     nil,
+		cmdWaitChan:             make(chan error, 1),
 		processLogger:           processLogger,
 		proxyLogger:             proxyLogger,
 		healthCheckTimeout:      healthCheckTimeout,
@@ -89,16 +93,17 @@ func (p *Process) swapState(expectedState, newState ProcessState) (ProcessState,
 	defer p.stateMutex.Unlock()

 	if p.state != expectedState {
+		p.proxyLogger.Warnf("swapState() Unexpected current state %s, expected %s", p.state, expectedState)
 		return p.state, ErrExpectedStateMismatch
 	}

 	if !isValidTransition(p.state, newState) {
-		p.proxyLogger.Warnf("Invalid state transition from %s to %s", p.state, newState)
+		p.proxyLogger.Warnf("swapState() Invalid state transition from %s to %s", p.state, newState)
 		return p.state, ErrInvalidStateTransition
 	}

-	p.proxyLogger.Debugf("State transition from %s to %s", expectedState, newState)
 	p.state = newState
+	p.proxyLogger.Debugf("swapState() State transitioned from %s to %s", expectedState, newState)
 	return p.state, nil
 }

@@ -179,6 +184,13 @@ func (p *Process) start() error {
 		return fmt.Errorf("start() failed: %v", err)
 	}

+	// Capture the exit error for later signaling
+	go func() {
+		exitErr := p.cmd.Wait()
+		p.proxyLogger.Debugf("cmd.Wait() returned for [%s] error: %v", p.ID, exitErr)
+		p.cmdWaitChan <- exitErr
+	}()
+
 	// One of three things can happen at this stage:
 	// 1. The command exits unexpectedly
 	// 2. The health check fails
@@ -222,6 +234,22 @@ func (p *Process) start() error {
 				}
 			case <-p.shutdownCtx.Done():
 				return errors.New("health check interrupted due to shutdown")
+			case exitErr := <-p.cmdWaitChan:
+				if exitErr != nil {
+					p.proxyLogger.Warnf("upstream command exited prematurely with error: %v", exitErr)
+					if curState, err := p.swapState(StateStarting, StateFailed); err != nil {
+						return fmt.Errorf("upstream command exited unexpectedly: %s AND state swap failed: %v, current state: %v", exitErr.Error(), err, curState)
+					} else {
+						return fmt.Errorf("upstream command exited unexpectedly: %s", exitErr.Error())
+					}
+				} else {
+					p.proxyLogger.Warnf("upstream command exited prematurely with no error")
+					if curState, err := p.swapState(StateStarting, StateFailed); err != nil {
+						return fmt.Errorf("upstream command exited prematurely with no error AND state swap failed: %v, current state: %v", err, curState)
+					} else {
+						return fmt.Errorf("upstream command exited prematurely with no error")
+					}
+				}
 			default:
 				if err := p.checkHealthEndpoint(healthURL); err == nil {
 					p.proxyLogger.Infof("Health check passed on %s", healthURL)
@@ -231,7 +259,7 @@ func (p *Process) start() error {
 					if strings.Contains(err.Error(), "connection refused") {
 						endTime, _ := checkDeadline.Deadline()
 						ttl := time.Until(endTime)
-						p.proxyLogger.Infof("Connection refused on %s, retrying in %.0fs", healthURL, ttl.Seconds())
+						p.proxyLogger.Infof("Connection refused on %s, giving up in %.0fs", healthURL, ttl.Seconds())
 					} else {
 						p.proxyLogger.Infof("Health check error on %s, %v", healthURL, err)
 					}
@@ -257,7 +285,6 @@ func (p *Process) start() error {
 				p.inFlightRequests.Wait()

 				if time.Since(p.lastRequestHandled) > maxDuration {
-
 					p.proxyLogger.Infof("Unloading model %s, TTL of %ds reached.", p.ID, p.config.UnloadAfter)
 					p.Stop()
 					return
@@ -276,6 +303,7 @@ func (p *Process) start() error {
 func (p *Process) Stop() {
 	// wait for any inflight requests before proceeding
 	p.inFlightRequests.Wait()
+	p.proxyLogger.Debugf("Stopping process [%s]", p.ID)

 	// calling Stop() when state is invalid is a no-op
 	if curState, err := p.swapState(StateReady, StateStopping); err != nil {
@@ -303,14 +331,14 @@ func (p *Process) Shutdown() {
 // stopCommand will send a SIGTERM to the process and wait for it to exit.
 // If it does not exit within 5 seconds, it will send a SIGKILL.
 func (p *Process) stopCommand(sigtermTTL time.Duration) {
+	stopStartTime := time.Now()
+	defer func() {
+		p.proxyLogger.Debugf("Process [%s] stopCommand took %v", p.ID, time.Since(stopStartTime))
+	}()
+
 	sigtermTimeout, cancelTimeout := context.WithTimeout(context.Background(), sigtermTTL)
 	defer cancelTimeout()

-	sigtermNormal := make(chan error, 1)
-	go func() {
-		sigtermNormal <- p.cmd.Wait()
-	}()
-
 	if p.cmd == nil || p.cmd.Process == nil {
 		p.proxyLogger.Warnf("Process [%s] cmd or cmd.Process is nil", p.ID)
 		return
@@ -324,7 +352,11 @@ func (p *Process) stopCommand(sigtermTTL time.Duration) {
 	case <-sigtermTimeout.Done():
 		p.proxyLogger.Infof("Process [%s] timed out waiting to stop, sending KILL signal", p.ID)
 		p.cmd.Process.Kill()
-	case err := <-sigtermNormal:
+	case err := <-p.cmdWaitChan:
+		// Note: in start(), p.cmdWaitChan also has a select { ... }. That should be OK
+		// because if we make it here then the cmd has been successfully running and made it
+		// through the health check. There is a possibility that ithe cmd crashed after the health check
+		// succeeded but that's not a case llama-swap is handling for now.
 		if err != nil {
 			if errno, ok := err.(syscall.Errno); ok {
 				p.proxyLogger.Errorf("Process [%s] errno >> %v", p.ID, errno)
@@ -369,6 +401,8 @@ func (p *Process) checkHealthEndpoint(healthURL string) error {
 }

 func (p *Process) ProxyRequest(w http.ResponseWriter, r *http.Request) {
+	requestBeginTime := time.Now()
+	var startDuration time.Duration

 	// prevent new requests from being made while stopping or irrecoverable
 	currentState := p.CurrentState()
@@ -385,11 +419,13 @@ func (p *Process) ProxyRequest(w http.ResponseWriter, r *http.Request) {

 	// start the process on demand
 	if p.CurrentState() != StateReady {
+		beginStartTime := time.Now()
 		if err := p.start(); err != nil {
 			errstr := fmt.Sprintf("unable to start process: %s", err)
 			http.Error(w, errstr, http.StatusBadGateway)
 			return
 		}
+		startDuration = time.Since(beginStartTime)
 	}

 	proxyTo := p.config.Proxy
@@ -433,4 +469,8 @@ func (p *Process) ProxyRequest(w http.ResponseWriter, r *http.Request) {
 			return
 		}
 	}
+
+	totalTime := time.Since(requestBeginTime)
+	p.proxyLogger.Debugf("Process [%s] request %s - start: %v, total: %v",
+		p.ID, r.RequestURI, startDuration, totalTime)
 }
@@ -2,9 +2,9 @@ package proxy

 import (
 	"fmt"
-	"io"
 	"net/http"
 	"net/http/httptest"
+	"os"
 	"sync"
 	"testing"
 	"time"
@@ -13,16 +13,25 @@ import (
 )

 var (
-	discardLogger = NewLogMonitorWriter(io.Discard)
+	debugLogger = NewLogMonitorWriter(os.Stdout)
 )

+func init() {
+	// flip to help with debugging tests
+	if false {
+		debugLogger.SetLogLevel(LevelDebug)
+	} else {
+		debugLogger.SetLogLevel(LevelError)
+	}
+}
+
 func TestProcess_AutomaticallyStartsUpstream(t *testing.T) {

 	expectedMessage := "testing91931"
 	config := getTestSimpleResponderConfig(expectedMessage)

 	// Create a process
-	process := NewProcess("test-process", 5, config, discardLogger, discardLogger)
+	process := NewProcess("test-process", 5, config, debugLogger, debugLogger)
 	defer process.Stop()

 	req := httptest.NewRequest("GET", "/test", nil)
@@ -58,7 +67,7 @@ func TestProcess_WaitOnMultipleStarts(t *testing.T) {
 	expectedMessage := "testing91931"
 	config := getTestSimpleResponderConfig(expectedMessage)

-	process := NewProcess("test-process", 5, config, discardLogger, discardLogger)
+	process := NewProcess("test-process", 5, config, debugLogger, debugLogger)
 	defer process.Stop()

 	var wg sync.WaitGroup
@@ -86,7 +95,7 @@ func TestProcess_BrokenModelConfig(t *testing.T) {
 		CheckEndpoint: "/health",
 	}

-	process := NewProcess("broken", 1, config, discardLogger, discardLogger)
+	process := NewProcess("broken", 1, config, debugLogger, debugLogger)

 	req := httptest.NewRequest("GET", "/", nil)
 	w := httptest.NewRecorder()
@@ -111,7 +120,7 @@ func TestProcess_UnloadAfterTTL(t *testing.T) {
 	config.UnloadAfter = 3 // seconds
 	assert.Equal(t, 3, config.UnloadAfter)

-	process := NewProcess("ttl_test", 2, config, discardLogger, discardLogger)
+	process := NewProcess("ttl_test", 2, config, debugLogger, debugLogger)
 	defer process.Stop()

 	// this should take 4 seconds
@@ -153,7 +162,7 @@ func TestProcess_LowTTLValue(t *testing.T) {
 	config.UnloadAfter = 1 // second
 	assert.Equal(t, 1, config.UnloadAfter)

-	process := NewProcess("ttl", 2, config, discardLogger, discardLogger)
+	process := NewProcess("ttl", 2, config, debugLogger, debugLogger)
 	defer process.Stop()

 	for i := 0; i < 100; i++ {
@@ -180,7 +189,7 @@ func TestProcess_HTTPRequestsHaveTimeToFinish(t *testing.T) {

 	expectedMessage := "12345"
 	config := getTestSimpleResponderConfig(expectedMessage)
-	process := NewProcess("t", 10, config, discardLogger, discardLogger)
+	process := NewProcess("t", 10, config, debugLogger, debugLogger)
 	defer process.Stop()

 	results := map[string]string{
@@ -257,7 +266,7 @@ func TestProcess_SwapState(t *testing.T) {

 	for _, test := range tests {
 		t.Run(test.name, func(t *testing.T) {
-			p := NewProcess("test", 10, getTestSimpleResponderConfig("test"), discardLogger, discardLogger)
+			p := NewProcess("test", 10, getTestSimpleResponderConfig("test"), debugLogger, debugLogger)
 			p.state = test.currentState

 			resultState, err := p.swapState(test.expectedState, test.newState)
@@ -290,7 +299,7 @@ func TestProcess_ShutdownInterruptsHealthCheck(t *testing.T) {
 	config.Proxy = "http://localhost:9998/test"

 	healthCheckTTLSeconds := 30
-	process := NewProcess("test-process", healthCheckTTLSeconds, config, discardLogger, discardLogger)
+	process := NewProcess("test-process", healthCheckTTLSeconds, config, debugLogger, debugLogger)

 	// make it a lot faster
 	process.healthCheckLoopInterval = time.Second
@@ -311,3 +320,23 @@ func TestProcess_ShutdownInterruptsHealthCheck(t *testing.T) {
 	assert.ErrorContains(t, err, "health check interrupted due to shutdown")
 	assert.Equal(t, StateShutdown, process.CurrentState())
 }
+
+func TestProcess_ExitInterruptsHealthCheck(t *testing.T) {
+	if testing.Short() {
+		t.Skip("skipping Exit Interrupts Health Check test")
+	}
+
+	// should run and exit but interrupt the long checkHealthTimeout
+	checkHealthTimeout := 5
+	config := ModelConfig{
+		Cmd:           "sleep 1",
+		Proxy:         "http://127.0.0.1:9913",
+		CheckEndpoint: "/health",
+	}
+
+	process := NewProcess("sleepy", checkHealthTimeout, config, debugLogger, debugLogger)
+	process.healthCheckLoopInterval = time.Second // make it faster
+	err := process.start()
+	assert.Equal(t, "upstream command exited prematurely with no error", err.Error())
+	assert.Equal(t, process.CurrentState(), StateFailed)
+}
@@ -49,14 +49,19 @@ func New(config *Config) *ProxyManager {
 	switch strings.ToLower(strings.TrimSpace(config.LogLevel)) {
 	case "debug":
 		proxyLogger.SetLogLevel(LevelDebug)
+		upstreamLogger.SetLogLevel(LevelDebug)
 	case "info":
 		proxyLogger.SetLogLevel(LevelInfo)
+		upstreamLogger.SetLogLevel(LevelInfo)
 	case "warn":
 		proxyLogger.SetLogLevel(LevelWarn)
+		upstreamLogger.SetLogLevel(LevelWarn)
 	case "error":
 		proxyLogger.SetLogLevel(LevelError)
+		upstreamLogger.SetLogLevel(LevelError)
 	default:
 		proxyLogger.SetLogLevel(LevelInfo)
+		upstreamLogger.SetLogLevel(LevelInfo)
 	}

 	pm := &ProxyManager{
@@ -22,6 +22,7 @@ func TestProxyManager_SwapProcessCorrectly(t *testing.T) {
 			"model1": getTestSimpleResponderConfig("model1"),
 			"model2": getTestSimpleResponderConfig("model2"),
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -62,6 +63,7 @@ func TestProxyManager_SwapMultiProcess(t *testing.T) {
 		Profiles: map[string][]string{
 			"test": {model1, model2},
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -103,6 +105,7 @@ func TestProxyManager_SwapMultiProcessParallelRequests(t *testing.T) {
 			"model2": getTestSimpleResponderConfig("model2"),
 			"model3": getTestSimpleResponderConfig("model3"),
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -153,6 +156,7 @@ func TestProxyManager_ListModelsHandler(t *testing.T) {
 			"model2": getTestSimpleResponderConfig("model2"),
 			"model3": getTestSimpleResponderConfig("model3"),
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -230,6 +234,7 @@ func TestProxyManager_ProfileNonMember(t *testing.T) {
 		Profiles: map[string][]string{
 			"test": {model1},
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -278,6 +283,7 @@ func TestProxyManager_Shutdown(t *testing.T) {
 			"model2": model2Config,
 			"model3": model3Config,
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -313,6 +319,7 @@ func TestProxyManager_Unload(t *testing.T) {
 		Models: map[string]ModelConfig{
 			"model1": getTestSimpleResponderConfig("model1"),
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -339,6 +346,7 @@ func TestProxyManager_StripProfileSlug(t *testing.T) {
 		Models: map[string]ModelConfig{
 			"TheExpectedModel": getTestSimpleResponderConfig("TheExpectedModel"),
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -365,6 +373,7 @@ func TestProxyManager_RunningEndpoint(t *testing.T) {
 		Profiles: map[string][]string{
 			"test": {"model1", "model2"},
 		},
+		LogLevel: "error",
 	}

 	// Define a helper struct to parse the JSON response.
@@ -472,6 +481,7 @@ func TestProxyManager_AudioTranscriptionHandler(t *testing.T) {
 		Models: map[string]ModelConfig{
 			"TheExpectedModel": getTestSimpleResponderConfig("TheExpectedModel"),
 		},
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -580,6 +590,8 @@ func TestProxyManager_UseModelName(t *testing.T) {
 		Models: map[string]ModelConfig{
 			"model1": modelConfig,
 		},
+
+		LogLevel: "error",
 	}

 	proxy := New(config)
@@ -647,7 +659,7 @@ func TestProxyManager_CORSOptionsHandler(t *testing.T) {
 		Models: map[string]ModelConfig{
 			"model1": getTestSimpleResponderConfig("model1"),
 		},
-		LogRequests: true,
+		LogLevel: "error",
 	}

 	tests := []struct {
Author	SHA1	Message	Date
Benson Wong	5fad24c16f	Make checkHealthTimeout Interruptable during startup (#102 ) interrupt and exit Process.start() early if the upstream process exits prematurely or unexpectedly.	2025-04-24 14:39:33 -07:00
Benson Wong	8404244fab	Moderate security update for golang/x/net -> v0.38.0	2025-04-24 09:58:40 -07:00
Benson Wong	712cd01081	fix confusing INFO message [no ci]	2025-04-24 09:56:20 -07:00
Benson Wong	1f7aa359b1	Update header image AI has finally made my dreams of llamas in funny clothing and stuck in a claw machine waiting to be picked come true!	2025-04-23 13:02:12 -07:00
Benson Wong	b138d6cf25	fix starhistory in README	2025-04-15 20:23:46 -07:00
Benson Wong	fb7c808082	add timing for Process start, stop, total request time (#91 )	2025-04-14 14:34:59 -07:00
Benson Wong	a7e640b0f7	add aider example	2025-04-07 12:37:14 -07:00
Benson Wong	593604dfdc	Show proxy and upstream logs in separate columns in logs UI	2025-04-05 10:36:54 -07:00