remove noisy debug print message

add guard to avoid unnecessary logic in Process.Shutdown
Add linux install and uninstall shell scripts (#139 )
2025-05-19 15:36:15 -07:00 · 2025-05-19 15:34:30 -07:00 · 2025-05-19 12:03:33 -07:00 · 2025-05-16 19:54:44 -07:00 · 2025-05-16 13:52:04 -07:00 · 2025-05-16 13:48:42 -07:00
11 changed files with 355 additions and 49 deletions
@@ -46,14 +46,14 @@ llama-swap's configuration is purposefully simple.
 models:
  "qwen2.5":
    proxy: "http://127.0.0.1:9999"
-    cmd: >
+    cmd: |
      /app/llama-server
      -hf bartowski/Qwen2.5-0.5B-Instruct-GGUF:Q4_K_M
      --port 9999
  "smollm2":
    proxy: "http://127.0.0.1:9999"
-    cmd: >
+    cmd: |
      /app/llama-server
      -hf bartowski/SmolLM2-135M-Instruct-GGUF:Q4_K_M
      --port 9999
@@ -82,7 +82,7 @@ startPort: 10001
 models:
  "llama":
    # multiline for readability
-    cmd: >
+    cmd: |
      llama-server --port 8999
      --model path/to/Qwen2.5-1.5B-Instruct-Q4_K_M.gguf
@@ -123,12 +123,18 @@ models:
  # Docker Support (v26.1.4+ required!)
  "docker-llama":
    proxy: "http://127.0.0.1:${PORT}"
-    cmd: >
+    cmd: |
      docker run --name dockertest
      --init --rm -p ${PORT}:8080 -v /mnt/nvme/models:/models
      ghcr.io/ggerganov/llama.cpp:server
      --model '/models/Qwen2.5-Coder-0.5B-Instruct-Q4_K_M.gguf'
    # use a custom command to stop the model when swapping. By default
    # this is SIGTERM on POSIX systems, and taskkill on Windows systems
    # the ${PID} variable can be used in cmdStop, it will be automatically replaced
    # with the PID of the running model
    cmdStop: docker stop dockertest
 # Groups provide advanced controls over model swapping behaviour. Using groups
 # some models can be kept loaded indefinitely, while others are swapped out.
 #
@@ -247,11 +253,11 @@ Pre-built binaries are available for Linux, FreeBSD and Darwin (OSX). These are
 1. Create a configuration file, see [config.example.yaml](config.example.yaml)
 1. Download a [release](https://github.com/mostlygeek/llama-swap/releases) appropriate for your OS and architecture.
 1. Run the binary with `llama-swap --config path/to/config.yaml`.
-  Available flags:
+   Available flags:
-    - `--config`: Path to the configuration file (default: `config.yaml`).
+   - `--config`: Path to the configuration file (default: `config.yaml`).
-    - `--listen`: Address and port to listen on (default: `:8080`).
+   - `--listen`: Address and port to listen on (default: `:8080`).
-    - `--version`: Show version information and exit.
+   - `--version`: Show version information and exit.
-    - `--watch-config`: Automatically reload the configuration file when it changes. This will wait for in-flight requests to complete then stop all running models (default: `false`).
+   - `--watch-config`: Automatically reload the configuration file when it changes. This will wait for in-flight requests to complete then stop all running models (default: `false`).
 ### Building from source
@@ -15,7 +15,7 @@ groups:
 models:
  "llama":
-    cmd: >
+    cmd: |
      models/llama-server-osx
      --port ${PORT}
      -m models/Llama-3.2-1B-Instruct-Q4_0.gguf
@@ -38,7 +38,7 @@ models:
  # Embedding example with Nomic
  # https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF
  "nomic":
-    cmd: >
+    cmd: |
      models/llama-server-osx --port ${PORT}
      -m models/nomic-embed-text-v1.5.Q8_0.gguf
      --ctx-size 8192
@@ -51,7 +51,7 @@ models:
  # Reranking example with bge-reranker
  # https://huggingface.co/gpustack/bge-reranker-v2-m3-GGUF
  "bge-reranker":
-    cmd: >
+    cmd: |
      models/llama-server-osx --port ${PORT}
      -m models/bge-reranker-v2-m3-Q4_K_M.gguf
      --ctx-size 8192
@@ -59,7 +59,7 @@ models:
  # Docker Support (v26.1.4+ required!)
  "dockertest":
-    cmd: >
+    cmd: |
      docker run --name dockertest
      --init --rm -p ${PORT}:8080 -v /mnt/nvme/models:/models
      ghcr.io/ggerganov/llama.cpp:server
@@ -17,6 +17,7 @@ const DEFAULT_GROUP_ID = "(default)"
 type ModelConfig struct {
 	Cmd           string   `yaml:"cmd"`
 	CmdStop       string   `yaml:"cmdStop"`
 	Proxy         string   `yaml:"proxy"`
 	Aliases       []string `yaml:"aliases"`
 	Env           []string `yaml:"env"`
@@ -135,7 +136,6 @@ func LoadConfigFromReader(r io.Reader) (Config, error) {
 		}
 	}
 	// iterate over the models and replace any ${PORT} with the next available port
 	// Get and sort all model IDs first, makes testing more consistent
 	modelIds := make([]string, 0, len(config.Models))
 	for modelId := range config.Models {
@@ -143,10 +143,10 @@ func LoadConfigFromReader(r io.Reader) (Config, error) {
 	}
 	sort.Strings(modelIds) // This guarantees stable iteration order
 	// iterate over the sorted models
 	nextPort := config.StartPort
 	for _, modelId := range modelIds {
 		modelConfig := config.Models[modelId]
 		// iterate over the models and replace any ${PORT} with the next available port
 		if strings.Contains(modelConfig.Cmd, "${PORT}") {
 			modelConfig.Cmd = strings.ReplaceAll(modelConfig.Cmd, "${PORT}", strconv.Itoa(nextPort))
 			if modelConfig.Proxy == "" {
@@ -160,6 +160,7 @@ func LoadConfigFromReader(r io.Reader) (Config, error) {
 			return Config{}, fmt.Errorf("model %s requires a proxy value when not using automatic ${PORT}", modelId)
 		}
 	}
 	config = AddDefaultGroupToConfig(config)
 	// check that members are all unique in the groups
 	memberUsage := make(map[string]string) // maps member to group it appears in
@@ -38,5 +38,4 @@ func TestConfig_SanitizeCommand(t *testing.T) {
 	args, err = SanitizeCommand("")
 	assert.Error(t, err)
 	assert.Nil(t, args)
 }
@@ -8,6 +8,7 @@ import (
 	"net/http"
 	"net/url"
 	"os/exec"
 	"runtime"
 	"strconv"
 	"strings"
 	"sync"
@@ -80,9 +81,8 @@ func NewProcess(ID string, healthCheckTimeout int, config ModelConfig, processLo
 	concurrentLimit := 10
 	if config.ConcurrencyLimit > 0 {
 		concurrentLimit = config.ConcurrencyLimit
 	} else {
 		proxyLogger.Debugf("Concurrency limit for model %s not set, defaulting to 10", ID)
 	}
 	return &Process{
 		ID:                      ID,
 		config:                  config,
@@ -148,7 +148,9 @@ func isValidTransition(from, to ProcessState) bool {
 		return to == StateStopping
 	case StateStopping:
 		return to == StateStopped || to == StateShutdown
-	case StateFailed, StateShutdown:
+	case StateFailed:
 		return to == StateStopping
 	case StateShutdown:
 		return false // No transitions allowed from these states
 	}
 	return false
@@ -358,12 +360,19 @@ func (p *Process) StopImmediately() {
 		return
 	}
-	p.proxyLogger.Debugf("<%s> Stopping process", p.ID)
+	p.proxyLogger.Debugf("<%s> Stopping process, current state: %s", p.ID, p.CurrentState())
 	currentState := p.CurrentState()
-	// calling Stop() when state is invalid is a no-op
+	if currentState == StateFailed {
-	if curState, err := p.swapState(StateReady, StateStopping); err != nil {
+		if curState, err := p.swapState(StateFailed, StateStopping); err != nil {
-		p.proxyLogger.Infof("<%s> Stop() Ready -> StateStopping err: %v, current state: %v", p.ID, err, curState)
+			p.proxyLogger.Infof("<%s> Stop() Failed -> StateStopping err: %v, current state: %v", p.ID, err, curState)
-		return
+			return
 		}
 	} else {
 		if curState, err := p.swapState(StateReady, StateStopping); err != nil {
 			p.proxyLogger.Infof("<%s> Stop() Ready -> StateStopping err: %v, current state: %v", p.ID, err, curState)
 			return
 		}
 	}
 	// stop the process with a graceful exit timeout
@@ -379,8 +388,14 @@ func (p *Process) StopImmediately() {
 // is in the state of starting, it will cancel it and shut it down. Once a process is in
 // the StateShutdown state, it can not be started again.
 func (p *Process) Shutdown() {
 	if !isValidTransition(p.CurrentState(), StateStopping) {
 		return
 	}
 	p.shutdownCancel()
 	p.stopCommand(p.gracefulStopTimeout)
 	// just force it to this state since there is no recovery from shutdown
 	p.state = StateShutdown
 }
@@ -400,8 +415,38 @@ func (p *Process) stopCommand(sigtermTTL time.Duration) {
 		return
 	}
-	if err := p.terminateProcess(); err != nil {
+	// if err := p.terminateProcess(); err != nil {
-		p.proxyLogger.Debugf("<%s> Process already terminated: %v (normal during shutdown)", p.ID, err)
+	// 	p.proxyLogger.Debugf("<%s> Process already terminated: %v (normal during shutdown)", p.ID, err)
 	// }
 	// the default cmdStop to taskkill /f /t /pid ${PID}
 	if runtime.GOOS == "windows" && strings.TrimSpace(p.config.CmdStop) == "" {
 		p.config.CmdStop = "taskkill /f /t /pid ${PID}"
 	}
 	if p.config.CmdStop != "" {
 		// replace ${PID} with the pid of the process
 		stopArgs, err := SanitizeCommand(strings.ReplaceAll(p.config.CmdStop, "${PID}", fmt.Sprintf("%d", p.cmd.Process.Pid)))
 		if err != nil {
 			p.proxyLogger.Errorf("<%s> Failed to sanitize stop command: %v", p.ID, err)
 			return
 		}
 		p.proxyLogger.Debugf("<%s> Executing stop command: %s", p.ID, strings.Join(stopArgs, " "))
 		stopCmd := exec.Command(stopArgs[0], stopArgs[1:]...)
 		stopCmd.Stdout = p.processLogger
 		stopCmd.Stderr = p.processLogger
 		stopCmd.Env = p.config.Env
 		if err := stopCmd.Run(); err != nil {
 			p.proxyLogger.Errorf("<%s> Failed to exec stop command: %v", p.ID, err)
 			return
 		}
 	} else {
 		if err := p.cmd.Process.Signal(syscall.SIGTERM); err != nil {
 			p.proxyLogger.Errorf("<%s> Failed to send SIGTERM to process: %v", p.ID, err)
 			return
 		}
 	}
 	select {
@@ -1,9 +0,0 @@
 //go:build !windows
 package proxy
 import "syscall"
 func (p *Process) terminateProcess() error {
 	return p.cmd.Process.Signal(syscall.SIGTERM)
 }
@@ -1,14 +0,0 @@
 //go:build windows
 package proxy
 import (
 	"fmt"
 	"os/exec"
 )
 func (p *Process) terminateProcess() error {
 	pid := fmt.Sprintf("%d", p.cmd.Process.Pid)
 	cmd := exec.Command("taskkill", "/f", "/t", "/pid", pid)
 	return cmd.Run()
 }
@@ -449,3 +449,22 @@ func TestProcess_ForceStopWithKill(t *testing.T) {
 	// the request should have been interrupted by SIGKILL
 	<-waitChan
 }
 func TestProcess_StopCmd(t *testing.T) {
 	config := getTestSimpleResponderConfig("test_stop_cmd")
 	if runtime.GOOS == "windows" {
 		config.CmdStop = "taskkill /f /t /pid ${PID}"
 	} else {
 		config.CmdStop = "kill -TERM ${PID}"
 	}
 	process := NewProcess("testStopCmd", 2, config, debugLogger, debugLogger)
 	defer process.Stop()
 	err := process.start()
 	assert.Nil(t, err)
 	assert.Equal(t, process.CurrentState(), StateReady)
 	process.StopImmediately()
 	assert.Equal(t, process.CurrentState(), StateStopped)
 }
@@ -352,6 +352,8 @@ func (pm *ProxyManager) upstreamIndex(c *gin.Context) {
 					stateStr = "Failed"
 				case StateShutdown:
 					stateStr = "Shutdown"
 				case StateStopped:
 					stateStr = "Stopped"
 				default:
 					stateStr = "Unknown"
 				}
@@ -0,0 +1,189 @@
 #!/bin/sh
 # This script installs llama-swap on Linux.
 # It detects the current operating system architecture and installs the appropriate version of llama-swap.
 set -eu
 red="$( (/usr/bin/tput bold || :; /usr/bin/tput setaf 1 || :) 2>&-)"
 plain="$( (/usr/bin/tput sgr0 || :) 2>&-)"
 status() { echo ">>> $*" >&2; }
 error() { echo "${red}ERROR:${plain} $*"; exit 1; }
 warning() { echo "${red}WARNING:${plain} $*"; }
 available() { command -v $1 >/dev/null; }
 require() {
    local MISSING=''
    for TOOL in $*; do
        if ! available $TOOL; then
            MISSING="$MISSING $TOOL"
        fi
    done
    echo $MISSING
 }
 SUDO=
 if [ "$(id -u)" -ne 0 ]; then
    if ! available sudo; then
        error "This script requires superuser permissions. Please re-run as root."
    fi
    SUDO="sudo"
 fi
 NEEDS=$(require curl tee jq tar)
 if [ -n "$NEEDS" ]; then
    status "ERROR: The following tools are required but missing:"
    for NEED in $NEEDS; do
        echo "  - $NEED"
    done
    exit 1
 fi
 [ "$(uname -s)" = "Linux" ] || error 'This script is intended to run on Linux only.'
 ARCH=$(uname -m)
 case "$ARCH" in
    x86_64) ARCH="amd64" ;;
    aarch64|arm64) ARCH="arm64" ;;
    *) error "Unsupported architecture: $ARCH" ;;
 esac
 IS_WSL2=false
 KERN=$(uname -r)
 case "$KERN" in
    *icrosoft*WSL2 | *icrosoft*wsl2) IS_WSL2=true;;
    *icrosoft) error "Microsoft WSL1 is not currently supported. Please use WSL2 with 'wsl --set-version <distro> 2'" ;;
    *) ;;
 esac
 download_binary() {
    ASSET_NAME="linux_$ARCH"
    # Fetch the latest release info and extract the matching asset URL
    DL_URL=$(curl -s "https://api.github.com/repos/mostlygeek/llama-swap/releases/latest" | \
        jq -r --arg name "$ASSET_NAME" \
        '.assets[] | select(.name | contains($name)) | .browser_download_url')
    # Check if a URL was successfully extracted
    if [ -z "$DL_URL" ]; then
        error "No matching asset found with name containing '$ASSET_NAME'."
    fi
    status "Downloading Linux $ARCH binary"
    curl -s -L "$DL_URL" | $SUDO tar -xzf - -C /usr/local/bin llama-swap
 }
 download_binary
 configure_systemd() {
    if ! id llama-swap >/dev/null 2>&1; then
        status "Creating llama-swap user..."
        $SUDO useradd -r -s /bin/false -U -m -d /usr/share/llama-swap llama-swap
    fi
    if getent group render >/dev/null 2>&1; then
        status "Adding llama-swap user to render group..."
        $SUDO usermod -a -G render llama-swap
    fi
    if getent group video >/dev/null 2>&1; then
        status "Adding llama-swap user to video group..."
        $SUDO usermod -a -G video llama-swap
    fi
    if getent group docker >/dev/null 2>&1; then
        status "Adding llama-swap user to docker group..."
        $SUDO usermod -a -G docker llama-swap
    fi
    status "Adding current user to llama-swap group..."
    $SUDO usermod -a -G llama-swap $(whoami)
    if [ ! -f "/usr/share/llama-swap/config.yaml" ]; then
        status "Creating default config.yaml..."
        cat <<EOF | $SUDO -u llama-swap tee /usr/share/llama-swap/config.yaml >/dev/null
 # default 15s likely to fail for default models due to downloading models
 healthCheckTimeout: 60
 models:
  "qwen2.5":
    cmd: |
      docker run
        --rm
        -p \${PORT}:8080
        --name qwen2.5
      ghcr.io/ggml-org/llama.cpp:server
        -hf bartowski/Qwen2.5-0.5B-Instruct-GGUF:Q4_K_M
    cmdStop: docker stop qwen2.5
  "smollm2":
    cmd: |
      docker run
        --rm
        -p \${PORT}:8080
        --name smollm2
      ghcr.io/ggml-org/llama.cpp:server
        -hf bartowski/SmolLM2-135M-Instruct-GGUF:Q4_K_M
    cmdStop: docker stop smollm2
 EOF
    fi
    status "Creating llama-swap systemd service..."
    cat <<EOF | $SUDO tee /etc/systemd/system/llama-swap.service >/dev/null
 [Unit]
 Description=llama-swap
 After=network.target
 [Service]
 User=llama-swap
 Group=llama-swap
 # set this to match your environment
 ExecStart=/usr/local/bin/llama-swap --config /usr/share/llama-swap/config.yaml --watch-config
 Restart=on-failure
 RestartSec=3
 StartLimitBurst=3
 StartLimitInterval=30
 [Install]
 WantedBy=multi-user.target
 EOF
    SYSTEMCTL_RUNNING="$(systemctl is-system-running || true)"
    case $SYSTEMCTL_RUNNING in
        running|degraded)
            status "Enabling and starting llama-swap service..."
            $SUDO systemctl daemon-reload
            $SUDO systemctl enable llama-swap
            start_service() { $SUDO systemctl restart llama-swap; }
            trap start_service EXIT
            ;;
        *)
            warning "systemd is not running"
            if [ "$IS_WSL2" = true ]; then
                warning "see https://learn.microsoft.com/en-us/windows/wsl/systemd#how-to-enable-systemd to enable it"
            fi
            ;;
    esac
 }
 if available systemctl; then
    configure_systemd
 fi
 install_success() {
    status 'The llama-swap API is now available at 127.0.0.1:8080.'
    status 'Customize the config file at /usr/share/llama-swap/config.yaml.'
    status 'Install complete.'
 }
 # WSL2 only supports GPUs via nvidia passthrough
 # so check for nvidia-smi to determine if GPU is available
 if [ "$IS_WSL2" = true ]; then
    if available nvidia-smi && [ -n "$(nvidia-smi | grep -o "CUDA Version: [0-9]*\.[0-9]*")" ]; then
        status "Nvidia GPU detected."
    fi
    exit 0
 fi
 install_success
@@ -0,0 +1,68 @@
 #!/bin/sh
 # This script uninstalls llama-swap on Linux.
 # It removes the binary, systemd service, config.yaml (optional), and llama-swap user and group.
 set -eu
 red="$( (/usr/bin/tput bold || :; /usr/bin/tput setaf 1 || :) 2>&-)"
 plain="$( (/usr/bin/tput sgr0 || :) 2>&-)"
 status() { echo ">>> $*" >&2; }
 error() { echo "${red}ERROR:${plain} $*"; exit 1; }
 warning() { echo "${red}WARNING:${plain} $*"; }
 available() { command -v $1 >/dev/null; }
 SUDO=
 if [ "$(id -u)" -ne 0 ]; then
    if ! available sudo; then
        error "This script requires superuser permissions. Please re-run as root."
    fi
    SUDO="sudo"
 fi
 configure_systemd() {
    status "Stopping llama-swap service..."
    $SUDO systemctl stop llama-swap
    status "Disabling llama-swap service..."
    $SUDO systemctl disable llama-swap
 }
 if available systemctl; then
    configure_systemd
 fi
 if available llama-swap; then
    status "Removing llama-swap binary..."
    $SUDO rm $(which llama-swap)
 fi
 if [ -f "/usr/share/llama-swap/config.yaml" ]; then
    while true; do
        printf "Delete config.yaml (/usr/share/llama-swap/config.yaml)? [y/N] " >&2
        read answer
        case "$answer" in
            [Yy]* ) 
                $SUDO rm -r /usr/share/llama-swap
                break
                ;;
            [Nn]* | "" ) 
                break
                ;;
            * ) 
                echo "Invalid input. Please enter y or n."
                ;;
        esac
    done
 fi
 if id llama-swap >/dev/null 2>&1; then
    status "Removing llama-swap user..."
    $SUDO userdel llama-swap
 fi
 if getent group llama-swap >/dev/null 2>&1; then
    status "Removing llama-swap group..."
    $SUDO groupdel llama-swap
 fi
Author	SHA1	Message	Date
Benson Wong	e7af671d8e	remove noisy debug print message	2025-05-19 15:36:15 -07:00
Benson Wong	8e62098eef	add guard to avoid unnecessary logic in Process.Shutdown	2025-05-19 15:34:30 -07:00
choyuansu	c260907415	Add linux install and uninstall shell scripts (#139 ) Contribution for install, and uninstall llama-swap in linux.	2025-05-19 12:03:33 -07:00
Benson Wong	b83a5fa291	make Failed stated recoverable (#137 ) A process in the failed state can transition to stopped either by calling /unload or swapping to another model.	2025-05-16 19:54:44 -07:00
Benson Wong	6e2ff28d59	improve cmdStop docs [no ci]	2025-05-16 13:52:04 -07:00
Benson Wong	a8b81f2799	Add stopCmd for custom stopping instructions (#136 ) Allow configuration of how a model is stopped before swapping. Setting `cmdStop` in the configuration will override the default behaviour and enables better integration with other process/container managers like docker or podman.	2025-05-16 13:48:42 -07:00
Benson Wong	f9ee7156dc	update configuration examples for multiline yaml commands #133	2025-05-16 11:45:39 -07:00