increase health check to a minimum of 5 seconds

2025-02-18 17:35:52 -08:00
14 changed files with 40 additions and 777 deletions
@@ -1,23 +0,0 @@
 # https://docs.github.com/en/actions/use-cases-and-examples/project-management/closing-inactive-issues
 name: Close inactive issues
 on:
  schedule:
    - cron: "32 1 * * *"
 jobs:
  close-issues:
    runs-on: ubuntu-latest
    permissions:
      issues: write
      pull-requests: write
    steps:
      - uses: actions/stale@v9
        with:
          days-before-issue-stale: 30
          days-before-issue-close: 14
          stale-issue-label: "stale"
          stale-issue-message: "This issue is stale because it has been open for 30 days with no activity."
          close-issue-message: "This issue was closed because it has been inactive for 14 days since being marked as stale."
          days-before-pr-stale: -1
          days-before-pr-close: -1
          repo-token: ${{ secrets.GITHUB_TOKEN }}
@@ -16,7 +16,6 @@ jobs:
    strategy:
      matrix:
        platform: [intel, cuda, vulkan, cpu, musa]
      fail-fast: false
    steps:
      - name: Checkout code
        uses: actions/checkout@v4
@@ -1,32 +0,0 @@
 # This workflow will build a golang project
 name: CI
 on:
  push:
    branches: [ "main" ]
  pull_request:
    branches: [ "main" ]
  # Allows manual triggering of the workflow
  workflow_dispatch:
 jobs:
  run-tests:
    runs-on: ubuntu-latest
    steps:
    - uses: actions/checkout@v4
    - name: Set up Go
      uses: actions/setup-go@v4
      with:
        go-version: '1.23'
    # necessary for testing proxy/Process swapping
    - name: Create simple-responder
      run: make simple-responder
    - name: Test all
      run: make test-all
@@ -11,23 +11,20 @@ Written in golang, it is very easy to install (single binary with no dependancie
 - ✅ Easy to deploy: single binary with no dependencies
 - ✅ Easy to config: single yaml file
 - ✅ On-demand model switching
 - ✅ Full control over server settings per model
 - ✅ OpenAI API supported endpoints:
  - `v1/completions`
  - `v1/chat/completions`
  - `v1/embeddings`
  - `v1/rerank`
  - `v1/audio/speech` ([#36](https://github.com/mostlygeek/llama-swap/issues/36))
-  - `v1/audio/transcriptions` ([docs](https://github.com/mostlygeek/llama-swap/issues/41#issuecomment-2722637867))
+- ✅ Multiple GPU support
 - ✅ llama-swap custom API endpoints
  - `/log` - remote log monitoring
  - `/upstream/:model_id` - direct access to upstream HTTP server ([demo](https://github.com/mostlygeek/llama-swap/pull/31))
  - `/unload` - manually unload running models ([#58](https://github.com/mostlygeek/llama-swap/issues/58))
  - `/running` - list currently running models ([#61](https://github.com/mostlygeek/llama-swap/issues/61))
 - ✅ Run multiple models at once with `profiles` ([docs](https://github.com/mostlygeek/llama-swap/issues/53#issuecomment-2660761741))
 - ✅ Automatic unloading of models after timeout by setting a `ttl`
 - ✅ Use any local OpenAI compatible server (llama.cpp, vllm, tabbyAPI, etc)
 - ✅ Docker and Podman support
- ✅ Full control over server settings per model
+- ✅ Run multiple models at once with `profiles` ([docs](https://github.com/mostlygeek/llama-swap/issues/53#issuecomment-2660761741))
 - ✅ Remote log monitoring at `/log`
 - ✅ Automatic unloading of models from GPUs after timeout
 - ✅ Use any local OpenAI compatible server (llama.cpp, vllm, tabbyAPI, etc)
 - ✅ Direct access to upstream HTTP server via `/upstream/:model_id` ([demo](https://github.com/mostlygeek/llama-swap/pull/31))
 ## How does llama-swap work?
@@ -117,13 +114,6 @@ models:
      ghcr.io/ggerganov/llama.cpp:server
      --model '/models/Qwen2.5-Coder-0.5B-Instruct-Q4_K_M.gguf'
  # `useModelName` will send a specific model name to the upstream server
  # overriding whatever was set in the request
  "qwq":
    proxy: http://127.0.0.1:11434
    cmd: my-server
    useModelName: "qwen:qwq"
 # profiles make it easy to managing multi model (and gpu) configurations.
 #
 # Tips:
@@ -136,16 +126,11 @@ profiles:
    - "llama"
 ```
-### Use Case Examples
+### Advanced Examples
 - [config.example.yaml](config.example.yaml) includes example for supporting `v1/embeddings` and `v1/rerank` endpoints
 - [Speculative Decoding](examples/speculative-decoding/README.md) - using a small draft model can increase inference speeds from 20% to 40%. This example includes a configurations Qwen2.5-Coder-32B (2.5x increase) and Llama-3.1-70B (1.4x increase) in the best cases.
 - [Optimizing Code Generation](examples/benchmark-snakegame/README.md) - find the optimal settings for your machine. This example demonstrates defining multiple configurations and testing which one is fastest.
 - [Restart on Config Change](examples/restart-on-config-change/README.md) - automatically restart llama-swap when trying out different configurations.
 ## Configuration
 llama-s
 </details>
@@ -264,11 +249,3 @@ StartLimitInterval=30
 [Install]
 WantedBy=multi-user.target
 ```
 ## Star History
 <picture>
   <source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date&theme=dark" />
   <source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date" />
   <img alt="Star History Chart" src="https://api.star-history.com/svg?repos=mostlygeek/llama-swap&type=Date" />
 </picture>
@@ -38,12 +38,6 @@ else
        | jq -r --arg arch "$ARCH" '.[] | select(.metadata.container.tags[] | startswith("server-\($arch)")) | .metadata.container.tags[]' \
        | sort -r | head -n1 | awk -F '-' '{print $3}')
    # Abort if LCPP_TAG is empty.
    if [[ -z "$LCPP_TAG" ]]; then
        echo "Abort: Could not find llama-server container for arch: $ARCH"
        exit 1
    fi
    CONTAINER_TAG="ghcr.io/mostlygeek/llama-swap:v${LS_VER}-${ARCH}-${LCPP_TAG}"
    CONTAINER_LATEST="ghcr.io/mostlygeek/llama-swap:${ARCH}"
    echo "Building ${CONTAINER_TAG} $LS_VER"
@@ -1,51 +0,0 @@
 # Restart llama-swap on config change
 Sometimes editing the configuration file can take a bit of trail and error to get a model configuration tuned just right. The `watch-and-restart.sh` script can be used to watch `config.yaml` for changes and restart `llama-swap` when it detects a change.
 ```bash
 #!/bin/bash
 #
 # A simple watch and restart llama-swap when its configuration
 # file changes. Useful for trying out configuration changes
 # without manually restarting the server each time.
 if [ -z "$1" ]; then
    echo "Usage: $0 <path to config.yaml>"
    exit 1
 fi
 while true; do
    # Start the process again
    ./llama-swap-linux-amd64 -config $1 -listen :1867 &
    PID=$!
    echo "Started llama-swap with PID $PID"
    # Wait for modifications in the specified directory or file
    inotifywait -e modify "$1"
    # Check if process exists before sending signal
    if kill -0 $PID 2>/dev/null; then
        echo "Sending SIGTERM to $PID"
        kill -SIGTERM $PID
        wait $PID
    else
        echo "Process $PID no longer exists"
    fi
    sleep 1
 done
 ```
 ## Usage and output example
 ```bash
 $ ./watch-and-restart.sh config.yaml
 Started llama-swap with PID 495455
 Setting up watches.
 Watches established.
 llama-swap listening on :1867
 Sending SIGTERM to 495455
 Shutting down llama-swap
 Started llama-swap with PID 495486
 Setting up watches.
 Watches established.
 llama-swap listening on :1867
 ```
@@ -3,11 +3,7 @@ module github.com/mostlygeek/llama-swap
 go 1.23.0
 require (
 	github.com/gin-gonic/gin v1.10.0
 	github.com/google/shlex v0.0.0-20191202100458-e7afc7fbc510
 	github.com/stretchr/testify v1.9.0
 	github.com/tidwall/gjson v1.18.0
 	github.com/tidwall/sjson v1.2.5
 	gopkg.in/yaml.v3 v3.0.1
 )
@@ -19,10 +15,12 @@ require (
 	github.com/davecgh/go-spew v1.1.1 // indirect
 	github.com/gabriel-vasile/mimetype v1.4.3 // indirect
 	github.com/gin-contrib/sse v0.1.0 // indirect
 	github.com/gin-gonic/gin v1.10.0 // indirect
 	github.com/go-playground/locales v0.14.1 // indirect
 	github.com/go-playground/universal-translator v0.18.1 // indirect
 	github.com/go-playground/validator/v10 v10.20.0 // indirect
 	github.com/goccy/go-json v0.10.2 // indirect
 	github.com/google/shlex v0.0.0-20191202100458-e7afc7fbc510 // indirect
 	github.com/json-iterator/go v1.1.12 // indirect
 	github.com/klauspost/cpuid/v2 v2.2.7 // indirect
 	github.com/leodido/go-urn v1.4.0 // indirect
@@ -31,14 +29,12 @@ require (
 	github.com/modern-go/reflect2 v1.0.2 // indirect
 	github.com/pelletier/go-toml/v2 v2.2.2 // indirect
 	github.com/pmezard/go-difflib v1.0.0 // indirect
 	github.com/tidwall/match v1.1.1 // indirect
 	github.com/tidwall/pretty v1.2.1 // indirect
 	github.com/twitchyliquid64/golang-asm v0.15.1 // indirect
 	github.com/ugorji/go/codec v1.2.12 // indirect
 	golang.org/x/arch v0.8.0 // indirect
-	golang.org/x/crypto v0.36.0 // indirect
+	golang.org/x/crypto v0.31.0 // indirect
-	golang.org/x/net v0.37.0 // indirect
+	golang.org/x/net v0.33.0 // indirect
-	golang.org/x/sys v0.31.0 // indirect
+	golang.org/x/sys v0.28.0 // indirect
-	golang.org/x/text v0.23.0 // indirect
+	golang.org/x/text v0.21.0 // indirect
 	google.golang.org/protobuf v1.34.1 // indirect
 )
@@ -57,16 +57,6 @@ github.com/stretchr/testify v1.8.1/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o
 github.com/stretchr/testify v1.8.4/go.mod h1:sz/lmYIOXD/1dqDmKjjqLyZ2RngseejIcXlSw2iwfAo=
 github.com/stretchr/testify v1.9.0 h1:HtqpIVDClZ4nwg75+f6Lvsy/wHu+3BoSGCbBAcpTsTg=
 github.com/stretchr/testify v1.9.0/go.mod h1:r2ic/lqez/lEtzL7wO/rwa5dbSLXVDPFyf8C91i36aY=
 github.com/tidwall/gjson v1.14.2/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk=
 github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY=
 github.com/tidwall/gjson v1.18.0/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk=
 github.com/tidwall/match v1.1.1 h1:+Ho715JplO36QYgwN9PGYNhgZvoUSc9X2c80KVTi+GA=
 github.com/tidwall/match v1.1.1/go.mod h1:eRSPERbgtNPcGhD8UCthc6PmLEQXEWd3PRB5JTxsfmM=
 github.com/tidwall/pretty v1.2.0/go.mod h1:ITEVvHYasfjBbM0u2Pg8T2nJnzm8xPwvNhhsoaGGjNU=
 github.com/tidwall/pretty v1.2.1 h1:qjsOFOWWQl+N3RsoF5/ssm1pHmJJwhjlSbZ51I6wMl4=
 github.com/tidwall/pretty v1.2.1/go.mod h1:ITEVvHYasfjBbM0u2Pg8T2nJnzm8xPwvNhhsoaGGjNU=
 github.com/tidwall/sjson v1.2.5 h1:kLy8mja+1c9jlljvWTlSazM7cKDRfJuR/bOJhcY5NcY=
 github.com/tidwall/sjson v1.2.5/go.mod h1:Fvgq9kS/6ociJEDnK0Fk1cpYF4FIW6ZF7LAe+6jwd28=
 github.com/twitchyliquid64/golang-asm v0.15.1 h1:SU5vSMR7hnwNxj24w34ZyCi/FmDZTkS4MhqMhdFk5YI=
 github.com/twitchyliquid64/golang-asm v0.15.1/go.mod h1:a1lVb/DtPvCB8fslRZhAngC2+aY1QWCk3Cedj/Gdt08=
 github.com/ugorji/go/codec v1.2.12 h1:9LC83zGrHhuUA9l16C9AHXAqEV/2wBQ4nkvumAE65EE=
@@ -78,28 +68,20 @@ golang.org/x/crypto v0.23.0 h1:dIJU/v2J8Mdglj/8rJ6UUOM3Zc9zLZxVZwwxMooUSAI=
 golang.org/x/crypto v0.23.0/go.mod h1:CKFgDieR+mRhux2Lsu27y0fO304Db0wZe70UKqHu0v8=
 golang.org/x/crypto v0.31.0 h1:ihbySMvVjLAeSH1IbfcRTkD/iNscyz8rGzjF/E5hV6U=
 golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
 golang.org/x/crypto v0.36.0 h1:AnAEvhDddvBdpY+uR+MyHmuZzzNqXSe/GvuDeob5L34=
 golang.org/x/crypto v0.36.0/go.mod h1:Y4J0ReaxCR1IMaabaSMugxJES1EpwhBHhv2bDHklZvc=
 golang.org/x/net v0.25.0 h1:d/OCCoBEUq33pjydKrGQhw7IlUPI2Oylr+8qLx49kac=
 golang.org/x/net v0.25.0/go.mod h1:JkAGAh7GEvH74S6FOH42FLoXpXbE/aqXSrIQjXgsiwM=
 golang.org/x/net v0.33.0 h1:74SYHlV8BIgHIFC/LrYkOGIwL19eTYXQ5wc6TBuO36I=
 golang.org/x/net v0.33.0/go.mod h1:HXLR5J+9DxmrqMwG9qjGCxZ+zKXxBru04zlTvWlWuN4=
 golang.org/x/net v0.37.0 h1:1zLorHbz+LYj7MQlSf1+2tPIIgibq2eL5xkrGk6f+2c=
 golang.org/x/net v0.37.0/go.mod h1:ivrbrMbzFq5J41QOQh0siUuly180yBYtLp+CKbEaFx8=
 golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
 golang.org/x/sys v0.20.0 h1:Od9JTbYCk261bKm4M/mw7AklTlFYIa0bIp9BgSm1S8Y=
 golang.org/x/sys v0.20.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
 golang.org/x/sys v0.28.0 h1:Fksou7UEQUWlKvIdsqzJmUmCX3cZuD2+P3XyyzwMhlA=
 golang.org/x/sys v0.28.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
 golang.org/x/sys v0.31.0 h1:ioabZlmFYtWhL+TRYpcnNlLwhyxaM9kWTDEmfnprqik=
 golang.org/x/sys v0.31.0/go.mod h1:BJP2sWEmIv4KK5OTEluFJCKSidICx8ciO85XgH3Ak8k=
 golang.org/x/text v0.15.0 h1:h1V/4gjBv8v9cjcR6+AR5+/cIYK5N/WAgiv4xlsEtAk=
 golang.org/x/text v0.15.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
 golang.org/x/text v0.21.0 h1:zyQAAkrwaneQ066sspRyJaG9VNi/YJ1NfzcGB3hZ/qo=
 golang.org/x/text v0.21.0/go.mod h1:4IBbMaMmOPCJ8SecivzSH54+73PCFmPWxNTLm+vZkEQ=
 golang.org/x/text v0.23.0 h1:D71I7dUrlY+VX0gQShAThNGHFxZ13dGLBHQLVl1mJlY=
 golang.org/x/text v0.23.0/go.mod h1:/BLNzu4aZCJ1+kcD0DNRotWKage4q2rGVAg4o22unh4=
 google.golang.org/protobuf v1.34.1 h1:9ddQBjfCyZPOHPUiPxpYESBLc+T8P3E+Vo4IbKZgFWg=
 google.golang.org/protobuf v1.34.1/go.mod h1:c6P6GXX6sHbq/GpV6MGZEdwhWPcYBgnhAHhKbcUYpos=
 gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
@@ -12,14 +12,12 @@ import (
 	"time"
 	"github.com/gin-gonic/gin"
 	"github.com/tidwall/gjson"
 )
 func main() {
 	gin.SetMode(gin.TestMode)
 	// Define a command-line flag for the port
 	port := flag.String("port", "8080", "port to listen on")
 	expectedModel := flag.String("model", "TheExpectedModel", "model name to expect")
 	// Define a command-line flag for the response message
 	responseMessage := flag.String("respond", "hi", "message to respond with")
@@ -43,70 +41,11 @@ func main() {
 		c.String(200, *responseMessage)
 	})
 	// for issue #62 to check model name strips profile slug
 	// has to be one of the openAI API endpoints that llama-swap proxies
 	// curl http://localhost:8080/v1/audio/speech -d '{"model":"profile:TheExpectedModel"}'
 	r.POST("/v1/audio/speech", func(c *gin.Context) {
 		body, err := io.ReadAll(c.Request.Body)
 		if err != nil {
 			c.JSON(http.StatusInternalServerError, gin.H{"error": "Failed to read request body"})
 			return
 		}
 		defer c.Request.Body.Close()
 		modelName := gjson.GetBytes(body, "model").String()
 		if modelName != *expectedModel {
 			c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("Invalid model: %s, expected: %s", modelName, *expectedModel)})
 			return
 		} else {
 			c.JSON(http.StatusOK, gin.H{"message": "ok"})
 		}
 	})
 	r.POST("/v1/completions", func(c *gin.Context) {
 		c.Header("Content-Type", "text/plain")
 		c.String(200, *responseMessage)
 	})
 	// issue #41
 	r.POST("/v1/audio/transcriptions", func(c *gin.Context) {
 		// Parse the multipart form
 		if err := c.Request.ParseMultipartForm(10 << 20); err != nil { // 10 MB max memory
 			c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("Error parsing multipart form: %s", err)})
 			return
 		}
 		// Get the model from the form values
 		model := c.Request.FormValue("model")
 		if model == "" {
 			c.JSON(http.StatusBadRequest, gin.H{"error": "Missing model parameter"})
 			return
 		}
 		// Get the file from the form
 		file, _, err := c.Request.FormFile("file")
 		if err != nil {
 			c.JSON(http.StatusBadRequest, gin.H{"error": fmt.Sprintf("Error getting file: %s", err)})
 			return
 		}
 		defer file.Close()
 		// Read the file content to get its size
 		fileBytes, err := io.ReadAll(file)
 		if err != nil {
 			c.JSON(http.StatusInternalServerError, gin.H{"error": fmt.Sprintf("Error reading file: %s", err)})
 			return
 		}
 		fileSize := len(fileBytes)
 		// Return a JSON response with the model and transcription text including file size
 		c.JSON(http.StatusOK, gin.H{
 			"text":  fmt.Sprintf("The length of the file is %d bytes", fileSize),
 			"model": model,
 		})
 	})
 	r.GET("/slow-respond", func(c *gin.Context) {
 		echo := c.Query("echo")
 		delay := c.Query("delay")
@@ -17,7 +17,6 @@ type ModelConfig struct {
 	CheckEndpoint string   `yaml:"checkEndpoint"`
 	UnloadAfter   int      `yaml:"ttl"`
 	Unlisted      bool     `yaml:"unlisted"`
 	UseModelName  string   `yaml:"useModelName"`
 }
 func (m *ModelConfig) SanitizedCommand() ([]string, error) {
@@ -116,9 +116,7 @@ func (p *Process) start() error {
 		return fmt.Errorf("can not start(), upstream proxy missing")
 	}
-	// multiple start() calls will wait for the one that is actually starting to
+	// wait for the other start() to complete
 	// complete before proceeding.
 	// ===========
 	curState := p.CurrentState()
 	if curState == StateReady {
@@ -134,26 +132,10 @@ func (p *Process) start() error {
 		return nil
 	}
 	// ===========
 	// There is the possibility of a hard to replicate race condition where
 	// curState *WAS* StateStopped but by the time we get to the p.stateMutex.Lock()
 	// below, it's value has changed!
 	p.stateMutex.Lock()
 	defer p.stateMutex.Unlock()
 	// with the exclusive lock, check if p.state is StateStopped, which is the only valid state
 	// to transition from to StateReady
 	if p.state != StateStopped {
 		if p.state == StateReady {
 			return nil
 		} else {
 			return fmt.Errorf("start() can not proceed expected StateReady but process is in %v", p.state)
 		}
 	}
 	if err := p.setState(StateStarting); err != nil {
 		return err
 	}
@@ -169,8 +169,6 @@ func TestProcess_LowTTLValue(t *testing.T) {
 }
 // issue #19
 // This test makes sure using Process.Stop() does not affect pending HTTP
 // requests. All HTTP requests in this test should complete successfully.
 func TestProcess_HTTPRequestsHaveTimeToFinish(t *testing.T) {
 	if testing.Short() {
 		t.Skip("skipping slow test")
@@ -194,9 +192,8 @@ func TestProcess_HTTPRequestsHaveTimeToFinish(t *testing.T) {
 		wg.Add(1)
 		go func(key string) {
 			defer wg.Done()
-			// send a request where simple-responder is will wait 300ms before responding
+			// send a request that should take 5 * 200ms (1 second) to complete
-			// this will simulate an in-progress request.
+			req := httptest.NewRequest("GET", fmt.Sprintf("/slow-respond?echo=%s&delay=200ms", key), nil)
 			req := httptest.NewRequest("GET", fmt.Sprintf("/slow-respond?echo=%s&delay=300ms", key), nil)
 			w := httptest.NewRecorder()
 			process.ProxyRequest(w, req)
@@ -212,9 +209,9 @@ func TestProcess_HTTPRequestsHaveTimeToFinish(t *testing.T) {
 		}(key)
 	}
-	// Stop the process while requests are still being processed
+	// stop the requests in the middle
 	go func() {
-		<-time.After(150 * time.Millisecond)
+		<-time.After(500 * time.Millisecond)
 		process.Stop()
 	}()
@@ -5,7 +5,6 @@ import (
 	"encoding/json"
 	"fmt"
 	"io"
 	"mime/multipart"
 	"net/http"
 	"sort"
 	"strconv"
@@ -14,8 +13,6 @@ import (
 	"time"
 	"github.com/gin-gonic/gin"
 	"github.com/tidwall/gjson"
 	"github.com/tidwall/sjson"
 )
 const (
@@ -96,7 +93,6 @@ func New(config *Config) *ProxyManager {
 	// Support audio/speech endpoint
 	pm.ginEngine.POST("/v1/audio/speech", pm.proxyOAIHandler)
 	pm.ginEngine.POST("/v1/audio/transcriptions", pm.proxyOAIPostFormHandler)
 	pm.ginEngine.GET("/v1/models", pm.listModelsHandler)
@@ -108,10 +104,6 @@ func New(config *Config) *ProxyManager {
 	pm.ginEngine.GET("/upstream", pm.upstreamIndex)
 	pm.ginEngine.Any("/upstream/:model_id/*upstreamPath", pm.proxyToUpstream)
 	pm.ginEngine.GET("/unload", pm.unloadAllModelsHandler)
 	pm.ginEngine.GET("/running", pm.listRunningProcessesHandler)
 	pm.ginEngine.GET("/", func(c *gin.Context) {
 		// Set the Content-Type header to text/html
 		c.Header("Content-Type", "text/html")
@@ -230,7 +222,11 @@ func (pm *ProxyManager) swapModel(requestedModel string) (*Process, error) {
 	defer pm.Unlock()
 	// Check if requestedModel contains a PROFILE_SPLIT_CHAR
-	profileName, modelName := splitRequestedModel(requestedModel)
+	profileName, modelName := "", requestedModel
 	if idx := strings.Index(requestedModel, PROFILE_SPLIT_CHAR); idx != -1 {
 		profileName = requestedModel[:idx]
 		modelName = requestedModel[idx+1:]
 	}
 	if profileName != "" {
 		if _, found := pm.config.Profiles[profileName]; !found {
@@ -346,148 +342,29 @@ func (pm *ProxyManager) proxyOAIHandler(c *gin.Context) {
 		return
 	}
-	requestedModel := gjson.GetBytes(bodyBytes, "model").String()
+	var requestBody map[string]interface{}
-	if requestedModel == "" {
+	if err := json.Unmarshal(bodyBytes, &requestBody); err != nil {
 		pm.sendErrorResponse(c, http.StatusBadRequest, fmt.Sprintf("invalid JSON: %s", err.Error()))
 		return
 	}
 	model, ok := requestBody["model"].(string)
 	if !ok {
 		pm.sendErrorResponse(c, http.StatusBadRequest, "missing or invalid 'model' key")
 	}
 	process, err := pm.swapModel(requestedModel)
 	if err != nil {
 		pm.sendErrorResponse(c, http.StatusNotFound, fmt.Sprintf("unable to swap to model, %s", err.Error()))
 		return
 	}
-	// issue #69 allow custom model names to be sent to upstream
+	if process, err := pm.swapModel(model); err != nil {
-	if process.config.UseModelName != "" {
+		pm.sendErrorResponse(c, http.StatusNotFound, fmt.Sprintf("unable to swap to model, %s", err.Error()))
-		bodyBytes, err = sjson.SetBytes(bodyBytes, "model", process.config.UseModelName)
+		return
 		if err != nil {
 			pm.sendErrorResponse(c, http.StatusInternalServerError, fmt.Sprintf("error updating JSON: %s", err.Error()))
 			return
 		}
 	} else {
-		profileName, modelName := splitRequestedModel(requestedModel)
+		c.Request.Body = io.NopCloser(bytes.NewBuffer(bodyBytes))
 		if profileName != "" {
 			bodyBytes, err = sjson.SetBytes(bodyBytes, "model", modelName)
 			if err != nil {
 				pm.sendErrorResponse(c, http.StatusInternalServerError, fmt.Sprintf("error updating JSON: %s", err.Error()))
 				return
 			}
 		}
 		// dechunk it as we already have all the body bytes see issue #11
 		c.Request.Header.Del("transfer-encoding")
 		c.Request.Header.Add("content-length", strconv.Itoa(len(bodyBytes)))
 		process.ProxyRequest(c.Writer, c.Request)
 	}
 	c.Request.Body = io.NopCloser(bytes.NewBuffer(bodyBytes))
 	// dechunk it as we already have all the body bytes see issue #11
 	c.Request.Header.Del("transfer-encoding")
 	c.Request.Header.Add("content-length", strconv.Itoa(len(bodyBytes)))
 	process.ProxyRequest(c.Writer, c.Request)
 }
 func (pm *ProxyManager) proxyOAIPostFormHandler(c *gin.Context) {
 	// We need to reconstruct the multipart form in any case since the body is consumed
 	// Create a new buffer for the reconstructed request
 	var requestBuffer bytes.Buffer
 	multipartWriter := multipart.NewWriter(&requestBuffer)
 	// Parse multipart form
 	if err := c.Request.ParseMultipartForm(32 << 20); err != nil { // 32MB max memory, larger files go to tmp disk
 		pm.sendErrorResponse(c, http.StatusBadRequest, fmt.Sprintf("error parsing multipart form: %s", err.Error()))
 		return
 	}
 	// Get model parameter from the form
 	requestedModel := c.Request.FormValue("model")
 	if requestedModel == "" {
 		pm.sendErrorResponse(c, http.StatusBadRequest, "missing or invalid 'model' parameter in form data")
 		return
 	}
 	// Swap to the requested model
 	process, err := pm.swapModel(requestedModel)
 	if err != nil {
 		pm.sendErrorResponse(c, http.StatusNotFound, fmt.Sprintf("unable to swap to model, %s", err.Error()))
 		return
 	}
 	// Get profile name and model name from the requested model
 	profileName, modelName := splitRequestedModel(requestedModel)
 	// Copy all form values
 	for key, values := range c.Request.MultipartForm.Value {
 		for _, value := range values {
 			fieldValue := value
 			// If this is the model field and we have a profile, use just the model name
 			if key == "model" {
 				if process.config.UseModelName != "" {
 					fieldValue = process.config.UseModelName
 				} else if profileName != "" {
 					fieldValue = modelName
 				}
 			}
 			field, err := multipartWriter.CreateFormField(key)
 			if err != nil {
 				pm.sendErrorResponse(c, http.StatusInternalServerError, "error recreating form field")
 				return
 			}
 			if _, err = field.Write([]byte(fieldValue)); err != nil {
 				pm.sendErrorResponse(c, http.StatusInternalServerError, "error writing form field")
 				return
 			}
 		}
 	}
 	// Copy all files from the original request
 	for key, fileHeaders := range c.Request.MultipartForm.File {
 		for _, fileHeader := range fileHeaders {
 			formFile, err := multipartWriter.CreateFormFile(key, fileHeader.Filename)
 			if err != nil {
 				pm.sendErrorResponse(c, http.StatusInternalServerError, "error recreating form file")
 				return
 			}
 			file, err := fileHeader.Open()
 			if err != nil {
 				pm.sendErrorResponse(c, http.StatusInternalServerError, "error opening uploaded file")
 				return
 			}
 			if _, err = io.Copy(formFile, file); err != nil {
 				file.Close()
 				pm.sendErrorResponse(c, http.StatusInternalServerError, "error copying file data")
 				return
 			}
 			file.Close()
 		}
 	}
 	// Close the multipart writer to finalize the form
 	if err := multipartWriter.Close(); err != nil {
 		pm.sendErrorResponse(c, http.StatusInternalServerError, "error finalizing multipart form")
 		return
 	}
 	// Create a new request with the reconstructed form data
 	modifiedReq, err := http.NewRequestWithContext(
 		c.Request.Context(),
 		c.Request.Method,
 		c.Request.URL.String(),
 		&requestBuffer,
 	)
 	if err != nil {
 		pm.sendErrorResponse(c, http.StatusInternalServerError, "error creating modified request")
 		return
 	}
 	// Copy the headers from the original request
 	modifiedReq.Header = c.Request.Header.Clone()
 	modifiedReq.Header.Set("Content-Type", multipartWriter.FormDataContentType())
 	// Use the modified request for proxying
 	process.ProxyRequest(c.Writer, modifiedReq)
 }
 func (pm *ProxyManager) sendErrorResponse(c *gin.Context, statusCode int, message string) {
@@ -500,42 +377,6 @@ func (pm *ProxyManager) sendErrorResponse(c *gin.Context, statusCode int, messag
 	}
 }
 func (pm *ProxyManager) unloadAllModelsHandler(c *gin.Context) {
 	pm.StopProcesses()
 	c.String(http.StatusOK, "OK")
 }
 func (pm *ProxyManager) listRunningProcessesHandler(context *gin.Context) {
 	context.Header("Content-Type", "application/json")
 	runningProcesses := make([]gin.H, 0) // Default to an empty response.
 	for _, process := range pm.currentProcesses {
 		// Append the process ID and State (multiple entries if profiles are being used).
 		runningProcesses = append(runningProcesses, gin.H{
 			"model": process.ID,
 			"state": process.state,
 		})
 	}
 	// Put the results under the `running` key.
 	response := gin.H{
 		"running": runningProcesses,
 	}
 	context.JSON(http.StatusOK, response) // Always return 200 OK
 }
 func ProcessKeyName(groupName, modelName string) string {
 	return groupName + PROFILE_SPLIT_CHAR + modelName
 }
 func splitRequestedModel(requestedModel string) (string, string) {
 	profileName, modelName := "", requestedModel
 	if idx := strings.Index(requestedModel, PROFILE_SPLIT_CHAR); idx != -1 {
 		profileName = requestedModel[:idx]
 		modelName = requestedModel[idx+1:]
 	}
 	return profileName, modelName
 }
@@ -4,8 +4,6 @@ import (
 	"bytes"
 	"encoding/json"
 	"fmt"
 	"math/rand"
 	"mime/multipart"
 	"net/http"
 	"net/http/httptest"
 	"sync"
@@ -306,338 +304,3 @@ func TestProxyManager_Shutdown(t *testing.T) {
 	}()
 	wg.Wait()
 }
 func TestProxyManager_Unload(t *testing.T) {
 	config := &Config{
 		HealthCheckTimeout: 15,
 		Models: map[string]ModelConfig{
 			"model1": getTestSimpleResponderConfig("model1"),
 		},
 	}
 	proxy := New(config)
 	proc, err := proxy.swapModel("model1")
 	assert.NoError(t, err)
 	assert.NotNil(t, proc)
 	assert.Len(t, proxy.currentProcesses, 1)
 	req := httptest.NewRequest("GET", "/unload", nil)
 	w := httptest.NewRecorder()
 	proxy.HandlerFunc(w, req)
 	assert.Equal(t, http.StatusOK, w.Code)
 	assert.Equal(t, w.Body.String(), "OK")
 	assert.Len(t, proxy.currentProcesses, 0)
 }
 // issue 62, strip profile slug from model name
 func TestProxyManager_StripProfileSlug(t *testing.T) {
 	config := &Config{
 		HealthCheckTimeout: 15,
 		Profiles: map[string][]string{
 			"test": {"TheExpectedModel"}, // TheExpectedModel is default in simple-responder.go
 		},
 		Models: map[string]ModelConfig{
 			"TheExpectedModel": getTestSimpleResponderConfig("TheExpectedModel"),
 		},
 	}
 	proxy := New(config)
 	defer proxy.StopProcesses()
 	reqBody := fmt.Sprintf(`{"model":"%s"}`, "test:TheExpectedModel")
 	req := httptest.NewRequest("POST", "/v1/audio/speech", bytes.NewBufferString(reqBody))
 	w := httptest.NewRecorder()
 	proxy.HandlerFunc(w, req)
 	assert.Equal(t, http.StatusOK, w.Code)
 	assert.Contains(t, w.Body.String(), "ok")
 }
 // Test issue #61 `Listing the current list of models and the loaded model.`
 func TestProxyManager_RunningEndpoint(t *testing.T) {
 	// Shared configuration
 	config := &Config{
 		HealthCheckTimeout: 15,
 		Models: map[string]ModelConfig{
 			"model1": getTestSimpleResponderConfig("model1"),
 			"model2": getTestSimpleResponderConfig("model2"),
 		},
 		Profiles: map[string][]string{
 			"test": {"model1", "model2"},
 		},
 	}
 	// Define a helper struct to parse the JSON response.
 	type RunningResponse struct {
 		Running []struct {
 			Model string `json:"model"`
 			State string `json:"state"`
 		} `json:"running"`
 	}
 	// Create proxy once for all tests
 	proxy := New(config)
 	defer proxy.StopProcesses()
 	t.Run("no models loaded", func(t *testing.T) {
 		req := httptest.NewRequest("GET", "/running", nil)
 		w := httptest.NewRecorder()
 		proxy.HandlerFunc(w, req)
 		assert.Equal(t, http.StatusOK, w.Code)
 		var response RunningResponse
 		// Check if this is a valid JSON object.
 		assert.NoError(t, json.Unmarshal(w.Body.Bytes(), &response))
 		// We should have an empty running array here.
 		assert.Empty(t, response.Running, "expected no running models")
 	})
 	t.Run("single model loaded", func(t *testing.T) {
 		// Load just a model.
 		reqBody := `{"model":"model1"}`
 		req := httptest.NewRequest("POST", "/v1/chat/completions", bytes.NewBufferString(reqBody))
 		w := httptest.NewRecorder()
 		proxy.HandlerFunc(w, req)
 		assert.Equal(t, http.StatusOK, w.Code)
 		// Simulate browser call for the `/running` endpoint.
 		req = httptest.NewRequest("GET", "/running", nil)
 		w = httptest.NewRecorder()
 		proxy.HandlerFunc(w, req)
 		var response RunningResponse
 		assert.NoError(t, json.Unmarshal(w.Body.Bytes(), &response))
 		// Check if we have a single array element.
 		assert.Len(t, response.Running, 1)
 		// Is this the right model?
 		assert.Equal(t, "model1", response.Running[0].Model)
 		// Is the model loaded?
 		assert.Equal(t, "ready", response.Running[0].State)
 	})
 	t.Run("multiple models via profile", func(t *testing.T) {
 		// Load more than one model.
 		for _, model := range []string{"model1", "model2"} {
 			profileModel := ProcessKeyName("test", model)
 			reqBody := fmt.Sprintf(`{"model":"%s"}`, profileModel)
 			req := httptest.NewRequest("POST", "/v1/chat/completions", bytes.NewBufferString(reqBody))
 			w := httptest.NewRecorder()
 			proxy.HandlerFunc(w, req)
 			assert.Equal(t, http.StatusOK, w.Code)
 		}
 		// Simulate the browser call.
 		req := httptest.NewRequest("GET", "/running", nil)
 		w := httptest.NewRecorder()
 		proxy.HandlerFunc(w, req)
 		var response RunningResponse
 		// The JSON response must be valid.
 		assert.NoError(t, json.Unmarshal(w.Body.Bytes(), &response))
 		// The response should contain 2 models.
 		assert.Len(t, response.Running, 2)
 		expectedModels := map[string]struct{}{
 			"model1": {},
 			"model2": {},
 		}
 		// Iterate through the models and check their states as well.
 		for _, entry := range response.Running {
 			_, exists := expectedModels[entry.Model]
 			assert.True(t, exists, "unexpected model %s", entry.Model)
 			assert.Equal(t, "ready", entry.State)
 			delete(expectedModels, entry.Model)
 		}
 		// Since we deleted each model while testing for its validity we should have no more models in the response.
 		assert.Empty(t, expectedModels, "unexpected additional models in response")
 	})
 }
 func TestProxyManager_AudioTranscriptionHandler(t *testing.T) {
 	config := &Config{
 		HealthCheckTimeout: 15,
 		Profiles: map[string][]string{
 			"test": {"TheExpectedModel"},
 		},
 		Models: map[string]ModelConfig{
 			"TheExpectedModel": getTestSimpleResponderConfig("TheExpectedModel"),
 		},
 	}
 	proxy := New(config)
 	defer proxy.StopProcesses()
 	testCases := []struct {
 		name        string
 		modelInput  string
 		expectModel string
 	}{
 		{
 			name:        "With Profile Prefix",
 			modelInput:  "test:TheExpectedModel",
 			expectModel: "TheExpectedModel", // Profile prefix should be stripped
 		},
 		{
 			name:        "Without Profile Prefix",
 			modelInput:  "TheExpectedModel",
 			expectModel: "TheExpectedModel", // Should remain the same
 		},
 	}
 	for _, tc := range testCases {
 		t.Run(tc.name, func(t *testing.T) {
 			// Create a buffer with multipart form data
 			var b bytes.Buffer
 			w := multipart.NewWriter(&b)
 			// Add the model field
 			fw, err := w.CreateFormField("model")
 			assert.NoError(t, err)
 			_, err = fw.Write([]byte(tc.modelInput))
 			assert.NoError(t, err)
 			// Add a file field
 			fw, err = w.CreateFormFile("file", "test.mp3")
 			assert.NoError(t, err)
 			// Generate random content length between 10 and 20
 			contentLength := rand.Intn(11) + 10 // 10 to 20
 			content := make([]byte, contentLength)
 			_, err = fw.Write(content)
 			assert.NoError(t, err)
 			w.Close()
 			// Create the request with the multipart form data
 			req := httptest.NewRequest("POST", "/v1/audio/transcriptions", &b)
 			req.Header.Set("Content-Type", w.FormDataContentType())
 			rec := httptest.NewRecorder()
 			proxy.HandlerFunc(rec, req)
 			// Verify the response
 			assert.Equal(t, http.StatusOK, rec.Code)
 			var response map[string]string
 			err = json.Unmarshal(rec.Body.Bytes(), &response)
 			assert.NoError(t, err)
 			assert.Equal(t, tc.expectModel, response["model"])
 			assert.Equal(t, response["text"], fmt.Sprintf("The length of the file is %d bytes", contentLength)) // matches simple-responder
 		})
 	}
 }
 func TestProxyManager_SplitRequestedModel(t *testing.T) {
 	tests := []struct {
 		name            string
 		requestedModel  string
 		expectedProfile string
 		expectedModel   string
 	}{
 		{"no profile", "gpt-4", "", "gpt-4"},
 		{"with profile", "profile1:gpt-4", "profile1", "gpt-4"},
 		{"only profile", "profile1:", "profile1", ""},
 		{"empty model", ":gpt-4", "", "gpt-4"},
 		{"empty profile", ":", "", ""},
 		{"no split char", "gpt-4", "", "gpt-4"},
 		{"profile and model with delimiter", "profile1:delimiter:gpt-4", "profile1", "delimiter:gpt-4"},
 	}
 	for _, tt := range tests {
 		t.Run(tt.name, func(t *testing.T) {
 			profileName, modelName := splitRequestedModel(tt.requestedModel)
 			if profileName != tt.expectedProfile {
 				t.Errorf("splitRequestedModel(%q) = %q, %q; want %q, %q", tt.requestedModel, profileName, modelName, tt.expectedProfile, tt.expectedModel)
 			}
 			if modelName != tt.expectedModel {
 				t.Errorf("splitRequestedModel(%q) = %q, %q; want %q, %q", tt.requestedModel, profileName, modelName, tt.expectedProfile, tt.expectedModel)
 			}
 		})
 	}
 }
 // Test useModelName in configuration sends overrides what is sent to upstream
 func TestProxyManager_UseModelName(t *testing.T) {
 	upstreamModelName := "upstreamModel"
 	modelConfig := getTestSimpleResponderConfig(upstreamModelName)
 	modelConfig.UseModelName = upstreamModelName
 	config := &Config{
 		HealthCheckTimeout: 15,
 		Profiles: map[string][]string{
 			"test": {"model1"},
 		},
 		Models: map[string]ModelConfig{
 			"model1": modelConfig,
 		},
 	}
 	proxy := New(config)
 	defer proxy.StopProcesses()
 	tests := []struct {
 		description    string
 		requestedModel string
 	}{
 		{"useModelName over rides requested model", "model1"},
 		{"useModelName over rides requested profile:model", "test:model1"},
 	}
 	for _, tt := range tests {
 		t.Run(tt.description+": /v1/chat/completions", func(t *testing.T) {
 			reqBody := fmt.Sprintf(`{"model":"%s"}`, tt.requestedModel)
 			req := httptest.NewRequest("POST", "/v1/chat/completions", bytes.NewBufferString(reqBody))
 			w := httptest.NewRecorder()
 			proxy.HandlerFunc(w, req)
 			assert.Equal(t, http.StatusOK, w.Code)
 			assert.Contains(t, w.Body.String(), upstreamModelName)
 		})
 	}
 	for _, tt := range tests {
 		t.Run(tt.description+": /v1/audio/transcriptions", func(t *testing.T) {
 			// Create a buffer with multipart form data
 			var b bytes.Buffer
 			w := multipart.NewWriter(&b)
 			// Add the model field
 			fw, err := w.CreateFormField("model")
 			assert.NoError(t, err)
 			_, err = fw.Write([]byte(tt.requestedModel))
 			assert.NoError(t, err)
 			// Add a file field
 			fw, err = w.CreateFormFile("file", "test.mp3")
 			assert.NoError(t, err)
 			_, err = fw.Write([]byte("test"))
 			assert.NoError(t, err)
 			w.Close()
 			// Create the request with the multipart form data
 			req := httptest.NewRequest("POST", "/v1/audio/transcriptions", &b)
 			req.Header.Set("Content-Type", w.FormDataContentType())
 			rec := httptest.NewRecorder()
 			proxy.HandlerFunc(rec, req)
 			// Verify the response
 			assert.Equal(t, http.StatusOK, rec.Code)
 			var response map[string]string
 			err = json.Unmarshal(rec.Body.Bytes(), &response)
 			assert.NoError(t, err)
 			assert.Equal(t, upstreamModelName, response["model"])
 		})
 	}
 }