Routes may share one lease (affinity = "route") and skip crossbar's queue (queue = false)
route.go gains Route.Affinity/Queue with PerRoute(), Queues() and affinity validation (checkRoutes moved here; config.go calls it once). limiter.Track counts a request without holding or refusing it; a release hands the slot to a waiter only while in flight <= parallel. The proxy leases a PerRoute() route under an empty fingerprint (the row keeps the real one) and uses Track when Queues() is false. Implemented by Ornith (OpenCode); owner review removed a release-on-first-flush workaround for a race in the owner's given test (see implementer log). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,180 @@
|
||||
package proxy_test
|
||||
|
||||
// v2.3 task 02: affinity = "route" puts every request on the route (every conversation, every
|
||||
// control call) on one lease, so one host; queue = false counts the route's requests on the host
|
||||
// without ever holding or refusing them, because the client pins its own llama-server slot and
|
||||
// the server's queue is the one that must show it.
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
const affinityHosts = `
|
||||
listen = "127.0.0.1:1"
|
||||
queue_max = 0
|
||||
lease_idle = "30m"
|
||||
[hosts.alpha]
|
||||
base_url = %q
|
||||
weight = 1.0
|
||||
models = { "shared" = { parallel = 1 } }
|
||||
[hosts.beta]
|
||||
base_url = %q
|
||||
weight = 1.0
|
||||
models = { "shared" = { parallel = 1 } }
|
||||
[routes.r]
|
||||
hosts = ["alpha", "beta"]
|
||||
default_model = "shared"
|
||||
[routes.bm]
|
||||
hosts = ["alpha", "beta"]
|
||||
default_model = "shared"
|
||||
affinity = "route"
|
||||
queue = false
|
||||
[routes."agent-*"]
|
||||
hosts = ["alpha", "beta"]
|
||||
default_model = "shared"
|
||||
affinity = "route"
|
||||
`
|
||||
|
||||
func TestRouteAffinityPutsEverythingOnOneHost(t *testing.T) {
|
||||
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
|
||||
r := newRig(t, affinityHosts, alpha, beta)
|
||||
|
||||
seen := map[string]int{}
|
||||
note := func(what string, resp *http.Response) {
|
||||
body := drain(resp)
|
||||
if resp.StatusCode != 200 {
|
||||
t.Fatalf("%s: %d %s", what, resp.StatusCode, body)
|
||||
}
|
||||
seen[resp.Header.Get("X-Crossbar-Host")]++
|
||||
}
|
||||
// Different conversations (different fingerprints), then control calls without any.
|
||||
for id := 1; id <= 4; id++ {
|
||||
note("chat", r.do(http.MethodPost, "/bm/v1/chat/completions", conversation(id, 1)))
|
||||
}
|
||||
note("slots", r.do(http.MethodGet, "/bm/slots?model=shared", ""))
|
||||
note("props", r.do(http.MethodGet, "/bm/props?model=shared", ""))
|
||||
note("control", r.do(http.MethodPost, "/bm/v1/chat/completions/control", `{"id":"chatcmpl-1","action":"reasoning_end","model":"shared"}`))
|
||||
if len(seen) != 1 {
|
||||
t.Fatalf("route-affinity requests spread over %v, want one host", seen)
|
||||
}
|
||||
|
||||
// Templated concrete routes each get their own route lease, and each is internally sticky.
|
||||
for _, route := range []string{"agent-a", "agent-b", "agent-c"} {
|
||||
hosts := map[string]bool{}
|
||||
for id := 1; id <= 3; id++ {
|
||||
resp := r.do(http.MethodPost, "/"+route+"/v1/chat/completions", conversation(id, 1))
|
||||
drain(resp)
|
||||
hosts[resp.Header.Get("X-Crossbar-Host")] = true
|
||||
}
|
||||
if len(hosts) != 1 {
|
||||
t.Errorf("%s spread over %v, want one host", route, hosts)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueFalseNeitherHoldsNorRefuses(t *testing.T) {
|
||||
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
|
||||
alpha.delay, beta.delay = 400*time.Millisecond, 400*time.Millisecond
|
||||
r := newRig(t, affinityHosts, alpha, beta)
|
||||
|
||||
// parallel = 1 and queue_max = 0: a queueing route would refuse the second and third.
|
||||
var wg sync.WaitGroup
|
||||
codes := make(chan int, 3)
|
||||
start := time.Now()
|
||||
for id := 1; id <= 3; id++ {
|
||||
wg.Add(1)
|
||||
go func(id int) {
|
||||
defer wg.Done()
|
||||
resp := r.do(http.MethodPost, "/bm/v1/chat/completions", conversation(id, 1))
|
||||
drain(resp)
|
||||
codes <- resp.StatusCode
|
||||
}(id)
|
||||
}
|
||||
// While they run, the host carries all three and a queueing route sees it full.
|
||||
var host string
|
||||
waitUntil(t, func() bool {
|
||||
for _, h := range []string{"alpha", "beta"} {
|
||||
if r.lim.InFlight(h, "shared") == 3 {
|
||||
host = h
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
})
|
||||
if n := r.lim.FreeSlots(host); n != 0 {
|
||||
t.Errorf("free slots on %s = %d while bm runs three, want 0", host, n)
|
||||
}
|
||||
wg.Wait()
|
||||
close(codes)
|
||||
for c := range codes {
|
||||
if c != 200 {
|
||||
t.Errorf("queue = false request: %d, want 200", c)
|
||||
}
|
||||
}
|
||||
// Concurrent, not serialised behind one slot: three 400 ms answers well under 1.2 s.
|
||||
if d := time.Since(start); d > 1100*time.Millisecond {
|
||||
t.Errorf("three queue = false requests took %v; they were held", d)
|
||||
}
|
||||
// The slot is given back just after the answer is sent (a deferred release), so wait for it.
|
||||
waitUntil(t, func() bool { return r.lim.InFlight(host, "shared") == 0 })
|
||||
// Accounting is unchanged: each chat is still a row.
|
||||
waitUntil(t, func() bool { return r.rows("bm") == 3 })
|
||||
}
|
||||
|
||||
// The default is unchanged: two conversations on a conversation-affinity route may land on
|
||||
// different hosts (they start where there is most room).
|
||||
func TestConversationAffinityStillSpreads(t *testing.T) {
|
||||
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
|
||||
alpha.delay, beta.delay = 300*time.Millisecond, 300*time.Millisecond
|
||||
r := newRig(t, affinityHosts, alpha, beta)
|
||||
var wg sync.WaitGroup
|
||||
var mu sync.Mutex
|
||||
hosts := map[string]bool{}
|
||||
for id := 1; id <= 2; id++ {
|
||||
wg.Add(1)
|
||||
go func(id int) {
|
||||
defer wg.Done()
|
||||
resp := r.do(http.MethodPost, "/r/v1/chat/completions", conversation(id, 1))
|
||||
drain(resp)
|
||||
mu.Lock()
|
||||
hosts[resp.Header.Get("X-Crossbar-Host")] = true
|
||||
mu.Unlock()
|
||||
}(id)
|
||||
time.Sleep(50 * time.Millisecond) // let the first take its slot so the second sees one host full
|
||||
}
|
||||
wg.Wait()
|
||||
if len(hosts) != 2 {
|
||||
t.Errorf("two concurrent conversations on route r used %v, want both hosts", hosts)
|
||||
}
|
||||
}
|
||||
|
||||
// A request counts against its host for as long as its answer is streaming, not only until the
|
||||
// first byte: a slot (queueing route) or a tracked place (queue = false) is given back when the
|
||||
// stream ends.
|
||||
func TestLoadIsHeldForTheWholeStream(t *testing.T) {
|
||||
for _, route := range []string{"r", "bm"} {
|
||||
t.Run(route, func(t *testing.T) {
|
||||
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
|
||||
r := newRig(t, affinityHosts, alpha, beta)
|
||||
body := strings.Replace(conversation(1, 1), `"stream":false`, `"stream":true`, 1)
|
||||
resp := r.do(http.MethodPost, "/"+route+"/v1/chat/completions", body)
|
||||
defer resp.Body.Close()
|
||||
host := resp.Header.Get("X-Crossbar-Host")
|
||||
line, err := bufio.NewReader(resp.Body).ReadString('\n')
|
||||
if err != nil || !strings.HasPrefix(line, "data:") {
|
||||
t.Fatalf("first line %q, err %v", line, err)
|
||||
}
|
||||
// The first chunk is here; the upstream sends more for another ~30 ms.
|
||||
if n := r.lim.InFlight(host, "shared"); n != 1 {
|
||||
t.Errorf("in flight on %s after the first chunk = %d, want 1 (released before the stream ended)", host, n)
|
||||
}
|
||||
drain(resp)
|
||||
waitUntil(t, func() bool { return r.lim.InFlight(host, "shared") == 0 })
|
||||
})
|
||||
}
|
||||
}
|
||||
+36
-21
@@ -217,6 +217,12 @@ func (p *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
model = resolveModel(model, r, routeCfg)
|
||||
fp := fingerprint.Of(body)
|
||||
// A route-affinity route puts every request (chat or control) on one lease per model, so the
|
||||
// lease key's fingerprint is "" for all of them; the real fingerprint is kept for the row below.
|
||||
leaseFP := fp
|
||||
if routeCfg.PerRoute() {
|
||||
leaseFP = ""
|
||||
}
|
||||
started := time.Now()
|
||||
|
||||
// v0 compatibility path: no lease table, no limiter, no recording.
|
||||
@@ -231,14 +237,14 @@ func (p *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
// Lease. The route's ordered host list is the candidate set.
|
||||
host, reused, err := p.leases.Acquire(lease.Key{Route: route, FP: fp, Model: model}, routeCfg.Hosts, time.Now())
|
||||
host, reused, err := p.leases.Acquire(lease.Key{Route: route, FP: leaseFP, Model: model}, routeCfg.Hosts, time.Now())
|
||||
if err != nil {
|
||||
switch {
|
||||
case errors.Is(err, lease.ErrNoHost):
|
||||
// No host healthy. Ask a waker to rouse a sleeping one; it answers
|
||||
// (served or 503) when it has had a turn, else falls through to the
|
||||
// plain 503.
|
||||
if p.waker != nil && p.wakeOnErrNoHost(w, r, route, routeCfg, rest, model, fp, started, lease.Key{Route: route, FP: fp, Model: model}) {
|
||||
if p.waker != nil && p.wakeOnErrNoHost(w, r, route, routeCfg, rest, model, fp, started, lease.Key{Route: route, FP: leaseFP, Model: model}) {
|
||||
return
|
||||
}
|
||||
p.writeError(w, http.StatusServiceUnavailable, "no healthy host")
|
||||
@@ -265,10 +271,30 @@ func (p *Handler) serveLeased(w http.ResponseWriter, r *http.Request, route stri
|
||||
p.forward(w, r, route, host, leaseState(reused), rest, fp, model, started, 0, 0, "", true)
|
||||
return
|
||||
}
|
||||
// Slot. A full queue is a 503; a context done while waiting means the client left.
|
||||
release, waited, err := p.lim.Acquire(r.Context(), host, model)
|
||||
if err != nil {
|
||||
if errors.Is(err, limiter.ErrQueueFull) {
|
||||
// Slot or track. A queue = false route leaves queueing to the client's own
|
||||
// llama-server slot: crossbar only counts the request on the host, never
|
||||
// holding it or refusing it.
|
||||
var waited time.Duration
|
||||
var release func()
|
||||
if routeCfg.Queues() {
|
||||
var err error
|
||||
release, waited, err = p.lim.Acquire(r.Context(), host, model)
|
||||
if err != nil {
|
||||
if errors.Is(err, limiter.ErrQueueFull) {
|
||||
p.writeRecord(store.Request{
|
||||
Route: route,
|
||||
FP: fp,
|
||||
Model: model,
|
||||
Host: host,
|
||||
Started: started,
|
||||
TotalMs: time.Since(started).Milliseconds(),
|
||||
Status: http.StatusServiceUnavailable,
|
||||
Err: "queue full",
|
||||
})
|
||||
p.writeError(w, http.StatusServiceUnavailable, "queue full")
|
||||
return
|
||||
}
|
||||
p.log.Warn("request", "route", route, "host", host, "method", r.Method, "path", rest, "status", 499)
|
||||
p.writeRecord(store.Request{
|
||||
Route: route,
|
||||
FP: fp,
|
||||
@@ -276,24 +302,13 @@ func (p *Handler) serveLeased(w http.ResponseWriter, r *http.Request, route stri
|
||||
Host: host,
|
||||
Started: started,
|
||||
TotalMs: time.Since(started).Milliseconds(),
|
||||
Status: http.StatusServiceUnavailable,
|
||||
Err: "queue full",
|
||||
Status: 499,
|
||||
Err: "client cancelled while queued",
|
||||
})
|
||||
p.writeError(w, http.StatusServiceUnavailable, "queue full")
|
||||
return
|
||||
}
|
||||
p.log.Warn("request", "route", route, "host", host, "method", r.Method, "path", rest, "status", 499)
|
||||
p.writeRecord(store.Request{
|
||||
Route: route,
|
||||
FP: fp,
|
||||
Model: model,
|
||||
Host: host,
|
||||
Started: started,
|
||||
TotalMs: time.Since(started).Milliseconds(),
|
||||
Status: 499,
|
||||
Err: "client cancelled while queued",
|
||||
})
|
||||
return
|
||||
} else {
|
||||
release = p.lim.Track(host, model)
|
||||
}
|
||||
defer release()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user