Routes may share one lease (affinity = "route") and skip crossbar's queue (queue = false)

route.go gains Route.Affinity/Queue with PerRoute(), Queues() and affinity
validation (checkRoutes moved here; config.go calls it once). limiter.Track
counts a request without holding or refusing it; a release hands the slot to a
waiter only while in flight <= parallel. The proxy leases a PerRoute() route
under an empty fingerprint (the row keeps the real one) and uses Track when
Queues() is false.

Implemented by Ornith (OpenCode); owner review removed a release-on-first-flush
workaround for a race in the owner's given test (see implementer log).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 19:33:46 -07:00
co-authored by Claude Opus 5.5
parent 33fa61bedb
commit aea2eeae2c
8 changed files with 486 additions and 85 deletions
+180
View File
@@ -0,0 +1,180 @@
package proxy_test
// v2.3 task 02: affinity = "route" puts every request on the route (every conversation, every
// control call) on one lease, so one host; queue = false counts the route's requests on the host
// without ever holding or refusing them, because the client pins its own llama-server slot and
// the server's queue is the one that must show it.
import (
"bufio"
"net/http"
"strings"
"sync"
"testing"
"time"
)
const affinityHosts = `
listen = "127.0.0.1:1"
queue_max = 0
lease_idle = "30m"
[hosts.alpha]
base_url = %q
weight = 1.0
models = { "shared" = { parallel = 1 } }
[hosts.beta]
base_url = %q
weight = 1.0
models = { "shared" = { parallel = 1 } }
[routes.r]
hosts = ["alpha", "beta"]
default_model = "shared"
[routes.bm]
hosts = ["alpha", "beta"]
default_model = "shared"
affinity = "route"
queue = false
[routes."agent-*"]
hosts = ["alpha", "beta"]
default_model = "shared"
affinity = "route"
`
func TestRouteAffinityPutsEverythingOnOneHost(t *testing.T) {
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
r := newRig(t, affinityHosts, alpha, beta)
seen := map[string]int{}
note := func(what string, resp *http.Response) {
body := drain(resp)
if resp.StatusCode != 200 {
t.Fatalf("%s: %d %s", what, resp.StatusCode, body)
}
seen[resp.Header.Get("X-Crossbar-Host")]++
}
// Different conversations (different fingerprints), then control calls without any.
for id := 1; id <= 4; id++ {
note("chat", r.do(http.MethodPost, "/bm/v1/chat/completions", conversation(id, 1)))
}
note("slots", r.do(http.MethodGet, "/bm/slots?model=shared", ""))
note("props", r.do(http.MethodGet, "/bm/props?model=shared", ""))
note("control", r.do(http.MethodPost, "/bm/v1/chat/completions/control", `{"id":"chatcmpl-1","action":"reasoning_end","model":"shared"}`))
if len(seen) != 1 {
t.Fatalf("route-affinity requests spread over %v, want one host", seen)
}
// Templated concrete routes each get their own route lease, and each is internally sticky.
for _, route := range []string{"agent-a", "agent-b", "agent-c"} {
hosts := map[string]bool{}
for id := 1; id <= 3; id++ {
resp := r.do(http.MethodPost, "/"+route+"/v1/chat/completions", conversation(id, 1))
drain(resp)
hosts[resp.Header.Get("X-Crossbar-Host")] = true
}
if len(hosts) != 1 {
t.Errorf("%s spread over %v, want one host", route, hosts)
}
}
}
func TestQueueFalseNeitherHoldsNorRefuses(t *testing.T) {
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
alpha.delay, beta.delay = 400*time.Millisecond, 400*time.Millisecond
r := newRig(t, affinityHosts, alpha, beta)
// parallel = 1 and queue_max = 0: a queueing route would refuse the second and third.
var wg sync.WaitGroup
codes := make(chan int, 3)
start := time.Now()
for id := 1; id <= 3; id++ {
wg.Add(1)
go func(id int) {
defer wg.Done()
resp := r.do(http.MethodPost, "/bm/v1/chat/completions", conversation(id, 1))
drain(resp)
codes <- resp.StatusCode
}(id)
}
// While they run, the host carries all three and a queueing route sees it full.
var host string
waitUntil(t, func() bool {
for _, h := range []string{"alpha", "beta"} {
if r.lim.InFlight(h, "shared") == 3 {
host = h
return true
}
}
return false
})
if n := r.lim.FreeSlots(host); n != 0 {
t.Errorf("free slots on %s = %d while bm runs three, want 0", host, n)
}
wg.Wait()
close(codes)
for c := range codes {
if c != 200 {
t.Errorf("queue = false request: %d, want 200", c)
}
}
// Concurrent, not serialised behind one slot: three 400 ms answers well under 1.2 s.
if d := time.Since(start); d > 1100*time.Millisecond {
t.Errorf("three queue = false requests took %v; they were held", d)
}
// The slot is given back just after the answer is sent (a deferred release), so wait for it.
waitUntil(t, func() bool { return r.lim.InFlight(host, "shared") == 0 })
// Accounting is unchanged: each chat is still a row.
waitUntil(t, func() bool { return r.rows("bm") == 3 })
}
// The default is unchanged: two conversations on a conversation-affinity route may land on
// different hosts (they start where there is most room).
func TestConversationAffinityStillSpreads(t *testing.T) {
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
alpha.delay, beta.delay = 300*time.Millisecond, 300*time.Millisecond
r := newRig(t, affinityHosts, alpha, beta)
var wg sync.WaitGroup
var mu sync.Mutex
hosts := map[string]bool{}
for id := 1; id <= 2; id++ {
wg.Add(1)
go func(id int) {
defer wg.Done()
resp := r.do(http.MethodPost, "/r/v1/chat/completions", conversation(id, 1))
drain(resp)
mu.Lock()
hosts[resp.Header.Get("X-Crossbar-Host")] = true
mu.Unlock()
}(id)
time.Sleep(50 * time.Millisecond) // let the first take its slot so the second sees one host full
}
wg.Wait()
if len(hosts) != 2 {
t.Errorf("two concurrent conversations on route r used %v, want both hosts", hosts)
}
}
// A request counts against its host for as long as its answer is streaming, not only until the
// first byte: a slot (queueing route) or a tracked place (queue = false) is given back when the
// stream ends.
func TestLoadIsHeldForTheWholeStream(t *testing.T) {
for _, route := range []string{"r", "bm"} {
t.Run(route, func(t *testing.T) {
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
r := newRig(t, affinityHosts, alpha, beta)
body := strings.Replace(conversation(1, 1), `"stream":false`, `"stream":true`, 1)
resp := r.do(http.MethodPost, "/"+route+"/v1/chat/completions", body)
defer resp.Body.Close()
host := resp.Header.Get("X-Crossbar-Host")
line, err := bufio.NewReader(resp.Body).ReadString('\n')
if err != nil || !strings.HasPrefix(line, "data:") {
t.Fatalf("first line %q, err %v", line, err)
}
// The first chunk is here; the upstream sends more for another ~30 ms.
if n := r.lim.InFlight(host, "shared"); n != 1 {
t.Errorf("in flight on %s after the first chunk = %d, want 1 (released before the stream ended)", host, n)
}
drain(resp)
waitUntil(t, func() bool { return r.lim.InFlight(host, "shared") == 0 })
})
}
}
+36 -21
View File
@@ -217,6 +217,12 @@ func (p *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
}
model = resolveModel(model, r, routeCfg)
fp := fingerprint.Of(body)
// A route-affinity route puts every request (chat or control) on one lease per model, so the
// lease key's fingerprint is "" for all of them; the real fingerprint is kept for the row below.
leaseFP := fp
if routeCfg.PerRoute() {
leaseFP = ""
}
started := time.Now()
// v0 compatibility path: no lease table, no limiter, no recording.
@@ -231,14 +237,14 @@ func (p *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
}
// Lease. The route's ordered host list is the candidate set.
host, reused, err := p.leases.Acquire(lease.Key{Route: route, FP: fp, Model: model}, routeCfg.Hosts, time.Now())
host, reused, err := p.leases.Acquire(lease.Key{Route: route, FP: leaseFP, Model: model}, routeCfg.Hosts, time.Now())
if err != nil {
switch {
case errors.Is(err, lease.ErrNoHost):
// No host healthy. Ask a waker to rouse a sleeping one; it answers
// (served or 503) when it has had a turn, else falls through to the
// plain 503.
if p.waker != nil && p.wakeOnErrNoHost(w, r, route, routeCfg, rest, model, fp, started, lease.Key{Route: route, FP: fp, Model: model}) {
if p.waker != nil && p.wakeOnErrNoHost(w, r, route, routeCfg, rest, model, fp, started, lease.Key{Route: route, FP: leaseFP, Model: model}) {
return
}
p.writeError(w, http.StatusServiceUnavailable, "no healthy host")
@@ -265,10 +271,30 @@ func (p *Handler) serveLeased(w http.ResponseWriter, r *http.Request, route stri
p.forward(w, r, route, host, leaseState(reused), rest, fp, model, started, 0, 0, "", true)
return
}
// Slot. A full queue is a 503; a context done while waiting means the client left.
release, waited, err := p.lim.Acquire(r.Context(), host, model)
if err != nil {
if errors.Is(err, limiter.ErrQueueFull) {
// Slot or track. A queue = false route leaves queueing to the client's own
// llama-server slot: crossbar only counts the request on the host, never
// holding it or refusing it.
var waited time.Duration
var release func()
if routeCfg.Queues() {
var err error
release, waited, err = p.lim.Acquire(r.Context(), host, model)
if err != nil {
if errors.Is(err, limiter.ErrQueueFull) {
p.writeRecord(store.Request{
Route: route,
FP: fp,
Model: model,
Host: host,
Started: started,
TotalMs: time.Since(started).Milliseconds(),
Status: http.StatusServiceUnavailable,
Err: "queue full",
})
p.writeError(w, http.StatusServiceUnavailable, "queue full")
return
}
p.log.Warn("request", "route", route, "host", host, "method", r.Method, "path", rest, "status", 499)
p.writeRecord(store.Request{
Route: route,
FP: fp,
@@ -276,24 +302,13 @@ func (p *Handler) serveLeased(w http.ResponseWriter, r *http.Request, route stri
Host: host,
Started: started,
TotalMs: time.Since(started).Milliseconds(),
Status: http.StatusServiceUnavailable,
Err: "queue full",
Status: 499,
Err: "client cancelled while queued",
})
p.writeError(w, http.StatusServiceUnavailable, "queue full")
return
}
p.log.Warn("request", "route", route, "host", host, "method", r.Method, "path", rest, "status", 499)
p.writeRecord(store.Request{
Route: route,
FP: fp,
Model: model,
Host: host,
Started: started,
TotalMs: time.Since(started).Milliseconds(),
Status: 499,
Err: "client cancelled while queued",
})
return
} else {
release = p.lim.Track(host, model)
}
defer release()