v1 plan: fix spread test config, queue test read, admin pin via lease.Candidates; note the findings
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
@@ -201,17 +201,37 @@ func TestConversationIsStickyAndLeaseHeaderTellsWhy(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// spreadHosts: beta is preferred (weight 10) until both of its "shared" slots are busy; then
|
||||
// alpha (2 free × 1) beats beta (0 free × 10), and a new conversation must start on alpha.
|
||||
const spreadHosts = `
|
||||
listen = "127.0.0.1:1"
|
||||
queue_max = 4
|
||||
lease_idle = "30m"
|
||||
[hosts.alpha]
|
||||
base_url = %q
|
||||
weight = 1.0
|
||||
models = { "shared" = { parallel = 2 } }
|
||||
[hosts.beta]
|
||||
base_url = %q
|
||||
weight = 10.0
|
||||
models = { "shared" = { parallel = 2 } }
|
||||
[routes.r]
|
||||
hosts = ["alpha", "beta"]
|
||||
default_model = "shared"
|
||||
`
|
||||
|
||||
func TestDifferentConversationsSpreadByFreeSlots(t *testing.T) {
|
||||
alpha, beta := newUpstream(t, "alpha"), newUpstream(t, "beta")
|
||||
beta.delay = 300 * time.Millisecond
|
||||
r := newRig(t, twoHosts, alpha, beta)
|
||||
beta.delay = 400 * time.Millisecond
|
||||
r := newRig(t, spreadHosts, alpha, beta)
|
||||
// Two slow conversations occupy beta's two "shared" slots…
|
||||
var wg sync.WaitGroup
|
||||
for i := 1; i <= 2; i++ {
|
||||
wg.Add(1)
|
||||
go func(i int) { defer wg.Done(); drain(r.post("/r/v1/chat/completions", conversation(i, 1))) }(i)
|
||||
time.Sleep(50 * time.Millisecond) // arrive one after the other so both pick beta (10 > 2)
|
||||
}
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
// …so a third conversation starting now is sent to alpha (beta has 0 free slots, alpha 2).
|
||||
resp := r.post("/r/v1/chat/completions", conversation(3, 1))
|
||||
drain(resp)
|
||||
@@ -219,6 +239,9 @@ func TestDifferentConversationsSpreadByFreeSlots(t *testing.T) {
|
||||
t.Errorf("third conversation went to %q, want alpha (free slots beat weight)", resp.Header.Get(proxy.HostHeader))
|
||||
}
|
||||
wg.Wait()
|
||||
if beta.hits.Load() != 2 || alpha.hits.Load() != 1 {
|
||||
t.Errorf("hits beta=%d alpha=%d, want 2 and 1", beta.hits.Load(), alpha.hits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
func TestQueueFullIs503(t *testing.T) {
|
||||
@@ -250,9 +273,18 @@ default_model = "shared"
|
||||
if got[200] != 2 || got[503] != 1 {
|
||||
t.Fatalf("status counts = %v, want two 200 and one 503", got)
|
||||
}
|
||||
rows, err := r.store.Usage(time.Time{}, store.ByRoute)
|
||||
if err != nil || len(rows) != 1 || rows[0].Requests != 3 || rows[0].Errors != 1 {
|
||||
t.Errorf("usage = %+v %v, want 3 requests, 1 error (the 503 is recorded too)", rows, err)
|
||||
// Rows are written after each response completes; allow the store a moment to catch up.
|
||||
var rows []store.UsageRow
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
rows, _ = r.store.Usage(time.Time{}, store.ByRoute)
|
||||
if len(rows) == 1 && rows[0].Requests == 3 {
|
||||
break
|
||||
}
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
}
|
||||
if len(rows) != 1 || rows[0].Requests != 3 || rows[0].Errors != 1 {
|
||||
t.Fatalf("usage = %+v, want 3 requests, 1 error (the 503 is recorded too)", rows)
|
||||
}
|
||||
if rows[0].QueuedMs <= 0 {
|
||||
t.Errorf("the queued request must record its wait: %+v", rows[0])
|
||||
|
||||
Reference in New Issue
Block a user