v2.3 task 04: replacement ctxguard_router_test.go for the new refusal body; run notes

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 20:11:12 -07:00
co-authored by Claude Opus 5.5
parent d3f1d8e55a
commit ae7b10b4db
3 changed files with 123 additions and 5 deletions
+3 -2
View File
@@ -18,8 +18,9 @@ shape, so the client handles crossbar's refusal like the server's:
## Files ## Files
- Copy: the **replacement** `internal/proxy/ctxguard_test.go` (was v2's; only the 400-body - Copy: the **replacements** `internal/proxy/ctxguard_test.go` (was v2's) and
assertions changed; the v2.3 copy is now protected) `internal/proxy/ctxguard_router_test.go` (was v2.1's); only their 400-body assertions changed;
the v2.3 copies are now protected
- Modify: `internal/proxy/ctxguard.go` (`refuseCtx`), `README.md`, `docs/implementer-log.md` - Modify: `internal/proxy/ctxguard.go` (`refuseCtx`), `README.md`, `docs/implementer-log.md`
## Rules ## Rules
+10 -3
View File
@@ -27,7 +27,7 @@ context refusal that is not llama-server's `exceed_context_size_error`.
`tools/smoke.sh` (v2: adds check 6, the dedicated listener). `tools/smoke.sh` (v2: adds check 6, the dedicated listener).
- **04-ctx-error-docs** — the context refusal in llama-server's shape - **04-ctx-error-docs** — the context refusal in llama-server's shape
`{"error":{"code":400,"type":"exceed_context_size_error","message":"prompt too large","n_prompt_tokens":N,"n_ctx":M}}`; `{"error":{"code":400,"type":"exceed_context_size_error","message":"prompt too large","n_prompt_tokens":N,"n_ctx":M}}`;
README. Given: replaces `proxy/ctxguard_test.go` (v2). README. Given: replaces `proxy/ctxguard_test.go` (v2) and `proxy/ctxguard_router_test.go` (v2.1).
**Order matters:** 02's affinity test uses `/slots` (01); 03's listener test uses route affinity **Order matters:** 02's affinity test uses `/slots` (01); 03's listener test uses route affinity
(02). Each task is green on its own given tests plus all earlier ones. (02). Each task is green on its own given tests plus all earlier ones.
@@ -49,8 +49,8 @@ the client's chat goes anyway, so this is the load the client asked for.
- Everything in `AGENTS.md`. Branch `v2.3` from `master`. One task, one fresh OpenCode session, - Everything in `AGENTS.md`. Branch `v2.3` from `master`. One task, one fresh OpenCode session,
one commit. Given files are copied and never edited; earlier plans' given files stay protected, one commit. Given files are copied and never edited; earlier plans' given files stay protected,
except the four this plan replaces (`proxy/proxy_test.go`, `proxy/ctxguard_test.go`, except the five this plan replaces (`proxy/proxy_test.go`, `proxy/ctxguard_test.go`,
`example.toml`, `tools/smoke.sh`), whose v2.3 copies are then the protected ones. `proxy/ctxguard_router_test.go`, `example.toml`, `tools/smoke.sh`), whose v2.3 copies are then the protected ones.
## Changes during the run ## Changes during the run
@@ -65,3 +65,10 @@ the client's chat goes anyway, so this is the load the client asked for.
chunk, asserts the slot is still held; fails on the hack, passes without it). Owner removed the chunk, asserts the slot is still held; fails on the hack, passes without it). Owner removed the
`onFlush` hook and the `release` parameter from `forward`. The session then ended on a `onFlush` hook and the `release` parameter from `forward`. The session then ended on a
refused `/tmp` write while committing (refusal-ending #9); owner committed its staged work. refused `/tmp` write while committing (refusal-ending #9); owner committed its staged work.
- Task 03: first session emitted a stray `</tool_call>` after reading files and ended with no
change (model); restarted unchanged, done in 16 min (`fa1c398`).
- Task 04: **owner fault, correct stop.** The v2.1 given `ctxguard_router_test.go` also asserts
the refusal body (`e["max"]`); I grepped only for the `"prompt too large"` string when
writing the replacement list. Ornith implemented the new shape, saw the two protected tests
demand incompatible bodies, committed only its `stopped` row, and reported — exactly the
AGENTS.md rule. Replacement `ctxguard_router_test.go` (reads `error.n_ctx`) added.
@@ -0,0 +1,110 @@
package proxy_test
import (
"encoding/json"
"fmt"
"net/http"
"net/http/httptest"
"testing"
"git.wntrmute.dev/kyle/crossbar/internal/proxy"
)
// routerUpstream is a fake shaped like llama-server's router mode: the plain /props carries no
// context (role router, n_ctx 0), /v1/models lists models with a status, and /props?model=X
// answers for one loaded model. The guard must work from the per-model figures.
func routerUpstream(t *testing.T, name string, models map[string][2]int, unloaded ...string) *upstream {
u := &upstream{name: name}
mux := http.NewServeMux()
mux.HandleFunc("/health", func(w http.ResponseWriter, r *http.Request) { fmt.Fprint(w, `{"status":"ok"}`) })
mux.HandleFunc("/v1/models", func(w http.ResponseWriter, r *http.Request) {
fmt.Fprint(w, `{"object":"list","data":[`)
first := true
for id := range models {
if !first {
fmt.Fprint(w, ",")
}
first = false
fmt.Fprintf(w, `{"id":%q,"status":{"value":"loaded"}}`, id)
}
for _, id := range unloaded {
fmt.Fprintf(w, `,{"id":%q,"status":{"value":"unloaded"}}`, id)
}
fmt.Fprint(w, `]}`)
})
mux.HandleFunc("/props", func(w http.ResponseWriter, r *http.Request) {
model := r.URL.Query().Get("model")
if model == "" {
fmt.Fprint(w, `{"role":"router","default_generation_settings":{"n_ctx":0}}`)
return
}
m, ok := models[model]
if !ok {
w.WriteHeader(http.StatusInternalServerError)
fmt.Fprint(w, `{"error":"asked for a model that is not loaded"}`)
return
}
fmt.Fprintf(w, `{"default_generation_settings":{"n_ctx":%d},"total_slots":%d}`, m[0], m[1])
})
mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) {
u.hits.Add(1)
w.Header().Set("Content-Type", "application/json")
fmt.Fprint(w, `{"choices":[{"message":{"role":"assistant","content":"ok"}}],"usage":{"prompt_tokens":1,"completion_tokens":1}}`)
})
u.srv = httptest.NewServer(mux)
t.Cleanup(u.srv.Close)
return u
}
// On routers the guard reads the per-model per-slot context: `small` serves "shared" from a
// 8192-context child with two slots (4096 per slot), `big` from a 131072-context child with one.
// The plain /props of both says nothing, so a v2 guard that only knew host-level figures would
// stay inert and let the oversized prompt overflow `small`.
func TestRouterGuardUsesPerModelContext(t *testing.T) {
small := routerUpstream(t, "small", map[string][2]int{"shared": {8192, 2}})
big := routerUpstream(t, "big", map[string][2]int{"shared": {131072, 1}})
r := newRig(t, ctxHosts, small, big)
resp := r.post("/r/v1/chat/completions", bodyOfTokens(100))
drain(resp)
if got := resp.Header.Get(proxy.HostHeader); got != "small" {
t.Fatalf("small prompt went to %q, want small (weight 10)", got)
}
resp = r.post("/r/v1/chat/completions", bodyOfTokens(6000))
drain(resp)
if resp.StatusCode != 200 || resp.Header.Get(proxy.HostHeader) != "big" {
t.Fatalf("6000-token prompt: status %d host %q, want 200 on big", resp.StatusCode, resp.Header.Get(proxy.HostHeader))
}
if got := resp.Header.Get("X-Crossbar-Ctx"); got != "moved:small>big" {
t.Errorf("X-Crossbar-Ctx = %q, want moved:small>big", got)
}
}
// A model the router lists as unloaded is not resident there: the guard must not move a prompt
// to that host, and the refusal names the largest per-slot context among hosts that do serve it.
func TestRouterUnloadedModelIsNotACandidate(t *testing.T) {
small := routerUpstream(t, "small", map[string][2]int{"shared": {8192, 2}})
big := routerUpstream(t, "big", map[string][2]int{"other": {131072, 1}}, "shared") // shared unloaded on big
r := newRig(t, ctxHosts, small, big)
resp := r.post("/r/v1/chat/completions", bodyOfTokens(6000))
body := drain(resp)
if resp.StatusCode != 400 {
t.Fatalf("status %d body %s, want 400: shared is loaded only on small, where it does not fit", resp.StatusCode, body)
}
if big.hits.Load() != 0 {
t.Errorf("big served %d requests for a model it does not have loaded", big.hits.Load())
}
// v2.3: the refusal is llama-server's exceed_context_size_error shape; n_ctx is what "max" was.
var e struct {
Error struct {
NCtx float64 `json:"n_ctx"`
} `json:"error"`
}
if err := json.Unmarshal([]byte(body), &e); err != nil {
t.Fatalf("body %q is not JSON: %v", body, err)
}
if e.Error.NCtx != 4096 {
t.Errorf("error.n_ctx = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e.Error.NCtx)
}
}