Context refusal in llama-server's exceed_context_size_error shape; README for v2.3

Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
2026-09-25 20:15:49 -07:00
parent fa4d06170f
commit 7a12ddcf5a
5 changed files with 106 additions and 22 deletions
+22 -7
View File
@@ -155,10 +155,21 @@ func largestSlotCtx(hosts []string, h Health, model string) int {
return best
}
// refuseCtx answers the 400 the guard's rule 4: the JSON body carries the
// estimate and the largest available per-slot context, plus the error text. It
// records the accounting row (status 400, Err "prompt too large") and never
// marks the host down.
// ctxErrorBody is llama-server's shape for a prompt that exceeds a host's
// context. A client that already handles the server's own overflow error keys
// on error.type and so recognises crossbar's refusal too. See refuseCtx.
type ctxErrorBody struct {
Code int `json:"code"`
Type string `json:"type"`
Message string `json:"message"`
NPromptTokens int `json:"n_prompt_tokens"`
NCtx int `json:"n_ctx"`
}
// refuseCtx answers the 400 the guard's rule 4: the body is llama-server's
// exceed_context_size_error, with the estimate as n_prompt_tokens and the
// largest available per-slot context as n_ctx. It records the accounting row
// (status 400, Err "prompt too large") and never marks the host down.
func (p *Handler) refuseCtx(w http.ResponseWriter, host, route, model, fp string, started time.Time, estimate, maxSlot int) {
p.writeRecord(store.Request{
Route: route,
@@ -173,9 +184,13 @@ func (p *Handler) refuseCtx(w http.ResponseWriter, host, route, model, fp string
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusBadRequest)
_ = json.NewEncoder(w).Encode(map[string]any{
"error": "prompt too large",
"estimate": estimate,
"max": maxSlot,
"error": ctxErrorBody{
Code: http.StatusBadRequest,
Type: "exceed_context_size_error",
Message: "prompt too large",
NPromptTokens: estimate,
NCtx: maxSlot,
},
})
}
+8 -3
View File
@@ -95,11 +95,16 @@ func TestRouterUnloadedModelIsNotACandidate(t *testing.T) {
if big.hits.Load() != 0 {
t.Errorf("big served %d requests for a model it does not have loaded", big.hits.Load())
}
var e map[string]any
// v2.3: the refusal is llama-server's exceed_context_size_error shape; n_ctx is what "max" was.
var e struct {
Error struct {
NCtx float64 `json:"n_ctx"`
} `json:"error"`
}
if err := json.Unmarshal([]byte(body), &e); err != nil {
t.Fatalf("body %q is not JSON: %v", body, err)
}
if max, _ := e["max"].(float64); max != 4096 {
t.Errorf("max = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e["max"])
if e.Error.NCtx != 4096 {
t.Errorf("error.n_ctx = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e.Error.NCtx)
}
}
+23 -7
View File
@@ -84,15 +84,31 @@ func TestOversizedPromptWithNoFitIs400(t *testing.T) {
if resp.StatusCode != http.StatusBadRequest {
t.Fatalf("status %d body %s, want 400", resp.StatusCode, body)
}
var e map[string]any
if err := json.Unmarshal([]byte(body), &e); err != nil || e["error"] != "prompt too large" {
t.Fatalf("body = %s, want error 'prompt too large'", body)
// v2.3: llama-server's own shape for this error, so a client handles crossbar's refusal the
// way it handles the server's (Boxmaker keys on error.type; the error JSON must come first).
if !strings.HasPrefix(body, `{"error":`) {
t.Errorf("body must start with the error object: %s", body)
}
if est, _ := e["estimate"].(float64); est < 8000 || est > 13000 {
t.Errorf("estimate = %v, want roughly 10000 tokens", e["estimate"])
var e struct {
Error struct {
Code int `json:"code"`
Type string `json:"type"`
Message string `json:"message"`
NPromptTokens float64 `json:"n_prompt_tokens"`
NCtx float64 `json:"n_ctx"`
} `json:"error"`
}
if max, _ := e["max"].(float64); max != 4096 {
t.Errorf("max = %v, want the largest per-slot context among the route's hosts (4096)", e["max"])
if err := json.Unmarshal([]byte(body), &e); err != nil || e.Error.Code != 400 || e.Error.Type != "exceed_context_size_error" || e.Error.Message != "prompt too large" {
t.Fatalf("body = %s, want {\"error\":{\"code\":400,\"type\":\"exceed_context_size_error\",\"message\":\"prompt too large\",…}}", body)
}
if est := e.Error.NPromptTokens; est < 8000 || est > 13000 {
t.Errorf("n_prompt_tokens = %v, want roughly 10000 tokens", est)
}
if max := e.Error.NCtx; max != 4096 {
t.Errorf("n_ctx = %v, want the largest per-slot context among the route's hosts (4096)", max)
}
if ct := resp.Header.Get("Content-Type"); !strings.HasPrefix(ct, "application/json") {
t.Errorf("Content-Type = %q, want application/json", ct)
}
if small.hits.Load()+tiny.hits.Load() != 0 {
t.Errorf("a refused prompt must not reach any upstream")