Context refusal in llama-server's exceed_context_size_error shape; README for v2.3
Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
@@ -155,10 +155,21 @@ func largestSlotCtx(hosts []string, h Health, model string) int {
|
||||
return best
|
||||
}
|
||||
|
||||
// refuseCtx answers the 400 the guard's rule 4: the JSON body carries the
|
||||
// estimate and the largest available per-slot context, plus the error text. It
|
||||
// records the accounting row (status 400, Err "prompt too large") and never
|
||||
// marks the host down.
|
||||
// ctxErrorBody is llama-server's shape for a prompt that exceeds a host's
|
||||
// context. A client that already handles the server's own overflow error keys
|
||||
// on error.type and so recognises crossbar's refusal too. See refuseCtx.
|
||||
type ctxErrorBody struct {
|
||||
Code int `json:"code"`
|
||||
Type string `json:"type"`
|
||||
Message string `json:"message"`
|
||||
NPromptTokens int `json:"n_prompt_tokens"`
|
||||
NCtx int `json:"n_ctx"`
|
||||
}
|
||||
|
||||
// refuseCtx answers the 400 the guard's rule 4: the body is llama-server's
|
||||
// exceed_context_size_error, with the estimate as n_prompt_tokens and the
|
||||
// largest available per-slot context as n_ctx. It records the accounting row
|
||||
// (status 400, Err "prompt too large") and never marks the host down.
|
||||
func (p *Handler) refuseCtx(w http.ResponseWriter, host, route, model, fp string, started time.Time, estimate, maxSlot int) {
|
||||
p.writeRecord(store.Request{
|
||||
Route: route,
|
||||
@@ -173,9 +184,13 @@ func (p *Handler) refuseCtx(w http.ResponseWriter, host, route, model, fp string
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.WriteHeader(http.StatusBadRequest)
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"error": "prompt too large",
|
||||
"estimate": estimate,
|
||||
"max": maxSlot,
|
||||
"error": ctxErrorBody{
|
||||
Code: http.StatusBadRequest,
|
||||
Type: "exceed_context_size_error",
|
||||
Message: "prompt too large",
|
||||
NPromptTokens: estimate,
|
||||
NCtx: maxSlot,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -95,11 +95,16 @@ func TestRouterUnloadedModelIsNotACandidate(t *testing.T) {
|
||||
if big.hits.Load() != 0 {
|
||||
t.Errorf("big served %d requests for a model it does not have loaded", big.hits.Load())
|
||||
}
|
||||
var e map[string]any
|
||||
// v2.3: the refusal is llama-server's exceed_context_size_error shape; n_ctx is what "max" was.
|
||||
var e struct {
|
||||
Error struct {
|
||||
NCtx float64 `json:"n_ctx"`
|
||||
} `json:"error"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(body), &e); err != nil {
|
||||
t.Fatalf("body %q is not JSON: %v", body, err)
|
||||
}
|
||||
if max, _ := e["max"].(float64); max != 4096 {
|
||||
t.Errorf("max = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e["max"])
|
||||
if e.Error.NCtx != 4096 {
|
||||
t.Errorf("error.n_ctx = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e.Error.NCtx)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -84,15 +84,31 @@ func TestOversizedPromptWithNoFitIs400(t *testing.T) {
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("status %d body %s, want 400", resp.StatusCode, body)
|
||||
}
|
||||
var e map[string]any
|
||||
if err := json.Unmarshal([]byte(body), &e); err != nil || e["error"] != "prompt too large" {
|
||||
t.Fatalf("body = %s, want error 'prompt too large'", body)
|
||||
// v2.3: llama-server's own shape for this error, so a client handles crossbar's refusal the
|
||||
// way it handles the server's (Boxmaker keys on error.type; the error JSON must come first).
|
||||
if !strings.HasPrefix(body, `{"error":`) {
|
||||
t.Errorf("body must start with the error object: %s", body)
|
||||
}
|
||||
if est, _ := e["estimate"].(float64); est < 8000 || est > 13000 {
|
||||
t.Errorf("estimate = %v, want roughly 10000 tokens", e["estimate"])
|
||||
var e struct {
|
||||
Error struct {
|
||||
Code int `json:"code"`
|
||||
Type string `json:"type"`
|
||||
Message string `json:"message"`
|
||||
NPromptTokens float64 `json:"n_prompt_tokens"`
|
||||
NCtx float64 `json:"n_ctx"`
|
||||
} `json:"error"`
|
||||
}
|
||||
if max, _ := e["max"].(float64); max != 4096 {
|
||||
t.Errorf("max = %v, want the largest per-slot context among the route's hosts (4096)", e["max"])
|
||||
if err := json.Unmarshal([]byte(body), &e); err != nil || e.Error.Code != 400 || e.Error.Type != "exceed_context_size_error" || e.Error.Message != "prompt too large" {
|
||||
t.Fatalf("body = %s, want {\"error\":{\"code\":400,\"type\":\"exceed_context_size_error\",\"message\":\"prompt too large\",…}}", body)
|
||||
}
|
||||
if est := e.Error.NPromptTokens; est < 8000 || est > 13000 {
|
||||
t.Errorf("n_prompt_tokens = %v, want roughly 10000 tokens", est)
|
||||
}
|
||||
if max := e.Error.NCtx; max != 4096 {
|
||||
t.Errorf("n_ctx = %v, want the largest per-slot context among the route's hosts (4096)", max)
|
||||
}
|
||||
if ct := resp.Header.Get("Content-Type"); !strings.HasPrefix(ct, "application/json") {
|
||||
t.Errorf("Content-Type = %q, want application/json", ct)
|
||||
}
|
||||
if small.hits.Load()+tiny.hits.Load() != 0 {
|
||||
t.Errorf("a refused prompt must not reach any upstream")
|
||||
|
||||
Reference in New Issue
Block a user