Context refusal in llama-server's exceed_context_size_error shape; README for v2.3

Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
2026-09-25 20:15:49 -07:00
parent fa4d06170f
commit 7a12ddcf5a
5 changed files with 106 additions and 22 deletions
+22 -7
View File
@@ -155,10 +155,21 @@ func largestSlotCtx(hosts []string, h Health, model string) int {
return best
}
// refuseCtx answers the 400 the guard's rule 4: the JSON body carries the
// estimate and the largest available per-slot context, plus the error text. It
// records the accounting row (status 400, Err "prompt too large") and never
// marks the host down.
// ctxErrorBody is llama-server's shape for a prompt that exceeds a host's
// context. A client that already handles the server's own overflow error keys
// on error.type and so recognises crossbar's refusal too. See refuseCtx.
type ctxErrorBody struct {
Code int `json:"code"`
Type string `json:"type"`
Message string `json:"message"`
NPromptTokens int `json:"n_prompt_tokens"`
NCtx int `json:"n_ctx"`
}
// refuseCtx answers the 400 the guard's rule 4: the body is llama-server's
// exceed_context_size_error, with the estimate as n_prompt_tokens and the
// largest available per-slot context as n_ctx. It records the accounting row
// (status 400, Err "prompt too large") and never marks the host down.
func (p *Handler) refuseCtx(w http.ResponseWriter, host, route, model, fp string, started time.Time, estimate, maxSlot int) {
p.writeRecord(store.Request{
Route: route,
@@ -173,9 +184,13 @@ func (p *Handler) refuseCtx(w http.ResponseWriter, host, route, model, fp string
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusBadRequest)
_ = json.NewEncoder(w).Encode(map[string]any{
"error": "prompt too large",
"estimate": estimate,
"max": maxSlot,
"error": ctxErrorBody{
Code: http.StatusBadRequest,
Type: "exceed_context_size_error",
Message: "prompt too large",
NPromptTokens: estimate,
NCtx: maxSlot,
},
})
}