Context refusal in llama-server's exceed_context_size_error shape; README for v2.3

Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
2026-09-25 20:15:49 -07:00
parent fa4d06170f
commit 7a12ddcf5a
5 changed files with 106 additions and 22 deletions
+8 -3
View File
@@ -95,11 +95,16 @@ func TestRouterUnloadedModelIsNotACandidate(t *testing.T) {
if big.hits.Load() != 0 {
t.Errorf("big served %d requests for a model it does not have loaded", big.hits.Load())
}
var e map[string]any
// v2.3: the refusal is llama-server's exceed_context_size_error shape; n_ctx is what "max" was.
var e struct {
Error struct {
NCtx float64 `json:"n_ctx"`
} `json:"error"`
}
if err := json.Unmarshal([]byte(body), &e); err != nil {
t.Fatalf("body %q is not JSON: %v", body, err)
}
if max, _ := e["max"].(float64); max != 4096 {
t.Errorf("max = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e["max"])
if e.Error.NCtx != 4096 {
t.Errorf("error.n_ctx = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e.Error.NCtx)
}
}