Context refusal in llama-server's exceed_context_size_error shape; README for v2.3
Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
@@ -95,11 +95,16 @@ func TestRouterUnloadedModelIsNotACandidate(t *testing.T) {
|
||||
if big.hits.Load() != 0 {
|
||||
t.Errorf("big served %d requests for a model it does not have loaded", big.hits.Load())
|
||||
}
|
||||
var e map[string]any
|
||||
// v2.3: the refusal is llama-server's exceed_context_size_error shape; n_ctx is what "max" was.
|
||||
var e struct {
|
||||
Error struct {
|
||||
NCtx float64 `json:"n_ctx"`
|
||||
} `json:"error"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(body), &e); err != nil {
|
||||
t.Fatalf("body %q is not JSON: %v", body, err)
|
||||
}
|
||||
if max, _ := e["max"].(float64); max != 4096 {
|
||||
t.Errorf("max = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e["max"])
|
||||
if e.Error.NCtx != 4096 {
|
||||
t.Errorf("error.n_ctx = %v, want 4096: the largest per-slot context among hosts that have shared loaded", e.Error.NCtx)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user