Smoke run for v1; README for leases, admin and accounting
Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
@@ -1,11 +1,13 @@
|
||||
// fakeupstream stands in for a llama-server router in tests and the smoke run. Do not edit.
|
||||
//
|
||||
// fakeupstream -listen 127.0.0.1:18081 -name alpha -models a,b -down-file /tmp/alpha.down
|
||||
// fakeupstream -listen 127.0.0.1:18081 -name alpha -models a,b -down-file /tmp/alpha.down -slow 0
|
||||
//
|
||||
// /health answers 503 while the down file exists, 200 otherwise. /v1/models lists -models.
|
||||
// /props answers a small JSON object. /v1/chat/completions echoes: a streamed answer of five
|
||||
// SSE chunks 200 ms apart when the body has "stream": true, one JSON answer otherwise. Every
|
||||
// response carries X-Upstream: <name>.
|
||||
// SSE chunks 200 ms apart when the body has "stream": true, then a final chunk carrying
|
||||
// "usage" and llama-server style "timings", then [DONE]; one JSON answer with usage and
|
||||
// timings otherwise. -slow adds that many milliseconds before answering (for queue tests).
|
||||
// Every response carries X-Upstream: <name>.
|
||||
package main
|
||||
|
||||
import (
|
||||
@@ -25,11 +27,14 @@ func main() {
|
||||
name := flag.String("name", "fake", "name reported in X-Upstream and answers")
|
||||
models := flag.String("models", "m", "comma-separated model ids for /v1/models")
|
||||
downFile := flag.String("down-file", "", "while this file exists, /health answers 503")
|
||||
slow := flag.Int("slow", 0, "milliseconds to wait before answering a completion")
|
||||
flag.Parse()
|
||||
|
||||
ids := strings.Split(*models, ",")
|
||||
mux := http.NewServeMux()
|
||||
stamp := func(w http.ResponseWriter) { w.Header().Set("X-Upstream", *name) }
|
||||
usage := map[string]any{"prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110}
|
||||
timings := map[string]any{"prompt_n": 100, "cache_n": 90, "predicted_n": 10, "predicted_ms": 50.0}
|
||||
|
||||
mux.HandleFunc("/health", func(w http.ResponseWriter, r *http.Request) {
|
||||
stamp(w)
|
||||
@@ -61,11 +66,12 @@ func main() {
|
||||
Stream bool `json:"stream"`
|
||||
}
|
||||
_ = json.Unmarshal(body, &req)
|
||||
time.Sleep(time.Duration(*slow) * time.Millisecond)
|
||||
if !req.Stream {
|
||||
writeJSON(w, map[string]any{
|
||||
"id": "chatcmpl-fake", "object": "chat.completion", "model": req.Model,
|
||||
"choices": []map[string]any{{"index": 0, "message": map[string]string{"role": "assistant", "content": "hello from " + *name}, "finish_reason": "stop"}},
|
||||
"usage": map[string]int{"prompt_tokens": 3, "completion_tokens": 3, "total_tokens": 6},
|
||||
"usage": usage, "timings": timings,
|
||||
})
|
||||
return
|
||||
}
|
||||
@@ -73,16 +79,24 @@ func main() {
|
||||
w.Header().Set("Cache-Control", "no-cache")
|
||||
w.WriteHeader(http.StatusOK)
|
||||
fl, _ := w.(http.Flusher)
|
||||
flush := func() {
|
||||
if fl != nil {
|
||||
fl.Flush()
|
||||
}
|
||||
}
|
||||
for i := 1; i <= 5; i++ {
|
||||
chunk := map[string]any{"id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model,
|
||||
"choices": []map[string]any{{"index": 0, "delta": map[string]string{"content": fmt.Sprintf("%s chunk %d ", *name, i)}}}}
|
||||
b, _ := json.Marshal(chunk)
|
||||
fmt.Fprintf(w, "data: %s\n\n", b)
|
||||
if fl != nil {
|
||||
fl.Flush()
|
||||
}
|
||||
flush()
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
final := map[string]any{"id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model,
|
||||
"choices": []map[string]any{}, "usage": usage, "timings": timings}
|
||||
b, _ := json.Marshal(final)
|
||||
fmt.Fprintf(w, "data: %s\n\n", b)
|
||||
flush()
|
||||
fmt.Fprint(w, "data: [DONE]\n\n")
|
||||
})
|
||||
mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -90,7 +104,7 @@ func main() {
|
||||
http.Error(w, `{"error":"not found"}`, http.StatusNotFound)
|
||||
})
|
||||
|
||||
log.Printf("fakeupstream %s listening on %s models=%v", *name, *listen, ids)
|
||||
log.Printf("fakeupstream %s listening on %s models=%v slow=%dms", *name, *listen, ids, *slow)
|
||||
srv := &http.Server{Addr: *listen, Handler: mux, ReadHeaderTimeout: 5 * time.Second}
|
||||
log.Fatal(srv.ListenAndServe())
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user