Smoke run for v1; README for leases, admin and accounting

Implemented-By: OpenCode session (model recorded in docs/implementer-log.md)
This commit is contained in:
2026-09-25 06:56:54 -07:00
parent 587ec7a1ec
commit cf2aa24393
5 changed files with 178 additions and 55 deletions
+22 -8
View File
@@ -1,11 +1,13 @@
// fakeupstream stands in for a llama-server router in tests and the smoke run. Do not edit.
//
// fakeupstream -listen 127.0.0.1:18081 -name alpha -models a,b -down-file /tmp/alpha.down
// fakeupstream -listen 127.0.0.1:18081 -name alpha -models a,b -down-file /tmp/alpha.down -slow 0
//
// /health answers 503 while the down file exists, 200 otherwise. /v1/models lists -models.
// /props answers a small JSON object. /v1/chat/completions echoes: a streamed answer of five
// SSE chunks 200 ms apart when the body has "stream": true, one JSON answer otherwise. Every
// response carries X-Upstream: <name>.
// SSE chunks 200 ms apart when the body has "stream": true, then a final chunk carrying
// "usage" and llama-server style "timings", then [DONE]; one JSON answer with usage and
// timings otherwise. -slow adds that many milliseconds before answering (for queue tests).
// Every response carries X-Upstream: <name>.
package main
import (
@@ -25,11 +27,14 @@ func main() {
name := flag.String("name", "fake", "name reported in X-Upstream and answers")
models := flag.String("models", "m", "comma-separated model ids for /v1/models")
downFile := flag.String("down-file", "", "while this file exists, /health answers 503")
slow := flag.Int("slow", 0, "milliseconds to wait before answering a completion")
flag.Parse()
ids := strings.Split(*models, ",")
mux := http.NewServeMux()
stamp := func(w http.ResponseWriter) { w.Header().Set("X-Upstream", *name) }
usage := map[string]any{"prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110}
timings := map[string]any{"prompt_n": 100, "cache_n": 90, "predicted_n": 10, "predicted_ms": 50.0}
mux.HandleFunc("/health", func(w http.ResponseWriter, r *http.Request) {
stamp(w)
@@ -61,11 +66,12 @@ func main() {
Stream bool `json:"stream"`
}
_ = json.Unmarshal(body, &req)
time.Sleep(time.Duration(*slow) * time.Millisecond)
if !req.Stream {
writeJSON(w, map[string]any{
"id": "chatcmpl-fake", "object": "chat.completion", "model": req.Model,
"choices": []map[string]any{{"index": 0, "message": map[string]string{"role": "assistant", "content": "hello from " + *name}, "finish_reason": "stop"}},
"usage": map[string]int{"prompt_tokens": 3, "completion_tokens": 3, "total_tokens": 6},
"usage": usage, "timings": timings,
})
return
}
@@ -73,16 +79,24 @@ func main() {
w.Header().Set("Cache-Control", "no-cache")
w.WriteHeader(http.StatusOK)
fl, _ := w.(http.Flusher)
flush := func() {
if fl != nil {
fl.Flush()
}
}
for i := 1; i <= 5; i++ {
chunk := map[string]any{"id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model,
"choices": []map[string]any{{"index": 0, "delta": map[string]string{"content": fmt.Sprintf("%s chunk %d ", *name, i)}}}}
b, _ := json.Marshal(chunk)
fmt.Fprintf(w, "data: %s\n\n", b)
if fl != nil {
fl.Flush()
}
flush()
time.Sleep(200 * time.Millisecond)
}
final := map[string]any{"id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model,
"choices": []map[string]any{}, "usage": usage, "timings": timings}
b, _ := json.Marshal(final)
fmt.Fprintf(w, "data: %s\n\n", b)
flush()
fmt.Fprint(w, "data: [DONE]\n\n")
})
mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) {
@@ -90,7 +104,7 @@ func main() {
http.Error(w, `{"error":"not found"}`, http.StatusNotFound)
})
log.Printf("fakeupstream %s listening on %s models=%v", *name, *listen, ids)
log.Printf("fakeupstream %s listening on %s models=%v slow=%dms", *name, *listen, ids, *slow)
srv := &http.Server{Addr: *listen, Handler: mux, ReadHeaderTimeout: 5 * time.Second}
log.Fatal(srv.ListenAndServe())
}