// fakeupstream stands in for a llama-server router in tests and the smoke run. Do not edit. // // fakeupstream -listen 127.0.0.1:18081 -name alpha -models a,b -down-file /tmp/alpha.down // // /health answers 503 while the down file exists, 200 otherwise. /v1/models lists -models. // /props answers a small JSON object. /v1/chat/completions echoes: a streamed answer of five // SSE chunks 200 ms apart when the body has "stream": true, one JSON answer otherwise. Every // response carries X-Upstream: . package main import ( "encoding/json" "flag" "fmt" "io" "log" "net/http" "os" "strings" "time" ) func main() { listen := flag.String("listen", "127.0.0.1:18081", "address to listen on") name := flag.String("name", "fake", "name reported in X-Upstream and answers") models := flag.String("models", "m", "comma-separated model ids for /v1/models") downFile := flag.String("down-file", "", "while this file exists, /health answers 503") flag.Parse() ids := strings.Split(*models, ",") mux := http.NewServeMux() stamp := func(w http.ResponseWriter) { w.Header().Set("X-Upstream", *name) } mux.HandleFunc("/health", func(w http.ResponseWriter, r *http.Request) { stamp(w) if *downFile != "" { if _, err := os.Stat(*downFile); err == nil { http.Error(w, `{"error":{"message":"Loading model"}}`, http.StatusServiceUnavailable) return } } writeJSON(w, map[string]string{"status": "ok"}) }) mux.HandleFunc("/v1/models", func(w http.ResponseWriter, r *http.Request) { stamp(w) data := []map[string]any{} for _, id := range ids { data = append(data, map[string]any{"id": id, "object": "model", "owned_by": *name}) } writeJSON(w, map[string]any{"object": "list", "data": data}) }) mux.HandleFunc("/props", func(w http.ResponseWriter, r *http.Request) { stamp(w) writeJSON(w, map[string]any{"default_generation_settings": map[string]any{"n_ctx": 8192}, "total_slots": 2, "model_path": *name}) }) mux.HandleFunc("/v1/chat/completions", func(w http.ResponseWriter, r *http.Request) { stamp(w) body, _ := io.ReadAll(io.LimitReader(r.Body, 1<<20)) var req struct { Model string `json:"model"` Stream bool `json:"stream"` } _ = json.Unmarshal(body, &req) if !req.Stream { writeJSON(w, map[string]any{ "id": "chatcmpl-fake", "object": "chat.completion", "model": req.Model, "choices": []map[string]any{{"index": 0, "message": map[string]string{"role": "assistant", "content": "hello from " + *name}, "finish_reason": "stop"}}, "usage": map[string]int{"prompt_tokens": 3, "completion_tokens": 3, "total_tokens": 6}, }) return } w.Header().Set("Content-Type", "text/event-stream") w.Header().Set("Cache-Control", "no-cache") w.WriteHeader(http.StatusOK) fl, _ := w.(http.Flusher) for i := 1; i <= 5; i++ { chunk := map[string]any{"id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model, "choices": []map[string]any{{"index": 0, "delta": map[string]string{"content": fmt.Sprintf("%s chunk %d ", *name, i)}}}} b, _ := json.Marshal(chunk) fmt.Fprintf(w, "data: %s\n\n", b) if fl != nil { fl.Flush() } time.Sleep(200 * time.Millisecond) } fmt.Fprint(w, "data: [DONE]\n\n") }) mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) { stamp(w) http.Error(w, `{"error":"not found"}`, http.StatusNotFound) }) log.Printf("fakeupstream %s listening on %s models=%v", *name, *listen, ids) srv := &http.Server{Addr: *listen, Handler: mux, ReadHeaderTimeout: 5 * time.Second} log.Fatal(srv.ListenAndServe()) } func writeJSON(w http.ResponseWriter, v any) { w.Header().Set("Content-Type", "application/json") _ = json.NewEncoder(w).Encode(v) }