68 lines
3.6 KiB
TOML
68 lines
3.6 KiB
TOML
# crossbar on hyperborea — the fleet's llama-server routers behind one tailnet endpoint.
|
|
# Clients: http://hyperborea.scylla-hammerhead.ts.net:7777/<route>/v1
|
|
listen = "100.112.40.10:7777" # hyperborea's tailnet address only; never a LAN or 0.0.0.0 bind
|
|
db = "/srv/crossbar/crossbar.db"
|
|
poll_interval = "60s"
|
|
lease_idle = "30m"
|
|
retention = "180d"
|
|
queue_max = 2 # waiting places per (host, model) beyond `parallel`; 503 past that
|
|
identity = "off" # switch to "tailscale" once routes carry `peers`
|
|
|
|
# `models` lists what each router is configured to serve, with that model's `parallel` from its
|
|
# preset; the poller learns which are actually loaded (only those count for stickiness and the
|
|
# context guard) and a request for an unloaded model still goes to a healthy host, where the
|
|
# router autoloads it as today.
|
|
|
|
[hosts.titan] # M3 Max 128 GB; ~2x straylight's decode speed
|
|
base_url = "http://titan.scylla-hammerhead.ts.net:8081"
|
|
weight = 2.0
|
|
models = { "ornith-1.5-35b-a3b" = { parallel = 4 }, "ornith-1.5-9b-uncensored" = { parallel = 2 }, "qwen3.8-27b-uncensored" = { parallel = 2 }, "qwen3.6-35b-a3b-abliterated" = { parallel = 2 }, "gemma4-26b-a4b-abliterated" = { parallel = 2 }, "laguna-s-2.1" = { parallel = 2 }, "hermes4-70b-heretic" = { parallel = 1 }, "llama33-70b-abliterated" = { parallel = 1 }, "qwen25-72b-abliterated" = { parallel = 1 } }
|
|
# Wake-on-LAN (best effort — Kyle 2026-09-25: titan is Wi-Fi only, no wired option, and moves
|
|
# between the infrastructure and generic Wi-Fi networks; the private Wi-Fi address is fixed).
|
|
# Magic packets are L2 broadcast; hyperborea is wired on the 192.168.88.0/24 segment, so this
|
|
# only reaches titan while it is on that network. Wake over Wi-Fi on Apple Silicon is unverified.
|
|
[hosts.titan.wake]
|
|
mac = "5e:fc:f2:3f:23:6b" # en0 active (private) address; hardware MAC is 60:3e:5f:33:6f:b8
|
|
broadcast = "192.168.88.255:9"
|
|
wait = "45s"
|
|
|
|
[hosts.straylight]
|
|
base_url = "http://straylight.scylla-hammerhead.ts.net:11434"
|
|
weight = 1.0
|
|
models = { "ornith-1.5-35b-a3b" = { parallel = 4 }, "ornith-1.5-9b-uncensored" = { parallel = 2 }, "qwen3.8-27b-uncensored" = { parallel = 2 }, "qwen3.6-35b-a3b-abliterated" = { parallel = 2 }, "gemma4-26b-a4b-abliterated" = { parallel = 2 }, "qwen3-vl-8b-abliterated" = { parallel = 2 }, "qwen3.8-flash-next-uncensored" = { parallel = 1 }, "ornith-1.0-35b" = { parallel = 2 } }
|
|
|
|
[hosts.dixie] # helper tier: the 9B only (honcho-embed is Honcho's lane, not routed)
|
|
base_url = "http://dixie.scylla-hammerhead.ts.net:11434"
|
|
weight = 0.5
|
|
models = { "ornith-1.5-9b-uncensored" = { parallel = 8 } }
|
|
|
|
# Routes: the first URL path segment (or X-Crossbar-Route). Each conversation on a route gets a
|
|
# sticky lease on the host with the most free slots x weight when it starts.
|
|
[routes.opencode-a]
|
|
hosts = ["titan", "straylight"]
|
|
default_model = "ornith-1.5-35b-a3b"
|
|
|
|
[routes.opencode-b]
|
|
hosts = ["titan", "straylight"]
|
|
default_model = "ornith-1.5-35b-a3b"
|
|
|
|
[routes.paper]
|
|
hosts = ["titan", "straylight"]
|
|
default_model = "ornith-1.5-35b-a3b"
|
|
|
|
[routes.hermes-straylight]
|
|
hosts = ["straylight", "titan", "dixie"]
|
|
default_model = "ornith-1.5-35b-a3b"
|
|
|
|
[routes.hermes-titan]
|
|
hosts = ["titan", "straylight", "dixie"]
|
|
default_model = "ornith-1.5-35b-a3b"
|
|
|
|
[routes.hermes-talos]
|
|
hosts = ["titan", "straylight", "dixie"]
|
|
default_model = "ornith-1.5-35b-a3b"
|
|
|
|
[routes.probe] # for operators: curl tests, never a real client
|
|
hosts = ["dixie", "straylight", "titan"]
|
|
default_model = "ornith-1.5-9b-uncensored"
|