deploy/hyperborea: production config, user unit, install script, README (wake block pending titan's MAC)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 12:57:25 -07:00
co-authored by Claude Fable 5.1
parent 4f03cb2c52
commit 6e9a70587a
4 changed files with 185 additions and 0 deletions
+65
View File
@@ -0,0 +1,65 @@
# crossbar on hyperborea — the fleet's llama-server routers behind one tailnet endpoint.
# Clients: http://hyperborea.scylla-hammerhead.ts.net:7777/<route>/v1
listen = "100.112.40.10:7777" # hyperborea's tailnet address only; never a LAN or 0.0.0.0 bind
db = "/srv/crossbar/crossbar.db"
poll_interval = "60s"
lease_idle = "30m"
retention = "180d"
queue_max = 2 # waiting places per (host, model) beyond `parallel`; 503 past that
identity = "off" # switch to "tailscale" once routes carry `peers`
# `models` lists what each router is configured to serve, with that model's `parallel` from its
# preset; the poller learns which are actually loaded (only those count for stickiness and the
# context guard) and a request for an unloaded model still goes to a healthy host, where the
# router autoloads it as today.
[hosts.titan] # M3 Max 128 GB; ~2x straylight's decode speed
base_url = "http://titan.scylla-hammerhead.ts.net:8081"
weight = 2.0
models = { "ornith-1.5-35b-a3b" = { parallel = 4 }, "ornith-1.5-9b-uncensored" = { parallel = 2 }, "qwen3.8-27b-uncensored" = { parallel = 2 }, "qwen3.6-35b-a3b-abliterated" = { parallel = 2 }, "gemma4-26b-a4b-abliterated" = { parallel = 2 }, "laguna-s-2.1" = { parallel = 2 }, "hermes4-70b-heretic" = { parallel = 1 }, "llama33-70b-abliterated" = { parallel = 1 }, "qwen25-72b-abliterated" = { parallel = 1 } }
# Wake-on-LAN: enable once titan's wake MAC is decided (see README). Magic packets are L2
# broadcast; hyperborea is wired on titan's segment (192.168.88.0/24).
# [hosts.titan.wake]
# mac = "d2:30:99:9a:ee:03" # dock Ethernet en4 if titan is wired; "5e:fc:f2:3f:23:6b" is the Wi-Fi private address
# broadcast = "192.168.88.255:9"
# wait = "45s"
[hosts.straylight]
base_url = "http://straylight.scylla-hammerhead.ts.net:11434"
weight = 1.0
models = { "ornith-1.5-35b-a3b" = { parallel = 4 }, "ornith-1.5-9b-uncensored" = { parallel = 2 }, "qwen3.8-27b-uncensored" = { parallel = 2 }, "qwen3.6-35b-a3b-abliterated" = { parallel = 2 }, "gemma4-26b-a4b-abliterated" = { parallel = 2 }, "qwen3-vl-8b-abliterated" = { parallel = 2 }, "qwen3.8-flash-next-uncensored" = { parallel = 1 }, "ornith-1.0-35b" = { parallel = 2 } }
[hosts.dixie] # helper tier: the 9B only (honcho-embed is Honcho's lane, not routed)
base_url = "http://dixie.scylla-hammerhead.ts.net:11434"
weight = 0.5
models = { "ornith-1.5-9b-uncensored" = { parallel = 8 } }
# Routes: the first URL path segment (or X-Crossbar-Route). Each conversation on a route gets a
# sticky lease on the host with the most free slots x weight when it starts.
[routes.opencode-a]
hosts = ["titan", "straylight"]
default_model = "ornith-1.5-35b-a3b"
[routes.opencode-b]
hosts = ["titan", "straylight"]
default_model = "ornith-1.5-35b-a3b"
[routes.paper]
hosts = ["titan", "straylight"]
default_model = "ornith-1.5-35b-a3b"
[routes.hermes-straylight]
hosts = ["straylight", "titan", "dixie"]
default_model = "ornith-1.5-35b-a3b"
[routes.hermes-titan]
hosts = ["titan", "straylight", "dixie"]
default_model = "ornith-1.5-35b-a3b"
[routes.hermes-talos]
hosts = ["titan", "straylight", "dixie"]
default_model = "ornith-1.5-35b-a3b"
[routes.probe] # for operators: curl tests, never a real client
hosts = ["dixie", "straylight", "titan"]
default_model = "ornith-1.5-9b-uncensored"