Files
imladris/hw/straylight/default.nix
T

434 lines
19 KiB
Nix

{ inputs, pkgs, lib, ... }:
let
pkgsUnstable = inputs.nixpkgs-unstable.legacyPackages.${pkgs.stdenv.hostPlatform.system};
# llama.cpp from nixpkgs-unstable (0.4.x as of Sept 2026) with the Vulkan backend for
# gfx1151 (Strix Halo). Laguna S 2.1 and Gemma 4 need a build newer than mid-July 2026,
# so keep the unstable input reasonably fresh (`nix flake update nixpkgs-unstable`).
llama-cpp = pkgsUnstable.llama-cpp.override { vulkanSupport = true; };
llamaModelsDir = "/var/lib/llama-server/models";
# llama-server runs in router mode: one endpoint, models load on demand, at most two
# resident at a time (see --models-max below). Section names are the model ids that
# clients pass in the "model" field. Weights are fetched by
# /var/lib/llama-server/models/download.sh, not by the service.
llamaModelsIni = pkgs.writeText "llama-models.ini" ''
version = 1
[*]
jinja = true
flash-attn = on
cache-type-k = q8_0
cache-type-v = q8_0
n-gpu-layers = 999
no-mmap = true
; llama.cpp splits ctx-size across parallel slots: 262144 / 2 = 131072 per slot.
ctx-size = 262144
parallel = 2
; Unload a model's weights and KV cache after six idle hours; the next request reloads it.
sleep-idle-seconds = 21600
; Qwen3.6-35B-A3B, refusal-ablated (HauhauCS "Aggressive"). Benchmark candidate A.
[qwen3.6-35b-a3b-abliterated]
model = ${llamaModelsDir}/qwen3.6-35b-a3b-abliterated/Qwen3.6-35B-A3B-Uncensored-HauhauCS-Aggressive-Q4_K_M.gguf
; Gemma 4 26B-A4B QAT, refusal-ablated (HauhauCS "Balanced"), with its MTP draft head.
; Benchmark candidate B (the Western model of the same MoE class).
[gemma4-26b-a4b-abliterated]
model = ${llamaModelsDir}/gemma4-26b-a4b-abliterated/Gemma4-26B-A4B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf
model-draft = ${llamaModelsDir}/gemma4-26b-a4b-abliterated/mtp-gemma-4-26B-A4B-it.gguf
spec-type = draft-mtp
; Poolside Laguna S 2.1 (118B total, 8B active): primary coding agent.
; ~69 GB of weights, so it only loads once the raised TTM/GTT limit is active (reboot).
; DFlash GGUF is on disk (laguna-s-2.1-DFlash-BF16.gguf). nixpkgs llama.cpp 0.4.0
; rejects it at load ("expected 76, got 69" tensors); Poolside's file needs their
; laguna-branch loader. Leave unwired so Laguna still starts.
[laguna-s-2.1]
model = ${llamaModelsDir}/laguna-s-2.1-UD-Q4_K_S/Laguna-S-2.1-UD-Q4_K_S-00001-of-00003.gguf
; Two slots sharing one unified 131072-token KV pool: a second request (subagent,
; another client) runs alongside instead of queueing, and either one may still use
; the full 131072 as long as their combined context fits.
parallel = 2
kv-unified = true
ctx-size = 131072
; Previous default model (a Qwen 3.5 derivative), kept for comparison runs. It was
; fetched by the old -hf flag into the HF cache; dedup hides the cache's own entry.
[ornith-1.0-35b]
hf-repo = deepreinforce-ai/Ornith-1.0-35B-GGUF:Q4_K_M
dedup-cache-models = true
; Ornith 1.5 9B, refusal-ablated (junafinity / mradermacher GGUF).
; Dense VLM: --hf-repo pulls mmproj-Q8_0; reports image input once loaded.
[ornith-1.5-9b-uncensored]
hf-repo = mradermacher/Ornith-1.5-9B-uncensored-GGUF:Q4_K_M
dedup-cache-models = true
; Qwen3-VL-8B Instruct, refusal-ablated (prithivMLmods v2 / mradermacher).
; Dedicated VLM in the same size class as Ornith 1.5 9B. Explicit mmproj-url
; because router-mode --hf-repo does not attach the projector on its own.
[qwen3-vl-8b-abliterated]
hf-repo = mradermacher/Qwen3-VL-8B-Instruct-abliterated-v2.0-GGUF:Q4_K_M
mmproj-url = https://huggingface.co/mradermacher/Qwen3-VL-8B-Instruct-abliterated-v2.0-GGUF/resolve/main/Qwen3-VL-8B-Instruct-abliterated-v2.0.mmproj-f16.gguf
dedup-cache-models = true
; Ornith 1.5 35B-A3B MoE (official GGUF). ~3B active, multimodal; mmproj
; comes with --hf-repo. Q4_K_M is ~22 GB.
; Sampling + a thinking cap: unrestricted <think> on this checkpoint stalls
; long agent turns (Talos). Clients that send their own temperature still win.
[ornith-1.5-35b-a3b]
hf-repo = ornith-ai/Ornith-1.5-35B-A3B-GGUF:Q4_K_M
temp = 0.6
top-p = 0.95
top-k = 20
reasoning-budget = 8192
dedup-cache-models = true
; Qwen 3.8 27B dense, refusal-ablated. Preferred uncensored coding agent.
; orcarouter/Qwen3.8-27B-Uncensored-GGUF is gated; huihui Q4_K_L keeps
; ablation-target tensors at Q8_0. llama-server fetches on first load.
[qwen3.8-27b-uncensored]
hf-repo = huihui-ai/Huihui-Qwen3.8-27B-abliterated-GGUF:Q4_K_L
dedup-cache-models = true
; Qwen3.8-Flash-Next (~177B / 6B active, qwen4exp), orcarouter-abliterated,
; cygnal IQ4_XS-NGQ4 GGUF (~98 GB) measured on gfx1151. Needs ~99 GB GTT
; plus KV: unload Laguna first. 32k ctx / 1 slot so it fits beside the
; 104 GiB TTM cap. mmproj is explicit; hf-repo alone misses it in router mode.
[qwen3.8-flash-next-uncensored]
hf-repo = cygnal/Qwen3.8-Flash-Next-Uncensored-IQ4XS-NGQ4-GGUF
hf-file = Qwen3.8-Flash-Next-Uncensored-IQ4XS-NGQ4.gguf
mmproj-url = https://huggingface.co/cygnal/Qwen3.8-Flash-Next-Uncensored-IQ4XS-NGQ4-GGUF/resolve/main/mmproj-Qwen3.8-Flash-Next-Uncensored-BF16.gguf
parallel = 1
ctx-size = 32768
dedup-cache-models = true
'';
# `llama-models` shows what the router has resident; `llama-unload` frees every loaded
# model (e.g. before a gaming session), or just the ones named on the command line.
llamaCtl = pkgs.writeText "llama-ctl.py" ''
import json, sys, urllib.request
URL = "http://127.0.0.1:11434"
def models():
with urllib.request.urlopen(URL + "/models", timeout=10) as r:
d = json.load(r)
return d.get("data", d if isinstance(d, list) else [])
def name(m):
return m.get("id") or m.get("model") or m.get("name")
cmd = sys.argv[1] if len(sys.argv) > 1 else "list"
if cmd == "list":
for m in models():
print("%-10s %s" % ((m.get("status") or {}).get("value", "?"), name(m)))
elif cmd == "unload":
loaded = [name(m) for m in models()
if (m.get("status") or {}).get("value") in ("loaded", "loading", "sleeping")]
targets = sys.argv[2:] or loaded
if not targets:
print("nothing loaded")
for t in targets:
req = urllib.request.Request(URL + "/models/unload", data=json.dumps({"model": t}).encode(),
headers={"Content-Type": "application/json"}, method="POST")
with urllib.request.urlopen(req, timeout=30) as r:
print("unload %s: %s" % (t, r.read().decode().strip()))
else:
sys.exit("usage: llama-ctl.py [list|unload [model ...]]")
'';
llamaModelsCmd = pkgs.writeShellScriptBin "llama-models" "exec ${pkgs.python3}/bin/python3 ${llamaCtl} list \"$@\"";
llamaUnloadCmd = pkgs.writeShellScriptBin "llama-unload" "exec ${pkgs.python3}/bin/python3 ${llamaCtl} unload \"$@\"";
# CPU whisper.cpp: keep GTT for llama-server. large-v3-turbo is ~1.5 GB.
whisperCpp = pkgsUnstable.whisper-cpp.override { withFFmpegSupport = true; };
whisperModel = "/var/lib/whisper-server/models/ggml-large-v3-turbo.bin";
whisperFetch = pkgs.writeShellScript "whisper-fetch-model" ''
set -eu
D=/var/lib/whisper-server/models
mkdir -p "$D"
F=${whisperModel}
if [ ! -s "$F" ]; then
echo "fetching whisper large-v3-turbo"
${pkgs.curl}/bin/curl -L --retry 20 --retry-delay 10 --retry-all-errors \
-o "$F.part" \
https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin
mv "$F.part" "$F"
fi
'';
whisperTranscribe = pkgs.writeShellScriptBin "whisper-transcribe" ''
exec ${whisperCpp}/bin/whisper-cli -m ${whisperModel} --language auto -f "$@"
'';
amdgpuDpmHigh = pkgs.writeShellScript "amdgpu-dpm-high" ''
set -eu
for f in /sys/class/drm/card*/device/power_dpm_force_performance_level; do
vendor="$(dirname "$f")/vendor"
[ -r "$vendor" ] || continue
grep -qx '0x1002' "$vendor" || continue
echo high > "$f"
done
'';
in
{
imports = [
./hardware-configuration.nix
../../configs/desktop.nix
../../configs/qemu.nix
../../configs/mcpkg.nix
../../configs/mcp.nix # MCP agent + mcp user (straylight is becoming the core host)
];
config = {
# Let the GPU address up to 104 GiB of the 128 GiB of unified memory (the TTM default
# caps GTT at 50% of RAM, ~62 GiB). Units are 4 KiB pages; takes effect after a reboot.
# amdgpu.gttsize is deprecated on recent kernels; ttm.* is the supported knob.
boot.kernelParams = [ "ttm.pages_limit=27262976" "ttm.page_pool_size=27262976" ];
# amdgpu defaults to powersave/auto and will clock the iGPU down under a
# desktop compositor. llama.cpp on gfx1151 wants the high DPM profile.
systemd.services.amdgpu-dpm-high = {
description = "Set amdgpu DPM performance level to high";
wantedBy = [ "multi-user.target" ];
after = [ "systemd-udev-settle.service" ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
ExecStart = amdgpuDpmHigh;
};
};
# llama.cpp server (router mode) with the Vulkan backend for gfx1151 (Strix Halo).
# Models and per-model settings live in llamaModelsIni above.
# NOTE: BIOS "UMA Frame Buffer Size" must be ≥32GB for a 23GB model to fit on-GPU.
users.users.llama-server = {
isSystemUser = true;
group = "llama-server";
home = "/var/lib/llama-server";
};
users.groups.llama-server = {};
systemd.services.llama-server = {
description = "llama.cpp inference server";
wantedBy = [ "multi-user.target" ];
after = [ "network-online.target" ];
wants = [ "network-online.target" ];
environment = {
HOME = "/var/lib/llama-server";
HF_HOME = "/var/lib/llama-server/huggingface";
};
serviceConfig = {
# Bind all interfaces so localhost and the tailnet can both reach the
# OpenAI-compatible API. LAN access is still blocked: 11434 is opened
# only on tailscale0, not in the global allowedTCPPorts list.
# "+" runs as root so we can pin amdgpu DPM before the server starts.
ExecStartPre = "+${amdgpuDpmHigh}";
ExecStart = ''
${llama-cpp}/bin/llama-server \
--host 0.0.0.0 \
--port 11434 \
--models-preset ${llamaModelsIni} \
--models-max 2 \
--timeout 7200
'';
User = "llama-server";
Group = "llama-server";
StateDirectory = "llama-server";
WorkingDirectory = "/var/lib/llama-server";
SupplementaryGroups = [ "render" "video" ];
Restart = "on-failure";
RestartSec = "10s";
TimeoutStartSec = "600";
};
};
# Local ASR: OpenAI-compatible /v1/audio/transcriptions on 11435.
# CPU-only so it does not contend with llama-server for GTT.
users.users.whisper-server = {
isSystemUser = true;
group = "whisper-server";
home = "/var/lib/whisper-server";
};
users.groups.whisper-server = {};
systemd.services.whisper-server = {
description = "whisper.cpp transcription server";
wantedBy = [ "multi-user.target" ];
after = [ "network-online.target" ];
wants = [ "network-online.target" ];
serviceConfig = {
ExecStartPre = whisperFetch;
ExecStart = ''
${whisperCpp}/bin/whisper-server \
--model ${whisperModel} \
--host 0.0.0.0 \
--port 11435 \
--inference-path /v1/audio/transcriptions \
--language auto \
--convert \
--threads 8 \
--no-gpu
'';
User = "whisper-server";
Group = "whisper-server";
StateDirectory = "whisper-server";
WorkingDirectory = "/var/lib/whisper-server";
Restart = "on-failure";
RestartSec = "10s";
TimeoutStartSec = "600";
};
};
# straylight is the unikernel host. The shared mcp.nix locks the agent
# down with PrivateDevices=true, which hides /dev/kvm and /dev/net/tun.
# Relax that here (only on straylight) so the agent can boot Nanos
# unikernel VMs under QEMU/KVM and (Phase 2) manage TAP devices.
systemd.services.mcp-agent.serviceConfig = {
PrivateDevices = lib.mkForce false;
DeviceAllow = [ "/dev/kvm rw" "/dev/net/tun rw" ];
SupplementaryGroups = [ "kvm" ];
AmbientCapabilities = [ "CAP_NET_ADMIN" ];
# The agent launches each unikernel as a daemonized QEMU process in its
# own cgroup. With the default KillMode=control-group, restarting the
# agent would SIGKILL every running VM. KillMode=process kills only the
# agent's main process on stop/restart, so VMs survive an agent upgrade.
# (If a VM does die, the agent's startup Recover restarts it.)
KillMode = lib.mkForce "process";
};
# Let the mcp user reach /dev/kvm directly as well.
users.users.mcp.extraGroups = [ "kvm" ];
# Isolated host-only bridge for unikernel VMs (Phase 2). Each unikernel
# gets a TAP on this bridge and a 10.99.0.0/24 static IP. The bridge has
# NO uplink and NO NAT, and the firewall drops any VM traffic leaving the
# bridge, so a VM can reach only the host gateway (10.99.0.1) -- mediation
# is enforced by network topology, not convention.
networking.bridges.mcp-br0.interfaces = [ ];
networking.interfaces.mcp-br0.ipv4.addresses = [
{ address = "10.99.0.1"; prefixLength = 24; }
];
# The host accepts traffic from VMs (so mc-proxy on the gateway can serve
# them); the FORWARD drop prevents VMs from routing anywhere off-bridge.
networking.firewall.trustedInterfaces = [ "mcp-br0" ];
networking.firewall.extraCommands = ''
iptables -D FORWARD -i mcp-br0 ! -o mcp-br0 -j DROP 2>/dev/null || true
iptables -A FORWARD -i mcp-br0 ! -o mcp-br0 -j DROP
'';
networking.firewall.extraStopCommands = ''
iptables -D FORWARD -i mcp-br0 ! -o mcp-br0 -j DROP 2>/dev/null || true
'';
# Allow rootless containers (podman) to bind low ports (53 for MCNS,
# 443/8443/9443 for mc-proxy) as straylight takes over the core role.
boot.kernel.sysctl."net.ipv4.ip_unprivileged_port_start" = 53;
hardware.bluetooth.enable = true; # enables BlueZ + bluetooth.service
programs.steam.enable = true;
programs.steam.protontricks.enable = true;
# herdr: terminal workspace manager for coding agents (https://herdr.dev).
# Pinned to the upstream flake (v0.9.0); nixpkgs-unstable only has 0.7.1.
environment.systemPackages = [
inputs.herdr.packages.x86_64-linux.default
llamaModelsCmd
llamaUnloadCmd
whisperTranscribe
whisperCpp
];
services.open-webui = {
enable = true;
host = "0.0.0.0";
port = 8080;
environment = {
SCARF_NO_ANALYTICS = "True";
DO_NOT_TRACK = "True";
ANONYMIZED_TELEMETRY = "False";
# Point at the local llama.cpp server (OpenAI-compatible API).
OPENAI_API_BASE_URLS = "http://127.0.0.1:11434/v1";
OPENAI_API_KEYS = "none";
ENABLE_OLLAMA_API = "False";
AUDIO_STT_ENGINE = "openai";
AUDIO_STT_OPENAI_API_BASE_URL = "http://127.0.0.1:11435/v1";
AUDIO_STT_OPENAI_API_KEY = "none";
};
};
# Project tracker (hackerman edition), tailnet-only.
services.tracker.enable = true;
# Hermes Agent gateway (messaging platforms, e.g. Mattermost).
# The CLI itself is installed imperatively via `uv tool install hermes-agent`
# (shim at ~/.local/bin/hermes); only the always-on gateway service is
# declared here. Credentials/state live in ~/.hermes (unmanaged; keep the
# bot token out of git). Do NOT also run `hermes gateway install` — it would
# write a mutable unit that shadows this one. Applies to every user manager
# on the host, but kyle is the only linger-enabled user, so effectively his.
users.users.kyle.linger = true;
systemd.user.services.hermes-gateway = {
description = "Hermes Agent Gateway - Messaging Platform Integration";
documentation = [ "https://hermes-agent.nousresearch.com/docs" ];
after = [ "network-online.target" ];
wants = [ "network-online.target" ];
wantedBy = [ "default.target" ];
startLimitIntervalSec = 0;
unitConfig.ConditionPathExists = "%h/.hermes/config.yaml";
serviceConfig = {
Type = "simple";
ExecStart = "%h/.local/bin/hermes gateway run";
WorkingDirectory = "%h/.hermes";
Environment = [
"PATH=%h/.local/bin:/run/current-system/sw/bin"
"HERMES_HOME=%h/.hermes"
# uv's foreign CPython has no CA bundle; point std-lib SSL at the
# NixOS store (needed for OAuth token refresh via urllib).
"SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt"
];
Restart = "always";
RestartSec = 5;
RestartForceExitStatus = 75;
RestartPreventExitStatus = 78;
KillMode = "mixed";
KillSignal = "SIGTERM";
ExecReload = "/run/current-system/sw/bin/kill -USR1 $MAINPID";
TimeoutStopSec = 60;
};
};
# Headless Chromium exposing a CDP endpoint for Hermes browser tools.
# Loopback only; browser.cdp_url in ~/.hermes/config.yaml points at it.
# Ungoogled-chromium (already in desktop-packages-light.nix), no ad-block
# extension so behavior stays stock; dedicated user-data-dir so this
# instance never touches kyle's browsing profile.
systemd.user.services.chromium-cdp = {
description = "Headless Chromium - CDP endpoint for Hermes browser tools";
after = [ "graphical-session.target" ];
wantedBy = [ "default.target" ];
serviceConfig = {
Type = "simple";
ExecStart =
let chromium = pkgs.ungoogled-chromium;
in "${chromium}/bin/chromium --headless=new --remote-debugging-port=9222 --remote-debugging-address=127.0.0.1 --user-data-dir=%h/.cache/chromium-cdp --no-first-run --no-default-browser-check --disable-background-networking";
Restart = "on-failure";
RestartSec = 5;
};
};
# Open ports: DNS (53), mc-proxy (443/8443/9443), agent (9444), master (9555), open-webui (8080).
networking.firewall.allowedTCPPorts = [ 53 443 8080 8443 9443 9444 9555 ];
networking.firewall.allowedUDPPorts = [ 53 ];
# llama.cpp OpenAI-compatible API: tailnet only (localhost is always allowed).
networking.firewall.interfaces.tailscale0.allowedTCPPorts = [ 11434 11435 ];
# DNS: MCNS for internal zones, public resolvers as fallback.
networking.nameservers = [
"192.168.88.181"
"100.95.252.120"
"1.1.1.1"
"8.8.8.8"
];
services.resolved.settings.Resolve.Domains = [
"~mcp.metacircular.net"
];
};
}