straylight: pin amdgpu DPM high and enable Laguna DFlash

Keep the iGPU out of powersave while llama-server runs, and pair Laguna
S 2.1 with Poolside's DFlash draft (f16 draft KV, n-max 4).
This commit is contained in:
2026-09-17 23:54:27 -07:00
parent c5fed8c604
commit ab89181213
+32
View File
@@ -39,8 +39,16 @@ let
; Poolside Laguna S 2.1 (118B total, 8B active): primary coding agent. ; Poolside Laguna S 2.1 (118B total, 8B active): primary coding agent.
; ~69 GB of weights, so it only loads once the raised TTM/GTT limit is active (reboot). ; ~69 GB of weights, so it only loads once the raised TTM/GTT limit is active (reboot).
; DFlash draft (~2.2 GB BF16) from poolside/Laguna-S-2.1-GGUF. Keep draft KV in
; f16; q8 draft KV collapses acceptance. spec-draft-n-max 4 is the gfx1151
; sweet spot (n=15 is the trained cap and is slower here).
[laguna-s-2.1] [laguna-s-2.1]
model = ${llamaModelsDir}/laguna-s-2.1-UD-Q4_K_S/Laguna-S-2.1-UD-Q4_K_S-00001-of-00003.gguf model = ${llamaModelsDir}/laguna-s-2.1-UD-Q4_K_S/Laguna-S-2.1-UD-Q4_K_S-00001-of-00003.gguf
model-draft = ${llamaModelsDir}/laguna-s-2.1-UD-Q4_K_S/laguna-s-2.1-DFlash-BF16.gguf
spec-type = draft-dflash
spec-draft-n-max = 4
cache-type-k-draft = f16
cache-type-v-draft = f16
parallel = 1 parallel = 1
ctx-size = 131072 ctx-size = 131072
@@ -143,6 +151,15 @@ let
whisperTranscribe = pkgs.writeShellScriptBin "whisper-transcribe" '' whisperTranscribe = pkgs.writeShellScriptBin "whisper-transcribe" ''
exec ${whisperCpp}/bin/whisper-cli -m ${whisperModel} --language auto -f "$@" exec ${whisperCpp}/bin/whisper-cli -m ${whisperModel} --language auto -f "$@"
''; '';
amdgpuDpmHigh = pkgs.writeShellScript "amdgpu-dpm-high" ''
set -eu
for f in /sys/class/drm/card*/device/power_dpm_force_performance_level; do
vendor="$(dirname "$f")/vendor"
[ -r "$vendor" ] || continue
grep -qx '0x1002' "$vendor" || continue
echo high > "$f"
done
'';
in in
{ {
imports = [ imports = [
@@ -159,6 +176,19 @@ in
# amdgpu.gttsize is deprecated on recent kernels; ttm.* is the supported knob. # amdgpu.gttsize is deprecated on recent kernels; ttm.* is the supported knob.
boot.kernelParams = [ "ttm.pages_limit=27262976" "ttm.page_pool_size=27262976" ]; boot.kernelParams = [ "ttm.pages_limit=27262976" "ttm.page_pool_size=27262976" ];
# amdgpu defaults to powersave/auto and will clock the iGPU down under a
# desktop compositor. llama.cpp on gfx1151 wants the high DPM profile.
systemd.services.amdgpu-dpm-high = {
description = "Set amdgpu DPM performance level to high";
wantedBy = [ "multi-user.target" ];
after = [ "systemd-udev-settle.service" ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
ExecStart = amdgpuDpmHigh;
};
};
# llama.cpp server (router mode) with the Vulkan backend for gfx1151 (Strix Halo). # llama.cpp server (router mode) with the Vulkan backend for gfx1151 (Strix Halo).
# Models and per-model settings live in llamaModelsIni above. # Models and per-model settings live in llamaModelsIni above.
# NOTE: BIOS "UMA Frame Buffer Size" must be ≥32GB for a 23GB model to fit on-GPU. # NOTE: BIOS "UMA Frame Buffer Size" must be ≥32GB for a 23GB model to fit on-GPU.
@@ -182,6 +212,8 @@ in
# Bind all interfaces so localhost and the tailnet can both reach the # Bind all interfaces so localhost and the tailnet can both reach the
# OpenAI-compatible API. LAN access is still blocked: 11434 is opened # OpenAI-compatible API. LAN access is still blocked: 11434 is opened
# only on tailscale0, not in the global allowedTCPPorts list. # only on tailscale0, not in the global allowedTCPPorts list.
# "+" runs as root so we can pin amdgpu DPM before the server starts.
ExecStartPre = "+${amdgpuDpmHigh}";
ExecStart = '' ExecStart = ''
${llama-cpp}/bin/llama-server \ ${llama-cpp}/bin/llama-server \
--host 0.0.0.0 \ --host 0.0.0.0 \