Dixie #1

Merged
kyle merged 12 commits from dixie into master 2026-09-22 04:56:26 +00:00
5 changed files with 247 additions and 0 deletions
Showing only changes of commit b632e20fd6 - Show all commits
+1
View File
@@ -67,6 +67,7 @@
lib = nixpkgs.lib;
in {
nixosConfigurations = lib.genAttrs [
"dixie"
"imladris"
"ono-sendai"
"orion"
+4
View File
@@ -0,0 +1,4 @@
ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIDXu7ME91tLfU+Q3fNFe8VyIouVvoktoYy8xVNpObxuv kyle@titan.local
ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIGea83yMIdCi0QUUPgmhRgIrii7lS1dYxZ6LSxSsDOph kyle@europa
ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIHukZAxnlMSoTlRtRF14mLbC0x+rb1zlxnSLVLg+6Xbh kyle@vade
ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAILw318jrtFJw1Cn3WnOk17Gkr8bo7NOACQwMQ7S4hvoL kyle@straylight
+186
View File
@@ -0,0 +1,186 @@
{ inputs, pkgs, lib, config, ... }:
let
# Unstable instantiated with unfree allowed: CUDA is unfree, and legacyPackages does not
# inherit the host's nixpkgs.config (straylight's Vulkan build never hit this).
pkgsUnstable = import inputs.nixpkgs-unstable {
system = pkgs.stdenv.hostPlatform.system;
config = { allowUnfree = true; cudaSupport = true; };
};
# llama.cpp with the CUDA backend, from unstable to match straylight's build (same model
# files, same router semantics). Built once on straylight and shipped with dixie-push
# --closure, or pulled from the cache when Dixie's link allows.
llama-cpp = pkgsUnstable.llama-cpp.override { cudaSupport = true; };
modelsDir = "/var/lib/llama-server/models";
# Dixie is the fleet's helper tier: everything that is latency-sensitive, small, and was
# starving on straylight's CPU (2026-09-21). One RTX 3060 (12 GB) today; card 2 joins
# when its fans are replaced.
#
# Router: the three models that must never be swapped out — approval guardian / Honcho
# deriver (9B), Honcho dialectic (4B, non-thinking), Honcho embeddings — resident together:
# 5.6 + 2.5 + 0.7 GB weights, ~3 GB left for KV. Vision does NOT go here: 5 GB + 1.2 GB
# mmproj does not fit beside them, so it gets its own on-demand process below.
llamaModelsIni = pkgs.writeText "llama-models.ini" ''
version = 1
[*]
jinja = true
flash-attn = on
cache-type-k = q8_0
cache-type-v = q8_0
n-gpu-layers = 999
; three resident models share 12 GB: keep contexts modest
ctx-size = 16384
parallel = 2
cache-ram = 512
; Ornith 1.5 9B, refusal-ablated. Hermes auxiliary model (approval guardian, title,
; web_extract) for all agents and Honcho's deriver/summary/dream model.
[ornith-1.5-9b-uncensored]
model = ${modelsDir}/ornith-1.5-9b-uncensored/Ornith-1.5-9B-uncensored.Q4_K_M.gguf
; Runaway guard carried over from straylight: helper tasks need short answers.
reasoning-budget = 2048
n-predict = 4096
; Honcho dialectic: non-thinking instruct model so answers are short and fast.
[honcho-dialectic]
model = ${modelsDir}/qwen3-4b-instruct-2507/Qwen3-4B-Instruct-2507-Q4_K_M.gguf
n-predict = 1024
; Honcho embeddings (1024-dim; EMBEDDING_VECTOR_DIMENSIONS=1024 on rift).
[honcho-embed]
model = ${modelsDir}/qwen3-embedding-0.6b/Qwen3-Embedding-0.6B-Q8_0.gguf
embedding = true
pooling = cls
ubatch-size = 8192
ctx-size = 8192
parallel = 4
'';
# Vision (auxiliary.vision for all agents): its own server on :11433, loaded on first
# request and dropped again after an idle hour so the router's three keep their VRAM.
visionModelsIni = pkgs.writeText "llama-vision.ini" ''
version = 1
[*]
jinja = true
flash-attn = on
n-gpu-layers = 999
ctx-size = 16384
parallel = 1
sleep-idle-seconds = 3600
[qwen3-vl-8b-abliterated]
model = ${modelsDir}/qwen3-vl-8b-abliterated/Qwen3-VL-8B-Instruct-abliterated-v2.0.Q4_K_M.gguf
mmproj = ${modelsDir}/qwen3-vl-8b-abliterated/Qwen3-VL-8B-Instruct-abliterated-v2.0.mmproj-f16.gguf
'';
llamaService = { description, port, preset, modelsMax }: {
inherit description;
wantedBy = [ "multi-user.target" ];
after = [ "network-online.target" "var-lib-llama\\x2dserver-models.mount" ];
wants = [ "network-online.target" ];
environment.HOME = "/var/lib/llama-server";
serviceConfig = {
ExecStart = ''
${llama-cpp}/bin/llama-server \
--host 0.0.0.0 \
--port ${toString port} \
--models-preset ${preset} \
--models-max ${toString modelsMax} \
--timeout 7200
'';
User = "llama-server";
Group = "llama-server";
StateDirectory = "llama-server";
WorkingDirectory = "/var/lib/llama-server";
SupplementaryGroups = [ "video" ];
Restart = "on-failure";
RestartSec = "10s";
TimeoutStartSec = "600";
};
};
in
{
imports = [
./hardware-configuration.nix
./disk-config.nix
];
config = {
# Headless. The base configuration.nix has no desktop; nothing GUI is added here.
boot.loader.systemd-boot.enable = true;
boot.loader.efi.canTouchEfiVariables = true;
# 6.12 LTS: mt7921u (the AX9L USB Wi-Fi) is in-tree from 5.18; avoids the 6.17/6.18
# mt7921 monitor-mode regression for good measure.
boot.kernelPackages = pkgs.linuxPackages_6_12;
# NVIDIA, compute only: no X, no modesetting needed for CUDA.
hardware.graphics.enable = true;
services.xserver.videoDrivers = [ "nvidia" ];
hardware.nvidia = {
open = false;
modesetting.enable = false;
nvidiaPersistenced = true;
package = config.boot.kernelPackages.nvidiaPackages.production;
};
hardware.enableRedistributableFirmware = true;
# Power cap for the 3060(s): thermal hour on 2026-09-21 held 75 °C at 140 W, case closed.
systemd.services.nvidia-power-cap = {
description = "Cap GPU power (thermals in the Aurora chassis)";
wantedBy = [ "multi-user.target" ];
after = [ "nvidia-persistenced.service" ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
ExecStart = "${config.hardware.nvidia.package.bin}/bin/nvidia-smi -pl 140";
};
};
# Headless and Wi-Fi only: there is no console to add a key from after install, so the
# keys go in with the system. Same set the live ISO used (~/src/dixie-iso).
users.users.kyle.openssh.authorizedKeys.keyFiles = [ ./authorized_keys.pub ];
users.users.root.openssh.authorizedKeys.keyFiles = [ ./authorized_keys.pub ];
users.users.llama-server = { isSystemUser = true; group = "llama-server"; home = "/var/lib/llama-server"; };
users.groups.llama-server = { };
# Model weights live on the USB cache stick (ext4, label dixie-cache), populated from
# straylight with dixie-push. Both servers wait for the mount.
fileSystems."/mnt/cache" = {
device = "/dev/disk/by-label/dixie-cache";
fsType = "ext4";
options = [ "noatime" "nofail" "x-systemd.device-timeout=30s" ];
};
fileSystems."/var/lib/llama-server/models" = {
device = "/mnt/cache/models";
fsType = "none";
options = [ "bind" "nofail" "x-systemd.requires=/mnt/cache" ];
};
systemd.services.llama-server = llamaService {
description = "llama.cpp router: helper tier (9B guardian/deriver, dialectic, embeddings)";
port = 11434; preset = llamaModelsIni; modelsMax = 3;
};
systemd.services.llama-vision = llamaService {
description = "llama.cpp vision server (qwen3-vl-8b, on demand)";
port = 11433; preset = visionModelsIni; modelsMax = 1;
};
# Wi-Fi only host. NetworkManager comes from the base config; the AX9L (mt7921u) becomes
# the primary once plugged in and the internal CNVi radio stays as fallback. Connection
# profiles are copied from the live-USB session at install time (see the install notes),
# not stored in this repo.
networking.networkmanager.wifi.backend = "wpa_supplicant";
# Reachable only over the tailnet: 11434 (router) and 11433 (vision) on tailscale0, and
# SSH from the LAN (base config opens 22) for bring-up.
networking.firewall.interfaces.tailscale0.allowedTCPPorts = [ 11433 11434 ];
environment.systemPackages = with pkgs; [
llama-cpp
pciutils usbutils lm_sensors dmidecode nvme-cli smartmontools ethtool iw wpa_supplicant
htop tmux git jq curl rsync nvtopPackages.nvidia
];
};
}
+38
View File
@@ -0,0 +1,38 @@
## (dixie) disko config
## Alienware Aurora R11: Samsung PM9A1 1 TB NVMe is the system disk. The 6 TB SATA HDD and the
## USB cache stick are left alone here (the stick is mounted by label in default.nix).
## Unencrypted on purpose: headless box that must come back from a power cut with nobody
## at the console to unlock it.
{ lib, ... }:
{
disko.devices.disk.main = {
type = "disk";
device = "/dev/disk/by-id/nvme-eui.36355630521205220025384500000001";
content = {
type = "gpt";
partitions = {
ESP = {
size = "1G";
type = "EF00";
content = {
type = "filesystem";
format = "vfat";
mountpoint = "/boot";
mountOptions = [ "umask=0077" ];
};
};
root = {
size = "100%";
content = {
type = "filesystem";
format = "ext4";
mountpoint = "/";
mountOptions = [ "noatime" ];
};
};
};
};
};
}
+18
View File
@@ -0,0 +1,18 @@
# Do not modify this file! It was generated by nixos-generate-config
# and may be overwritten by future invocations. Please make changes
# to /etc/nixos/configuration.nix instead.
{ config, lib, pkgs, modulesPath, ... }:
{
imports =
[ (modulesPath + "/installer/scan/not-detected.nix")
];
boot.initrd.availableKernelModules = [ "xhci_pci" "ahci" "nvme" "usbhid" "usb_storage" "sd_mod" ];
boot.initrd.kernelModules = [ ];
boot.kernelModules = [ "kvm-intel" ];
boot.extraModulePackages = [ ];
nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux";
hardware.cpu.intel.updateMicrocode = lib.mkDefault config.hardware.enableRedistributableFirmware;
}