new service files - updates to ll-stack-architecture.md

This commit is contained in:
2026-08-25 09:48:34 +02:00
parent 14f888843f
commit 6c8c89ffce
18 changed files with 644 additions and 34 deletions

View File

@@ -0,0 +1,32 @@
[Unit]
Description=llama.cpp server - alpha (meta/combat/action_eval on Ada #1)
After=network-online.target
[Service]
Type=simple
User=nyx-organs
Group=nimmerverse-agents
Environment="CUDA_VISIBLE_DEVICES=0"
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
ExecStart=/data/llama.cpp/build/bin/llama-server \
--model /data/models/prod/alpha/Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf \
--model-draft /data/models/prod/alpha/mtp-gemma-4-31B-it.gguf \
--spec-type draft-mtp \
-fa on \
--alias alpha \
--host 0.0.0.0 \
--port 31001 \
--n-gpu-layers 40 \
--ctx-size 65536 \
--cache-type-k q4_0 \
--cache-type-v q4_0 \
--batch-size 2048 \
--ubatch-size 512 \
--threads 8 \
--cont-batching \
--reasoning off
RestartSec=5
WorkingDirectory=/data
[Install]
WantedBy=multi-user.target

View File

@@ -0,0 +1,32 @@
[Unit]
Description=llama.cpp server - beta (meta/combat/action_eval on Ada #1)
After=network-online.target
[Service]
Type=simple
User=nyx-organs
Group=nimmerverse-agents
Environment="CUDA_VISIBLE_DEVICES=1"
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
ExecStart=/data/llama.cpp/build/bin/llama-server \
--model /data/models/prod/beta/Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf \
--model-draft /data/models/prod/beta/mtp-gemma-4-31B-it.gguf \
--spec-type draft-mtp \
-fa on \
--alias beta \
--host 0.0.0.0 \
--port 31002 \
--n-gpu-layers 40 \
--ctx-size 65536 \
--cache-type-k q4_0 \
--cache-type-v q4_0 \
--batch-size 2048 \
--ubatch-size 512 \
--threads 8 \
--cont-batching \
--reasoning off
RestartSec=5
WorkingDirectory=/data
[Install]
WantedBy=multi-user.target

View File

@@ -0,0 +1,31 @@
[Unit]
Description=llama.cpp server - delta
After=network-online.target
[Service]
Type=simple
User=nyx-organs
Group=nimmerverse-agents
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
ExecStart=/usr/local/bin/llama-server \
--model /data/models/prod/delta/Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf \
--model-draft /data/models/prod/delta/mtp-gemma-4-31B-it.gguf \
--spec-type draft-mtp \
-fa on \
--alias delta \
--host 0.0.0.0 \
--port 31001 \
--n-gpu-layers 60 \
--cache-type-k q4_0 \
--cache-type-v q4_0 \
--ctx-size 262144 \
--batch-size 2048 \
--ubatch-size 512 \
--threads 8 \
--cont-batching \
--reasoning off
RestartSec=5
WorkingDirectory=/data
[Install]
WantedBy=multi-user.target

View File

@@ -0,0 +1,31 @@
[Unit]
Description=llama.cpp server - epsilon
After=network-online.target
[Service]
Type=simple
User=nyx-organs
Group=nimmerverse-agents
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
ExecStart=/usr/local/bin/llama-server \
--model /data/models/prod/epsilon/Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf \
--model-draft /data/models/prod/epsilon/mtp-gemma-4-31B-it.gguf \
--spec-type draft-mtp \
-fa on \
--alias epsilon \
--host 0.0.0.0 \
--port 31002 \
--n-gpu-layers 60 \
--cache-type-k q4_0 \
--cache-type-v q4_0 \
--ctx-size 262144 \
--batch-size 2048 \
--ubatch-size 512 \
--threads 8 \
--cont-batching \
--reasoning off
RestartSec=5
WorkingDirectory=/data
[Install]
WantedBy=multi-user.target

View File

@@ -0,0 +1,31 @@
[Unit]
Description=llama.cpp server - gamma (memory/profile/vision/translator on RTX 3090)
After=network-online.target
[Service]
Type=simple
User=nyx-organs
Group=nimmerverse-agents
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
ExecStart=/data/llama.cpp/build/bin/llama-server \
--model /data/models/prod/gamma/Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf \
--model-draft /data/models/prod/gamma/mtp-gemma-4-31B-it.gguf \
--spec-type draft-mtp \
-fa on \
--alias gamma \
--host 0.0.0.0 \
--port 31003 \
--n-gpu-layers 40 \
--ctx-size 262144 \
--cache-type-k q4_0 \
--cache-type-v q4_0 \
--batch-size 2048 \
--ubatch-size 512 \
--threads 8 \
--cont-batching \
--reasoning off
RestartSec=5
WorkingDirectory=/data
[Install]
WantedBy=multi-user.target

View File

@@ -0,0 +1,23 @@
[Unit]
Description=llama.cpp server - zeta
After=network-online.target
[Service]
Type=simple
User=nyx-organs
Group=nimmerverse-agents
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
ExecStart=llama-server \
--model /data/models/prod/zeta/Huihui-Qwen3-VL-8B-Instruct-abliterated.Q8_0.gguf \
--alias zeta \
--mmproj /data/models/prod/zeta/zeta-mmproj-Q8_0.gguf \
--host 0.0.0.0 \
--port 31003 \
--n-gpu-layers 9999 \
--ctx-size 64000 \
--chat-template-kwargs '{"enable_thinking": false}'
RestartSec=5
WorkingDirectory=/data
[Install]
WantedBy=multi-user.target

View File

@@ -5,44 +5,42 @@ Chat endpoint: `/v1/chat/completions` with `{"chat_template_kwargs": {"enable_th
## Instances
| Instance | Node | GPU | Port | Model | Quant | VRAM | Vision |
|----------|------|-----|------|-------|-------|------|--------|
| **nyx** | theia | Blackwell 6000 | 31000 | self-trained ~41B | BF16 | ~84GB | ✅ |
| **alpha1** | dioscuri | Ada #1 | 31001 | Qwen3.6-27B-Fable-711 | IQ4_XS | 16.6GB | ❌ |
| **alpha2** | dioscuri | Ada #2 | 31002 | Qwen3.6-27B-Fable-711 | IQ4_XS | 16.6GB | ❌ |
| **gamma** | comfy-dev | RTX 3090 | 31003 | Qwen3.6-27B-Fable-711-MTP | Q4_K_S MTP | 19.9GB | ✅ |
| Instance | Node | GPU | Port | Model | Quant | Vision |
|----------|------|-----|------|-------|-------|------|
| **alpha** | dioscuri | Ada #1 | 31001 | Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf-MTP | Q4_K_M MTP
| **beta** | dioscuri | Ada #2 | 31002 | Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf-MTP | Q4_K_M MTP
| **gamma** | comfy-dev | RTX 3090 | 31003 | Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf-MTP | Q4_K_M MTP
| **delta** | theia | Blackwell 6000 | 31001 | Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf-MTP | Q4_K_M MTP
| **epsilon** | theia | Blackwell 6000 | 31002 | Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf-MTP | Q4_K_M MTP
| **zeta** | theia | Blackwell 6000 | 31003 | Huihui-Qwen3-VL-8B-Instruct-abliterated.Q8_0.gguf | Q8_0 | mmproj-Q8_0.gguf
## Role Mapping
## Service Files under:
/nimmerverse/nimmersky/guides-stack/
- **nyx (31000):** default dialogue, diary, agent_helper
- **alpha1 (31001):** meta, combat, action_evaluation
- **alpha2 (31002):** gamemaster, overflow
- **gamma (31003):** memory, character_profile, vision, universal_translator
llama-alpha.service
llama-beta.service
llama-delta.service
llama-epsilon.service
llama-gamma.service
llama-zeta.service
##
## Theia specific
## Endpoints for SkyrimNet
sudo nvidia-cuda-mps-control -d
```yaml
default: 10.0.40.21:31000/v1
diary: 10.0.40.21:31000/v1
agent_helper: 10.0.40.21:31000/v1
meta: 10.0.40.22:31001/v1
combat: 10.0.40.22:31001/v1
action_evaluation: 10.0.40.22:31001/v1
gamemaster: 10.0.40.22:31002/v1
memory: 10.0.30.124:31003/v1
character_profile: 10.0.30.124:31003/v1
vision: 10.0.30.124:31003/v1
universal_translator: 10.0.30.124:31003/v1
```
This command starts the CUDA Multi-Process Service (MPS) control daemon in the background.
## Notes
The control program used to manage the MPS server system.-d (Daemon mode): This flag launches the MPS control daemon in the background.What it does: Normally, if multiple Linux processes try to use the same GPU simultaneously, the GPU time-slices between them (hardware context switching), which adds overhead. MPS allows multiple different processes to share a single GPU context. This lets them execute kernels simultaneously on the same GPU, filling up underutilized GPU hardware and improving overall throughput.
##
- Throughput: gamma ~20 tok/s, alpha1 ~10 tok/s, alpha2 ~14 tok/s
- alpha1/alpha2 identical models (SHA256 confirmed)
- gamma has MTP quant; alpha1/alpha2 have non-MTP
- All llama.cpp services run as nyx-organs:nimmerverse-agents
- OStimNet: Fable-Fusion needs explicit x-rated vocabulary in system prompts
sudo nvidia-smi -i 0 -c EXCLUSIVE_PROCESS
## Service Files
This command changes the Compute Mode of a specific GPU so that only one process can attach to it at a time.
`~/tmp/skyrimnet-analysis/services/` — llama-gamma.service, llama-alpha1.service, llama-alpha2.service
nvidia-smi: The NVIDIA System Management Interface utility.
-i 0 (Index 0): Targets the specific GPU with an ID of 0.
-c EXCLUSIVE_PROCESS (Compute Mode): Sets the compute mode to "Exclusive Process"
.What it does: It restricts the GPU so that only one single CUDA context (process) can access it at any given time. If a second process tries to use GPU 0 while the first is running, it will immediately throw an out-of-memory or initialization error.
##
💡 How They Work TogetherAt first glance, these two commands seem to contradict each other: one allows sharing (MPS), and the other forbids sharing (EXCLUSIVE_PROCESS).However, they are designed to be used together. When you set a GPU to EXCLUSIVE_PROCESS, the MPS server counts as that single allowed process.EXCLUSIVE_PROCESS blocks all regular user applications from touching GPU 0 directly.MPS daemon steps in as the sole owner of GPU 0.Your applications then talk to the MPS daemon, which safely multiplexes their workloads onto the GPU together.This specific combination is the recommended way to deploy MPS because it prevents rogue, non-MPS processes from accidentally hijacking the GPU and ruining performance.