new service files - updates to ll-stack-architecture.md
This commit is contained in:
31
guides-stack/llama-delta.service
Normal file
31
guides-stack/llama-delta.service
Normal file
@@ -0,0 +1,31 @@
|
||||
[Unit]
|
||||
Description=llama.cpp server - delta
|
||||
After=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=nyx-organs
|
||||
Group=nimmerverse-agents
|
||||
Environment="PATH=/usr/local/cuda/bin:/data/venvs/llama/bin:/usr/local/bin:/usr/bin:/usr/sbin:/bin"
|
||||
ExecStart=/usr/local/bin/llama-server \
|
||||
--model /data/models/prod/delta/Gemma4-31B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf \
|
||||
--model-draft /data/models/prod/delta/mtp-gemma-4-31B-it.gguf \
|
||||
--spec-type draft-mtp \
|
||||
-fa on \
|
||||
--alias delta \
|
||||
--host 0.0.0.0 \
|
||||
--port 31001 \
|
||||
--n-gpu-layers 60 \
|
||||
--cache-type-k q4_0 \
|
||||
--cache-type-v q4_0 \
|
||||
--ctx-size 262144 \
|
||||
--batch-size 2048 \
|
||||
--ubatch-size 512 \
|
||||
--threads 8 \
|
||||
--cont-batching \
|
||||
--reasoning off
|
||||
RestartSec=5
|
||||
WorkingDirectory=/data
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
Reference in New Issue
Block a user