vLLM Production Install — RTX PRO 6000 Blackwell — systemd, no Docker
Target: unsloth/Qwen3.6-27B-NVFP4, full 262144 context, single GPU
systemd-managed, no Docker
Correct NVFP4 / SM120 kernel path (cute-DSL, confirmed — not Marlin fallback)
Full 262,144 native context, single GPU
fp8 KV cache (speed-prioritized)
MTP speculative decoding, num_speculative_tokens: 2
Reasoning output separated via --reasoning-parser qwen3
Structured tool calls via --enable-auto-tool-choice + --tool-call-parser qwen3_coder
1. Install Nvidia drivers
sudo apt-get --purge remove -y ' *nvidia*'
sudo add-apt-repository ppa:graphics-drivers/ppa
sudo apt install linux-headers-$( uname -r) build-essential dkms
sudo apt update
sudo apt install ubuntu-drivers-common
sudo ubuntu-drivers devices
sudo apt install apt install nvidia-driver-610-open
sudo apt install cuda-nvcc-13-3
1. Remove old CUDA versions
sudo apt-get --purge remove -y ' cuda-toolkit-*' ' cuda-*'
sudo apt-get autoremove -y
sudo apt-get autoclean -y
sudo rm -f /etc/profile.d/cuda.sh
3. Install CUDA 13.3 toolkit
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64/cuda-keyring_1.1-1_all.deb
sudo dpkg -i cuda-keyring_1.1-1_all.deb
sudo apt-get update
sudo apt-get install -y cuda-toolkit-13-3
rm -f cuda-keyring_1.1-1_all.deb
cat << 'EOF ' | sudo tee /etc/profile.d/cuda.sh
export PATH=/usr/local/cuda-13.3/bin:$PATH
export LD_LIBRARY_PATH=/usr/local/cuda-13.3/lib64:$LD_LIBRARY_PATH
EOF
source /etc/profile.d/cuda.sh
4. Verify driver + toolkit
nvidia-smi
nvcc --version
sudo mkdir -p /opt/uv-python
sudo chown " $USER " :" $USER " /opt/uv-python
export UV_PYTHON_INSTALL_DIR=/opt/uv-python
uv python install 3.13
sudo mkdir -p /opt/vllm-env
sudo chown " $USER " :" $USER " /opt/vllm-env
uv venv /opt/vllm-env --python 3.13
source /opt/vllm-env/bin/activate
uv pip install " vllm>=0.25.0" " flashinfer-python>=0.6.13" " nvidia-cutlass-dsl>=4.5.2" --torch-backend=auto
deactivate
sudo chmod -R o+rX /opt/uv-python
6. Verify NVFP4 / SM120 kernel path (must print True True)
/opt/vllm-env/bin/python -c " import torch; from vllm.utils.flashinfer import has_flashinfer_b12x_gemm as g, has_flashinfer_b12x_moe as m; print(torch.cuda.get_device_capability(), g(), m())"
7. Download model weights
sudo mkdir -p /opt/models/Qwen3.6-27B-NVFP4 /opt/models/hf-cache
sudo chown -R " $USER " :" $USER " /opt/models
hf download unsloth/Qwen3.6-27B-NVFP4 --local-dir /opt/models/Qwen3.6-27B-NVFP4
8. Service user + permissions
sudo useradd --system --no-create-home --shell /usr/sbin/nologin vllm
sudo chown -R vllm:vllm /opt/vllm-env /opt/models
sudo mkdir -p /var/lib/vllm
sudo chown vllm:vllm /var/lib/vllm
sudo mkdir -p /etc/vllm
cat << 'EOF ' | sudo tee /etc/vllm/vllm.env
HOME=/var/lib/vllm
TRITON_CACHE_DIR=/var/lib/vllm/.triton-cache
HF_HOME=/opt/models/hf-cache
VLLM_API_KEY=changeme
FLASHINFER_CUDA_ARCH_LIST=12.0f
FLASHINFER_FORCE_SM=120f
EOF
sudo chmod 600 /etc/vllm/vllm.env
sudo chown vllm:vllm /etc/vllm/vllm.env
10. systemd unit — full 262144 context, single card, NVFP4 auto-backend
cat << 'EOF ' | sudo tee /etc/systemd/system/vllm.service
[Unit]
Description=vLLM server - Qwen3.6-27B-NVFP4 - full context
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
User=vllm
Group=vllm
EnvironmentFile=/etc/vllm/vllm.env
Environment=PATH=/usr/local/cuda-13.3/bin:/usr/bin:/bin
Environment=LD_LIBRARY_PATH=/usr/local/cuda-13.3/lib64
WorkingDirectory=/opt/models
ExecStart=/opt/vllm-env/bin/vllm serve /opt/models/Qwen3.6-27B-NVFP4 \
--served-model-name Qwen3.6-27B-NVFP4 \
--trust-remote-code \
--dtype bfloat16 \
--max-model-len 262144 \
--kv-cache-dtype fp8 \
--gpu-memory-utilization 0.90 \
--enable-chunked-prefill \
--enable-prefix-caching \
--reasoning-parser qwen3 \
--enable-auto-tool-choice \
--tool-call-parser qwen3_coder \
--speculative-config '{"method":"mtp","num_speculative_tokens":2}' \
--host 127.0.0.1 \
--port 8000
Restart=on-failure
RestartSec=5
LimitNOFILE=1048576
LimitMEMLOCK=infinity
[Install]
WantedBy=multi-user.target
EOF
sudo systemctl daemon-reload
sudo systemctl enable --now vllm.service
sudo systemctl status vllm.service --no-pager
journalctl -u vllm.service -f
curl -s http://127.0.0.1:8000/v1/models -H " Authorization: Bearer changeme"
curl -s http://127.0.0.1:8000/v1/chat/completions \
-H " Authorization: Bearer changeme" \
-H " Content-Type: application/json" \
-d ' {
"model":"Qwen3.6-27B-NVFP4",
"messages":[{"role":"user","content":"What is the weather in Brussels?"}],
"tools":[{
"type":"function",
"function":{
"name":"get_weather",
"description":"Get current weather for a location",
"parameters":{"type":"object","properties":{"location":{"type":"string"}},"required":["location"]}
}
}],
"tool_choice":"auto"
}' | jq ' .choices[0].message.tool_calls'