brew install make
git clone https://github.com/ggml-org/llama.cpp
cd llama.cpp
cmake -B build
cmake --build build --config Release -j $(sysctl -n hw.ncpu)
brew install hf
hf auth login
open browser -> https://huggingface.co/settings/tokens get token
mkdir -p ~/llama.cpp/models cd ~/llama.cpp/models
hf download ggml-org/Qwen3.6-27B-MTP-GGUF
Qwen3.6-27B-MTP-Q8_0.gguf
--local-dir .
hf download RDson/Qwen3.6-27B-MTP-Q4_K_M-GGUF
Qwen3.6-27B-MTP-Q4_K_M.gguf
--local-dir .
or local dir inside llama.cpp
hf download Radamanthys11/Qwen3.6-27B-MTP-Q8_0-GGUF
Qwen3.6-27B-MTP-Q8_0.gguf
--local-dir ./models
only mtp - 16.5gb model
./build/bin/llama-server
-m models/qwen3.6-27b-mtp-Q4_K_M.gguf
-ngl 999
-c 4096
--spec-type draft-mtp
--spec-draft-n-max 2
--port 9000
only mtp - 30gb model
./build/bin/llama-server
-m ~/llama.cpp/models/Qwen3.6-27B-MTP-Q8_0.gguf
-ngl 999
-c 16384
--flash-attn on
--spec-type draft-mtp
--spec-draft-n-max 2
--host 0.0.0.0
--port 9000
mtp + ngram - 16.5gb model
./build/bin/llama-server
-m ~/llama.cpp/models/Qwen3.6-27B-MTP-Q4_K_M.gguf
-ngl 999
-c 16384
--flash-attn on
--spec-type ngram-mod,draft-mtp
--spec-draft-n-max 2
--spec-ngram-mod-n-match 24
--spec-ngram-mod-n-min 48
--spec-ngram-mod-n-max 64
--host 0.0.0.0
--port 9000
mtp + ngram - 30gb model
./build/bin/llama-server
-m ~/llama.cpp/models/Qwen3.6-27B-MTP-Q8_0.gguf
-ngl 999
-c 16384
--flash-attn on
--spec-type ngram-mod,draft-mtp
--spec-draft-n-max 2
--spec-ngram-mod-n-match 24
--spec-ngram-mod-n-min 48
--spec-ngram-mod-n-max 64
--host 0.0.0.0
--port 9000
curl http://localhost:9000/v1/chat/completions
-H "Content-Type: application/json"
-d '{"model":"qwen3","messages":[{"role":"user","content":"Hello"}]}'