Created
September 26, 2026 21:46
-
-
Save dbsanfte/588e0d3618926a4f6d3ea5eeeff5b56a to your computer and use it in GitHub Desktop.
Benchmark Llaminar (change the models dir variable)
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env bash | |
| set -euo pipefail | |
| LLAMINAR_MODELS_DIR="/path/to/models" | |
| LLAMINAR_GPU=0 | |
| LLAMINAR_REV="557f0f97d15d447850b77b7ce4e91bb0663331b3" | |
| LLAMINAR_IMAGE="ghcr.io/llaminar/llaminar:develop-${LLAMINAR_REV}-avx2" | |
| LLAMINAR_MODELS_DIR="$(realpath "$LLAMINAR_MODELS_DIR")" | |
| LLAMINAR_RESULTS="$(mktemp -d "$PWD/llaminar-benchmark.XXXXXX")" | |
| MODELS=( | |
| "Qwen3.8-27B-IQ4_XS.gguf" | |
| "Qwen3.6-35B-A3B-UD-IQ3_S.gguf" | |
| ) | |
| for model in "${MODELS[@]}"; do | |
| test -r "$LLAMINAR_MODELS_DIR/$model" || { | |
| echo "Missing model: $LLAMINAR_MODELS_DIR/$model" >&2 | |
| exit 1 | |
| } | |
| done | |
| docker pull "$LLAMINAR_IMAGE" | |
| # Download the exact 512-token prompt, without adding a trailing newline. | |
| curl -fsSL \ | |
| "https://raw.githubusercontent.com/Llaminar/llaminar/${LLAMINAR_REV}/.agents/llama-cpp-comparison/assets/prompt-512-qwen38.json" \ | |
| | jq -jr '.prompt' >"$LLAMINAR_RESULTS/prompt.txt" | |
| # Run sequentially: one warmup and five measured requests per model. | |
| for model in "${MODELS[@]}"; do | |
| stem="${model%.gguf}" | |
| docker run --rm \ | |
| --user 0:0 \ | |
| --network host \ | |
| --ipc host \ | |
| --security-opt seccomp=unconfined \ | |
| --cap-add SYS_NICE \ | |
| --cap-add SYS_PTRACE \ | |
| --device=/dev/kfd \ | |
| --device=/dev/dri \ | |
| --mount "type=bind,src=$LLAMINAR_MODELS_DIR,dst=/models,readonly" \ | |
| --mount "type=bind,src=$LLAMINAR_RESULTS,dst=/results" \ | |
| -e HIP_VISIBLE_DEVICES="$LLAMINAR_GPU" \ | |
| -e LLAMINAR_BENCHMARK_WARMUP_ITERATIONS=1 \ | |
| -e LLAMINAR_BENCHMARK_ITERATIONS=5 \ | |
| "$LLAMINAR_IMAGE" benchmark \ | |
| -m "/models/$model" \ | |
| --only-backends rocm \ | |
| --auto-device-counts rocm=1 \ | |
| -c 4096 \ | |
| --prompt-file /results/prompt.txt \ | |
| -n 256 \ | |
| --temperature 0 \ | |
| --seed 42 \ | |
| --mtp \ | |
| --mtp-depth-policy dynamic \ | |
| --benchmark-json-output "/results/$stem.json" \ | |
| 2>&1 | tee "$LLAMINAR_RESULTS/$stem.log" | |
| # Validate the workload and print median prefill/decode throughput. | |
| jq -e ' | |
| def median: sort | .[length / 2 | floor]; | |
| if ( | |
| .success | |
| and (.perf_stats.enabled == false) | |
| and .mtp.request.adaptive_depth_enabled | |
| and (.prefix_cache.matched_tokens == 0) | |
| and (.iterations | length == 5) | |
| and all( | |
| .iterations[]; | |
| .tokens.prefill == 512 | |
| and .tokens.decode == 256 | |
| and .mtp.draft_steps > 0 | |
| ) | |
| ) | |
| then { | |
| model: .config.model_path, | |
| prefill_tok_s: | |
| ([.iterations[].throughput_tokens_per_sec.prefill] | median), | |
| decode_tok_s: | |
| ([.iterations[].throughput_tokens_per_sec.decode_after_prefill] | median) | |
| } | |
| else | |
| error("Benchmark did not match the requested workload") | |
| end | |
| ' "$LLAMINAR_RESULTS/$stem.json" | |
| done | |
| printf '\nLogs and full JSON results: %s\n' "$LLAMINAR_RESULTS" |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment