Last active
May 19, 2026 20:46
-
-
Save aneeshkp/d23c1be8139b939d3d888cd6e7def332 to your computer and use it in GitHub Desktop.
Deploy LLMInferenceService with local NVMe model cache on AKS
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/bin/bash | |
| # Deploy LLMInferenceService using pre-downloaded models from local NVMe storage | |
| # Models are in HuggingFace cache format at /mnt/local-nvme-storage/models/hub/ | |
| # | |
| # The operator normally generates "vllm serve /mnt/models" which breaks with HF cache | |
| # (symlinks in snapshots/ point to ../../blobs/ which is outside the mount boundary). | |
| # | |
| # Workaround: override the container command to use the model NAME instead of a path. | |
| # Mount the HF cache and set HF_HUB_CACHE so vLLM resolves from cache — no download. | |
| # Must include SSL flags since the operator's default command template is bypassed. | |
| # | |
| # Available models on NVMe (as of 2026-05-19): | |
| # | |
| # Model Size GPUs | |
| # --------------------------------------------------------------------------------------------------------------- | |
| # Qwen/Qwen2.5-3B-Instruct 5.8G 1 | |
| # RedHatAI/Llama-3.2-3B-Instruct-FP8 4.1G 1 | |
| # RedHatAI/Llama-3.2-3B-Instruct-FP8-dynamic 4.1G 1 | |
| # Qwen/Qwen3-30B-A3B-Instruct-2507-FP8 29.1G 1 | |
| # Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8 29.1G 1 | |
| # openai/gpt-oss-20b 38.5G 1-2 | |
| # zai-org/GLM-Z1-32B-0414 60.7G 1-2 | |
| # RedHatAI/Llama-3.3-70B-Instruct-FP8-dynamic 67.7G 2 | |
| # openai/gpt-oss-120b 182.3G 4 | |
| # Qwen/Qwen3-235B-A22B-Instruct-2507-FP8 220.2G 4-8 | |
| # zai-org/GLM-4.7-FP8 337.2G 8 | |
| # deepseek-ai/DeepSeek-V3-0324 641.3G 8+ | |
| # --- Configuration: change these for different models --- | |
| MODEL_NAME="Qwen/Qwen2.5-3B-Instruct" | |
| GPU_COUNT="1" | |
| MEMORY_LIMIT="16Gi" | |
| MEMORY_REQUEST="8Gi" | |
| DEPLOY_NAME="qwen-test" | |
| NAMESPACE="llm-test" | |
| # --------------------------------------------------------- | |
| # 1. Create namespace | |
| kubectl create namespace ${NAMESPACE} 2>/dev/null || true | |
| # 2. Copy pull secret | |
| kubectl create secret generic rhai-pull-secret \ | |
| --from-file=.dockerconfigjson=~/pull-secret.txt \ | |
| --type=kubernetes.io/dockerconfigjson \ | |
| -n ${NAMESPACE} 2>/dev/null || true | |
| # 3. Deploy | |
| kubectl apply -n ${NAMESPACE} -f - <<EOF | |
| apiVersion: serving.kserve.io/v1alpha2 | |
| kind: LLMInferenceService | |
| metadata: | |
| name: ${DEPLOY_NAME} | |
| spec: | |
| storageInitializer: | |
| enabled: false | |
| model: | |
| uri: hf://${MODEL_NAME} | |
| name: ${MODEL_NAME} | |
| replicas: 1 | |
| router: | |
| scheduler: | |
| template: | |
| imagePullSecrets: | |
| - name: rhai-pull-secret | |
| containers: | |
| - name: main | |
| - name: tokenizer | |
| route: {} | |
| gateway: {} | |
| template: | |
| imagePullSecrets: | |
| - name: rhai-pull-secret | |
| nodeSelector: | |
| agentpool: gpunp | |
| volumes: | |
| # Mount the entire HF cache directory | |
| - name: hf-cache | |
| hostPath: | |
| path: /mnt/local-nvme-storage/models | |
| type: Directory | |
| containers: | |
| - name: main | |
| # Override command to use model NAME (not /mnt/models path) | |
| # vLLM resolves the model from HF_HUB_CACHE via huggingface_hub | |
| # No snapshot SHA needed, no symlink issues | |
| command: | |
| - vllm | |
| - serve | |
| - ${MODEL_NAME} | |
| - --port | |
| - "8000" | |
| - --served-model-name | |
| - ${MODEL_NAME} | |
| - --disable-uvicorn-access-log | |
| - --enable-ssl-refresh | |
| - --ssl-certfile | |
| - /var/run/kserve/tls/tls.crt | |
| - --ssl-keyfile | |
| - /var/run/kserve/tls/tls.key | |
| env: | |
| - name: HF_HUB_CACHE | |
| value: /hf-cache/hub | |
| volumeMounts: | |
| - name: hf-cache | |
| mountPath: /hf-cache | |
| readOnly: true | |
| resources: | |
| limits: | |
| nvidia.com/gpu: "${GPU_COUNT}" | |
| memory: ${MEMORY_LIMIT} | |
| cpu: "4" | |
| requests: | |
| nvidia.com/gpu: "${GPU_COUNT}" | |
| memory: ${MEMORY_REQUEST} | |
| cpu: "2" | |
| EOF | |
| echo "" | |
| echo "Monitor: kubectl get llmisvc -n ${NAMESPACE} -w" | |
| echo "Pods: kubectl get pods -n ${NAMESPACE} -w" |
Author
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
Line 23 —
storageInitializer.enabled: falseSkips the storage-initializer init container that KServe injects to download the model. Without this, KServe runs an init container that downloads the model from
hf://even when the model is already on local storage.Line 26 —
uri: hf://Qwen/Qwen2.5-3B-InstructRequired by the CRD — cannot be omitted. With
storageInitializer.enabled: false, this value is only used by vLLM as the model identifier (translated to--model Qwen/Qwen2.5-3B-Instruct). No download happens.