Created
May 19, 2026 20:00
-
-
Save aneeshkp/a8d2ca681a9eadf810963b67b9b7116b to your computer and use it in GitHub Desktop.
Deploy LLMInferenceService with HuggingFace download (storage-initializer) on AKS
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/bin/bash | |
| # Deploy LLMInferenceService — download model from HuggingFace via storage-initializer | |
| # No local models needed. The storage-initializer init container downloads the model | |
| # before vLLM starts. | |
| # 1. Create namespace | |
| kubectl create namespace llm-test 2>/dev/null || true | |
| # 2. Copy pull secret | |
| kubectl create secret generic rhai-pull-secret \ | |
| --from-file=.dockerconfigjson=~/pull-secret.txt \ | |
| --type=kubernetes.io/dockerconfigjson \ | |
| -n llm-test 2>/dev/null || true | |
| # 3. (Optional) Create HuggingFace token secret for gated models | |
| # kubectl create secret generic hf-token \ | |
| # --from-literal=HF_TOKEN=hf_xxxxxxxxxxxxx \ | |
| # -n llm-test | |
| # 4. Deploy (using Qwen2.5-3B-Instruct — small, fast to download) | |
| kubectl apply -n llm-test -f - <<'EOF' | |
| apiVersion: serving.kserve.io/v1alpha2 | |
| kind: LLMInferenceService | |
| metadata: | |
| name: qwen-test | |
| spec: | |
| model: | |
| uri: hf://Qwen/Qwen2.5-3B-Instruct | |
| name: Qwen/Qwen2.5-3B-Instruct | |
| replicas: 1 | |
| router: | |
| scheduler: | |
| template: | |
| imagePullSecrets: | |
| - name: rhai-pull-secret | |
| containers: | |
| - name: main | |
| - name: tokenizer | |
| route: {} | |
| gateway: {} | |
| template: | |
| imagePullSecrets: | |
| - name: rhai-pull-secret | |
| nodeSelector: | |
| agentpool: gpunp | |
| containers: | |
| - name: main | |
| resources: | |
| limits: | |
| nvidia.com/gpu: "1" | |
| memory: 16Gi | |
| cpu: "4" | |
| requests: | |
| nvidia.com/gpu: "1" | |
| memory: 8Gi | |
| cpu: "2" | |
| EOF | |
| echo "" | |
| echo "The storage-initializer will download the model before vLLM starts." | |
| echo "This may take a few minutes depending on model size and network speed." | |
| echo "" | |
| echo "Monitor: kubectl get llmisvc -n llm-test -w" | |
| echo "Pods: kubectl get pods -n llm-test -w" | |
| echo "Logs: kubectl logs -n llm-test -l app=qwen-test -c storage-initializer -f" |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment