vllm deployment.yaml
vllm-deployment.yaml
Deployment for a vLLM server that serves the
NousResearch/Hermes-3-Llama-3.1-8B model.
apiVersion: apps/v1
kind: Deployment
metadata:
name: vllm-server
namespace: mk8s-docs-examples
spec:
replicas: 1
selector:
matchLabels:
app: vllm-server
template:
metadata:
labels:
app: vllm-server
spec:
runtimeClassName: nvidia
tolerations:
- key: "nvidia.com/gpu"
operator: "Equal"
value: "true"
effect: "NoSchedule"
containers:
- name: vllm
image: vllm/vllm-openai:v0.9.1
command: ["/bin/bash", "-c"]
args:
- |
exec python3 -m vllm.entrypoints.openai.api_server
--model "$VLLM_MODEL"
ports:
- containerPort: 8000
env:
- name: HF_HOME
value: /root/.cache/huggingface
- name: VLLM_MODEL
value: NousResearch/Hermes-3-Llama-3.1-8B
volumeMounts:
- name: hf-cache
mountPath: /root/.cache/huggingface
- name: dev-shm
mountPath: /dev/shm
resources:
limits:
memory: "32Gi"
nvidia.com/gpu: "1"
volumes:
- name: hf-cache
persistentVolumeClaim:
claimName: huggingface-cache
- name: dev-shm
emptyDir:
medium: Memory
sizeLimit: 1Gi
restartPolicy: Always