# multiplication-eval-deepseek-r1-distill-llama-70b.yaml

# Kubernetes Job to evaluate how well a language model solves basic
# multiplication problems using vLLM. The Job downloads and runs a Python script
# from SCRIPT_URL, which:
# - Downloads the specified Hugging Face model (MODEL_ID)
# - Generates 1,000 random multiplication problems
# - Queries the model for answers
# - Computes and prints accuracy to stdout

apiVersion: batch/v1
kind: Job
metadata:
  name: multiplication-eval-deepseek-r1-distill-llama-70b
  namespace: mk8s-docs-examples
spec:
  template:
    spec:
      runtimeClassName: nvidia
      tolerations:
        - key: "nvidia.com/gpu"
          operator: "Equal"
          value: "true"
          effect: "NoSchedule"
      containers:
        - name: vllm-eval-runner
          image: vllm/vllm-openai:v0.9.1
          command: ["/bin/bash"]
          args:
            - -c
            - |
              curl -L "$SCRIPT_URL" -o eval_multiplication.py && \
              python3 eval_multiplication.py "$MODEL_ID" --stdout
          env:
            - name: MODEL_ID
              value: deepseek-ai/DeepSeek-R1-Distill-Llama-70B
            - name: SCRIPT_URL
              value: https://docs.lambda.ai/assets/code/eval_multiplication.py
            - name: HF_HOME
              value: /root/.cache/huggingface
          volumeMounts:
            - name: hf-cache
              mountPath: /root/.cache/huggingface
            - name: dev-shm
              mountPath: /dev/shm
          resources:
            limits:
              memory: "32Gi"
              nvidia.com/gpu: "4"
      volumes:
        - name: hf-cache
          persistentVolumeClaim:
            claimName: huggingface-cache
        - name: dev-shm
          emptyDir:
            medium: Memory
            sizeLimit: 1Gi
      restartPolicy: Never
      backoffLimit: 2
      ttlSecondsAfterFinished: 300
