apiVersion: apps/v1
kind: Deployment
metadata:
  name: llama-3-1-8b-instruct  # normalized: dots replaced with dashes
  namespace: your-namespace
  labels:
    app: llama-3-1-8b-instruct  # normalized: dots replaced with dashes
spec:
  replicas: 1
  selector:
    matchLabels:
      app: llama-3-1-8b-instruct  # normalized: dots replaced with dashes
  template:
    metadata:
      labels:
        app: llama-3-1-8b-instruct  # normalized: dots replaced with dashes
    spec:
      nodeName: <your-node-name>
      volumes:
        - name: llama-3-1-8b-instruct-pvc
          persistentVolumeClaim:
            claimName: llama-3-1-8b-instruct-pvc
        # vLLM needs to access the host's shared memory for tensor parallel inference.
        - name: shm
          emptyDir:
            medium: Memory
            sizeLimit: "2Gi"
      containers:
        - name: vllm-container
          image: vllm/vllm-openai:latest
          imagePullPolicy: IfNotPresent #Always
          command: ["/bin/sh", "-c"]
          args:
            - >
             vllm serve /mnt/OSSbuket/Llama-3.1-8B-Instruct
             --host 0.0.0.0
             --port 8000
             --served-model-name Llama-3.1-8B-Instruct
             --enable-chunked-prefill
             --max-num-batched-tokens 1024
             --max-model-len 4096
             --gpu-memory-utilization 0.90
             --tensor-parallel-size 1
             --dtype auto
             --enable-auto-tool-choice
             --tool-call-parser llama3_json


          ports:
            - containerPort: 8000
              protocol: TCP
          resources:
            limits:
              cpu: "10"
              memory: 20G
              nvidia.com/gpu: "1"
            requests:
              cpu: "2"
              memory: 6G
              nvidia.com/gpu: "1"
          volumeMounts:
            - name: llama-3-1-8b-instruct-pvc
              mountPath: /mnt/OSSbuket/Llama-3.1-8B-Instruct
            - name: shm
              mountPath: /dev/shm
          livenessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 120
            periodSeconds: 10
          readinessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 120
            periodSeconds: 5
      restartPolicy: Always
      terminationGracePeriodSeconds: 30
      dnsPolicy: ClusterFirst
      schedulerName: default-scheduler
  strategy:
    type: RollingUpdate
    rollingUpdate:
      maxUnavailable: 25%
      maxSurge: 25%
  revisionHistoryLimit: 10
  progressDeadlineSeconds: 600
---
apiVersion: v1
kind: Service
metadata:
  name: llama-3-1-8b-instruct
  namespace: your-namespace
spec:
  type: ClusterIP  # <-- CHANGED: Now internal only
  ports:
    - name: http
      port: 8000
      protocol: TCP
      targetPort: 8000
  selector:
    app: llama-3-1-8b-instruct  # normalized: dots replaced with dashes


#Llama-3.1-8B-Instruct