apiVersion: apps/v1
kind: Deployment
metadata:
  name: qwen3-30b-a3b  # normalized: dots replaced with dashes
  namespace: your-namespace
  labels:
    app: qwen3-30b-a3b  # normalized: dots replaced with dashes
spec:
  replicas: 1
  selector:
    matchLabels:
      app: qwen3-30b-a3b  # normalized: dots replaced with dashes
  template:
    metadata:
      labels:
        app: qwen3-30b-a3b  # normalized: dots replaced with dashes
    spec:
      nodeName: <your-node-name>
      volumes:
        - name: qwen3-30b-a3b-pvc
          persistentVolumeClaim:
            claimName: qwen3-30b-a3b-pvc
        # vLLM needs to access the host's shared memory for tensor parallel inference.
        - name: shm
          emptyDir:
            medium: Memory
            sizeLimit: "2Gi"
      containers:
        - name: vllm-container
          image: vllm/vllm-openai:latest
          imagePullPolicy: IfNotPresent #Always
          command: ["/bin/sh", "-c"]
          args:
            - >

             vllm serve /mnt/OSSbuket/Qwen3-30B-A3B
             --host 0.0.0.0
             --port 8000
             --served-model-name Qwen3-30B-A3B
             --dtype bfloat16
             --max-model-len 4096
             --gpu-memory-utilization 0.90
             --quantization bitsandbytes
             --load-format bitsandbytes
             --trust-remote-code
             --enforce-eager
             --enable-chunked-prefill
             --enable-auto-tool-choice
             --tool-call-parser hermes




          ports:
            - containerPort: 8000
              protocol: TCP
          resources:
            limits:
              cpu: "10"
              memory: 40G
              nvidia.com/gpu: "1"
            requests:
              cpu: "4"
              memory: 20G
              nvidia.com/gpu: "1"
          env:
            - name: PYTORCH_CUDA_ALLOC_CONF
              value: expandable_segments:True
          volumeMounts:
            - name: qwen3-30b-a3b-pvc
              mountPath: /mnt/OSSbuket/Qwen3-30B-A3B
            - name: shm
              mountPath: /dev/shm
          livenessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 300
            periodSeconds: 30
            timeoutSeconds: 10
            failureThreshold: 3
          readinessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 300
            periodSeconds: 10
            timeoutSeconds: 10
            failureThreshold: 3
      restartPolicy: Always
      terminationGracePeriodSeconds: 30
      dnsPolicy: ClusterFirst
      schedulerName: default-scheduler
  strategy:
    type: RollingUpdate
    rollingUpdate:
      maxUnavailable: 25%
      maxSurge: 25%
  revisionHistoryLimit: 10
  progressDeadlineSeconds: 600
---
apiVersion: v1
kind: Service
metadata:
  name: qwen3-30b-a3b
  namespace: your-namespace
spec:
  type: ClusterIP
  ports:
    - name: http
      port: 8000
      protocol: TCP
      targetPort: 8000
  selector:
    app: qwen3-30b-a3b