apiVersion: apps/v1
kind: Deployment
metadata:
  name: qwen3-32b-awq  # normalized: dots replaced with dashes
  namespace:  your-namespace
  labels:
    app: qwen3-32b-awq  # normalized: dots replaced with dashes
spec:
  replicas: 1
  selector:
    matchLabels:
      app: qwen3-32b-awq  # normalized: dots replaced with dashes
  template:
    metadata:
      labels:
        app: qwen3-32b-awq  # normalized: dots replaced with dashes
    spec:
      nodeName: <your-node-name>
      volumes:
        - name: qwen3-32b-awq-pvc
          persistentVolumeClaim:
            claimName: qwen3-32b-awq-pvc
        # vLLM needs to access the host's shared memory for tensor parallel inference.
        - name: shm
          emptyDir:
            medium: Memory
            sizeLimit: "2Gi"
      containers:
        - name: vllm-container
          image: vllm/vllm-openai:latest
          imagePullPolicy: IfNotPresent #Always
          command: ["/bin/sh", "-c"]
          args:
            - >
              vllm serve /mnt/OSSbuket/Qwen3-32B-AWQ
              --host 0.0.0.0
              --port 8000
              --served-model-name Qwen3-32B-AWQ
              --dtype half
              --tensor-parallel-size 1
              --quantization awq
              --kv-cache-dtype fp8
              --gpu-memory-utilization 0.85
              --swap-space 16
              --max-model-len 10240
              --max-num-batched-tokens 512
              --max-num-seqs 16
              --enable-chunked-prefill
              --enable-auto-tool-choice
              --tool-call-parser hermes

          ports:
            - containerPort: 8000
              protocol: TCP
          resources:
            limits:
              cpu: "10"
              memory: 20G
              nvidia.com/gpu: "1"
            requests:
              cpu: "2"
              memory: 6G
              nvidia.com/gpu: "1"
          volumeMounts:
            - name: qwen3-32b-awq-pvc
              mountPath: /mnt/OSSbuket/Qwen3-32B-AWQ
            - name: shm
              mountPath: /dev/shm
          livenessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 120
            periodSeconds: 10
          readinessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 120
            periodSeconds: 5
      restartPolicy: Always
      terminationGracePeriodSeconds: 30
      dnsPolicy: ClusterFirst
      schedulerName: default-scheduler
  strategy:
    type: RollingUpdate
    rollingUpdate:
      maxUnavailable: 25%
      maxSurge: 25%
  revisionHistoryLimit: 10
  progressDeadlineSeconds: 600
---
apiVersion: v1
kind: Service
metadata:
  name: qwen3-32b-awq
  namespace: your-namespace
spec:
  type: ClusterIP  # <-- CHANGED: Now internal only
  ports:
    - name: http
      port: 8000
      protocol: TCP
      targetPort: 8000
  selector:
    app: qwen3-32b-awq  # normalized: dots replaced with dashes




    # Remote deployment command example:
    # --chat-template "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\\nYou are a helpful assistant<|im_end|>\\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\\n' + message['content'] + '<|im_end|>' + '\\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\\n' }}{% endif %}"