diff --git a/apps/base/customer1/openclaw/kustomization.yaml b/apps/base/customer1/openclaw/kustomization.yaml index c7e532b..32de28c 100644 --- a/apps/base/customer1/openclaw/kustomization.yaml +++ b/apps/base/customer1/openclaw/kustomization.yaml @@ -7,3 +7,4 @@ resources: - service.yaml - openclaw-secrets.yaml - pg-cluster-openclaw.yaml + - vllm-gemma.yaml diff --git a/apps/base/customer1/openclaw/vllm-gemma.yaml b/apps/base/customer1/openclaw/vllm-gemma.yaml new file mode 100644 index 0000000..445b7c5 --- /dev/null +++ b/apps/base/customer1/openclaw/vllm-gemma.yaml @@ -0,0 +1,68 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: openclaw-brain-vllm + namespace: customer1 + labels: + app: openclaw-brain +spec: + replicas: 1 + selector: + matchLabels: + app: openclaw-brain + template: + metadata: + labels: + app: openclaw-brain + spec: + nodeSelector: + cloud.google.com/gke-accelerator: "nvidia-rtx-pro-6000" + tolerations: + - key: "nvidia.com/gpu-nvidia-rtx-pro-6000" + operator: "Equal" + value: "present" + effect: "NO_SCHEDULE" + containers: + - name: vllm-brain + image: vllm/vllm-openai:latest + command: ["python3", "-m", "vllm.entrypoints.openai.api_server"] + args: + - "--model" + - "cognitivecomputations/dolphin-2.9.4-qwen2.5-32b-AWQ" # Uncensored 32B Reasoning Heavyweight (Quantized) + - "--quantization" + - "awq" + - "--dtype" + - "half" + - "--max-model-len" + - "16384" + ports: + - containerPort: 8000 + resources: + limits: + nvidia.com/gpu: 1 + memory: "32Gi" + cpu: "8" + requests: + nvidia.com/gpu: 1 + memory: "16Gi" + cpu: "4" + volumeMounts: + - name: model-cache + mountPath: /root/.cache/huggingface + volumes: + - name: model-cache + emptyDir: {} +--- +apiVersion: v1 +kind: Service +metadata: + name: openclaw-brain-service + namespace: customer1 +spec: + selector: + app: openclaw-brain + ports: + - protocol: TCP + port: 8000 + targetPort: 8000 + type: ClusterIP