76 lines
1.8 KiB
YAML
76 lines
1.8 KiB
YAML
apiVersion: apps/v1
|
|
kind: Deployment
|
|
metadata:
|
|
name: openclaw-brain-vllm
|
|
namespace: customer1
|
|
labels:
|
|
app: openclaw-brain
|
|
spec:
|
|
replicas: 1
|
|
selector:
|
|
matchLabels:
|
|
app: openclaw-brain
|
|
template:
|
|
metadata:
|
|
labels:
|
|
app: openclaw-brain
|
|
spec:
|
|
nodeSelector:
|
|
cloud.google.com/gke-accelerator: "nvidia-a100-80gb"
|
|
tolerations:
|
|
- key: "nvidia.com/gpu-a100-80gb"
|
|
operator: "Equal"
|
|
value: "present"
|
|
effect: "NoSchedule"
|
|
containers:
|
|
- name: vllm-brain
|
|
image: vllm/vllm-openai:latest
|
|
command: ["python3", "-m", "vllm.entrypoints.openai.api_server"]
|
|
env:
|
|
- name: VLLM_ATTENTION_BACKEND
|
|
value: "XFORMERS"
|
|
- name: HF_TOKEN
|
|
value: ""
|
|
- name: VLLM_TOKENIZER_MODE
|
|
value: "auto"
|
|
args:
|
|
- "--model"
|
|
- "dphn/Dolphin-Mistral-24B-Venice-Edition" # Uncensored 24B (Native)
|
|
- "--dtype"
|
|
- "auto"
|
|
- "--max-model-len"
|
|
- "8192"
|
|
- "--trust-remote-code"
|
|
- "--gpu-memory-utilization"
|
|
- "0.95"
|
|
ports:
|
|
- containerPort: 8000
|
|
resources:
|
|
limits:
|
|
nvidia.com/gpu: 1
|
|
memory: "32Gi"
|
|
cpu: "8"
|
|
requests:
|
|
nvidia.com/gpu: 1
|
|
memory: "16Gi"
|
|
cpu: "4"
|
|
volumeMounts:
|
|
- name: model-cache
|
|
mountPath: /root/.cache/huggingface
|
|
volumes:
|
|
- name: model-cache
|
|
emptyDir: {}
|
|
---
|
|
apiVersion: v1
|
|
kind: Service
|
|
metadata:
|
|
name: openclaw-brain-service
|
|
namespace: customer1
|
|
spec:
|
|
selector:
|
|
app: openclaw-brain
|
|
ports:
|
|
- protocol: TCP
|
|
port: 8000
|
|
targetPort: 8000
|
|
type: ClusterIP
|