Merge pull request #53 from sirius0xdev/feat/keda-a100-scaling
feat: KEDA scaling for A100 vLLM (scale to 0 when idle)
This commit is contained in:
commit
d2feb008da
1 changed files with 21 additions and 0 deletions
21
apps/base/customer1/openclaw/keda-vllm.yaml
Normal file
21
apps/base/customer1/openclaw/keda-vllm.yaml
Normal file
|
|
@ -0,0 +1,21 @@
|
||||||
|
apiVersion: keda.sh/v1alpha1
|
||||||
|
kind: ScaledObject
|
||||||
|
metadata:
|
||||||
|
name: openclaw-brain-a100-scaling
|
||||||
|
namespace: customer1
|
||||||
|
spec:
|
||||||
|
scaleTargetRef:
|
||||||
|
name: openclaw-brain-vllm
|
||||||
|
kind: Deployment
|
||||||
|
service:
|
||||||
|
name: openclaw-brain-service
|
||||||
|
port: 8000
|
||||||
|
minReplicaCount: 0
|
||||||
|
maxReplicaCount: 1
|
||||||
|
cooldownPeriod: 900 # 15 minutes idle before scale-down
|
||||||
|
triggers:
|
||||||
|
- type: http
|
||||||
|
metadata:
|
||||||
|
path: "/v1/models"
|
||||||
|
host: "openclaw-brain-service.customer1.svc.cluster.local"
|
||||||
|
requestCount: "1" # Scale up on 1+ request (wake script handles cold start)
|
||||||
Loading…
Add table
Reference in a new issue