Switch L4 vLLM to Dolphin Qwen2 7B AWQ for speed and uncensored compliance

This commit is contained in:
Sirius Claw 2026-04-20 20:47:30 +00:00
parent ca2f14f25b
commit e1cf661692

View file

@ -30,7 +30,8 @@ spec:
- name: HF_TOKEN
value: ""
args:
- --model=trohrbaugh/Qwen2.5-Coder-7B-Instruct-heretic
- --model=cognitivecomputations/dolphin-2.9.3-qwen2-7b-awq
- --quantization=awq
- --host=0.0.0.0
- --port=8000
- --tensor-parallel-size=1