diff --git a/infrastructure/gpus/base/vllm-servers/rtx6000-vllm.yaml b/infrastructure/gpus/base/vllm-servers/rtx6000-vllm.yaml index 65374f0..7f58b97 100644 --- a/infrastructure/gpus/base/vllm-servers/rtx6000-vllm.yaml +++ b/infrastructure/gpus/base/vllm-servers/rtx6000-vllm.yaml @@ -40,7 +40,7 @@ spec: - --tokenizer-mode=hf - --gpu-memory-utilization=0.95 - --kv-cache-dtype=fp8_e5m2 - - --max-model-len=131072 + - --max-model-len=256389 - --enable-auto-tool-choice - --enable-chunked-prefill - --max-num-batched-tokens=8192 @@ -48,7 +48,7 @@ spec: - --trust-remote-code - --dtype=auto - --enable-prefix-caching - - --tool-call-parser=qwen3_xml + - --tool-call-parser=qwen3_coder - --reasoning-parser=qwen3 - --disable-custom-all-reduce ports: