docker run --rm -it --gpus all --ipc=host --network host
-v ~/.cache/huggingface:/root/.cache/huggingface
nvcr.io/nvidia/vllm:25.09
vllm serve "nvidia/Qwen3.8-27B-NVFP4"
--host 0.0.0.0
--port 8000
--gpu-memory-utilization 0.45
--max-model-len 65536
--kv-cache-dtype fp8
--reasoning-parser qwen3
--tool-call-parser qwen3_xml
--trust-remote-code

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