vllm serve google/gemma-4-E2B-it --host 0.0.0.0 --port 8000 \ --max-model-len 32768 --gpu-memory-utilization 0.90 \ --enable-auto-tool-choice --reasoning-parser gemma4 --tool-call-parser gemma4 \ --chat-template /app/vllm/examples/tool_chat_template_gemma4.jinja \ --limit-mm-per-prompt '{"image": 4, "audio": 0}' --async-scheduling