docker run --rm --gpus all -p 8000:8000 \ vllm/vllm-openai:latest \ --model TheBloke/Mistral-7B-v0.1-AWQ \ --quantization awq __ __