docker run --gpus all -it \
--ipc=host \
-e VLLM_DISABLE_PYNCCL=1 \
-p 127.0.0.1:8001:8000 \
-v /mnt/models:/models \
-v /mnt/lectures:/lectures \
absence209/vllm-sm70 \
--model /models/Qwen2.5-VL-7B-Instruct-AWQ \
--served-model-name qwen-vl \
--dtype float16 \
--quantization awq \
--tensor-parallel-size 2 \
--gpu-memory-utilization 0.80 \
--max-model-len 8192 \
--max-num-seqs 1 \
--max-num-batched-tokens 1024 \
--attention-backend TRITON_ATTN \
--disable-custom-all-reduce \
--trust-remote-code \
--limit-mm-per-prompt '{"image":1,"video":0}' \
--allowed-local-media-path /lectures \
--mm-processor-kwargs '{"max_pixels": 802816}' \
--host 0.0.0.0 \
--port 8000