LM/vLLM

From Fundamental Ramen
Jump to navigation Jump to search

compose

services:
  vllm-xpu:
    image: vllm/vllm-openai-xpu:v0.30.0
    container_name: vllm-xpu-qwen38-exp

    devices:
      - /dev/dri:/dev/dri

    volumes:
      - ${HOME}/.cache/huggingface:/root/.cache/huggingface

    environment:
      - HF_HOME=/root/.cache/huggingface
      - HF_TOKEN=${HF_TOKEN:-}
      - no_proxy=localhost,127.0.0.1
      - ZE_FLAT_DEVICE_HIERARCHY=COMPOSITE
      - ZE_AFFINITY_MASK=0
      - VLLM_WORKER_MULTIPROC_METHOD=spawn
      - PYTORCH_ALLOC_CONF=expandable_segments:True
      - VLLM_XPU_ENABLE_XPU_GRAPH=1

    ports:
      - "9000:9000"

    entrypoint: ["vllm", "serve"]
    command:
      - "SergiioB/Qwen3.8-27B-GPTQ-Int4-sym-G128-MTP-BF16"
      - "--served-model-name=qwen38"
      - "--port=9000"
      - "--host=0.0.0.0"
      - "--quantization=gptq"
      - "--dtype=float16"
      - "--max-model-len=131072"
      - "--gpu-memory-utilization=0.88"
      - "--kv-cache-dtype=fp8"
      - "--max-num-seqs=1"
      - "--max-num-batched-tokens=8192"
      - "--trust-remote-code"
      - "--enable-auto-tool-choice"
      - "--tool-call-parser=qwen3_xml"
      - "--reasoning-parser=qwen3"
      - "--limit-mm-per-prompt={\"image\":10}"
      - "--speculative-config={\"method\":\"mtp\",\"num_speculative_tokens\":4}"

    restart: no

Verify environments in container

docker exec -it vllm-xpu-qwen38-prod bash

docker exec vllm-xpu-qwen38-prod bash -c "pip list | awk 'NR<3 || tolower(\$0) ~ /torch|intel|triton|vllm|oneccl|mkl|dpc|level/'"

docker exec vllm-xpu-qwen38-prod bash -c 'pip show vllm-xpu-kernels'

docker exec vllm-xpu-qwen38-prod bash -c 'lsb_release -a'

TODO

References