LM/vLLM: Difference between revisions

From Fundamental Ramen
< LM
Jump to navigation Jump to search
Line 68: Line 68:
Python 3.12.3
Python 3.12.3
</pre>
</pre>
|-
| valign="top" |
<syntaxhighlight lang="bash">
pip list
</syntaxhighlight>
|
<div style="max-height:250px; overflow: scroll-y;">
<pre>
Package                                  Version      Editable project location                        12:20 [237/1957]
---------------------------------------- ------------- -------------------------------------
absl-py                                  2.5.0
accelerate                              1.14.0
agent-detector                          2.0.0
aiohappyeyeballs                        2.7.1
aiohttp                                  3.14.3
aiosignal                                1.4.0
albucore                                0.0.24
albumentations                          2.0.8
annotated-doc                            0.0.5
annotated-types                          0.8.0
anthropic                                0.122.0
anyio                                    4.14.2
apache-tvm-ffi                          0.1.13.post3
arctic_inference                        0.2.0
astor                                    0.8.1
attrs                                    26.1.0
audioread                                3.1.0
auto-round-lib                          0.14.2
blake3                                  1.0.9
blobfile                                3.2.0
bm25s                                    0.3.10
bounded-pool-executor                    0.0.3
cachetools                              7.1.7
cbor2                                    6.1.4
certifi                                  2026.7.22
cffi                                    2.1.1
chardet                                  6.0.0.post1
charset-normalizer                      3.5.1
chz                                      0.4.0
click                                    8.4.2
cloudpickle                              3.1.2
cmake                                    4.4.3
colorama                                0.4.6
compressed-tensors                      0.17.0
coverage                                7.15.4
cryptography                            50.0.0
DataProperty                            1.1.1
datasets                                5.0.1
decorator                                5.3.1
defusedxml                              0.7.1
depyf                                    0.20.0
detect-installer                        0.1.0
dill                                    0.4.1
distro                                  1.9.0
dnspython                                2.8.0
docker                                  7.2.0
docopt                                  0.6.2
docstring_parser                        0.18.0
dpcpp-cpp-rt                            2026.0.0
einops                                  0.8.2
email-validator                          2.3.0
</pre>
</div>
|}
|}



Revision as of 04:22, 18 September 2026

Verify environments in container

docker run -it --rm \
  --device=/dev/dri \
  --group-add video \
  -v /dev/dri/by-path:/dev/dri/by-path \
  --entrypoint bash \
  vllm/vllm-openai-xpu:latest
Command Result
lsb_release -a
No LSB modules are available.
Distributor ID: Ubuntu
Description:    Ubuntu 24.04.4 LTS
Release:        24.04
Codename:       noble
xpu-smi
+----------------------------------------------+----------------------------+------------------------+
|            Intel XPU-SMI v2.1    Driver: 9913356BC2F8B14798CAB3D    Level Zero: 1.32.0             |
+----------------------------------------------+----------------------------+------------------------+
| GPU  Name                    Persistence-M   | Bus-Id            Disp.A   |   Volatile Uncorr. ECC |
| Fan              Temp  Pwr:Usage/Cap         | Memory-Usage               |   GPU-Util  Compute M. |
+----------------------------------------------+----------------------------+------------------------+
|   0  Intel(R) Arc(TM) Pro B  Off             | 0000:03:00.0      Off      |               Disabled |
|      70 Graphics                             |                            |                        |
| 0%                N/A  0W / 230W             | 42MiB / 32656MiB           |      N/A       Default |
+----------------------------------------------+----------------------------+------------------------+

+-------+-------+--------+----------------+----------------------------------------------------------+
| Processes:                                                                                         |
|   GPU     PID    Type    Process Name                                             GPU Memory Usage |
+-------+-------+--------+----------------+----------------------------------------------------------+
|                                  No running GPU processes found.                                   |
+-------+-------+--------+----------------+----------------------------------------------------------+
which python
/opt/venv/bin/python
python -V
Python 3.12.3
pip list
Package                                  Version       Editable project location                         12:20 [237/1957]
---------------------------------------- ------------- -------------------------------------
absl-py                                  2.5.0
accelerate                               1.14.0
agent-detector                           2.0.0
aiohappyeyeballs                         2.7.1
aiohttp                                  3.14.3
aiosignal                                1.4.0
albucore                                 0.0.24
albumentations                           2.0.8
annotated-doc                            0.0.5
annotated-types                          0.8.0
anthropic                                0.122.0
anyio                                    4.14.2
apache-tvm-ffi                           0.1.13.post3
arctic_inference                         0.2.0
astor                                    0.8.1
attrs                                    26.1.0
audioread                                3.1.0
auto-round-lib                           0.14.2
blake3                                   1.0.9
blobfile                                 3.2.0
bm25s                                    0.3.10
bounded-pool-executor                    0.0.3
cachetools                               7.1.7
cbor2                                    6.1.4
certifi                                  2026.7.22
cffi                                     2.1.1
chardet                                  6.0.0.post1
charset-normalizer                       3.5.1
chz                                      0.4.0
click                                    8.4.2
cloudpickle                              3.1.2
cmake                                    4.4.3
colorama                                 0.4.6
compressed-tensors                       0.17.0
coverage                                 7.15.4
cryptography                             50.0.0
DataProperty                             1.1.1
datasets                                 5.0.1
decorator                                5.3.1
defusedxml                               0.7.1
depyf                                    0.20.0
detect-installer                         0.1.0
dill                                     0.4.1
distro                                   1.9.0
dnspython                                2.8.0
docker                                   7.2.0
docopt                                   0.6.2
docstring_parser                         0.18.0
dpcpp-cpp-rt                             2026.0.0
einops                                   0.8.2
email-validator                          2.3.0

compose (Works!)

services:
  vllm-xpu:
    image: vllm/vllm-openai-xpu:latest
    container_name: vllm-xpu-qwen38
    privileged: true
    network_mode: host
    shm_size: "32g"

    devices:
      - /dev/dri:/dev/dri

    volumes:
      - ${HOME}/.cache/huggingface:/root/.cache/huggingface

    environment:
      - HF_HOME=/root/.cache/huggingface
      - HF_TOKEN=${HF_TOKEN:-}
      - no_proxy=localhost,127.0.0.1
      - ZE_FLAT_DEVICE_HIERARCHY=COMPOSITE
      - ZE_AFFINITY_MASK=0
      - VLLM_WORKER_MULTIPROC_METHOD=spawn
      - PYTORCH_ALLOC_CONF=expandable_segments:True
      - VLLM_XPU_ENABLE_XPU_GRAPH=1

    ports:
      - "9000:9000"

    entrypoint: ["vllm", "serve"]
    command:
      - "SergiioB/Qwen3.8-27B-GPTQ-Int4-sym-G128-MTP-BF16"
      - "--served-model-name=qwen38"
      - "--port=9000"
      - "--host=0.0.0.0"
      - "--quantization=gptq"
      - "--dtype=float16"
      - "--max-model-len=131072"
      - "--gpu-memory-utilization=0.88"
      - "--kv-cache-dtype=fp8"
      - "--max-num-seqs=1"
      - "--max-num-batched-tokens=8192"
      - "--trust-remote-code"
      - "--enable-auto-tool-choice"
      - "--tool-call-parser=qwen3_xml"
      - "--reasoning-parser=qwen3"
      - "--limit-mm-per-prompt={\"image\":10}"
      - "--speculative-config={\"method\":\"mtp\",\"num_speculative_tokens\":4}"

    restart: no
sudo usermod -aG docker $USER
docker compose up

docker.io

sudo apt install docker.io
sudo docker pull intel/vllm

docker run -t -d --shm-size 10g --net=host --ipc=host --privileged \
  -v /dev/dri/by-path:/dev/dri/by-path --name=vllm-test \
  --device /dev/dri:/dev/dri --entrypoint=/bin/bash intel/vllm:latest

source

uv venv --python 3.12 --seed --managed-python
source .venv/bin/activate

git clone https://github.com/vllm-project/vllm.git
cd vllm
pip install --upgrade pip
pip install -v -r requirements/xpu.txt

pip uninstall -y triton triton-xpu
pip install triton-xpu==3.7.2 --extra-index-url https://download.pytorch.org/whl/xpu

VLLM_TARGET_DEVICE=xpu pip install --no-build-isolation -e . -v

# check device
python -c "import torch; print(torch.xpu.is_available()); print(torch.xpu.device_count())"

# download model
hf download ciocan/gemma-4-E4B-it-W4A16

# Run
vllm serve unsloth/gemma-4-E4B-it-qat-GGUF --device xpu

TODO