LM/vLLM: Difference between revisions
< LM
| Line 60: | Line 60: | ||
docker exec vllm-xpu-qwen38-prod bash -c 'lsb_release -a' | docker exec vllm-xpu-qwen38-prod bash -c 'lsb_release -a' | ||
</syntaxhighlight> | </syntaxhighlight> | ||
{| class="wikitable" | |||
! Command || Results | |||
|- | |||
| <syntaxhightlight lang="bash"> | |||
</syntaxhightlight> | |||
| <syntaxhightlight lang="bash"> | |||
</syntaxhightlight> | |||
|} | |||
== TODO == | == TODO == | ||
Revision as of 06:07, 30 September 2026
compose
services:
vllm-xpu:
image: vllm/vllm-openai-xpu:v0.30.0
container_name: vllm-xpu-qwen38-exp
devices:
- /dev/dri:/dev/dri
volumes:
- ${HOME}/.cache/huggingface:/root/.cache/huggingface
environment:
- HF_HOME=/root/.cache/huggingface
- HF_TOKEN=${HF_TOKEN:-}
- no_proxy=localhost,127.0.0.1
- ZE_FLAT_DEVICE_HIERARCHY=COMPOSITE
- ZE_AFFINITY_MASK=0
- VLLM_WORKER_MULTIPROC_METHOD=spawn
- PYTORCH_ALLOC_CONF=expandable_segments:True
- VLLM_XPU_ENABLE_XPU_GRAPH=1
ports:
- "9000:9000"
entrypoint: ["vllm", "serve"]
command:
- "SergiioB/Qwen3.8-27B-GPTQ-Int4-sym-G128-MTP-BF16"
- "--served-model-name=qwen38"
- "--port=9000"
- "--host=0.0.0.0"
- "--quantization=gptq"
- "--dtype=float16"
- "--max-model-len=131072"
- "--gpu-memory-utilization=0.88"
- "--kv-cache-dtype=fp8"
- "--max-num-seqs=1"
- "--max-num-batched-tokens=8192"
- "--trust-remote-code"
- "--enable-auto-tool-choice"
- "--tool-call-parser=qwen3_xml"
- "--reasoning-parser=qwen3"
- "--limit-mm-per-prompt={\"image\":10}"
- "--speculative-config={\"method\":\"mtp\",\"num_speculative_tokens\":4}"
restart: no
Verify environments in container
docker exec -it vllm-xpu-qwen38-prod bash
docker exec vllm-xpu-qwen38-prod bash -c "pip list | awk 'NR<3 || tolower(\$0) ~ /torch|intel|triton|vllm|oneccl|mkl|dpc|level/'"
docker exec vllm-xpu-qwen38-prod bash -c 'pip show vllm-xpu-kernels'
docker exec vllm-xpu-qwen38-prod bash -c 'lsb_release -a'
| Command | Results |
|---|---|
| <syntaxhightlight lang="bash">
</syntaxhightlight> |
<syntaxhightlight lang="bash">
</syntaxhightlight> |
TODO
- https://docs.vllm.ai/en/stable/cli/serve/
- https://lcz.me/topic/1293/5
- https://jonathanmann.tech/blog/qwen38-intel-arc-b70/
- https://hub.docker.com/r/intel/vllm
- https://hub.docker.com/u/intel/?page=1&search=vllm
- https://docs.vllm.ai/en/stable/getting_started/installation/gpu/#requirements
- https://www.pugetsystems.com/labs/articles/intel-arc-pro-b70-multi-gpu-ai-inference-performance/
- https://github.com/Hal9000AIML/arc-pro-b70-inference-setup-ubuntu-server
- https://sergiiob.dev/posts/intel-arc-b70-vllm-initial-mxfp4-test/
- https://forum.level1techs.com/t/intel-b70-launch-unboxed-and-tested/247873
- https://huggingface.co/blog/MatrixYao/intel-gpu