vllm/.buildkite/run-cpu-test.sh

#!/bin/bash

# This script build the CPU docker image and run the offline inference inside the container.
# It serves a sanity check for compilation and basic model usage.
set -ex

# allow to bind to different cores
CORE_RANGE=${CORE_RANGE:-48-95}
NUMA_NODE=${NUMA_NODE:-1}

# Try building the docker image
numactl -C "$CORE_RANGE" -N "$NUMA_NODE" docker build -t cpu-test-"$BUILDKITE_BUILD_NUMBER" -f Dockerfile.cpu .
numactl -C "$CORE_RANGE" -N "$NUMA_NODE" docker build --build-arg VLLM_CPU_DISABLE_AVX512="true" -t cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2 -f Dockerfile.cpu .

# Setup cleanup
remove_docker_container() { set -e; docker rm -f cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2-"$NUMA_NODE" || true; }
trap remove_docker_container EXIT
remove_docker_container

# Run the image, setting --shm-size=4g for tensor parallel.
docker run -itd --entrypoint /bin/bash -v ~/.cache/huggingface:/root/.cache/huggingface --cpuset-cpus="$CORE_RANGE"  \
 --cpuset-mems="$NUMA_NODE" --privileged=true --network host -e HF_TOKEN --env VLLM_CPU_KVCACHE_SPACE=4 --shm-size=4g --name cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" cpu-test-"$BUILDKITE_BUILD_NUMBER"
docker run -itd --entrypoint /bin/bash -v ~/.cache/huggingface:/root/.cache/huggingface --cpuset-cpus="$CORE_RANGE" \
 --cpuset-mems="$NUMA_NODE" --privileged=true --network host -e HF_TOKEN --env VLLM_CPU_KVCACHE_SPACE=4 --shm-size=4g --name cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2-"$NUMA_NODE" cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2

function cpu_tests() {
  set -e
  export NUMA_NODE=$2

  # offline inference
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2-"$NUMA_NODE" bash -c "
    set -e
    python3 examples/offline_inference/basic/generate.py --model facebook/opt-125m"

  # Run basic model test
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "
    set -e
    pip install -r vllm/requirements/test.txt
    pytest -v -s tests/models/decoder_only/language -m cpu_model
    pytest -v -s tests/models/embedding/language -m cpu_model
    pytest -v -s tests/models/encoder_decoder/language -m cpu_model
    pytest -v -s tests/models/decoder_only/audio_language -m cpu_model
    pytest -v -s tests/models/decoder_only/vision_language -m cpu_model"

  # Run compressed-tensor test
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "
    set -e
    pytest -s -v \
    tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_static_setup \
    tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_dynamic_per_token"

  # Run AWQ test
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "
    set -e
    pytest -s -v \
    tests/quantization/test_ipex_quant.py"

  # Run chunked-prefill and prefix-cache test
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "
    set -e
    pytest -s -v -k cpu_model \
    tests/basic_correctness/test_chunked_prefill.py"  

  # online serving
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "
    set -e
    export VLLM_CPU_KVCACHE_SPACE=10 
    export VLLM_CPU_OMP_THREADS_BIND=$1
    python3 -m vllm.entrypoints.openai.api_server --model facebook/opt-125m --dtype half & 
    timeout 600 bash -c 'until curl localhost:8000/v1/models; do sleep 1; done' || exit 1
    python3 benchmarks/benchmark_serving.py \
      --backend vllm \
      --dataset-name random \
      --model facebook/opt-125m \
      --num-prompts 20 \
      --endpoint /v1/completions \
      --tokenizer facebook/opt-125m"

  # Run multi-lora tests
  docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "
    set -e
    pytest -s -v \
    tests/lora/test_qwen2vl.py"
}

# All of CPU tests are expected to be finished less than 40 mins.
export -f cpu_tests
timeout 40m bash -c "cpu_tests $CORE_RANGE $NUMA_NODE"
[CI/Build] Add shell script linting using shellcheck (#7925) Signed-off-by: Russell Bryant <rbryant@redhat.com> 2024-11-07 13:17:29 -05:00			`#!/bin/bash`

[Hardware][Intel] Add CPU inference backend (#3634) Co-authored-by: Kunshang Ji <kunshang.ji@intel.com> Co-authored-by: Yuan Zhou <yuan.zhou@intel.com> 2024-04-02 13:07:30 +08:00			`# This script build the CPU docker image and run the offline inference inside the container.`
			`# It serves a sanity check for compilation and basic model usage.`
			`set -ex`

[CI][CPU]refactor CPU tests to allow to bind with different cores (#10222) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2024-11-12 18:07:32 +08:00			`# allow to bind to different cores`
			`CORE_RANGE=${CORE_RANGE:-48-95}`
			`NUMA_NODE=${NUMA_NODE:-1}`

[Hardware][Intel] Add CPU inference backend (#3634) Co-authored-by: Kunshang Ji <kunshang.ji@intel.com> Co-authored-by: Yuan Zhou <yuan.zhou@intel.com> 2024-04-02 13:07:30 +08:00			`# Try building the docker image`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`numactl -C "$CORE_RANGE" -N "$NUMA_NODE" docker build -t cpu-test-"$BUILDKITE_BUILD_NUMBER" -f Dockerfile.cpu .`
			`numactl -C "$CORE_RANGE" -N "$NUMA_NODE" docker build --build-arg VLLM_CPU_DISABLE_AVX512="true" -t cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2 -f Dockerfile.cpu .`
[Hardware][Intel] Add CPU inference backend (#3634) Co-authored-by: Kunshang Ji <kunshang.ji@intel.com> Co-authored-by: Yuan Zhou <yuan.zhou@intel.com> 2024-04-02 13:07:30 +08:00
			`# Setup cleanup`
[CI/Build][Bugfix] Fix CPU CI image clean up (#11836) Signed-off-by: jiang1.li <jiang1.li@intel.com> 2025-01-08 23:18:28 +08:00			`remove_docker_container() { set -e; docker rm -f cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2-"$NUMA_NODE" \|\| true; }`
[Hardware][Intel] Add CPU inference backend (#3634) Co-authored-by: Kunshang Ji <kunshang.ji@intel.com> Co-authored-by: Yuan Zhou <yuan.zhou@intel.com> 2024-04-02 13:07:30 +08:00			`trap remove_docker_container EXIT`
			`remove_docker_container`

[Hardware] [Intel] Enable Multiprocessing and tensor parallel in CPU backend and update documentation (#6125) 2024-07-27 04:50:10 +08:00			`# Run the image, setting --shm-size=4g for tensor parallel.`
[CI/Build] Make shellcheck happy (#10285) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-14 17:47:53 +08:00			`docker run -itd --entrypoint /bin/bash -v ~/.cache/huggingface:/root/.cache/huggingface --cpuset-cpus="$CORE_RANGE" \`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`--cpuset-mems="$NUMA_NODE" --privileged=true --network host -e HF_TOKEN --env VLLM_CPU_KVCACHE_SPACE=4 --shm-size=4g --name cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" cpu-test-"$BUILDKITE_BUILD_NUMBER"`
[CI/Build] Make shellcheck happy (#10285) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-14 17:47:53 +08:00			`docker run -itd --entrypoint /bin/bash -v ~/.cache/huggingface:/root/.cache/huggingface --cpuset-cpus="$CORE_RANGE" \`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`--cpuset-mems="$NUMA_NODE" --privileged=true --network host -e HF_TOKEN --env VLLM_CPU_KVCACHE_SPACE=4 --shm-size=4g --name cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2-"$NUMA_NODE" cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2`
[CI/BUILD] enable intel queue for longer CPU tests (#4113) 2024-06-04 01:39:50 +08:00
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`function cpu_tests() {`
[Bugfix][Hardware][CPU] Fix broken encoder-decoder CPU runner (#10218) Signed-off-by: Isotr0py <2037008807@qq.com> 2024-11-11 20:37:58 +08:00			`set -e`
[Hardware][CPU] Support chunked-prefill and prefix-caching on CPU (#10355) Signed-off-by: jiang1.li <jiang1.li@intel.com> 2024-11-20 18:57:39 +08:00			`export NUMA_NODE=$2`
[Bugfix][Hardware][CPU] Fix broken encoder-decoder CPU runner (#10218) Signed-off-by: Isotr0py <2037008807@qq.com> 2024-11-11 20:37:58 +08:00
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`# offline inference`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-avx2-"$NUMA_NODE" bash -c "`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`set -e`
Merge similar examples in `offline_inference` into single `basic` example (#12737) 2025-02-20 12:53:51 +00:00			`python3 examples/offline_inference/basic/generate.py --model facebook/opt-125m"`
[CI/BUILD] enable intel queue for longer CPU tests (#4113) 2024-06-04 01:39:50 +08:00
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`# Run basic model test`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`set -e`
Move requirements into their own directory (#12547) Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> 2025-03-08 17:44:35 +01:00			`pip install -r vllm/requirements/test.txt`
[Model] Support Qwen2 embeddings and use tags to select model tests (#10184) 2024-11-15 12:23:09 +08:00			`pytest -v -s tests/models/decoder_only/language -m cpu_model`
			`pytest -v -s tests/models/embedding/language -m cpu_model`
			`pytest -v -s tests/models/encoder_decoder/language -m cpu_model`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`pytest -v -s tests/models/decoder_only/audio_language -m cpu_model`
			`pytest -v -s tests/models/decoder_only/vision_language -m cpu_model"`
[Hardware] [Intel] Enable Multiprocessing and tensor parallel in CPU backend and update documentation (#6125) 2024-07-27 04:50:10 +08:00
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`# Run compressed-tensor test`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`set -e`
			`pytest -s -v \`
			`tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_static_setup \`
			`tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_dynamic_per_token"`
[Hardware][CPU] Support AWQ for CPU backend (#7515) 2024-10-10 00:28:08 +08:00
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`# Run AWQ test`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`set -e`
			`pytest -s -v \`
			`tests/quantization/test_ipex_quant.py"`
[Hardware][Intel] Support compressed-tensor W8A8 for CPU backend (#7257) 2024-09-12 00:46:46 +08:00
[Hardware][CPU] Support chunked-prefill and prefix-caching on CPU (#10355) Signed-off-by: jiang1.li <jiang1.li@intel.com> 2024-11-20 18:57:39 +08:00			`# Run chunked-prefill and prefix-cache test`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "`
[Hardware][CPU] Support chunked-prefill and prefix-caching on CPU (#10355) Signed-off-by: jiang1.li <jiang1.li@intel.com> 2024-11-20 18:57:39 +08:00			`set -e`
			`pytest -s -v -k cpu_model \`
			`tests/basic_correctness/test_chunked_prefill.py"`

Replace "online inference" with "online serving" (#11923) Signed-off-by: Harry Mellor <19981378+hmellor@users.noreply.github.com> 2025-01-10 12:05:56 +00:00			`# online serving`
[CI][CPU] adding build number to docker image name (#11788) Signed-off-by: Yuan Zhou <yuan.zhou@intel.com> 2025-01-07 15:28:01 +08:00			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`set -e`
			`export VLLM_CPU_KVCACHE_SPACE=10`
[CI/Build] Fix CPU CI online inference timeout (#10314) Signed-off-by: Isotr0py <2037008807@qq.com> 2024-11-14 16:45:32 +08:00			`export VLLM_CPU_OMP_THREADS_BIND=$1`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`python3 -m vllm.entrypoints.openai.api_server --model facebook/opt-125m --dtype half &`
			`timeout 600 bash -c 'until curl localhost:8000/v1/models; do sleep 1; done' \|\| exit 1`
			`python3 benchmarks/benchmark_serving.py \`
			`--backend vllm \`
			`--dataset-name random \`
			`--model facebook/opt-125m \`
			`--num-prompts 20 \`
			`--endpoint /v1/completions \`
			`--tokenizer facebook/opt-125m"`
[Hardware][CPU] Multi-LoRA implementation for the CPU backend (#11100) Signed-off-by: Akshat Tripathi <akshat@krai.ai> Signed-off-by: Oleg Mosalov <oleg@krai.ai> Signed-off-by: Jee Jee Li <pandaleefree@gmail.com> Co-authored-by: Oleg Mosalov <oleg@krai.ai> Co-authored-by: Jee Jee Li <pandaleefree@gmail.com> Co-authored-by: Isotr0py <2037008807@qq.com> 2025-01-12 13:01:52 +00:00
			`# Run multi-lora tests`
			`docker exec cpu-test-"$BUILDKITE_BUILD_NUMBER"-"$NUMA_NODE" bash -c "`
			`set -e`
			`pytest -s -v \`
			`tests/lora/test_qwen2vl.py"`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`}`

[CI/Build][CPU][Bugfix] Fix CPU CI (#12150) Signed-off-by: jiang1.li <jiang1.li@intel.com> 2025-01-17 19:39:52 +08:00			`# All of CPU tests are expected to be finished less than 40 mins.`
[CI/Build] Adding timeout in CPU CI to avoid CPU test queue blocking (#6892) Signed-off-by: DarkLight1337 <tlleungac@connect.ust.hk> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-11-09 11:27:11 +08:00			`export -f cpu_tests`
[CI/Build][CPU][Bugfix] Fix CPU CI (#12150) Signed-off-by: jiang1.li <jiang1.li@intel.com> 2025-01-17 19:39:52 +08:00			`timeout 40m bash -c "cpu_tests $CORE_RANGE $NUMA_NODE"`