@@ -12,15 +12,19 @@ steps:
1212 - vllm/_custom_ops.py
1313 - tests/kernels/attention/test_cpu_attn.py
1414 - tests/kernels/moe/test_cpu_fused_moe.py
15+ - tests/kernels/moe/test_cpu_quant_fused_moe.py
1516 - tests/kernels/test_onednn.py
1617 - tests/kernels/test_awq_int4_to_int8.py
18+ - tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
1719 commands :
1820 - |
19- bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 20m "
21+ bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
2022 pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
2123 pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
24+ pytest -x -v -s tests/kernels/moe/test_cpu_quant_fused_moe.py
2225 pytest -x -v -s tests/kernels/test_onednn.py
23- pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py"
26+ pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py
27+ pytest -x -v -s tests/kernels/quantization/test_cpu_fp8_scaled_mm.py"
2428
2529 - label : CPU-Compatibility Tests
2630 depends_on : []
@@ -50,30 +54,45 @@ steps:
5054 pytest -x -v -s tests/models/language/generation -m cpu_model
5155 pytest -x -v -s tests/models/language/pooling -m cpu_model"
5256
57+ - label : CPU-ModelRunnerV2 Tests
58+ depends_on : []
59+ device : intel_cpu
60+ no_plugin : true
61+ soft_fail : true
62+ source_file_dependencies :
63+ - vllm/v1/worker/cpu/
64+ - vllm/v1/worker/gpu/
65+ commands :
66+ - |
67+ bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
68+ uv pip install git+https://github.com/triton-lang/triton-cpu.git@270e696d
69+ VLLM_USE_V2_MODEL_RUNNER=1 pytest -x -v -s tests/models/language/generation/test_granite.py -m cpu_model"
70+
5371 - label : CPU-Quantization Model Tests
5472 depends_on : []
5573 device : intel_cpu
5674 no_plugin : true
5775 source_file_dependencies :
5876 - csrc/cpu/
5977 - vllm/model_executor/layers/quantization/cpu_wna16.py
60- - vllm/model_executor/layers/quantization/gptq_marlin .py
78+ - vllm/model_executor/layers/quantization/auto_gptq .py
6179 - vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w8a8_int8.py
6280 - vllm/model_executor/layers/quantization/kernels/scaled_mm/cpu.py
6381 - vllm/model_executor/layers/quantization/kernels/mixed_precision/cpu.py
82+ - vllm/model_executor/layers/fused_moe/experts/cpu_moe.py
6483 - tests/quantization/test_compressed_tensors.py
6584 - tests/quantization/test_cpu_wna16.py
6685 commands :
6786 - |
68- bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 20m "
87+ bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
6988 pytest -x -v -s tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_logprobs
7089 pytest -x -v -s tests/quantization/test_cpu_wna16.py"
7190
72- - label : CPU-Distributed Tests
91+ - label : CPU-Distributed Tests (PP+TP)
7392 depends_on : []
7493 device : intel_cpu
7594 no_plugin : true
76- source_file_dependencies :
95+ source_file_dependencies : &cpu_distributed_deps
7796 - csrc/cpu/shm.cpp
7897 - vllm/v1/worker/cpu_worker.py
7998 - vllm/v1/worker/gpu_worker.py
@@ -82,10 +101,21 @@ steps:
82101 - vllm/platforms/cpu.py
83102 - vllm/distributed/parallel_state.py
84103 - vllm/distributed/device_communicators/cpu_communicator.py
104+ - .buildkite/scripts/hardware_ci/run-cpu-distributed-smoke-test.sh
105+ commands :
106+ - |
107+ bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 10m "
108+ bash .buildkite/scripts/hardware_ci/run-cpu-distributed-smoke-test.sh tp_pp"
109+
110+ - label : CPU-Distributed Tests (DP+TP)
111+ depends_on : []
112+ device : intel_cpu
113+ no_plugin : true
114+ source_file_dependencies : *cpu_distributed_deps
85115 commands :
86116 - |
87117 bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 10m "
88- bash .buildkite/scripts/hardware_ci/run-cpu-distributed-smoke-test.sh"
118+ bash .buildkite/scripts/hardware_ci/run-cpu-distributed-smoke-test.sh dp_tp "
89119
90120 - label : CPU-Multi-Modal Model Tests %N
91121 depends_on : []
0 commit comments