[CI] Move Distributed Tests from H200 -> H100 (#32555)

2026-01-18 13:25:23 -05:00
parent 327a02d8db
commit afc3622602
1 changed files with 5 additions and 4 deletions
--- a/.buildkite/test-pipeline.yaml
+++ b/.buildkite/test-pipeline.yaml
@@ -1400,18 +1400,19 @@ steps:
    - pytest -v -s tests/distributed/test_sequence_parallel.py
    - pytest -v -s tests/compile/distributed/test_sequence_parallelism.py

-##### H200 test #####
- label: Distributed Tests (H200) # optional
-  gpu: h200
+- label: Distributed Tests (H100) # optional
+  gpu: h100
  optional: true
  working_dir: "/vllm-workspace/"
  num_gpus: 2
  commands:
    - VLLM_TEST_CLEAN_GPU_MEMORY=1 pytest -v -s tests/compile/distributed/test_async_tp.py
    - pytest -v -s tests/distributed/test_context_parallel.py
-    - CUDA_VISIBLE_DEVICES=1,2 VLLM_USE_DEEP_GEMM=1 VLLM_LOGGING_LEVEL=DEBUG python3 examples/offline_inference/data_parallel.py --model=Qwen/Qwen1.5-MoE-A2.7B -tp=1 -dp=2 --max-model-len=2048 --all2all-backend=deepep_high_throughput
+    - VLLM_USE_DEEP_GEMM=1 VLLM_LOGGING_LEVEL=DEBUG python3 examples/offline_inference/data_parallel.py --model=Qwen/Qwen1.5-MoE-A2.7B -tp=1 -dp=2 --max-model-len=2048 --all2all-backend=deepep_high_throughput
    - pytest -v -s tests/v1/distributed/test_dbo.py

+##### H200 test #####
+
 - label: LM Eval Large Models (H200) # optional
  timeout_in_minutes: 60
  gpu: h200