diff --git a/.buildkite/test-amd.yaml b/.buildkite/test-amd.yaml index c5369950fb3..4085ab0b1bd 100644 --- a/.buildkite/test-amd.yaml +++ b/.buildkite/test-amd.yaml @@ -460,7 +460,7 @@ steps: - tests/lora - vllm/platforms/rocm.py commands: - - pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_llm_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py + - pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_qwen3_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py #------------------------------------------------------ mi250 ยท model_executor -------------------------------------------------------# @@ -1760,7 +1760,7 @@ steps: - export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True - pytest -v -s -x lora/test_chatglm3_tp.py - pytest -v -s -x lora/test_llama_tp.py - - pytest -v -s -x lora/test_llm_with_multi_loras.py + - pytest -v -s -x lora/test_qwen3_with_multi_loras.py - pytest -v -s -x lora/test_olmoe_tp.py - pytest -v -s -x lora/test_gptoss_tp.py - pytest -v -s -x lora/test_qwen35_densemodel_lora.py diff --git a/.buildkite/test_areas/lora.yaml b/.buildkite/test_areas/lora.yaml index ebffcc59fc3..8107f9b37ff 100644 --- a/.buildkite/test_areas/lora.yaml +++ b/.buildkite/test_areas/lora.yaml @@ -9,7 +9,7 @@ steps: - vllm/lora - tests/lora commands: - - pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_llm_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py + - pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_qwen3_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py parallelism: 4 @@ -31,7 +31,7 @@ steps: # requires multi-GPU testing for validation. - pytest -v -s -x lora/test_chatglm3_tp.py - pytest -v -s -x lora/test_llama_tp.py - - pytest -v -s -x lora/test_llm_with_multi_loras.py + - pytest -v -s -x lora/test_qwen3_with_multi_loras.py - pytest -v -s -x lora/test_olmoe_tp.py - pytest -v -s -x lora/test_gptoss_tp.py - pytest -v -s -x lora/test_qwen35_densemodel_lora.py \ No newline at end of file diff --git a/tests/lora/test_chatglm3_tp.py b/tests/lora/test_chatglm3_tp.py index 8f42243387d..ace4fb5f50e 100644 --- a/tests/lora/test_chatglm3_tp.py +++ b/tests/lora/test_chatglm3_tp.py @@ -1,9 +1,12 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM project +import pytest + import vllm import vllm.config from vllm.lora.request import LoRARequest +from vllm.platforms import current_platform from ..utils import create_new_process_for_each_test, multi_gpu_test @@ -50,6 +53,9 @@ def do_sample(llm: vllm.LLM, lora_path: str, lora_id: int) -> list[str]: return generated_texts +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @create_new_process_for_each_test() def test_chatglm3_lora(chatglm3_lora_files): llm = vllm.LLM( @@ -70,6 +76,9 @@ def test_chatglm3_lora(chatglm3_lora_files): assert output2[i] == EXPECTED_LORA_OUTPUT[i] +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @multi_gpu_test(num_gpus=4) def test_chatglm3_lora_tp4(chatglm3_lora_files): llm = vllm.LLM( diff --git a/tests/lora/test_default_mm_loras.py b/tests/lora/test_default_mm_loras.py index c76d3c6e798..673e8e85555 100644 --- a/tests/lora/test_default_mm_loras.py +++ b/tests/lora/test_default_mm_loras.py @@ -11,6 +11,7 @@ import pytest from huggingface_hub import snapshot_download from vllm.lora.request import LoRARequest +from vllm.platforms import current_platform from ..conftest import AudioTestAssets, VllmRunner from ..utils import create_new_process_for_each_test @@ -76,6 +77,9 @@ def test_active_default_mm_lora( ) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @create_new_process_for_each_test() def test_inactive_default_mm_lora( vllm_runner: type[VllmRunner], @@ -92,6 +96,9 @@ def test_inactive_default_mm_lora( ) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @create_new_process_for_each_test() def test_default_mm_lora_succeeds_with_redundant_lora_request( vllm_runner: type[VllmRunner], @@ -107,6 +114,9 @@ def test_default_mm_lora_succeeds_with_redundant_lora_request( ) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @create_new_process_for_each_test() def test_default_mm_lora_fails_with_overridden_lora_request( vllm_runner: type[VllmRunner], diff --git a/tests/lora/test_llama_tp.py b/tests/lora/test_llama_tp.py index 99c823238dd..42f6ddc2f69 100644 --- a/tests/lora/test_llama_tp.py +++ b/tests/lora/test_llama_tp.py @@ -10,6 +10,7 @@ import vllm.config from vllm import LLM from vllm.lora.request import LoRARequest from vllm.model_executor.model_loader.tensorizer import TensorizerConfig +from vllm.platforms import current_platform from ..utils import VLLM_PATH, create_new_process_for_each_test, multi_gpu_test @@ -139,6 +140,9 @@ def test_llama_lora(llama32_lora_files, cudagraph_specialize_lora: bool): generate_and_test(llm, llama32_lora_files) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @multi_gpu_test(num_gpus=4) def test_llama_lora_tp4(llama32_lora_files): llm = vllm.LLM( diff --git a/tests/lora/test_minicpmv_tp.py b/tests/lora/test_minicpmv_tp.py index 3d6484a710a..0090f9c569b 100644 --- a/tests/lora/test_minicpmv_tp.py +++ b/tests/lora/test_minicpmv_tp.py @@ -68,6 +68,9 @@ def do_sample(llm: vllm.LLM, lora_path: str, lora_id: int) -> list[str]: return generated_texts +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) def test_minicpmv_lora(minicpmv_lora_files): llm = vllm.LLM( MODEL_PATH, diff --git a/tests/lora/test_olmoe_tp.py b/tests/lora/test_olmoe_tp.py index 0b477062205..bbc25cb6b8e 100644 --- a/tests/lora/test_olmoe_tp.py +++ b/tests/lora/test_olmoe_tp.py @@ -11,6 +11,7 @@ from safetensors.torch import load_file, save_file import vllm from vllm.lora.request import LoRARequest +from vllm.platforms import current_platform from ..utils import multi_gpu_test @@ -110,6 +111,9 @@ def generate_and_test( ) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) def test_olmoe_lora(olmoe_lora_files, maybe_enable_lora_dual_stream): # We enable enforce_eager=True here to reduce VRAM usage for lora-test CI, # Otherwise, the lora-test will fail due to CUDA OOM. @@ -178,6 +182,9 @@ def test_olmoe_lora_mixed_random( assert outputs[0].outputs[0].text.strip().startswith(EXPECTED_LORA_OUTPUT[0]) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @pytest.mark.parametrize("fully_sharded_loras", [False, True]) @multi_gpu_test(num_gpus=2) def test_olmoe_lora_tp2(olmoe_lora_files, fully_sharded_loras): diff --git a/tests/lora/test_qwen35_densemodel_lora.py b/tests/lora/test_qwen35_densemodel_lora.py index a9ee5fac8cb..e926bbcef27 100644 --- a/tests/lora/test_qwen35_densemodel_lora.py +++ b/tests/lora/test_qwen35_densemodel_lora.py @@ -1,12 +1,14 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM project +import pytest from transformers import AutoTokenizer import vllm import vllm.config from vllm.assets.image import ImageAsset from vllm.lora.request import LoRARequest +from vllm.platforms import current_platform from ..utils import create_new_process_for_each_test, multi_gpu_test @@ -311,6 +313,9 @@ def _assert_qwen35_text_vl_and_mixed_lora( ) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) @create_new_process_for_each_test() def test_qwen35_text_lora( qwen35_text_lora_files, qwen35_vl_lora_files, maybe_enable_lora_dual_stream diff --git a/tests/lora/test_llm_with_multi_loras.py b/tests/lora/test_qwen3_with_multi_loras.py similarity index 100% rename from tests/lora/test_llm_with_multi_loras.py rename to tests/lora/test_qwen3_with_multi_loras.py diff --git a/tests/lora/test_qwenvl.py b/tests/lora/test_qwenvl.py index 5f8fc26c16d..a4a32278db0 100644 --- a/tests/lora/test_qwenvl.py +++ b/tests/lora/test_qwenvl.py @@ -2,12 +2,14 @@ # SPDX-FileCopyrightText: Copyright contributors to the vLLM project from dataclasses import dataclass +import pytest from packaging.version import Version from transformers import __version__ as TRANSFORMERS_VERSION import vllm from vllm.assets.image import ImageAsset from vllm.lora.request import LoRARequest +from vllm.platforms import current_platform from vllm.sampling_params import BeamSearchParams @@ -206,6 +208,9 @@ def test_qwen2vl_lora_beam_search(qwen2vl_lora_files): ) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) def test_qwen25vl_lora(qwen25vl_lora_files): """Test Qwen 2.5 VL model with LoRA""" config = TestConfig(model_path=QWEN25VL_MODEL_PATH, lora_path=qwen25vl_lora_files) @@ -216,6 +221,9 @@ def test_qwen25vl_lora(qwen25vl_lora_files): tester.run_test(TEST_IMAGES, expected_outputs=EXPECTED_OUTPUTS, lora_id=lora_id) +@pytest.mark.skipif( + current_platform.is_cuda_alike(), reason="Skipping to avoid redundant model tests" +) def test_qwen25vl_vision_lora(qwen25vl_vision_lora_files): config = TestConfig( model_path=QWEN25VL_MODEL_PATH, diff --git a/tests/lora/test_whisper.py b/tests/lora/test_whisper.py index 83b814d49f7..ea8179a9c66 100644 --- a/tests/lora/test_whisper.py +++ b/tests/lora/test_whisper.py @@ -124,30 +124,3 @@ def test_whisper_multi_lora(whisper_lora_files): f"Expected same outputs for same adapter with different IDs. " f"Got: {outputs_lora1} vs {outputs_lora2}" ) - - -@create_new_process_for_each_test() -def test_whisper_with_and_without_lora(whisper_lora_files): - """Test that Whisper produces different outputs with and without LoRA. - - This test verifies that the LoRA adapter actually affects the model output. - """ - llm = create_whisper_llm(enable_lora=True) - - # Run with LoRA - outputs_with_lora = run_whisper_inference( - llm, lora_path=whisper_lora_files, lora_id=1 - ) - - # Run without LoRA (base model only) - outputs_without_lora = run_whisper_inference(llm, lora_path=None) - - # Both should produce valid outputs - assert len(outputs_with_lora[0]) > 0 - assert len(outputs_without_lora[0]) > 0 - - print(f"Output with LoRA: {outputs_with_lora[0]}") - print(f"Output without LoRA: {outputs_without_lora[0]}") - - # Note: Outputs may or may not differ depending on the adapter - # The main verification is that both configurations work