mirror of
https://github.com/vllm-project/vllm.git
synced 2026-08-23 14:10:14 +00:00
[CI/Build][AMD] Use float16 in test_reset_prefix_cache_e2e to avoid accuracy issues (#29997)
Signed-off-by: Randall Smith <[email protected]> Co-authored-by: Randall Smith <[email protected]>
This commit is contained in:
@@ -21,6 +21,7 @@ def test_reset_prefix_cache_e2e(monkeypatch):
|
||||
max_num_batched_tokens=32,
|
||||
max_model_len=2048,
|
||||
compilation_config={"mode": 0},
|
||||
dtype="float16",
|
||||
)
|
||||
engine = LLMEngine.from_engine_args(engine_args)
|
||||
sampling_params = SamplingParams(
|
||||
|
||||
Reference in New Issue
Block a user