From b88abb5036aa6b7191a3e9341ab652e6562a49d2 Mon Sep 17 00:00:00 2001 From: Daoyuan Li <94409450+DaoyuanLi2816@users.noreply.github.com> Date: Thu, 16 Jul 2026 17:00:47 -0700 Subject: [PATCH] [Misc] Remove orphaned env vars and stale env-var references (#44749) Signed-off-by: Daoyuan Li <94409450+DaoyuanLi2816@users.noreply.github.com> --- docs/configuration/env_vars.md | 2 +- docs/serving/expert_parallel_deployment.md | 2 -- vllm/envs.py | 30 ---------------------- 3 files changed, 1 insertion(+), 33 deletions(-) diff --git a/docs/configuration/env_vars.md b/docs/configuration/env_vars.md index f6d548a19d9..38de2a49760 100644 --- a/docs/configuration/env_vars.md +++ b/docs/configuration/env_vars.md @@ -5,7 +5,7 @@ vLLM uses the following environment variables to configure the system: !!! warning Please note that `VLLM_PORT` and `VLLM_HOST_IP` set the port and ip for vLLM's **internal usage**. It is not the port and ip for the API server. If you use `--host $VLLM_HOST_IP` and `--port $VLLM_PORT` to start the API server, it will not work. - All environment variables used by vLLM are prefixed with `VLLM_`. **Special care should be taken for Kubernetes users**: please do not name the service as `vllm`, otherwise environment variables set by Kubernetes might conflict with vLLM's environment variables, because [Kubernetes sets environment variables for each service with the capitalized service name as the prefix](https://kubernetes.io/docs/concepts/services-networking/service/#environment-variables). + Most vLLM-specific environment variables are prefixed with `VLLM_` (a handful of standard names — for example `CUDA_VISIBLE_DEVICES`, `MAX_JOBS`, `S3_ACCESS_KEY_ID`/`S3_SECRET_ACCESS_KEY`/`S3_ENDPOINT_URL`, `DO_NOT_TRACK`, `NO_COLOR` — are also read directly when set). **Special care should be taken for Kubernetes users**: please do not name the service as `vllm`, otherwise environment variables set by Kubernetes might conflict with vLLM's environment variables, because [Kubernetes sets environment variables for each service with the capitalized service name as the prefix](https://kubernetes.io/docs/concepts/services-networking/service/#environment-variables). ```python --8<-- "vllm/envs.py:env-vars-definition" diff --git a/docs/serving/expert_parallel_deployment.md b/docs/serving/expert_parallel_deployment.md index b7c2ee87375..9a8bb68bc11 100644 --- a/docs/serving/expert_parallel_deployment.md +++ b/docs/serving/expert_parallel_deployment.md @@ -220,8 +220,6 @@ For multi-node deployment, add these EPLB flags to each node's command. We recom - Use simulator flags `VLLM_MOE_ROUTING_SIMULATION_STRATEGY=uniform_random` and `VLLM_RANDOMIZE_DP_DUMMY_INPUTS=1` so token routing is balanced across EP ranks. -- Increasing `VLLM_MOE_DP_CHUNK_SIZE` may increase throughput by increasing the maximum batch size for inter-rank token transfers. This may cause DeepEP to throw `assert self.nvshmem_qp_depth >= (num_max_dispatch_tokens_per_rank + 1) * 2`, which can be fixed by increasing environment variable `NVSHMEM_QP_DEPTH`. - ## Disaggregated Serving (Prefill/Decode Split) For production deployments requiring strict SLA guarantees for time-to-first-token and inter-token latency, disaggregated serving allows independent scaling of prefill and decode operations. diff --git a/vllm/envs.py b/vllm/envs.py index 5b601cda8cb..67d8fa5a2c6 100755 --- a/vllm/envs.py +++ b/vllm/envs.py @@ -127,7 +127,6 @@ if TYPE_CHECKING: VLLM_MXFP8_EMULATION_DEQUANT_AT_LOAD: bool = True VLLM_ROCM_USE_AITER: bool = False VLLM_ROCM_USE_AITER_CUSTOM_AR: bool = True - VLLM_ROCM_USE_AITER_PAGED_ATTN: bool = False VLLM_ROCM_USE_AITER_LINEAR: bool = True VLLM_ROCM_USE_AITER_LINEAR_HIPBMM: bool = False VLLM_ROCM_USE_AITER_MOE: bool = True @@ -182,8 +181,6 @@ if TYPE_CHECKING: VLLM_HUMMING_MOE_GEMM_TYPE: Literal["indexed", "grouped", "auto"] | None = None VLLM_DEEPEPLL_NVFP4_DISPATCH: bool = False VLLM_V1_USE_OUTLINES_CACHE: bool = False - VLLM_TPU_BUCKET_PADDING_GAP: int = 0 - VLLM_TPU_MOST_MODEL_LEN: int | None = None VLLM_TPU_USING_PATHWAYS: bool = False VLLM_USE_DEEP_GEMM: bool = True VLLM_MOE_USE_DEEP_GEMM: bool = True @@ -1194,11 +1191,6 @@ environment_variables: dict[str, Callable[[], Any]] = { "VLLM_ROCM_USE_AITER_CUSTOM_AR": lambda: ( os.getenv("VLLM_ROCM_USE_AITER_CUSTOM_AR", "True").lower() in ("true", "1") ), - # Whether to use aiter paged attention. - # By default is disabled. - "VLLM_ROCM_USE_AITER_PAGED_ATTN": lambda: ( - os.getenv("VLLM_ROCM_USE_AITER_PAGED_ATTN", "False").lower() in ("true", "1") - ), # use aiter linear op if aiter ops are enabled # The following list of related ops # - scaled_mm (per-tensor / rowwise) @@ -1432,8 +1424,6 @@ environment_variables: dict[str, Callable[[], Any]] = { "VLLM_RAY_EXTRA_ENV_VARS_TO_COPY": lambda: os.getenv( "VLLM_RAY_EXTRA_ENV_VARS_TO_COPY", "" ), - # Whether to use S3 path for model loading in CI via RunAI Streamer - "VLLM_CI_USE_S3": lambda: os.environ.get("VLLM_CI_USE_S3", "0") == "1", # Use model_redirect to redirect the model name to a local folder. # `model_redirect` can be a json file mapping the model between # repo_id and local folder: @@ -1482,16 +1472,6 @@ environment_variables: dict[str, Callable[[], Any]] = { "VLLM_V1_USE_OUTLINES_CACHE": lambda: ( os.environ.get("VLLM_V1_USE_OUTLINES_CACHE", "0") == "1" ), - # Gap between padding buckets for the forward pass. So we have - # 8, we will run forward pass with [16, 24, 32, ...]. - "VLLM_TPU_BUCKET_PADDING_GAP": lambda: ( - int(os.environ["VLLM_TPU_BUCKET_PADDING_GAP"]) - if "VLLM_TPU_BUCKET_PADDING_GAP" in os.environ - else 0 - ), - "VLLM_TPU_MOST_MODEL_LEN": lambda: maybe_convert_int( - os.environ.get("VLLM_TPU_MOST_MODEL_LEN", None) - ), # Whether using Pathways "VLLM_TPU_USING_PATHWAYS": lambda: bool( "proxy" in os.getenv("JAX_PLATFORMS", "").lower() @@ -1667,15 +1647,6 @@ environment_variables: dict[str, Callable[[], Any]] = { "VLLM_MAX_TOKENS_PER_EXPERT_FP4_MOE": lambda: int( os.getenv("VLLM_MAX_TOKENS_PER_EXPERT_FP4_MOE", "163840") ), - # Specifies the thresholds of the communicated tensor sizes under which - # vllm should use flashinfer fused allreduce. The variable should be a - # JSON with the following format: - # { : } - # Unspecified world sizes will fall back to - # { 2: 64, 4: 1, : 0.5 } - "VLLM_FLASHINFER_ALLREDUCE_FUSION_THRESHOLDS_MB": lambda: json.loads( - os.getenv("VLLM_FLASHINFER_ALLREDUCE_FUSION_THRESHOLDS_MB", "{}") - ), # MoE routing strategy selector. # See `RoutingSimulator.get_available_strategies()` # for available # strategies. @@ -2147,7 +2118,6 @@ def compile_factors() -> dict[str, object]: "VLLM_DP_MASTER_PORT", "VLLM_NIXL_SIDE_CHANNEL_HOST", "VLLM_RANDOMIZE_DP_DUMMY_INPUTS", - "VLLM_CI_USE_S3", "VLLM_MODEL_REDIRECT_PATH", "VLLM_HOST_IP", "VLLM_FORCE_AOT_LOAD",