Skip to content

vllm.v1.worker.gpu.eplb_utils

Functions:

preserve_serving_state(model_runner)

Keep the elastic EP warmup out of the request pool and the KV cache.

Source code in vllm/v1/worker/gpu/eplb_utils.py
@contextmanager
def preserve_serving_state(model_runner: "GPUModelRunner") -> Iterator[None]:
    """Keep the elastic EP warmup out of the request pool and the KV cache."""
    # After the drain only parked streaming sessions still hold a slot, and
    # those are re-added from NewRequestData on their next chunk.
    _remove_all_requests(model_runner)
    model_runner.block_tables.redirect_writes_to_null_block = True
    try:
        yield
    finally:
        _remove_all_requests(model_runner)
        model_runner.block_tables.redirect_writes_to_null_block = False
        if model_runner.kv_block_zeroer is not None:
            model_runner.kv_block_zeroer.zero_block_ids([0])

step_eplb_after(*, is_dummy=False)

Step EPLB after a model runner method completes successfully.

Source code in vllm/v1/worker/gpu/eplb_utils.py
def step_eplb_after(*, is_dummy: bool = False) -> Callable:
    """Step EPLB after a model runner method completes successfully."""

    def decorator(fn: Callable) -> Callable:
        @wraps(fn)
        def wrapper(self: Any, *args, **kwargs) -> Any:
            result = fn(self, *args, **kwargs)
            if kwargs.get("skip_eplb", False):
                return result

            is_profile = kwargs.get("is_profile", False) if is_dummy else False
            self.eplb.step(is_dummy=is_dummy, is_profile=is_profile)
            return result

        return wrapper

    return decorator