Skip to content

vllm.v1.outputs

Classes:

Functions:

AsyncModelRunnerOutput

Bases: ABC

Methods:

  • get_output –

    Get the ModelRunnerOutput for this async output.

Source code in vllm/v1/outputs.py
class AsyncModelRunnerOutput(ABC):
    @abstractmethod
    def get_output(self) -> ModelRunnerOutput:
        """Get the ModelRunnerOutput for this async output.

        This is a blocking call that waits until the results are ready, which
        might involve copying device tensors to the host.
        This method should only be called once per AsyncModelRunnerOutput.
        """
        pass

get_output() abstractmethod

Get the ModelRunnerOutput for this async output.

This is a blocking call that waits until the results are ready, which might involve copying device tensors to the host. This method should only be called once per AsyncModelRunnerOutput.

Source code in vllm/v1/outputs.py
@abstractmethod
def get_output(self) -> ModelRunnerOutput:
    """Get the ModelRunnerOutput for this async output.

    This is a blocking call that waits until the results are ready, which
    might involve copying device tensors to the host.
    This method should only be called once per AsyncModelRunnerOutput.
    """
    pass

LogprobsTensors

Bases: NamedTuple

Methods:

  • cat –

    Concatenate flattened logprob tensors.

  • empty_cpu –

    Create empty LogprobsTensors on CPU.

  • filter –

    Filter the logprobs tensors with the given bool mask.

Source code in vllm/v1/outputs.py
class LogprobsTensors(NamedTuple):
    # [num_reqs x num_generated_tokens, max_num_logprobs + 1]
    logprob_token_ids: torch.Tensor
    # [num_reqs x num_generated_tokens, max_num_logprobs + 1]
    logprobs: torch.Tensor
    # [num_reqs x num_generated_tokens]
    selected_token_ranks: torch.Tensor
    # [num_reqs + 1]
    cu_num_generated_tokens: list[int] | None = None
    # [num_reqs + 1]. Set instead of cu_num_generated_tokens when the
    # boundaries only exist on device (adaptive verification); rides along on
    # the async D2H copy.
    cu_num_generated_tokens_tensor: torch.Tensor | None = None

    def tolists(self, cu_num_generated_tokens: list[int] | None = None):
        if cu_num_generated_tokens is None:
            if self.cu_num_generated_tokens_tensor is not None:
                cu_num_generated_tokens = self.cu_num_generated_tokens_tensor.tolist()
            else:
                cu_num_generated_tokens = self.cu_num_generated_tokens
        return LogprobsLists(
            self.logprob_token_ids.cpu().numpy(),
            self.logprobs.cpu().numpy(),
            self.selected_token_ranks.cpu().numpy(),
            cu_num_generated_tokens,
        )

    def to_cpu_nonblocking(self) -> "LogprobsTensors":
        if self.logprob_token_ids.device.type == "cpu":
            return self
        cu_tensor = self.cu_num_generated_tokens_tensor
        if cu_tensor is not None:
            cu_tensor = cu_tensor.to("cpu", non_blocking=True)
        return LogprobsTensors(
            self.logprob_token_ids.to("cpu", non_blocking=True),
            self.logprobs.to("cpu", non_blocking=True),
            self.selected_token_ranks.to("cpu", non_blocking=True),
            self.cu_num_generated_tokens,
            cu_tensor,
        )

    def filter(self, mask: torch.Tensor) -> "LogprobsTensors":
        """Filter the logprobs tensors with the given bool mask."""
        assert self.cu_num_generated_tokens is None, (
            "filter can't be used with cu_num_generated_tokens"
        )
        assert self.cu_num_generated_tokens_tensor is None, (
            "filter can't be used with cu_num_generated_tokens_tensor"
        )
        return LogprobsTensors(
            self.logprob_token_ids[mask],
            self.logprobs[mask],
            self.selected_token_ranks[mask],
        )

    @staticmethod
    def cat(
        tensors: Sequence["LogprobsTensors"],
        cu_num_generated_tokens: list[int] | None = None,
    ) -> "LogprobsTensors":
        """Concatenate flattened logprob tensors."""
        assert tensors
        assert cu_num_generated_tokens is not None or all(
            tensor.cu_num_generated_tokens is None for tensor in tensors
        )
        if len(tensors) == 1:
            tensor = tensors[0]
            if cu_num_generated_tokens is None:
                return tensor
            return tensor._replace(cu_num_generated_tokens=cu_num_generated_tokens)
        # The multi-chunk path rebuilds boundaries from the CPU layout, which
        # the device-only boundaries of adaptive verification never use.
        assert all(tensor.cu_num_generated_tokens_tensor is None for tensor in tensors)
        return LogprobsTensors(
            logprob_token_ids=torch.cat(
                [tensor.logprob_token_ids for tensor in tensors]
            ),
            logprobs=torch.cat([tensor.logprobs for tensor in tensors]),
            selected_token_ranks=torch.cat(
                [tensor.selected_token_ranks for tensor in tensors]
            ),
            cu_num_generated_tokens=cu_num_generated_tokens,
        )

    @staticmethod
    def empty_cpu(
        num_positions: int, num_tokens_per_position: int
    ) -> "LogprobsTensors":
        """Create empty LogprobsTensors on CPU."""
        logprob_token_ids = torch.empty(
            (num_positions, num_tokens_per_position),
            dtype=torch.int32,
            device="cpu",
            pin_memory=PIN_MEMORY,
        )
        logprobs = logprob_token_ids.new_empty(
            (num_positions, num_tokens_per_position),
            dtype=torch.float32,
            pin_memory=PIN_MEMORY,
        )
        selected_token_ranks = logprob_token_ids.new_empty(
            num_positions, pin_memory=PIN_MEMORY
        )
        return LogprobsTensors(
            logprob_token_ids=logprob_token_ids,
            logprobs=logprobs,
            selected_token_ranks=selected_token_ranks,
        )

cat(tensors, cu_num_generated_tokens=None) staticmethod

Concatenate flattened logprob tensors.

Source code in vllm/v1/outputs.py
@staticmethod
def cat(
    tensors: Sequence["LogprobsTensors"],
    cu_num_generated_tokens: list[int] | None = None,
) -> "LogprobsTensors":
    """Concatenate flattened logprob tensors."""
    assert tensors
    assert cu_num_generated_tokens is not None or all(
        tensor.cu_num_generated_tokens is None for tensor in tensors
    )
    if len(tensors) == 1:
        tensor = tensors[0]
        if cu_num_generated_tokens is None:
            return tensor
        return tensor._replace(cu_num_generated_tokens=cu_num_generated_tokens)
    # The multi-chunk path rebuilds boundaries from the CPU layout, which
    # the device-only boundaries of adaptive verification never use.
    assert all(tensor.cu_num_generated_tokens_tensor is None for tensor in tensors)
    return LogprobsTensors(
        logprob_token_ids=torch.cat(
            [tensor.logprob_token_ids for tensor in tensors]
        ),
        logprobs=torch.cat([tensor.logprobs for tensor in tensors]),
        selected_token_ranks=torch.cat(
            [tensor.selected_token_ranks for tensor in tensors]
        ),
        cu_num_generated_tokens=cu_num_generated_tokens,
    )

empty_cpu(num_positions, num_tokens_per_position) staticmethod

Create empty LogprobsTensors on CPU.

Source code in vllm/v1/outputs.py
@staticmethod
def empty_cpu(
    num_positions: int, num_tokens_per_position: int
) -> "LogprobsTensors":
    """Create empty LogprobsTensors on CPU."""
    logprob_token_ids = torch.empty(
        (num_positions, num_tokens_per_position),
        dtype=torch.int32,
        device="cpu",
        pin_memory=PIN_MEMORY,
    )
    logprobs = logprob_token_ids.new_empty(
        (num_positions, num_tokens_per_position),
        dtype=torch.float32,
        pin_memory=PIN_MEMORY,
    )
    selected_token_ranks = logprob_token_ids.new_empty(
        num_positions, pin_memory=PIN_MEMORY
    )
    return LogprobsTensors(
        logprob_token_ids=logprob_token_ids,
        logprobs=logprobs,
        selected_token_ranks=selected_token_ranks,
    )

filter(mask)

Filter the logprobs tensors with the given bool mask.

Source code in vllm/v1/outputs.py
def filter(self, mask: torch.Tensor) -> "LogprobsTensors":
    """Filter the logprobs tensors with the given bool mask."""
    assert self.cu_num_generated_tokens is None, (
        "filter can't be used with cu_num_generated_tokens"
    )
    assert self.cu_num_generated_tokens_tensor is None, (
        "filter can't be used with cu_num_generated_tokens_tensor"
    )
    return LogprobsTensors(
        self.logprob_token_ids[mask],
        self.logprobs[mask],
        self.selected_token_ranks[mask],
    )

ModelRunnerOutput dataclass

Methods:

Source code in vllm/v1/outputs.py
@dataclass
class ModelRunnerOutput:
    # [num_reqs]
    req_ids: list[str]
    # req_id -> index
    req_id_to_index: dict[str, int]

    # num_reqs x num_generated_tokens
    # num_generated_tokens is the number of tokens
    # generated in the current step. It can be different for
    # each request due to speculative/jump decoding.
    sampled_token_ids: list[list[int]] = field(default_factory=list)

    # [num_reqs, max_num_logprobs + 1]
    # [num_reqs, max_num_logprobs + 1]
    # [num_reqs]
    logprobs: LogprobsLists | None = None

    # req_id -> (token_ids, logprobs, ranks)
    # [prompt_len, num_prompt_logprobs]
    # [prompt_len, num_prompt_logprobs]
    # [prompt_len]
    prompt_logprobs_dict: dict[str, LogprobsTensors | None] = field(
        default_factory=dict
    )

    # req_id -> [num_scored_rows, num_token_ids] prompt_logprob_token_ids scores.
    prompt_token_id_logprobs_dict: dict[str, torch.Tensor] = field(default_factory=dict)

    # [num_reqs, hidden_size]
    pooler_output: list[torch.Tensor | None] | None = None

    kv_connector_output: KVConnectorOutput | None = None

    ec_connector_output: ECConnectorOutput | None = None

    # req_id -> num_nans_in_logits
    num_nans_in_logits: dict[str, int] | None = None

    # information related to cudagraph execution
    cudagraph_stats: CUDAGraphStat | None = None

    aux_output_connector_output: dict[str, AuxRequestOutput] | None = None

    # ``None`` when ``return_sampling_mask`` is off.
    sampling_masks: SamplingMaskLists | None = None

    @staticmethod
    def with_kv_conn_output_only(
        kv_connector_output: KVConnectorOutput | None,
    ) -> "ModelRunnerOutput":
        """Return ModelRunnerOutput containing the provided KVConnectorOutput,
        otherwise empty. Returns None if kv_connector_output is passed as None.
        """
        if kv_connector_output is None or kv_connector_output.is_empty():
            return EMPTY_MODEL_RUNNER_OUTPUT
        output = copy(EMPTY_MODEL_RUNNER_OUTPUT)
        output.kv_connector_output = kv_connector_output
        return output

    @staticmethod
    def with_ec_conn_output_only(
        ec_connector_output: ECConnectorOutput | None,
    ) -> "ModelRunnerOutput":
        """Return an otherwise-empty output carrying `ec_connector_output`."""
        return ModelRunnerOutput.with_ec_conn_output(
            EMPTY_MODEL_RUNNER_OUTPUT, ec_connector_output
        )

    @staticmethod
    def with_ec_conn_output(
        output: "ModelRunnerOutput",
        ec_connector_output: ECConnectorOutput | None,
    ) -> "ModelRunnerOutput":
        """Return `output` carrying `ec_connector_output`.

        The shared empty output is copied rather than written to, so callers
        must use the return value.
        """
        if ec_connector_output is None or ec_connector_output.is_empty():
            return output
        if output is EMPTY_MODEL_RUNNER_OUTPUT:
            output = copy(EMPTY_MODEL_RUNNER_OUTPUT)
        output.ec_connector_output = ec_connector_output
        return output

with_ec_conn_output(output, ec_connector_output) staticmethod

Return output carrying ec_connector_output.

The shared empty output is copied rather than written to, so callers must use the return value.

Source code in vllm/v1/outputs.py
@staticmethod
def with_ec_conn_output(
    output: "ModelRunnerOutput",
    ec_connector_output: ECConnectorOutput | None,
) -> "ModelRunnerOutput":
    """Return `output` carrying `ec_connector_output`.

    The shared empty output is copied rather than written to, so callers
    must use the return value.
    """
    if ec_connector_output is None or ec_connector_output.is_empty():
        return output
    if output is EMPTY_MODEL_RUNNER_OUTPUT:
        output = copy(EMPTY_MODEL_RUNNER_OUTPUT)
    output.ec_connector_output = ec_connector_output
    return output

with_ec_conn_output_only(ec_connector_output) staticmethod

Return an otherwise-empty output carrying ec_connector_output.

Source code in vllm/v1/outputs.py
@staticmethod
def with_ec_conn_output_only(
    ec_connector_output: ECConnectorOutput | None,
) -> "ModelRunnerOutput":
    """Return an otherwise-empty output carrying `ec_connector_output`."""
    return ModelRunnerOutput.with_ec_conn_output(
        EMPTY_MODEL_RUNNER_OUTPUT, ec_connector_output
    )

with_kv_conn_output_only(kv_connector_output) staticmethod

Return ModelRunnerOutput containing the provided KVConnectorOutput, otherwise empty. Returns None if kv_connector_output is passed as None.

Source code in vllm/v1/outputs.py
@staticmethod
def with_kv_conn_output_only(
    kv_connector_output: KVConnectorOutput | None,
) -> "ModelRunnerOutput":
    """Return ModelRunnerOutput containing the provided KVConnectorOutput,
    otherwise empty. Returns None if kv_connector_output is passed as None.
    """
    if kv_connector_output is None or kv_connector_output.is_empty():
        return EMPTY_MODEL_RUNNER_OUTPUT
    output = copy(EMPTY_MODEL_RUNNER_OUTPUT)
    output.kv_connector_output = kv_connector_output
    return output

SamplingMaskLists

Bases: NamedTuple

CSR sampling masks; a request slice may contain multiple positions.

Source code in vllm/v1/outputs.py
class SamplingMaskLists(NamedTuple):
    """CSR sampling masks; a request slice may contain multiple positions."""

    # [num_kept_tokens]
    token_ids: np.ndarray
    # [num_positions + 1], or None for a single position
    offsets: np.ndarray | None = None
    # [num_requests + 1] for multi-position request batches.
    cu_num_generated_tokens: list[int] | None = None

    def slice_request(self, req_idx: int, num_positions: int) -> "SamplingMaskLists":
        assert self.offsets is not None
        cu = self.cu_num_generated_tokens
        start = req_idx if cu is None else cu[req_idx]
        end = start + num_positions
        lo, hi = self.offsets[start], self.offsets[end]
        if num_positions == 1:
            return SamplingMaskLists(self.token_ids[lo:hi])
        return SamplingMaskLists(
            self.token_ids[lo:hi], self.offsets[start : end + 1] - lo
        )

    def to_nested_list(self) -> list[list[int]]:
        token_ids = self.token_ids.tolist()
        if self.offsets is None:
            return [token_ids]
        offsets = self.offsets.tolist()
        return [token_ids[offsets[i] : offsets[i + 1]] for i in range(len(offsets) - 1)]

make_empty_encoder_model_runner_output(scheduler_output)

Create a ModelRunnerOutput stub that contains the correct per-request bookkeeping but no generated data yet.

Source code in vllm/v1/outputs.py
def make_empty_encoder_model_runner_output(
    scheduler_output: "SchedulerOutput",
) -> ModelRunnerOutput:
    """Create a ModelRunnerOutput stub that contains the correct
    per-request bookkeeping but no generated data yet.
    """
    if not scheduler_output.num_scheduled_tokens:
        return EMPTY_MODEL_RUNNER_OUTPUT

    # Convert to list so we get a deterministic, indexable sequence
    req_ids: list[str] = list(scheduler_output.num_scheduled_tokens.keys())

    # Give every request its own contiguous index
    req_id_to_index: dict[str, int] = {rid: idx for idx, rid in enumerate(req_ids)}

    # An encoder instance never samples, so it emits no tokens at all. The
    # scheduler finishes these requests once their prompt is fully encoded
    # (see `Scheduler.update_from_output`).
    sampled_token_ids: list[list[int]] = [[] for _ in req_ids]

    # Pooler outputs are not available yet ⇒ use None placeholders
    pooler_output: list[torch.Tensor | None] = [None for _ in req_ids]

    return ModelRunnerOutput(
        req_ids=req_ids,
        req_id_to_index=req_id_to_index,
        sampled_token_ids=sampled_token_ids,
        pooler_output=pooler_output,
    )