Skip to content

vllm.config.observability

Classes:

ObservabilityConfig

Configuration for observability - metrics and tracing.

Methods:

  • compute_hash –

    WARNING: Whenever a new field is added to this config,

Attributes:

Source code in vllm/config/observability.py
@config
class ObservabilityConfig:
    """Configuration for observability - metrics and tracing."""

    show_hidden_metrics_for_version: str | None = None
    """Enable deprecated Prometheus metrics that have been hidden since the
    specified version. For example, if a previously deprecated metric has been
    hidden since the v0.7.0 release, you use
    `--show-hidden-metrics-for-version=0.7` as a temporary escape hatch while
    you migrate to new metrics. The metric is likely to be removed completely
    in an upcoming release."""

    @cached_property
    def show_hidden_metrics(self) -> bool:
        """Check if the hidden metrics should be shown."""
        if self.show_hidden_metrics_for_version is None:
            return False
        return version._prev_minor_version_was(self.show_hidden_metrics_for_version)

    otlp_traces_endpoint: str | None = None
    """Target URL to which OpenTelemetry traces will be sent."""

    collect_detailed_traces: list[DetailedTraceModules] | None = None
    """It makes sense to set this only if `--otlp-traces-endpoint` is set. If
    set, it will collect detailed traces for the specified modules. This
    involves use of possibly costly and or blocking operations and hence might
    have a performance impact.

    Note that collecting detailed timing information for each request can be
    expensive."""

    per_request_spec_decode_metrics: Literal["none", "summary", "detailed"] = "none"
    """Include per-request speculative-decoding acceptance metrics in the
    response under `metrics.speculative_decoding`. `none` disables; `summary` adds mean
    acceptance length, draft acceptance rate, and the step-by-draft-length
    histogram; `detailed` additionally records the ordered per-step
    accepted/proposed arrays (one entry per verify step). Only reported for
    single-sequence requests (`n == 1`), mirroring the timing metrics. No effect
    unless speculative decoding is enabled. Independent of `--disable-log-stats`.
    This is the per-request response-body counterpart of the aggregate
    `vllm:spec_decode_*` Prometheus metrics. The response field is experimental
    and its shape may change in a future release."""

    kv_cache_metrics: bool = False
    """Enable KV cache residency metrics (lifetime, idle time, reuse gaps).
    Uses sampling to minimize overhead.
    Requires log stats to be enabled (i.e., --disable-log-stats not set)."""

    kv_cache_metrics_sample: float = Field(default=0.01, gt=0, le=1)
    """Sampling rate for KV cache metrics (0.0, 1.0]. Default 0.01 = 1% of blocks."""

    custom_histogram_buckets: dict[str, list[float]] | None = None
    """Custom Prometheus histogram bucket boundaries, as a JSON mapping from
    bucket-family key to a list of strictly increasing, positive, finite
    upper bounds. When a family key is present, its list replaces the default
    buckets for every histogram in that family; families not listed keep
    their defaults. Known families: `request_latency`, `time_to_first_token`,
    `inter_token_latency`, `iteration_tokens`, `request_params_n`,
    `request_num_preemptions`, `request_tokens`, `kv_cache_residency`. Example:
    `--custom-histogram-buckets '{"request_latency": [0.01, 0.05, 0.1, 0.5]}'`.
    Note that every extra bucket adds one time series per metric and label
    set."""

    cudagraph_metrics: bool = False
    """Enable CUDA graph metrics (number of padded/unpadded tokens, runtime cudagraph
    dispatch modes, and their observed frequencies at every logging interval)."""

    enable_layerwise_nvtx_tracing: bool = False
    """Enable layerwise NVTX tracing. This traces the execution of each layer or
    module in the model and attach information such as input/output shapes to
    nvtx range markers. Noted that this doesn't work with CUDA graphs enabled."""

    enable_mfu_metrics: bool = False
    """Enable Model FLOPs Utilization (MFU) metrics."""

    enable_mm_processor_stats: bool = False
    """Enable collection of timing statistics for multimodal processor operations.
    This is for internal use only (e.g., benchmarks) and is not exposed as a CLI
    argument."""

    enable_logging_iteration_details: bool = False
    """Enable detailed logging of iteration details.
    If set, vllm EngineCore will log iteration details
    This includes number of context/generation requests and tokens
    and the elapsed cpu time for the iteration."""

    jit_monitor_mode: Literal["warn", "error"] = "warn"
    """How to handle post-warmup JIT compilation events."""

    jit_monitor_verbose: bool = False
    """Log every monitored JIT compile with runtime details. This can emit many
    logs and add overhead, so it is intended for debugging."""

    @cached_property
    def collect_model_forward_time(self) -> bool:
        """Whether to collect model forward time for the request."""
        return self.collect_detailed_traces is not None and (
            "model" in self.collect_detailed_traces
            or "all" in self.collect_detailed_traces
        )

    @cached_property
    def collect_model_execute_time(self) -> bool:
        """Whether to collect model execute time for the request."""
        return self.collect_detailed_traces is not None and (
            "worker" in self.collect_detailed_traces
            or "all" in self.collect_detailed_traces
        )

    def compute_hash(self) -> str:
        """WARNING: Whenever a new field is added to this config,
        ensure that it is included in the factors list if
        it affects the computation graph.

        Provide a hash that uniquely identifies all the configs
        that affect the structure of the computation
        graph from input ids/embeddings to the final hidden states,
        excluding anything before input ids/embeddings and after
        the final hidden states.
        """
        # no factors to consider.
        # this config will not affect the computation graph.
        factors: list[Any] = []
        hash_str = safe_hash(str(factors).encode(), usedforsecurity=False).hexdigest()
        return hash_str

    @field_validator("show_hidden_metrics_for_version")
    @classmethod
    def _validate_show_hidden_metrics_for_version(cls, value: str | None) -> str | None:
        if value is not None:
            # Raises an exception if the string is not a valid version.
            parse(value)
        return value

    @field_validator("otlp_traces_endpoint")
    @classmethod
    def _validate_otlp_traces_endpoint(cls, value: str | None) -> str | None:
        if value is not None:
            from vllm.tracing import is_tracing_available, otel_import_error_traceback

            if not is_tracing_available():
                raise ValueError(
                    "OpenTelemetry is not available. Unable to configure "
                    "'otlp_traces_endpoint'. Ensure OpenTelemetry packages are "
                    f"installed. Original error:\n{otel_import_error_traceback}"
                )
        return value

    @field_validator("custom_histogram_buckets", mode="before")
    @classmethod
    def _reject_bool_histogram_bounds(cls, value: object) -> object:
        """Reject booleans before pydantic silently coerces them to floats."""
        if isinstance(value, dict):
            for family, buckets in value.items():
                if not isinstance(buckets, list):
                    continue
                for bound in buckets:
                    if isinstance(bound, bool):
                        raise ValueError(
                            f"custom_histogram_buckets[{family!r}]: bound "
                            f"{bound!r} must be a number, not a boolean"
                        )
        return value

    @field_validator("custom_histogram_buckets")
    @classmethod
    def _validate_custom_histogram_buckets(
        cls, value: dict[str, list[float]] | None
    ) -> dict[str, list[float]] | None:
        if value is None:
            return value
        from vllm.v1.metrics.buckets import BUCKET_FAMILY_KEYS

        for family, buckets in value.items():
            if family not in BUCKET_FAMILY_KEYS:
                raise ValueError(
                    f"custom_histogram_buckets: unknown bucket family "
                    f"{family!r}; known families: {sorted(BUCKET_FAMILY_KEYS)}"
                )
            if not buckets:
                raise ValueError(
                    f"custom_histogram_buckets[{family!r}]: bucket list "
                    "must not be empty"
                )
            for bound in buckets:
                if not math.isfinite(bound) or bound <= 0:
                    raise ValueError(
                        f"custom_histogram_buckets[{family!r}]: bound "
                        f"{bound!r} must be finite and greater than 0"
                    )
            if any(a >= b for a, b in pairwise(buckets)):
                raise ValueError(
                    f"custom_histogram_buckets[{family!r}]: bounds {buckets} "
                    "must be strictly increasing"
                )
        return value

    @model_validator(mode="after")
    def _validate_tracing_config(self):
        if self.collect_detailed_traces and not self.otlp_traces_endpoint:
            raise ValueError(
                "collect_detailed_traces requires `--otlp-traces-endpoint` to be set."
            )
        return self

collect_detailed_traces = None class-attribute instance-attribute

It makes sense to set this only if --otlp-traces-endpoint is set. If set, it will collect detailed traces for the specified modules. This involves use of possibly costly and or blocking operations and hence might have a performance impact.

Note that collecting detailed timing information for each request can be expensive.

collect_model_execute_time cached property

Whether to collect model execute time for the request.

collect_model_forward_time cached property

Whether to collect model forward time for the request.

cudagraph_metrics = False class-attribute instance-attribute

Enable CUDA graph metrics (number of padded/unpadded tokens, runtime cudagraph dispatch modes, and their observed frequencies at every logging interval).

custom_histogram_buckets = None class-attribute instance-attribute

Custom Prometheus histogram bucket boundaries, as a JSON mapping from bucket-family key to a list of strictly increasing, positive, finite upper bounds. When a family key is present, its list replaces the default buckets for every histogram in that family; families not listed keep their defaults. Known families: request_latency, time_to_first_token, inter_token_latency, iteration_tokens, request_params_n, request_num_preemptions, request_tokens, kv_cache_residency. Example: --custom-histogram-buckets '{"request_latency": [0.01, 0.05, 0.1, 0.5]}'. Note that every extra bucket adds one time series per metric and label set.

enable_layerwise_nvtx_tracing = False class-attribute instance-attribute

Enable layerwise NVTX tracing. This traces the execution of each layer or module in the model and attach information such as input/output shapes to nvtx range markers. Noted that this doesn't work with CUDA graphs enabled.

enable_logging_iteration_details = False class-attribute instance-attribute

Enable detailed logging of iteration details. If set, vllm EngineCore will log iteration details This includes number of context/generation requests and tokens and the elapsed cpu time for the iteration.

enable_mfu_metrics = False class-attribute instance-attribute

Enable Model FLOPs Utilization (MFU) metrics.

enable_mm_processor_stats = False class-attribute instance-attribute

Enable collection of timing statistics for multimodal processor operations. This is for internal use only (e.g., benchmarks) and is not exposed as a CLI argument.

jit_monitor_mode = 'warn' class-attribute instance-attribute

How to handle post-warmup JIT compilation events.

jit_monitor_verbose = False class-attribute instance-attribute

Log every monitored JIT compile with runtime details. This can emit many logs and add overhead, so it is intended for debugging.

kv_cache_metrics = False class-attribute instance-attribute

Enable KV cache residency metrics (lifetime, idle time, reuse gaps). Uses sampling to minimize overhead. Requires log stats to be enabled (i.e., --disable-log-stats not set).

kv_cache_metrics_sample = Field(default=0.01, gt=0, le=1) class-attribute instance-attribute

Sampling rate for KV cache metrics (0.0, 1.0]. Default 0.01 = 1% of blocks.

otlp_traces_endpoint = None class-attribute instance-attribute

Target URL to which OpenTelemetry traces will be sent.

per_request_spec_decode_metrics = 'none' class-attribute instance-attribute

Include per-request speculative-decoding acceptance metrics in the response under metrics.speculative_decoding. none disables; summary adds mean acceptance length, draft acceptance rate, and the step-by-draft-length histogram; detailed additionally records the ordered per-step accepted/proposed arrays (one entry per verify step). Only reported for single-sequence requests (n == 1), mirroring the timing metrics. No effect unless speculative decoding is enabled. Independent of --disable-log-stats. This is the per-request response-body counterpart of the aggregate vllm:spec_decode_* Prometheus metrics. The response field is experimental and its shape may change in a future release.

show_hidden_metrics cached property

Check if the hidden metrics should be shown.

show_hidden_metrics_for_version = None class-attribute instance-attribute

Enable deprecated Prometheus metrics that have been hidden since the specified version. For example, if a previously deprecated metric has been hidden since the v0.7.0 release, you use --show-hidden-metrics-for-version=0.7 as a temporary escape hatch while you migrate to new metrics. The metric is likely to be removed completely in an upcoming release.

_reject_bool_histogram_bounds(value) classmethod

Reject booleans before pydantic silently coerces them to floats.

Source code in vllm/config/observability.py
@field_validator("custom_histogram_buckets", mode="before")
@classmethod
def _reject_bool_histogram_bounds(cls, value: object) -> object:
    """Reject booleans before pydantic silently coerces them to floats."""
    if isinstance(value, dict):
        for family, buckets in value.items():
            if not isinstance(buckets, list):
                continue
            for bound in buckets:
                if isinstance(bound, bool):
                    raise ValueError(
                        f"custom_histogram_buckets[{family!r}]: bound "
                        f"{bound!r} must be a number, not a boolean"
                    )
    return value

compute_hash()

WARNING: Whenever a new field is added to this config, ensure that it is included in the factors list if it affects the computation graph.

Provide a hash that uniquely identifies all the configs that affect the structure of the computation graph from input ids/embeddings to the final hidden states, excluding anything before input ids/embeddings and after the final hidden states.

Source code in vllm/config/observability.py
def compute_hash(self) -> str:
    """WARNING: Whenever a new field is added to this config,
    ensure that it is included in the factors list if
    it affects the computation graph.

    Provide a hash that uniquely identifies all the configs
    that affect the structure of the computation
    graph from input ids/embeddings to the final hidden states,
    excluding anything before input ids/embeddings and after
    the final hidden states.
    """
    # no factors to consider.
    # this config will not affect the computation graph.
    factors: list[Any] = []
    hash_str = safe_hash(str(factors).encode(), usedforsecurity=False).hexdigest()
    return hash_str