Skip to content

vllm.config.engram

Classes:

  • EngramConfig –

    Configuration for Engram embedding storage and sharding.

Functions:

EngramConfig

Configuration for Engram embedding storage and sharding.

Methods:

Attributes:

  • cpu_offload (bool) –

    Store embedding weights in pinned CPU memory for UVA lookup.

  • dp_shared_memory (bool | None) –

    Share CPU-offloaded embedding weights between co-located

  • embedding_across_dp (bool) –

    Shard embeddings across TP and all DP ranks when enabled.

  • use_thp (bool) –

    Back private CPU-offloaded tables with transparent huge pages (best

Source code in vllm/config/engram.py
@config
class EngramConfig:
    """Configuration for Engram embedding storage and sharding."""

    cpu_offload: bool = True
    """Store embedding weights in pinned CPU memory for UVA lookup."""

    embedding_across_dp: bool = False
    """Shard embeddings across TP and all DP ranks when enabled.
    Otherwise, each DP rank has a separate TP-sharded embedding replica."""

    dp_shared_memory: bool | None = None
    """Share CPU-offloaded embedding weights between co-located
    DP replicas. Each node stores one copy of every TP shard, reducing host
    memory without per-step Engram DP collectives. Requires sufficient
    /dev/shm capacity and a shared IPC namespace. Defaults to enabled whenever
    the other settings allow it, falling back to per-replica tables when DP
    replicas are not co-located on one node or /dev/shm cannot hold them."""

    use_thp: bool = False
    """Back private CPU-offloaded tables with transparent huge pages (best
    effort, falls back to ordinary pinned pages). Prefaulting the tables at
    startup takes longer. Requires cpu_offload without dp_shared_memory."""

    @model_validator(mode="after")
    def _validate_shared_memory(self) -> Self:
        if self.dp_shared_memory and not self.cpu_offload:
            raise ValueError("dp_shared_memory requires cpu_offload=True")
        if self.use_thp and (not self.cpu_offload or self.dp_shared_memory):
            raise ValueError(
                "use_thp requires cpu_offload=True and dp_shared_memory=False"
            )
        return self

    def verify_model_config(self, model_config: "ModelConfig | None") -> None:
        """Reject Engram configuration for models without n-gram embeddings."""
        field = (
            _NGRAM_LAYER_FIELDS.get(model_config.architecture)
            if model_config is not None
            else None
        )
        if (
            model_config is None
            or field is None
            or not getattr(model_config.hf_text_config, field, None)
        ):
            raise ValueError(
                "EngramConfig requires a model with supported Engram "
                "embeddings and non-empty n-gram layer ids."
            )

    def resolve_dp_shared_memory(self, parallel_config: "ParallelConfig") -> None:
        """Share host tables by default wherever the configuration permits."""
        if self.dp_shared_memory is None:
            self.dp_shared_memory = (
                self.cpu_offload
                and not self.use_thp
                and parallel_config.data_parallel_size > 1
                and not parallel_config.enable_elastic_ep
            )

    def verify_parallel_config(self, parallel_config: "ParallelConfig") -> None:
        """Reject unsupported embedding parallel topologies."""
        if self.dp_shared_memory:
            if parallel_config.data_parallel_size <= 1:
                raise ValueError("dp_shared_memory requires data_parallel_size > 1.")
            if parallel_config.enable_elastic_ep:
                raise ValueError("dp_shared_memory is not supported with elastic EP.")
        if (
            self.embedding_across_dp
            and parallel_config.data_parallel_size > 1
            and parallel_config.enable_elastic_ep
        ):
            raise ValueError(
                "Engram embedding_across_dp is not supported with elastic EP yet."
            )

    def get_parallel_size(self, parallel_config: "ParallelConfig") -> int:
        """Derive the embedding group size from the parallel configuration."""
        size = parallel_config.tensor_parallel_size
        if self.embedding_across_dp and parallel_config.data_parallel_size > 1:
            size *= parallel_config.data_parallel_size
        return size

    def compute_hash(self) -> str:
        """Hash settings that affect embedding execution and graph structure."""
        return hash_factors(get_hash_factors(self, set()))

cpu_offload = True class-attribute instance-attribute

Store embedding weights in pinned CPU memory for UVA lookup.

dp_shared_memory = None class-attribute instance-attribute

Share CPU-offloaded embedding weights between co-located DP replicas. Each node stores one copy of every TP shard, reducing host memory without per-step Engram DP collectives. Requires sufficient /dev/shm capacity and a shared IPC namespace. Defaults to enabled whenever the other settings allow it, falling back to per-replica tables when DP replicas are not co-located on one node or /dev/shm cannot hold them.

embedding_across_dp = False class-attribute instance-attribute

Shard embeddings across TP and all DP ranks when enabled. Otherwise, each DP rank has a separate TP-sharded embedding replica.

use_thp = False class-attribute instance-attribute

Back private CPU-offloaded tables with transparent huge pages (best effort, falls back to ordinary pinned pages). Prefaulting the tables at startup takes longer. Requires cpu_offload without dp_shared_memory.

compute_hash()

Hash settings that affect embedding execution and graph structure.

Source code in vllm/config/engram.py
def compute_hash(self) -> str:
    """Hash settings that affect embedding execution and graph structure."""
    return hash_factors(get_hash_factors(self, set()))

get_parallel_size(parallel_config)

Derive the embedding group size from the parallel configuration.

Source code in vllm/config/engram.py
def get_parallel_size(self, parallel_config: "ParallelConfig") -> int:
    """Derive the embedding group size from the parallel configuration."""
    size = parallel_config.tensor_parallel_size
    if self.embedding_across_dp and parallel_config.data_parallel_size > 1:
        size *= parallel_config.data_parallel_size
    return size

resolve_dp_shared_memory(parallel_config)

Share host tables by default wherever the configuration permits.

Source code in vllm/config/engram.py
def resolve_dp_shared_memory(self, parallel_config: "ParallelConfig") -> None:
    """Share host tables by default wherever the configuration permits."""
    if self.dp_shared_memory is None:
        self.dp_shared_memory = (
            self.cpu_offload
            and not self.use_thp
            and parallel_config.data_parallel_size > 1
            and not parallel_config.enable_elastic_ep
        )

verify_model_config(model_config)

Reject Engram configuration for models without n-gram embeddings.

Source code in vllm/config/engram.py
def verify_model_config(self, model_config: "ModelConfig | None") -> None:
    """Reject Engram configuration for models without n-gram embeddings."""
    field = (
        _NGRAM_LAYER_FIELDS.get(model_config.architecture)
        if model_config is not None
        else None
    )
    if (
        model_config is None
        or field is None
        or not getattr(model_config.hf_text_config, field, None)
    ):
        raise ValueError(
            "EngramConfig requires a model with supported Engram "
            "embeddings and non-empty n-gram layer ids."
        )

verify_parallel_config(parallel_config)

Reject unsupported embedding parallel topologies.

Source code in vllm/config/engram.py
def verify_parallel_config(self, parallel_config: "ParallelConfig") -> None:
    """Reject unsupported embedding parallel topologies."""
    if self.dp_shared_memory:
        if parallel_config.data_parallel_size <= 1:
            raise ValueError("dp_shared_memory requires data_parallel_size > 1.")
        if parallel_config.enable_elastic_ep:
            raise ValueError("dp_shared_memory is not supported with elastic EP.")
    if (
        self.embedding_across_dp
        and parallel_config.data_parallel_size > 1
        and parallel_config.enable_elastic_ep
    ):
        raise ValueError(
            "Engram embedding_across_dp is not supported with elastic EP yet."
        )

model_has_engram_layers(model_config)

Whether the model carries n-gram embedding layers.

Source code in vllm/config/engram.py
def model_has_engram_layers(model_config: "ModelConfig | None") -> bool:
    """Whether the model carries n-gram embedding layers."""
    if model_config is None:
        return False
    field = _NGRAM_LAYER_FIELDS.get(model_config.architecture)
    if field is None:
        return False
    return bool(getattr(model_config.hf_text_config, field, None))