Skip to content

vllm.multimodal.cache

Modules:

  • base –
  • factories –
  • lru –

    Implementation of Key-Replicated Cache (see docs/configuration/optimization.md).

  • shm –

    Implementation of Shared Memory Cache (see docs/configuration/optimization.md).

Classes:

Functions:

BaseMultiModalProcessorCache

Bases: BaseMultiModalCache[MultiModalProcessorCacheInItem, MultiModalProcessorCacheOutItem]

The required interface for caches on P0.

Methods:

  • close –

    Close the underlying cache, if needed.

  • invalidate –

    Drop mm_hash from this P0 shadow cache to recover from P0/P1 drift.

  • is_cached –

    Check whether a sequence of multi-modal items are

  • is_cached_item –

    Check whether a multi-modal item is

  • make_stats –

    Get (and reset) the multi-modal cache stats.

  • release_sender_touches –

    Release the items that were touched but not updated afterwards.

  • touch_sender_cache_item –

    Update the cache eviction order for a multi-modal item.

  • validate_input_item –

    Validate externally supplied cache metadata before engine handoff.

Source code in vllm/multimodal/cache/base.py
class BaseMultiModalProcessorCache(
    BaseMultiModalCache[MultiModalProcessorCacheInItem, MultiModalProcessorCacheOutItem]
):
    """The required interface for caches on P0."""

    @abstractmethod
    def is_cached_item(self, mm_hash: str) -> bool:
        """Check whether a multi-modal item is
        in the underlying cache.

        This **DOES NOT** update the cache eviction order.

        Args:
            mm_hash: The hash of the item to check.

        Returns:
            `True` if the item is cached, otherwise `False`.

        """
        raise NotImplementedError

    def is_cached(self, mm_hashes: list[str]) -> list[bool]:
        """Check whether a sequence of multi-modal items are
        in the underlying cache.

        This **DOES NOT** update the cache eviction order.

        Args:
            mm_hashes: The hash of each item to check.

        Returns:
            For each item, `True` if the item is cached, otherwise `False`.

        """
        return [self.is_cached_item(mm_hash) for mm_hash in mm_hashes]

    def invalidate(self, mm_hash: str) -> None:
        """Drop ``mm_hash`` from this P0 shadow cache to recover from P0/P1 drift.

        No-op by default; shadow caches that can drift from P1 override this.
        """
        pass

    def close(self) -> None:
        """Close the underlying cache, if needed."""
        pass

    def validate_input_item(
        self,
        mm_item: MultiModalKwargsItem,
        mm_hash: str,
    ) -> None:
        """Validate externally supplied cache metadata before engine handoff."""
        return None

    @abstractmethod
    def touch_sender_cache_item(self, mm_hash: str) -> None:
        """Update the cache eviction order for a multi-modal item.

        This is used to touch the item in the cache without changing
        its value.

        Args:
            mm_hash: The hash of the multi-modal item.

        """
        raise NotImplementedError

    def release_sender_touches(self) -> None:
        """Release the items that were touched but not updated afterwards."""
        return None

    @abstractmethod
    def make_stats(self, *, delta: bool = False) -> CacheInfo:
        """Get (and reset) the multi-modal cache stats.

        Returns:
            The current multi-modal caching stats.

        """
        raise NotImplementedError

close()

Close the underlying cache, if needed.

Source code in vllm/multimodal/cache/base.py
def close(self) -> None:
    """Close the underlying cache, if needed."""
    pass

invalidate(mm_hash)

Drop mm_hash from this P0 shadow cache to recover from P0/P1 drift.

No-op by default; shadow caches that can drift from P1 override this.

Source code in vllm/multimodal/cache/base.py
def invalidate(self, mm_hash: str) -> None:
    """Drop ``mm_hash`` from this P0 shadow cache to recover from P0/P1 drift.

    No-op by default; shadow caches that can drift from P1 override this.
    """
    pass

is_cached(mm_hashes)

Check whether a sequence of multi-modal items are in the underlying cache.

This DOES NOT update the cache eviction order.

Parameters:

  • mm_hashes

    (list[str]) –

    The hash of each item to check.

Returns:

  • list[bool] –

    For each item, True if the item is cached, otherwise False.

Source code in vllm/multimodal/cache/base.py
def is_cached(self, mm_hashes: list[str]) -> list[bool]:
    """Check whether a sequence of multi-modal items are
    in the underlying cache.

    This **DOES NOT** update the cache eviction order.

    Args:
        mm_hashes: The hash of each item to check.

    Returns:
        For each item, `True` if the item is cached, otherwise `False`.

    """
    return [self.is_cached_item(mm_hash) for mm_hash in mm_hashes]

is_cached_item(mm_hash) abstractmethod

Check whether a multi-modal item is in the underlying cache.

This DOES NOT update the cache eviction order.

Parameters:

  • mm_hash

    (str) –

    The hash of the item to check.

Returns:

  • bool –

    True if the item is cached, otherwise False.

Source code in vllm/multimodal/cache/base.py
@abstractmethod
def is_cached_item(self, mm_hash: str) -> bool:
    """Check whether a multi-modal item is
    in the underlying cache.

    This **DOES NOT** update the cache eviction order.

    Args:
        mm_hash: The hash of the item to check.

    Returns:
        `True` if the item is cached, otherwise `False`.

    """
    raise NotImplementedError

make_stats(*, delta=False) abstractmethod

Get (and reset) the multi-modal cache stats.

Returns:

  • CacheInfo –

    The current multi-modal caching stats.

Source code in vllm/multimodal/cache/base.py
@abstractmethod
def make_stats(self, *, delta: bool = False) -> CacheInfo:
    """Get (and reset) the multi-modal cache stats.

    Returns:
        The current multi-modal caching stats.

    """
    raise NotImplementedError

release_sender_touches()

Release the items that were touched but not updated afterwards.

Source code in vllm/multimodal/cache/base.py
def release_sender_touches(self) -> None:
    """Release the items that were touched but not updated afterwards."""
    return None

touch_sender_cache_item(mm_hash) abstractmethod

Update the cache eviction order for a multi-modal item.

This is used to touch the item in the cache without changing its value.

Parameters:

  • mm_hash

    (str) –

    The hash of the multi-modal item.

Source code in vllm/multimodal/cache/base.py
@abstractmethod
def touch_sender_cache_item(self, mm_hash: str) -> None:
    """Update the cache eviction order for a multi-modal item.

    This is used to touch the item in the cache without changing
    its value.

    Args:
        mm_hash: The hash of the multi-modal item.

    """
    raise NotImplementedError

validate_input_item(mm_item, mm_hash)

Validate externally supplied cache metadata before engine handoff.

Source code in vllm/multimodal/cache/base.py
def validate_input_item(
    self,
    mm_item: MultiModalKwargsItem,
    mm_hash: str,
) -> None:
    """Validate externally supplied cache metadata before engine handoff."""
    return None

BaseMultiModalReceiverCache

Bases: BaseMultiModalCache[MultiModalKwargsItem | None, MultiModalKwargsItem]

The required interface for caches on P1.

Methods:

Source code in vllm/multimodal/cache/base.py
class BaseMultiModalReceiverCache(
    BaseMultiModalCache[MultiModalKwargsItem | None, MultiModalKwargsItem]
):
    """The required interface for caches on P1."""

    def get_and_update_features(
        self,
        mm_features: list["MultiModalFeatureSpec"],
    ) -> list["MultiModalFeatureSpec"]:
        """Update multimodal features with cached encoder outputs.
        Touch all identifier at first before update to avoid
        item in updated list evict during update.

        Uses mm_hash for cache key to share across LoRAs (falls back to
        identifier for backward compatibility).
        """
        for feature in mm_features:
            cache_key = feature.mm_hash or feature.identifier
            self.touch_receiver_cache_item(cache_key, feature.data)

        missing_mm_hashes: list[str] = []
        for feature in mm_features:
            cache_key = feature.mm_hash or feature.identifier
            try:
                feature.data = self.get_and_update_item(feature.data, cache_key)
            except MultiModalCacheMissError as e:
                # Collect every drifted hash in this request before raising, so the
                # engine can have P0 invalidate them all at once -- otherwise a
                # request with k drifted items needs k client retries (each resend
                # only un-shadows the one reported hash).
                missing_mm_hashes.extend(e.mm_hashes)
        if missing_mm_hashes:
            raise MultiModalCacheMissError(missing_mm_hashes)
        return mm_features

    @abstractmethod
    def touch_receiver_cache_item(
        self,
        mm_hash: str,
        mm_item: MultiModalKwargsItem | None = None,
    ) -> None:
        """Update the cache eviction order for a multi-modal item.

        This is used to touch the item in the cache without changing
        its value.

        Args:
            mm_hash: The hash of the multi-modal item.
            mm_item: The multi-modal item itself. This is optional and
                may not be needed by some cache implementations.

        """
        raise NotImplementedError

get_and_update_features(mm_features)

Update multimodal features with cached encoder outputs. Touch all identifier at first before update to avoid item in updated list evict during update.

Uses mm_hash for cache key to share across LoRAs (falls back to identifier for backward compatibility).

Source code in vllm/multimodal/cache/base.py
def get_and_update_features(
    self,
    mm_features: list["MultiModalFeatureSpec"],
) -> list["MultiModalFeatureSpec"]:
    """Update multimodal features with cached encoder outputs.
    Touch all identifier at first before update to avoid
    item in updated list evict during update.

    Uses mm_hash for cache key to share across LoRAs (falls back to
    identifier for backward compatibility).
    """
    for feature in mm_features:
        cache_key = feature.mm_hash or feature.identifier
        self.touch_receiver_cache_item(cache_key, feature.data)

    missing_mm_hashes: list[str] = []
    for feature in mm_features:
        cache_key = feature.mm_hash or feature.identifier
        try:
            feature.data = self.get_and_update_item(feature.data, cache_key)
        except MultiModalCacheMissError as e:
            # Collect every drifted hash in this request before raising, so the
            # engine can have P0 invalidate them all at once -- otherwise a
            # request with k drifted items needs k client retries (each resend
            # only un-shadows the one reported hash).
            missing_mm_hashes.extend(e.mm_hashes)
    if missing_mm_hashes:
        raise MultiModalCacheMissError(missing_mm_hashes)
    return mm_features

touch_receiver_cache_item(mm_hash, mm_item=None) abstractmethod

Update the cache eviction order for a multi-modal item.

This is used to touch the item in the cache without changing its value.

Parameters:

  • mm_hash

    (str) –

    The hash of the multi-modal item.

  • mm_item

    (MultiModalKwargsItem | None, default: None ) –

    The multi-modal item itself. This is optional and may not be needed by some cache implementations.

Source code in vllm/multimodal/cache/base.py
@abstractmethod
def touch_receiver_cache_item(
    self,
    mm_hash: str,
    mm_item: MultiModalKwargsItem | None = None,
) -> None:
    """Update the cache eviction order for a multi-modal item.

    This is used to touch the item in the cache without changing
    its value.

    Args:
        mm_hash: The hash of the multi-modal item.
        mm_item: The multi-modal item itself. This is optional and
            may not be needed by some cache implementations.

    """
    raise NotImplementedError

LruKeyReplicatedReceiverCache

Bases: BaseMultiModalReceiverCache

The cache which is used on P1 when LRU caching is enabled.

How to update each item:

  • If the caller sent tensor data, store it (replacing any cached item under the same key) and return that data. P0 can miss after independent LRU eviction and resend a different item for the same identity.
  • If the caller sent no data and the item is cached, return the cached item.
  • If the caller sent no data and the item is not cached, raise MultiModalCacheMissError.
Source code in vllm/multimodal/cache/lru.py
class LruKeyReplicatedReceiverCache(BaseMultiModalReceiverCache):
    """The cache which is used on P1 when LRU caching is enabled.

    How to update each item:

    - If the caller sent tensor data, store it (replacing any cached item
      under the same key) and return that data. P0 can miss after independent
      LRU eviction and resend a different item for the same identity.
    - If the caller sent no data and the item is cached, return the cached item.
    - If the caller sent no data and the item is not cached, raise
      `MultiModalCacheMissError`.
    """

    def __init__(self, model_config: ModelConfig) -> None:
        super().__init__()

        mm_config = model_config.get_multimodal_config()

        self._cache = MultiModalCache.get_lru_cache(
            mm_config.mm_processor_cache_gb,
            MultiModalKwargsItem,
        )

    @override
    def get_and_update_item(
        self,
        mm_item: MultiModalKwargsItem | None,
        mm_hash: str,
    ) -> MultiModalKwargsItem:
        if mm_item is not None:
            # P0 sent a payload. Never keep a stale cached tensor under this
            # identity: a later P0 hit would pair new placeholders with the old
            # item and crash the engine. Drop first so a too-large replacement
            # cannot leave the previous entry in place.
            self._cache.pop(mm_hash, None)
            self.cache_if_fits(self._cache, mm_hash, mm_item)
            return mm_item

        if (cached_item := self._cache.get(mm_hash)) is not None:
            return cached_item

        # No data and not cached here: P0 sent data=None trusting its shadow, but
        # the P0/P1 caches have drifted. Raise a retryable error (not assert) so the
        # engine can have P0 drop the stale entry and the client resend the data.
        raise MultiModalCacheMissError([mm_hash])

    @override
    def touch_receiver_cache_item(
        self,
        mm_hash: str,
        mm_item: MultiModalKwargsItem | None = None,
    ) -> None:
        self._cache.touch(mm_hash)

    @override
    def clear_cache(self) -> None:
        self._cache.clear()

LruKeyReplicatedSenderCache

Bases: BaseMultiModalProcessorCache

The cache which is used on P0 when LRU caching is enabled.

How to update each item:

  • If the item is already in the cache, clear the input to avoid unnecessary IPC.

  • If the item is not in the cache, store the metadata of that item so that the eviction policy remains the same as the cache on P1, and return the input. By only storing the metadata, we avoid keeping the data itself in memory inside P0.

Source code in vllm/multimodal/cache/lru.py
class LruKeyReplicatedSenderCache(BaseMultiModalProcessorCache):
    """The cache which is used on P0 when LRU caching is enabled.

    How to update each item:

    - If the item is already in the cache, clear the input to avoid
      unnecessary IPC.

    - If the item is not in the cache, store the metadata of that item so
      that the eviction policy remains the same as the cache on P1,
      and return the input.
      By only storing the metadata, we avoid keeping the data itself in
      memory inside P0.
    """

    def __init__(self, model_config: ModelConfig) -> None:
        super().__init__()

        mm_config = model_config.get_multimodal_config()

        self._cache = MultiModalCache.get_lru_cache(
            mm_config.mm_processor_cache_gb,
            MultiModalProcessorCacheItemMetadata,
        )

    @override
    def is_cached_item(self, mm_hash: str) -> bool:
        return mm_hash in self._cache

    @override
    def get_and_update_item(
        self,
        mm_item: MultiModalProcessorCacheInItem,
        mm_hash: str,
    ) -> MultiModalProcessorCacheOutItem:
        if (cached_item := self._cache.get(mm_hash)) is not None:
            return None, cached_item.prompt_updates

        assert mm_item is not None, f"Expected a cached item for {mm_hash=}"

        self.cache_if_fits(
            self._cache, mm_hash, MultiModalProcessorCacheItemMetadata(*mm_item)
        )
        return mm_item

    @override
    def touch_sender_cache_item(self, mm_hash: str) -> None:
        self._cache.touch(mm_hash)

    @override
    def clear_cache(self) -> None:
        self._cache.clear()

    @override
    def make_stats(self, *, delta: bool = False) -> CacheInfo:
        return self._cache.stat(delta=delta)

    @override
    def invalidate(self, mm_hash: str) -> None:
        # Drop our stale shadow entry so the next request for this hash re-sends the
        # data and repopulates P1 (see MultiModalCacheMissError).
        self._cache.pop(mm_hash, None)

MultiModalCacheMissError

Bases: RuntimeError

Raised by the P1 receiver cache when items are requested with no data and are not cached.

P0 (frontend) keeps a metadata-only shadow of P1 (engine) and sends data=None on a shadow hit. The two caches are updated in different orders across processes, so they can drift -- leaving P0 referencing items P1 has evicted. Raising (instead of asserting) lets the engine return a retryable response and have P0 drop the stale entries (BaseMultiModalProcessorCache.invalidate) so the client resends the data. Carries every drifted mm_hash in the request so P0 can drop them all in one pass -- one retry then recovers the whole request, not one item per retry.

Source code in vllm/multimodal/cache/base.py
class MultiModalCacheMissError(RuntimeError):
    """Raised by the P1 receiver cache when items are requested with no data and
    are not cached.

    P0 (frontend) keeps a metadata-only shadow of P1 (engine) and sends
    ``data=None`` on a shadow hit. The two caches are updated in different orders
    across processes, so they can drift -- leaving P0 referencing items P1 has
    evicted. Raising (instead of asserting) lets the engine return a retryable
    response and have P0 drop the stale entries
    (``BaseMultiModalProcessorCache.invalidate``) so the client resends the data.
    Carries every drifted ``mm_hash`` in the request so P0 can drop them all in one
    pass -- one retry then recovers the whole request, not one item per retry.
    """

    def __init__(self, mm_hashes: list[str]) -> None:
        super().__init__(
            f"Multi-modal items {mm_hashes} are not in the receiver (P1) cache and "
            "no data was provided to recompute them (P0/P1 cache drift); the request "
            "should be retried with the multi-modal data attached."
        )
        self.mm_hashes = mm_hashes

MultiModalProcessorOnlyCache

Bases: BaseMultiModalProcessorCache

The cache which is used on P0 when IPC caching is disabled.

How to update each item:

  • If the item is in the cache, replace the input with the cached item.
  • If the item is not in the cache, store that item (which includes tensor data and metadata) into the cache, and return the input.
Source code in vllm/multimodal/cache/base.py
class MultiModalProcessorOnlyCache(BaseMultiModalProcessorCache):
    """The cache which is used on P0 when IPC caching is disabled.

    How to update each item:

    - If the item is in the cache, replace the input with the cached item.
    - If the item is not in the cache, store that item (which includes
      tensor data and metadata) into the cache, and return the input.
    """

    def __init__(self, model_config: ModelConfig) -> None:
        super().__init__()

        mm_config = model_config.get_multimodal_config()

        self._cache = MultiModalCache.get_lru_cache(
            mm_config.mm_processor_cache_gb,
            MultiModalProcessorCacheItem,
        )

    @override
    def is_cached_item(self, mm_hash: str) -> bool:
        return mm_hash in self._cache

    @override
    def get_and_update_item(
        self,
        mm_item: MultiModalProcessorCacheInItem,
        mm_hash: str,
    ) -> MultiModalProcessorCacheOutItem:
        if (cached_item := self._cache.get(mm_hash)) is not None:
            return cached_item.item, cached_item.prompt_updates

        assert mm_item is not None, f"Expected a cached item for {mm_hash=}"

        self.cache_if_fits(self._cache, mm_hash, MultiModalProcessorCacheItem(*mm_item))
        return mm_item

    @override
    def touch_sender_cache_item(self, mm_hash: str) -> None:
        self._cache.touch(mm_hash)

    @override
    def clear_cache(self) -> None:
        self._cache.clear()

    @override
    def make_stats(self, *, delta: bool = False) -> CacheInfo:
        return self._cache.stat(delta=delta)

ShmObjectStoreReceiverCache

Bases: BaseMultiModalReceiverCache

The cache which is used on P1 Worker Process when SHM caching is enabled.

How to update each item:

  • If the item has an address, replace the input with the cached item.
  • If not, return the input.

Methods:

Source code in vllm/multimodal/cache/shm.py
class ShmObjectStoreReceiverCache(BaseMultiModalReceiverCache):
    """The cache which is used on P1 Worker Process when SHM caching is enabled.

    How to update each item:

    - If the item has an address, replace the input with the cached item.
    - If not, return the input.
    """

    def __init__(
        self,
        vllm_config: VllmConfig,
        shared_worker_lock: LockType,
    ) -> None:
        super().__init__()

        self.world_size = vllm_config.parallel_config.world_size
        mm_config = vllm_config.model_config.get_multimodal_config()

        ring_buffer = SingleWriterShmRingBuffer(
            data_buffer_size=int(mm_config.mm_processor_cache_gb * GiB_bytes),
            name=envs.VLLM_OBJECT_STORAGE_SHM_BUFFER_NAME,
            create=False,  # Server is a reader
        )
        self._shm_cache = SingleWriterShmObjectStorage(
            max_object_size=mm_config.mm_shm_cache_max_object_size_mb * MiB_bytes,
            n_readers=self.world_size,
            ring_buffer=ring_buffer,
            serde_class=MsgpackSerde,
            reader_lock=shared_worker_lock,
        )

    @override
    def get_and_update_features(
        self, mm_features: list["MultiModalFeatureSpec"]
    ) -> list["MultiModalFeatureSpec"]:
        # strip_covered_mm_data preserves address items, so None here represents
        # a stripped uncached payload and has no SHM reference to acknowledge.
        features_with_data = [
            feature for feature in mm_features if feature.data is not None
        ]
        super().get_and_update_features(features_with_data)
        return mm_features

    @override
    def get_and_update_item(
        self,
        mm_item: MultiModalKwargsItem | None,
        mm_hash: str,
    ) -> MultiModalKwargsItem:
        assert mm_item is not None, f"Expected an address item for {mm_hash=}"
        if (handle := _get_shm_handle(mm_item)) is not None:
            address, monotonic_id, signature = handle
            return self._shm_cache.get(address, monotonic_id, signature, mm_hash)

        return mm_item

    @override
    def touch_receiver_cache_item(
        self,
        mm_hash: str,
        mm_item: MultiModalKwargsItem | None = None,
    ) -> None:
        """Validate the item's handle in shared memory cache."""
        assert mm_item is not None
        if (handle := _get_shm_handle(mm_item)) is not None:
            address, monotonic_id, signature = handle
            self._shm_cache.touch(
                mm_hash,
                address=address,
                monotonic_id=monotonic_id,
                signature=signature,
            )

    @override
    def clear_cache(self) -> None:
        self._shm_cache.clear()

touch_receiver_cache_item(mm_hash, mm_item=None)

Validate the item's handle in shared memory cache.

Source code in vllm/multimodal/cache/shm.py
@override
def touch_receiver_cache_item(
    self,
    mm_hash: str,
    mm_item: MultiModalKwargsItem | None = None,
) -> None:
    """Validate the item's handle in shared memory cache."""
    assert mm_item is not None
    if (handle := _get_shm_handle(mm_item)) is not None:
        address, monotonic_id, signature = handle
        self._shm_cache.touch(
            mm_hash,
            address=address,
            monotonic_id=monotonic_id,
            signature=signature,
        )

ShmObjectStoreSenderCache

Bases: BaseMultiModalProcessorCache

The cache which is used on P0 when SHM caching is enabled.

How to update each item:

  • If the item is already in the cache, clear the input to avoid unnecessary IPC.

  • If the item is not in the cache, store the data in shared memory.

Methods:

Source code in vllm/multimodal/cache/shm.py
class ShmObjectStoreSenderCache(BaseMultiModalProcessorCache):
    """The cache which is used on P0 when SHM caching is enabled.

    How to update each item:

    - If the item is already in the cache, clear the input to avoid
      unnecessary IPC.

    - If the item is not in the cache, store the data in shared memory.
    """

    def __init__(self, vllm_config: VllmConfig) -> None:
        super().__init__()

        self.world_size = vllm_config.parallel_config.world_size
        mm_config = vllm_config.model_config.get_multimodal_config()

        ring_buffer = SingleWriterShmRingBuffer(
            data_buffer_size=int(mm_config.mm_processor_cache_gb * GiB_bytes),
            name=envs.VLLM_OBJECT_STORAGE_SHM_BUFFER_NAME,
            create=True,  # sender is the writer
        )
        self._shm_cache = SingleWriterShmObjectStorage(
            max_object_size=mm_config.mm_shm_cache_max_object_size_mb * MiB_bytes,
            n_readers=self.world_size,
            ring_buffer=ring_buffer,
            serde_class=MsgpackSerde,
        )
        # cache prompt_updates for P0 only
        self._p0_cache: dict[str, Sequence[ResolvedPromptUpdate]] = {}

        self._hits = 0
        self._total = 0
        self._last_info = CacheInfo(hits=0, total=0)

    def _stat(self, *, delta: bool = False) -> CacheInfo:
        info = CacheInfo(hits=self._hits, total=self._total)

        if delta:
            info_delta = info - self._last_info
            self._last_info = info
            info = info_delta

        return info

    @override
    def is_cached_item(self, mm_hash: str) -> bool:
        return self._shm_cache.is_cached(mm_hash)

    @override
    def get_and_update_item(
        self,
        mm_item: MultiModalProcessorCacheInItem,
        mm_hash: str,
    ) -> MultiModalProcessorCacheOutItem:
        if self._shm_cache.is_cached(mm_hash):
            self._hits += 1
            self._total += 1

            address, monotonic_id = self._shm_cache.get_cached(mm_hash)
            signature = self._shm_cache.get_signature(mm_hash)
            prompt_updates = self._p0_cache[mm_hash]
            return (
                self.address_as_item(address, monotonic_id, signature),
                prompt_updates,
            )

        assert mm_item is not None, f"Expected a cached item for {mm_hash=}"
        item, prompt_updates = mm_item

        self._total += 1

        try:
            address, monotonic_id = self._shm_cache.put(mm_hash, item)
            signature = self._shm_cache.get_signature(mm_hash)
            # Try to remove dangling items if p0 cache is too large.
            if len(self._p0_cache) >= 2 * len(self._shm_cache.key_index):
                self.remove_dangling_items()

            self._p0_cache[mm_hash] = prompt_updates
            return (
                self.address_as_item(address, monotonic_id, signature),
                prompt_updates,
            )
        except ValueError as e:
            # `put` raises ValueError either for an oversize item or for a
            # duplicate key (concurrent insert); the latter is benign so we
            # only warn on the oversize case. Subsequent UUID-only requests
            # for an oversize item will fail with a cache miss.
            if "already exists" not in str(e):
                logger.warning_once(
                    "mm_input %s too large to cache; "
                    "raise --mm-shm-cache-max-object-size-mb. (%s)",
                    mm_hash,
                    str(e),
                )
            return mm_item
        except MemoryError as e:
            # Cache full and protected items prevent eviction.
            logger.debug(
                "mm_input %s not cached; shm cache full, "
                "consider raising --mm-processor-cache-gb. (%s)",
                mm_hash,
                str(e),
            )
            return mm_item

    @override
    def touch_sender_cache_item(self, mm_hash: str) -> None:
        """Touch the item in shared memory cache to prevent eviction."""
        self._shm_cache.touch(mm_hash)

    @override
    def release_sender_touches(self) -> None:
        self._shm_cache.release_touches()

    @override
    def validate_input_item(
        self,
        mm_item: MultiModalKwargsItem,
        mm_hash: str,
    ) -> None:
        if (handle := _get_shm_handle(mm_item)) is not None:
            address, monotonic_id, signature = handle
            self._shm_cache.verify_signature(
                mm_hash,
                address,
                monotonic_id,
                signature,
            )

    @override
    def clear_cache(self) -> None:
        self._shm_cache.clear()
        self._p0_cache.clear()

        self._hits = 0
        self._total = 0
        self._last_info = CacheInfo(hits=0, total=0)

    @override
    def make_stats(self, *, delta: bool = False) -> CacheInfo:
        return self._stat(delta=delta)

    @override
    def close(self) -> None:
        self._shm_cache.close()

    def remove_dangling_items(self) -> None:
        """Remove items that are no longer in the shared memory cache."""
        cached_hashes = self._shm_cache.key_index.keys()
        dangling_hashes = set(self._p0_cache.keys()) - cached_hashes
        for mm_hash in dangling_hashes:
            del self._p0_cache[mm_hash]

    def address_as_item(
        self,
        address: int,
        monotonic_id: int,
        signature: list[int],
    ) -> MultiModalKwargsItem:
        addr_elem = MultiModalFieldElem(
            data=address,
            field=MultiModalBatchedField(),
        )
        id_elem = MultiModalFieldElem(
            data=monotonic_id,
            field=MultiModalBatchedField(),
        )

        signature_elem = MultiModalFieldElem(
            data=signature,
            field=MultiModalBatchedField(),
        )

        return MultiModalKwargsItem(
            {
                "address": addr_elem,
                "monotonic_id": id_elem,
                "signature": signature_elem,
            }
        )

remove_dangling_items()

Remove items that are no longer in the shared memory cache.

Source code in vllm/multimodal/cache/shm.py
def remove_dangling_items(self) -> None:
    """Remove items that are no longer in the shared memory cache."""
    cached_hashes = self._shm_cache.key_index.keys()
    dangling_hashes = set(self._p0_cache.keys()) - cached_hashes
    for mm_hash in dangling_hashes:
        del self._p0_cache[mm_hash]

touch_sender_cache_item(mm_hash)

Touch the item in shared memory cache to prevent eviction.

Source code in vllm/multimodal/cache/shm.py
@override
def touch_sender_cache_item(self, mm_hash: str) -> None:
    """Touch the item in shared memory cache to prevent eviction."""
    self._shm_cache.touch(mm_hash)

engine_receiver_cache_from_config(vllm_config)

Return a BaseMultiModalReceiverCache for the engine process.

Source code in vllm/multimodal/cache/factories.py
def engine_receiver_cache_from_config(
    vllm_config: VllmConfig,
) -> BaseMultiModalReceiverCache | None:
    """Return a `BaseMultiModalReceiverCache` for the engine process."""
    cache_type = _get_cache_type(vllm_config)
    if cache_type in (None, "processor_only", "shm"):
        return None
    elif cache_type == "lru":
        return LruKeyReplicatedReceiverCache(vllm_config.model_config)
    else:
        raise ValueError(f"Unknown cache type: {cache_type!r}")

processor_cache_from_config(vllm_config)

Return a BaseMultiModalProcessorCache, if enabled.

Source code in vllm/multimodal/cache/factories.py
def processor_cache_from_config(
    vllm_config: VllmConfig,
) -> BaseMultiModalProcessorCache | None:
    """Return a `BaseMultiModalProcessorCache`, if enabled."""
    cache_type = _get_cache_type(vllm_config)
    if cache_type is None:
        return None
    elif cache_type == "processor_only":
        return MultiModalProcessorOnlyCache(vllm_config.model_config)
    elif cache_type == "lru":
        return LruKeyReplicatedSenderCache(vllm_config.model_config)
    elif cache_type == "shm":
        return ShmObjectStoreSenderCache(vllm_config)
    else:
        raise ValueError(f"Unknown cache type: {cache_type!r}")

processor_only_cache_from_config(vllm_config)

Return a MultiModalProcessorOnlyCache, if enabled.

Source code in vllm/multimodal/cache/factories.py
def processor_only_cache_from_config(
    vllm_config: VllmConfig,
) -> MultiModalProcessorOnlyCache | None:
    """Return a `MultiModalProcessorOnlyCache`, if enabled."""
    cache_type = _get_cache_type(vllm_config)
    if cache_type is None:
        return None

    return MultiModalProcessorOnlyCache(vllm_config.model_config)

worker_receiver_cache_from_config(vllm_config, shared_worker_lock)

Return a BaseMultiModalReceiverCache for the worker process.

Source code in vllm/multimodal/cache/factories.py
def worker_receiver_cache_from_config(
    vllm_config: VllmConfig,
    shared_worker_lock: LockType | None,
) -> BaseMultiModalReceiverCache | None:
    """Return a `BaseMultiModalReceiverCache` for the worker process."""
    cache_type = _get_cache_type(vllm_config)
    if cache_type in (None, "processor_only", "lru"):
        return None
    elif cache_type == "shm":
        if shared_worker_lock is None:
            raise ValueError(
                "Missing `shared_worker_lock` argument from executor. "
                "This argument is needed for mm_processor_cache_type='shm'."
            )

        return ShmObjectStoreReceiverCache(vllm_config, shared_worker_lock)
    else:
        raise ValueError(f"Unknown cache type: {cache_type!r}")