Skip to content

vllm.models.deepseek_v4.common.mm_preprocess

Multimodal preprocessing for the DeepSeek-V4 vision variants (DeepSeek-V4-Flash-Vision-Exp).

The image transform and sentinel-block construction are ported from the official repository's image_processor.py so that token counts bit-match the reference. Each <|deepseek_image|> placeholder in the prompt expands to a variable-length block of sentinel tokens; only positions with type == IMAGE receive vision embeddings, the other sentinels are looked up from learned embedding vectors in the model.

Unlike the reference (which uses out-of-vocab ids vocab_size + type), the sentinel block borrows five consecutive reserved tokenizer tokens (<|place_holder_mm_span_0431|> .. _0435|>): they are special tokens the tokenizer never emits from plain text, so the ids stay in-vocabulary and work with stock token validation, logprobs and detokenization. The ids are pure markers — every sentinel position's embedding is overwritten with vision/sentinel vectors before the decoder layers see it, exactly like the reference's out-of-vocab scheme.

Classes:

Functions:

DeepseekV4VLImageProcessor

Per-image transform (the PIL-input equivalent of the reference load_image).

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
class DeepseekV4VLImageProcessor:
    """Per-image transform (the PIL-input equivalent of the reference
    ``load_image``)."""

    def __init__(self, config: DeepseekV4Config) -> None:
        super().__init__()
        self.patch_size = config.vision_patch_size
        self.downsample_ratio = config.vision_downsample_ratio
        self.max_n_token = config.vision_max_n_token
        self.min_pixels = config.vision_min_pixels
        self.max_wh_ratio = config.vision_max_wh_ratio

    def __call__(self, image: Image.Image):
        return load_image(
            image,
            patch_size=self.patch_size,
            downsample_ratio=self.downsample_ratio,
            max_n_token=self.max_n_token,
            min_pixels=self.min_pixels,
            max_wh_ratio=self.max_wh_ratio,
        )

DeepseekV4VLProcessor

Minimal stand-in for the HF processor of DeepSeek-V4 vision models.

The official repository ships image preprocessing as plain functions in image_processor.py (no auto_map processor), so this class wraps their ports directly and the model loads without --trust-remote-code.

__call__ returns a BatchFeature with one entry per image (flattened across images):

  • patches: (sum(n_vit_h * n_vit_w), 3, p, p) bf16 ViT patches.
  • vit_grid: (num_images, 2) int64 [n_vit_h, n_vit_w].
  • llm_grid: (num_images, 2) int64 [n_llm_h, n_llm_w].
  • perm: concatenated per-image (n_llm_h * n_llm_w,) int64 index selecting aligner outputs into the final N-layout order.
  • types: concatenated per-image pad-free sentinel block types; block ids = IMAGE_SENTINEL_BASE_ID + types.
Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
class DeepseekV4VLProcessor:
    """Minimal stand-in for the HF processor of DeepSeek-V4 vision models.

    The official repository ships image preprocessing as plain functions in
    ``image_processor.py`` (no ``auto_map`` processor), so this class wraps
    their ports directly and the model loads without ``--trust-remote-code``.

    ``__call__`` returns a ``BatchFeature`` with one entry per image
    (flattened across images):

    - ``patches``: ``(sum(n_vit_h * n_vit_w), 3, p, p)`` bf16 ViT patches.
    - ``vit_grid``: ``(num_images, 2)`` int64 ``[n_vit_h, n_vit_w]``.
    - ``llm_grid``: ``(num_images, 2)`` int64 ``[n_llm_h, n_llm_w]``.
    - ``perm``: concatenated per-image ``(n_llm_h * n_llm_w,)`` int64 index
      selecting aligner outputs into the final N-layout order.
    - ``types``: concatenated per-image pad-free sentinel block types;
      ``block ids = IMAGE_SENTINEL_BASE_ID + types``.
    """

    def __init__(self, config: DeepseekV4Config) -> None:
        super().__init__()
        self.config = config
        self.image_processor = DeepseekV4VLImageProcessor(config)

    def __call__(
        self,
        text: str | None = None,
        images: Sequence[Image.Image] | None = None,
        return_tensors: str | None = None,
        **kwargs: Any,
    ) -> BatchFeature:
        patches_list = []
        vit_grid = []
        llm_grid = []
        perm_list = []
        types_list = []
        for image in images or []:
            patches, n_vit_h, n_vit_w, n_llm_h, n_llm_w = self.image_processor(image)
            types, perm = build_image_block_pad_free(n_llm_h, n_llm_w)
            patches_list.append(patches)
            vit_grid.append((n_vit_h, n_vit_w))
            llm_grid.append((n_llm_h, n_llm_w))
            perm_list.append(perm)
            types_list.append(types)

        if not patches_list:
            return BatchFeature({})

        return BatchFeature(
            {
                "patches": torch.cat(patches_list),
                "vit_grid": torch.tensor(vit_grid, dtype=torch.int64),
                "llm_grid": torch.tensor(llm_grid, dtype=torch.int64),
                "perm": torch.cat(perm_list),
                "types": torch.cat(types_list),
            }
        )

build_image_block(n_llm_h, n_llm_w, start_pos)

Builds the N-layout token types (final order) and the aligner-row order for IMAGE slots.

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
def build_image_block(n_llm_h: int, n_llm_w: int, start_pos: int):
    """Builds the N-layout token types (final order) and the aligner-row order
    for IMAGE slots."""
    compress_pad = COMPRESS_PAD_TO - 1 - start_pos % COMPRESS_PAD_TO
    pad_h = n_llm_h % 2
    rows = n_llm_h + pad_h
    row_len = n_llm_w + 1
    pad_last = rows // 2 * row_len % 2 * 2
    types = torch.tensor(
        ([IMAGE] * n_llm_w + [IMAGE_NEW_LINE]) * n_llm_h
        + [IMAGE_PAD] * (row_len * pad_h),
        dtype=torch.int64,
    )
    order = torch.arange(rows * row_len).view(rows // 2, 2, row_len)
    order = order.transpose(1, 2).reshape(-1)
    image_idx = torch.full((rows * row_len,), -1, dtype=torch.int64)
    image_idx.view(rows, row_len)[:n_llm_h, :n_llm_w] = torch.arange(
        n_llm_h * n_llm_w
    ).view(n_llm_h, n_llm_w)
    perm = image_idx[order]
    perm = perm[perm >= 0]
    types = torch.cat(
        [
            torch.full((compress_pad,), IMAGE_PAD, dtype=torch.int64),
            torch.tensor([IMAGE_START]),
            types[order],
            torch.full((pad_last,), IMAGE_PAD, dtype=torch.int64),
            torch.tensor([IMAGE_END]),
        ]
    )
    return types, perm

build_image_block_pad_free(n_llm_h, n_llm_w)

build_image_block without the position-dependent compressor pad.

start_pos is chosen so that compress_pad == 0; the pad is instead prepended when the block is spliced into the final prompt, where its position is known.

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
def build_image_block_pad_free(n_llm_h: int, n_llm_w: int):
    """``build_image_block`` without the position-dependent compressor pad.

    ``start_pos`` is chosen so that ``compress_pad == 0``; the pad is instead
    prepended when the block is spliced into the final prompt, where its
    position is known.
    """
    return build_image_block(n_llm_h, n_llm_w, COMPRESS_PAD_TO - 1)

grid_tokens(best_height, best_width, patch_size, downsample_ratio)

Number of LLM tokens the aligner grid occupies (N-layout, including row/align padding).

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
def grid_tokens(best_height, best_width, patch_size, downsample_ratio):
    """Number of LLM tokens the aligner grid occupies (N-layout, including
    row/align padding)."""
    n_llm_h = math.ceil((best_height // patch_size) / downsample_ratio)
    n_llm_w = math.ceil((best_width // patch_size) / downsample_ratio)
    num_tokens = n_llm_h * (n_llm_w + 1) + 2
    if n_llm_h % 2 == 1:
        num_tokens += n_llm_w + 1
    num_tokens += (n_llm_h + 1) // 2 * (n_llm_w + 1) % 2 * 2
    return n_llm_h, n_llm_w, num_tokens

image_sentinel_mask(token_ids)

Boolean mask for image-block sentinel positions (in-vocab ids).

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
def image_sentinel_mask(token_ids: torch.Tensor) -> torch.Tensor:
    """Boolean mask for image-block sentinel positions (in-vocab ids)."""
    return (token_ids >= IMAGE_SENTINEL_BASE_ID) & (
        token_ids < IMAGE_SENTINEL_BASE_ID + len(IMAGE_SENTINEL_TOKEN_NAMES)
    )

load_image(image, *, patch_size, downsample_ratio, max_n_token, min_pixels, max_wh_ratio)

Transform one PIL image into ViT patches.

Same math as the reference load_image, except the image is already decoded (vLLM supplies PIL images instead of a record dict).

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
def load_image(
    image: Image.Image,
    *,
    patch_size: int,
    downsample_ratio: int,
    max_n_token: int,
    min_pixels: int,
    max_wh_ratio: float | None,
):
    """Transform one PIL image into ViT patches.

    Same math as the reference ``load_image``, except the image is already
    decoded (vLLM supplies PIL images instead of a record dict).
    """
    p = patch_size
    image = image.convert("RGB")
    width, height = image.size
    if max_wh_ratio is not None and width > height * max_wh_ratio:
        width = height * max_wh_ratio
    if 0 < width * height < min_pixels:
        ratio = (min_pixels / (width * height)) ** 0.5
        width = int(width * ratio)
        height = int(height * ratio)
    best_width = math.ceil(width / p) * p
    best_height = math.ceil(height / p) * p
    n_llm_h, n_llm_w, best_height, best_width = safe_resize(
        height, width, best_height, best_width, p, downsample_ratio, max_n_token
    )
    n_vit_h, n_vit_w = best_height // p, best_width // p
    if max_wh_ratio is not None and image.width >= max_wh_ratio * image.height:
        image = image.resize((best_width, best_height))
    else:
        image = ImageOps.pad(image, (best_width, best_height), color=(127, 127, 127))
    x = torch.from_numpy(np.asarray(image, dtype=np.float32)).permute(2, 0, 1) / 255
    x = ((x - 0.5) / 0.5).to(torch.bfloat16)
    patches = (
        x.reshape(3, n_vit_h, p, n_vit_w, p)
        .permute(1, 3, 0, 2, 4)
        .reshape(n_vit_h * n_vit_w, 3, p, p)
    )
    return patches, n_vit_h, n_vit_w, n_llm_h, n_llm_w

validate_image_sentinel_ids(tokenizer)

Check the borrowed sentinel ids against the tokenizer.

Source code in vllm/models/deepseek_v4/common/mm_preprocess.py
def validate_image_sentinel_ids(tokenizer) -> None:
    """Check the borrowed sentinel ids against the tokenizer."""
    for i, name in enumerate(IMAGE_SENTINEL_TOKEN_NAMES):
        token_id = tokenizer.convert_tokens_to_ids(name)
        if token_id != IMAGE_SENTINEL_BASE_ID + i:
            raise ValueError(
                f"Image sentinel token {name!r} has id {token_id}, expected "
                f"{IMAGE_SENTINEL_BASE_ID + i}; the DeepSeek-V4 vision "
                "sentinel block requires these consecutive reserved ids."
            )