Skip to content

vllm.v1.structured_output.backend_outlines

Classes:

Functions:

OutlinesGrammar dataclass

Bases: StructuredOutputGrammar

Methods:

  • accept_tokens –

    Accepts a list of tokens and advances the FSM.

Source code in vllm/v1/structured_output/backend_outlines.py
@dataclass
class OutlinesGrammar(StructuredOutputGrammar):
    vocab_size: int
    eos_token_id: int
    guide: oc.Guide = field(hash=False)
    num_processed_tokens: int = field(
        default_factory=lambda: 0, repr=False, hash=False, init=False
    )

    # outlines_core signals done on DFA accept and never advances on EOS
    # (its mask allows only EOS in a final state, but advance(EOS) fails),
    # so accepting EOS in a final state is tracked here instead.
    _is_terminated: bool = field(default=False, init=False, repr=False, hash=False)

    def accept_tokens(self, request_id: str, tokens: list[int]) -> bool:
        """Accepts a list of tokens and advances the FSM.

        Returns True if all grammar-constrained tokens were accepted.
        Tokens after termination (EOS) are ignored. Returns False if the FSM
        failed to advance.
        """
        if self._is_terminated:
            return True
        eos_index = next(
            (i for i, t in enumerate(tokens) if t == self.eos_token_id), None
        )
        if eos_index is not None:
            tokens = tokens[:eos_index]
        # Advance can fail when the next state reached after advancing with
        # the current tokens is a dead state. This is because Guide.accepts_tokens()
        # only checks whether the current tokens can be accepted,
        # whereas guide.advance() additionally checks the next state
        # after all tokens are accepted.
        # We need to be aware that the FSM must be prepared without dead states.
        if tokens and not self.guide.accepts_tokens(tokens):
            return False
        for t in tokens:
            self.guide.advance(t)
        if eos_index is not None:
            if not self.guide.is_finished():
                if tokens:
                    self.guide.rollback_state(len(tokens))
                return False
            self._is_terminated = True
        self.num_processed_tokens += len(tokens) + self._is_terminated
        return True

    def rollback(self, num_tokens: int) -> None:
        if num_tokens <= 0:
            return
        self.num_processed_tokens -= num_tokens
        if self._is_terminated:
            self._is_terminated = False
            num_tokens -= 1
        if num_tokens:
            self.guide.rollback_state(num_tokens)

    def validate_tokens(self, tokens: list[int]) -> list[int]:
        if self._is_terminated:
            return []
        accepted: list[int] = []
        for tok in tokens:
            if tok == self.eos_token_id:
                if self._is_finished_after(accepted):
                    accepted.append(tok)
                break
            accepted.append(tok)
            if not self.guide.accepts_tokens(accepted):
                accepted.pop()
                break
        return accepted

    def _is_finished_after(self, tokens: list[int]) -> bool:
        if not tokens:
            return self.guide.is_finished()
        for t in tokens:
            self.guide.advance(t)
        finished = self.guide.is_finished()
        self.guide.rollback_state(len(tokens))
        return finished

    def fill_bitmask(self, bitmask: torch.Tensor, idx: int) -> None:
        mask = bitmask[idx]
        self.guide.write_mask_into(mask.data_ptr(), mask.numel(), mask.element_size())

    def is_terminated(self) -> bool:
        return self._is_terminated

    def reset(self):
        self.num_processed_tokens = 0
        self._is_terminated = False
        self.guide.reset()

accept_tokens(request_id, tokens)

Accepts a list of tokens and advances the FSM.

Returns True if all grammar-constrained tokens were accepted. Tokens after termination (EOS) are ignored. Returns False if the FSM failed to advance.

Source code in vllm/v1/structured_output/backend_outlines.py
def accept_tokens(self, request_id: str, tokens: list[int]) -> bool:
    """Accepts a list of tokens and advances the FSM.

    Returns True if all grammar-constrained tokens were accepted.
    Tokens after termination (EOS) are ignored. Returns False if the FSM
    failed to advance.
    """
    if self._is_terminated:
        return True
    eos_index = next(
        (i for i, t in enumerate(tokens) if t == self.eos_token_id), None
    )
    if eos_index is not None:
        tokens = tokens[:eos_index]
    # Advance can fail when the next state reached after advancing with
    # the current tokens is a dead state. This is because Guide.accepts_tokens()
    # only checks whether the current tokens can be accepted,
    # whereas guide.advance() additionally checks the next state
    # after all tokens are accepted.
    # We need to be aware that the FSM must be prepared without dead states.
    if tokens and not self.guide.accepts_tokens(tokens):
        return False
    for t in tokens:
        self.guide.advance(t)
    if eos_index is not None:
        if not self.guide.is_finished():
            if tokens:
                self.guide.rollback_state(len(tokens))
            return False
        self._is_terminated = True
    self.num_processed_tokens += len(tokens) + self._is_terminated
    return True

_check_unsupported(parsed)

Check for regex features unsupported by regex-automata.

Source code in vllm/v1/structured_output/backend_outlines.py
def _check_unsupported(parsed) -> None:
    """Check for regex features unsupported by regex-automata."""
    tokens = parsed.data if hasattr(parsed, "data") else parsed
    for ttype, tval in tokens:
        # backreference
        if ttype in (sre_parse.GROUPREF, sre_parse.GROUPREF_EXISTS):
            raise ValueError("Backreferences are unsupported.")

        # look-around assertion
        elif ttype in (sre_constants.ASSERT, sre_constants.ASSERT_NOT):
            raise ValueError("Look-Around assertion are unsupported.")

        # unicode word boundaries
        elif ttype == sre_parse.AT:
            if tval in (sre_constants.AT_BOUNDARY, sre_constants.AT_NON_BOUNDARY):
                raise ValueError("Unicode word boundaries are unsupported.")

        elif ttype == sre_parse.BRANCH:
            # tval is (None, branches)
            for branch in tval[1]:
                _check_unsupported(branch)

        # tval is (min, max, subpattern)
        elif ttype == sre_parse.MAX_REPEAT:
            _check_unsupported(tval[2])

_prefix_needs_context(parsed)

Return True if there's a look-around/anchor before any consumer.

Source code in vllm/v1/structured_output/backend_outlines.py
def _prefix_needs_context(parsed) -> bool:
    """Return True if there's a look-around/anchor before any consumer."""

    def subpattern_consumes(parsed) -> bool:
        """Return True if subpattern can consume at least one character."""
        tokens = parsed.data if hasattr(parsed, "data") else parsed
        for ttype, tval in tokens:
            # literal, character class, or dot always consumes
            if ttype in (sre_parse.LITERAL, sre_parse.IN, sre_parse.ANY):
                return True
            # quantified subpattern: check inner pattern
            elif ttype == sre_parse.MAX_REPEAT:
                _, mx, sub = tval
                if mx != 0 and subpattern_consumes(sub):
                    return True
            # alternation: if any branch consumes, the whole does
            elif ttype == sre_parse.BRANCH:
                _, branches = tval
                if any(subpattern_consumes(br) for br in branches):
                    return True
            # grouped subpattern: recurse into its contents
            elif ttype == sre_parse.SUBPATTERN and subpattern_consumes(tval[3]):
                return True
        # No consumers, return False
        return False

    tokens = parsed.data if hasattr(parsed, "data") else parsed
    for ttype, tval in tokens:
        # Direct anchors or look-around
        if ttype == sre_parse.AT or ttype in (
            sre_constants.ASSERT,
            sre_constants.ASSERT_NOT,
        ):
            return True

        # Nested subpattern: check
        if ttype == sre_parse.SUBPATTERN:
            # tval: (group, add_flags, del_flags, subpattern)
            if _prefix_needs_context(tval[3]):
                return True
            if subpattern_consumes(tval[3]):
                return False

        # if any branch has a prefix anchor => True,
        # else if at least one branch consumes => prefix ends => False
        elif ttype == sre_parse.BRANCH:
            saw_consumer = False
            for br in tval[1]:
                if _prefix_needs_context(br):
                    return True
                if subpattern_consumes(br):
                    saw_consumer = True
            if saw_consumer:
                return False

        # Immediate consumer tokens
        elif ttype in (sre_parse.LITERAL, sre_parse.IN, sre_parse.ANY):
            return False

        # if subpattern has anchor => True, if it can consume => stop
        elif ttype == sre_parse.MAX_REPEAT:
            if _prefix_needs_context(tval[2]):
                return True
            if subpattern_consumes(tval[2]):
                return False

    return False

validate_regex_is_buildable(pattern)

Validates that the input regex is not using unsupported features of the regex-automata crate (outlines_core regex engine) and has a universal start state. definition of universal start state used can be found at: https://docs.rs/regex-automata/latest/regex_automata/dfa/trait.Automaton.html#method.universal_start_state

Source code in vllm/v1/structured_output/backend_outlines.py
def validate_regex_is_buildable(pattern: str) -> None:
    """Validates that the input regex is not using unsupported features
    of the `regex-automata` crate (outlines_core regex engine) and has a
    universal start state.
    definition of universal start state used can be found at:
    https://docs.rs/regex-automata/latest/regex_automata/dfa/trait.Automaton.html#method.universal_start_state
    """
    try:
        parsed = sre_parse.parse(pattern)

    except sre_constants.error as e:
        raise VLLMValidationError(f"Error parsing regex: {e}") from e

    try:
        _check_unsupported(parsed)
    except ValueError as e:
        raise VLLMValidationError(
            f"Regex uses unsupported feature for structured outputs: {e}. "
            "Only basic matching constructs are supported—lookarounds, "
            "backreferences, and unicode boundaries are not."
        ) from e

    if _prefix_needs_context(parsed):
        raise VLLMValidationError(
            "Regex does not have a anchored universal start state"
            "This means that the Regex uses anchors (^) or look-arounds "
            "in a way which requires context before any token is matched."
            "structured outputs needs regexes that can match without needing "
            "that context. Try rewriting the pattern without using these "
            f"constructs. Pattern:\n{pattern}"
        )