Skip to content

vllm.parser.chat_parsing.content_parsers

This file contains the parsers used by chat response parsing. Each parser takes a chunk of captured text and parses it into a single key in the output message dictionary. Functions are generally boilerplate.

Functions:

_apply_transform(transform, scope)

Recursively walk a transform template, which is used to restructure parsed output into the actual shape we want. A dotted placeholder like {content.args} descends into keys of the looked-up value.

Source code in vllm/parser/chat_parsing/content_parsers.py
def _apply_transform(transform: Any, scope: dict) -> Any:
    """Recursively walk a transform template, which is used to restructure
    parsed output into the actual shape we want. A dotted placeholder like
    `{content.args}` descends into keys of the looked-up value."""
    if isinstance(transform, dict):
        return {k: _apply_transform(v, scope) for k, v in transform.items()}
    if isinstance(transform, list):
        return [_apply_transform(v, scope) for v in transform]
    if not isinstance(transform, str):
        return transform
    whole = _PLACEHOLDER.fullmatch(transform)
    if not whole:
        return transform
    path = whole.group(1)
    root, *keys = path.split(".")
    if root not in scope:
        raise KeyError(f"transform placeholder '{{{path}}}' is not defined. Available: {sorted(scope)}")
    value = scope[root]
    for key in keys:
        if not isinstance(value, dict):
            raise ValueError(f"transform placeholder '{{{path}}}' cannot index into {type(value).__name__} at '{key}'")
        if key not in value:
            raise ValueError(f"transform placeholder '{{{path}}}' is missing key '{key}'. Available: {sorted(value)}")
        value = value[key]
    return value

_json(text, args)

JSON parser with optional dialect knobs for LLM-emitted quirks.

args: - unquoted_keys (bool): quote bare-identifier keys before parsing. - string_delims ([[open, close], ...]): strings delimited by these custom markers are pre-extracted, then restored as standard JSON strings. - allow_non_json (bool): return stripped text if parsing fails.

Source code in vllm/parser/chat_parsing/content_parsers.py
def _json(text: str, args: dict) -> Any:
    """JSON parser with optional dialect knobs for LLM-emitted quirks.

    `args`:
      - `unquoted_keys` (bool): quote bare-identifier keys before parsing.
      - `string_delims` ([[open, close], ...]): strings delimited by these
        custom markers are pre-extracted, then restored as standard JSON strings.
      - `allow_non_json` (bool): return stripped text if parsing fails.
    """
    string_delims = args.get("string_delims", [])
    unquoted_keys = args.get("unquoted_keys", False)

    if string_delims and (_LAX_OPEN in text or _LAX_CLOSE in text):
        raise ValueError("json: input contains reserved sentinel characters (\\x01/\\x02); cannot parse safely.")

    working = text
    captured: list[str] = []
    for open_d, close_d in string_delims:
        pattern = re.escape(open_d) + r"(.*?)" + re.escape(close_d)

        def _capture(m: Any) -> str:
            captured.append(m.group(1))
            return f"{_LAX_OPEN}{len(captured) - 1}{_LAX_CLOSE}"

        working = re.sub(pattern, _capture, working, flags=re.DOTALL)

    if unquoted_keys:
        working = re.sub(r"(?<=[{,])(\w+):", r'"\1":', working)

    for i, s in enumerate(captured):
        working = working.replace(f"{_LAX_OPEN}{i}{_LAX_CLOSE}", json.dumps(s))

    try:
        return json.loads(working)
    except json.JSONDecodeError as e:
        if args.get("allow_non_json"):
            return _text(text, args)
        if working == text:
            raise ValueError(f"json parser could not parse region as JSON.\nContent: {text!r}\nError: {e}") from e
        raise ValueError(
            f"json: could not parse after dialect transforms.\n"
            f"Original: {text!r}\nTransformed: {working!r}\nError: {e}"
        ) from e

_kv_lines(text, args)

Parse line-delimited key<sep>value pairs into a dict.

Source code in vllm/parser/chat_parsing/content_parsers.py
def _kv_lines(text: str, args: dict) -> dict:
    """Parse line-delimited `key<sep>value` pairs into a dict."""
    line_sep = args.get("line_sep", "\n")
    kv_sep = args.get("kv_sep", ":")
    value_parser = args.get("value_parser")

    out: dict[str, Any] = {}
    for line in text.split(line_sep):
        line = _text(line, args)
        if not line or kv_sep not in line:
            continue
        k, v = line.split(kv_sep, 1)
        k, v = _text(k, args), _text(v, args)
        out[k] = _sub_parse(v, value_parser)
    return out

_xml_inline(text, args)

Parse shallow XML-ish tags into a dict. tag_pattern regex must have named groups key and value. Optional value_parser recurses; merge_duplicates collects duplicate keys into a list.

Source code in vllm/parser/chat_parsing/content_parsers.py
def _xml_inline(text: str, args: dict) -> dict:
    """Parse shallow XML-ish tags into a dict. `tag_pattern` regex must have named
    groups `key` and `value`. Optional `value_parser` recurses; `merge_duplicates`
    collects duplicate keys into a list."""
    tag_pattern = args.get("tag_pattern")
    if tag_pattern is None:
        raise ValueError("xml-inline: 'tag_pattern' content_arg is required")
    value_parser = args.get("value_parser")
    merge = args.get("merge_duplicates", False)

    out: dict[str, Any] = {}
    for m in re.finditer(tag_pattern, text, flags=re.DOTALL):
        groups = m.groupdict()
        key = groups.get("key")
        if key is None:
            raise ValueError(f"xml-inline: tag_pattern must have a named group 'key'. Pattern: {tag_pattern}")
        value = _sub_parse(groups.get("value", ""), value_parser)
        if key in out and merge:
            if not isinstance(out[key], list):
                out[key] = [out[key]]
            out[key].append(value)
        else:
            out[key] = value
    return out

process_field(body, field, captures)

Run body through the field's content parser, then optionally apply the transform template. When transform_each is set, the parsed content must be a list and the template is applied to each element (with the element's keys unpacked into the template scope, alongside any regex captures).

field is a spec.Field; typed via duck-typing to avoid a cyclic import.

Source code in vllm/parser/chat_parsing/content_parsers.py
def process_field(body: str, field, captures: dict) -> Any:
    """Run `body` through the field's content parser, then optionally apply the
    transform template. When `transform_each` is set, the parsed content must
    be a list and the template is applied to each element (with the element's
    keys unpacked into the template scope, alongside any regex captures).

    `field` is a `spec.Field`; typed via duck-typing to avoid a cyclic import."""
    value = parse_content(body, field.content, field.content_args)
    if field.transform is None:
        return value
    if field.transform_each:
        if not isinstance(value, list):
            raise ValueError(
                f"Field '{field.name}': transform_each requires the parsed content to be a list, "
                f"got {type(value).__name__}."
            )
        out = []
        for item in value:
            if not isinstance(item, dict):
                raise ValueError(
                    f"Field '{field.name}': transform_each requires each list element to be a dict, "
                    f"got {type(item).__name__}."
                )
            out.append(_apply_transform(field.transform, {**captures, **item}))
        return out
    return _apply_transform(field.transform, {**captures, "content": value})

validate_transform_strings(scope, transform)

Walk a transform template and reject any string that mixes a {name} placeholder with literal text. Only whole-string placeholders (e.g. "{content}") and plain literals are supported. Called from the template loader so authors get a clear error at load time, not at parse time.

Source code in vllm/parser/chat_parsing/content_parsers.py
def validate_transform_strings(scope: str, transform: Any) -> None:
    """Walk a transform template and reject any string that mixes a `{name}`
    placeholder with literal text. Only whole-string placeholders (e.g.
    `"{content}"`) and plain literals are supported. Called from the template
    loader so authors get a clear error at load time, not at parse time."""
    if isinstance(transform, dict):
        for v in transform.values():
            validate_transform_strings(scope, v)
        return
    if isinstance(transform, list):
        for v in transform:
            validate_transform_strings(scope, v)
        return
    if not isinstance(transform, str):
        return
    if _PLACEHOLDER.search(transform) and not _PLACEHOLDER.fullmatch(transform):
        raise ValueError(
            f"{scope}: transform string {transform!r} mixes a {{placeholder}} with literal text. "
            'Use either a whole-string placeholder (e.g. "{content}") or a plain literal; '
            "string interpolation is not supported."
        )