Skip to content

vllm.parser.abstract_parser

Classes:

  • DelegatingParser –

    A Parser implementation that delegates to separate ReasoningParser and

  • Parser –

    Parse model output into reasoning, content and tool calls.

  • StreamState –

    Mutable state for Parser.parse_delta(). One per stream.

Functions:

DelegatingParser

Bases: Parser

A Parser implementation that delegates to separate ReasoningParser and ToolParser instances.

This is the recommended base class for creating model-specific parsers that combine existing reasoning and tool parser implementations. Subclasses should set self._reasoning_parser and self._tool_parser in their __init__ method.

If either parser is None, the corresponding methods will return default values (no reasoning extraction, no tool calls).

Methods:

Source code in vllm/parser/abstract_parser.py
 330
 331
 332
 333
 334
 335
 336
 337
 338
 339
 340
 341
 342
 343
 344
 345
 346
 347
 348
 349
 350
 351
 352
 353
 354
 355
 356
 357
 358
 359
 360
 361
 362
 363
 364
 365
 366
 367
 368
 369
 370
 371
 372
 373
 374
 375
 376
 377
 378
 379
 380
 381
 382
 383
 384
 385
 386
 387
 388
 389
 390
 391
 392
 393
 394
 395
 396
 397
 398
 399
 400
 401
 402
 403
 404
 405
 406
 407
 408
 409
 410
 411
 412
 413
 414
 415
 416
 417
 418
 419
 420
 421
 422
 423
 424
 425
 426
 427
 428
 429
 430
 431
 432
 433
 434
 435
 436
 437
 438
 439
 440
 441
 442
 443
 444
 445
 446
 447
 448
 449
 450
 451
 452
 453
 454
 455
 456
 457
 458
 459
 460
 461
 462
 463
 464
 465
 466
 467
 468
 469
 470
 471
 472
 473
 474
 475
 476
 477
 478
 479
 480
 481
 482
 483
 484
 485
 486
 487
 488
 489
 490
 491
 492
 493
 494
 495
 496
 497
 498
 499
 500
 501
 502
 503
 504
 505
 506
 507
 508
 509
 510
 511
 512
 513
 514
 515
 516
 517
 518
 519
 520
 521
 522
 523
 524
 525
 526
 527
 528
 529
 530
 531
 532
 533
 534
 535
 536
 537
 538
 539
 540
 541
 542
 543
 544
 545
 546
 547
 548
 549
 550
 551
 552
 553
 554
 555
 556
 557
 558
 559
 560
 561
 562
 563
 564
 565
 566
 567
 568
 569
 570
 571
 572
 573
 574
 575
 576
 577
 578
 579
 580
 581
 582
 583
 584
 585
 586
 587
 588
 589
 590
 591
 592
 593
 594
 595
 596
 597
 598
 599
 600
 601
 602
 603
 604
 605
 606
 607
 608
 609
 610
 611
 612
 613
 614
 615
 616
 617
 618
 619
 620
 621
 622
 623
 624
 625
 626
 627
 628
 629
 630
 631
 632
 633
 634
 635
 636
 637
 638
 639
 640
 641
 642
 643
 644
 645
 646
 647
 648
 649
 650
 651
 652
 653
 654
 655
 656
 657
 658
 659
 660
 661
 662
 663
 664
 665
 666
 667
 668
 669
 670
 671
 672
 673
 674
 675
 676
 677
 678
 679
 680
 681
 682
 683
 684
 685
 686
 687
 688
 689
 690
 691
 692
 693
 694
 695
 696
 697
 698
 699
 700
 701
 702
 703
 704
 705
 706
 707
 708
 709
 710
 711
 712
 713
 714
 715
 716
 717
 718
 719
 720
 721
 722
 723
 724
 725
 726
 727
 728
 729
 730
 731
 732
 733
 734
 735
 736
 737
 738
 739
 740
 741
 742
 743
 744
 745
 746
 747
 748
 749
 750
 751
 752
 753
 754
 755
 756
 757
 758
 759
 760
 761
 762
 763
 764
 765
 766
 767
 768
 769
 770
 771
 772
 773
 774
 775
 776
 777
 778
 779
 780
 781
 782
 783
 784
 785
 786
 787
 788
 789
 790
 791
 792
 793
 794
 795
 796
 797
 798
 799
 800
 801
 802
 803
 804
 805
 806
 807
 808
 809
 810
 811
 812
 813
 814
 815
 816
 817
 818
 819
 820
 821
 822
 823
 824
 825
 826
 827
 828
 829
 830
 831
 832
 833
 834
 835
 836
 837
 838
 839
 840
 841
 842
 843
 844
 845
 846
 847
 848
 849
 850
 851
 852
 853
 854
 855
 856
 857
 858
 859
 860
 861
 862
 863
 864
 865
 866
 867
 868
 869
 870
 871
 872
 873
 874
 875
 876
 877
 878
 879
 880
 881
 882
 883
 884
 885
 886
 887
 888
 889
 890
 891
 892
 893
 894
 895
 896
 897
 898
 899
 900
 901
 902
 903
 904
 905
 906
 907
 908
 909
 910
 911
 912
 913
 914
 915
 916
 917
 918
 919
 920
 921
 922
 923
 924
 925
 926
 927
 928
 929
 930
 931
 932
 933
 934
 935
 936
 937
 938
 939
 940
 941
 942
 943
 944
 945
 946
 947
 948
 949
 950
 951
 952
 953
 954
 955
 956
 957
 958
 959
 960
 961
 962
 963
 964
 965
 966
 967
 968
 969
 970
 971
 972
 973
 974
 975
 976
 977
 978
 979
 980
 981
 982
 983
 984
 985
 986
 987
 988
 989
 990
 991
 992
 993
 994
 995
 996
 997
 998
 999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
class DelegatingParser(Parser):
    """A Parser implementation that delegates to separate ReasoningParser and
    ToolParser instances.

    This is the recommended base class for creating model-specific parsers
    that combine existing reasoning and tool parser implementations.
    Subclasses should set `self._reasoning_parser` and `self._tool_parser`
    in their `__init__` method.

    If either parser is None, the corresponding methods will return default
    values (no reasoning extraction, no tool calls).
    """

    def extract_reasoning(
        self,
        model_output: str,
        request: ChatCompletionRequest | ResponsesRequest,
    ) -> tuple[str | None, str | None]:
        if self._reasoning_parser is None:
            return None, model_output
        return self._reasoning_parser.extract_reasoning(model_output, request)

    def _get_function_name(
        self, request: ChatCompletionRequest | ResponsesRequest
    ) -> str:
        if request.tool_choice and isinstance(request.tool_choice, ToolChoiceFunction):
            return request.tool_choice.name
        if request.tool_choice and isinstance(
            request.tool_choice, ChatCompletionNamedToolChoiceParam
        ):
            return request.tool_choice.function.name
        raise ValueError("Invalid tool_choice for function name extraction.")

    def _make_tool_call_id(self, function_name: str) -> str | None:
        state = self._stream_state
        if state.tool_call_id_type != "kimi_k2":
            return None
        tool_call_id = make_tool_call_id(
            id_type=state.tool_call_id_type,
            func_name=function_name,
            idx=state.history_tool_call_cnt,
        )
        state.history_tool_call_cnt += 1
        return tool_call_id

    def _extract_tool_calls(
        self,
        content: str | None,
        request: ChatCompletionRequest | ResponsesRequest,
        enable_auto_tools: bool = False,
    ) -> tuple[list[FunctionCall] | None, str | None]:
        tool_parser = self._tool_parser
        if tool_parser is None:
            return [], content

        if request.tool_choice == "none":
            if self._engine_based:
                result = self.extract_tool_calls(content or "", request=request)
                return [], result.content
            return [], content

        supports_required_and_named = tool_parser.supports_required_and_named
        is_named_tool_choice = request.tool_choice and isinstance(
            request.tool_choice,
            (ToolChoiceFunction, ChatCompletionNamedToolChoiceParam),
        )
        is_required_tool_choice = request.tool_choice == "required"
        is_auto_tool_choice = enable_auto_tools and (
            request.tool_choice == "auto"
            or request.tool_choice is None
            or (
                not supports_required_and_named
                and (is_named_tool_choice or is_required_tool_choice)
            )
        )

        tool_calls = list[FunctionCall]()
        if is_named_tool_choice and supports_required_and_named:
            if content is None or (isinstance(content, str) and not content.strip()):
                return [], None
            function_name = self._get_function_name(request)
            tool_calls.append(
                FunctionCall(
                    id=self._make_tool_call_id(function_name),
                    name=function_name,
                    arguments=content,
                )
            )
            content = None
        elif is_required_tool_choice and supports_required_and_named:
            # "required" with standard JSON-based parsing
            parsed_calls = []
            with contextlib.suppress(ValidationError):
                content = content or ""
                parsed_calls = TypeAdapter(list[FunctionDefinition]).validate_json(
                    content
                )
            for tc in parsed_calls:
                tool_calls.append(
                    FunctionCall(
                        id=self._make_tool_call_id(tc.name),
                        name=tc.name,
                        arguments=json.dumps(tc.parameters, ensure_ascii=False),
                    )
                )
            content = None
        elif is_auto_tool_choice:
            # Automatic Tool Call Parsing (also used as fallback for
            # required/named when supports_required_and_named=False)
            tool_call_info = self.extract_tool_calls(
                content if content is not None else "",
                request=request,
            )
            if tool_call_info is not None and tool_call_info.tools_called:
                tool_calls.extend(
                    FunctionCall(
                        id=tc.id,
                        name=tc.function.name,
                        arguments=tc.function.arguments,
                    )
                    for tc in tool_call_info.tool_calls
                )
                content = tool_call_info.content
                if content and content.strip() == "":
                    content = None
            else:
                # No tool calls.
                # For required/named tool choice (when falling back to auto
                # parsing), if content is empty or whitespace-only, return
                # empty list with None content.
                if (is_required_tool_choice or is_named_tool_choice) and (
                    content is None
                    or (isinstance(content, str) and not content.strip())
                ):
                    return [], None
                # No complete tool calls: for engine-based parsers, return
                # the tool parser's content, which drops incomplete
                # tool-call markup (e.g. a <tool_call> opener truncated by
                # max_tokens or a stop string), so the non-streaming path
                # matches streaming. Legacy parsers keep their existing
                # behavior of returning the raw content.
                if self._engine_based and tool_call_info is not None:
                    return None, tool_call_info.content or None
                return None, content

        return tool_calls, content

    def adjust_request(
        self, request: ChatCompletionRequest | ResponsesRequest
    ) -> ChatCompletionRequest | ResponsesRequest:
        if self._reasoning_parser is not None:
            request = self._reasoning_parser.adjust_request(request)
        if self._tool_parser is not None:
            request = self._apply_structural_tag(request)
            request = self._tool_parser.adjust_request(request)
        return request

    def _apply_structural_tag(
        self, request: ChatCompletionRequest | ResponsesRequest
    ) -> ChatCompletionRequest | ResponsesRequest:
        tool_parser = self._tool_parser
        if tool_parser is None or not request.tools:
            return request

        need_tool_calling = (
            request.tool_choice == "auto"
            or request.tool_choice == "required"
            or isinstance(
                request.tool_choice,
                (ChatCompletionNamedToolChoiceParam, ToolChoiceFunction),
            )
        )
        if not need_tool_calling:
            return request

        structured_outputs = request.extract_structured_outputs()
        is_auto = request.tool_choice == "auto"
        single_call = request.parallel_tool_calls is False
        strict_level = self.tool_strict_level
        if (
            strict_level == ToolStrictLevel.AUTO
            and tool_parser.default_tool_strict_level is not None
        ):
            strict_level = tool_parser.default_tool_strict_level
        if single_call and not (is_auto and structured_outputs is not None):
            # Limiting the call count needs the call envelope in the grammar.
            # Auto with structured outputs keeps its format-only grammar, so the
            # flag never makes calls possible that are impossible without it.
            strict_level = max(strict_level, ToolStrictLevel.FUNCTION)

        resolved_tools = None
        if tool_parser.structural_tag_model is not None:
            resolved_tools = resolve_tool_strictness(
                request.tools,
                request.tool_choice,
                strict_level,
            )

        output_format = None
        if resolved_tools and is_auto and structured_outputs:
            if not _xgrammar_supports(structured_outputs):
                logger.warning_once(
                    "Tool calls are not constrained for tool_choice=auto with "
                    "structured outputs because xgrammar does not support the "
                    "structured output constraint; the structured output "
                    "constraint applies.",
                    scope="local",
                )
                return request
            output_format = structured_outputs_to_format(structured_outputs)

        if resolved_tools is not None:
            tag_request = request
            if output_format is not None:
                # IMPORTANT(arpera):
                # The "auto" tag allows plain text in response.
                # Use tag "required" here to ensure structured-output branch
                # remains meaningful.
                tag_request = request.model_copy(update={"tool_choice": "required"})
            tools_structural_tag = tool_parser.get_structural_tag(
                tag_request,
                reasoning=False,
                strict_level=strict_level,
            )
        else:
            tools_structural_tag = None

        if tools_structural_tag is None:
            if is_auto and structured_outputs is not None:
                logger.warning_once(
                    "Tool calls are not constrained for tool_choice=auto with "
                    "structured outputs because structural tags are unavailable; "
                    "the structured output constraint applies.",
                    scope="local",
                )
            return request

        if single_call:
            limit_to_single_tool_call(tools_structural_tag)
        structural_tag = tools_structural_tag
        if output_format is not None:
            structural_tag = StructuralTag(
                format=OrFormat(elements=[tools_structural_tag.format, output_format])
            )
        elif structured_outputs is not None and not is_auto:
            logger.warning_once(
                "structured outputs are ignored because tool_choice forces tool call.",
                scope="local",
            )

        request.structured_outputs = StructuredOutputsParams(
            structural_tag=json.dumps(structural_tag.model_dump()),
        )
        if isinstance(request, ResponsesRequest):
            request.text = None
        else:
            request.response_format = None
        return request

    def extract_reasoning_streaming(
        self,
        previous_text: str,
        current_text: str,
        delta_text: str,
        previous_token_ids: Sequence[int],
        current_token_ids: Sequence[int],
        delta_token_ids: Sequence[int],
    ) -> DeltaMessage | None:
        if self._reasoning_parser is None:
            return DeltaMessage(content=delta_text)
        return self._reasoning_parser.extract_reasoning_streaming(
            previous_text,
            current_text,
            delta_text,
            previous_token_ids,
            current_token_ids,
            delta_token_ids,
        )

    def extract_tool_calls(
        self,
        model_output: str,
        request: ChatCompletionRequest | ResponsesRequest,
    ) -> ExtractedToolCallInformation:
        if self._tool_parser is None:
            return ExtractedToolCallInformation(
                tools_called=False, tool_calls=[], content=model_output
            )
        result = None
        is_tool_called: bool | Exception = False
        try:
            result = self._tool_parser.extract_tool_calls(
                model_output,
                request=request,  # type: ignore[arg-type]
            )
            is_tool_called = bool(result.tools_called)
        except Exception as e:
            is_tool_called = e
            raise
        finally:
            record_tool_parser_invocation(
                is_tool_called=is_tool_called,
                is_streaming=False,
                request=request,
            )
        return result

    def extract_tool_calls_streaming(
        self,
        previous_text: str,
        current_text: str,
        delta_text: str,
        previous_token_ids: Sequence[int],
        current_token_ids: Sequence[int],
        delta_token_ids: Sequence[int],
        request: ChatCompletionRequest | ResponsesRequest,
    ) -> DeltaMessage | None:
        if self._tool_parser is None:
            return None
        result = None
        is_tool_called: bool | Exception = False
        try:
            result = self._tool_parser.extract_tool_calls_streaming(
                previous_text,
                current_text,
                delta_text,
                previous_token_ids,
                current_token_ids,
                delta_token_ids,
                request,  # type: ignore[arg-type]
            )
            is_tool_called = bool(result and result.tool_calls)
        except Exception as e:
            is_tool_called = e
            raise
        finally:
            record_tool_parser_invocation(
                is_tool_called=is_tool_called,
                is_streaming=True,
                request=request,
            )
        return result

    def _extract_tool_calls_streaming(
        self,
        previous_text: str,
        current_text: str,
        delta_text: str,
        previous_token_ids: Sequence[int],
        current_token_ids: Sequence[int],
        delta_token_ids: Sequence[int],
        request: ChatCompletionRequest | ResponsesRequest,
        # The following parameters are used for "required" tool choice parsing and are
        # tracked in StreamState for streaming parsing.
        tool_call_idx: int | None = None,
        tool_call_id_type: str = "random",
        function_name_returned: bool = False,
    ) -> tuple[DeltaMessage | None, bool]:
        assert self._tool_parser is not None
        supports_required_and_named = self._tool_parser.supports_required_and_named

        if request.tool_choice == "none":
            if self._engine_based:
                # Engine-backed parsers route content extraction through
                # extract_tool_calls_streaming, so run the full pipeline
                # and strip tool_calls after.
                delta_message = self.extract_tool_calls_streaming(
                    previous_text,
                    current_text,
                    delta_text,
                    previous_token_ids,
                    current_token_ids,
                    delta_token_ids,
                    request,  # type: ignore[arg-type]
                )
                if delta_message:
                    delta_message.tool_calls = []
                return delta_message, False
            return (DeltaMessage(content=delta_text) if delta_text else None), False

        if (
            supports_required_and_named
            and request.tool_choice
            and isinstance(
                request.tool_choice,
                (ToolChoiceFunction, ChatCompletionNamedToolChoiceParam),
            )
        ):
            delta_message, function_name_returned = extract_named_tool_call_streaming(
                delta_text=delta_text,
                function_name=self._get_function_name(request),
                function_name_returned=function_name_returned,
                tool_call_idx=tool_call_idx,
                tool_call_id_type=tool_call_id_type,
                tokenizer=self.model_tokenizer,
            )
            return delta_message, function_name_returned

        if supports_required_and_named and request.tool_choice == "required":
            delta_message, function_name_returned = (
                extract_required_tool_call_streaming(
                    previous_text=previous_text,
                    current_text=current_text,
                    delta_text=delta_text,
                    function_name_returned=function_name_returned,
                    tool_call_idx=tool_call_idx,
                    tool_call_id_type=tool_call_id_type,
                )
            )
            return delta_message, function_name_returned
        return self.extract_tool_calls_streaming(
            previous_text,
            current_text,
            delta_text,
            previous_token_ids,
            current_token_ids,
            delta_token_ids,
            request,
        ), False

    def is_reasoning_end(self, input_ids: list[int]) -> bool:
        if self._reasoning_parser is None:
            return False
        return self._reasoning_parser.is_reasoning_end(input_ids)

    def _is_reasoning_end_streaming(
        self, input_ids: list[int], delta_ids: list[int]
    ) -> bool:
        if self._reasoning_parser is None:
            return False
        return self._reasoning_parser.is_reasoning_end_streaming(input_ids, delta_ids)

    def _extract_content_ids(self, input_ids: list[int]) -> list[int]:
        if self._reasoning_parser is None:
            return input_ids
        return self._reasoning_parser.extract_content_ids(input_ids)

    def _in_reasoning_phase(self, state: StreamState) -> bool:
        if self._reasoning_parser is None:
            return False
        return not state.reasoning_ended

    def _in_tool_call_phase(self, state: StreamState) -> bool:
        if self._tool_parser is None:
            return False
        return state.reasoning_ended

    def _append_unstreamed_tool_args(
        self,
        delta_message: DeltaMessage | None,
    ) -> None:
        """Append parsed-but-unstreamed tool-call arguments to *delta_message*."""
        if (
            self._tool_parser is not None
            and delta_message
            and delta_message.tool_calls
            and (last_tc := delta_message.tool_calls[-1]).function
        ):
            last_tc.function.arguments = (
                last_tc.function.arguments or ""
            ) + self._tool_parser.get_remaining_unstreamed_args()

    def finalize_generation(
        self,
        delta_message: DeltaMessage | None,
        request: ChatCompletionRequest | ResponsesRequest,
        state: StreamState,
    ) -> DeltaMessage | None:
        """Finalize generation for cases where generation was incomplete.
        For example, if streaming terminated before reasoning ended
        """
        fallback_fn = getattr(
            self._reasoning_parser, "get_streaming_fallback_content", None
        )
        if fallback_fn is not None and not state.reasoning_ended:
            promoted = fallback_fn(state.previous_text, request)
            if promoted:
                if delta_message is None:
                    delta_message = DeltaMessage()
                delta_message.content = (delta_message.content or "") + promoted

        self._append_unstreamed_tool_args(delta_message)
        return delta_message

    def parse(
        self,
        model_output: str,
        request: ChatCompletionRequest | ResponsesRequest,
        enable_auto_tools: bool = False,
        model_output_token_ids: Sequence[int] = (),
    ) -> tuple[str | None, str | None, list[FunctionCall] | None]:
        self._initialize_history_tool_call_cnt(request)
        reasoning, content = self.extract_reasoning(model_output, request)
        tool_calls, content = self._extract_tool_calls(
            content=content,
            request=request,
            enable_auto_tools=enable_auto_tools,
        )
        return reasoning, content, tool_calls

    def parse_delta(
        self,
        delta_text: str,
        delta_token_ids: list[int],
        request: ChatCompletionRequest | ResponsesRequest,
        prompt_token_ids: list[int] | None = None,
        *,
        finished: bool,
    ) -> DeltaMessage | None:
        self._initialize_history_tool_call_cnt(request)
        state = self._stream_state

        if not state.prompt_reasoning_checked and prompt_token_ids is not None:
            state.prompt_reasoning_checked = True
            if self._reasoning_parser is None or self.is_reasoning_end(
                prompt_token_ids
            ):
                state.reasoning_ended = True
            else:
                # Reasoning is still open at the end of the prompt; let the
                # reasoning parser adjust its initial parsing state so the
                # first generated tokens are classified correctly.
                self._reasoning_parser.adjust_initial_state_from_prompt(
                    prompt_token_ids
                )

        current_text, current_token_ids = state.advance(delta_text, delta_token_ids)
        delta_message: DeltaMessage | None = None
        reasoning_transitioned = False

        # Reasoning extraction
        if self._in_reasoning_phase(state):
            delta_message = self.extract_reasoning_streaming(
                previous_text=state.previous_text,
                current_text=current_text,
                delta_text=delta_text,
                previous_token_ids=state.previous_token_ids,
                current_token_ids=current_token_ids,
                delta_token_ids=delta_token_ids,
            )
            reasoning_parser = self._reasoning_parser
            if reasoning_parser is not None and reasoning_parser.engine_based_streaming:
                should_transition = (
                    reasoning_parser.has_engine_confirmed_reasoning_end()
                )
            else:
                should_transition = self._is_reasoning_end_streaming(
                    current_token_ids, delta_token_ids
                )
            if should_transition:
                state.reasoning_ended = True
                reasoning_transitioned = True
                current_token_ids = self._extract_content_ids(delta_token_ids)
                # Flush whenever the reasoning parser is engine-based (not only
                # when _engine_based is True): it buffers the post-marker text
                # (e.g. the "<" of "<tool_call>"), surfaced via finish_streaming().
                flush_delta = (
                    reasoning_parser.finish_streaming()  # type: ignore[union-attr, attr-defined]
                    if reasoning_parser is not None
                    and reasoning_parser.engine_based_streaming
                    else None
                )
                current_text = (
                    (delta_message.content if delta_message else None) or ""
                ) + ((flush_delta.content if flush_delta else None) or "")
                if self._engine_based:
                    if delta_message and self._tool_parser is not None:
                        delta_message.content = None
                else:
                    delta_text = current_text

        # Tool call extraction
        if self._in_tool_call_phase(state):
            if not state.tool_call_text_started:
                state.tool_call_text_started = True
                state.previous_text = ""
                state.previous_token_ids = []
                delta_text = current_text
                delta_token_ids = current_token_ids

            reasoning_from_this_batch = (
                delta_message.reasoning if delta_message else None
            )

            delta_message, state.function_name_returned = (
                self._extract_tool_calls_streaming(
                    previous_text=state.previous_text,
                    current_text=current_text,
                    delta_text=delta_text,
                    previous_token_ids=state.previous_token_ids,
                    current_token_ids=current_token_ids,
                    delta_token_ids=delta_token_ids,
                    request=request,  # type: ignore[arg-type]
                    tool_call_idx=state.history_tool_call_cnt,
                    tool_call_id_type=state.tool_call_id_type,
                    function_name_returned=state.function_name_returned,
                )
            )

            if reasoning_from_this_batch:
                if delta_message is None:
                    delta_message = DeltaMessage(reasoning=reasoning_from_this_batch)
                elif not delta_message.reasoning:
                    delta_message.reasoning = reasoning_from_this_batch

            if (
                delta_message
                and delta_message.tool_calls
                and delta_message.tool_calls[0].id is not None
            ):
                state.history_tool_call_cnt += 1

        # No phase active: pass through as content.
        # Skip when reasoning just ended in this delta — the engine already
        # consumed the end-of-reasoning marker (e.g. </think>) and
        # delta_text still contains the raw marker text.
        if (
            delta_message is None
            and not reasoning_transitioned
            and not self._in_reasoning_phase(state)
            and not self._in_tool_call_phase(state)
        ):
            delta_message = DeltaMessage(content=delta_text)

        state.commit(current_text, current_token_ids)

        if finished:
            delta_message = self.finalize_generation(delta_message, request, state)
            delta_message = self._flush_engine_parsers(delta_message)

        # Suppress reasoning deltas if not requested
        if delta_message and not request.include_reasoning:
            delta_message.reasoning = None

            # If only reasoning was in the message (no content, no tool_calls)
            # skip emitting entirely
            if not delta_message.content and not delta_message.tool_calls:
                delta_message = None

        return delta_message

    def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
        """Count reasoning tokens through the configured reasoning parser."""
        if self._reasoning_parser is None:
            return 0
        return self._reasoning_parser.count_reasoning_tokens(token_ids)

    def _flush_engine_parsers(
        self, delta_message: DeltaMessage | None
    ) -> DeltaMessage | None:
        """Flush buffered state from engine-based parsers at stream end."""
        reasoning_ended = self._stream_state.reasoning_ended
        for parser in (self._reasoning_parser, self._tool_parser):
            if not getattr(parser, "engine_based_streaming", False):
                continue
            # When reasoning has ended and we transitioned to the tool
            # phase, the reasoning parser's engine may still have buffered
            # characters from tool-call markup it saw with
            # skip_tool_parsing=True.  Flushing that would leak spurious
            # content (e.g. a stray '"'), so skip it.
            if parser is self._reasoning_parser and reasoning_ended:
                continue
            finish = getattr(parser, "finish_streaming", None)
            if finish is None:
                continue
            flush_delta = finish()
            if flush_delta is None:
                continue
            if delta_message is None:
                delta_message = flush_delta
            else:
                if flush_delta.content:
                    delta_message.content = (
                        delta_message.content or ""
                    ) + flush_delta.content
                if flush_delta.reasoning:
                    delta_message.reasoning = (
                        delta_message.reasoning or ""
                    ) + flush_delta.reasoning
                if flush_delta.tool_calls:
                    delta_message.tool_calls = (
                        delta_message.tool_calls or []
                    ) + flush_delta.tool_calls
        return delta_message

_append_unstreamed_tool_args(delta_message)

Append parsed-but-unstreamed tool-call arguments to delta_message.

Source code in vllm/parser/abstract_parser.py
def _append_unstreamed_tool_args(
    self,
    delta_message: DeltaMessage | None,
) -> None:
    """Append parsed-but-unstreamed tool-call arguments to *delta_message*."""
    if (
        self._tool_parser is not None
        and delta_message
        and delta_message.tool_calls
        and (last_tc := delta_message.tool_calls[-1]).function
    ):
        last_tc.function.arguments = (
            last_tc.function.arguments or ""
        ) + self._tool_parser.get_remaining_unstreamed_args()

_flush_engine_parsers(delta_message)

Flush buffered state from engine-based parsers at stream end.

Source code in vllm/parser/abstract_parser.py
def _flush_engine_parsers(
    self, delta_message: DeltaMessage | None
) -> DeltaMessage | None:
    """Flush buffered state from engine-based parsers at stream end."""
    reasoning_ended = self._stream_state.reasoning_ended
    for parser in (self._reasoning_parser, self._tool_parser):
        if not getattr(parser, "engine_based_streaming", False):
            continue
        # When reasoning has ended and we transitioned to the tool
        # phase, the reasoning parser's engine may still have buffered
        # characters from tool-call markup it saw with
        # skip_tool_parsing=True.  Flushing that would leak spurious
        # content (e.g. a stray '"'), so skip it.
        if parser is self._reasoning_parser and reasoning_ended:
            continue
        finish = getattr(parser, "finish_streaming", None)
        if finish is None:
            continue
        flush_delta = finish()
        if flush_delta is None:
            continue
        if delta_message is None:
            delta_message = flush_delta
        else:
            if flush_delta.content:
                delta_message.content = (
                    delta_message.content or ""
                ) + flush_delta.content
            if flush_delta.reasoning:
                delta_message.reasoning = (
                    delta_message.reasoning or ""
                ) + flush_delta.reasoning
            if flush_delta.tool_calls:
                delta_message.tool_calls = (
                    delta_message.tool_calls or []
                ) + flush_delta.tool_calls
    return delta_message

count_reasoning_tokens(token_ids)

Count reasoning tokens through the configured reasoning parser.

Source code in vllm/parser/abstract_parser.py
def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
    """Count reasoning tokens through the configured reasoning parser."""
    if self._reasoning_parser is None:
        return 0
    return self._reasoning_parser.count_reasoning_tokens(token_ids)

finalize_generation(delta_message, request, state)

Finalize generation for cases where generation was incomplete. For example, if streaming terminated before reasoning ended

Source code in vllm/parser/abstract_parser.py
def finalize_generation(
    self,
    delta_message: DeltaMessage | None,
    request: ChatCompletionRequest | ResponsesRequest,
    state: StreamState,
) -> DeltaMessage | None:
    """Finalize generation for cases where generation was incomplete.
    For example, if streaming terminated before reasoning ended
    """
    fallback_fn = getattr(
        self._reasoning_parser, "get_streaming_fallback_content", None
    )
    if fallback_fn is not None and not state.reasoning_ended:
        promoted = fallback_fn(state.previous_text, request)
        if promoted:
            if delta_message is None:
                delta_message = DeltaMessage()
            delta_message.content = (delta_message.content or "") + promoted

    self._append_unstreamed_tool_args(delta_message)
    return delta_message

Parser

Parse model output into reasoning, content and tool calls.

The serving layer holds one Parser per request and calls only the members defined here. ParserEngine implements them over a single declarative engine; DelegatingParser composes a legacy ReasoningParser / ToolParser pair.

Methods:

  • adjust_request –

    Adjust the request parameters for tool calling.

  • count_reasoning_tokens –

    Return the number of reasoning tokens in generated token IDs.

  • is_reasoning_end –

    Check if the reasoning content ends in the input_ids.

  • parse –

    Parse a complete model output, extracting reasoning and tool calls.

  • parse_delta –

    Parse a single streaming delta, orchestrating reasoning then

  • set_prompt_token_ids –

    Provide the exact rendered prompt to parsers that need prefix state.

Attributes:

Source code in vllm/parser/abstract_parser.py
class Parser:
    """Parse model output into reasoning, content and tool calls.

    The serving layer holds one ``Parser`` per request and calls only the
    members defined here. ``ParserEngine`` implements them over a single
    declarative engine; ``DelegatingParser`` composes a legacy
    ``ReasoningParser`` / ``ToolParser`` pair.
    """

    # Class-level parser classes for compatibility with existing patterns
    # Subclasses should override these if they use specific parser classes
    reasoning_parser_cls: type[ReasoningParser] | None = None
    tool_parser_cls: type[ToolParser] | None = None
    # Server-side floor for tool-call structural tags (--tool-strict-level).
    tool_strict_level: ToolStrictLevel = ToolStrictLevel.AUTO
    always_adjust_request: bool = False

    def __init__(
        self,
        tokenizer: TokenizerLike,
        tools: list[Tool] | None = None,
        *args,
        model_config=None,
        **kwargs,
    ):
        self.model_tokenizer = tokenizer
        self._reasoning_parser: ReasoningParser | None = None
        self._tool_parser: ToolParser | None = None
        if self.__class__.reasoning_parser_cls is not None:
            self._reasoning_parser = self.__class__.reasoning_parser_cls(
                tokenizer, *args, model_config=model_config, **kwargs
            )
        if self.__class__.tool_parser_cls is not None:
            # Engine-based adapters take the same construction kwargs as
            # the reasoning parser (e.g. chat_template_kwargs for per-request
            # thinking toggles); legacy ToolParser classes take none.
            if issubclass(self.__class__.tool_parser_cls, ParserEngineToolAdapter):
                self._tool_parser = self.__class__.tool_parser_cls(
                    tokenizer, tools, model_config=model_config, **kwargs
                )
            else:
                self._tool_parser = self.__class__.tool_parser_cls(tokenizer, tools)

        self._engine_based = (
            self._reasoning_parser is None
            or self._reasoning_parser.engine_based_streaming
        ) and (self._tool_parser is None or self._tool_parser.engine_based_streaming)
        if (
            self._reasoning_parser is None
            and self._tool_parser is not None
            and hasattr(self._tool_parser, "skip_reasoning_parsing")
        ):
            # With no reasoning parser configured, reasoning markup is
            # plain content: an engine-based tool parser should pass it
            # through verbatim where its grammar allows, not consume or
            # reclassify it. The engine ignores the flag for markers
            # shared with non-reasoning structure.
            self._tool_parser.skip_reasoning_parsing = True
        self._stream_state = StreamState(
            tool_call_id_type=(
                get_tool_call_id_type(model_config)
                if model_config is not None
                else "random"
            ),
            engine_based=self._engine_based,
        )

    @property
    def reasoning_parser(self) -> ReasoningParser | None:
        """The underlying reasoning parser, if any."""
        return self._reasoning_parser

    @property
    def tool_parser(self) -> ToolParser | None:
        """The underlying tool parser, if any."""
        return self._tool_parser

    def _initialize_history_tool_call_cnt(
        self,
        request: ChatCompletionRequest | ResponsesRequest,
    ) -> None:
        state = self._stream_state
        if state.history_tool_call_cnt_initialized:
            return
        if state.tool_call_id_type != "kimi_k2":
            state.history_tool_call_cnt_initialized = True
            return
        state.history_tool_call_cnt = count_history_tool_calls(request)
        state.history_tool_call_cnt_initialized = True

    def adjust_request(
        self, request: ChatCompletionRequest | ResponsesRequest
    ) -> ChatCompletionRequest | ResponsesRequest:
        """Adjust the request parameters for tool calling.

        Can be overridden by subclasses to modify request parameters
        (e.g., setting structured output schemas for tool calling).

        Args:
            request: The original request.

        Returns:
            The adjusted request.

        """
        return request

    def set_prompt_token_ids(self, prompt_token_ids: Sequence[int]) -> None:
        """Provide the exact rendered prompt to parsers that need prefix state."""
        return

    @abstractmethod
    def is_reasoning_end(self, input_ids: list[int]) -> bool:
        """Check if the reasoning content ends in the input_ids.

        Called with the rendered prompt to decide whether generation starts
        after reasoning. Must be a pure function of the input_ids.

        Args:
            input_ids: The token IDs of the model output.

        Returns:
            True if the reasoning content ends in the input_ids.

        """

    @abstractmethod
    def parse(
        self,
        model_output: str,
        request: ChatCompletionRequest | ResponsesRequest,
        enable_auto_tools: bool = False,
        model_output_token_ids: Sequence[int] = (),
    ) -> tuple[str | None, str | None, list[FunctionCall] | None]:
        """Parse a complete model output, extracting reasoning and tool calls.

        Args:
            model_output: The complete model-generated string.
            request: The request object used to generate the output.
            enable_auto_tools: Whether to enable automatic tool call parsing.
            model_output_token_ids: The generated raw output token IDs.

        Returns:
            A tuple of (reasoning, content, tool_calls).

        """

    @abstractmethod
    def parse_delta(
        self,
        delta_text: str,
        delta_token_ids: list[int],
        request: ChatCompletionRequest | ResponsesRequest,
        prompt_token_ids: list[int] | None = None,
        *,
        finished: bool,
    ) -> DeltaMessage | None:
        """Parse a single streaming delta, orchestrating reasoning then
        tool call extraction via internal stream state.
        """

    def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
        """Return the number of reasoning tokens in generated token IDs."""
        return 0

reasoning_parser property

The underlying reasoning parser, if any.

tool_parser property

The underlying tool parser, if any.

adjust_request(request)

Adjust the request parameters for tool calling.

Can be overridden by subclasses to modify request parameters (e.g., setting structured output schemas for tool calling).

Parameters:

Returns:

Source code in vllm/parser/abstract_parser.py
def adjust_request(
    self, request: ChatCompletionRequest | ResponsesRequest
) -> ChatCompletionRequest | ResponsesRequest:
    """Adjust the request parameters for tool calling.

    Can be overridden by subclasses to modify request parameters
    (e.g., setting structured output schemas for tool calling).

    Args:
        request: The original request.

    Returns:
        The adjusted request.

    """
    return request

count_reasoning_tokens(token_ids)

Return the number of reasoning tokens in generated token IDs.

Source code in vllm/parser/abstract_parser.py
def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
    """Return the number of reasoning tokens in generated token IDs."""
    return 0

is_reasoning_end(input_ids) abstractmethod

Check if the reasoning content ends in the input_ids.

Called with the rendered prompt to decide whether generation starts after reasoning. Must be a pure function of the input_ids.

Parameters:

  • input_ids

    (list[int]) –

    The token IDs of the model output.

Returns:

  • bool –

    True if the reasoning content ends in the input_ids.

Source code in vllm/parser/abstract_parser.py
@abstractmethod
def is_reasoning_end(self, input_ids: list[int]) -> bool:
    """Check if the reasoning content ends in the input_ids.

    Called with the rendered prompt to decide whether generation starts
    after reasoning. Must be a pure function of the input_ids.

    Args:
        input_ids: The token IDs of the model output.

    Returns:
        True if the reasoning content ends in the input_ids.

    """

parse(model_output, request, enable_auto_tools=False, model_output_token_ids=()) abstractmethod

Parse a complete model output, extracting reasoning and tool calls.

Parameters:

  • model_output

    (str) –

    The complete model-generated string.

  • request

    (ChatCompletionRequest | ResponsesRequest) –

    The request object used to generate the output.

  • enable_auto_tools

    (bool, default: False ) –

    Whether to enable automatic tool call parsing.

  • model_output_token_ids

    (Sequence[int], default: () ) –

    The generated raw output token IDs.

Returns:

  • tuple[str | None, str | None, list[FunctionCall] | None] –

    A tuple of (reasoning, content, tool_calls).

Source code in vllm/parser/abstract_parser.py
@abstractmethod
def parse(
    self,
    model_output: str,
    request: ChatCompletionRequest | ResponsesRequest,
    enable_auto_tools: bool = False,
    model_output_token_ids: Sequence[int] = (),
) -> tuple[str | None, str | None, list[FunctionCall] | None]:
    """Parse a complete model output, extracting reasoning and tool calls.

    Args:
        model_output: The complete model-generated string.
        request: The request object used to generate the output.
        enable_auto_tools: Whether to enable automatic tool call parsing.
        model_output_token_ids: The generated raw output token IDs.

    Returns:
        A tuple of (reasoning, content, tool_calls).

    """

parse_delta(delta_text, delta_token_ids, request, prompt_token_ids=None, *, finished) abstractmethod

Parse a single streaming delta, orchestrating reasoning then tool call extraction via internal stream state.

Source code in vllm/parser/abstract_parser.py
@abstractmethod
def parse_delta(
    self,
    delta_text: str,
    delta_token_ids: list[int],
    request: ChatCompletionRequest | ResponsesRequest,
    prompt_token_ids: list[int] | None = None,
    *,
    finished: bool,
) -> DeltaMessage | None:
    """Parse a single streaming delta, orchestrating reasoning then
    tool call extraction via internal stream state.
    """

set_prompt_token_ids(prompt_token_ids)

Provide the exact rendered prompt to parsers that need prefix state.

Source code in vllm/parser/abstract_parser.py
def set_prompt_token_ids(self, prompt_token_ids: Sequence[int]) -> None:
    """Provide the exact rendered prompt to parsers that need prefix state."""
    return

StreamState dataclass

Mutable state for Parser.parse_delta(). One per stream.

Source code in vllm/parser/abstract_parser.py
@dataclass
class StreamState:
    """Mutable state for ``Parser.parse_delta()``. One per stream."""

    reasoning_ended: bool = False
    tool_call_text_started: bool = False
    prompt_reasoning_checked: bool = False
    previous_text: str = ""
    previous_token_ids: list[int] = field(default_factory=list)
    history_tool_call_cnt: int = 0
    history_tool_call_cnt_initialized: bool = False
    tool_call_id_type: str = "random"
    # only used for "required" and "named tool" choices,
    # tracks whether function name has been fully returned in the stream yet
    function_name_returned: bool = False
    engine_based: bool = False

    def advance(
        self,
        delta_text: str,
        delta_token_ids: list[int],
    ) -> tuple[str, list[int]]:
        if self.engine_based:
            return delta_text, delta_token_ids
        return (
            self.previous_text + delta_text,
            self.previous_token_ids + delta_token_ids,
        )

    def commit(
        self,
        current_text: str,
        current_token_ids: list[int],
    ) -> None:
        if self.engine_based:
            self.previous_text = ""
            self.previous_token_ids = []
        else:
            self.previous_text = current_text
            self.previous_token_ids = current_token_ids

_xgrammar_supports(structured_outputs)

Whether the constraint passes xgrammar validation on its own.

Source code in vllm/parser/abstract_parser.py
def _xgrammar_supports(structured_outputs: StructuredOutputsParams) -> bool:
    """Whether the constraint passes xgrammar validation on its own."""
    from vllm.v1.structured_output.backend_xgrammar import validate_xgrammar_grammar

    try:
        # validate_xgrammar_grammar rewrites `choice` into `grammar` in place.
        validate_xgrammar_grammar(
            SamplingParams(structured_outputs=replace(structured_outputs))
        )
    except (TypeError, ValueError, VLLMValidationError):
        return False
    return True

structured_outputs_to_format(params)

Map StructuredOutputsParams in a XGrammar Format.

Source code in vllm/parser/abstract_parser.py
def structured_outputs_to_format(params: StructuredOutputsParams) -> Format | None:
    """Map StructuredOutputsParams in a XGrammar Format."""
    if params.json_object:
        return JSONSchemaFormat(json_schema={"type": "object"})
    if params.json is not None:
        schema = params.json
        if isinstance(schema, str):
            schema = json.loads(schema)
        return JSONSchemaFormat(json_schema=schema)
    if params.regex is not None:
        return RegexFormat(pattern=params.regex)
    if params.choice is not None:
        return OrFormat(
            elements=[ConstStringFormat(value=choice) for choice in params.choice]
        )
    if params.grammar is not None:
        from vllm.v1.structured_output.utils import grammar_is_likely_lark

        grammar = params.grammar
        # IMPORTANT(arpera):
        # GrammarFormat only accepts EBNF.
        if grammar_is_likely_lark(grammar):
            try:
                grammar = str(Grammar.from_lark(grammar))
            except Exception as e:
                raise VLLMValidationError("Invalid grammar specification.") from e
        return GrammarFormat(grammar=grammar)
    if params.structural_tag is not None:
        s_tag = json.loads(params.structural_tag)
        if "structures" in s_tag:
            # LegacyStructuralTagResponseFormat
            return TriggeredTagsFormat(
                triggers=s_tag["triggers"],
                tags=[
                    TagFormat(
                        begin=structure["begin"],
                        content=JSONSchemaFormat(json_schema=structure["schema"]),
                        end=structure["end"],
                    )
                    for structure in s_tag["structures"]
                ],
            )
        # StructuralTagResponseFormat
        return StructuralTag.model_validate(s_tag).format
    return None