diff --git a/backend/python/vllm/backend.py b/backend/python/vllm/backend.py
index 7235c8e0737a..6936096c35b8 100644
--- a/backend/python/vllm/backend.py
+++ b/backend/python/vllm/backend.py
@@ -558,6 +558,73 @@ def _build_sampling_params(self, request):
return sampling_params
+ def _new_reasoning_parser(self, chat_template_kwargs):
+ """Build the reasoning parser, telling it whether thinking is on.
+
+ vLLM's newer parser engines decide their *initial state* from
+ ``chat_template_kwargs``: ``Qwen3Parser`` reads
+ ``chat_template_kwargs["enable_thinking"]`` and defaults to ``True``,
+ starting in the REASONING state. Constructed without it, a completion
+ produced with thinking disabled is classified as reasoning end to end,
+ and the answer is reported in both ``reasoning_content`` and
+ ``content``.
+
+ vLLM's own OpenAI server forwards the request's chat template kwargs
+ here; this backend renders the template itself, so it forwards the
+ same dict. Older parsers do not accept the argument — fall back to the
+ plain constructor for those.
+ """
+ try:
+ return self.reasoning_parser_cls(
+ self.tokenizer, chat_template_kwargs=chat_template_kwargs or {},
+ )
+ except TypeError:
+ return self.reasoning_parser_cls(self.tokenizer)
+
+ @staticmethod
+ def _split_reasoning(rp, generated_text, prompt, reasoning, content):
+ """Decide what the reasoning parser's output actually means.
+
+ Covers the *older* parser shape, which has no initial state to set:
+ ``BaseThinkingReasoningParser.extract_reasoning`` documents its own
+ fallback — "For models that may not generate start token, assume the
+ reasoning content is always at the start." When no end token is
+ present it returns *everything* as reasoning and ``None`` as content,
+ which is right for a truncated reasoning run and wrong for a
+ completion that never contained reasoning at all.
+
+ Taking ``None`` content to mean "keep the raw text" then duplicates
+ the answer into both fields.
+
+ The prompt says which case it is. A template with thinking on leaves
+ the reasoning block open (the prompt ends with the start token); with
+ thinking off it closes the block in the prompt, so the completion is
+ plain content. Parsers that expose no token pair (the engine-based
+ adapters, which take the ``chat_template_kwargs`` route above) keep
+ the parser's verdict unchanged.
+ """
+ start = getattr(rp, "start_token", None)
+ end = getattr(rp, "end_token", None)
+
+ if end and end in generated_text:
+ # The parser split on the end token. Empty content here means the
+ # model stopped right after it, not that parsing failed.
+ return reasoning or "", content or ""
+
+ if not start:
+ # Unknown token layout — keep the previous behaviour rather than
+ # guess.
+ return reasoning or "", content if content is not None else generated_text
+
+ if not (start in generated_text or (prompt or "").rstrip().endswith(start)):
+ # No end token and the block was never open: the "reasoning starts
+ # at the beginning" fallback does not apply to this completion.
+ return "", generated_text
+
+ # Block was open and the end token never arrived — reasoning ran out of
+ # budget. It is all reasoning, and there is no answer to report.
+ return reasoning or "", content or ""
+
async def _predict(self, request, context, streaming=False):
# Build the sampling parameters
sampling_params = self._build_sampling_params(request)
@@ -572,6 +639,9 @@ async def _predict(self, request, context, streaming=False):
# Extract image paths and process images
prompt = request.Prompt
+ # Kept in scope: the reasoning parser needs to know which chat
+ # template kwargs produced this prompt.
+ template_kwargs = {}
image_paths = request.Images
image_data = [self.load_image(img_path) for img_path in image_paths]
@@ -582,7 +652,7 @@ async def _predict(self, request, context, streaming=False):
# If tokenizer template is enabled and messages are provided instead of prompt, apply the tokenizer template
if not request.Prompt and request.UseTokenizerTemplate and request.Messages:
messages_dicts = self._messages_to_dicts(request.Messages)
- template_kwargs = {"tokenize": False, "add_generation_prompt": True}
+ template_kwargs.update({"tokenize": False, "add_generation_prompt": True})
# Pass tools for tool calling
if request.Tools:
@@ -757,10 +827,11 @@ async def _predict(self, request, context, streaming=False):
if self.reasoning_parser_cls:
try:
- rp = self.reasoning_parser_cls(self.tokenizer)
+ rp = self._new_reasoning_parser(template_kwargs)
r, c = rp.extract_reasoning(generated_text, request=None)
- reasoning_content = r or ""
- content = c if c is not None else generated_text
+ reasoning_content, content = self._split_reasoning(
+ rp, generated_text, prompt, r, c,
+ )
except Exception as e:
print(f"Reasoning parser error: {e}", file=sys.stderr)
diff --git a/backend/python/vllm/test.py b/backend/python/vllm/test.py
index a0679d4ff8e5..0bf223b7e370 100644
--- a/backend/python/vllm/test.py
+++ b/backend/python/vllm/test.py
@@ -549,3 +549,116 @@ def test_no_tool_parser_unchanged_per_delta_stream(self):
intermediate, ["Hello ", "world", "!"],
f"plain streaming changed; got {intermediate!r}",
)
+
+
+class TestReasoningSplit(unittest.TestCase):
+ """Server-less tests for BackendServicer._split_reasoning.
+
+ vLLM's BaseThinkingReasoningParser returns the whole completion as
+ reasoning and None as content whenever the end token is missing. Taken
+ literally that duplicates a thinking-disabled answer into both fields.
+ """
+
+ class _Parser:
+ start_token = ""
+ end_token = ""
+
+ def _split(self, generated, prompt, reasoning, content):
+ import sys, os
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+ from backend import BackendServicer
+ return BackendServicer._split_reasoning(
+ self._Parser(), generated, prompt, reasoning, content,
+ )
+
+ def test_thinking_off_is_not_duplicated_into_reasoning(self):
+ """No tags anywhere: the answer is content, and only content."""
+ r, c = self._split(
+ "391", "user: 17*23?\n\n\n\n\n",
+ reasoning="391", content=None,
+ )
+ self.assertEqual(r, "")
+ self.assertEqual(c, "391")
+
+ def test_prefilled_start_tag_keeps_truncated_reasoning(self):
+ """Prompt left the block open and the end token never arrived
+ (budget exhausted): that really is all reasoning."""
+ r, c = self._split(
+ "thinking and thinking", "user: hi\n\n",
+ reasoning="thinking and thinking", content=None,
+ )
+ self.assertEqual(r, "thinking and thinking")
+ self.assertEqual(c, "")
+
+ def test_end_token_present_keeps_parser_split(self):
+ r, c = self._split(
+ "adding two and two4", "user: hi\n\n",
+ reasoning="adding two and two", content="4",
+ )
+ self.assertEqual(r, "adding two and two")
+ self.assertEqual(c, "4")
+
+ def test_stop_right_after_end_token_yields_empty_content(self):
+ """Content must not fall back to the raw text — that would put the
+ reasoning into the answer."""
+ r, c = self._split(
+ "reasoned", "user: hi\n\n",
+ reasoning="reasoned", content=None,
+ )
+ self.assertEqual(r, "reasoned")
+ self.assertEqual(c, "")
+
+ def test_unknown_token_layout_keeps_previous_behaviour(self):
+ class _Bare:
+ pass
+ import sys, os
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+ from backend import BackendServicer
+ r, c = BackendServicer._split_reasoning(
+ _Bare(), "raw", "prompt", "raw", None,
+ )
+ self.assertEqual(r, "raw")
+ self.assertEqual(c, "raw")
+
+
+class TestReasoningParserConstruction(unittest.TestCase):
+ """The parser must learn whether thinking was on for this request.
+
+ vLLM's engine-based parsers (Qwen3Parser and friends) read
+ chat_template_kwargs["enable_thinking"] and default to True, so a parser
+ built without it treats a thinking-disabled completion as pure reasoning.
+ """
+
+ def _servicer(self):
+ import sys, os
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+ from backend import BackendServicer
+ s = BackendServicer()
+ s.tokenizer = object()
+ return s
+
+ def test_chat_template_kwargs_are_forwarded(self):
+ seen = {}
+
+ class _Parser:
+ def __init__(self, tokenizer, **kwargs):
+ seen.update(kwargs)
+
+ s = self._servicer()
+ s.reasoning_parser_cls = _Parser
+ s._new_reasoning_parser({"enable_thinking": False})
+ self.assertEqual(
+ seen.get("chat_template_kwargs"), {"enable_thinking": False},
+ )
+
+ def test_parser_without_the_kwarg_still_builds(self):
+ """Older parsers take only the tokenizer — must not break them."""
+ class _Old:
+ def __init__(self, tokenizer):
+ self.tokenizer = tokenizer
+
+ s = self._servicer()
+ s.reasoning_parser_cls = _Old
+ self.assertIsInstance(
+ s._new_reasoning_parser({"enable_thinking": False}), _Old,
+ )