Skip to content

vllm.reasoning.kimi_k3_reasoning_parser

Reasoning parser for the Kimi K3 (XTML) chat format.

This strips the think channel out of generated text and hands the remainder (response + tools channels) downstream. Kimi K3 wraps the thinking channel as an XTML element built from special tokens::

<|open|>think<|sep|> <reasoning> <|close|>think<|sep|>
Two subtleties drive the implementation
  • Unlike Kimi-K2 (a single <think> token), each K3 marker is a 3-token sequence, so the token-id helpers search for the marker subsequence rather than a single id.
  • In thinking mode the serving layer may feed <|open|>think<|sep|> as the generation prefix, so the model's output can begin inside the think channel with no open marker. The text paths therefore treat a missing open marker as "reasoning starts at offset 0".

When thinking is disabled (chat_template_kwargs={"thinking": False} or {"enable_thinking": False}, i.e. instruct mode) the parser returns every delta as normal content; there is simply no think channel to extract.

Classes:

KimiK3ReasoningParser

Bases: ReasoningParser

Reasoning parser for the Kimi K3 (XTML) think channel.

Methods:

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
class KimiK3ReasoningParser(ReasoningParser):
    """Reasoning parser for the Kimi K3 (XTML) think channel."""

    def __init__(self, tokenizer: PreTrainedTokenizerBase, *args, **kwargs):
        super().__init__(tokenizer)

        if not self.model_tokenizer:
            raise ValueError(
                "The model tokenizer must be passed to the ReasoningParser "
                "constructor during construction."
            )

        # thinking can be disabled via chat_template_kwargs -> identity fallthrough
        chat_kwargs = kwargs.get("chat_template_kwargs", {}) or {}
        thinking = chat_kwargs.get("thinking", None)
        if thinking is None:
            thinking = chat_kwargs.get("enable_thinking", True)
        self._thinking_enabled = bool(thinking)

        # XTML markers as literal strings (skip_special_tokens=False at serve time)
        self._think_open = "<|open|>think<|sep|>"
        self._think_close = "<|close|>think<|sep|>"

        # Content-channel markers that must be stripped from the final output.
        # The tool parser handles these when active, but when tool_choice is
        # "none" (the default for requests without tools) the tool parser is
        # bypassed and the reasoning parser must strip them itself.
        self._response_open = "<|open|>response<|sep|>"
        self._response_close = "<|close|>response<|sep|>"
        self._message_close = "<|close|>message<|sep|>"

        # Tolerant matchers: defense-in-depth against vLLM's added-token spacing
        # ("<|open|> think <|sep|>"). The `\s*` is a no-op on clean input, so the
        # normal path stays byte-exact; it only helps if some serving path leaves
        # spaces_between_special_tokens on. See adjust_request for the real fix.
        open_marker = r"<\|open\|>"
        close_marker = r"<\|close\|>"
        sep_marker = r"<\|sep\|>"
        self._think_open_re = re.compile(open_marker + r"\s*think\s*" + sep_marker)
        self._think_close_re = re.compile(close_marker + r"\s*think\s*" + sep_marker)
        self._response_open_re = re.compile(
            open_marker + r"\s*response\s*" + sep_marker
        )
        self._response_close_re = re.compile(
            close_marker + r"\s*response\s*" + sep_marker
        )
        self._message_close_re = re.compile(
            close_marker + r"\s*message\s*" + sep_marker
        )

        # marker token-id subsequences (each marker is 3 tokens)
        self._think_open_ids = tokenizer.encode(
            self._think_open, add_special_tokens=False
        )
        self._think_close_ids = tokenizer.encode(
            self._think_close, add_special_tokens=False
        )
        self._response_open_ids = tokenizer.encode(
            self._response_open, add_special_tokens=False
        )
        self._last_streaming_delta_token_ids: tuple[int, ...] | None = None
        self._last_streaming_content_token_ids: list[int] | None = None

    @property
    def reasoning_start_str(self) -> str | None:
        return self._think_open

    @property
    def reasoning_end_str(self) -> str | None:
        return self._think_close

    def adjust_request(
        self,
        request: "ChatCompletionRequest | ResponsesRequest",
    ) -> "ChatCompletionRequest | ResponsesRequest":
        request.skip_special_tokens = False
        if hasattr(request, "spaces_between_special_tokens"):
            request.spaces_between_special_tokens = False
        return request

    def is_reasoning_end(self, input_ids: Sequence[int]) -> bool:
        if not self._thinking_enabled:
            return True
        # Reasoning has ended only if the *most recent* think block is closed:
        # the last close marker must come after the last open marker. A plain
        # "a close marker exists anywhere" check false-positives in multi-turn /
        # agent continuations, where the chat template keeps a prior turn's
        # think channel (with its <|close|>think<|sep|>) in the prompt while the
        # current turn is still reasoning (its <|open|>think<|sep|> is the newest
        # marker). A missing open marker (e.g. it was consumed as the generation
        # prefix) means a close marker alone ends reasoning, which is what
        # "close is the newest marker" already encodes.
        return (
            _newest_marker(input_ids, self._think_close_ids, self._think_open_ids) == 0
        )

    def is_reasoning_end_streaming(
        self, input_ids: Sequence[int], delta_ids: Iterable[int]
    ) -> bool:
        """Reasoning-end check for a single decode step.

        The engine calls this once per structured-output request per decode step
        while the request is still inside the think channel, so it only has to
        look at the tokens generated this step, plus ``len(marker) - 1`` tokens
        of context in case a marker straddles the step boundary.

        The inherited default re-runs the full-sequence ``is_reasoning_end``,
        which makes the scheduler O(context) per request per step and starves the
        GPU on long agentic contexts.
        """
        if not self._thinking_enabled:
            return True
        delta = list(delta_ids)
        if not delta:
            return False
        carry = max(len(self._think_close_ids), len(self._think_open_ids)) - 1
        head = len(input_ids) - len(delta)
        window = list(input_ids[max(0, head - carry) : head]) + delta
        return _newest_marker(window, self._think_close_ids, self._think_open_ids) == 0

    def _extract_content_ids(self, input_ids: list[int]) -> list[int]:
        if not self._thinking_enabled:
            return input_ids
        idx = _subseq_index(input_ids, self._think_close_ids)
        if idx == -1:
            return []  # still reasoning
        return input_ids[idx + len(self._think_close_ids) :]

    def extract_content_ids(self, input_ids: list[int]) -> list[int]:
        cached_delta_ids = self._last_streaming_delta_token_ids
        cached_content_ids = self._last_streaming_content_token_ids
        self._last_streaming_delta_token_ids = None
        self._last_streaming_content_token_ids = None
        if cached_delta_ids == tuple(input_ids) and cached_content_ids is not None:
            return cached_content_ids
        return self._extract_content_ids(input_ids)

    def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
        if not self._thinking_enabled:
            return 0
        think_open = self._think_open_ids
        think_close = self._think_close_ids
        response_open = self._response_open_ids
        first_ids = {m[0] for m in (think_open, think_close, response_open) if m}
        n = len(token_ids)
        count = 0
        in_reasoning = True
        seen_open = False
        i = 0
        while i < n:
            head = token_ids[i]
            if head in first_ids:
                if i + len(think_open) <= n and _match_at(token_ids, i, think_open):
                    in_reasoning = True
                    seen_open = True
                    i += len(think_open)
                    continue
                if i + len(think_close) <= n and _match_at(token_ids, i, think_close):
                    in_reasoning = False
                    i += len(think_close)
                    continue
                if (
                    not seen_open
                    and i + len(response_open) <= n
                    and _match_at(token_ids, i, response_open)
                ):
                    in_reasoning = False
                    i += len(response_open)
                    continue
            if in_reasoning:
                count += 1
            i += 1
        return count

    def _strip_content_wrapper(self, text: str) -> str:
        """Strip ``<|open|>response<|sep|>…<|close|>response<|sep|>`` wrapper and
        ``<|close|>message<|sep|>`` from *text*.

        When Kimi K3 tool parsing is active it needs the raw XTML
        ``response`` + ``tools`` channels. Otherwise the reasoning parser cleans
        up the response wrapper itself so API users do not see XTML markers.
        """
        # Try structured unwrap first: extract body of response channel
        m_ro = self._response_open_re.search(text)
        m_rc = self._response_close_re.search(text, m_ro.end() if m_ro else 0)
        if m_ro is not None and m_rc is not None:
            text = text[m_ro.end() : m_rc.start()]
        elif m_ro is not None:
            # response opened but not closed (truncated)
            text = text[m_ro.end() :]
        else:
            # No response wrapper — strip stray markers if present
            text = self._response_open_re.sub("", text)
            text = self._response_close_re.sub("", text)
        text = self._message_close_re.sub("", text)
        return text

    @staticmethod
    def _should_preserve_tool_channels(
        request: "ChatCompletionRequest | ResponsesRequest",
    ) -> bool:
        return bool(getattr(request, "tools", None)) and (
            getattr(request, "tool_choice", None) != "none"
        )

    def _content_after_reasoning(
        self,
        text: str,
        request: "ChatCompletionRequest | ResponsesRequest",
    ) -> str | None:
        if self._should_preserve_tool_channels(request):
            return text or None
        return self._strip_content_wrapper(text) or None

    def extract_reasoning(
        self, model_output: str, request: "ChatCompletionRequest | ResponsesRequest"
    ) -> tuple[str | None, str | None]:
        """Split full text into ``(reasoning, rest)`` for the non-streaming path.

        Handles four shapes:
          * response opener without think markers -> no think channel; all content
          * open marker present       -> reasoning starts after ``<|open|>think<|sep|>``
          * open marker absent but a   close marker exists (gen-prefix consumed
            the open) -> reasoning starts at offset 0
          * neither marker present     -> truncated reasoning after a consumed
            generation prefix
        ``rest`` is whatever follows the close marker, fed on to the tool parser.
        """
        if not self._thinking_enabled:
            return None, self._content_after_reasoning(model_output, request)

        m_open = self._think_open_re.search(model_output)
        # reasoning content begins right after think-open (or at start if the
        # open marker was already consumed as a generation prefix)
        content_start = m_open.end() if m_open is not None else 0

        m_close = self._think_close_re.search(model_output, content_start)
        if (
            m_open is None
            and m_close is None
            and self._response_open_re.search(model_output) is not None
        ):
            return None, self._content_after_reasoning(model_output, request)
        if m_close is not None:
            reasoning = model_output[content_start : m_close.start()]
            rest = model_output[m_close.end() :]
            return (reasoning or None, self._content_after_reasoning(rest, request))
        # think not closed -> still reasoning, no content yet
        return (model_output[content_start:] or None, None)

    def _reasoning_text_ready_to_emit(self, text: str) -> str:
        """Return the reasoning prefix that is safe to stream now.

        Work from accumulated text, not from the current delta alone. That turns
        split-open and split-close handling into the same prefix-diff problem.

        Example 1: split think-open marker.
          chunks: ``<|open|>`` / ``think`` / ``<|sep|>reasoning``
          current text after chunk 1: ``<|open|>`` -> emit ``""``
          current text after chunk 2: ``<|open|>think`` -> emit ``""``
          current text after chunk 3: ``<|open|>think<|sep|>reasoning``
          -> emit ``reasoning``

        Example 2: split think-close marker.
          chunks: ``reasoning`` / ``<|close|>`` / ``think<|sep|>...``
          after ``<|close|>``, the suffix is only a partial close marker, so the
          sendable reasoning is still ``reasoning`` and the delta is empty. Once
          ``think<|sep|>`` arrives, the close branch hands the following response
          or tools text to the downstream parser.
        """
        m_open = self._think_open_re.search(text)
        if m_open is not None:
            text = text[m_open.end() :]
        overlap = 0
        for marker in (self._think_open, self._think_close):
            max_check = min(len(marker) - 1, len(text))
            for n in range(max_check, 0, -1):
                if text.endswith(marker[:n]):
                    overlap = max(overlap, n)
                    break
        return text[:-overlap] if overlap else text

    def _content_ready_to_emit(self, text: str) -> str:
        """Return the content prefix that is safe to stream now.

        Mirrors ``_reasoning_text_ready_to_emit`` but for the post-reasoning
        content phase.  Strips the ``<|open|>response<|sep|>`` prefix, holds back
        any partial marker suffix, and removes complete
        ``<|close|>response<|sep|>`` / ``<|close|>message<|sep|>`` markers.
        """
        # Strip response-open prefix
        m_open = self._response_open_re.search(text)
        if m_open is not None:
            text = text[m_open.end() :]

        # Remove complete close/message markers
        text = self._response_close_re.sub("", text)
        text = self._message_close_re.sub("", text)

        # Hold back partial markers at the end
        overlap = 0
        for marker in (
            self._response_open,
            self._response_close,
            self._message_close,
        ):
            max_check = min(len(marker) - 1, len(text))
            for n in range(max_check, 0, -1):
                if text.endswith(marker[:n]):
                    overlap = max(overlap, n)
                    break
        return text[:-overlap] if overlap else text

    def strip_content_streaming(
        self,
        previous_text: str,
        current_text: str,
    ) -> DeltaMessage | None:
        """Strip XTML content wrappers from streaming deltas after reasoning.

        Called by KimiK3Parser when no tool parser is configured, so the
        reasoning parser handles ``<|open|>response<|sep|>`` /
        ``<|close|>response<|sep|>`` / ``<|close|>message<|sep|>`` stripping itself.

        Works from accumulated text (``previous_text`` / ``current_text``
        already contain only post-reasoning content).
        """
        current_safe = self._content_ready_to_emit(current_text)
        previous_safe = self._content_ready_to_emit(previous_text)
        if current_safe.startswith(previous_safe):
            delta = current_safe[len(previous_safe) :]
        else:
            delta = current_safe
        return DeltaMessage(content=delta) if delta else None

    def extract_reasoning_streaming(
        self,
        previous_text: str,
        current_text: str,
        delta_text: str,
        previous_token_ids: Sequence[int],
        current_token_ids: Sequence[int],
        delta_token_ids: Sequence[int],
    ) -> DeltaMessage | None:
        self._last_streaming_delta_token_ids = None
        self._last_streaming_content_token_ids = None
        if not self._thinking_enabled:
            return DeltaMessage(content=delta_text)

        # reasoning already ended -> downstream content
        if self._think_close_re.search(previous_text):
            return DeltaMessage(content=delta_text)

        # the close marker completes within this delta's accumulated text:
        # split the buffer at the close marker into reasoning vs trailing content.
        m_close = self._think_close_re.search(current_text)
        if m_close is not None:
            self._last_streaming_delta_token_ids = tuple(delta_token_ids)
            self._last_streaming_content_token_ids = self._extract_content_ids(
                list(current_token_ids)
            )
            m_open = self._think_open_re.search(current_text)
            r_start = m_open.end() if m_open is not None else 0
            reasoning = current_text[r_start : m_close.start()]
            already_sent = self._reasoning_text_ready_to_emit(previous_text)
            if reasoning.startswith(already_sent):
                reasoning_delta = reasoning[len(already_sent) :]
            else:
                reasoning_delta = reasoning
            content = current_text[m_close.end() :]
            return DeltaMessage(
                reasoning=reasoning_delta or None,
                content=content or None,
            )

        current_reasoning = self._reasoning_text_ready_to_emit(current_text)
        previous_reasoning = self._reasoning_text_ready_to_emit(previous_text)
        if current_reasoning.startswith(previous_reasoning):
            reasoning_delta = current_reasoning[len(previous_reasoning) :]
        else:
            reasoning_delta = current_reasoning
        if not reasoning_delta:
            return None
        return DeltaMessage(reasoning=reasoning_delta)

    # Backward-compatible aliases for existing unit tests and downstream users
    # that still call the pre-split method names.
    extract_reasoning_content = extract_reasoning
    extract_reasoning_content_streaming = extract_reasoning_streaming

_content_ready_to_emit(text)

Return the content prefix that is safe to stream now.

Mirrors _reasoning_text_ready_to_emit but for the post-reasoning content phase. Strips the <|open|>response<|sep|> prefix, holds back any partial marker suffix, and removes complete <|close|>response<|sep|> / <|close|>message<|sep|> markers.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def _content_ready_to_emit(self, text: str) -> str:
    """Return the content prefix that is safe to stream now.

    Mirrors ``_reasoning_text_ready_to_emit`` but for the post-reasoning
    content phase.  Strips the ``<|open|>response<|sep|>`` prefix, holds back
    any partial marker suffix, and removes complete
    ``<|close|>response<|sep|>`` / ``<|close|>message<|sep|>`` markers.
    """
    # Strip response-open prefix
    m_open = self._response_open_re.search(text)
    if m_open is not None:
        text = text[m_open.end() :]

    # Remove complete close/message markers
    text = self._response_close_re.sub("", text)
    text = self._message_close_re.sub("", text)

    # Hold back partial markers at the end
    overlap = 0
    for marker in (
        self._response_open,
        self._response_close,
        self._message_close,
    ):
        max_check = min(len(marker) - 1, len(text))
        for n in range(max_check, 0, -1):
            if text.endswith(marker[:n]):
                overlap = max(overlap, n)
                break
    return text[:-overlap] if overlap else text

_reasoning_text_ready_to_emit(text)

Return the reasoning prefix that is safe to stream now.

Work from accumulated text, not from the current delta alone. That turns split-open and split-close handling into the same prefix-diff problem.

split think-open marker.

chunks: <|open|> / think / <|sep|>reasoning current text after chunk 1: <|open|> -> emit "" current text after chunk 2: <|open|>think -> emit "" current text after chunk 3: <|open|>think<|sep|>reasoning -> emit reasoning

split think-close marker.

chunks: reasoning / <|close|> / think<|sep|>... after <|close|>, the suffix is only a partial close marker, so the sendable reasoning is still reasoning and the delta is empty. Once think<|sep|> arrives, the close branch hands the following response or tools text to the downstream parser.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def _reasoning_text_ready_to_emit(self, text: str) -> str:
    """Return the reasoning prefix that is safe to stream now.

    Work from accumulated text, not from the current delta alone. That turns
    split-open and split-close handling into the same prefix-diff problem.

    Example 1: split think-open marker.
      chunks: ``<|open|>`` / ``think`` / ``<|sep|>reasoning``
      current text after chunk 1: ``<|open|>`` -> emit ``""``
      current text after chunk 2: ``<|open|>think`` -> emit ``""``
      current text after chunk 3: ``<|open|>think<|sep|>reasoning``
      -> emit ``reasoning``

    Example 2: split think-close marker.
      chunks: ``reasoning`` / ``<|close|>`` / ``think<|sep|>...``
      after ``<|close|>``, the suffix is only a partial close marker, so the
      sendable reasoning is still ``reasoning`` and the delta is empty. Once
      ``think<|sep|>`` arrives, the close branch hands the following response
      or tools text to the downstream parser.
    """
    m_open = self._think_open_re.search(text)
    if m_open is not None:
        text = text[m_open.end() :]
    overlap = 0
    for marker in (self._think_open, self._think_close):
        max_check = min(len(marker) - 1, len(text))
        for n in range(max_check, 0, -1):
            if text.endswith(marker[:n]):
                overlap = max(overlap, n)
                break
    return text[:-overlap] if overlap else text

_strip_content_wrapper(text)

Strip <|open|>response<|sep|>…<|close|>response<|sep|> wrapper and <|close|>message<|sep|> from text.

When Kimi K3 tool parsing is active it needs the raw XTML response + tools channels. Otherwise the reasoning parser cleans up the response wrapper itself so API users do not see XTML markers.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def _strip_content_wrapper(self, text: str) -> str:
    """Strip ``<|open|>response<|sep|>…<|close|>response<|sep|>`` wrapper and
    ``<|close|>message<|sep|>`` from *text*.

    When Kimi K3 tool parsing is active it needs the raw XTML
    ``response`` + ``tools`` channels. Otherwise the reasoning parser cleans
    up the response wrapper itself so API users do not see XTML markers.
    """
    # Try structured unwrap first: extract body of response channel
    m_ro = self._response_open_re.search(text)
    m_rc = self._response_close_re.search(text, m_ro.end() if m_ro else 0)
    if m_ro is not None and m_rc is not None:
        text = text[m_ro.end() : m_rc.start()]
    elif m_ro is not None:
        # response opened but not closed (truncated)
        text = text[m_ro.end() :]
    else:
        # No response wrapper — strip stray markers if present
        text = self._response_open_re.sub("", text)
        text = self._response_close_re.sub("", text)
    text = self._message_close_re.sub("", text)
    return text

extract_reasoning(model_output, request)

Split full text into (reasoning, rest) for the non-streaming path.

Handles four shapes
  • response opener without think markers -> no think channel; all content
  • open marker present -> reasoning starts after <|open|>think<|sep|>
  • open marker absent but a close marker exists (gen-prefix consumed the open) -> reasoning starts at offset 0
  • neither marker present -> truncated reasoning after a consumed generation prefix

rest is whatever follows the close marker, fed on to the tool parser.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def extract_reasoning(
    self, model_output: str, request: "ChatCompletionRequest | ResponsesRequest"
) -> tuple[str | None, str | None]:
    """Split full text into ``(reasoning, rest)`` for the non-streaming path.

    Handles four shapes:
      * response opener without think markers -> no think channel; all content
      * open marker present       -> reasoning starts after ``<|open|>think<|sep|>``
      * open marker absent but a   close marker exists (gen-prefix consumed
        the open) -> reasoning starts at offset 0
      * neither marker present     -> truncated reasoning after a consumed
        generation prefix
    ``rest`` is whatever follows the close marker, fed on to the tool parser.
    """
    if not self._thinking_enabled:
        return None, self._content_after_reasoning(model_output, request)

    m_open = self._think_open_re.search(model_output)
    # reasoning content begins right after think-open (or at start if the
    # open marker was already consumed as a generation prefix)
    content_start = m_open.end() if m_open is not None else 0

    m_close = self._think_close_re.search(model_output, content_start)
    if (
        m_open is None
        and m_close is None
        and self._response_open_re.search(model_output) is not None
    ):
        return None, self._content_after_reasoning(model_output, request)
    if m_close is not None:
        reasoning = model_output[content_start : m_close.start()]
        rest = model_output[m_close.end() :]
        return (reasoning or None, self._content_after_reasoning(rest, request))
    # think not closed -> still reasoning, no content yet
    return (model_output[content_start:] or None, None)

is_reasoning_end_streaming(input_ids, delta_ids)

Reasoning-end check for a single decode step.

The engine calls this once per structured-output request per decode step while the request is still inside the think channel, so it only has to look at the tokens generated this step, plus len(marker) - 1 tokens of context in case a marker straddles the step boundary.

The inherited default re-runs the full-sequence is_reasoning_end, which makes the scheduler O(context) per request per step and starves the GPU on long agentic contexts.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def is_reasoning_end_streaming(
    self, input_ids: Sequence[int], delta_ids: Iterable[int]
) -> bool:
    """Reasoning-end check for a single decode step.

    The engine calls this once per structured-output request per decode step
    while the request is still inside the think channel, so it only has to
    look at the tokens generated this step, plus ``len(marker) - 1`` tokens
    of context in case a marker straddles the step boundary.

    The inherited default re-runs the full-sequence ``is_reasoning_end``,
    which makes the scheduler O(context) per request per step and starves the
    GPU on long agentic contexts.
    """
    if not self._thinking_enabled:
        return True
    delta = list(delta_ids)
    if not delta:
        return False
    carry = max(len(self._think_close_ids), len(self._think_open_ids)) - 1
    head = len(input_ids) - len(delta)
    window = list(input_ids[max(0, head - carry) : head]) + delta
    return _newest_marker(window, self._think_close_ids, self._think_open_ids) == 0

strip_content_streaming(previous_text, current_text)

Strip XTML content wrappers from streaming deltas after reasoning.

Called by KimiK3Parser when no tool parser is configured, so the reasoning parser handles <|open|>response<|sep|> / <|close|>response<|sep|> / <|close|>message<|sep|> stripping itself.

Works from accumulated text (previous_text / current_text already contain only post-reasoning content).

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def strip_content_streaming(
    self,
    previous_text: str,
    current_text: str,
) -> DeltaMessage | None:
    """Strip XTML content wrappers from streaming deltas after reasoning.

    Called by KimiK3Parser when no tool parser is configured, so the
    reasoning parser handles ``<|open|>response<|sep|>`` /
    ``<|close|>response<|sep|>`` / ``<|close|>message<|sep|>`` stripping itself.

    Works from accumulated text (``previous_text`` / ``current_text``
    already contain only post-reasoning content).
    """
    current_safe = self._content_ready_to_emit(current_text)
    previous_safe = self._content_ready_to_emit(previous_text)
    if current_safe.startswith(previous_safe):
        delta = current_safe[len(previous_safe) :]
    else:
        delta = current_safe
    return DeltaMessage(content=delta) if delta else None

_match_at(haystack, i, needle)

Whether needle occurs in haystack starting at i.

Compares element by element rather than slicing: haystack is usually a ConstantList, whose slices cost a Python __getitem__ plus a list allocation on every probe.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def _match_at(haystack: Sequence[int], i: int, needle: Sequence[int]) -> bool:
    """Whether *needle* occurs in *haystack* starting at *i*.

    Compares element by element rather than slicing: ``haystack`` is usually a
    ``ConstantList``, whose slices cost a Python ``__getitem__`` plus a list
    allocation on every probe.
    """
    return all(haystack[i + k] == needle[k] for k in range(len(needle)))

_newest_marker(haystack, a, b)

Report which of a / b occurs last in haystack.

Returns 0 if a is the newest marker, 1 if b is, -1 if neither occurs. Equivalent to comparing two _subseq_index results, but a single backward pass that stops at the first hit instead of walking to index 0 twice.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def _newest_marker(haystack: Sequence[int], a: Sequence[int], b: Sequence[int]) -> int:
    """Report which of *a* / *b* occurs last in *haystack*.

    Returns 0 if *a* is the newest marker, 1 if *b* is, -1 if neither occurs.
    Equivalent to comparing two ``_subseq_index`` results, but a single backward
    pass that stops at the first hit instead of walking to index 0 twice.
    """
    if not a or not b:
        return -1
    a0, b0 = a[0], b[0]
    for i in range(len(haystack) - min(len(a), len(b)), -1, -1):
        head = haystack[i]
        if head == a0 and i + len(a) <= len(haystack) and _match_at(haystack, i, a):
            return 0
        if head == b0 and i + len(b) <= len(haystack) and _match_at(haystack, i, b):
            return 1
    return -1

_subseq_index(haystack, needle)

Return start index of the last occurrence of needle in haystack, or -1.

Source code in vllm/reasoning/kimi_k3_reasoning_parser.py
def _subseq_index(haystack: Sequence[int], needle: Sequence[int]) -> int:
    """Return start index of the last occurrence of needle in haystack, or -1."""
    n = len(needle)
    if n == 0:
        return -1
    first = needle[0]
    for i in range(len(haystack) - n, -1, -1):
        if haystack[i] == first and _match_at(haystack, i, needle):
            return i
    return -1