From 61b3c6cd623fde1dcd6487c2e0d760ecf1e0a435 Mon Sep 17 00:00:00 2001 From: prosolis <5590409+prosolis@users.noreply.github.com> Date: Sun, 26 Jul 2026 21:25:56 -0700 Subject: [PATCH] Disable thinking on the vLLM backend MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Qwen3-family models reason by default and prepend a plain-text preamble ahead of the answer — not a block, so it cannot be stripped after the fact. Every Petal pass parses a JSON object out of the completion, so an unsuppressed preamble fails the parse outright. Send chat_template_kwargs {"enable_thinking": false} on every request, matching the unconditional think:false the Ollama backend already sends. --- internal/llm/vllm.go | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/internal/llm/vllm.go b/internal/llm/vllm.go index 1f9f950..97d2919 100644 --- a/internal/llm/vllm.go +++ b/internal/llm/vllm.go @@ -28,6 +28,12 @@ type vllmRequest struct { TopP float64 `json:"top_p"` Stop []string `json:"stop,omitempty"` Stream bool `json:"stream"` + // ChatTemplateKwargs is a vLLM extension. Qwen3-family models reason by + // default and prepend a plain-text preamble ("Here's a thinking process:") + // ahead of the answer — not a block, so it cannot be stripped after + // the fact. Every Petal pass parses a JSON object out of the completion, so + // an unsuppressed preamble fails the parse outright. + ChatTemplateKwargs map[string]any `json:"chat_template_kwargs,omitempty"` } func (c *VLLMClient) body(req CompletionRequest) vllmRequest { @@ -40,6 +46,8 @@ func (c *VLLMClient) body(req CompletionRequest) vllmRequest { TopP: req.TopP, Stop: req.Stop, Stream: req.Stream, + + ChatTemplateKwargs: map[string]any{"enable_thinking": false}, } }