diff --git a/internal/llm/vllm.go b/internal/llm/vllm.go index 1f9f950..97d2919 100644 --- a/internal/llm/vllm.go +++ b/internal/llm/vllm.go @@ -28,6 +28,12 @@ type vllmRequest struct { TopP float64 `json:"top_p"` Stop []string `json:"stop,omitempty"` Stream bool `json:"stream"` + // ChatTemplateKwargs is a vLLM extension. Qwen3-family models reason by + // default and prepend a plain-text preamble ("Here's a thinking process:") + // ahead of the answer — not a block, so it cannot be stripped after + // the fact. Every Petal pass parses a JSON object out of the completion, so + // an unsuppressed preamble fails the parse outright. + ChatTemplateKwargs map[string]any `json:"chat_template_kwargs,omitempty"` } func (c *VLLMClient) body(req CompletionRequest) vllmRequest { @@ -40,6 +46,8 @@ func (c *VLLMClient) body(req CompletionRequest) vllmRequest { TopP: req.TopP, Stop: req.Stop, Stream: req.Stream, + + ChatTemplateKwargs: map[string]any{"enable_thinking": false}, } }