AI: add enable_thinking reasoning toggle plumbed to llama.cpp
New optional SamplingOverride forwarded to llama-server as chat_template_kwargs.enable_thinking (gates Qwen3-style reasoning blocks). None leaves the template default; other backends ignore it. Wired through the agentic-insight and chat-turn request bodies/handlers. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -70,6 +70,10 @@ pub struct ChatTurnRequest {
|
||||
pub top_p: Option<f32>,
|
||||
pub top_k: Option<i32>,
|
||||
pub min_p: Option<f32>,
|
||||
/// Reasoning toggle for thinking-capable models. Forwarded to the
|
||||
/// llama.cpp backend as `chat_template_kwargs.enable_thinking`; ignored
|
||||
/// by other backends. None defers to the model/template default.
|
||||
pub enable_thinking: Option<bool>,
|
||||
pub max_iterations: Option<usize>,
|
||||
/// Per-turn system-prompt override. In append mode (default), applied
|
||||
/// ephemerally — original system message restored before persistence.
|
||||
@@ -344,6 +348,7 @@ impl InsightChatService {
|
||||
top_p: req.top_p,
|
||||
top_k: req.top_k,
|
||||
min_p: req.min_p,
|
||||
enable_thinking: req.enable_thinking,
|
||||
};
|
||||
let backend = self.generator.resolve_backend(kind, &overrides).await?;
|
||||
let model_used = backend.model().to_string();
|
||||
@@ -847,6 +852,7 @@ impl InsightChatService {
|
||||
top_p: req.top_p,
|
||||
top_k: req.top_k,
|
||||
min_p: req.min_p,
|
||||
enable_thinking: req.enable_thinking,
|
||||
};
|
||||
let backend = self.generator.resolve_backend(kind, &overrides).await?;
|
||||
let model_used = backend.model().to_string();
|
||||
@@ -1017,6 +1023,7 @@ impl InsightChatService {
|
||||
top_p: req.top_p,
|
||||
top_k: req.top_k,
|
||||
min_p: req.min_p,
|
||||
enable_thinking: req.enable_thinking,
|
||||
};
|
||||
let backend = self.generator.resolve_backend(kind, &overrides).await?;
|
||||
let model_used = backend.model().to_string();
|
||||
@@ -1425,6 +1432,7 @@ impl InsightChatService {
|
||||
top_p: req.top_p,
|
||||
top_k: req.top_k,
|
||||
min_p: req.min_p,
|
||||
enable_thinking: req.enable_thinking,
|
||||
};
|
||||
let backend = self.generator.resolve_backend(kind, &overrides).await?;
|
||||
let model_used = backend.model().to_string();
|
||||
@@ -1607,6 +1615,7 @@ impl InsightChatService {
|
||||
top_p: req.top_p,
|
||||
top_k: req.top_k,
|
||||
min_p: req.min_p,
|
||||
enable_thinking: req.enable_thinking,
|
||||
};
|
||||
let backend = self.generator.resolve_backend(kind, &overrides).await?;
|
||||
let model_used = backend.model().to_string();
|
||||
|
||||
Reference in New Issue
Block a user