fix: handle reasoning_content for Kimi/thinking models (#7252)
Signed-off-by: clayarnoldg2m <carnold@g2m.ai> Co-authored-by: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -1346,8 +1346,9 @@ impl Agent {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Preserve thinking content from the original response
|
// Preserve thinking/reasoning content from the original response
|
||||||
// Gemini (and other thinking models) require thinking to be echoed back
|
// Gemini (and other thinking models) require thinking to be echoed back
|
||||||
|
// Kimi/DeepSeek require reasoning_content on assistant tool call messages
|
||||||
let thinking_content: Vec<MessageContent> = response.content.iter()
|
let thinking_content: Vec<MessageContent> = response.content.iter()
|
||||||
.filter(|c| matches!(c, MessageContent::Thinking(_)))
|
.filter(|c| matches!(c, MessageContent::Thinking(_)))
|
||||||
.cloned()
|
.cloned()
|
||||||
@@ -1361,10 +1362,25 @@ impl Agent {
|
|||||||
messages_to_add.push(thinking_msg);
|
messages_to_add.push(thinking_msg);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Collect reasoning content to attach to tool request messages
|
||||||
|
let reasoning_content: Vec<MessageContent> = response.content.iter()
|
||||||
|
.filter(|c| matches!(c, MessageContent::Reasoning(_)))
|
||||||
|
.cloned()
|
||||||
|
.collect();
|
||||||
|
|
||||||
for (idx, request) in frontend_requests.iter().chain(remaining_requests.iter()).enumerate() {
|
for (idx, request) in frontend_requests.iter().chain(remaining_requests.iter()).enumerate() {
|
||||||
if request.tool_call.is_ok() {
|
if request.tool_call.is_ok() {
|
||||||
let request_msg = Message::assistant()
|
let mut request_msg = Message::assistant()
|
||||||
.with_id(format!("msg_{}", Uuid::new_v4()))
|
.with_id(format!("msg_{}", Uuid::new_v4()));
|
||||||
|
|
||||||
|
// Attach reasoning content to EVERY split tool request message.
|
||||||
|
// Providers like Kimi require reasoning_content on all assistant
|
||||||
|
// messages with tool_calls when thinking mode is enabled.
|
||||||
|
for rc in &reasoning_content {
|
||||||
|
request_msg = request_msg.with_content(rc.clone());
|
||||||
|
}
|
||||||
|
|
||||||
|
request_msg = request_msg
|
||||||
.with_tool_request_with_metadata(
|
.with_tool_request_with_metadata(
|
||||||
request.id.clone(),
|
request.id.clone(),
|
||||||
request.tool_call.clone(),
|
request.tool_call.clone(),
|
||||||
|
|||||||
@@ -290,8 +290,17 @@ impl AppsManagerClient {
|
|||||||
let messages = vec![Message::user().with_text(&user_prompt)];
|
let messages = vec![Message::user().with_text(&user_prompt)];
|
||||||
let tools = vec![Self::create_app_content_tool()];
|
let tools = vec![Self::create_app_content_tool()];
|
||||||
|
|
||||||
|
let mut model_config = provider.get_model_config();
|
||||||
|
model_config.max_tokens = Some(16384);
|
||||||
|
|
||||||
let (response, _usage) = provider
|
let (response, _usage) = provider
|
||||||
.complete(session_id, &system_prompt, &messages, &tools)
|
.complete_with_model(
|
||||||
|
Some(session_id),
|
||||||
|
&model_config,
|
||||||
|
&system_prompt,
|
||||||
|
&messages,
|
||||||
|
&tools,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("LLM call failed: {}", e))?;
|
.map_err(|e| format!("LLM call failed: {}", e))?;
|
||||||
|
|
||||||
@@ -321,8 +330,17 @@ impl AppsManagerClient {
|
|||||||
let messages = vec![Message::user().with_text(&user_prompt)];
|
let messages = vec![Message::user().with_text(&user_prompt)];
|
||||||
let tools = vec![Self::update_app_content_tool()];
|
let tools = vec![Self::update_app_content_tool()];
|
||||||
|
|
||||||
|
let mut model_config = provider.get_model_config();
|
||||||
|
model_config.max_tokens = Some(16384);
|
||||||
|
|
||||||
let (response, _usage) = provider
|
let (response, _usage) = provider
|
||||||
.complete(session_id, &system_prompt, &messages, &tools)
|
.complete_with_model(
|
||||||
|
Some(session_id),
|
||||||
|
&model_config,
|
||||||
|
&system_prompt,
|
||||||
|
&messages,
|
||||||
|
&tools,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("LLM call failed: {}", e))?;
|
.map_err(|e| format!("LLM call failed: {}", e))?;
|
||||||
|
|
||||||
|
|||||||
@@ -82,7 +82,7 @@ pub fn format_messages(messages: &[Message], image_format: &ImageFormat) -> Vec<
|
|||||||
let mut output = Vec::new();
|
let mut output = Vec::new();
|
||||||
let mut content_array = Vec::new();
|
let mut content_array = Vec::new();
|
||||||
let mut text_array = Vec::new();
|
let mut text_array = Vec::new();
|
||||||
let mut reasoning_text: Option<String> = None;
|
let mut reasoning_text = String::new();
|
||||||
|
|
||||||
for content in &message.content {
|
for content in &message.content {
|
||||||
match content {
|
match content {
|
||||||
@@ -116,7 +116,7 @@ pub fn format_messages(messages: &[Message], image_format: &ImageFormat) -> Vec<
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
MessageContent::Reasoning(r) => {
|
MessageContent::Reasoning(r) => {
|
||||||
reasoning_text = Some(r.text.clone());
|
reasoning_text.push_str(&r.text);
|
||||||
}
|
}
|
||||||
MessageContent::ToolRequest(request) => match &request.tool_call {
|
MessageContent::ToolRequest(request) => match &request.tool_call {
|
||||||
Ok(tool_call) => {
|
Ok(tool_call) => {
|
||||||
@@ -278,15 +278,11 @@ pub fn format_messages(messages: &[Message], image_format: &ImageFormat) -> Vec<
|
|||||||
converted["content"] = json!(null);
|
converted["content"] = json!(null);
|
||||||
}
|
}
|
||||||
|
|
||||||
// DeepSeek requires reasoning_content field when tool_calls are present
|
// Include reasoning_content only when non-empty.
|
||||||
// Set it to the captured reasoning text, or empty string if not present
|
// Kimi rejects empty reasoning_content (""), so we must omit it entirely
|
||||||
if converted.get("tool_calls").is_some() {
|
// when there's no reasoning to send.
|
||||||
let reasoning = reasoning_text.unwrap_or_default();
|
if !reasoning_text.is_empty() {
|
||||||
converted["reasoning_content"] = json!(reasoning);
|
converted["reasoning_content"] = json!(reasoning_text);
|
||||||
} else if let Some(reasoning) = reasoning_text {
|
|
||||||
if !reasoning.is_empty() {
|
|
||||||
converted["reasoning_content"] = json!(reasoning);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if converted.get("content").is_some() || converted.get("tool_calls").is_some() {
|
if converted.get("content").is_some() || converted.get("tool_calls").is_some() {
|
||||||
@@ -542,6 +538,7 @@ where
|
|||||||
use futures::StreamExt;
|
use futures::StreamExt;
|
||||||
|
|
||||||
let mut accumulated_reasoning: Vec<Value> = Vec::new();
|
let mut accumulated_reasoning: Vec<Value> = Vec::new();
|
||||||
|
let mut accumulated_reasoning_content = String::new();
|
||||||
|
|
||||||
'outer: while let Some(response) = stream.next().await {
|
'outer: while let Some(response) = stream.next().await {
|
||||||
if response.as_ref().is_ok_and(|s| s == "data: [DONE]") {
|
if response.as_ref().is_ok_and(|s| s == "data: [DONE]") {
|
||||||
@@ -562,6 +559,9 @@ where
|
|||||||
if let Some(details) = &chunk.choices[0].delta.reasoning_details {
|
if let Some(details) = &chunk.choices[0].delta.reasoning_details {
|
||||||
accumulated_reasoning.extend(details.iter().cloned());
|
accumulated_reasoning.extend(details.iter().cloned());
|
||||||
}
|
}
|
||||||
|
if let Some(rc) = &chunk.choices[0].delta.reasoning_content {
|
||||||
|
accumulated_reasoning_content.push_str(rc);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut usage = extract_usage_with_output_tokens(&chunk);
|
let mut usage = extract_usage_with_output_tokens(&chunk);
|
||||||
@@ -602,6 +602,9 @@ where
|
|||||||
if let Some(details) = &tool_chunk.choices[0].delta.reasoning_details {
|
if let Some(details) = &tool_chunk.choices[0].delta.reasoning_details {
|
||||||
accumulated_reasoning.extend(details.iter().cloned());
|
accumulated_reasoning.extend(details.iter().cloned());
|
||||||
}
|
}
|
||||||
|
if let Some(rc) = &tool_chunk.choices[0].delta.reasoning_content {
|
||||||
|
accumulated_reasoning_content.push_str(rc);
|
||||||
|
}
|
||||||
if let Some(delta_tool_calls) = &tool_chunk.choices[0].delta.tool_calls {
|
if let Some(delta_tool_calls) = &tool_chunk.choices[0].delta.tool_calls {
|
||||||
for delta_call in delta_tool_calls {
|
for delta_call in delta_tool_calls {
|
||||||
if let Some(index) = delta_call.index {
|
if let Some(index) = delta_call.index {
|
||||||
@@ -642,6 +645,10 @@ where
|
|||||||
};
|
};
|
||||||
|
|
||||||
let mut contents = Vec::new();
|
let mut contents = Vec::new();
|
||||||
|
if !accumulated_reasoning_content.is_empty() {
|
||||||
|
contents.push(MessageContent::reasoning(&accumulated_reasoning_content));
|
||||||
|
accumulated_reasoning_content.clear();
|
||||||
|
}
|
||||||
let mut sorted_indices: Vec<_> = tool_call_data.keys().cloned().collect();
|
let mut sorted_indices: Vec<_> = tool_call_data.keys().cloned().collect();
|
||||||
sorted_indices.sort();
|
sorted_indices.sort();
|
||||||
|
|
||||||
|
|||||||
@@ -221,10 +221,18 @@ impl GeminiCliProvider {
|
|||||||
|
|
||||||
fn parse_stream_json_response(events: &[Value]) -> Result<(Message, Usage), ProviderError> {
|
fn parse_stream_json_response(events: &[Value]) -> Result<(Message, Usage), ProviderError> {
|
||||||
let mut all_text_content = Vec::new();
|
let mut all_text_content = Vec::new();
|
||||||
|
let mut all_thinking_content = Vec::new();
|
||||||
let mut usage = Usage::default();
|
let mut usage = Usage::default();
|
||||||
|
|
||||||
for parsed in events {
|
for parsed in events {
|
||||||
match parsed.get("type").and_then(|t| t.as_str()) {
|
match parsed.get("type").and_then(|t| t.as_str()) {
|
||||||
|
Some("thinking") => {
|
||||||
|
if let Some(content) = parsed.get("content").and_then(|c| c.as_str()) {
|
||||||
|
if !content.is_empty() {
|
||||||
|
all_thinking_content.push(content.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Some("message") => {
|
Some("message") => {
|
||||||
if parsed.get("role").and_then(|r| r.as_str()) == Some("assistant") {
|
if parsed.get("role").and_then(|r| r.as_str()) == Some("assistant") {
|
||||||
if let Some(content) = parsed.get("content").and_then(|c| c.as_str()) {
|
if let Some(content) = parsed.get("content").and_then(|c| c.as_str()) {
|
||||||
@@ -253,11 +261,16 @@ impl GeminiCliProvider {
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
let message = Message::new(
|
let mut content = Vec::new();
|
||||||
Role::Assistant,
|
|
||||||
chrono::Utc::now().timestamp(),
|
let combined_thinking = all_thinking_content.join("");
|
||||||
vec![MessageContent::text(combined_text)],
|
if !combined_thinking.is_empty() {
|
||||||
);
|
content.push(MessageContent::thinking(combined_thinking, String::new()));
|
||||||
|
}
|
||||||
|
|
||||||
|
content.push(MessageContent::text(combined_text));
|
||||||
|
|
||||||
|
let message = Message::new(Role::Assistant, chrono::Utc::now().timestamp(), content);
|
||||||
|
|
||||||
Ok((message, usage))
|
Ok((message, usage))
|
||||||
}
|
}
|
||||||
@@ -395,6 +408,44 @@ mod tests {
|
|||||||
assert!(GeminiCliProvider::parse_stream_json_response(&empty).is_err());
|
assert!(GeminiCliProvider::parse_stream_json_response(&empty).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_parse_thinking_blocks() {
|
||||||
|
let events = vec![
|
||||||
|
json!({"type":"init","session_id":"abc","model":"gemini-2.5-pro"}),
|
||||||
|
json!({"type":"thinking","content":"Let me reason about this...","delta":true}),
|
||||||
|
json!({"type":"thinking","content":" Step 1: analyze the problem.","delta":true}),
|
||||||
|
json!({"type":"message","role":"assistant","content":"Here is the answer.","delta":true}),
|
||||||
|
json!({"type":"result","status":"success","stats":{"input_tokens":30,"output_tokens":15,"total_tokens":45}}),
|
||||||
|
];
|
||||||
|
let (message, usage) = GeminiCliProvider::parse_stream_json_response(&events).unwrap();
|
||||||
|
assert_eq!(message.role, Role::Assistant);
|
||||||
|
|
||||||
|
// Should have thinking content followed by text content
|
||||||
|
assert_eq!(message.content.len(), 2);
|
||||||
|
let thinking = message.content[0]
|
||||||
|
.as_thinking()
|
||||||
|
.expect("first content should be thinking");
|
||||||
|
assert_eq!(
|
||||||
|
thinking.thinking,
|
||||||
|
"Let me reason about this... Step 1: analyze the problem."
|
||||||
|
);
|
||||||
|
assert_eq!(message.as_concat_text(), "Here is the answer.");
|
||||||
|
assert_eq!(usage.input_tokens, Some(30));
|
||||||
|
assert_eq!(usage.output_tokens, Some(15));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_parse_no_thinking_blocks() {
|
||||||
|
// When there's no thinking, message should only have text content
|
||||||
|
let events = vec![
|
||||||
|
json!({"type":"message","role":"assistant","content":"Direct answer.","delta":true}),
|
||||||
|
json!({"type":"result","status":"success","stats":{"input_tokens":10,"output_tokens":5,"total_tokens":15}}),
|
||||||
|
];
|
||||||
|
let (message, _usage) = GeminiCliProvider::parse_stream_json_response(&events).unwrap();
|
||||||
|
assert_eq!(message.content.len(), 1);
|
||||||
|
assert_eq!(message.as_concat_text(), "Direct answer.");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_build_prompt_first_and_resume() {
|
fn test_build_prompt_first_and_resume() {
|
||||||
let provider = make_provider();
|
let provider = make_provider();
|
||||||
|
|||||||
@@ -179,15 +179,17 @@ impl ProviderTester {
|
|||||||
.complete(session_id, "You are a helpful assistant.", &[message], &[])
|
.complete(session_id, "You are a helpful assistant.", &[message], &[])
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
assert_eq!(
|
assert!(
|
||||||
response.content.len(),
|
!response.content.is_empty(),
|
||||||
1,
|
"Expected at least one content item in response"
|
||||||
"Expected single content item in response"
|
|
||||||
);
|
);
|
||||||
|
|
||||||
assert!(
|
assert!(
|
||||||
matches!(response.content[0], MessageContent::Text(_)),
|
response
|
||||||
"Expected text response"
|
.content
|
||||||
|
.iter()
|
||||||
|
.any(|c| matches!(c, MessageContent::Text(_))),
|
||||||
|
"Expected at least one text content item in response"
|
||||||
);
|
);
|
||||||
|
|
||||||
println!(
|
println!(
|
||||||
|
|||||||
@@ -105,8 +105,15 @@ export default function ProviderConfigurationModal({
|
|||||||
|
|
||||||
const toSubmit = Object.fromEntries(
|
const toSubmit = Object.fromEntries(
|
||||||
Object.entries(configValues)
|
Object.entries(configValues)
|
||||||
.filter(([_k, entry]) => !!entry.value)
|
.filter(
|
||||||
.map(([k, entry]) => [k, entry.value || ''])
|
([_k, entry]) =>
|
||||||
|
!!entry.value ||
|
||||||
|
(entry.serverValue != null && typeof entry.serverValue === 'string')
|
||||||
|
)
|
||||||
|
.map(([k, entry]) => [
|
||||||
|
k,
|
||||||
|
entry.value ?? (typeof entry.serverValue === 'string' ? entry.serverValue : ''),
|
||||||
|
])
|
||||||
);
|
);
|
||||||
|
|
||||||
try {
|
try {
|
||||||
|
|||||||
Reference in New Issue
Block a user