From 6af43341c8c3ed4fe7102e4807b245e4d9125636 Mon Sep 17 00:00:00 2001 From: john Date: Wed, 23 Sep 2026 14:25:12 +0800 Subject: [PATCH] fix(web): slice fetched HTML on UTF-8 char boundaries Lowercasing non-ASCII text can change byte length, so offsets found in the lowercased copy panicked when reused on the original (seen on 103 in parse_ddg_results / strip_html). Use ASCII-only lowercasing, clamp slice ends to char boundaries, and stop parsing on truncated result tags. Co-authored-by: Cursor --- .../src/agents/platform_extensions/web.rs | 123 ++++++++++++++---- 1 file changed, 98 insertions(+), 25 deletions(-) diff --git a/crates/goose/src/agents/platform_extensions/web.rs b/crates/goose/src/agents/platform_extensions/web.rs index 8b8f38787..926a621d2 100644 --- a/crates/goose/src/agents/platform_extensions/web.rs +++ b/crates/goose/src/agents/platform_extensions/web.rs @@ -60,32 +60,45 @@ impl WebClient { }) } + fn floor_char_boundary(s: &str, index: usize) -> usize { + if index >= s.len() { + return s.len(); + } + let mut i = index; + while !s.is_char_boundary(i) { + i -= 1; + } + i + } + fn truncate(s: String) -> String { if s.len() > MAX_OUTPUT_CHARS { - format!("{}\n[output truncated]", &s[..MAX_OUTPUT_CHARS]) + let end = Self::floor_char_boundary(&s, MAX_OUTPUT_CHARS); + format!("{}\n[output truncated]", &s[..end]) } else { s } } + // Byte offsets found in the lowercased copy are reused to slice the original, + // so lowercasing must preserve byte length: ASCII-only. + fn remove_blocks(mut text: String, open: &str, close: &str) -> String { + loop { + let lower = text.to_ascii_lowercase(); + let Some(start) = lower.find(open) else { + break; + }; + let Some(end) = lower[start..].find(close) else { + break; + }; + text.replace_range(start..start + end + close.len(), ""); + } + text + } + fn strip_html(html: &str) -> String { - let mut text = html.to_string(); - - while let Some(start) = text.to_lowercase().find("") { - text = format!("{}{}", &text[..start], &text[start + end + 9..]); - } else { - break; - } - } - - while let Some(start) = text.to_lowercase().find("") { - text = format!("{}{}", &text[..start], &text[start + end + 8..]); - } else { - break; - } - } + let text = Self::remove_blocks(html.to_string(), ""); + let text = Self::remove_blocks(text, ""); let mut result = String::new(); let mut in_tag = false; @@ -183,7 +196,7 @@ impl WebClient { fn parse_ddg_results(html: &str, max: usize) -> String { let mut results = Vec::new(); - let lower = html.to_lowercase(); + let lower = html.to_ascii_lowercase(); let mut pos = 0; while results.len() < max { @@ -194,15 +207,15 @@ impl WebClient { let before = &html[..link_start + search_str.len()]; let a_start = before.rfind('<').unwrap_or(0); - let a_block = &html[a_start - ..link_start - + search_str.len() - + 100.min(html.len() - link_start - search_str.len())]; + let a_end = Self::floor_char_boundary(html, link_start + search_str.len() + 100); + let a_block = &html[a_start..a_end]; let href = Self::extract_attr(a_block, "href").unwrap_or_default(); let after_a_tag_end = link_start + search_str.len(); - let close_offset = html[after_a_tag_end..].find('>').unwrap_or(0); + let Some(close_offset) = html[after_a_tag_end..].find('>') else { + break; + }; let content_start = after_a_tag_end + close_offset + 1; let close_a = html[content_start..].find("").unwrap_or(0); let title = Self::strip_html(&html[content_start..content_start + close_a]); @@ -247,7 +260,7 @@ impl WebClient { fn extract_attr(tag: &str, attr: &str) -> Option { let search = format!("{attr}=\""); - let lower = tag.to_lowercase(); + let lower = tag.to_ascii_lowercase(); let start = lower.find(&search)? + search.len(); let end = tag[start..].find('"')?; Some(tag[start..start + end].to_string()) @@ -329,3 +342,63 @@ impl McpClientTrait for WebClient { Some(&self.info) } } + +#[cfg(test)] +mod tests { + use super::WebClient; + use super::MAX_OUTPUT_CHARS; + + #[test] + fn truncate_does_not_split_multibyte_chars() { + let s = format!("a{}", "中".repeat(MAX_OUTPUT_CHARS)); + let out = WebClient::truncate(s); + assert!(out.ends_with("\n[output truncated]")); + let body = out.trim_end_matches("\n[output truncated]"); + assert!(body.len() <= MAX_OUTPUT_CHARS); + assert!(body.chars().skip(1).all(|c| c == '中')); + } + + #[test] + fn truncate_keeps_short_input() { + assert_eq!(WebClient::truncate("短文本".to_string()), "短文本"); + } + + #[test] + fn strip_html_handles_chars_that_change_length_when_lowercased() { + let html = + "

İstanbul Ⱥ 中文

尾部

"; + let text = WebClient::strip_html(html); + assert_eq!(text, "İstanbul Ⱥ 中文尾部"); + } + + #[test] + fn strip_html_keeps_text_when_block_is_unclosed() { + assert_eq!(WebClient::strip_html("

前