diff --git a/crates/goose/src/agents/platform_extensions/web.rs b/crates/goose/src/agents/platform_extensions/web.rs index 8b8f38787..926a621d2 100644 --- a/crates/goose/src/agents/platform_extensions/web.rs +++ b/crates/goose/src/agents/platform_extensions/web.rs @@ -60,32 +60,45 @@ impl WebClient { }) } + fn floor_char_boundary(s: &str, index: usize) -> usize { + if index >= s.len() { + return s.len(); + } + let mut i = index; + while !s.is_char_boundary(i) { + i -= 1; + } + i + } + fn truncate(s: String) -> String { if s.len() > MAX_OUTPUT_CHARS { - format!("{}\n[output truncated]", &s[..MAX_OUTPUT_CHARS]) + let end = Self::floor_char_boundary(&s, MAX_OUTPUT_CHARS); + format!("{}\n[output truncated]", &s[..end]) } else { s } } + // Byte offsets found in the lowercased copy are reused to slice the original, + // so lowercasing must preserve byte length: ASCII-only. + fn remove_blocks(mut text: String, open: &str, close: &str) -> String { + loop { + let lower = text.to_ascii_lowercase(); + let Some(start) = lower.find(open) else { + break; + }; + let Some(end) = lower[start..].find(close) else { + break; + }; + text.replace_range(start..start + end + close.len(), ""); + } + text + } + fn strip_html(html: &str) -> String { - let mut text = html.to_string(); - - while let Some(start) = text.to_lowercase().find(""); + let text = Self::remove_blocks(text, "
尾部
"; + let text = WebClient::strip_html(html); + assert_eq!(text, "İstanbul Ⱥ 中文尾部"); + } + + #[test] + fn strip_html_keeps_text_when_block_is_unclosed() { + assert_eq!(WebClient::strip_html("前