fix(web): slice fetched HTML on UTF-8 char boundaries

Lowercasing non-ASCII text can change byte length, so offsets found in the
lowercased copy panicked when reused on the original (seen on 103 in
parse_ddg_results / strip_html). Use ASCII-only lowercasing, clamp slice
ends to char boundaries, and stop parsing on truncated result tags.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
john
2026-09-23 14:25:12 +08:00
parent e622befc30
commit 6af43341c8
@@ -60,32 +60,45 @@ impl WebClient {
})
}
fn floor_char_boundary(s: &str, index: usize) -> usize {
if index >= s.len() {
return s.len();
}
let mut i = index;
while !s.is_char_boundary(i) {
i -= 1;
}
i
}
fn truncate(s: String) -> String {
if s.len() > MAX_OUTPUT_CHARS {
format!("{}\n[output truncated]", &s[..MAX_OUTPUT_CHARS])
let end = Self::floor_char_boundary(&s, MAX_OUTPUT_CHARS);
format!("{}\n[output truncated]", &s[..end])
} else {
s
}
}
// Byte offsets found in the lowercased copy are reused to slice the original,
// so lowercasing must preserve byte length: ASCII-only.
fn remove_blocks(mut text: String, open: &str, close: &str) -> String {
loop {
let lower = text.to_ascii_lowercase();
let Some(start) = lower.find(open) else {
break;
};
let Some(end) = lower[start..].find(close) else {
break;
};
text.replace_range(start..start + end + close.len(), "");
}
text
}
fn strip_html(html: &str) -> String {
let mut text = html.to_string();
while let Some(start) = text.to_lowercase().find("<script") {
if let Some(end) = text.to_lowercase()[start..].find("</script>") {
text = format!("{}{}", &text[..start], &text[start + end + 9..]);
} else {
break;
}
}
while let Some(start) = text.to_lowercase().find("<style") {
if let Some(end) = text.to_lowercase()[start..].find("</style>") {
text = format!("{}{}", &text[..start], &text[start + end + 8..]);
} else {
break;
}
}
let text = Self::remove_blocks(html.to_string(), "<script", "</script>");
let text = Self::remove_blocks(text, "<style", "</style>");
let mut result = String::new();
let mut in_tag = false;
@@ -183,7 +196,7 @@ impl WebClient {
fn parse_ddg_results(html: &str, max: usize) -> String {
let mut results = Vec::new();
let lower = html.to_lowercase();
let lower = html.to_ascii_lowercase();
let mut pos = 0;
while results.len() < max {
@@ -194,15 +207,15 @@ impl WebClient {
let before = &html[..link_start + search_str.len()];
let a_start = before.rfind('<').unwrap_or(0);
let a_block = &html[a_start
..link_start
+ search_str.len()
+ 100.min(html.len() - link_start - search_str.len())];
let a_end = Self::floor_char_boundary(html, link_start + search_str.len() + 100);
let a_block = &html[a_start..a_end];
let href = Self::extract_attr(a_block, "href").unwrap_or_default();
let after_a_tag_end = link_start + search_str.len();
let close_offset = html[after_a_tag_end..].find('>').unwrap_or(0);
let Some(close_offset) = html[after_a_tag_end..].find('>') else {
break;
};
let content_start = after_a_tag_end + close_offset + 1;
let close_a = html[content_start..].find("</a>").unwrap_or(0);
let title = Self::strip_html(&html[content_start..content_start + close_a]);
@@ -247,7 +260,7 @@ impl WebClient {
fn extract_attr(tag: &str, attr: &str) -> Option<String> {
let search = format!("{attr}=\"");
let lower = tag.to_lowercase();
let lower = tag.to_ascii_lowercase();
let start = lower.find(&search)? + search.len();
let end = tag[start..].find('"')?;
Some(tag[start..start + end].to_string())
@@ -329,3 +342,63 @@ impl McpClientTrait for WebClient {
Some(&self.info)
}
}
#[cfg(test)]
mod tests {
use super::WebClient;
use super::MAX_OUTPUT_CHARS;
#[test]
fn truncate_does_not_split_multibyte_chars() {
let s = format!("a{}", "中".repeat(MAX_OUTPUT_CHARS));
let out = WebClient::truncate(s);
assert!(out.ends_with("\n[output truncated]"));
let body = out.trim_end_matches("\n[output truncated]");
assert!(body.len() <= MAX_OUTPUT_CHARS);
assert!(body.chars().skip(1).all(|c| c == '中'));
}
#[test]
fn truncate_keeps_short_input() {
assert_eq!(WebClient::truncate("短文本".to_string()), "短文本");
}
#[test]
fn strip_html_handles_chars_that_change_length_when_lowercased() {
let html =
"<p>İstanbul Ⱥ 中文</p><SCRIPT>var x = 'İ';</SCRIPT><Style>.a{}</Style><p>尾部</p>";
let text = WebClient::strip_html(html);
assert_eq!(text, "İstanbul Ⱥ 中文尾部");
}
#[test]
fn strip_html_keeps_text_when_block_is_unclosed() {
assert_eq!(WebClient::strip_html("<p>前</p><script>x"), "前x");
}
#[test]
fn parse_ddg_results_handles_multibyte_near_link() {
let html = format!(
"<div>İİİ<a class=\"result__a\" href=\"https://example.com/中\">{}标题</a>\
<a class=\"result__snippet\">摘要 İ</a></div>",
"长".repeat(60)
);
let out = WebClient::parse_ddg_results(&html, 5);
assert!(out.starts_with("1. "));
assert!(out.contains("标题"));
assert!(out.contains("https://example.com/中"));
assert!(out.contains("摘要 İ"));
}
#[test]
fn parse_ddg_results_handles_truncated_link_tag() {
assert_eq!(
WebClient::parse_ddg_results("<a class=\"result__a\"", 5),
"No results found."
);
assert_eq!(
WebClient::parse_ddg_results("<a class=\"result__a\"中文", 5),
"No results found."
);
}
}