fix(web): slice fetched HTML on UTF-8 char boundaries
Lowercasing non-ASCII text can change byte length, so offsets found in the lowercased copy panicked when reused on the original (seen on 103 in parse_ddg_results / strip_html). Use ASCII-only lowercasing, clamp slice ends to char boundaries, and stop parsing on truncated result tags. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -60,32 +60,45 @@ impl WebClient {
|
||||
})
|
||||
}
|
||||
|
||||
fn floor_char_boundary(s: &str, index: usize) -> usize {
|
||||
if index >= s.len() {
|
||||
return s.len();
|
||||
}
|
||||
let mut i = index;
|
||||
while !s.is_char_boundary(i) {
|
||||
i -= 1;
|
||||
}
|
||||
i
|
||||
}
|
||||
|
||||
fn truncate(s: String) -> String {
|
||||
if s.len() > MAX_OUTPUT_CHARS {
|
||||
format!("{}\n[output truncated]", &s[..MAX_OUTPUT_CHARS])
|
||||
let end = Self::floor_char_boundary(&s, MAX_OUTPUT_CHARS);
|
||||
format!("{}\n[output truncated]", &s[..end])
|
||||
} else {
|
||||
s
|
||||
}
|
||||
}
|
||||
|
||||
// Byte offsets found in the lowercased copy are reused to slice the original,
|
||||
// so lowercasing must preserve byte length: ASCII-only.
|
||||
fn remove_blocks(mut text: String, open: &str, close: &str) -> String {
|
||||
loop {
|
||||
let lower = text.to_ascii_lowercase();
|
||||
let Some(start) = lower.find(open) else {
|
||||
break;
|
||||
};
|
||||
let Some(end) = lower[start..].find(close) else {
|
||||
break;
|
||||
};
|
||||
text.replace_range(start..start + end + close.len(), "");
|
||||
}
|
||||
text
|
||||
}
|
||||
|
||||
fn strip_html(html: &str) -> String {
|
||||
let mut text = html.to_string();
|
||||
|
||||
while let Some(start) = text.to_lowercase().find("<script") {
|
||||
if let Some(end) = text.to_lowercase()[start..].find("</script>") {
|
||||
text = format!("{}{}", &text[..start], &text[start + end + 9..]);
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
while let Some(start) = text.to_lowercase().find("<style") {
|
||||
if let Some(end) = text.to_lowercase()[start..].find("</style>") {
|
||||
text = format!("{}{}", &text[..start], &text[start + end + 8..]);
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
let text = Self::remove_blocks(html.to_string(), "<script", "</script>");
|
||||
let text = Self::remove_blocks(text, "<style", "</style>");
|
||||
|
||||
let mut result = String::new();
|
||||
let mut in_tag = false;
|
||||
@@ -183,7 +196,7 @@ impl WebClient {
|
||||
|
||||
fn parse_ddg_results(html: &str, max: usize) -> String {
|
||||
let mut results = Vec::new();
|
||||
let lower = html.to_lowercase();
|
||||
let lower = html.to_ascii_lowercase();
|
||||
|
||||
let mut pos = 0;
|
||||
while results.len() < max {
|
||||
@@ -194,15 +207,15 @@ impl WebClient {
|
||||
|
||||
let before = &html[..link_start + search_str.len()];
|
||||
let a_start = before.rfind('<').unwrap_or(0);
|
||||
let a_block = &html[a_start
|
||||
..link_start
|
||||
+ search_str.len()
|
||||
+ 100.min(html.len() - link_start - search_str.len())];
|
||||
let a_end = Self::floor_char_boundary(html, link_start + search_str.len() + 100);
|
||||
let a_block = &html[a_start..a_end];
|
||||
|
||||
let href = Self::extract_attr(a_block, "href").unwrap_or_default();
|
||||
|
||||
let after_a_tag_end = link_start + search_str.len();
|
||||
let close_offset = html[after_a_tag_end..].find('>').unwrap_or(0);
|
||||
let Some(close_offset) = html[after_a_tag_end..].find('>') else {
|
||||
break;
|
||||
};
|
||||
let content_start = after_a_tag_end + close_offset + 1;
|
||||
let close_a = html[content_start..].find("</a>").unwrap_or(0);
|
||||
let title = Self::strip_html(&html[content_start..content_start + close_a]);
|
||||
@@ -247,7 +260,7 @@ impl WebClient {
|
||||
|
||||
fn extract_attr(tag: &str, attr: &str) -> Option<String> {
|
||||
let search = format!("{attr}=\"");
|
||||
let lower = tag.to_lowercase();
|
||||
let lower = tag.to_ascii_lowercase();
|
||||
let start = lower.find(&search)? + search.len();
|
||||
let end = tag[start..].find('"')?;
|
||||
Some(tag[start..start + end].to_string())
|
||||
@@ -329,3 +342,63 @@ impl McpClientTrait for WebClient {
|
||||
Some(&self.info)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::WebClient;
|
||||
use super::MAX_OUTPUT_CHARS;
|
||||
|
||||
#[test]
|
||||
fn truncate_does_not_split_multibyte_chars() {
|
||||
let s = format!("a{}", "中".repeat(MAX_OUTPUT_CHARS));
|
||||
let out = WebClient::truncate(s);
|
||||
assert!(out.ends_with("\n[output truncated]"));
|
||||
let body = out.trim_end_matches("\n[output truncated]");
|
||||
assert!(body.len() <= MAX_OUTPUT_CHARS);
|
||||
assert!(body.chars().skip(1).all(|c| c == '中'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncate_keeps_short_input() {
|
||||
assert_eq!(WebClient::truncate("短文本".to_string()), "短文本");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_html_handles_chars_that_change_length_when_lowercased() {
|
||||
let html =
|
||||
"<p>İstanbul Ⱥ 中文</p><SCRIPT>var x = 'İ';</SCRIPT><Style>.a{}</Style><p>尾部</p>";
|
||||
let text = WebClient::strip_html(html);
|
||||
assert_eq!(text, "İstanbul Ⱥ 中文尾部");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_html_keeps_text_when_block_is_unclosed() {
|
||||
assert_eq!(WebClient::strip_html("<p>前</p><script>x"), "前x");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_ddg_results_handles_multibyte_near_link() {
|
||||
let html = format!(
|
||||
"<div>İİİ<a class=\"result__a\" href=\"https://example.com/中\">{}标题</a>\
|
||||
<a class=\"result__snippet\">摘要 İ</a></div>",
|
||||
"长".repeat(60)
|
||||
);
|
||||
let out = WebClient::parse_ddg_results(&html, 5);
|
||||
assert!(out.starts_with("1. "));
|
||||
assert!(out.contains("标题"));
|
||||
assert!(out.contains("https://example.com/中"));
|
||||
assert!(out.contains("摘要 İ"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_ddg_results_handles_truncated_link_tag() {
|
||||
assert_eq!(
|
||||
WebClient::parse_ddg_results("<a class=\"result__a\"", 5),
|
||||
"No results found."
|
||||
);
|
||||
assert_eq!(
|
||||
WebClient::parse_ddg_results("<a class=\"result__a\"中文", 5),
|
||||
"No results found."
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user