| 1 | //! Content-type routing and readable-document extraction for web tools. |
| 2 | //! |
| 3 | //! Networking deliberately lives elsewhere. This module accepts already |
| 4 | //! fetched bytes and turns them into one normalized document so `fetch_url` |
| 5 | //! and `web.run` cannot disagree about HTML, Markdown, PDF, or media handling. |
| 6 | |
| 7 | use super::adapter::{AdapterFailure, AdapterResult}; |
| 8 | use std::sync::OnceLock; |
| 9 | |
| 10 | use encoding_rs::{Encoding, UTF_8, UTF_16BE, UTF_16LE}; |
| 11 | use regex::Regex; |
| 12 | |
| 13 | use crate::tools::spec::ToolError; |
| 14 | |
| 15 | #[derive(Debug, Clone, Copy, PartialEq, Eq)] |
| 16 | pub(crate) enum DocumentKind { |
| 17 | Html, |
| 18 | Markdown, |
| 19 | Text, |
| 20 | Pdf, |
| 21 | Media, |
| 22 | } |
| 23 | |
| 24 | #[derive(Debug, Clone)] |
| 25 | pub(crate) struct ExtractedDocument { |
| 26 | pub(crate) kind: DocumentKind, |
| 27 | pub(crate) title: Option<String>, |
| 28 | pub(crate) text: String, |
| 29 | pub(crate) markdown: String, |
| 30 | /// Readability-cleaned HTML. `web.run` consumes this to retain clickable |
| 31 | /// links while avoiding page chrome and consent-banner noise. |
| 32 | pub(crate) cleaned_html: Option<String>, |
| 33 | pub(crate) pdf_pages: Option<Vec<Vec<String>>>, |
| 34 | /// Validated extension for image/audio/video artifacts. |
| 35 | pub(crate) media_extension: Option<&'static str>, |
| 36 | } |
| 37 | |
| 38 | #[derive(Debug, Clone, Copy, PartialEq, Eq)] |
| 39 | struct MediaSignature { |
| 40 | extension: &'static str, |
| 41 | family: MediaFamily, |
| 42 | } |
| 43 | |
| 44 | #[derive(Debug, Clone, Copy, PartialEq, Eq)] |
| 45 | enum MediaFamily { |
| 46 | Image, |
| 47 | Audio, |
| 48 | Video, |
| 49 | } |
| 50 | |
| 51 | static TITLE_RE: OnceLock<Regex> = OnceLock::new(); |
| 52 | static FALLBACK_RE: OnceLock<[Regex; 3]> = OnceLock::new(); |
| 53 | static PAGE_CHROME_RE: OnceLock<Regex> = OnceLock::new(); |
| 54 | static PAGE_HEADER_RE: OnceLock<Regex> = OnceLock::new(); |
| 55 | static TAG_RE: OnceLock<Regex> = OnceLock::new(); |
| 56 | static WHITESPACE_RE: OnceLock<Regex> = OnceLock::new(); |
| 57 | |
| 58 | /// HTML's encoding declaration prescan is intentionally small. Keeping the |
| 59 | /// bound here prevents a late body string, script, or injected fragment from |
| 60 | /// changing how an already-started document is decoded. |
| 61 | const HTML_ENCODING_SNIFF_BYTES: usize = 1_024; |
| 62 | |
| 63 | pub(crate) async fn extract_document( |
| 64 | url: &str, |
| 65 | content_type: Option<&str>, |
| 66 | bytes: &[u8], |
| 67 | context: Option<&super::super::spec::ToolContext>, |
| 68 | ) -> AdapterResult<ExtractedDocument> { |
| 69 | extract_document_with_pdf_command( |
| 70 | url, |
| 71 | content_type, |
| 72 | bytes, |
| 73 | super::super::pdf::PdfTextCommand::system(context), |
| 74 | ) |
| 75 | .await |
| 76 | } |
| 77 | |
| 78 | pub(crate) async fn extract_document_with_pdf_command( |
| 79 | url: &str, |
| 80 | content_type: Option<&str>, |
| 81 | bytes: &[u8], |
| 82 | pdf_command: super::super::pdf::PdfTextCommand<'_>, |
| 83 | ) -> AdapterResult<ExtractedDocument> { |
| 84 | let context = pdf_command.context(); |
| 85 | let declared = normalized_content_type(content_type); |
| 86 | let declared = declared.as_deref(); |
| 87 | |
| 88 | if bytes.is_empty() { |
| 89 | return Ok(ExtractedDocument { |
| 90 | kind: DocumentKind::Text, |
| 91 | title: None, |
| 92 | text: String::new(), |
| 93 | markdown: String::new(), |
| 94 | cleaned_html: None, |
| 95 | pdf_pages: None, |
| 96 | media_extension: None, |
| 97 | }); |
| 98 | } |
| 99 | |
| 100 | if validate_pdf_response(url, content_type, bytes)? { |
| 101 | return extract_pdf(bytes, pdf_command).await.map_err(|error| { |
| 102 | if context |
| 103 | .is_some_and(|context| context.features.enabled(crate::features::Feature::PdfHost)) |
| 104 | { |
| 105 | AdapterFailure::host(error) |
| 106 | } else { |
| 107 | error.into() |
| 108 | } |
| 109 | }); |
| 110 | } |
| 111 | |
| 112 | if let Some(signature) = sniff_media(bytes) { |
| 113 | if let Some(declared_family) = declared_media_family(declared) |
| 114 | && declared_family != signature.family |
| 115 | { |
| 116 | return Err((ToolError::execution_failed(format!( |
| 117 | "Response media type `{}` did not match its bytes", |
| 118 | declared.unwrap_or("unknown") |
| 119 | ))) |
| 120 | .into()); |
| 121 | } |
| 122 | return Ok(ExtractedDocument { |
| 123 | kind: DocumentKind::Media, |
| 124 | title: None, |
| 125 | text: String::new(), |
| 126 | markdown: String::new(), |
| 127 | cleaned_html: None, |
| 128 | pdf_pages: None, |
| 129 | media_extension: Some(signature.extension), |
| 130 | }); |
| 131 | } |
| 132 | |
| 133 | if declared_media_family(declared).is_some() { |
| 134 | return Err((ToolError::execution_failed(format!( |
| 135 | "Response claimed media type `{}`, but its bytes did not match a supported media signature", |
| 136 | declared.unwrap_or("unknown") |
| 137 | ))).into()); |
| 138 | } |
| 139 | |
| 140 | let sniff_html = should_sniff_html_encoding(declared, url, bytes); |
| 141 | let body = decode_response_body(bytes, content_type, sniff_html)?; |
| 142 | if sniff_html || is_html(declared, url, &body) { |
| 143 | if let Some(context) = context.filter(|context| { |
| 144 | context |
| 145 | .features |
| 146 | .enabled(crate::features::Feature::WebExtractHost) |
| 147 | }) { |
| 148 | return extract_html_with_host(url, &body, context).await; |
| 149 | } |
| 150 | return extract_html(url, &body).map_err(Into::into); |
| 151 | } |
| 152 | if is_markdown(declared, url) { |
| 153 | return Ok(ExtractedDocument { |
| 154 | kind: DocumentKind::Markdown, |
| 155 | title: markdown_title(&body), |
| 156 | text: body.clone(), |
| 157 | markdown: body, |
| 158 | cleaned_html: None, |
| 159 | pdf_pages: None, |
| 160 | media_extension: None, |
| 161 | }); |
| 162 | } |
| 163 | if is_textual(declared, url) { |
| 164 | return Ok(ExtractedDocument { |
| 165 | kind: DocumentKind::Text, |
| 166 | title: None, |
| 167 | text: body.clone(), |
| 168 | markdown: body, |
| 169 | cleaned_html: None, |
| 170 | pdf_pages: None, |
| 171 | media_extension: None, |
| 172 | }); |
| 173 | } |
| 174 | |
| 175 | Err((ToolError::execution_failed(format!( |
| 176 | "Unsupported binary response type `{}`; use a dedicated download tool", |
| 177 | declared.unwrap_or("unknown") |
| 178 | ))) |
| 179 | .into()) |
| 180 | } |
| 181 | |
| 182 | pub(crate) fn validate_pdf_response( |
| 183 | url: &str, |
| 184 | content_type: Option<&str>, |
| 185 | bytes: &[u8], |
| 186 | ) -> Result<bool, ToolError> { |
| 187 | let declared = normalized_content_type(content_type); |
| 188 | let declared = declared.as_deref(); |
| 189 | let signed = looks_like_pdf(bytes); |
| 190 | if signed && declared_media_family(declared).is_some() { |
| 191 | return Err(ToolError::execution_failed(format!( |
| 192 | "Response media type `{}` did not match its PDF bytes", |
| 193 | declared.unwrap_or("unknown") |
| 194 | ))); |
| 195 | } |
| 196 | let claimed = signed || declared == Some("application/pdf") || url_is_pdf(url); |
| 197 | if claimed && !signed { |
| 198 | return Err(ToolError::execution_failed( |
| 199 | "Response claimed to be a PDF, but its bytes did not contain a PDF signature", |
| 200 | )); |
| 201 | } |
| 202 | Ok(claimed) |
| 203 | } |
| 204 | |
| 205 | async fn extract_html_with_host( |
| 206 | url: &str, |
| 207 | html: &str, |
| 208 | context: &super::super::spec::ToolContext, |
| 209 | ) -> AdapterResult<ExtractedDocument> { |
| 210 | #[derive(serde::Deserialize)] |
| 211 | #[serde(deny_unknown_fields)] |
| 212 | struct Choice { |
| 213 | kind: String, |
| 214 | candidate: Option<u8>, |
| 215 | } |
| 216 | let parsed_url = reqwest::Url::parse(url) |
| 217 | .map_err(|error| ToolError::invalid_input(format!("invalid URL: {error}")))?; |
| 218 | let candidates = main_html_candidates(html); |
| 219 | let facts=candidates.iter().map(|(id,html)| { |
| 220 | let text=html_to_plain_text(html); |
| 221 | serde_json::json!({"id":id,"non_whitespace":text.chars().filter(|value|!value.is_whitespace()).count(),"words":text.split_whitespace().count()}) |
| 222 | }).collect::<Vec<_>>(); |
| 223 | let budget = context |
| 224 | .turn_deadline |
| 225 | .map(|deadline| deadline.saturating_duration_since(tokio::time::Instant::now())) |
| 226 | .unwrap_or(std::time::Duration::from_secs(15)); |
| 227 | let choice: Choice = super::adapter::transform( |
| 228 | crate::extension_host::StockOperation::WebExtract, |
| 229 | serde_json::json!({"candidates":facts}), |
| 230 | context, |
| 231 | budget, |
| 232 | ) |
| 233 | .await?; |
| 234 | if choice.kind != "web_extract" { |
| 235 | return Err(AdapterFailure::host(ToolError::execution_failed( |
| 236 | "Web Host returned a malformed HTML region choice", |
| 237 | ))); |
| 238 | } |
| 239 | let expected = candidates |
| 240 | .iter() |
| 241 | .find(|(_, html)| meaningful_html(html)) |
| 242 | .map(|(id, _)| *id); |
| 243 | if choice.candidate != expected { |
| 244 | return Err(AdapterFailure::host(ToolError::execution_failed( |
| 245 | "Web Host changed mandatory readable-region order or meaningfulness", |
| 246 | ))); |
| 247 | } |
| 248 | let Some(selected) = choice.candidate else { |
| 249 | return Err(js_required_error(url).into()); |
| 250 | }; |
| 251 | let cleaned = candidates |
| 252 | .into_iter() |
| 253 | .find(|(id, _)| *id == selected) |
| 254 | .map(|(_, html)| html) |
| 255 | .ok_or_else(|| { |
| 256 | AdapterFailure::host(ToolError::execution_failed( |
| 257 | "Web Host returned an unknown HTML region", |
| 258 | )) |
| 259 | })?; |
| 260 | document_from_html_region(url, &parsed_url, html_title(html), cleaned).map_err(Into::into) |
| 261 | } |
| 262 | |
| 263 | fn extract_html(url: &str, html: &str) -> Result<ExtractedDocument, ToolError> { |
| 264 | let parsed_url = reqwest::Url::parse(url) |
| 265 | .map_err(|err| ToolError::invalid_input(format!("invalid URL: {err}")))?; |
| 266 | let original_title = html_title(html); |
| 267 | |
| 268 | // Readability-based extraction was removed to consolidate the HTML |
| 269 | // pipeline onto a single stack (htmd 0.5 + html5ever 0.38). The |
| 270 | // previous dual-stack (readability 0.3 / html5ever 0.26 + htmd / 0.38) |
| 271 | // compiled two incompatible html5ever/markup5ever trees. The fallback |
| 272 | // main-content regex retains the meaningful-content signal used by the |
| 273 | // tests (≥32 non-whitespace chars, ≥5 words) without the duplicate tree. |
| 274 | let cleaned_html = fallback_main_html(html).ok_or_else(|| js_required_error(url))?; |
| 275 | document_from_html_region(url, &parsed_url, original_title, cleaned_html) |
| 276 | } |
| 277 | |
| 278 | fn document_from_html_region( |
| 279 | url: &str, |
| 280 | parsed_url: &reqwest::Url, |
| 281 | original_title: Option<String>, |
| 282 | cleaned_html: String, |
| 283 | ) -> Result<ExtractedDocument, ToolError> { |
| 284 | let markdown = html_to_markdown_with_base_url(&cleaned_html, parsed_url).map_err(|err| { |
| 285 | ToolError::execution_failed(format!( |
| 286 | "Failed to convert readable HTML to Markdown: {err}" |
| 287 | )) |
| 288 | })?; |
| 289 | let text = html_to_plain_text(&cleaned_html); |
| 290 | |
| 291 | if !meaningful_text(&text) && !meaningful_text(&markdown) { |
| 292 | return Err(js_required_error(url)); |
| 293 | } |
| 294 | |
| 295 | let title = original_title; |
| 296 | |
| 297 | Ok(ExtractedDocument { |
| 298 | kind: DocumentKind::Html, |
| 299 | title, |
| 300 | text, |
| 301 | markdown, |
| 302 | cleaned_html: Some(cleaned_html), |
| 303 | pdf_pages: None, |
| 304 | media_extension: None, |
| 305 | }) |
| 306 | } |
| 307 | |
| 308 | /// Resolve relative anchors in `htmd`'s parsed DOM, not with an HTML regex. |
| 309 | /// Absolute, fragment, non-HTTP, and malformed destinations fall through to |
| 310 | /// the built-in handler unchanged. |
| 311 | fn html_to_markdown_with_base_url( |
| 312 | html: &str, |
| 313 | base_url: &reqwest::Url, |
| 314 | ) -> Result<String, std::io::Error> { |
| 315 | let base_url = base_url.clone(); |
| 316 | htmd::HtmlToMarkdown::builder() |
| 317 | .add_handler( |
| 318 | vec!["a"], |
| 319 | move |handlers: &dyn htmd::element_handler::Handlers, element: htmd::Element<'_>| { |
| 320 | let href = element.attrs.iter().find_map(|attr| { |
| 321 | (attr.name.local.as_ref() == "href").then(|| attr.value.to_string()) |
| 322 | }); |
| 323 | let Some(destination) = href |
| 324 | .as_deref() |
| 325 | .and_then(|href| resolve_relative_http_href(&base_url, href)) |
| 326 | else { |
| 327 | return handlers.fallback(element); |
| 328 | }; |
| 329 | let content = handlers.walk_children(element.node).content; |
| 330 | let trailing = &content[content.trim_end().len()..]; |
| 331 | let destination = destination.replace('(', "\\(").replace(')', "\\)"); |
| 332 | let title = element |
| 333 | .attrs |
| 334 | .iter() |
| 335 | .find_map(|attr| { |
| 336 | (attr.name.local.as_ref() == "title").then(|| { |
| 337 | attr.value |
| 338 | .split_whitespace() |
| 339 | .collect::<Vec<_>>() |
| 340 | .join(" ") |
| 341 | .replace('"', "\\\"") |
| 342 | }) |
| 343 | }) |
| 344 | .map_or_else(String::new, |title| format!(" \"{title}\"")); |
| 345 | Some(format!("[{}]({destination}{title}){trailing}", content.trim()).into()) |
| 346 | }, |
| 347 | ) |
| 348 | .build() |
| 349 | .convert(html) |
| 350 | } |
| 351 | |
| 352 | fn resolve_relative_http_href(base_url: &reqwest::Url, href: &str) -> Option<String> { |
| 353 | if !matches!(base_url.scheme(), "http" | "https") { |
| 354 | return None; |
| 355 | } |
| 356 | |
| 357 | let href = href.trim(); |
| 358 | if href.is_empty() || href.starts_with('#') || reqwest::Url::parse(href).is_ok() { |
| 359 | return None; |
| 360 | } |
| 361 | |
| 362 | base_url.join(href).ok().map(Into::into) |
| 363 | } |
| 364 | |
| 365 | /// Pick the readable region of a page with bounded regexes. |
| 366 | /// |
| 367 | /// Order: every `<article>` (listing, news and forum pages carry one per |
| 368 | /// card), then `<main>`, then `<body>`. Page chrome is removed from the chosen |
| 369 | /// region; a `<header>` is chrome only at body level, because inside an |
| 370 | /// article or `<main>` it carries the title and byline. Forms are unwrapped, |
| 371 | /// not dropped: ASP.NET-style pages wrap the whole body in one `<form>`, so |
| 372 | /// only the controls themselves are stripped. |
| 373 | /// |
| 374 | /// Known limits: the lazy regexes do not balance nested same-name elements |
| 375 | /// (an `<article>` inside an `<article>` ends at the inner close tag), and no |
| 376 | /// JavaScript runs, so a client-rendered shell still yields nothing. |
| 377 | fn main_html_candidates(html: &str) -> Vec<(u8, String)> { |
| 378 | let [article, main, body] = FALLBACK_RE.get_or_init(|| { |
| 379 | ["article", "main", "body"].map(|tag| { |
| 380 | Regex::new(&format!(r"(?is)<{tag}(?:\s[^>]*)?>(.*?)</{tag}\s*>")) |
| 381 | .expect("fallback element regex") |
| 382 | }) |
| 383 | }); |
| 384 | let articles = article |
| 385 | .captures_iter(html) |
| 386 | .filter_map(|capture| capture.get(1)) |
| 387 | .map(|content| strip_page_chrome(content.as_str(), false)) |
| 388 | .filter(|content| !html_to_plain_text(content).is_empty()) |
| 389 | .collect::<Vec<_>>() |
| 390 | .join("\n"); |
| 391 | let mut candidates = vec![(0, articles)]; |
| 392 | for (id, re, strip_header) in [(1, main, false), (2, body, true)] { |
| 393 | if let Some(content) = re.captures(html).and_then(|capture| capture.get(1)) { |
| 394 | candidates.push((id, strip_page_chrome(content.as_str(), strip_header))); |
| 395 | } |
| 396 | } |
| 397 | candidates |
| 398 | } |
| 399 | |
| 400 | fn fallback_main_html(html: &str) -> Option<String> { |
| 401 | main_html_candidates(html) |
| 402 | .into_iter() |
| 403 | .find_map(|(_, html)| meaningful_html(&html).then_some(html)) |
| 404 | } |
| 405 | |
| 406 | fn strip_page_chrome(html: &str, strip_header: bool) -> String { |
| 407 | let page_chrome = PAGE_CHROME_RE.get_or_init(|| { |
| 408 | Regex::new(concat!( |
| 409 | r"(?is)(?:<script(?:\s[^>]*)?>.*?</script\s*>", |
| 410 | r"|<style(?:\s[^>]*)?>.*?</style\s*>", |
| 411 | r"|<noscript(?:\s[^>]*)?>.*?</noscript\s*>", |
| 412 | r"|<nav(?:\s[^>]*)?>.*?</nav\s*>", |
| 413 | r"|<footer(?:\s[^>]*)?>.*?</footer\s*>", |
| 414 | r"|<aside(?:\s[^>]*)?>.*?</aside\s*>", |
| 415 | // Form controls, and the form tags themselves (unwrapped). |
| 416 | r"|<select(?:\s[^>]*)?>.*?</select\s*>", |
| 417 | r"|<textarea(?:\s[^>]*)?>.*?</textarea\s*>", |
| 418 | r"|<button(?:\s[^>]*)?>.*?</button\s*>", |
| 419 | r"|<input(?:\s[^>]*)?>", |
| 420 | r"|</?form(?:\s[^>]*)?>)", |
| 421 | )) |
| 422 | .expect("page chrome regex") |
| 423 | }); |
| 424 | let cleaned = page_chrome.replace_all(html, ""); |
| 425 | if !strip_header { |
| 426 | return cleaned.into_owned(); |
| 427 | } |
| 428 | PAGE_HEADER_RE |
| 429 | .get_or_init(|| { |
| 430 | Regex::new(r"(?is)<header(?:\s[^>]*)?>.*?</header\s*>").expect("page header regex") |
| 431 | }) |
| 432 | .replace_all(&cleaned, "") |
| 433 | .into_owned() |
| 434 | } |
| 435 | |
| 436 | fn meaningful_html(html: &str) -> bool { |
| 437 | meaningful_text(&html_to_plain_text(html)) |
| 438 | } |
| 439 | |
| 440 | fn meaningful_text(text: &str) -> bool { |
| 441 | text.chars().filter(|ch| !ch.is_whitespace()).count() >= 32 |
| 442 | && text.split_whitespace().count() >= 5 |
| 443 | } |
| 444 | |
| 445 | fn html_to_plain_text(html: &str) -> String { |
| 446 | let without_tags = TAG_RE |
| 447 | .get_or_init(|| Regex::new(r"(?s)<[^>]+>").expect("tag regex")) |
| 448 | .replace_all(html, " "); |
| 449 | normalize_text(&decode_common_entities(&without_tags)) |
| 450 | } |
| 451 | |
| 452 | fn normalize_text(text: &str) -> String { |
| 453 | WHITESPACE_RE |
| 454 | .get_or_init(|| Regex::new(r"\s+").expect("whitespace regex")) |
| 455 | .replace_all(text.trim(), " ") |
| 456 | .into_owned() |
| 457 | } |
| 458 | |
| 459 | fn decode_common_entities(value: &str) -> String { |
| 460 | value |
| 461 | .replace(" ", " ") |
| 462 | .replace("&", "&") |
| 463 | .replace("<", "<") |
| 464 | .replace(">", ">") |
| 465 | .replace(""", "\"") |
| 466 | .replace("'", "'") |
| 467 | } |
| 468 | |
| 469 | fn html_title(html: &str) -> Option<String> { |
| 470 | let capture = TITLE_RE |
| 471 | .get_or_init(|| { |
| 472 | Regex::new(r"(?is)<title(?:\s[^>]*)?>(.*?)</title\s*>").expect("title regex") |
| 473 | }) |
| 474 | .captures(html)?; |
| 475 | let title = normalize_text(&decode_common_entities(capture.get(1)?.as_str())); |
| 476 | (!title.is_empty()).then_some(title) |
| 477 | } |
| 478 | |
| 479 | fn markdown_title(body: &str) -> Option<String> { |
| 480 | body.lines().find_map(|line| { |
| 481 | let title = line.trim().strip_prefix("# ")?.trim(); |
| 482 | (!title.is_empty()).then(|| title.to_string()) |
| 483 | }) |
| 484 | } |
| 485 | |
| 486 | /// Stable prefix for "the response parsed, but carried no readable body". |
| 487 | /// |
| 488 | /// The fetch pipeline matches on this to decide whether a cache-busting |
| 489 | /// re-fetch is worth one more request, and to attach the role-aware recovery |
| 490 | /// text. Extraction itself has no `ToolContext`, so it cannot know which |
| 491 | /// escalation the calling role actually owns; it states the fact and leaves |
| 492 | /// the remedy to [`super::fetch`]. |
| 493 | pub(crate) const JS_SHELL_MARKER: &str = "No readable page content was found at"; |
| 494 | |
| 495 | /// Whether `error` is the JS-shell extraction failure (a parsed response whose |
| 496 | /// body held no readable content), as opposed to a transport or type failure. |
| 497 | pub(crate) fn is_js_shell_error(error: &ToolError) -> bool { |
| 498 | error.to_string().contains(JS_SHELL_MARKER) |
| 499 | } |
| 500 | |
| 501 | fn js_required_error(url: &str) -> ToolError { |
| 502 | ToolError::execution_failed(format!( |
| 503 | "{JS_SHELL_MARKER} {url}; the response parsed but its body held no readable content, so the page may require JavaScript." |
| 504 | )) |
| 505 | } |
| 506 | |
| 507 | /// Decode one response body without guessing from language statistics. |
| 508 | /// |
| 509 | /// Precedence is receipt-grade and deterministic: BOM, recognized transport |
| 510 | /// charset, an HTML-only bounded meta prescan, then UTF-8. Unknown transport |
| 511 | /// labels deliberately fall through to a valid HTML declaration. `html_sniff` |
| 512 | /// must come from MIME/URL/ASCII markup evidence; JSON and plain text callers |
| 513 | /// pass `false`, so a body string cannot impersonate an HTML declaration. |
| 514 | pub(crate) fn decode_response_body( |
| 515 | bytes: &[u8], |
| 516 | content_type: Option<&str>, |
| 517 | html_sniff: bool, |
| 518 | ) -> Result<String, ToolError> { |
| 519 | if let Some((encoding, bom_len)) = Encoding::for_bom(bytes) { |
| 520 | let (decoded, _) = encoding.decode_without_bom_handling(&bytes[bom_len..]); |
| 521 | reject_binary_nul(bytes, encoding, &decoded)?; |
| 522 | return Ok(decoded.into_owned()); |
| 523 | } |
| 524 | |
| 525 | let transport_encoding = content_type.and_then(content_type_encoding); |
| 526 | let encoding = transport_encoding |
| 527 | .or_else(|| html_sniff.then(|| html_meta_encoding(bytes)).flatten()) |
| 528 | .unwrap_or(UTF_8); |
| 529 | let (decoded, _) = encoding.decode_without_bom_handling(bytes); |
| 530 | reject_binary_nul(bytes, encoding, &decoded)?; |
| 531 | Ok(decoded.into_owned()) |
| 532 | } |
| 533 | |
| 534 | fn reject_binary_nul( |
| 535 | bytes: &[u8], |
| 536 | encoding: &'static Encoding, |
| 537 | decoded: &str, |
| 538 | ) -> Result<(), ToolError> { |
| 539 | // UTF-16 uses zero bytes structurally for many characters, so inspect its |
| 540 | // decoded scalar values. Every other encoding must be NUL-free across the |
| 541 | // complete response; neither a BOM nor a late byte may bypass the guard. |
| 542 | let contains_nul = if encoding == UTF_16LE || encoding == UTF_16BE { |
| 543 | decoded.contains('\0') |
| 544 | } else { |
| 545 | bytes.contains(&0) |
| 546 | }; |
| 547 | if contains_nul { |
| 548 | return Err(ToolError::execution_failed( |
| 549 | "Unsupported binary response contained NUL bytes", |
| 550 | )); |
| 551 | } |
| 552 | Ok(()) |
| 553 | } |
| 554 | |
| 555 | /// Parse only exact semicolon-delimited `charset` parameters. A random |
| 556 | /// `charset=` substring inside another parameter is not transport authority. |
| 557 | fn content_type_encoding(value: &str) -> Option<&'static Encoding> { |
| 558 | value.split(';').skip(1).find_map(|parameter| { |
| 559 | let (name, raw_value) = parameter.split_once('=')?; |
| 560 | if !name.trim().eq_ignore_ascii_case("charset") { |
| 561 | return None; |
| 562 | } |
| 563 | let value = raw_value.trim(); |
| 564 | let value = match (value.as_bytes().first(), value.as_bytes().last()) { |
| 565 | (Some(b'"'), Some(b'"')) | (Some(b'\''), Some(b'\'')) if value.len() >= 2 => { |
| 566 | &value[1..value.len() - 1] |
| 567 | } |
| 568 | _ if value.contains('"') || value.contains('\'') => return None, |
| 569 | _ => value, |
| 570 | }; |
| 571 | let label = value.trim(); |
| 572 | (!label.is_empty()) |
| 573 | .then(|| Encoding::for_label(label.as_bytes())) |
| 574 | .flatten() |
| 575 | }) |
| 576 | } |
| 577 | |
| 578 | fn should_sniff_html_encoding(content_type: Option<&str>, url: &str, bytes: &[u8]) -> bool { |
| 579 | match content_type { |
| 580 | Some("text/html" | "application/xhtml+xml") => true, |
| 581 | // Explicit non-HTML text and structured formats never consult markup |
| 582 | // embedded in their body. |
| 583 | Some(value) if value.starts_with("text/") || is_structured_text_type(value) => false, |
| 584 | Some("application/octet-stream") | None => { |
| 585 | url_path_ends_with(url, &[".html", ".htm"]) || looks_like_html_bytes(bytes) |
| 586 | } |
| 587 | Some(_) => false, |
| 588 | } |
| 589 | } |
| 590 | |
| 591 | fn is_structured_text_type(content_type: &str) -> bool { |
| 592 | content_type.contains("json") |
| 593 | || content_type.contains("xml") |
| 594 | || content_type.contains("yaml") |
| 595 | || content_type.contains("javascript") |
| 596 | } |
| 597 | |
| 598 | fn looks_like_html_bytes(bytes: &[u8]) -> bool { |
| 599 | let start = Encoding::for_bom(bytes).map_or(0, |(_, length)| length); |
| 600 | let end = bytes |
| 601 | .len() |
| 602 | .min(start.saturating_add(HTML_ENCODING_SNIFF_BYTES)); |
| 603 | let ascii = ascii_lowercase_projection(&bytes[start..end]); |
| 604 | let Some(prefix) = html_prefix_after_leading_declarations(&ascii) else { |
| 605 | return false; |
| 606 | }; |
| 607 | prefix.starts_with("<!doctype html") |
| 608 | || prefix.starts_with("<html") |
| 609 | || prefix.starts_with("<head") |
| 610 | || prefix.starts_with("<meta") |
| 611 | } |
| 612 | |
| 613 | fn html_prefix_after_leading_declarations(mut prefix: &str) -> Option<&str> { |
| 614 | loop { |
| 615 | prefix = prefix.trim_start(); |
| 616 | if let Some(comment) = prefix.strip_prefix("<!--") { |
| 617 | let end = comment.find("-->")?; |
| 618 | prefix = &comment[end + 3..]; |
| 619 | continue; |
| 620 | } |
| 621 | if let Some(declaration) = prefix.strip_prefix("<?xml") { |
| 622 | let end = declaration.find("?>")?; |
| 623 | prefix = &declaration[end + 2..]; |
| 624 | continue; |
| 625 | } |
| 626 | return Some(prefix); |
| 627 | } |
| 628 | } |
| 629 | |
| 630 | fn html_meta_encoding(bytes: &[u8]) -> Option<&'static Encoding> { |
| 631 | let sniff_len = bytes.len().min(HTML_ENCODING_SNIFF_BYTES); |
| 632 | let html = ascii_lowercase_projection(&bytes[..sniff_len]); |
| 633 | let mut cursor = 0usize; |
| 634 | |
| 635 | while let Some(relative) = html[cursor..].find('<') { |
| 636 | let start = cursor + relative; |
| 637 | if html[start..].starts_with("<!--") { |
| 638 | cursor = html[start + 4..] |
| 639 | .find("-->") |
| 640 | .map_or(html.len(), |end| start + 4 + end + 3); |
| 641 | continue; |
| 642 | } |
| 643 | if tag_starts_at(&html, start, "script") || tag_starts_at(&html, start, "style") { |
| 644 | let name = if tag_starts_at(&html, start, "script") { |
| 645 | "script" |
| 646 | } else { |
| 647 | "style" |
| 648 | }; |
| 649 | let close = format!("</{name}"); |
| 650 | cursor = html[start..] |
| 651 | .find(&close) |
| 652 | .and_then(|close_start| { |
| 653 | html[start + close_start..] |
| 654 | .find('>') |
| 655 | .map(|end| start + close_start + end + 1) |
| 656 | }) |
| 657 | .unwrap_or(html.len()); |
| 658 | continue; |
| 659 | } |
| 660 | if !tag_starts_at(&html, start, "meta") { |
| 661 | cursor = start + 1; |
| 662 | continue; |
| 663 | } |
| 664 | let relative_end = html[start..].find('>')?; |
| 665 | let end = start + relative_end + 1; |
| 666 | let tag = &html[start..end]; |
| 667 | if let Some(label) = html_attribute_value(tag, "charset") |
| 668 | && let Some(encoding) = Encoding::for_label(label.as_bytes()) |
| 669 | { |
| 670 | return Some(normalize_meta_encoding(encoding)); |
| 671 | } |
| 672 | let is_content_type = html_attribute_value(tag, "http-equiv") |
| 673 | .is_some_and(|value| value.eq_ignore_ascii_case("content-type")); |
| 674 | if is_content_type |
| 675 | && let Some(content) = html_attribute_value(tag, "content") |
| 676 | && let Some(encoding) = content_type_encoding(&content) |
| 677 | { |
| 678 | return Some(normalize_meta_encoding(encoding)); |
| 679 | } |
| 680 | cursor = end; |
| 681 | } |
| 682 | None |
| 683 | } |
| 684 | |
| 685 | fn normalize_meta_encoding(encoding: &'static Encoding) -> &'static Encoding { |
| 686 | if encoding == UTF_16LE || encoding == UTF_16BE { |
| 687 | UTF_8 |
| 688 | } else { |
| 689 | encoding |
| 690 | } |
| 691 | } |
| 692 | |
| 693 | fn ascii_lowercase_projection(bytes: &[u8]) -> String { |
| 694 | bytes |
| 695 | .iter() |
| 696 | .map(|byte| { |
| 697 | if byte.is_ascii() { |
| 698 | char::from(byte.to_ascii_lowercase()) |
| 699 | } else { |
| 700 | ' ' |
| 701 | } |
| 702 | }) |
| 703 | .collect() |
| 704 | } |
| 705 | |
| 706 | fn tag_starts_at(html: &str, start: usize, name: &str) -> bool { |
| 707 | let Some(after_name) = html.get(start + 1 + name.len()..) else { |
| 708 | return false; |
| 709 | }; |
| 710 | html[start + 1..].starts_with(name) |
| 711 | && after_name |
| 712 | .chars() |
| 713 | .next() |
| 714 | .is_some_and(|ch| ch.is_ascii_whitespace() || matches!(ch, '/' | '>')) |
| 715 | } |
| 716 | |
| 717 | fn html_attribute_value(tag: &str, wanted: &str) -> Option<String> { |
| 718 | let bytes = tag.as_bytes(); |
| 719 | let mut cursor = 1usize; |
| 720 | while cursor < bytes.len() && !bytes[cursor].is_ascii_whitespace() && bytes[cursor] != b'>' { |
| 721 | cursor += 1; |
| 722 | } |
| 723 | while cursor < bytes.len() { |
| 724 | while cursor < bytes.len() && (bytes[cursor].is_ascii_whitespace() || bytes[cursor] == b'/') |
| 725 | { |
| 726 | cursor += 1; |
| 727 | } |
| 728 | if cursor >= bytes.len() || bytes[cursor] == b'>' { |
| 729 | break; |
| 730 | } |
| 731 | let name_start = cursor; |
| 732 | while cursor < bytes.len() |
| 733 | && !bytes[cursor].is_ascii_whitespace() |
| 734 | && !matches!(bytes[cursor], b'=' | b'/' | b'>') |
| 735 | { |
| 736 | cursor += 1; |
| 737 | } |
| 738 | let name = &tag[name_start..cursor]; |
| 739 | while cursor < bytes.len() && bytes[cursor].is_ascii_whitespace() { |
| 740 | cursor += 1; |
| 741 | } |
| 742 | if cursor >= bytes.len() || bytes[cursor] != b'=' { |
| 743 | continue; |
| 744 | } |
| 745 | cursor += 1; |
| 746 | while cursor < bytes.len() && bytes[cursor].is_ascii_whitespace() { |
| 747 | cursor += 1; |
| 748 | } |
| 749 | if cursor >= bytes.len() { |
| 750 | break; |
| 751 | } |
| 752 | let (value_start, value_end) = if matches!(bytes[cursor], b'"' | b'\'') { |
| 753 | let quote = bytes[cursor]; |
| 754 | cursor += 1; |
| 755 | let start = cursor; |
| 756 | while cursor < bytes.len() && bytes[cursor] != quote { |
| 757 | cursor += 1; |
| 758 | } |
| 759 | let end = cursor; |
| 760 | cursor = cursor.saturating_add(1); |
| 761 | (start, end) |
| 762 | } else { |
| 763 | let start = cursor; |
| 764 | while cursor < bytes.len() |
| 765 | && !bytes[cursor].is_ascii_whitespace() |
| 766 | && bytes[cursor] != b'>' |
| 767 | { |
| 768 | cursor += 1; |
| 769 | } |
| 770 | (start, cursor) |
| 771 | }; |
| 772 | if name.eq_ignore_ascii_case(wanted) { |
| 773 | return Some(tag[value_start..value_end].trim().to_string()); |
| 774 | } |
| 775 | } |
| 776 | None |
| 777 | } |
| 778 | |
| 779 | fn normalized_content_type(content_type: Option<&str>) -> Option<String> { |
| 780 | content_type |
| 781 | .and_then(|value| value.split(';').next()) |
| 782 | .map(str::trim) |
| 783 | .filter(|value| !value.is_empty()) |
| 784 | .map(str::to_ascii_lowercase) |
| 785 | } |
| 786 | |
| 787 | fn is_html(content_type: Option<&str>, url: &str, body: &str) -> bool { |
| 788 | matches!(content_type, Some("text/html" | "application/xhtml+xml")) |
| 789 | || url_path_ends_with(url, &[".html", ".htm"]) |
| 790 | || { |
| 791 | let prefix = body.trim_start().chars().take(64).collect::<String>(); |
| 792 | let prefix = prefix.to_ascii_lowercase(); |
| 793 | prefix.contains("<!doctype html") || prefix.contains("<html") |
| 794 | } |
| 795 | } |
| 796 | |
| 797 | fn is_markdown(content_type: Option<&str>, url: &str) -> bool { |
| 798 | matches!( |
| 799 | content_type, |
| 800 | Some("text/markdown" | "text/x-markdown" | "application/markdown") |
| 801 | ) || url_path_ends_with(url, &[".md", ".markdown"]) |
| 802 | } |
| 803 | |
| 804 | fn is_textual(content_type: Option<&str>, url: &str) -> bool { |
| 805 | content_type.is_some_and(|value| { |
| 806 | value.starts_with("text/") |
| 807 | || value.contains("json") |
| 808 | || value.contains("xml") |
| 809 | || value.contains("yaml") |
| 810 | || value.contains("javascript") |
| 811 | || value == "application/sql" |
| 812 | }) || url_path_ends_with( |
| 813 | url, |
| 814 | &[ |
| 815 | ".txt", ".json", ".jsonl", ".xml", ".yaml", ".yml", ".csv", ".tsv", ".rs", ".py", |
| 816 | ".js", ".ts", ".toml", |
| 817 | ], |
| 818 | ) |
| 819 | } |
| 820 | |
| 821 | fn url_is_pdf(url: &str) -> bool { |
| 822 | url_path_ends_with(url, &[".pdf"]) |
| 823 | } |
| 824 | |
| 825 | fn url_path_ends_with(url: &str, extensions: &[&str]) -> bool { |
| 826 | reqwest::Url::parse(url) |
| 827 | .ok() |
| 828 | .map(|parsed| parsed.path().to_ascii_lowercase()) |
| 829 | .is_some_and(|path| extensions.iter().any(|extension| path.ends_with(extension))) |
| 830 | } |
| 831 | |
| 832 | fn looks_like_pdf(bytes: &[u8]) -> bool { |
| 833 | bytes.starts_with(b"%PDF-") |
| 834 | } |
| 835 | |
| 836 | fn declared_media_family(content_type: Option<&str>) -> Option<MediaFamily> { |
| 837 | let content_type = content_type?; |
| 838 | if content_type.starts_with("image/") { |
| 839 | Some(MediaFamily::Image) |
| 840 | } else if content_type.starts_with("audio/") { |
| 841 | Some(MediaFamily::Audio) |
| 842 | } else if content_type.starts_with("video/") { |
| 843 | Some(MediaFamily::Video) |
| 844 | } else { |
| 845 | None |
| 846 | } |
| 847 | } |
| 848 | |
| 849 | fn sniff_media(bytes: &[u8]) -> Option<MediaSignature> { |
| 850 | let trimmed = bytes |
| 851 | .iter() |
| 852 | .position(|byte| !byte.is_ascii_whitespace()) |
| 853 | .map(|start| &bytes[start..]) |
| 854 | .unwrap_or(bytes); |
| 855 | let signature = if bytes.starts_with(b"\x89PNG\r\n\x1a\n") { |
| 856 | MediaSignature { |
| 857 | extension: "png", |
| 858 | family: MediaFamily::Image, |
| 859 | } |
| 860 | } else if bytes.starts_with(b"\xff\xd8\xff") { |
| 861 | MediaSignature { |
| 862 | extension: "jpg", |
| 863 | family: MediaFamily::Image, |
| 864 | } |
| 865 | } else if bytes.starts_with(b"GIF87a") || bytes.starts_with(b"GIF89a") { |
| 866 | MediaSignature { |
| 867 | extension: "gif", |
| 868 | family: MediaFamily::Image, |
| 869 | } |
| 870 | } else if bytes.len() >= 12 && bytes.starts_with(b"RIFF") && &bytes[8..12] == b"WEBP" { |
| 871 | MediaSignature { |
| 872 | extension: "webp", |
| 873 | family: MediaFamily::Image, |
| 874 | } |
| 875 | } else if bytes.starts_with(b"ID3") || bytes.starts_with(b"\xff\xfb") { |
| 876 | MediaSignature { |
| 877 | extension: "mp3", |
| 878 | family: MediaFamily::Audio, |
| 879 | } |
| 880 | } else if bytes.starts_with(b"fLaC") { |
| 881 | MediaSignature { |
| 882 | extension: "flac", |
| 883 | family: MediaFamily::Audio, |
| 884 | } |
| 885 | } else if bytes.starts_with(b"OggS") { |
| 886 | MediaSignature { |
| 887 | extension: "ogg", |
| 888 | family: MediaFamily::Audio, |
| 889 | } |
| 890 | } else if bytes.len() >= 12 && bytes.starts_with(b"RIFF") && &bytes[8..12] == b"WAVE" { |
| 891 | MediaSignature { |
| 892 | extension: "wav", |
| 893 | family: MediaFamily::Audio, |
| 894 | } |
| 895 | } else if bytes.len() >= 12 && &bytes[4..8] == b"ftyp" { |
| 896 | MediaSignature { |
| 897 | extension: "mp4", |
| 898 | family: MediaFamily::Video, |
| 899 | } |
| 900 | } else if bytes.starts_with(b"\x1aE\xdf\xa3") { |
| 901 | MediaSignature { |
| 902 | extension: "webm", |
| 903 | family: MediaFamily::Video, |
| 904 | } |
| 905 | } else if trimmed.starts_with(b"<svg") |
| 906 | || (trimmed.starts_with(b"<?xml") |
| 907 | && trimmed |
| 908 | .windows(4) |
| 909 | .take(1_024) |
| 910 | .any(|window| window.eq_ignore_ascii_case(b"<svg"))) |
| 911 | { |
| 912 | MediaSignature { |
| 913 | extension: "svg", |
| 914 | family: MediaFamily::Image, |
| 915 | } |
| 916 | } else { |
| 917 | return None; |
| 918 | }; |
| 919 | Some(signature) |
| 920 | } |
| 921 | |
| 922 | async fn extract_pdf( |
| 923 | bytes: &[u8], |
| 924 | command: super::super::pdf::PdfTextCommand<'_>, |
| 925 | ) -> Result<ExtractedDocument, ToolError> { |
| 926 | let text = super::super::pdf::extract_bytes(bytes, command) |
| 927 | .await |
| 928 | .map_err(super::super::pdf::into_tool_error)?; |
| 929 | let pages = split_pdf_pages(&text); |
| 930 | let text = pages |
| 931 | .iter() |
| 932 | .map(|page| page.join("\n")) |
| 933 | .collect::<Vec<_>>() |
| 934 | .join("\n\n"); |
| 935 | Ok(ExtractedDocument { |
| 936 | kind: DocumentKind::Pdf, |
| 937 | title: Some("PDF Document".to_string()), |
| 938 | markdown: text.clone(), |
| 939 | text, |
| 940 | cleaned_html: None, |
| 941 | pdf_pages: Some(pages), |
| 942 | media_extension: None, |
| 943 | }) |
| 944 | } |
| 945 | |
| 946 | fn split_pdf_pages(text: &str) -> Vec<Vec<String>> { |
| 947 | text.split('\x0C') |
| 948 | .map(|page| { |
| 949 | page.lines() |
| 950 | .map(str::trim) |
| 951 | .filter(|line| !line.is_empty()) |
| 952 | .map(ToOwned::to_owned) |
| 953 | .collect::<Vec<_>>() |
| 954 | }) |
| 955 | .collect() |
| 956 | } |
| 957 | |
| 958 | #[cfg(test)] |
| 959 | mod tests { |
| 960 | use super::*; |
| 961 | |
| 962 | #[tokio::test] |
| 963 | async fn html_becomes_readable_markdown_without_page_chrome() { |
| 964 | let html = br#"<!doctype html><html><head><title>Whale & Signal</title></head><body> |
| 965 | <nav>Products Pricing Log in Cookies</nav> |
| 966 | <article><h1>Fetch once</h1><p>This is the important article body with enough words to be useful.</p> |
| 967 | <a href="/proof">Read the proof</a></article> |
| 968 | <footer>Privacy Cookies Terms</footer></body></html>"#; |
| 969 | let document = extract_document("https://example.com/post", Some("text/html"), html, None) |
| 970 | .await |
| 971 | .expect("extract html"); |
| 972 | |
| 973 | assert_eq!(document.kind, DocumentKind::Html); |
| 974 | assert_eq!(document.title.as_deref(), Some("Whale & Signal")); |
| 975 | assert!(document.markdown.contains("Fetch once") || document.title.is_some()); |
| 976 | assert!( |
| 977 | document |
| 978 | .markdown |
| 979 | .contains("[Read the proof](https://example.com/proof)") |
| 980 | ); |
| 981 | assert!(!document.markdown.contains("Products Pricing")); |
| 982 | assert!(!document.markdown.contains("Privacy Cookies")); |
| 983 | } |
| 984 | |
| 985 | #[test] |
| 986 | fn relative_http_href_resolution_preserves_other_destination_kinds() { |
| 987 | let base = reqwest::Url::parse("https://example.com/guides/page").expect("base URL"); |
| 988 | assert_eq!( |
| 989 | resolve_relative_http_href(&base, "../proof?q=1#receipt").as_deref(), |
| 990 | Some("https://example.com/proof?q=1#receipt") |
| 991 | ); |
| 992 | for href in [ |
| 993 | "#receipt", |
| 994 | "mailto:maintainer@example.com", |
| 995 | "data:text/plain,proof", |
| 996 | "codewhale:session/123", |
| 997 | "https://other.example/proof", |
| 998 | "http://[::1", |
| 999 | "", |
| 1000 | ] { |
| 1001 | assert_eq!( |
| 1002 | resolve_relative_http_href(&base, href), |
| 1003 | None, |
| 1004 | "destination must be left to htmd unchanged: {href:?}" |
| 1005 | ); |
| 1006 | } |
| 1007 | let file = reqwest::Url::parse("file:///tmp/page").expect("file URL"); |
| 1008 | assert!(resolve_relative_http_href(&file, "proof").is_none()); |
| 1009 | } |
| 1010 | |
| 1011 | #[tokio::test] |
| 1012 | async fn sparse_document_uses_article_fallback() { |
| 1013 | let html = br#"<html><head><title>Fallback</title></head><body><nav>cookie banner</nav> |
| 1014 | <article><h2>Small source</h2><p>Five useful words survive this compact article fallback path.</p></article> |
| 1015 | </body></html>"#; |
| 1016 | let document = extract_document("https://example.com/short", Some("text/html"), html, None) |
| 1017 | .await |
| 1018 | .expect("extract fallback"); |
| 1019 | |
| 1020 | assert!(document.markdown.contains("## Small source")); |
| 1021 | assert!(!document.markdown.contains("cookie banner")); |
| 1022 | } |
| 1023 | |
| 1024 | #[tokio::test] |
| 1025 | async fn javascript_shell_returns_actionable_error() { |
| 1026 | let error = extract_document( |
| 1027 | "https://example.com/app", |
| 1028 | Some("text/html"), |
| 1029 | b"<html><body><div id='root'></div><script>boot()</script></body></html>", |
| 1030 | None, |
| 1031 | ) |
| 1032 | .await |
| 1033 | .expect_err("empty app shell must fail"); |
| 1034 | |
| 1035 | let message = error.to_string(); |
| 1036 | assert!(message.contains("may require JavaScript"), "{message}"); |
| 1037 | assert!( |
| 1038 | message.contains("https://example.com/app"), |
| 1039 | "the shell failure must name the URL: {message}" |
| 1040 | ); |
| 1041 | assert!( |
| 1042 | is_js_shell_error(&error.error), |
| 1043 | "the fetch pipeline recognizes this failure by marker: {message}" |
| 1044 | ); |
| 1045 | assert!( |
| 1046 | !is_js_shell_error(&ToolError::execution_failed("connection reset")), |
| 1047 | "transport failures must not look like a JS shell" |
| 1048 | ); |
| 1049 | } |
| 1050 | |
| 1051 | #[tokio::test] |
| 1052 | async fn markdown_passes_through_unchanged() { |
| 1053 | let body = b"# Release note\n\nA complete markdown response remains intact.\n"; |
| 1054 | let document = extract_document( |
| 1055 | "https://example.com/release.md", |
| 1056 | Some("text/markdown; charset=utf-8"), |
| 1057 | body, |
| 1058 | None, |
| 1059 | ) |
| 1060 | .await |
| 1061 | .expect("extract markdown"); |
| 1062 | |
| 1063 | assert_eq!(document.kind, DocumentKind::Markdown); |
| 1064 | assert_eq!(document.markdown.as_bytes(), body); |
| 1065 | assert_eq!(document.title.as_deref(), Some("Release note")); |
| 1066 | } |
| 1067 | |
| 1068 | #[tokio::test] |
| 1069 | async fn media_requires_matching_magic_bytes() { |
| 1070 | let error = extract_document( |
| 1071 | "https://example.com/not-image.png", |
| 1072 | Some("image/png"), |
| 1073 | b"<html>not really an image</html>", |
| 1074 | None, |
| 1075 | ) |
| 1076 | .await |
| 1077 | .expect_err("spoofed media must fail"); |
| 1078 | assert!(error.to_string().contains("did not match")); |
| 1079 | |
| 1080 | let mut png = b"\x89PNG\r\n\x1a\n".to_vec(); |
| 1081 | png.extend_from_slice(b"fake test payload"); |
| 1082 | let document = extract_document( |
| 1083 | "https://example.com/image", |
| 1084 | Some("application/octet-stream"), |
| 1085 | &png, |
| 1086 | None, |
| 1087 | ) |
| 1088 | .await |
| 1089 | .expect("sniff png"); |
| 1090 | assert_eq!(document.kind, DocumentKind::Media); |
| 1091 | assert_eq!(document.media_extension, Some("png")); |
| 1092 | } |
| 1093 | |
| 1094 | #[tokio::test] |
| 1095 | async fn arbitrary_binary_is_rejected() { |
| 1096 | let error = extract_document( |
| 1097 | "https://example.com/archive.bin", |
| 1098 | Some("application/octet-stream"), |
| 1099 | b"PK\x03\x04archive bytes", |
| 1100 | None, |
| 1101 | ) |
| 1102 | .await |
| 1103 | .expect_err("archive must be rejected"); |
| 1104 | assert!(error.to_string().contains("Unsupported binary response")); |
| 1105 | } |
| 1106 | |
| 1107 | #[tokio::test] |
| 1108 | async fn empty_success_body_is_valid_text() { |
| 1109 | let document = extract_document( |
| 1110 | "https://example.com/no-content", |
| 1111 | Some("application/octet-stream"), |
| 1112 | b"", |
| 1113 | None, |
| 1114 | ) |
| 1115 | .await |
| 1116 | .expect("empty body"); |
| 1117 | assert_eq!(document.kind, DocumentKind::Text); |
| 1118 | assert!(document.text.is_empty()); |
| 1119 | } |
| 1120 | |
| 1121 | #[tokio::test] |
| 1122 | async fn content_type_matching_is_case_insensitive() { |
| 1123 | let document = extract_document( |
| 1124 | "https://example.com/document", |
| 1125 | Some("Application/JSON; Charset=UTF-8"), |
| 1126 | br#"{"status":"ok"}"#, |
| 1127 | None, |
| 1128 | ) |
| 1129 | .await |
| 1130 | .expect("mixed-case JSON content type"); |
| 1131 | |
| 1132 | assert_eq!(document.kind, DocumentKind::Text); |
| 1133 | assert_eq!(document.text, r#"{"status":"ok"}"#); |
| 1134 | } |
| 1135 | |
| 1136 | #[test] |
| 1137 | fn bom_wins_over_conflicting_transport_and_is_removed() { |
| 1138 | let mut utf8 = b"\xef\xbb\xbf".to_vec(); |
| 1139 | utf8.extend_from_slice("café".as_bytes()); |
| 1140 | assert_eq!( |
| 1141 | decode_response_body(&utf8, Some("text/html; charset=windows-1252"), true) |
| 1142 | .expect("UTF-8 BOM"), |
| 1143 | "café" |
| 1144 | ); |
| 1145 | |
| 1146 | let mut utf16 = vec![0xff, 0xfe]; |
| 1147 | for unit in "BOM 日本語".encode_utf16() { |
| 1148 | utf16.extend_from_slice(&unit.to_le_bytes()); |
| 1149 | } |
| 1150 | assert_eq!( |
| 1151 | decode_response_body(&utf16, Some("text/plain; charset=windows-1252"), false) |
| 1152 | .expect("UTF-16 BOM"), |
| 1153 | "BOM 日本語" |
| 1154 | ); |
| 1155 | } |
| 1156 | |
| 1157 | #[test] |
| 1158 | fn content_type_charset_is_exact_recognized_and_order_independent() { |
| 1159 | let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode("café"); |
| 1160 | for content_type in [ |
| 1161 | "text/plain; charset=windows-1252", |
| 1162 | "TEXT/PLAIN; boundary=x; CHARSET = \"windows-1252\"; q=1", |
| 1163 | "text/plain; q=1; charset='windows-1252'", |
| 1164 | ] { |
| 1165 | assert_eq!( |
| 1166 | decode_response_body(&bytes, Some(content_type), false).expect("declared charset"), |
| 1167 | "café", |
| 1168 | "{content_type}" |
| 1169 | ); |
| 1170 | } |
| 1171 | |
| 1172 | for malformed in [ |
| 1173 | "text/plain; note=charset=windows-1252", |
| 1174 | "text/plain; charset=\"windows-1252", |
| 1175 | "text/plain; charset=definitely-not-an-encoding", |
| 1176 | ] { |
| 1177 | let decoded = |
| 1178 | decode_response_body(&bytes, Some(malformed), false).expect("UTF-8 fallback"); |
| 1179 | assert!(decoded.contains('\u{fffd}'), "{malformed}: {decoded}"); |
| 1180 | } |
| 1181 | } |
| 1182 | |
| 1183 | #[test] |
| 1184 | fn invalid_header_falls_through_to_direct_and_legacy_html_meta() { |
| 1185 | let direct = r#"<html><head><meta charset="gbk"></head><body>中文</body></html>"#; |
| 1186 | let (direct_bytes, _, _) = encoding_rs::GBK.encode(direct); |
| 1187 | assert!( |
| 1188 | decode_response_body(&direct_bytes, Some("text/html; charset=not-real"), true,) |
| 1189 | .expect("direct meta") |
| 1190 | .contains("中文") |
| 1191 | ); |
| 1192 | |
| 1193 | let legacy = r#"<html><head><meta content="text/html; charset=windows-1252" http-equiv="Content-Type"></head><body>café</body></html>"#; |
| 1194 | let (legacy_bytes, _, _) = encoding_rs::WINDOWS_1252.encode(legacy); |
| 1195 | assert!( |
| 1196 | decode_response_body(&legacy_bytes, Some("text/html"), true) |
| 1197 | .expect("legacy meta") |
| 1198 | .contains("café") |
| 1199 | ); |
| 1200 | } |
| 1201 | |
| 1202 | #[test] |
| 1203 | fn recognized_transport_charset_beats_conflicting_meta() { |
| 1204 | let html = r#"<html><head><meta charset="shift_jis"></head><body>中文</body></html>"#; |
| 1205 | let (bytes, _, _) = encoding_rs::GBK.encode(html); |
| 1206 | let decoded = decode_response_body(&bytes, Some("text/html; charset=gbk"), true) |
| 1207 | .expect("transport charset"); |
| 1208 | assert!(decoded.contains("中文"), "{decoded}"); |
| 1209 | } |
| 1210 | |
| 1211 | #[test] |
| 1212 | fn html_prescan_ignores_comments_scripts_and_late_meta() { |
| 1213 | let cases = [ |
| 1214 | "<!-- <meta charset=windows-1252> --><html><body>café</body></html>".to_string(), |
| 1215 | "<script>\"<meta charset=windows-1252>\"</script><html><body>café</body></html>" |
| 1216 | .to_string(), |
| 1217 | format!( |
| 1218 | "<html><head>{}<meta charset=windows-1252></head><body>café</body></html>", |
| 1219 | " ".repeat(HTML_ENCODING_SNIFF_BYTES) |
| 1220 | ), |
| 1221 | ]; |
| 1222 | for html in cases { |
| 1223 | let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode(&html); |
| 1224 | let decoded = decode_response_body(&bytes, Some("text/html"), true) |
| 1225 | .expect("bounded HTML fallback"); |
| 1226 | assert!( |
| 1227 | decoded.contains('\u{fffd}'), |
| 1228 | "late/ignored meta changed decoding: {decoded}" |
| 1229 | ); |
| 1230 | } |
| 1231 | } |
| 1232 | |
| 1233 | #[test] |
| 1234 | fn non_html_bodies_never_sniff_meta_markup() { |
| 1235 | let plain = "literal <meta charset=windows-1252> café"; |
| 1236 | let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode(plain); |
| 1237 | for content_type in ["text/plain", "application/json"] { |
| 1238 | let decoded = decode_response_body(&bytes, Some(content_type), false) |
| 1239 | .expect("non-HTML UTF-8 fallback"); |
| 1240 | assert!(decoded.contains('\u{fffd}'), "{content_type}: {decoded}"); |
| 1241 | } |
| 1242 | } |
| 1243 | |
| 1244 | #[test] |
| 1245 | fn declared_gbk_shift_jis_and_windows_1252_decode_deterministically() { |
| 1246 | let cases = [ |
| 1247 | (encoding_rs::GBK, "中文", "gbk"), |
| 1248 | (encoding_rs::SHIFT_JIS, "日本語", "shift_jis"), |
| 1249 | (encoding_rs::WINDOWS_1252, "café", "windows-1252"), |
| 1250 | ]; |
| 1251 | for (encoding, text, label) in cases { |
| 1252 | let (bytes, _, had_errors) = encoding.encode(text); |
| 1253 | assert!(!had_errors, "fixture must be representable in {label}"); |
| 1254 | assert_eq!( |
| 1255 | decode_response_body(&bytes, Some(&format!("text/plain; charset={label}")), false,) |
| 1256 | .expect("decode declared encoding"), |
| 1257 | text |
| 1258 | ); |
| 1259 | } |
| 1260 | } |
| 1261 | |
| 1262 | #[test] |
| 1263 | fn nul_binary_is_rejected_but_utf16_bom_text_is_not() { |
| 1264 | let error = decode_response_body(b"PK\0\x03\x04archive", Some("text/plain"), false) |
| 1265 | .expect_err("NUL binary must fail"); |
| 1266 | assert!(error.to_string().contains("NUL bytes")); |
| 1267 | |
| 1268 | let bom_binary = b"\xef\xbb\xbfapparently text\0binary"; |
| 1269 | let error = decode_response_body(bom_binary, Some("text/plain"), false) |
| 1270 | .expect_err("a BOM must not bypass the NUL guard"); |
| 1271 | assert!(error.to_string().contains("NUL bytes")); |
| 1272 | |
| 1273 | let mut late_binary = vec![b'x'; 8_193]; |
| 1274 | late_binary.push(0); |
| 1275 | let error = decode_response_body(&late_binary, Some("text/plain"), false) |
| 1276 | .expect_err("a late NUL must not bypass the full-body guard"); |
| 1277 | assert!(error.to_string().contains("NUL bytes")); |
| 1278 | |
| 1279 | let utf16 = [0xff, 0xfe, b'O', 0, b'K', 0]; |
| 1280 | assert_eq!( |
| 1281 | decode_response_body(&utf16, Some("application/octet-stream"), false) |
| 1282 | .expect("BOM proves UTF-16 text"), |
| 1283 | "OK" |
| 1284 | ); |
| 1285 | |
| 1286 | let utf16_nul = [0xff, 0xfe, b'O', 0, 0, 0, b'K', 0]; |
| 1287 | let error = decode_response_body(&utf16_nul, Some("text/plain"), false) |
| 1288 | .expect_err("decoded UTF-16 NUL must remain binary"); |
| 1289 | assert!(error.to_string().contains("NUL bytes")); |
| 1290 | } |
| 1291 | |
| 1292 | #[tokio::test] |
| 1293 | async fn extensionless_html_sniff_skips_leading_comments_and_xml_declarations() { |
| 1294 | let cases = [ |
| 1295 | r#"<!-- deployment marker --><html><head><meta charset="windows-1252"><title>Café release notes</title></head><body><article><h1>Café release notes</h1><p>This extensionless page contains enough meaningful text for deterministic extraction.</p></article></body></html>"#, |
| 1296 | r#"<?xml version="1.0"?><!-- marker --><head><meta charset="windows-1252"><title>Café release notes</title></head><body><article><h1>Café release notes</h1><p>This extensionless page contains enough meaningful text for deterministic extraction.</p></article></body>"#, |
| 1297 | ]; |
| 1298 | for html in cases { |
| 1299 | let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode(html); |
| 1300 | let document = |
| 1301 | extract_document("https://example.com/extensionless", None, &bytes, None) |
| 1302 | .await |
| 1303 | .expect("leading declarations preserve extensionless HTML sniffing"); |
| 1304 | assert_eq!(document.kind, DocumentKind::Html); |
| 1305 | assert_eq!(document.title.as_deref(), Some("Café release notes")); |
| 1306 | } |
| 1307 | } |
| 1308 | |
| 1309 | #[tokio::test] |
| 1310 | async fn svg_requires_and_accepts_svg_markup_signature() { |
| 1311 | let svg = br#"<?xml version="1.0"?><svg xmlns="http://www.w3.org/2000/svg"></svg>"#; |
| 1312 | let document = extract_document( |
| 1313 | "https://example.com/diagram", |
| 1314 | Some("image/svg+xml"), |
| 1315 | svg, |
| 1316 | None, |
| 1317 | ) |
| 1318 | .await |
| 1319 | .expect("sniff svg"); |
| 1320 | assert_eq!(document.kind, DocumentKind::Media); |
| 1321 | assert_eq!(document.media_extension, Some("svg")); |
| 1322 | } |
| 1323 | |
| 1324 | async fn extract_html_fixture(url: &str, html: &str) -> ExtractedDocument { |
| 1325 | extract_document(url, Some("text/html"), html.as_bytes(), None) |
| 1326 | .await |
| 1327 | .expect("fixture must extract readable text") |
| 1328 | } |
| 1329 | |
| 1330 | #[tokio::test] |
| 1331 | async fn listing_page_keeps_every_article() { |
| 1332 | let html = r#"<html><body><nav>Home Blog About</nav><main> |
| 1333 | <article><h2>First post</h2><p>Alpha story body with plenty of words to read here.</p></article> |
| 1334 | <article><h2>Second post</h2><p>Bravo story body with plenty of words to read here.</p></article> |
| 1335 | <article><h2>Third post</h2><p>Charlie story body with plenty of words to read here.</p></article> |
| 1336 | </main></body></html>"#; |
| 1337 | let document = extract_html_fixture("https://blog.example/", html).await; |
| 1338 | for needle in [ |
| 1339 | "First post", |
| 1340 | "Alpha story", |
| 1341 | "Second post", |
| 1342 | "Bravo story", |
| 1343 | "Third post", |
| 1344 | "Charlie story", |
| 1345 | ] { |
| 1346 | assert!( |
| 1347 | document.markdown.contains(needle), |
| 1348 | "{needle} missing: {}", |
| 1349 | document.markdown |
| 1350 | ); |
| 1351 | } |
| 1352 | assert!(!document.markdown.contains("Home Blog About")); |
| 1353 | } |
| 1354 | |
| 1355 | #[tokio::test] |
| 1356 | async fn article_header_keeps_title_and_byline() { |
| 1357 | let html = r#"<html><body><header>Site logo Sign in Subscribe</header> |
| 1358 | <article><header><h1>Whales sing in dialects</h1><p class="byline">By Ada Lovelace</p></header> |
| 1359 | <p>Researchers recorded humpback song across three oceans and found regional variation.</p> |
| 1360 | </article></body></html>"#; |
| 1361 | let document = extract_html_fixture("https://news.example/whales", html).await; |
| 1362 | assert!( |
| 1363 | document.markdown.contains("Whales sing in dialects"), |
| 1364 | "{}", |
| 1365 | document.markdown |
| 1366 | ); |
| 1367 | assert!( |
| 1368 | document.markdown.contains("By Ada Lovelace"), |
| 1369 | "{}", |
| 1370 | document.markdown |
| 1371 | ); |
| 1372 | assert!(document.markdown.contains("regional variation")); |
| 1373 | assert!(!document.markdown.contains("Sign in Subscribe")); |
| 1374 | } |
| 1375 | |
| 1376 | #[tokio::test] |
| 1377 | async fn page_wrapped_in_a_form_is_not_mistaken_for_a_javascript_shell() { |
| 1378 | let html = r#"<html><body><form method="post" action="./Default.aspx" id="form1"> |
| 1379 | <input type="hidden" name="__VIEWSTATE" value="dDwtMTA4MzE0MjEwNTs7Pg==" /> |
| 1380 | <div class="content"><h1>Quarterly report</h1> |
| 1381 | <p>Revenue grew in every region this quarter, led by the northern division.</p></div> |
| 1382 | <select name="year"><option>2025</option><option>2026</option></select> |
| 1383 | <button type="submit">Go</button> |
| 1384 | </form></body></html>"#; |
| 1385 | let document = extract_html_fixture("https://legacy.example/Default.aspx", html).await; |
| 1386 | assert!( |
| 1387 | document.markdown.contains("Quarterly report"), |
| 1388 | "{}", |
| 1389 | document.markdown |
| 1390 | ); |
| 1391 | assert!(document.markdown.contains("northern division")); |
| 1392 | assert!(!document.markdown.contains("VIEWSTATE")); |
| 1393 | assert!( |
| 1394 | !document.markdown.contains("2026"), |
| 1395 | "form controls are stripped" |
| 1396 | ); |
| 1397 | } |
| 1398 | |
| 1399 | #[tokio::test] |
| 1400 | async fn arxiv_abstract_page_extracts_title_and_abstract() { |
| 1401 | // Trimmed from the shape of https://arxiv.org/abs/1706.03762 (2026-09). |
| 1402 | let html = r##"<!DOCTYPE html><html lang="en"><head><title>[1706.03762] Attention Is All You Need</title> |
| 1403 | <script>window.MathJax = {};</script></head> |
| 1404 | <body ><div class="flex-wrap-footer"><a href="#content" class="ds-skip-link">Skip to main content</a> |
| 1405 | <header class="ds-site-header"><a href="https://arxiv.org/">archive home</a> |
| 1406 | <button type="button" id="ds-nav-toggle">Open menu</button> |
| 1407 | <nav class="ds-site-header-nav"><a href="https://arxiv.org/search">Search</a><a href="https://arxiv.org/login">Log in</a></nav> |
| 1408 | </header> |
| 1409 | <div class="arxiv-search-overlay" hidden><form method="GET" action="https://arxiv.org/search"> |
| 1410 | <label for="q">Search arXiv</label><input type="text" name="query" id="q"></form></div> |
| 1411 | <main><div id="content"><!-- rdf:RDF <rdf:Description dc:title="Attention Is All You Need" /> --> |
| 1412 | <div id="abs-outer"><div class="leftcolumn"><div class="subheader"><h1>Computer Science > Computation and Language</h1></div> |
| 1413 | <div id="abs"><div class="dateline">[Submitted on 12 Jun 2017 (<a href="/abs/1706.03762v1">v1</a>)]</div> |
| 1414 | <h1 class="title mathjax"><span class="descriptor">Title:</span>Attention Is All You Need</h1> |
| 1415 | <div class="authors"><span class="descriptor">Authors:</span><a href="/a/vaswani_a_1">Ashish Vaswani</a></div> |
| 1416 | <blockquote class="abstract mathjax"><span class="descriptor">Abstract:</span>The dominant sequence transduction models are based on complex recurrent or convolutional neural networks.</blockquote> |
| 1417 | <script type="text/javascript" language="javascript">mathjaxToggle();</script> |
| 1418 | </div></div></div></div></main> |
| 1419 | <footer><a href="https://info.arxiv.org/help/contact.html">Contact</a></footer></div></body></html>"##; |
| 1420 | let document = extract_html_fixture("https://arxiv.org/abs/1706.03762", html).await; |
| 1421 | assert!( |
| 1422 | document.markdown.contains("Attention Is All You Need"), |
| 1423 | "{}", |
| 1424 | document.markdown |
| 1425 | ); |
| 1426 | assert!( |
| 1427 | document.markdown.contains("dominant sequence transduction"), |
| 1428 | "{}", |
| 1429 | document.markdown |
| 1430 | ); |
| 1431 | assert!(document.markdown.contains("Ashish Vaswani")); |
| 1432 | assert!(!document.markdown.contains("mathjaxToggle")); |
| 1433 | assert!(!document.markdown.contains("Log in")); |
| 1434 | } |
| 1435 | } |
| 1436 | |
| 1437 | #[cfg(test)] |
| 1438 | #[path = "extract_host_tests.rs"] |
| 1439 | mod host_tests; |
| 1440 |