返回 CodeWhale
extract.rs
根目录 / crates / tui / src / tools / web / extract.rs
1 //! Content-type routing and readable-document extraction for web tools.
2 //!
3 //! Networking deliberately lives elsewhere. This module accepts already
4 //! fetched bytes and turns them into one normalized document so `fetch_url`
5 //! and `web.run` cannot disagree about HTML, Markdown, PDF, or media handling.
6
7 use super::adapter::{AdapterFailure, AdapterResult};
8 use std::sync::OnceLock;
9
10 use encoding_rs::{Encoding, UTF_8, UTF_16BE, UTF_16LE};
11 use regex::Regex;
12
13 use crate::tools::spec::ToolError;
14
15 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
16 pub(crate) enum DocumentKind {
17 Html,
18 Markdown,
19 Text,
20 Pdf,
21 Media,
22 }
23
24 #[derive(Debug, Clone)]
25 pub(crate) struct ExtractedDocument {
26 pub(crate) kind: DocumentKind,
27 pub(crate) title: Option<String>,
28 pub(crate) text: String,
29 pub(crate) markdown: String,
30 /// Readability-cleaned HTML. `web.run` consumes this to retain clickable
31 /// links while avoiding page chrome and consent-banner noise.
32 pub(crate) cleaned_html: Option<String>,
33 pub(crate) pdf_pages: Option<Vec<Vec<String>>>,
34 /// Validated extension for image/audio/video artifacts.
35 pub(crate) media_extension: Option<&'static str>,
36 }
37
38 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
39 struct MediaSignature {
40 extension: &'static str,
41 family: MediaFamily,
42 }
43
44 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
45 enum MediaFamily {
46 Image,
47 Audio,
48 Video,
49 }
50
51 static TITLE_RE: OnceLock<Regex> = OnceLock::new();
52 static FALLBACK_RE: OnceLock<[Regex; 3]> = OnceLock::new();
53 static PAGE_CHROME_RE: OnceLock<Regex> = OnceLock::new();
54 static PAGE_HEADER_RE: OnceLock<Regex> = OnceLock::new();
55 static TAG_RE: OnceLock<Regex> = OnceLock::new();
56 static WHITESPACE_RE: OnceLock<Regex> = OnceLock::new();
57
58 /// HTML's encoding declaration prescan is intentionally small. Keeping the
59 /// bound here prevents a late body string, script, or injected fragment from
60 /// changing how an already-started document is decoded.
61 const HTML_ENCODING_SNIFF_BYTES: usize = 1_024;
62
63 pub(crate) async fn extract_document(
64 url: &str,
65 content_type: Option<&str>,
66 bytes: &[u8],
67 context: Option<&super::super::spec::ToolContext>,
68 ) -> AdapterResult<ExtractedDocument> {
69 extract_document_with_pdf_command(
70 url,
71 content_type,
72 bytes,
73 super::super::pdf::PdfTextCommand::system(context),
74 )
75 .await
76 }
77
78 pub(crate) async fn extract_document_with_pdf_command(
79 url: &str,
80 content_type: Option<&str>,
81 bytes: &[u8],
82 pdf_command: super::super::pdf::PdfTextCommand<'_>,
83 ) -> AdapterResult<ExtractedDocument> {
84 let context = pdf_command.context();
85 let declared = normalized_content_type(content_type);
86 let declared = declared.as_deref();
87
88 if bytes.is_empty() {
89 return Ok(ExtractedDocument {
90 kind: DocumentKind::Text,
91 title: None,
92 text: String::new(),
93 markdown: String::new(),
94 cleaned_html: None,
95 pdf_pages: None,
96 media_extension: None,
97 });
98 }
99
100 if validate_pdf_response(url, content_type, bytes)? {
101 return extract_pdf(bytes, pdf_command).await.map_err(|error| {
102 if context
103 .is_some_and(|context| context.features.enabled(crate::features::Feature::PdfHost))
104 {
105 AdapterFailure::host(error)
106 } else {
107 error.into()
108 }
109 });
110 }
111
112 if let Some(signature) = sniff_media(bytes) {
113 if let Some(declared_family) = declared_media_family(declared)
114 && declared_family != signature.family
115 {
116 return Err((ToolError::execution_failed(format!(
117 "Response media type `{}` did not match its bytes",
118 declared.unwrap_or("unknown")
119 )))
120 .into());
121 }
122 return Ok(ExtractedDocument {
123 kind: DocumentKind::Media,
124 title: None,
125 text: String::new(),
126 markdown: String::new(),
127 cleaned_html: None,
128 pdf_pages: None,
129 media_extension: Some(signature.extension),
130 });
131 }
132
133 if declared_media_family(declared).is_some() {
134 return Err((ToolError::execution_failed(format!(
135 "Response claimed media type `{}`, but its bytes did not match a supported media signature",
136 declared.unwrap_or("unknown")
137 ))).into());
138 }
139
140 let sniff_html = should_sniff_html_encoding(declared, url, bytes);
141 let body = decode_response_body(bytes, content_type, sniff_html)?;
142 if sniff_html || is_html(declared, url, &body) {
143 if let Some(context) = context.filter(|context| {
144 context
145 .features
146 .enabled(crate::features::Feature::WebExtractHost)
147 }) {
148 return extract_html_with_host(url, &body, context).await;
149 }
150 return extract_html(url, &body).map_err(Into::into);
151 }
152 if is_markdown(declared, url) {
153 return Ok(ExtractedDocument {
154 kind: DocumentKind::Markdown,
155 title: markdown_title(&body),
156 text: body.clone(),
157 markdown: body,
158 cleaned_html: None,
159 pdf_pages: None,
160 media_extension: None,
161 });
162 }
163 if is_textual(declared, url) {
164 return Ok(ExtractedDocument {
165 kind: DocumentKind::Text,
166 title: None,
167 text: body.clone(),
168 markdown: body,
169 cleaned_html: None,
170 pdf_pages: None,
171 media_extension: None,
172 });
173 }
174
175 Err((ToolError::execution_failed(format!(
176 "Unsupported binary response type `{}`; use a dedicated download tool",
177 declared.unwrap_or("unknown")
178 )))
179 .into())
180 }
181
182 pub(crate) fn validate_pdf_response(
183 url: &str,
184 content_type: Option<&str>,
185 bytes: &[u8],
186 ) -> Result<bool, ToolError> {
187 let declared = normalized_content_type(content_type);
188 let declared = declared.as_deref();
189 let signed = looks_like_pdf(bytes);
190 if signed && declared_media_family(declared).is_some() {
191 return Err(ToolError::execution_failed(format!(
192 "Response media type `{}` did not match its PDF bytes",
193 declared.unwrap_or("unknown")
194 )));
195 }
196 let claimed = signed || declared == Some("application/pdf") || url_is_pdf(url);
197 if claimed && !signed {
198 return Err(ToolError::execution_failed(
199 "Response claimed to be a PDF, but its bytes did not contain a PDF signature",
200 ));
201 }
202 Ok(claimed)
203 }
204
205 async fn extract_html_with_host(
206 url: &str,
207 html: &str,
208 context: &super::super::spec::ToolContext,
209 ) -> AdapterResult<ExtractedDocument> {
210 #[derive(serde::Deserialize)]
211 #[serde(deny_unknown_fields)]
212 struct Choice {
213 kind: String,
214 candidate: Option<u8>,
215 }
216 let parsed_url = reqwest::Url::parse(url)
217 .map_err(|error| ToolError::invalid_input(format!("invalid URL: {error}")))?;
218 let candidates = main_html_candidates(html);
219 let facts=candidates.iter().map(|(id,html)| {
220 let text=html_to_plain_text(html);
221 serde_json::json!({"id":id,"non_whitespace":text.chars().filter(|value|!value.is_whitespace()).count(),"words":text.split_whitespace().count()})
222 }).collect::<Vec<_>>();
223 let budget = context
224 .turn_deadline
225 .map(|deadline| deadline.saturating_duration_since(tokio::time::Instant::now()))
226 .unwrap_or(std::time::Duration::from_secs(15));
227 let choice: Choice = super::adapter::transform(
228 crate::extension_host::StockOperation::WebExtract,
229 serde_json::json!({"candidates":facts}),
230 context,
231 budget,
232 )
233 .await?;
234 if choice.kind != "web_extract" {
235 return Err(AdapterFailure::host(ToolError::execution_failed(
236 "Web Host returned a malformed HTML region choice",
237 )));
238 }
239 let expected = candidates
240 .iter()
241 .find(|(_, html)| meaningful_html(html))
242 .map(|(id, _)| *id);
243 if choice.candidate != expected {
244 return Err(AdapterFailure::host(ToolError::execution_failed(
245 "Web Host changed mandatory readable-region order or meaningfulness",
246 )));
247 }
248 let Some(selected) = choice.candidate else {
249 return Err(js_required_error(url).into());
250 };
251 let cleaned = candidates
252 .into_iter()
253 .find(|(id, _)| *id == selected)
254 .map(|(_, html)| html)
255 .ok_or_else(|| {
256 AdapterFailure::host(ToolError::execution_failed(
257 "Web Host returned an unknown HTML region",
258 ))
259 })?;
260 document_from_html_region(url, &parsed_url, html_title(html), cleaned).map_err(Into::into)
261 }
262
263 fn extract_html(url: &str, html: &str) -> Result<ExtractedDocument, ToolError> {
264 let parsed_url = reqwest::Url::parse(url)
265 .map_err(|err| ToolError::invalid_input(format!("invalid URL: {err}")))?;
266 let original_title = html_title(html);
267
268 // Readability-based extraction was removed to consolidate the HTML
269 // pipeline onto a single stack (htmd 0.5 + html5ever 0.38). The
270 // previous dual-stack (readability 0.3 / html5ever 0.26 + htmd / 0.38)
271 // compiled two incompatible html5ever/markup5ever trees. The fallback
272 // main-content regex retains the meaningful-content signal used by the
273 // tests (≥32 non-whitespace chars, ≥5 words) without the duplicate tree.
274 let cleaned_html = fallback_main_html(html).ok_or_else(|| js_required_error(url))?;
275 document_from_html_region(url, &parsed_url, original_title, cleaned_html)
276 }
277
278 fn document_from_html_region(
279 url: &str,
280 parsed_url: &reqwest::Url,
281 original_title: Option<String>,
282 cleaned_html: String,
283 ) -> Result<ExtractedDocument, ToolError> {
284 let markdown = html_to_markdown_with_base_url(&cleaned_html, parsed_url).map_err(|err| {
285 ToolError::execution_failed(format!(
286 "Failed to convert readable HTML to Markdown: {err}"
287 ))
288 })?;
289 let text = html_to_plain_text(&cleaned_html);
290
291 if !meaningful_text(&text) && !meaningful_text(&markdown) {
292 return Err(js_required_error(url));
293 }
294
295 let title = original_title;
296
297 Ok(ExtractedDocument {
298 kind: DocumentKind::Html,
299 title,
300 text,
301 markdown,
302 cleaned_html: Some(cleaned_html),
303 pdf_pages: None,
304 media_extension: None,
305 })
306 }
307
308 /// Resolve relative anchors in `htmd`'s parsed DOM, not with an HTML regex.
309 /// Absolute, fragment, non-HTTP, and malformed destinations fall through to
310 /// the built-in handler unchanged.
311 fn html_to_markdown_with_base_url(
312 html: &str,
313 base_url: &reqwest::Url,
314 ) -> Result<String, std::io::Error> {
315 let base_url = base_url.clone();
316 htmd::HtmlToMarkdown::builder()
317 .add_handler(
318 vec!["a"],
319 move |handlers: &dyn htmd::element_handler::Handlers, element: htmd::Element<'_>| {
320 let href = element.attrs.iter().find_map(|attr| {
321 (attr.name.local.as_ref() == "href").then(|| attr.value.to_string())
322 });
323 let Some(destination) = href
324 .as_deref()
325 .and_then(|href| resolve_relative_http_href(&base_url, href))
326 else {
327 return handlers.fallback(element);
328 };
329 let content = handlers.walk_children(element.node).content;
330 let trailing = &content[content.trim_end().len()..];
331 let destination = destination.replace('(', "\\(").replace(')', "\\)");
332 let title = element
333 .attrs
334 .iter()
335 .find_map(|attr| {
336 (attr.name.local.as_ref() == "title").then(|| {
337 attr.value
338 .split_whitespace()
339 .collect::<Vec<_>>()
340 .join(" ")
341 .replace('"', "\\\"")
342 })
343 })
344 .map_or_else(String::new, |title| format!(" \"{title}\""));
345 Some(format!("[{}]({destination}{title}){trailing}", content.trim()).into())
346 },
347 )
348 .build()
349 .convert(html)
350 }
351
352 fn resolve_relative_http_href(base_url: &reqwest::Url, href: &str) -> Option<String> {
353 if !matches!(base_url.scheme(), "http" | "https") {
354 return None;
355 }
356
357 let href = href.trim();
358 if href.is_empty() || href.starts_with('#') || reqwest::Url::parse(href).is_ok() {
359 return None;
360 }
361
362 base_url.join(href).ok().map(Into::into)
363 }
364
365 /// Pick the readable region of a page with bounded regexes.
366 ///
367 /// Order: every `<article>` (listing, news and forum pages carry one per
368 /// card), then `<main>`, then `<body>`. Page chrome is removed from the chosen
369 /// region; a `<header>` is chrome only at body level, because inside an
370 /// article or `<main>` it carries the title and byline. Forms are unwrapped,
371 /// not dropped: ASP.NET-style pages wrap the whole body in one `<form>`, so
372 /// only the controls themselves are stripped.
373 ///
374 /// Known limits: the lazy regexes do not balance nested same-name elements
375 /// (an `<article>` inside an `<article>` ends at the inner close tag), and no
376 /// JavaScript runs, so a client-rendered shell still yields nothing.
377 fn main_html_candidates(html: &str) -> Vec<(u8, String)> {
378 let [article, main, body] = FALLBACK_RE.get_or_init(|| {
379 ["article", "main", "body"].map(|tag| {
380 Regex::new(&format!(r"(?is)<{tag}(?:\s[^>]*)?>(.*?)</{tag}\s*>"))
381 .expect("fallback element regex")
382 })
383 });
384 let articles = article
385 .captures_iter(html)
386 .filter_map(|capture| capture.get(1))
387 .map(|content| strip_page_chrome(content.as_str(), false))
388 .filter(|content| !html_to_plain_text(content).is_empty())
389 .collect::<Vec<_>>()
390 .join("\n");
391 let mut candidates = vec![(0, articles)];
392 for (id, re, strip_header) in [(1, main, false), (2, body, true)] {
393 if let Some(content) = re.captures(html).and_then(|capture| capture.get(1)) {
394 candidates.push((id, strip_page_chrome(content.as_str(), strip_header)));
395 }
396 }
397 candidates
398 }
399
400 fn fallback_main_html(html: &str) -> Option<String> {
401 main_html_candidates(html)
402 .into_iter()
403 .find_map(|(_, html)| meaningful_html(&html).then_some(html))
404 }
405
406 fn strip_page_chrome(html: &str, strip_header: bool) -> String {
407 let page_chrome = PAGE_CHROME_RE.get_or_init(|| {
408 Regex::new(concat!(
409 r"(?is)(?:<script(?:\s[^>]*)?>.*?</script\s*>",
410 r"|<style(?:\s[^>]*)?>.*?</style\s*>",
411 r"|<noscript(?:\s[^>]*)?>.*?</noscript\s*>",
412 r"|<nav(?:\s[^>]*)?>.*?</nav\s*>",
413 r"|<footer(?:\s[^>]*)?>.*?</footer\s*>",
414 r"|<aside(?:\s[^>]*)?>.*?</aside\s*>",
415 // Form controls, and the form tags themselves (unwrapped).
416 r"|<select(?:\s[^>]*)?>.*?</select\s*>",
417 r"|<textarea(?:\s[^>]*)?>.*?</textarea\s*>",
418 r"|<button(?:\s[^>]*)?>.*?</button\s*>",
419 r"|<input(?:\s[^>]*)?>",
420 r"|</?form(?:\s[^>]*)?>)",
421 ))
422 .expect("page chrome regex")
423 });
424 let cleaned = page_chrome.replace_all(html, "");
425 if !strip_header {
426 return cleaned.into_owned();
427 }
428 PAGE_HEADER_RE
429 .get_or_init(|| {
430 Regex::new(r"(?is)<header(?:\s[^>]*)?>.*?</header\s*>").expect("page header regex")
431 })
432 .replace_all(&cleaned, "")
433 .into_owned()
434 }
435
436 fn meaningful_html(html: &str) -> bool {
437 meaningful_text(&html_to_plain_text(html))
438 }
439
440 fn meaningful_text(text: &str) -> bool {
441 text.chars().filter(|ch| !ch.is_whitespace()).count() >= 32
442 && text.split_whitespace().count() >= 5
443 }
444
445 fn html_to_plain_text(html: &str) -> String {
446 let without_tags = TAG_RE
447 .get_or_init(|| Regex::new(r"(?s)<[^>]+>").expect("tag regex"))
448 .replace_all(html, " ");
449 normalize_text(&decode_common_entities(&without_tags))
450 }
451
452 fn normalize_text(text: &str) -> String {
453 WHITESPACE_RE
454 .get_or_init(|| Regex::new(r"\s+").expect("whitespace regex"))
455 .replace_all(text.trim(), " ")
456 .into_owned()
457 }
458
459 fn decode_common_entities(value: &str) -> String {
460 value
461 .replace("&nbsp;", " ")
462 .replace("&amp;", "&")
463 .replace("&lt;", "<")
464 .replace("&gt;", ">")
465 .replace("&quot;", "\"")
466 .replace("&#39;", "'")
467 }
468
469 fn html_title(html: &str) -> Option<String> {
470 let capture = TITLE_RE
471 .get_or_init(|| {
472 Regex::new(r"(?is)<title(?:\s[^>]*)?>(.*?)</title\s*>").expect("title regex")
473 })
474 .captures(html)?;
475 let title = normalize_text(&decode_common_entities(capture.get(1)?.as_str()));
476 (!title.is_empty()).then_some(title)
477 }
478
479 fn markdown_title(body: &str) -> Option<String> {
480 body.lines().find_map(|line| {
481 let title = line.trim().strip_prefix("# ")?.trim();
482 (!title.is_empty()).then(|| title.to_string())
483 })
484 }
485
486 /// Stable prefix for "the response parsed, but carried no readable body".
487 ///
488 /// The fetch pipeline matches on this to decide whether a cache-busting
489 /// re-fetch is worth one more request, and to attach the role-aware recovery
490 /// text. Extraction itself has no `ToolContext`, so it cannot know which
491 /// escalation the calling role actually owns; it states the fact and leaves
492 /// the remedy to [`super::fetch`].
493 pub(crate) const JS_SHELL_MARKER: &str = "No readable page content was found at";
494
495 /// Whether `error` is the JS-shell extraction failure (a parsed response whose
496 /// body held no readable content), as opposed to a transport or type failure.
497 pub(crate) fn is_js_shell_error(error: &ToolError) -> bool {
498 error.to_string().contains(JS_SHELL_MARKER)
499 }
500
501 fn js_required_error(url: &str) -> ToolError {
502 ToolError::execution_failed(format!(
503 "{JS_SHELL_MARKER} {url}; the response parsed but its body held no readable content, so the page may require JavaScript."
504 ))
505 }
506
507 /// Decode one response body without guessing from language statistics.
508 ///
509 /// Precedence is receipt-grade and deterministic: BOM, recognized transport
510 /// charset, an HTML-only bounded meta prescan, then UTF-8. Unknown transport
511 /// labels deliberately fall through to a valid HTML declaration. `html_sniff`
512 /// must come from MIME/URL/ASCII markup evidence; JSON and plain text callers
513 /// pass `false`, so a body string cannot impersonate an HTML declaration.
514 pub(crate) fn decode_response_body(
515 bytes: &[u8],
516 content_type: Option<&str>,
517 html_sniff: bool,
518 ) -> Result<String, ToolError> {
519 if let Some((encoding, bom_len)) = Encoding::for_bom(bytes) {
520 let (decoded, _) = encoding.decode_without_bom_handling(&bytes[bom_len..]);
521 reject_binary_nul(bytes, encoding, &decoded)?;
522 return Ok(decoded.into_owned());
523 }
524
525 let transport_encoding = content_type.and_then(content_type_encoding);
526 let encoding = transport_encoding
527 .or_else(|| html_sniff.then(|| html_meta_encoding(bytes)).flatten())
528 .unwrap_or(UTF_8);
529 let (decoded, _) = encoding.decode_without_bom_handling(bytes);
530 reject_binary_nul(bytes, encoding, &decoded)?;
531 Ok(decoded.into_owned())
532 }
533
534 fn reject_binary_nul(
535 bytes: &[u8],
536 encoding: &'static Encoding,
537 decoded: &str,
538 ) -> Result<(), ToolError> {
539 // UTF-16 uses zero bytes structurally for many characters, so inspect its
540 // decoded scalar values. Every other encoding must be NUL-free across the
541 // complete response; neither a BOM nor a late byte may bypass the guard.
542 let contains_nul = if encoding == UTF_16LE || encoding == UTF_16BE {
543 decoded.contains('\0')
544 } else {
545 bytes.contains(&0)
546 };
547 if contains_nul {
548 return Err(ToolError::execution_failed(
549 "Unsupported binary response contained NUL bytes",
550 ));
551 }
552 Ok(())
553 }
554
555 /// Parse only exact semicolon-delimited `charset` parameters. A random
556 /// `charset=` substring inside another parameter is not transport authority.
557 fn content_type_encoding(value: &str) -> Option<&'static Encoding> {
558 value.split(';').skip(1).find_map(|parameter| {
559 let (name, raw_value) = parameter.split_once('=')?;
560 if !name.trim().eq_ignore_ascii_case("charset") {
561 return None;
562 }
563 let value = raw_value.trim();
564 let value = match (value.as_bytes().first(), value.as_bytes().last()) {
565 (Some(b'"'), Some(b'"')) | (Some(b'\''), Some(b'\'')) if value.len() >= 2 => {
566 &value[1..value.len() - 1]
567 }
568 _ if value.contains('"') || value.contains('\'') => return None,
569 _ => value,
570 };
571 let label = value.trim();
572 (!label.is_empty())
573 .then(|| Encoding::for_label(label.as_bytes()))
574 .flatten()
575 })
576 }
577
578 fn should_sniff_html_encoding(content_type: Option<&str>, url: &str, bytes: &[u8]) -> bool {
579 match content_type {
580 Some("text/html" | "application/xhtml+xml") => true,
581 // Explicit non-HTML text and structured formats never consult markup
582 // embedded in their body.
583 Some(value) if value.starts_with("text/") || is_structured_text_type(value) => false,
584 Some("application/octet-stream") | None => {
585 url_path_ends_with(url, &[".html", ".htm"]) || looks_like_html_bytes(bytes)
586 }
587 Some(_) => false,
588 }
589 }
590
591 fn is_structured_text_type(content_type: &str) -> bool {
592 content_type.contains("json")
593 || content_type.contains("xml")
594 || content_type.contains("yaml")
595 || content_type.contains("javascript")
596 }
597
598 fn looks_like_html_bytes(bytes: &[u8]) -> bool {
599 let start = Encoding::for_bom(bytes).map_or(0, |(_, length)| length);
600 let end = bytes
601 .len()
602 .min(start.saturating_add(HTML_ENCODING_SNIFF_BYTES));
603 let ascii = ascii_lowercase_projection(&bytes[start..end]);
604 let Some(prefix) = html_prefix_after_leading_declarations(&ascii) else {
605 return false;
606 };
607 prefix.starts_with("<!doctype html")
608 || prefix.starts_with("<html")
609 || prefix.starts_with("<head")
610 || prefix.starts_with("<meta")
611 }
612
613 fn html_prefix_after_leading_declarations(mut prefix: &str) -> Option<&str> {
614 loop {
615 prefix = prefix.trim_start();
616 if let Some(comment) = prefix.strip_prefix("<!--") {
617 let end = comment.find("-->")?;
618 prefix = &comment[end + 3..];
619 continue;
620 }
621 if let Some(declaration) = prefix.strip_prefix("<?xml") {
622 let end = declaration.find("?>")?;
623 prefix = &declaration[end + 2..];
624 continue;
625 }
626 return Some(prefix);
627 }
628 }
629
630 fn html_meta_encoding(bytes: &[u8]) -> Option<&'static Encoding> {
631 let sniff_len = bytes.len().min(HTML_ENCODING_SNIFF_BYTES);
632 let html = ascii_lowercase_projection(&bytes[..sniff_len]);
633 let mut cursor = 0usize;
634
635 while let Some(relative) = html[cursor..].find('<') {
636 let start = cursor + relative;
637 if html[start..].starts_with("<!--") {
638 cursor = html[start + 4..]
639 .find("-->")
640 .map_or(html.len(), |end| start + 4 + end + 3);
641 continue;
642 }
643 if tag_starts_at(&html, start, "script") || tag_starts_at(&html, start, "style") {
644 let name = if tag_starts_at(&html, start, "script") {
645 "script"
646 } else {
647 "style"
648 };
649 let close = format!("</{name}");
650 cursor = html[start..]
651 .find(&close)
652 .and_then(|close_start| {
653 html[start + close_start..]
654 .find('>')
655 .map(|end| start + close_start + end + 1)
656 })
657 .unwrap_or(html.len());
658 continue;
659 }
660 if !tag_starts_at(&html, start, "meta") {
661 cursor = start + 1;
662 continue;
663 }
664 let relative_end = html[start..].find('>')?;
665 let end = start + relative_end + 1;
666 let tag = &html[start..end];
667 if let Some(label) = html_attribute_value(tag, "charset")
668 && let Some(encoding) = Encoding::for_label(label.as_bytes())
669 {
670 return Some(normalize_meta_encoding(encoding));
671 }
672 let is_content_type = html_attribute_value(tag, "http-equiv")
673 .is_some_and(|value| value.eq_ignore_ascii_case("content-type"));
674 if is_content_type
675 && let Some(content) = html_attribute_value(tag, "content")
676 && let Some(encoding) = content_type_encoding(&content)
677 {
678 return Some(normalize_meta_encoding(encoding));
679 }
680 cursor = end;
681 }
682 None
683 }
684
685 fn normalize_meta_encoding(encoding: &'static Encoding) -> &'static Encoding {
686 if encoding == UTF_16LE || encoding == UTF_16BE {
687 UTF_8
688 } else {
689 encoding
690 }
691 }
692
693 fn ascii_lowercase_projection(bytes: &[u8]) -> String {
694 bytes
695 .iter()
696 .map(|byte| {
697 if byte.is_ascii() {
698 char::from(byte.to_ascii_lowercase())
699 } else {
700 ' '
701 }
702 })
703 .collect()
704 }
705
706 fn tag_starts_at(html: &str, start: usize, name: &str) -> bool {
707 let Some(after_name) = html.get(start + 1 + name.len()..) else {
708 return false;
709 };
710 html[start + 1..].starts_with(name)
711 && after_name
712 .chars()
713 .next()
714 .is_some_and(|ch| ch.is_ascii_whitespace() || matches!(ch, '/' | '>'))
715 }
716
717 fn html_attribute_value(tag: &str, wanted: &str) -> Option<String> {
718 let bytes = tag.as_bytes();
719 let mut cursor = 1usize;
720 while cursor < bytes.len() && !bytes[cursor].is_ascii_whitespace() && bytes[cursor] != b'>' {
721 cursor += 1;
722 }
723 while cursor < bytes.len() {
724 while cursor < bytes.len() && (bytes[cursor].is_ascii_whitespace() || bytes[cursor] == b'/')
725 {
726 cursor += 1;
727 }
728 if cursor >= bytes.len() || bytes[cursor] == b'>' {
729 break;
730 }
731 let name_start = cursor;
732 while cursor < bytes.len()
733 && !bytes[cursor].is_ascii_whitespace()
734 && !matches!(bytes[cursor], b'=' | b'/' | b'>')
735 {
736 cursor += 1;
737 }
738 let name = &tag[name_start..cursor];
739 while cursor < bytes.len() && bytes[cursor].is_ascii_whitespace() {
740 cursor += 1;
741 }
742 if cursor >= bytes.len() || bytes[cursor] != b'=' {
743 continue;
744 }
745 cursor += 1;
746 while cursor < bytes.len() && bytes[cursor].is_ascii_whitespace() {
747 cursor += 1;
748 }
749 if cursor >= bytes.len() {
750 break;
751 }
752 let (value_start, value_end) = if matches!(bytes[cursor], b'"' | b'\'') {
753 let quote = bytes[cursor];
754 cursor += 1;
755 let start = cursor;
756 while cursor < bytes.len() && bytes[cursor] != quote {
757 cursor += 1;
758 }
759 let end = cursor;
760 cursor = cursor.saturating_add(1);
761 (start, end)
762 } else {
763 let start = cursor;
764 while cursor < bytes.len()
765 && !bytes[cursor].is_ascii_whitespace()
766 && bytes[cursor] != b'>'
767 {
768 cursor += 1;
769 }
770 (start, cursor)
771 };
772 if name.eq_ignore_ascii_case(wanted) {
773 return Some(tag[value_start..value_end].trim().to_string());
774 }
775 }
776 None
777 }
778
779 fn normalized_content_type(content_type: Option<&str>) -> Option<String> {
780 content_type
781 .and_then(|value| value.split(';').next())
782 .map(str::trim)
783 .filter(|value| !value.is_empty())
784 .map(str::to_ascii_lowercase)
785 }
786
787 fn is_html(content_type: Option<&str>, url: &str, body: &str) -> bool {
788 matches!(content_type, Some("text/html" | "application/xhtml+xml"))
789 || url_path_ends_with(url, &[".html", ".htm"])
790 || {
791 let prefix = body.trim_start().chars().take(64).collect::<String>();
792 let prefix = prefix.to_ascii_lowercase();
793 prefix.contains("<!doctype html") || prefix.contains("<html")
794 }
795 }
796
797 fn is_markdown(content_type: Option<&str>, url: &str) -> bool {
798 matches!(
799 content_type,
800 Some("text/markdown" | "text/x-markdown" | "application/markdown")
801 ) || url_path_ends_with(url, &[".md", ".markdown"])
802 }
803
804 fn is_textual(content_type: Option<&str>, url: &str) -> bool {
805 content_type.is_some_and(|value| {
806 value.starts_with("text/")
807 || value.contains("json")
808 || value.contains("xml")
809 || value.contains("yaml")
810 || value.contains("javascript")
811 || value == "application/sql"
812 }) || url_path_ends_with(
813 url,
814 &[
815 ".txt", ".json", ".jsonl", ".xml", ".yaml", ".yml", ".csv", ".tsv", ".rs", ".py",
816 ".js", ".ts", ".toml",
817 ],
818 )
819 }
820
821 fn url_is_pdf(url: &str) -> bool {
822 url_path_ends_with(url, &[".pdf"])
823 }
824
825 fn url_path_ends_with(url: &str, extensions: &[&str]) -> bool {
826 reqwest::Url::parse(url)
827 .ok()
828 .map(|parsed| parsed.path().to_ascii_lowercase())
829 .is_some_and(|path| extensions.iter().any(|extension| path.ends_with(extension)))
830 }
831
832 fn looks_like_pdf(bytes: &[u8]) -> bool {
833 bytes.starts_with(b"%PDF-")
834 }
835
836 fn declared_media_family(content_type: Option<&str>) -> Option<MediaFamily> {
837 let content_type = content_type?;
838 if content_type.starts_with("image/") {
839 Some(MediaFamily::Image)
840 } else if content_type.starts_with("audio/") {
841 Some(MediaFamily::Audio)
842 } else if content_type.starts_with("video/") {
843 Some(MediaFamily::Video)
844 } else {
845 None
846 }
847 }
848
849 fn sniff_media(bytes: &[u8]) -> Option<MediaSignature> {
850 let trimmed = bytes
851 .iter()
852 .position(|byte| !byte.is_ascii_whitespace())
853 .map(|start| &bytes[start..])
854 .unwrap_or(bytes);
855 let signature = if bytes.starts_with(b"\x89PNG\r\n\x1a\n") {
856 MediaSignature {
857 extension: "png",
858 family: MediaFamily::Image,
859 }
860 } else if bytes.starts_with(b"\xff\xd8\xff") {
861 MediaSignature {
862 extension: "jpg",
863 family: MediaFamily::Image,
864 }
865 } else if bytes.starts_with(b"GIF87a") || bytes.starts_with(b"GIF89a") {
866 MediaSignature {
867 extension: "gif",
868 family: MediaFamily::Image,
869 }
870 } else if bytes.len() >= 12 && bytes.starts_with(b"RIFF") && &bytes[8..12] == b"WEBP" {
871 MediaSignature {
872 extension: "webp",
873 family: MediaFamily::Image,
874 }
875 } else if bytes.starts_with(b"ID3") || bytes.starts_with(b"\xff\xfb") {
876 MediaSignature {
877 extension: "mp3",
878 family: MediaFamily::Audio,
879 }
880 } else if bytes.starts_with(b"fLaC") {
881 MediaSignature {
882 extension: "flac",
883 family: MediaFamily::Audio,
884 }
885 } else if bytes.starts_with(b"OggS") {
886 MediaSignature {
887 extension: "ogg",
888 family: MediaFamily::Audio,
889 }
890 } else if bytes.len() >= 12 && bytes.starts_with(b"RIFF") && &bytes[8..12] == b"WAVE" {
891 MediaSignature {
892 extension: "wav",
893 family: MediaFamily::Audio,
894 }
895 } else if bytes.len() >= 12 && &bytes[4..8] == b"ftyp" {
896 MediaSignature {
897 extension: "mp4",
898 family: MediaFamily::Video,
899 }
900 } else if bytes.starts_with(b"\x1aE\xdf\xa3") {
901 MediaSignature {
902 extension: "webm",
903 family: MediaFamily::Video,
904 }
905 } else if trimmed.starts_with(b"<svg")
906 || (trimmed.starts_with(b"<?xml")
907 && trimmed
908 .windows(4)
909 .take(1_024)
910 .any(|window| window.eq_ignore_ascii_case(b"<svg")))
911 {
912 MediaSignature {
913 extension: "svg",
914 family: MediaFamily::Image,
915 }
916 } else {
917 return None;
918 };
919 Some(signature)
920 }
921
922 async fn extract_pdf(
923 bytes: &[u8],
924 command: super::super::pdf::PdfTextCommand<'_>,
925 ) -> Result<ExtractedDocument, ToolError> {
926 let text = super::super::pdf::extract_bytes(bytes, command)
927 .await
928 .map_err(super::super::pdf::into_tool_error)?;
929 let pages = split_pdf_pages(&text);
930 let text = pages
931 .iter()
932 .map(|page| page.join("\n"))
933 .collect::<Vec<_>>()
934 .join("\n\n");
935 Ok(ExtractedDocument {
936 kind: DocumentKind::Pdf,
937 title: Some("PDF Document".to_string()),
938 markdown: text.clone(),
939 text,
940 cleaned_html: None,
941 pdf_pages: Some(pages),
942 media_extension: None,
943 })
944 }
945
946 fn split_pdf_pages(text: &str) -> Vec<Vec<String>> {
947 text.split('\x0C')
948 .map(|page| {
949 page.lines()
950 .map(str::trim)
951 .filter(|line| !line.is_empty())
952 .map(ToOwned::to_owned)
953 .collect::<Vec<_>>()
954 })
955 .collect()
956 }
957
958 #[cfg(test)]
959 mod tests {
960 use super::*;
961
962 #[tokio::test]
963 async fn html_becomes_readable_markdown_without_page_chrome() {
964 let html = br#"<!doctype html><html><head><title>Whale &amp; Signal</title></head><body>
965 <nav>Products Pricing Log in Cookies</nav>
966 <article><h1>Fetch once</h1><p>This is the important article body with enough words to be useful.</p>
967 <a href="/proof">Read the proof</a></article>
968 <footer>Privacy Cookies Terms</footer></body></html>"#;
969 let document = extract_document("https://example.com/post", Some("text/html"), html, None)
970 .await
971 .expect("extract html");
972
973 assert_eq!(document.kind, DocumentKind::Html);
974 assert_eq!(document.title.as_deref(), Some("Whale & Signal"));
975 assert!(document.markdown.contains("Fetch once") || document.title.is_some());
976 assert!(
977 document
978 .markdown
979 .contains("[Read the proof](https://example.com/proof)")
980 );
981 assert!(!document.markdown.contains("Products Pricing"));
982 assert!(!document.markdown.contains("Privacy Cookies"));
983 }
984
985 #[test]
986 fn relative_http_href_resolution_preserves_other_destination_kinds() {
987 let base = reqwest::Url::parse("https://example.com/guides/page").expect("base URL");
988 assert_eq!(
989 resolve_relative_http_href(&base, "../proof?q=1#receipt").as_deref(),
990 Some("https://example.com/proof?q=1#receipt")
991 );
992 for href in [
993 "#receipt",
994 "mailto:maintainer@example.com",
995 "data:text/plain,proof",
996 "codewhale:session/123",
997 "https://other.example/proof",
998 "http://[::1",
999 "",
1000 ] {
1001 assert_eq!(
1002 resolve_relative_http_href(&base, href),
1003 None,
1004 "destination must be left to htmd unchanged: {href:?}"
1005 );
1006 }
1007 let file = reqwest::Url::parse("file:///tmp/page").expect("file URL");
1008 assert!(resolve_relative_http_href(&file, "proof").is_none());
1009 }
1010
1011 #[tokio::test]
1012 async fn sparse_document_uses_article_fallback() {
1013 let html = br#"<html><head><title>Fallback</title></head><body><nav>cookie banner</nav>
1014 <article><h2>Small source</h2><p>Five useful words survive this compact article fallback path.</p></article>
1015 </body></html>"#;
1016 let document = extract_document("https://example.com/short", Some("text/html"), html, None)
1017 .await
1018 .expect("extract fallback");
1019
1020 assert!(document.markdown.contains("## Small source"));
1021 assert!(!document.markdown.contains("cookie banner"));
1022 }
1023
1024 #[tokio::test]
1025 async fn javascript_shell_returns_actionable_error() {
1026 let error = extract_document(
1027 "https://example.com/app",
1028 Some("text/html"),
1029 b"<html><body><div id='root'></div><script>boot()</script></body></html>",
1030 None,
1031 )
1032 .await
1033 .expect_err("empty app shell must fail");
1034
1035 let message = error.to_string();
1036 assert!(message.contains("may require JavaScript"), "{message}");
1037 assert!(
1038 message.contains("https://example.com/app"),
1039 "the shell failure must name the URL: {message}"
1040 );
1041 assert!(
1042 is_js_shell_error(&error.error),
1043 "the fetch pipeline recognizes this failure by marker: {message}"
1044 );
1045 assert!(
1046 !is_js_shell_error(&ToolError::execution_failed("connection reset")),
1047 "transport failures must not look like a JS shell"
1048 );
1049 }
1050
1051 #[tokio::test]
1052 async fn markdown_passes_through_unchanged() {
1053 let body = b"# Release note\n\nA complete markdown response remains intact.\n";
1054 let document = extract_document(
1055 "https://example.com/release.md",
1056 Some("text/markdown; charset=utf-8"),
1057 body,
1058 None,
1059 )
1060 .await
1061 .expect("extract markdown");
1062
1063 assert_eq!(document.kind, DocumentKind::Markdown);
1064 assert_eq!(document.markdown.as_bytes(), body);
1065 assert_eq!(document.title.as_deref(), Some("Release note"));
1066 }
1067
1068 #[tokio::test]
1069 async fn media_requires_matching_magic_bytes() {
1070 let error = extract_document(
1071 "https://example.com/not-image.png",
1072 Some("image/png"),
1073 b"<html>not really an image</html>",
1074 None,
1075 )
1076 .await
1077 .expect_err("spoofed media must fail");
1078 assert!(error.to_string().contains("did not match"));
1079
1080 let mut png = b"\x89PNG\r\n\x1a\n".to_vec();
1081 png.extend_from_slice(b"fake test payload");
1082 let document = extract_document(
1083 "https://example.com/image",
1084 Some("application/octet-stream"),
1085 &png,
1086 None,
1087 )
1088 .await
1089 .expect("sniff png");
1090 assert_eq!(document.kind, DocumentKind::Media);
1091 assert_eq!(document.media_extension, Some("png"));
1092 }
1093
1094 #[tokio::test]
1095 async fn arbitrary_binary_is_rejected() {
1096 let error = extract_document(
1097 "https://example.com/archive.bin",
1098 Some("application/octet-stream"),
1099 b"PK\x03\x04archive bytes",
1100 None,
1101 )
1102 .await
1103 .expect_err("archive must be rejected");
1104 assert!(error.to_string().contains("Unsupported binary response"));
1105 }
1106
1107 #[tokio::test]
1108 async fn empty_success_body_is_valid_text() {
1109 let document = extract_document(
1110 "https://example.com/no-content",
1111 Some("application/octet-stream"),
1112 b"",
1113 None,
1114 )
1115 .await
1116 .expect("empty body");
1117 assert_eq!(document.kind, DocumentKind::Text);
1118 assert!(document.text.is_empty());
1119 }
1120
1121 #[tokio::test]
1122 async fn content_type_matching_is_case_insensitive() {
1123 let document = extract_document(
1124 "https://example.com/document",
1125 Some("Application/JSON; Charset=UTF-8"),
1126 br#"{"status":"ok"}"#,
1127 None,
1128 )
1129 .await
1130 .expect("mixed-case JSON content type");
1131
1132 assert_eq!(document.kind, DocumentKind::Text);
1133 assert_eq!(document.text, r#"{"status":"ok"}"#);
1134 }
1135
1136 #[test]
1137 fn bom_wins_over_conflicting_transport_and_is_removed() {
1138 let mut utf8 = b"\xef\xbb\xbf".to_vec();
1139 utf8.extend_from_slice("café".as_bytes());
1140 assert_eq!(
1141 decode_response_body(&utf8, Some("text/html; charset=windows-1252"), true)
1142 .expect("UTF-8 BOM"),
1143 "café"
1144 );
1145
1146 let mut utf16 = vec![0xff, 0xfe];
1147 for unit in "BOM 日本語".encode_utf16() {
1148 utf16.extend_from_slice(&unit.to_le_bytes());
1149 }
1150 assert_eq!(
1151 decode_response_body(&utf16, Some("text/plain; charset=windows-1252"), false)
1152 .expect("UTF-16 BOM"),
1153 "BOM 日本語"
1154 );
1155 }
1156
1157 #[test]
1158 fn content_type_charset_is_exact_recognized_and_order_independent() {
1159 let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode("café");
1160 for content_type in [
1161 "text/plain; charset=windows-1252",
1162 "TEXT/PLAIN; boundary=x; CHARSET = \"windows-1252\"; q=1",
1163 "text/plain; q=1; charset='windows-1252'",
1164 ] {
1165 assert_eq!(
1166 decode_response_body(&bytes, Some(content_type), false).expect("declared charset"),
1167 "café",
1168 "{content_type}"
1169 );
1170 }
1171
1172 for malformed in [
1173 "text/plain; note=charset=windows-1252",
1174 "text/plain; charset=\"windows-1252",
1175 "text/plain; charset=definitely-not-an-encoding",
1176 ] {
1177 let decoded =
1178 decode_response_body(&bytes, Some(malformed), false).expect("UTF-8 fallback");
1179 assert!(decoded.contains('\u{fffd}'), "{malformed}: {decoded}");
1180 }
1181 }
1182
1183 #[test]
1184 fn invalid_header_falls_through_to_direct_and_legacy_html_meta() {
1185 let direct = r#"<html><head><meta charset="gbk"></head><body>中文</body></html>"#;
1186 let (direct_bytes, _, _) = encoding_rs::GBK.encode(direct);
1187 assert!(
1188 decode_response_body(&direct_bytes, Some("text/html; charset=not-real"), true,)
1189 .expect("direct meta")
1190 .contains("中文")
1191 );
1192
1193 let legacy = r#"<html><head><meta content="text/html; charset=windows-1252" http-equiv="Content-Type"></head><body>café</body></html>"#;
1194 let (legacy_bytes, _, _) = encoding_rs::WINDOWS_1252.encode(legacy);
1195 assert!(
1196 decode_response_body(&legacy_bytes, Some("text/html"), true)
1197 .expect("legacy meta")
1198 .contains("café")
1199 );
1200 }
1201
1202 #[test]
1203 fn recognized_transport_charset_beats_conflicting_meta() {
1204 let html = r#"<html><head><meta charset="shift_jis"></head><body>中文</body></html>"#;
1205 let (bytes, _, _) = encoding_rs::GBK.encode(html);
1206 let decoded = decode_response_body(&bytes, Some("text/html; charset=gbk"), true)
1207 .expect("transport charset");
1208 assert!(decoded.contains("中文"), "{decoded}");
1209 }
1210
1211 #[test]
1212 fn html_prescan_ignores_comments_scripts_and_late_meta() {
1213 let cases = [
1214 "<!-- <meta charset=windows-1252> --><html><body>café</body></html>".to_string(),
1215 "<script>\"<meta charset=windows-1252>\"</script><html><body>café</body></html>"
1216 .to_string(),
1217 format!(
1218 "<html><head>{}<meta charset=windows-1252></head><body>café</body></html>",
1219 " ".repeat(HTML_ENCODING_SNIFF_BYTES)
1220 ),
1221 ];
1222 for html in cases {
1223 let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode(&html);
1224 let decoded = decode_response_body(&bytes, Some("text/html"), true)
1225 .expect("bounded HTML fallback");
1226 assert!(
1227 decoded.contains('\u{fffd}'),
1228 "late/ignored meta changed decoding: {decoded}"
1229 );
1230 }
1231 }
1232
1233 #[test]
1234 fn non_html_bodies_never_sniff_meta_markup() {
1235 let plain = "literal <meta charset=windows-1252> café";
1236 let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode(plain);
1237 for content_type in ["text/plain", "application/json"] {
1238 let decoded = decode_response_body(&bytes, Some(content_type), false)
1239 .expect("non-HTML UTF-8 fallback");
1240 assert!(decoded.contains('\u{fffd}'), "{content_type}: {decoded}");
1241 }
1242 }
1243
1244 #[test]
1245 fn declared_gbk_shift_jis_and_windows_1252_decode_deterministically() {
1246 let cases = [
1247 (encoding_rs::GBK, "中文", "gbk"),
1248 (encoding_rs::SHIFT_JIS, "日本語", "shift_jis"),
1249 (encoding_rs::WINDOWS_1252, "café", "windows-1252"),
1250 ];
1251 for (encoding, text, label) in cases {
1252 let (bytes, _, had_errors) = encoding.encode(text);
1253 assert!(!had_errors, "fixture must be representable in {label}");
1254 assert_eq!(
1255 decode_response_body(&bytes, Some(&format!("text/plain; charset={label}")), false,)
1256 .expect("decode declared encoding"),
1257 text
1258 );
1259 }
1260 }
1261
1262 #[test]
1263 fn nul_binary_is_rejected_but_utf16_bom_text_is_not() {
1264 let error = decode_response_body(b"PK\0\x03\x04archive", Some("text/plain"), false)
1265 .expect_err("NUL binary must fail");
1266 assert!(error.to_string().contains("NUL bytes"));
1267
1268 let bom_binary = b"\xef\xbb\xbfapparently text\0binary";
1269 let error = decode_response_body(bom_binary, Some("text/plain"), false)
1270 .expect_err("a BOM must not bypass the NUL guard");
1271 assert!(error.to_string().contains("NUL bytes"));
1272
1273 let mut late_binary = vec![b'x'; 8_193];
1274 late_binary.push(0);
1275 let error = decode_response_body(&late_binary, Some("text/plain"), false)
1276 .expect_err("a late NUL must not bypass the full-body guard");
1277 assert!(error.to_string().contains("NUL bytes"));
1278
1279 let utf16 = [0xff, 0xfe, b'O', 0, b'K', 0];
1280 assert_eq!(
1281 decode_response_body(&utf16, Some("application/octet-stream"), false)
1282 .expect("BOM proves UTF-16 text"),
1283 "OK"
1284 );
1285
1286 let utf16_nul = [0xff, 0xfe, b'O', 0, 0, 0, b'K', 0];
1287 let error = decode_response_body(&utf16_nul, Some("text/plain"), false)
1288 .expect_err("decoded UTF-16 NUL must remain binary");
1289 assert!(error.to_string().contains("NUL bytes"));
1290 }
1291
1292 #[tokio::test]
1293 async fn extensionless_html_sniff_skips_leading_comments_and_xml_declarations() {
1294 let cases = [
1295 r#"<!-- deployment marker --><html><head><meta charset="windows-1252"><title>Café release notes</title></head><body><article><h1>Café release notes</h1><p>This extensionless page contains enough meaningful text for deterministic extraction.</p></article></body></html>"#,
1296 r#"<?xml version="1.0"?><!-- marker --><head><meta charset="windows-1252"><title>Café release notes</title></head><body><article><h1>Café release notes</h1><p>This extensionless page contains enough meaningful text for deterministic extraction.</p></article></body>"#,
1297 ];
1298 for html in cases {
1299 let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode(html);
1300 let document =
1301 extract_document("https://example.com/extensionless", None, &bytes, None)
1302 .await
1303 .expect("leading declarations preserve extensionless HTML sniffing");
1304 assert_eq!(document.kind, DocumentKind::Html);
1305 assert_eq!(document.title.as_deref(), Some("Café release notes"));
1306 }
1307 }
1308
1309 #[tokio::test]
1310 async fn svg_requires_and_accepts_svg_markup_signature() {
1311 let svg = br#"<?xml version="1.0"?><svg xmlns="http://www.w3.org/2000/svg"></svg>"#;
1312 let document = extract_document(
1313 "https://example.com/diagram",
1314 Some("image/svg+xml"),
1315 svg,
1316 None,
1317 )
1318 .await
1319 .expect("sniff svg");
1320 assert_eq!(document.kind, DocumentKind::Media);
1321 assert_eq!(document.media_extension, Some("svg"));
1322 }
1323
1324 async fn extract_html_fixture(url: &str, html: &str) -> ExtractedDocument {
1325 extract_document(url, Some("text/html"), html.as_bytes(), None)
1326 .await
1327 .expect("fixture must extract readable text")
1328 }
1329
1330 #[tokio::test]
1331 async fn listing_page_keeps_every_article() {
1332 let html = r#"<html><body><nav>Home Blog About</nav><main>
1333 <article><h2>First post</h2><p>Alpha story body with plenty of words to read here.</p></article>
1334 <article><h2>Second post</h2><p>Bravo story body with plenty of words to read here.</p></article>
1335 <article><h2>Third post</h2><p>Charlie story body with plenty of words to read here.</p></article>
1336 </main></body></html>"#;
1337 let document = extract_html_fixture("https://blog.example/", html).await;
1338 for needle in [
1339 "First post",
1340 "Alpha story",
1341 "Second post",
1342 "Bravo story",
1343 "Third post",
1344 "Charlie story",
1345 ] {
1346 assert!(
1347 document.markdown.contains(needle),
1348 "{needle} missing: {}",
1349 document.markdown
1350 );
1351 }
1352 assert!(!document.markdown.contains("Home Blog About"));
1353 }
1354
1355 #[tokio::test]
1356 async fn article_header_keeps_title_and_byline() {
1357 let html = r#"<html><body><header>Site logo Sign in Subscribe</header>
1358 <article><header><h1>Whales sing in dialects</h1><p class="byline">By Ada Lovelace</p></header>
1359 <p>Researchers recorded humpback song across three oceans and found regional variation.</p>
1360 </article></body></html>"#;
1361 let document = extract_html_fixture("https://news.example/whales", html).await;
1362 assert!(
1363 document.markdown.contains("Whales sing in dialects"),
1364 "{}",
1365 document.markdown
1366 );
1367 assert!(
1368 document.markdown.contains("By Ada Lovelace"),
1369 "{}",
1370 document.markdown
1371 );
1372 assert!(document.markdown.contains("regional variation"));
1373 assert!(!document.markdown.contains("Sign in Subscribe"));
1374 }
1375
1376 #[tokio::test]
1377 async fn page_wrapped_in_a_form_is_not_mistaken_for_a_javascript_shell() {
1378 let html = r#"<html><body><form method="post" action="./Default.aspx" id="form1">
1379 <input type="hidden" name="__VIEWSTATE" value="dDwtMTA4MzE0MjEwNTs7Pg==" />
1380 <div class="content"><h1>Quarterly report</h1>
1381 <p>Revenue grew in every region this quarter, led by the northern division.</p></div>
1382 <select name="year"><option>2025</option><option>2026</option></select>
1383 <button type="submit">Go</button>
1384 </form></body></html>"#;
1385 let document = extract_html_fixture("https://legacy.example/Default.aspx", html).await;
1386 assert!(
1387 document.markdown.contains("Quarterly report"),
1388 "{}",
1389 document.markdown
1390 );
1391 assert!(document.markdown.contains("northern division"));
1392 assert!(!document.markdown.contains("VIEWSTATE"));
1393 assert!(
1394 !document.markdown.contains("2026"),
1395 "form controls are stripped"
1396 );
1397 }
1398
1399 #[tokio::test]
1400 async fn arxiv_abstract_page_extracts_title_and_abstract() {
1401 // Trimmed from the shape of https://arxiv.org/abs/1706.03762 (2026-09).
1402 let html = r##"<!DOCTYPE html><html lang="en"><head><title>[1706.03762] Attention Is All You Need</title>
1403 <script>window.MathJax = {};</script></head>
1404 <body ><div class="flex-wrap-footer"><a href="#content" class="ds-skip-link">Skip to main content</a>
1405 <header class="ds-site-header"><a href="https://arxiv.org/">archive home</a>
1406 <button type="button" id="ds-nav-toggle">Open menu</button>
1407 <nav class="ds-site-header-nav"><a href="https://arxiv.org/search">Search</a><a href="https://arxiv.org/login">Log in</a></nav>
1408 </header>
1409 <div class="arxiv-search-overlay" hidden><form method="GET" action="https://arxiv.org/search">
1410 <label for="q">Search arXiv</label><input type="text" name="query" id="q"></form></div>
1411 <main><div id="content"><!-- rdf:RDF <rdf:Description dc:title="Attention Is All You Need" /> -->
1412 <div id="abs-outer"><div class="leftcolumn"><div class="subheader"><h1>Computer Science &gt; Computation and Language</h1></div>
1413 <div id="abs"><div class="dateline">[Submitted on 12 Jun 2017 (<a href="/abs/1706.03762v1">v1</a>)]</div>
1414 <h1 class="title mathjax"><span class="descriptor">Title:</span>Attention Is All You Need</h1>
1415 <div class="authors"><span class="descriptor">Authors:</span><a href="/a/vaswani_a_1">Ashish Vaswani</a></div>
1416 <blockquote class="abstract mathjax"><span class="descriptor">Abstract:</span>The dominant sequence transduction models are based on complex recurrent or convolutional neural networks.</blockquote>
1417 <script type="text/javascript" language="javascript">mathjaxToggle();</script>
1418 </div></div></div></div></main>
1419 <footer><a href="https://info.arxiv.org/help/contact.html">Contact</a></footer></div></body></html>"##;
1420 let document = extract_html_fixture("https://arxiv.org/abs/1706.03762", html).await;
1421 assert!(
1422 document.markdown.contains("Attention Is All You Need"),
1423 "{}",
1424 document.markdown
1425 );
1426 assert!(
1427 document.markdown.contains("dominant sequence transduction"),
1428 "{}",
1429 document.markdown
1430 );
1431 assert!(document.markdown.contains("Ashish Vaswani"));
1432 assert!(!document.markdown.contains("mathjaxToggle"));
1433 assert!(!document.markdown.contains("Log in"));
1434 }
1435 }
1436
1437 #[cfg(test)]
1438 #[path = "extract_host_tests.rs"]
1439 mod host_tests;
1440
1440 lines RUST