返回 CodeWhale
fetch_url.rs
根目录 / crates / tui / src / tools / fetch_url.rs
1 //! Direct-fetch HTTP tool. Complements `web_search` for cases where the user
2 //! already knows the URL — a known repo, a blog post, a spec page — and
3 //! search is overkill or actively unhelpful.
4 //!
5 //! Returns a structured `{url, status, content_type, content, truncated}`
6 //! payload. HTML responses are stripped to readable text by default
7 //! (`format = "markdown"`); pass `format = "raw"` to keep the bytes intact
8 //! when the model wants to do its own parsing.
9
10 use super::handle::query_jsonpath;
11 use super::pdf::PdfTextCommand;
12 use super::spec::{
13 ApprovalRequirement, ToolCapability, ToolContext, ToolError, ToolResult, ToolSpec, optional_u64,
14 };
15 use super::web::extract::{DocumentKind, ExtractedDocument, decode_response_body};
16 use super::web::fetch::{
17 DEFAULT_MAX_BYTES, DEFAULT_TIMEOUT, FetchAttempt, FetchOptions, HARD_MAX_BYTES,
18 HARD_MAX_TIMEOUT, fetch_readable,
19 };
20 use super::web::overflow::bound_text as bound_web_text;
21 #[cfg(test)]
22 use super::web::overflow::inline_char_budget;
23 use async_trait::async_trait;
24 use serde::Serialize;
25 use serde_json::{Value, json};
26 use std::collections::BTreeMap;
27 use std::time::Duration;
28
29 const FETCH_ACCEPT: &str = "text/html,text/markdown,text/plain,application/json,application/pdf,image/*,audio/*,video/*,*/*;q=0.5";
30
31 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
32 enum Format {
33 Text,
34 Markdown,
35 Raw,
36 }
37
38 impl Format {
39 fn parse(value: Option<&str>) -> Result<Self, ToolError> {
40 match value
41 .unwrap_or("markdown")
42 .trim()
43 .to_ascii_lowercase()
44 .as_str()
45 {
46 "text" | "txt" | "plain" => Ok(Self::Text),
47 "markdown" | "md" => Ok(Self::Markdown),
48 "raw" | "html" | "bytes" => Ok(Self::Raw),
49 other => Err(ToolError::invalid_input(format!(
50 "unknown format `{other}` (allowed: text, markdown, raw)"
51 ))),
52 }
53 }
54 }
55
56 #[derive(Debug, Serialize)]
57 struct FetchResponse {
58 ref_id: String,
59 url: String,
60 status: u16,
61 headers: BTreeMap<String, String>,
62 content_type: String,
63 content: String,
64 truncated: bool,
65 receipt: FetchReceipt,
66 #[serde(skip_serializing_if = "Option::is_none")]
67 artifact: Option<String>,
68 #[serde(skip_serializing_if = "Option::is_none")]
69 fields: Option<BTreeMap<String, Vec<Value>>>,
70 }
71
72 #[derive(Debug, Serialize)]
73 struct FetchReceipt {
74 cache_hit: bool,
75 retries: usize,
76 redirects: usize,
77 /// Every request this fetch made, in order: session `cache_hit`, whether
78 /// the attempt bypassed caches, which one produced content, and the
79 /// response headers that explain the cache state (#5904).
80 attempts: Vec<FetchAttempt>,
81 }
82
83 #[derive(Debug)]
84 struct ArtifactWrite {
85 session_id: String,
86 absolute_path: std::path::PathBuf,
87 relative_path: std::path::PathBuf,
88 byte_size: u64,
89 preview: String,
90 }
91
92 pub struct FetchUrlTool;
93
94 #[async_trait]
95 impl ToolSpec for FetchUrlTool {
96 fn name(&self) -> &'static str {
97 "fetch_url"
98 }
99
100 fn model_visible(&self) -> bool {
101 false
102 }
103
104 fn description(&self) -> &'static str {
105 "Fetch a known URL directly (HTTP GET) and return its content with a session-scoped citation ref_id. Use this instead of `curl` in `exec_shell` — sandboxed, network-policy aware, and properly decoded. Plain-text endpoints (`.md`, `.txt`, `.json`, `.yaml`, `raw.githubusercontent.com`, public APIs) prefer this over the browser/automation stack. For unknown queries, use `web_search` first. If a login or authorization wall is returned, treat the wall as the result; do not claim the protected page was read."
106 }
107
108 fn input_schema(&self) -> Value {
109 json!({
110 "type": "object",
111 "properties": {
112 "url": {
113 "type": "string",
114 "description": "Absolute HTTP/HTTPS URL to fetch."
115 },
116 "format": {
117 "type": "string",
118 "enum": ["text", "markdown", "raw"],
119 "description": "Post-processing for the response body. `markdown` (default) uses readability extraction and real HTML-to-Markdown conversion; `text` returns readable plain text; `raw` preserves textual response bytes. Binary media is saved as a session artifact."
120 },
121 "max_bytes": {
122 "type": "integer",
123 "description": "Truncate response body after this many bytes (default 1,000,000; hard max 10,485,760)."
124 },
125 "timeout_ms": {
126 "type": "integer",
127 "description": "Request timeout in milliseconds (default 15,000; max 60,000)."
128 },
129 "fields": {
130 "type": "array",
131 "items": { "type": "string" },
132 "description": "Optional JSONPath projections for JSON responses. Supports $, .field, [index], [*], and ['field']; returns matches under `fields`."
133 }
134 },
135 "required": ["url"]
136 })
137 }
138
139 fn capabilities(&self) -> Vec<ToolCapability> {
140 vec![ToolCapability::ReadOnly, ToolCapability::Network]
141 }
142
143 fn approval_requirement(&self) -> ApprovalRequirement {
144 // Read-only HTTP can still disclose local data through a URL or query.
145 // Host allowlisting controls reachability, not approval of this payload.
146 ApprovalRequirement::Required
147 }
148
149 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
150 let url = input
151 .get("url")
152 .and_then(Value::as_str)
153 .ok_or_else(|| ToolError::invalid_input("`url` is required"))?
154 .trim()
155 .to_string();
156
157 if url.is_empty() {
158 return Err(ToolError::invalid_input("`url` cannot be empty"));
159 }
160 let scheme_ok = url.starts_with("http://") || url.starts_with("https://");
161 if !scheme_ok {
162 return Err(ToolError::invalid_input(
163 "only http:// and https:// URLs are supported",
164 ));
165 }
166
167 let format = Format::parse(input.get("format").and_then(Value::as_str))?;
168 let max_bytes =
169 usize::try_from(optional_u64(&input, "max_bytes", DEFAULT_MAX_BYTES as u64)?)
170 .unwrap_or(HARD_MAX_BYTES)
171 .clamp(1, HARD_MAX_BYTES);
172 let timeout_ms = optional_u64(&input, "timeout_ms", DEFAULT_TIMEOUT.as_millis() as u64)?
173 .clamp(1, HARD_MAX_TIMEOUT.as_millis() as u64);
174 let requested_fields = parse_fields(&input)?;
175 // Bound as a reference so the `Fn` extractor can run twice without
176 // moving the field list into its first future.
177 let requested_fields = &requested_fields;
178 // A 2xx that extracts to nothing is re-fetched once past every cache
179 // before it becomes an error; the receipt keeps both attempts (#5904).
180 let readable = fetch_readable(
181 &url,
182 &FetchOptions::new(Duration::from_millis(timeout_ms), max_bytes, FETCH_ACCEPT),
183 context,
184 "fetch_url",
185 |fetched: super::web::fetch::FetchedPayload| {
186 Box::pin(async move {
187 let is_success = (200..300).contains(&fetched.status);
188 let body_text = if requested_fields.is_empty() {
189 None
190 } else {
191 // JSON is never allowed to discover an encoding from body markup.
192 Some(decode_response_body(
193 &fetched.bytes,
194 Some(&fetched.content_type),
195 false,
196 )?)
197 };
198 let fields = match body_text.as_deref() {
199 Some(body) => {
200 project_json_fields(body, &fetched.content_type, requested_fields)?
201 }
202 None => None,
203 };
204 let extracted = extract_fetched_document(
205 format,
206 &fetched.url,
207 &fetched.content_type,
208 &fetched.bytes,
209 is_success,
210 body_text.as_deref(),
211 PdfTextCommand::system(Some(context)),
212 )
213 .await?;
214 Ok((extracted, fields))
215 })
216 },
217 )
218 .await?;
219 let super::web::fetch::ReadableFetch {
220 payload: fetched,
221 document: (extracted, fields),
222 attempts,
223 } = readable;
224 let is_success = (200..300).contains(&fetched.status);
225
226 let citation_title = extracted.title.clone();
227 let (processed, artifact_write) = render_extracted(
228 &fetched.url,
229 &fetched.content_type,
230 format,
231 extracted,
232 &fetched.bytes,
233 context,
234 )?;
235 let artifact = artifact_write
236 .as_ref()
237 .map(|write| crate::artifacts::format_artifact_relative_path(&write.relative_path));
238
239 let citation = super::web::citations::register(
240 &context.state_namespace,
241 &fetched.url,
242 citation_title.as_deref(),
243 )
244 .ok_or_else(|| ToolError::execution_failed("fetched URL could not be registered"))?;
245 let response = FetchResponse {
246 ref_id: citation.ref_id,
247 url: citation.url,
248 status: fetched.status,
249 headers: fetched.headers,
250 content_type: fetched.content_type,
251 content: processed,
252 truncated: fetched.truncated,
253 receipt: FetchReceipt {
254 cache_hit: fetched.cache_hit,
255 retries: fetched.retries,
256 redirects: fetched.redirects,
257 attempts,
258 },
259 artifact,
260 fields,
261 };
262
263 let content = serde_json::to_string_pretty(&response).map_err(|error| {
264 ToolError::execution_failed(format!("failed to serialize response: {error}"))
265 })?;
266 let metadata = artifact_write.map(artifact_metadata);
267
268 if !is_success {
269 // Don't `Err` on 4xx/5xx — the caller often wants to see the body
270 // (e.g. a JSON error envelope). Mark the result as a failure so the
271 // engine renders it as such.
272 return Ok(ToolResult {
273 content,
274 success: false,
275 metadata,
276 });
277 }
278
279 Ok(ToolResult {
280 content,
281 success: true,
282 metadata,
283 })
284 }
285 }
286
287 async fn extract_fetched_document(
288 format: Format,
289 url: &str,
290 content_type: &str,
291 bytes: &[u8],
292 is_success: bool,
293 decoded_body: Option<&str>,
294 pdf_command: PdfTextCommand<'_>,
295 ) -> super::web::adapter::AdapterResult<ExtractedDocument> {
296 let extraction = if format == Format::Raw
297 && super::web::extract::validate_pdf_response(url, Some(content_type), bytes)?
298 {
299 Ok(ExtractedDocument {
300 kind: DocumentKind::Pdf,
301 title: Some("PDF Document".to_string()),
302 text: String::new(),
303 markdown: String::new(),
304 cleaned_html: None,
305 pdf_pages: None,
306 media_extension: None,
307 })
308 } else {
309 super::web::extract::extract_document_with_pdf_command(
310 url,
311 Some(content_type),
312 bytes,
313 pdf_command,
314 )
315 .await
316 };
317 match extraction {
318 Ok(document) => Ok(document),
319 Err(error)
320 if error.content()
321 && (format == Format::Raw || !is_success)
322 && is_declared_textual(content_type) =>
323 {
324 let body_text = match decoded_body {
325 Some(body_text) => body_text.to_string(),
326 None => {
327 decode_response_body(bytes, Some(content_type), is_declared_html(content_type))?
328 }
329 };
330 Ok(ExtractedDocument {
331 kind: DocumentKind::Text,
332 title: None,
333 text: body_text.clone(),
334 markdown: body_text,
335 cleaned_html: None,
336 pdf_pages: None,
337 media_extension: None,
338 })
339 }
340 Err(error) => Err(error),
341 }
342 }
343
344 fn is_declared_textual(content_type: &str) -> bool {
345 let content_type = content_type
346 .split(';')
347 .next()
348 .unwrap_or(content_type)
349 .trim()
350 .to_ascii_lowercase();
351 content_type.starts_with("text/")
352 || content_type.contains("html")
353 || content_type.contains("json")
354 || content_type.contains("xml")
355 || content_type.contains("yaml")
356 || content_type.contains("javascript")
357 }
358
359 fn is_declared_html(content_type: &str) -> bool {
360 matches!(
361 content_type
362 .split(';')
363 .next()
364 .unwrap_or(content_type)
365 .trim()
366 .to_ascii_lowercase()
367 .as_str(),
368 "text/html" | "application/xhtml+xml"
369 )
370 }
371
372 fn render_extracted(
373 url: &str,
374 content_type: &str,
375 format: Format,
376 document: ExtractedDocument,
377 bytes: &[u8],
378 context: &ToolContext,
379 ) -> Result<(String, Option<ArtifactWrite>), ToolError> {
380 if document.kind == DocumentKind::Pdf && format != Format::Raw {
381 let extracted = match format {
382 Format::Text => document.text,
383 Format::Markdown => document.markdown,
384 Format::Raw => unreachable!("raw PDF handled below"),
385 };
386 return bound_text(url, extracted, context);
387 }
388
389 if document.kind == DocumentKind::Media || document.kind == DocumentKind::Pdf {
390 let extension = document
391 .media_extension
392 .unwrap_or(if document.kind == DocumentKind::Pdf {
393 "pdf"
394 } else {
395 "bin"
396 });
397 let artifact = write_binary_artifact(url, extension, bytes, context)?;
398 let relative = crate::artifacts::format_artifact_relative_path(&artifact.relative_path);
399 let label = if document.kind == DocumentKind::Pdf {
400 "PDF"
401 } else {
402 "media"
403 };
404 let content =
405 format!("[{label} response saved to {relative}; content type: {content_type}.]");
406 return Ok((content, Some(artifact)));
407 }
408
409 let content = match format {
410 Format::Raw => decode_response_body(
411 bytes,
412 Some(content_type),
413 document.kind == DocumentKind::Html,
414 )?,
415 Format::Text => document.text,
416 Format::Markdown => document.markdown,
417 };
418 bound_text(url, content, context)
419 }
420
421 fn bound_text(
422 url: &str,
423 content: String,
424 context: &ToolContext,
425 ) -> Result<(String, Option<ArtifactWrite>), ToolError> {
426 let bounded = bound_web_text(
427 content,
428 context,
429 |body| fetch_artifact_id(url, body.as_bytes()),
430 "page",
431 )?;
432 let artifact = bounded.artifact.map(|artifact| ArtifactWrite {
433 session_id: artifact.session_id,
434 absolute_path: artifact.absolute_path,
435 relative_path: artifact.relative_path,
436 byte_size: artifact.byte_size,
437 preview: artifact.preview,
438 });
439 Ok((bounded.content, artifact))
440 }
441
442 fn write_binary_artifact(
443 url: &str,
444 extension: &str,
445 bytes: &[u8],
446 context: &ToolContext,
447 ) -> Result<ArtifactWrite, ToolError> {
448 let artifact_id = fetch_artifact_id(url, bytes);
449 let (absolute_path, relative_path) = crate::artifacts::write_session_artifact_bytes(
450 &context.state_namespace,
451 &artifact_id,
452 extension,
453 bytes,
454 )
455 .map_err(|error| {
456 ToolError::execution_failed(format!(
457 "failed to preserve fetched media artifact: {error}"
458 ))
459 })?;
460 Ok(ArtifactWrite {
461 session_id: context.state_namespace.clone(),
462 absolute_path,
463 relative_path,
464 byte_size: bytes.len() as u64,
465 preview: format!("Fetched {extension} artifact from {url}"),
466 })
467 }
468
469 fn fetch_artifact_id(url: &str, bytes: &[u8]) -> String {
470 let mut identity = Vec::with_capacity(url.len() + bytes.len());
471 identity.extend_from_slice(url.as_bytes());
472 identity.extend_from_slice(bytes);
473 let digest = crate::hashing::sha256_hex(&identity);
474 format!("fetch_{}", &digest[..16])
475 }
476
477 fn artifact_metadata(write: ArtifactWrite) -> Value {
478 json!({
479 "spillover_path": write.absolute_path.display().to_string(),
480 "artifact_session_id": write.session_id,
481 "artifact_relative_path": crate::artifacts::format_artifact_relative_path(&write.relative_path),
482 "artifact_byte_size": write.byte_size,
483 "artifact_preview": write.preview,
484 // Every artifact this tool writes is retrievable evidence, so flag it
485 // for the engine's `activate_result_dependencies` (same contract as
486 // the shell-truncation spillover): the next turn auto-activates
487 // `retrieve_tool_result`. Text overflows name that tool in their
488 // footer; binary PDF/media saves name only the saved-artifact path in
489 // their inline pointer, so this flag is what makes that pointer
490 // actionable rather than a dead end.
491 "evidence_available": true,
492 })
493 }
494
495 fn parse_fields(input: &Value) -> Result<Vec<String>, ToolError> {
496 let Some(values) = input.get("fields") else {
497 return Ok(Vec::new());
498 };
499 let Some(values) = values.as_array() else {
500 return Err(ToolError::invalid_input("`fields` must be an array"));
501 };
502 let mut fields = Vec::new();
503 for value in values {
504 let Some(field) = value.as_str() else {
505 return Err(ToolError::invalid_input(
506 "`fields` entries must be JSONPath strings",
507 ));
508 };
509 let field = field.trim();
510 if !field.is_empty() {
511 fields.push(field.to_string());
512 }
513 }
514 Ok(fields)
515 }
516
517 fn project_json_fields(
518 body_text: &str,
519 content_type: &str,
520 fields: &[String],
521 ) -> Result<Option<BTreeMap<String, Vec<Value>>>, ToolError> {
522 if fields.is_empty() {
523 return Ok(None);
524 }
525 if !content_type.to_ascii_lowercase().contains("json") {
526 return Err(ToolError::invalid_input(
527 "`fields` can only be used with JSON responses",
528 ));
529 }
530 let body_json: Value = serde_json::from_str(body_text).map_err(|e| {
531 ToolError::execution_failed(format!("response body is not valid JSON for `fields`: {e}"))
532 })?;
533 let mut out = BTreeMap::new();
534 for field in fields {
535 let matches = query_jsonpath(&body_json, field).map_err(|e| {
536 ToolError::invalid_input(format!("invalid JSONPath `{field}` in `fields`: {e}"))
537 })?;
538 out.insert(field.clone(), matches);
539 }
540 Ok(Some(out))
541 }
542
543 #[cfg(test)]
544 #[path = "fetch_url/tests.rs"]
545 mod pdf_tests;
546
547 #[cfg(test)]
548 mod tests {
549 use super::*;
550 use crate::tools::spec::ToolContext;
551 use std::path::PathBuf;
552
553 struct ArtifactRootRestore(Option<PathBuf>);
554
555 impl Drop for ArtifactRootRestore {
556 fn drop(&mut self) {
557 crate::artifacts::set_test_artifact_sessions_root(self.0.take());
558 }
559 }
560
561 fn ctx() -> ToolContext {
562 ToolContext::new(PathBuf::from("."))
563 }
564
565 /// `fetch_url` can disclose local data through a URL or query, so it always
566 /// asks: the tool declares `Required`, and the default-ask policy resolves
567 /// that to a prompt rather than running it.
568 #[test]
569 fn fetch_url_always_requires_approval() {
570 use crate::tools::spec::{ApprovalRequirement, ToolSpec};
571 assert_eq!(
572 FetchUrlTool.approval_requirement(),
573 ApprovalRequirement::Required
574 );
575 assert!(
576 FetchUrlTool
577 .capabilities()
578 .contains(&ToolCapability::Network)
579 );
580
581 // Under the default Ask posture that requirement resolves to a prompt,
582 // and only full access (or an explicit bypass) lets it through.
583 use crate::core::authority::{ToolPermission, TurnAuthority, resolve_tool_permission};
584 use codewhale_config::AppMode;
585 use codewhale_execpolicy::ApprovalMode;
586 let requirement = FetchUrlTool.approval_requirement();
587 let ask = TurnAuthority::from_effective_fields(
588 AppMode::Agent,
589 true,
590 false,
591 false,
592 ApprovalMode::Suggest,
593 );
594 assert_eq!(
595 resolve_tool_permission(&ask, requirement, false),
596 ToolPermission::Prompt
597 );
598 let never = TurnAuthority::from_effective_fields(
599 AppMode::Agent,
600 true,
601 false,
602 false,
603 ApprovalMode::Never,
604 );
605 assert_eq!(
606 resolve_tool_permission(&never, requirement, false),
607 ToolPermission::Deny
608 );
609 }
610
611 #[test]
612 fn format_parse_accepts_aliases_and_rejects_unknown() {
613 assert_eq!(Format::parse(Some("markdown")).unwrap(), Format::Markdown);
614 assert_eq!(Format::parse(Some("MD")).unwrap(), Format::Markdown);
615 assert_eq!(Format::parse(Some("text")).unwrap(), Format::Text);
616 assert_eq!(Format::parse(Some("raw")).unwrap(), Format::Raw);
617 assert_eq!(Format::parse(None).unwrap(), Format::Markdown);
618 assert!(Format::parse(Some("yaml")).is_err());
619 }
620
621 #[test]
622 fn raw_text_uses_declared_charset_and_rejects_nul_binary() {
623 let document = || ExtractedDocument {
624 kind: DocumentKind::Text,
625 title: None,
626 text: String::new(),
627 markdown: String::new(),
628 cleaned_html: None,
629 pdf_pages: None,
630 media_extension: None,
631 };
632 let (bytes, _, _) = encoding_rs::WINDOWS_1252.encode("café");
633 let (content, artifact) = render_extracted(
634 "https://example.com/plain",
635 "text/plain; charset=windows-1252",
636 Format::Raw,
637 document(),
638 &bytes,
639 &ctx(),
640 )
641 .expect("decode raw text");
642 assert_eq!(content, "café");
643 assert!(artifact.is_none());
644
645 let error = render_extracted(
646 "https://example.com/not-text",
647 "text/plain",
648 Format::Raw,
649 document(),
650 b"binary\0payload",
651 &ctx(),
652 )
653 .expect_err("raw text must retain the binary guard");
654 assert!(error.to_string().contains("NUL bytes"));
655 }
656
657 #[test]
658 fn textual_and_html_fallback_classification_is_exact() {
659 assert!(is_declared_textual("Application/JSON; charset=utf-8"));
660 assert!(is_declared_html("TEXT/HTML; charset=gbk"));
661 assert!(!is_declared_html("text/plain; note=text/html"));
662 assert!(!is_declared_textual("application/octet-stream"));
663 }
664
665 #[test]
666 fn route_budget_overflow_round_trips_through_session_artifact() {
667 let _lock = crate::artifacts::TEST_ARTIFACT_SESSIONS_GUARD
668 .lock()
669 .unwrap_or_else(|error| error.into_inner());
670 let tmp = tempfile::tempdir().unwrap();
671 let prior =
672 crate::artifacts::set_test_artifact_sessions_root(Some(tmp.path().join("sessions")));
673 let _restore = ArtifactRootRestore(prior);
674 let context = ToolContext::new(".")
675 .with_state_namespace("fetch-overflow")
676 .with_route_context_window(10_000);
677 let full = "Whale content. ".repeat(200);
678
679 let (inline, artifact) =
680 bound_text("https://example.com/large", full.clone(), &context).unwrap();
681 let artifact = artifact.expect("overflow artifact");
682
683 assert!(inline.contains("retrieve_tool_result"));
684 assert!(inline.chars().count() <= inline_char_budget(&context));
685 assert_eq!(
686 std::fs::read_to_string(&artifact.absolute_path).unwrap(),
687 full
688 );
689 // The footer names `retrieve_tool_result` as the recovery path; the
690 // evidence flag is what makes the engine auto-activate that tool on
691 // the next turn, so the named tool is actually present.
692 let metadata = artifact_metadata(artifact);
693 assert_eq!(
694 metadata.get("evidence_available"),
695 Some(&json!(true)),
696 "artifact metadata must flag retrievable evidence:\n{metadata}"
697 );
698 }
699
700 #[test]
701 fn project_json_fields_returns_requested_jsonpath_matches() {
702 let fields = vec!["$.items[*].name".to_string(), "$.count".to_string()];
703 let projected = project_json_fields(
704 r#"{"items":[{"name":"alpha"},{"name":"beta"}],"count":2}"#,
705 "application/json",
706 &fields,
707 )
708 .expect("project")
709 .expect("some");
710
711 assert_eq!(
712 projected.get("$.items[*].name").unwrap(),
713 &vec![json!("alpha"), json!("beta")]
714 );
715 assert_eq!(projected.get("$.count").unwrap(), &vec![json!(2)]);
716 }
717
718 #[test]
719 fn project_json_fields_rejects_non_json_content_type() {
720 let fields = vec!["$.name".to_string()];
721 let err = project_json_fields("{}", "text/plain", &fields).expect_err("must reject");
722 assert!(format!("{err}").contains("JSON responses"));
723 }
724
725 #[tokio::test]
726 async fn rejects_non_http_schemes() {
727 let tool = FetchUrlTool;
728 let res = tool
729 .execute(json!({"url": "file:///etc/passwd"}), &ctx())
730 .await;
731 let err = res.unwrap_err();
732 assert!(format!("{err:?}").contains("http"));
733 }
734
735 #[tokio::test]
736 async fn rejects_empty_url() {
737 let tool = FetchUrlTool;
738 let res = tool.execute(json!({"url": " "}), &ctx()).await;
739 assert!(res.is_err());
740 }
741
742 #[tokio::test]
743 async fn rejects_missing_url() {
744 let tool = FetchUrlTool;
745 let res = tool.execute(json!({}), &ctx()).await;
746 assert!(res.is_err());
747 }
748
749 #[tokio::test]
750 async fn rejects_localhost_hostname() {
751 let tool = FetchUrlTool;
752 let res = tool
753 .execute(json!({"url": "http://localhost:8080/admin"}), &ctx())
754 .await;
755 let err = res.unwrap_err();
756 assert!(format!("{err}").contains("localhost"));
757 }
758
759 #[tokio::test]
760 async fn network_policy_denies_blocked_host() {
761 use crate::network_policy::{Decision, NetworkPolicy, NetworkPolicyDecider};
762 let policy = NetworkPolicy {
763 default: Decision::Deny.into(),
764 allow: vec!["api.deepseek.com".to_string()],
765 deny: vec![],
766 proxy: Vec::new(),
767 proxy_fake_ip_cidrs: Vec::new(),
768 audit: false,
769 };
770 let decider = NetworkPolicyDecider::new(policy, None);
771 let ctx = ToolContext::new(PathBuf::from(".")).with_network_policy(decider);
772 let tool = FetchUrlTool;
773 let res = tool
774 .execute(json!({"url": "https://example.com/foo"}), &ctx)
775 .await;
776 let err = res.expect_err("blocked host should fail");
777 assert!(format!("{err}").contains("blocked"));
778 }
779
780 #[tokio::test]
781 async fn proxy_opt_in_does_not_allow_restricted_ip_literal() {
782 use crate::network_policy::{Decision, NetworkPolicy, NetworkPolicyDecider};
783
784 let policy = NetworkPolicy {
785 default: Decision::Allow.into(),
786 allow: Vec::new(),
787 deny: Vec::new(),
788 proxy: vec!["198.18.0.1".to_string()],
789 proxy_fake_ip_cidrs: vec!["198.18.0.0/15".to_string()],
790 audit: false,
791 };
792 let decider = NetworkPolicyDecider::new(policy, None);
793 let ctx = ToolContext::new(PathBuf::from(".")).with_network_policy(decider);
794 let tool = FetchUrlTool;
795
796 let err = tool
797 .execute(json!({"url": "http://198.18.0.1/status"}), &ctx)
798 .await
799 .expect_err("literal restricted IP URLs must stay blocked");
800
801 assert!(format!("{err}").contains("IP 198.18.0.1 is a restricted address"));
802 }
803 }
804
804 lines RUST