| 1 | //! Turning a local image file into a wire-ready image content block. |
| 2 | //! |
| 3 | //! CodeWhale's message model has been multimodal for a long time — |
| 4 | //! [`ContentBlock::ImageUrl`] round-trips |
| 5 | //! through session persistence, compaction and all three wire builders. What it |
| 6 | //! never had was a *faucet*: nothing outside `#[cfg(test)]` ever constructed |
| 7 | //! one, so a user who attached a screenshot got the literal text |
| 8 | //! `[Attached image: /path/to/shot.png]` and a model that correctly concluded it |
| 9 | //! could not see the picture. |
| 10 | //! |
| 11 | //! This module is that faucet. It reads a file, proves it is an image the |
| 12 | //! providers actually accept, holds it to a size budget, and emits a |
| 13 | //! `data:` URL. |
| 14 | //! |
| 15 | //! # Two stages, because the two failure kinds differ |
| 16 | //! |
| 17 | //! [`expand_attachment_blocks`] runs when the message is built and decides the |
| 18 | //! *permanent* questions: does this file exist, is it really an image, is it |
| 19 | //! small enough. Those answers cannot change, so they are baked into history. |
| 20 | //! |
| 21 | //! [`strip_images_when_unsupported`] runs per outbound request and decides the |
| 22 | //! one *contingent* question: can the model this request is going to actually |
| 23 | //! see images. Routes change mid-session, so answering that at build time |
| 24 | //! would mean attaching a screenshot under a text-only model and losing it |
| 25 | //! permanently, even after switching to a vision model. History keeps the |
| 26 | //! image; each request is normalized against its own route. |
| 27 | //! |
| 28 | //! # Provider neutrality |
| 29 | //! |
| 30 | //! There is exactly one internal representation — a `data:<media-type>;base64,…` |
| 31 | //! URL on `ContentBlock::ImageUrl` — and each wire builder projects it: |
| 32 | //! |
| 33 | //! | wire format | shape | |
| 34 | //! |---|---| |
| 35 | //! | Chat Completions | `{"type":"image_url","image_url":{"url":"data:…"}}` | |
| 36 | //! | Responses | `{"type":"input_image","image_url":"data:…"}` | |
| 37 | //! | Anthropic Messages | `{"type":"image","source":{"type":"base64","media_type":…,"data":…}}` | |
| 38 | //! |
| 39 | //! The Anthropic split lives in [`parse_data_url`], which |
| 40 | //! `client::anthropic` calls. Anthropic is the reason the accepted-format list |
| 41 | //! below is not simply "whatever the `image` crate can decode": it accepts only |
| 42 | //! PNG/JPEG/GIF/WebP, and so, therefore, do we. Refusing a BMP here with a |
| 43 | //! readable message beats letting one through to a provider-side 400. |
| 44 | |
| 45 | use anyhow::{Result, bail}; |
| 46 | use codewhale_protocol::runtime::{ |
| 47 | MAX_RUNTIME_IMAGE_BYTES, MAX_RUNTIME_IMAGE_TOTAL_BYTES, MAX_RUNTIME_IMAGES, RuntimeImageInput, |
| 48 | }; |
| 49 | use image::codecs::jpeg::JpegEncoder; |
| 50 | use image::codecs::png::{CompressionType, FilterType as PngFilter, PngEncoder}; |
| 51 | use image::imageops::FilterType; |
| 52 | use image::{DynamicImage, ExtendedColorType, GenericImageView, ImageEncoder, ImageReader, Limits}; |
| 53 | use std::io::Cursor; |
| 54 | use std::path::Path; |
| 55 | |
| 56 | use base64::{Engine as _, engine::general_purpose::STANDARD}; |
| 57 | |
| 58 | use crate::model_profile::SupportState; |
| 59 | use codewhale_models::{ContentBlock, ImageUrlContent}; |
| 60 | |
| 61 | /// Largest source image accepted, in bytes, before base64 expansion. |
| 62 | /// |
| 63 | /// Base64 inflates by 4/3, so this admits roughly 6.7 MB of request body per |
| 64 | /// image. The number is Anthropic's documented per-image ceiling; keeping the |
| 65 | /// tightest provider limit as the shared limit is what makes a "CodeWhale |
| 66 | /// accepted it" verdict portable across routes. |
| 67 | pub const MAX_IMAGE_BYTES: usize = 5 * 1024 * 1024; |
| 68 | |
| 69 | /// Maximum width or height admitted for an input image (8192 px). |
| 70 | pub const MAX_IMAGE_DIMENSION: u32 = 8192; |
| 71 | |
| 72 | /// Maximum total pixels admitted before decoding is aborted (~33.5 megapixels). |
| 73 | pub const MAX_IMAGE_PIXELS: u64 = 33_554_432; |
| 74 | |
| 75 | /// Memory allocation limit for image decoding (64 MiB). |
| 76 | pub const MAX_DECODE_ALLOC_BYTES: u64 = 64 * 1024 * 1024; |
| 77 | |
| 78 | pub(crate) fn decode_and_guard_image(bytes: &[u8]) -> Result<(DynamicImage, u32, u32)> { |
| 79 | let limits = || { |
| 80 | let mut limits = Limits::default(); |
| 81 | limits.max_alloc = Some(MAX_DECODE_ALLOC_BYTES); |
| 82 | limits.max_image_width = Some(MAX_IMAGE_DIMENSION); |
| 83 | limits.max_image_height = Some(MAX_IMAGE_DIMENSION); |
| 84 | limits |
| 85 | }; |
| 86 | let mut reader = ImageReader::new(Cursor::new(bytes)).with_guessed_format()?; |
| 87 | reader.limits(limits()); |
| 88 | let (width, height) = reader |
| 89 | .into_dimensions() |
| 90 | .map_err(|_| anyhow::anyhow!("invalid image header or decompression bomb guard"))?; |
| 91 | if u64::from(width) * u64::from(height) > MAX_IMAGE_PIXELS |
| 92 | || width > MAX_IMAGE_DIMENSION |
| 93 | || height > MAX_IMAGE_DIMENSION |
| 94 | { |
| 95 | bail!("image dimensions exceed the decompression bomb guard; downscale or crop first"); |
| 96 | } |
| 97 | let mut reader = ImageReader::new(Cursor::new(bytes)).with_guessed_format()?; |
| 98 | reader.limits(limits()); |
| 99 | let decoded = reader |
| 100 | .decode() |
| 101 | .map_err(|_| anyhow::anyhow!("invalid image content or decode allocation limit"))?; |
| 102 | Ok((decoded, width, height)) |
| 103 | } |
| 104 | |
| 105 | /// Validate untrusted inline input before route selection or durable admission. |
| 106 | /// Return the existing provider-neutral history representation; no file is opened. |
| 107 | pub(crate) fn prepare_runtime_images(images: &[RuntimeImageInput]) -> Result<Vec<ContentBlock>> { |
| 108 | if images.len() > MAX_RUNTIME_IMAGES { |
| 109 | bail!("images exceed the {MAX_RUNTIME_IMAGES} attachment limit"); |
| 110 | } |
| 111 | prepare_images_with_limit( |
| 112 | images, |
| 113 | MAX_RUNTIME_IMAGE_BYTES, |
| 114 | Some(MAX_RUNTIME_IMAGE_TOTAL_BYTES), |
| 115 | ) |
| 116 | } |
| 117 | |
| 118 | /// Internal Engine/history input retains the established local 5 MiB ceiling. |
| 119 | /// Network callers must first pass `prepare_runtime_images` (4 MiB per image, |
| 120 | /// 10 images and 5 MiB total). Local history never had those aggregate/count |
| 121 | /// limits; impose only its existing per-image bound and bounded full decode. |
| 122 | pub(crate) fn prepare_stored_images(images: &[RuntimeImageInput]) -> Result<Vec<ContentBlock>> { |
| 123 | prepare_images_with_limit(images, MAX_IMAGE_BYTES, None) |
| 124 | } |
| 125 | |
| 126 | fn prepare_images_with_limit( |
| 127 | images: &[RuntimeImageInput], |
| 128 | per_image_limit: usize, |
| 129 | total_limit: Option<usize>, |
| 130 | ) -> Result<Vec<ContentBlock>> { |
| 131 | let mut total = 0usize; |
| 132 | images |
| 133 | .iter() |
| 134 | .enumerate() |
| 135 | .map(|(index, image)| { |
| 136 | if image.data_base64.len() > per_image_limit.div_ceil(3) * 4 { |
| 137 | bail!( |
| 138 | "image {} exceeds the {} MiB limit", |
| 139 | index + 1, |
| 140 | per_image_limit / (1024 * 1024) |
| 141 | ); |
| 142 | } |
| 143 | let bytes = STANDARD |
| 144 | .decode(&image.data_base64) |
| 145 | .map_err(|_| anyhow::anyhow!("image {} has invalid base64", index + 1))?; |
| 146 | if bytes.len() > per_image_limit { |
| 147 | bail!( |
| 148 | "image {} exceeds the {} MiB limit", |
| 149 | index + 1, |
| 150 | per_image_limit / (1024 * 1024) |
| 151 | ); |
| 152 | } |
| 153 | total = total.saturating_add(bytes.len()); |
| 154 | if total_limit.is_some_and(|limit| total > limit) { |
| 155 | bail!("images exceed the 5 MiB total limit"); |
| 156 | } |
| 157 | let attached = encode_image_bytes(&bytes, &format!("image {}", index + 1))?; |
| 158 | if image.mime != attached.media_type { |
| 159 | bail!("image {} MIME does not match its content", index + 1); |
| 160 | } |
| 161 | decode_and_guard_image(&bytes)?; |
| 162 | // Standard padded base64 is the one replay representation. |
| 163 | if STANDARD.encode(&bytes) != image.data_base64 { |
| 164 | bail!("image {} base64 is not canonical", index + 1); |
| 165 | } |
| 166 | Ok(attached.content_block()) |
| 167 | }) |
| 168 | .collect() |
| 169 | } |
| 170 | |
| 171 | /// Reuse durable canonical bytes for retry, never reread a path or URL. |
| 172 | pub(crate) fn runtime_images_from_blocks( |
| 173 | blocks: &[ContentBlock], |
| 174 | ) -> Result<Vec<RuntimeImageInput>> { |
| 175 | let mut images = Vec::new(); |
| 176 | for block in blocks { |
| 177 | if let ContentBlock::ImageUrl { image_url } = block { |
| 178 | if image_url.url.len() > MAX_IMAGE_BYTES.div_ceil(3) * 4 + 32 { |
| 179 | bail!("stored image exceeds the attachment limit"); |
| 180 | } |
| 181 | let (mime, data) = parse_data_url(&image_url.url) |
| 182 | .ok_or_else(|| anyhow::anyhow!("stored image requires canonical inline content"))?; |
| 183 | images.push(RuntimeImageInput { |
| 184 | mime: mime.to_string(), |
| 185 | data_base64: data.to_string(), |
| 186 | }); |
| 187 | } |
| 188 | } |
| 189 | prepare_stored_images(&images)?; |
| 190 | Ok(images) |
| 191 | } |
| 192 | |
| 193 | /// Validate new image-bearing durable records without rewriting their block order. |
| 194 | /// Legacy schema 2 history continues to use its original interpretation. |
| 195 | pub(crate) fn validate_stored_image_content(blocks: &[ContentBlock]) -> Result<()> { |
| 196 | if blocks.iter().any(|block| { |
| 197 | !matches!( |
| 198 | block, |
| 199 | ContentBlock::Text { .. } | ContentBlock::ImageUrl { .. } |
| 200 | ) |
| 201 | }) { |
| 202 | bail!("invalid persisted user image content kind"); |
| 203 | } |
| 204 | if runtime_images_from_blocks(blocks)?.is_empty() { |
| 205 | bail!("persisted image input must contain an image"); |
| 206 | } |
| 207 | Ok(()) |
| 208 | } |
| 209 | |
| 210 | /// Why a file could not be attached as an image. |
| 211 | /// |
| 212 | /// Every variant renders to a sentence naming the file and the reason. These |
| 213 | /// strings reach both the user (as a command error) and the model (as an |
| 214 | /// in-band notice), so they say what to do next rather than only what failed. |
| 215 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 216 | pub enum ImageAttachError { |
| 217 | /// The file could not be read at all. |
| 218 | Unreadable { path: String, reason: String }, |
| 219 | /// The file is zero bytes. |
| 220 | Empty { path: String }, |
| 221 | /// Over `limit` bytes: [`MAX_IMAGE_BYTES`] for already-encoded bytes, the |
| 222 | /// larger source bound when attach-time downscaling applies. |
| 223 | TooLarge { |
| 224 | path: String, |
| 225 | bytes: usize, |
| 226 | limit: usize, |
| 227 | }, |
| 228 | /// Magic bytes identify a format no provider in the set accepts. |
| 229 | UnsupportedFormat { path: String, detected: String }, |
| 230 | /// Magic bytes match nothing we recognize as an image. |
| 231 | NotAnImage { path: String }, |
| 232 | } |
| 233 | |
| 234 | impl std::fmt::Display for ImageAttachError { |
| 235 | fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { |
| 236 | match self { |
| 237 | Self::Unreadable { path, reason } => { |
| 238 | write!(f, "Cannot attach {path}: {reason}") |
| 239 | } |
| 240 | Self::Empty { path } => { |
| 241 | write!(f, "Cannot attach {path}: the file is empty") |
| 242 | } |
| 243 | Self::TooLarge { path, bytes, limit } => write!( |
| 244 | f, |
| 245 | "Cannot attach {path}: {} exceeds the {} per-image limit. \ |
| 246 | Downscale or crop it first.", |
| 247 | human_bytes(*bytes), |
| 248 | human_bytes(*limit), |
| 249 | ), |
| 250 | Self::UnsupportedFormat { path, detected } => write!( |
| 251 | f, |
| 252 | "Cannot attach {path}: {detected} is not accepted by vision \ |
| 253 | models. Convert it to PNG, JPEG, GIF or WebP.", |
| 254 | ), |
| 255 | Self::NotAnImage { path } => write!( |
| 256 | f, |
| 257 | "Cannot attach {path}: the file is not a PNG, JPEG, GIF or \ |
| 258 | WebP image (its contents do not match any of those formats).", |
| 259 | ), |
| 260 | } |
| 261 | } |
| 262 | } |
| 263 | |
| 264 | impl std::error::Error for ImageAttachError {} |
| 265 | |
| 266 | pub(crate) fn human_bytes(bytes: usize) -> String { |
| 267 | if bytes >= 1024 * 1024 { |
| 268 | format!("{:.1} MB", bytes as f64 / (1024.0 * 1024.0)) |
| 269 | } else if bytes >= 1024 { |
| 270 | format!("{:.1} KB", bytes as f64 / 1024.0) |
| 271 | } else { |
| 272 | format!("{bytes} bytes") |
| 273 | } |
| 274 | } |
| 275 | |
| 276 | /// An image that is ready to go on the wire. |
| 277 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 278 | pub struct AttachedImage { |
| 279 | /// e.g. `"image/png"`. |
| 280 | pub media_type: &'static str, |
| 281 | /// `data:<media_type>;base64,<payload>`. |
| 282 | pub data_url: String, |
| 283 | /// Size of the source file, before base64. |
| 284 | pub source_bytes: usize, |
| 285 | } |
| 286 | |
| 287 | /// Validated image content returned by lowercase `read`. |
| 288 | pub struct PreparedToolImage { |
| 289 | pub block: Option<codewhale_tools::ToolResultContentBlock>, |
| 290 | pub note: String, |
| 291 | } |
| 292 | |
| 293 | /// Prepare one provider-neutral tool-result image using Codewhale's existing |
| 294 | /// cross-provider format and size policy. Failure is a visible text receipt, |
| 295 | /// not a failed file read. |
| 296 | #[must_use] |
| 297 | pub fn prepare_tool_image_bytes(bytes: &[u8], mime_type: &str) -> PreparedToolImage { |
| 298 | let mime_type = mime_type.split(';').next().unwrap_or(mime_type).trim(); |
| 299 | let valid = bytes.len() <= MAX_IMAGE_BYTES |
| 300 | && sniff_media_type(bytes) == Some(mime_type) |
| 301 | && decode_and_guard_image(bytes).is_ok(); |
| 302 | if !valid { |
| 303 | return PreparedToolImage { |
| 304 | block: None, |
| 305 | note: format!( |
| 306 | "Read image file [{mime_type}]\n[Image omitted: unsupported, invalid, or above the 5 MiB inline limit.]" |
| 307 | ), |
| 308 | }; |
| 309 | } |
| 310 | PreparedToolImage { |
| 311 | block: Some(codewhale_tools::ToolResultContentBlock::Image { |
| 312 | mime_type: mime_type.to_string(), |
| 313 | data: STANDARD.encode(bytes), |
| 314 | }), |
| 315 | note: format!("Read image file [{mime_type}]"), |
| 316 | } |
| 317 | } |
| 318 | |
| 319 | fn valid_tool_image(mime_type: &str, data: &str) -> bool { |
| 320 | matches!( |
| 321 | mime_type, |
| 322 | "image/png" | "image/jpeg" | "image/gif" | "image/webp" |
| 323 | ) && data.len() <= MAX_IMAGE_BYTES.div_ceil(3) * 4 |
| 324 | && STANDARD.decode(data).is_ok_and(|bytes| { |
| 325 | bytes.len() <= MAX_IMAGE_BYTES |
| 326 | && sniff_media_type(&bytes) == Some(mime_type) |
| 327 | && decode_and_guard_image(&bytes).is_ok() |
| 328 | }) |
| 329 | } |
| 330 | |
| 331 | /// Enforce the same one-image limit at the tool execution boundary so plugin |
| 332 | /// or future rich tools cannot create unbounded history. |
| 333 | #[must_use] |
| 334 | pub(crate) fn bound_rich_tool_result( |
| 335 | mut rich: crate::tools::spec::RichToolResult, |
| 336 | ) -> crate::tools::spec::RichToolResult { |
| 337 | let mut kept = Vec::with_capacity(1); |
| 338 | let mut omitted = 0usize; |
| 339 | for block in rich.content_blocks.drain(..) { |
| 340 | let codewhale_tools::ToolResultContentBlock::Image { mime_type, data } = █ |
| 341 | if kept.is_empty() && valid_tool_image(mime_type, data) { |
| 342 | kept.push(block); |
| 343 | } else { |
| 344 | omitted += 1; |
| 345 | } |
| 346 | } |
| 347 | rich.result.content = tool_result_text_with_omission(&rich.result.content, omitted); |
| 348 | rich.content_blocks = kept; |
| 349 | rich |
| 350 | } |
| 351 | |
| 352 | /// Borrow the first valid inline image and count everything omitted. |
| 353 | #[must_use] |
| 354 | pub(crate) fn provider_tool_result_image_refs( |
| 355 | blocks: Option<&[serde_json::Value]>, |
| 356 | ) -> (Option<(&str, &str)>, usize) { |
| 357 | let mut image = None; |
| 358 | let mut omitted = 0usize; |
| 359 | for block in blocks.unwrap_or_default() { |
| 360 | let fields = block |
| 361 | .get("type") |
| 362 | .and_then(serde_json::Value::as_str) |
| 363 | .filter(|kind| *kind == "image") |
| 364 | .and_then(|_| { |
| 365 | block |
| 366 | .get("mime_type") |
| 367 | .and_then(serde_json::Value::as_str) |
| 368 | .zip(block.get("data").and_then(serde_json::Value::as_str)) |
| 369 | }); |
| 370 | if image.is_none() |
| 371 | && let Some((mime_type, data)) = fields |
| 372 | && valid_tool_image(mime_type, data) |
| 373 | { |
| 374 | image = Some((mime_type, data)); |
| 375 | } else { |
| 376 | omitted += 1; |
| 377 | } |
| 378 | } |
| 379 | (image, omitted) |
| 380 | } |
| 381 | |
| 382 | #[must_use] |
| 383 | pub(crate) fn tool_result_text_with_omission(content: &str, omitted: usize) -> String { |
| 384 | if omitted == 0 { |
| 385 | return content.to_string(); |
| 386 | } |
| 387 | format!( |
| 388 | "{content}\n[{omitted} tool-result image block(s) omitted: invalid, unsupported, oversized, or additional.]" |
| 389 | ) |
| 390 | } |
| 391 | |
| 392 | /// Copy/export projection that retains metadata but never inline base64. |
| 393 | #[must_use] |
| 394 | pub(crate) fn safe_tool_result_content_blocks( |
| 395 | blocks: Option<&[serde_json::Value]>, |
| 396 | ) -> Option<Vec<serde_json::Value>> { |
| 397 | blocks.map(|blocks| { |
| 398 | blocks |
| 399 | .iter() |
| 400 | .map(|block| { |
| 401 | if block.get("type").and_then(serde_json::Value::as_str) == Some("image") { |
| 402 | serde_json::json!({ |
| 403 | "type": "image", |
| 404 | "mime_type": block.get("mime_type").and_then(serde_json::Value::as_str).unwrap_or("application/octet-stream"), |
| 405 | "omission_code": "inline_or_local_image_payload", |
| 406 | "omitted_base64_bytes": block.get("data").and_then(serde_json::Value::as_str).map_or(0, str::len), |
| 407 | }) |
| 408 | } else { |
| 409 | block.clone() |
| 410 | } |
| 411 | }) |
| 412 | .collect() |
| 413 | }) |
| 414 | } |
| 415 | |
| 416 | #[must_use] |
| 417 | pub(crate) fn safe_tool_result_message_projection( |
| 418 | messages: &[codewhale_models::Message], |
| 419 | ) -> Vec<codewhale_models::Message> { |
| 420 | let mut projected = messages.to_vec(); |
| 421 | for message in &mut projected { |
| 422 | for block in &mut message.content { |
| 423 | if let ContentBlock::ToolResult { content_blocks, .. } = block { |
| 424 | *content_blocks = safe_tool_result_content_blocks(content_blocks.as_deref()); |
| 425 | } |
| 426 | } |
| 427 | } |
| 428 | projected |
| 429 | } |
| 430 | |
| 431 | impl AttachedImage { |
| 432 | /// The content block this image becomes in a message. |
| 433 | #[must_use] |
| 434 | pub fn content_block(&self) -> ContentBlock { |
| 435 | ContentBlock::ImageUrl { |
| 436 | image_url: ImageUrlContent { |
| 437 | url: self.data_url.clone(), |
| 438 | }, |
| 439 | } |
| 440 | } |
| 441 | } |
| 442 | |
| 443 | /// Identify an image format from its leading bytes. |
| 444 | /// |
| 445 | /// Extension sniffing is not enough here: the extension is attacker- and |
| 446 | /// typo-controlled, while what the provider validates is the payload. A |
| 447 | /// `.png` holding JPEG bytes must be declared `image/jpeg` or the request is |
| 448 | /// rejected with a media-type mismatch that reads like a CodeWhale bug. |
| 449 | /// |
| 450 | /// Returns `None` for anything that is not one of the four accepted formats; |
| 451 | /// [`detect_rejected_format`] names the near-misses so the error can be |
| 452 | /// specific. |
| 453 | #[must_use] |
| 454 | pub fn sniff_media_type(bytes: &[u8]) -> Option<&'static str> { |
| 455 | if bytes.starts_with(b"\x89PNG\r\n\x1a\n") { |
| 456 | return Some("image/png"); |
| 457 | } |
| 458 | if bytes.starts_with(b"\xff\xd8\xff") { |
| 459 | return Some("image/jpeg"); |
| 460 | } |
| 461 | if bytes.starts_with(b"GIF87a") || bytes.starts_with(b"GIF89a") { |
| 462 | return Some("image/gif"); |
| 463 | } |
| 464 | if bytes.len() >= 12 && bytes.starts_with(b"RIFF") && &bytes[8..12] == b"WEBP" { |
| 465 | return Some("image/webp"); |
| 466 | } |
| 467 | None |
| 468 | } |
| 469 | |
| 470 | /// Name a format we can recognize but deliberately refuse. |
| 471 | /// |
| 472 | /// These are real images, so `NotAnImage` would be a lie and would send the |
| 473 | /// user looking for a corrupt file. Naming the format points at the fix |
| 474 | /// (convert it) instead. |
| 475 | #[must_use] |
| 476 | pub fn detect_rejected_format(bytes: &[u8]) -> Option<&'static str> { |
| 477 | if bytes.starts_with(b"BM") { |
| 478 | return Some("BMP"); |
| 479 | } |
| 480 | if bytes.starts_with(b"II\x2a\x00") || bytes.starts_with(b"MM\x00\x2a") { |
| 481 | return Some("TIFF"); |
| 482 | } |
| 483 | if bytes.len() >= 12 && bytes.starts_with(b"\0\0\0") && &bytes[4..8] == b"ftyp" { |
| 484 | return Some("HEIC/AVIF"); |
| 485 | } |
| 486 | if bytes.starts_with(b"<svg") || bytes.starts_with(b"<?xml") { |
| 487 | return Some("SVG"); |
| 488 | } |
| 489 | if bytes.starts_with(b"%PDF") { |
| 490 | return Some("PDF"); |
| 491 | } |
| 492 | None |
| 493 | } |
| 494 | |
| 495 | /// Validate and encode raw bytes that were read from `path`. |
| 496 | /// |
| 497 | /// Split from [`attach_image_from_path`] so the whole policy — order of |
| 498 | /// checks, limits, format verdicts — is testable without touching a |
| 499 | /// filesystem. |
| 500 | pub fn encode_image_bytes(bytes: &[u8], path: &str) -> Result<AttachedImage, ImageAttachError> { |
| 501 | if bytes.is_empty() { |
| 502 | return Err(ImageAttachError::Empty { |
| 503 | path: path.to_string(), |
| 504 | }); |
| 505 | } |
| 506 | // Size is checked before format so a huge file is rejected on the cheap |
| 507 | // fact rather than after we have decided what it is. |
| 508 | if bytes.len() > MAX_IMAGE_BYTES { |
| 509 | return Err(ImageAttachError::TooLarge { |
| 510 | path: path.to_string(), |
| 511 | bytes: bytes.len(), |
| 512 | limit: MAX_IMAGE_BYTES, |
| 513 | }); |
| 514 | } |
| 515 | let Some(media_type) = sniff_media_type(bytes) else { |
| 516 | return Err(format_error(bytes, path)); |
| 517 | }; |
| 518 | let payload = STANDARD.encode(bytes); |
| 519 | Ok(AttachedImage { |
| 520 | media_type, |
| 521 | data_url: format!("data:{media_type};base64,{payload}"), |
| 522 | source_bytes: bytes.len(), |
| 523 | }) |
| 524 | } |
| 525 | |
| 526 | fn format_error(bytes: &[u8], path: &str) -> ImageAttachError { |
| 527 | match detect_rejected_format(bytes) { |
| 528 | Some(detected) => ImageAttachError::UnsupportedFormat { |
| 529 | path: path.to_string(), |
| 530 | detected: detected.to_string(), |
| 531 | }, |
| 532 | None => ImageAttachError::NotAnImage { |
| 533 | path: path.to_string(), |
| 534 | }, |
| 535 | } |
| 536 | } |
| 537 | |
| 538 | /// Longest edge an attached image is sent at. A Retina screenshot is 3–6k px |
| 539 | /// and often over [`MAX_IMAGE_BYTES`] as PNG; larger images are downscaled |
| 540 | /// and re-encoded when attached rather than refused. |
| 541 | pub const ATTACH_MAX_EDGE_PX: u32 = 2048; |
| 542 | |
| 543 | /// Read, validate and encode an image file. |
| 544 | /// |
| 545 | /// Images over [`ATTACH_MAX_EDGE_PX`] or [`MAX_IMAGE_BYTES`] are decoded under |
| 546 | /// the decompression-bomb guard, fitted to the edge and re-encoded on |
| 547 | /// `read_media`'s budget ladder (PNG for flat or alpha content, JPEG for |
| 548 | /// photos). Known limit: an animated GIF that needs downscaling keeps only |
| 549 | /// its first frame. Sources above `read_media`'s source bound are refused. |
| 550 | pub fn attach_image_from_path(path: &Path) -> Result<AttachedImage, ImageAttachError> { |
| 551 | let display = path.display().to_string(); |
| 552 | let source_limit = crate::tools::read_media::MAX_SOURCE_IMAGE_BYTES; |
| 553 | // Check the size from metadata first so a multi-gigabyte file is refused |
| 554 | // without being read into memory. |
| 555 | if let Ok(meta) = std::fs::metadata(path) { |
| 556 | let len = meta.len(); |
| 557 | if len > source_limit as u64 { |
| 558 | return Err(ImageAttachError::TooLarge { |
| 559 | path: display, |
| 560 | bytes: usize::try_from(len).unwrap_or(usize::MAX), |
| 561 | limit: source_limit, |
| 562 | }); |
| 563 | } |
| 564 | } |
| 565 | let bytes = std::fs::read(path).map_err(|error| ImageAttachError::Unreadable { |
| 566 | path: display.clone(), |
| 567 | reason: error.to_string(), |
| 568 | })?; |
| 569 | if bytes.len() > source_limit { |
| 570 | return Err(ImageAttachError::TooLarge { |
| 571 | path: display, |
| 572 | bytes: bytes.len(), |
| 573 | limit: source_limit, |
| 574 | }); |
| 575 | } |
| 576 | let oversized_edge = ImageReader::new(Cursor::new(&bytes)) |
| 577 | .with_guessed_format() |
| 578 | .ok() |
| 579 | .and_then(|reader| reader.into_dimensions().ok()) |
| 580 | .is_some_and(|(width, height)| width.max(height) > ATTACH_MAX_EDGE_PX); |
| 581 | if bytes.len() <= MAX_IMAGE_BYTES && !oversized_edge { |
| 582 | return encode_image_bytes(&bytes, &display); |
| 583 | } |
| 584 | if sniff_media_type(&bytes).is_none() { |
| 585 | return Err(format_error(&bytes, &display)); |
| 586 | } |
| 587 | let unreadable = |reason: String| ImageAttachError::Unreadable { |
| 588 | path: display.clone(), |
| 589 | reason, |
| 590 | }; |
| 591 | let (image, _, _) = |
| 592 | decode_and_guard_image(&bytes).map_err(|error| unreadable(error.to_string()))?; |
| 593 | let (encoded, _) = |
| 594 | crate::tools::read_media::fit_and_encode(&image, ATTACH_MAX_EDGE_PX, MAX_IMAGE_BYTES, path) |
| 595 | .map_err(|error| unreadable(error.to_string()))?; |
| 596 | encode_image_bytes(&encoded, &display) |
| 597 | } |
| 598 | |
| 599 | /// Image blocks sent from the latest user prompt onward: this turn's |
| 600 | /// attachments and tool-result images, not ones replayed from history. |
| 601 | #[must_use] |
| 602 | pub fn images_since_last_user_prompt(messages: &[codewhale_models::Message]) -> usize { |
| 603 | let is_prompt = |message: &codewhale_models::Message| { |
| 604 | message.role == codewhale_models::Role::User |
| 605 | && message.content.iter().any(|block| { |
| 606 | matches!( |
| 607 | block, |
| 608 | ContentBlock::Text { .. } | ContentBlock::ImageUrl { .. } |
| 609 | ) |
| 610 | }) |
| 611 | }; |
| 612 | let start = messages.iter().rposition(is_prompt).unwrap_or(0); |
| 613 | messages[start..] |
| 614 | .iter() |
| 615 | .flat_map(|message| &message.content) |
| 616 | .map(|block| match block { |
| 617 | ContentBlock::ImageUrl { .. } => 1, |
| 618 | ContentBlock::ToolResult { content_blocks, .. } => { |
| 619 | content_blocks.as_ref().map_or(0, Vec::len) |
| 620 | } |
| 621 | _ => 0, |
| 622 | }) |
| 623 | .sum() |
| 624 | } |
| 625 | |
| 626 | /// Split a `data:<media-type>;base64,<payload>` URL. |
| 627 | /// |
| 628 | /// Anthropic's Messages API models an image as a tagged `source` rather than a |
| 629 | /// URL, so the native route has to take the data URL back apart. Returns |
| 630 | /// `None` for `http(s)` URLs and for anything malformed, which the caller |
| 631 | /// renders as a remote source or a visible degradation respectively. |
| 632 | #[must_use] |
| 633 | pub fn parse_data_url(url: &str) -> Option<(&str, &str)> { |
| 634 | let rest = url.strip_prefix("data:")?; |
| 635 | let (header, payload) = rest.split_once(',')?; |
| 636 | let media_type = header.strip_suffix(";base64")?; |
| 637 | if media_type.is_empty() || payload.is_empty() { |
| 638 | return None; |
| 639 | } |
| 640 | Some((media_type, payload)) |
| 641 | } |
| 642 | |
| 643 | /// Whether a URL is one a provider can fetch for itself. |
| 644 | #[must_use] |
| 645 | pub fn is_remote_image_url(url: &str) -> bool { |
| 646 | url.starts_with("https://") || url.starts_with("http://") |
| 647 | } |
| 648 | |
| 649 | /// The outcome of expanding a user turn's attachment placeholders. |
| 650 | #[derive(Debug, Clone, Default, PartialEq)] |
| 651 | pub struct ExpandedAttachments { |
| 652 | /// Image blocks to append to the user message, in placeholder order. |
| 653 | pub blocks: Vec<ContentBlock>, |
| 654 | /// One line per attachment that could not be sent. These are shown to the |
| 655 | /// user and also handed to the model, because a model that is not told an |
| 656 | /// image was dropped will confidently discuss it from the filename. |
| 657 | pub notices: Vec<String>, |
| 658 | } |
| 659 | |
| 660 | /// Build the image blocks for a user turn from its `[Attached image: …]` lines. |
| 661 | /// |
| 662 | /// This is the ingest half of the composer's placeholder design: the buffer |
| 663 | /// holds a path-bearing text line (which survives editing, history and session |
| 664 | /// reload for free), and the bytes are read here, once, as the message is |
| 665 | /// built. |
| 666 | /// |
| 667 | /// Only *permanent* failures are decided here — a file that is missing, |
| 668 | /// oversized, or not an image will still be all of those things next turn, so |
| 669 | /// baking the verdict into history costs nothing. Whether the *model* can see |
| 670 | /// images is deliberately not decided here; that is contingent on the active |
| 671 | /// route and is re-decided per request by |
| 672 | /// [`strip_images_when_unsupported`]. |
| 673 | /// |
| 674 | /// Each image is bracketed by text tags naming its path. Without them a turn |
| 675 | /// carrying three screenshots gives the model three anonymous images in a row |
| 676 | /// and no way to say which is which. |
| 677 | /// |
| 678 | /// Never returns an error: a turn with one bad attachment should still be |
| 679 | /// sent, with the failure stated in-band rather than swallowed. |
| 680 | #[must_use] |
| 681 | pub fn expand_attachment_blocks(text: &str) -> ExpandedAttachments { |
| 682 | let references = codewhale_core::media_attachment_references(text); |
| 683 | let mut out = ExpandedAttachments::default(); |
| 684 | for reference in references { |
| 685 | if reference.kind != "image" { |
| 686 | // Video and any future kind: left as the text reference it |
| 687 | // already was. Silently ignoring it here is not a drop — the |
| 688 | // path is still in the prompt, exactly as before this module |
| 689 | // existed. |
| 690 | continue; |
| 691 | } |
| 692 | match attach_image_from_path(Path::new(&reference.path)) { |
| 693 | Ok(image) => { |
| 694 | out.blocks |
| 695 | .push(tag_block(&format!("<image path=\"{}\">", reference.path))); |
| 696 | out.blocks.push(image.content_block()); |
| 697 | out.blocks.push(tag_block("</image>")); |
| 698 | } |
| 699 | Err(error) => out.notices.push(error.to_string()), |
| 700 | } |
| 701 | } |
| 702 | out |
| 703 | } |
| 704 | |
| 705 | fn tag_block(text: &str) -> ContentBlock { |
| 706 | ContentBlock::Text { |
| 707 | text: text.to_string(), |
| 708 | cache_control: None, |
| 709 | } |
| 710 | } |
| 711 | |
| 712 | /// Replace every image in a request with text when the route cannot see them. |
| 713 | /// |
| 714 | /// Capability is a property of the *route*, not of the attachment, and the |
| 715 | /// route changes freely mid-session. Deciding this when the message is built |
| 716 | /// would burn the answer into history: attach a screenshot while a text-only |
| 717 | /// model is selected, switch to a vision model, and the image would be gone |
| 718 | /// for good. So history always keeps the real image and each outbound request |
| 719 | /// is normalized against the model it is actually going to. |
| 720 | /// |
| 721 | /// Only a known `Unsupported` strips. Most routes report `Unknown` because |
| 722 | /// models.dev has no modality data for them, and treating unknown as "no" |
| 723 | /// would make the feature dead on arrival for exactly the self-hosted and |
| 724 | /// custom routes that most need it — so `Unknown` sends the image and lets the |
| 725 | /// provider be the authority. |
| 726 | /// |
| 727 | /// The image is replaced in place rather than removed, so the model is told |
| 728 | /// why it is looking at a gap instead of being left to invent one. |
| 729 | pub fn strip_images_when_unsupported( |
| 730 | messages: &mut [codewhale_models::Message], |
| 731 | vision: SupportState, |
| 732 | model: &str, |
| 733 | ) -> usize { |
| 734 | if vision != SupportState::Unsupported { |
| 735 | return 0; |
| 736 | } |
| 737 | let mut stripped = 0; |
| 738 | for message in messages.iter_mut() { |
| 739 | for block in &mut message.content { |
| 740 | match block { |
| 741 | ContentBlock::ImageUrl { .. } => { |
| 742 | *block = ContentBlock::Text { |
| 743 | text: format!( |
| 744 | "[image content omitted: the active model ({model}) does \ |
| 745 | not accept image input. Use the image_ocr tool to read \ |
| 746 | text from it, or switch to a vision-capable model with \ |
| 747 | /model.]" |
| 748 | ), |
| 749 | cache_control: None, |
| 750 | }; |
| 751 | stripped += 1; |
| 752 | } |
| 753 | ContentBlock::ToolResult { |
| 754 | content, |
| 755 | content_blocks, |
| 756 | .. |
| 757 | } => { |
| 758 | let count = content_blocks.as_ref().map_or(0, Vec::len); |
| 759 | if count > 0 { |
| 760 | *content_blocks = None; |
| 761 | *content = format!( |
| 762 | "{content}\n[{count} image block(s) omitted: the active model ({model}) does not accept image input. Use the image_ocr tool to read text from it, or switch to a vision-capable model with /model.]" |
| 763 | ); |
| 764 | stripped += count; |
| 765 | } |
| 766 | } |
| 767 | _ => {} |
| 768 | } |
| 769 | } |
| 770 | } |
| 771 | stripped |
| 772 | } |
| 773 | |
| 774 | /// Total inline-image byte budget for one compaction retry that follows a |
| 775 | /// provider body-size rejection. |
| 776 | /// |
| 777 | /// An HTTP 413 boundary caps the request *body*, and the token-side context |
| 778 | /// budget that governs compaction cannot see it: an image that costs a flat |
| 779 | /// token estimate can still spend megabytes of base64. Nothing consults this |
| 780 | /// budget on the happy path — it exists only to recover a summary call that |
| 781 | /// has already been refused, where the cheapest useful move is to re-encode |
| 782 | /// the inline images under a cap at or below the smallest provider body limits |
| 783 | /// seen in practice and retry. |
| 784 | pub(crate) const COMPACTION_IMAGE_TOTAL_BUDGET_BYTES: usize = 2 * 1024 * 1024; |
| 785 | |
| 786 | /// Smallest per-image share of that budget, so an image-heavy session still |
| 787 | /// re-encodes each image instead of dividing the budget down to nothing. |
| 788 | const COMPACTION_IMAGE_MIN_BUDGET_BYTES: usize = 96 * 1024; |
| 789 | |
| 790 | /// Longest edge a re-encoded inline image keeps, in pixels. |
| 791 | const COMPACTION_IMAGE_MAX_EDGE: u32 = 1024; |
| 792 | |
| 793 | /// Longest edge the shrink ladder descends to before giving up on an image. |
| 794 | const COMPACTION_IMAGE_MIN_EDGE: u32 = 128; |
| 795 | |
| 796 | /// JPEG quality for re-encoded images that carry no meaningful alpha. |
| 797 | const COMPACTION_IMAGE_JPEG_QUALITY: u8 = 80; |
| 798 | |
| 799 | /// What one inline-image shrink pass changed. |
| 800 | #[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] |
| 801 | pub(crate) struct ShrunkInlineImages { |
| 802 | /// Images that were re-encoded smaller. |
| 803 | pub images: usize, |
| 804 | /// Inline images the request carried, rewritten or not. |
| 805 | /// |
| 806 | /// `images == 0` alone cannot tell "nothing to do because every image |
| 807 | /// already fits" from "there were no images"; a request-size ladder must |
| 808 | /// not conflate the two, because only the second makes the next rung |
| 809 | /// pointless. |
| 810 | pub images_seen: usize, |
| 811 | /// Decoded bytes of those images before the pass. |
| 812 | pub bytes_before: usize, |
| 813 | /// Decoded bytes of those images after the pass. |
| 814 | pub bytes_after: usize, |
| 815 | } |
| 816 | |
| 817 | /// Re-encode every inline image in `messages` under a total byte budget. |
| 818 | /// |
| 819 | /// Covers both carriers an outbound request can hold: an `image_url` block, |
| 820 | /// and the stored tool-result shape (`{"type":"image","mime_type","data"}`) |
| 821 | /// that the wire projection reads back out via |
| 822 | /// [`provider_tool_result_image_refs`]. |
| 823 | /// |
| 824 | /// `images == 0` means nothing was rewritten — either there were no inline |
| 825 | /// images, or each was already under its share of the budget. Check |
| 826 | /// [`ShrunkInlineImages::images_seen`] to tell those apart: a caller that is |
| 827 | /// still looking at a body-size rejection must climb to the next rung rather |
| 828 | /// than resend the same bytes, and when images are present but nothing was |
| 829 | /// rewritten, the next rung is the only one that can still change the payload. |
| 830 | pub(crate) fn shrink_images_for_request( |
| 831 | messages: &mut [codewhale_models::Message], |
| 832 | ) -> ShrunkInlineImages { |
| 833 | shrink_images_for_request_with_budget(messages, COMPACTION_IMAGE_TOTAL_BUDGET_BYTES) |
| 834 | } |
| 835 | |
| 836 | /// Budget-parameterized core of [`shrink_images_for_request`], kept separate |
| 837 | /// so tests can drive it with small budgets and small images. |
| 838 | pub(crate) fn shrink_images_for_request_with_budget( |
| 839 | messages: &mut [codewhale_models::Message], |
| 840 | budget: usize, |
| 841 | ) -> ShrunkInlineImages { |
| 842 | let sizes = inline_image_sizes(messages); |
| 843 | let total: usize = sizes.iter().sum(); |
| 844 | let images_seen = sizes.len(); |
| 845 | if total <= budget { |
| 846 | // Images may be present and simply already fit: report them as seen |
| 847 | // without rewriting, so a caller can tell "nothing to shrink" from |
| 848 | // "nothing there". |
| 849 | return ShrunkInlineImages { |
| 850 | images_seen, |
| 851 | ..ShrunkInlineImages::default() |
| 852 | }; |
| 853 | } |
| 854 | // A share of the budget per image, floored so a few large images still come |
| 855 | // back readable. With many images the floor can push the total past |
| 856 | // `budget`; the caller's next rung (replace with notes) covers that case, |
| 857 | // not a tighter share here. |
| 858 | let per_image = (budget / sizes.len()).max(COMPACTION_IMAGE_MIN_BUDGET_BYTES); |
| 859 | let mut outcome = ShrunkInlineImages::default(); |
| 860 | for message in messages.iter_mut() { |
| 861 | for block in &mut message.content { |
| 862 | match block { |
| 863 | ContentBlock::ImageUrl { image_url } => { |
| 864 | if let Some((url, shrunk)) = shrink_data_url(&image_url.url, per_image) { |
| 865 | image_url.url = url; |
| 866 | outcome.images += 1; |
| 867 | outcome.bytes_before += shrunk.before; |
| 868 | outcome.bytes_after += shrunk.after; |
| 869 | } |
| 870 | } |
| 871 | ContentBlock::ToolResult { content_blocks, .. } => { |
| 872 | let Some(blocks) = content_blocks.as_mut() else { |
| 873 | continue; |
| 874 | }; |
| 875 | for value in blocks.iter_mut() { |
| 876 | if let Some(shrunk) = shrink_stored_tool_image(value, per_image) { |
| 877 | outcome.images += 1; |
| 878 | outcome.bytes_before += shrunk.before; |
| 879 | outcome.bytes_after += shrunk.after; |
| 880 | } |
| 881 | } |
| 882 | } |
| 883 | _ => {} |
| 884 | } |
| 885 | } |
| 886 | } |
| 887 | outcome.images_seen = images_seen; |
| 888 | outcome |
| 889 | } |
| 890 | |
| 891 | /// Replace every inline image with a text note that keeps the image's |
| 892 | /// existence visible without its bytes. |
| 893 | /// |
| 894 | /// This is the last rung of the request-size ladder. A summary pass that still |
| 895 | /// exceeds the provider's body cap after re-encoding has nothing left to give |
| 896 | /// but the pixels; the note keeps the fact that a tool or the user supplied an |
| 897 | /// image — and how large it was — so the handoff can still say so instead of |
| 898 | /// silently dropping the detail. Session history keeps the real image; this |
| 899 | /// only rewrites the outbound copy. |
| 900 | /// |
| 901 | /// Returns the number of images replaced. |
| 902 | pub(crate) fn replace_images_with_placeholders( |
| 903 | messages: &mut [codewhale_models::Message], |
| 904 | reason: &str, |
| 905 | ) -> usize { |
| 906 | let mut replaced = 0; |
| 907 | for message in messages.iter_mut() { |
| 908 | for block in &mut message.content { |
| 909 | match block { |
| 910 | ContentBlock::ImageUrl { image_url } => { |
| 911 | let bytes = parse_data_url(&image_url.url) |
| 912 | .map(|(_, payload)| decoded_len_estimate(payload)); |
| 913 | *block = ContentBlock::Text { |
| 914 | text: image_placeholder_note(1, bytes, reason), |
| 915 | cache_control: None, |
| 916 | }; |
| 917 | replaced += 1; |
| 918 | } |
| 919 | ContentBlock::ToolResult { |
| 920 | content, |
| 921 | content_blocks, |
| 922 | .. |
| 923 | } => { |
| 924 | let Some(blocks) = content_blocks.as_ref() else { |
| 925 | continue; |
| 926 | }; |
| 927 | let mut count = 0usize; |
| 928 | let mut bytes = 0usize; |
| 929 | for value in blocks { |
| 930 | if let Some((_, payload)) = stored_tool_image_payload(value) { |
| 931 | count += 1; |
| 932 | bytes += decoded_len_estimate(payload); |
| 933 | } |
| 934 | } |
| 935 | if count == 0 { |
| 936 | continue; |
| 937 | } |
| 938 | // Same shape as `strip_images_when_unsupported`: the |
| 939 | // wire projection reads tool images back out of |
| 940 | // `content_blocks` and would count a text block placed |
| 941 | // there as an omitted image. |
| 942 | *content_blocks = None; |
| 943 | *content = format!( |
| 944 | "{content}\n{}", |
| 945 | image_placeholder_note(count, Some(bytes), reason) |
| 946 | ); |
| 947 | replaced += count; |
| 948 | } |
| 949 | _ => {} |
| 950 | } |
| 951 | } |
| 952 | } |
| 953 | replaced |
| 954 | } |
| 955 | |
| 956 | /// The in-band note that replaces an inline image byte payload for one |
| 957 | /// summary pass. It tells the summarizer what was there and what to do |
| 958 | /// about it (refer to it as an image; never invent its contents), rather than |
| 959 | /// leaving a silent gap. |
| 960 | fn image_placeholder_note(count: usize, bytes: Option<usize>, reason: &str) -> String { |
| 961 | let size = bytes.map_or(String::new(), |bytes| format!(" (~{})", human_bytes(bytes))); |
| 962 | format!( |
| 963 | "[{count} image(s){size} omitted from this summary pass: {reason}. \ |
| 964 | The image(s) remain in the session. Refer to them only as images the \ |
| 965 | conversation included; do not describe or guess what they showed.]" |
| 966 | ) |
| 967 | } |
| 968 | |
| 969 | /// Base64 length to decoded length: four characters carry three bytes. |
| 970 | fn decoded_len_estimate(payload: &str) -> usize { |
| 971 | payload.len() / 4 * 3 |
| 972 | } |
| 973 | |
| 974 | /// Decoded-size estimates for every inline image, in request order. |
| 975 | fn inline_image_sizes(messages: &[codewhale_models::Message]) -> Vec<usize> { |
| 976 | let mut sizes = Vec::new(); |
| 977 | for message in messages { |
| 978 | for block in &message.content { |
| 979 | match block { |
| 980 | ContentBlock::ImageUrl { image_url } => { |
| 981 | if let Some((_, payload)) = parse_data_url(&image_url.url) { |
| 982 | sizes.push(decoded_len_estimate(payload)); |
| 983 | } |
| 984 | } |
| 985 | ContentBlock::ToolResult { content_blocks, .. } => { |
| 986 | if let Some(blocks) = content_blocks.as_ref() { |
| 987 | for value in blocks { |
| 988 | if let Some((_, payload)) = stored_tool_image_payload(value) { |
| 989 | sizes.push(decoded_len_estimate(payload)); |
| 990 | } |
| 991 | } |
| 992 | } |
| 993 | } |
| 994 | _ => {} |
| 995 | } |
| 996 | } |
| 997 | } |
| 998 | sizes |
| 999 | } |
| 1000 | |
| 1001 | /// The `(mime_type, data)` pair of a stored tool-result image block, if the |
| 1002 | /// block is one. |
| 1003 | fn stored_tool_image_payload(value: &serde_json::Value) -> Option<(&str, &str)> { |
| 1004 | if value.get("type").and_then(serde_json::Value::as_str) != Some("image") { |
| 1005 | return None; |
| 1006 | } |
| 1007 | let mime_type = value.get("mime_type").and_then(serde_json::Value::as_str)?; |
| 1008 | let data = value.get("data").and_then(serde_json::Value::as_str)?; |
| 1009 | Some((mime_type, data)) |
| 1010 | } |
| 1011 | |
| 1012 | /// One rewritten image, in decoded bytes. |
| 1013 | #[derive(Debug, Clone, Copy)] |
| 1014 | struct ShrunkPayload { |
| 1015 | before: usize, |
| 1016 | after: usize, |
| 1017 | } |
| 1018 | |
| 1019 | /// Re-encode a `data:` URL image under `budget`, returning the replacement |
| 1020 | /// URL plus the byte delta. |
| 1021 | fn shrink_data_url(url: &str, budget: usize) -> Option<(String, ShrunkPayload)> { |
| 1022 | let (_, payload) = parse_data_url(url)?; |
| 1023 | let (mime, data, shrunk) = shrink_base64_image(payload, budget)?; |
| 1024 | Some((format!("data:{mime};base64,{data}"), shrunk)) |
| 1025 | } |
| 1026 | |
| 1027 | /// Re-encode one stored tool-result image block in place. |
| 1028 | fn shrink_stored_tool_image(value: &mut serde_json::Value, budget: usize) -> Option<ShrunkPayload> { |
| 1029 | let (_, payload) = stored_tool_image_payload(value)?; |
| 1030 | let (mime, data, shrunk) = shrink_base64_image(payload, budget)?; |
| 1031 | let object = value.as_object_mut()?; |
| 1032 | object.insert("mime_type".to_string(), serde_json::Value::String(mime)); |
| 1033 | object.insert("data".to_string(), serde_json::Value::String(data)); |
| 1034 | Some(shrunk) |
| 1035 | } |
| 1036 | |
| 1037 | /// Re-encode one base64 inline image to fit `budget` decoded bytes. |
| 1038 | /// |
| 1039 | /// `None` means the image already fits, cannot be decoded, or the re-encode |
| 1040 | /// came back no smaller — so a caller's "did anything change" tally stays |
| 1041 | /// honest about what a retry would actually send. |
| 1042 | fn shrink_base64_image(data: &str, budget: usize) -> Option<(String, String, ShrunkPayload)> { |
| 1043 | let bytes = STANDARD.decode(data).ok()?; |
| 1044 | if bytes.len() <= budget { |
| 1045 | return None; |
| 1046 | } |
| 1047 | let (decoded, _, _) = decode_and_guard_image(&bytes).ok()?; |
| 1048 | let (mime, encoded) = reencode_within_budget(decoded, budget)?; |
| 1049 | if encoded.len() >= bytes.len() { |
| 1050 | return None; |
| 1051 | } |
| 1052 | Some(( |
| 1053 | mime.to_string(), |
| 1054 | STANDARD.encode(&encoded), |
| 1055 | ShrunkPayload { |
| 1056 | before: bytes.len(), |
| 1057 | after: encoded.len(), |
| 1058 | }, |
| 1059 | )) |
| 1060 | } |
| 1061 | |
| 1062 | /// Encode `image` down a short ladder until it fits `budget`. |
| 1063 | /// |
| 1064 | /// Rung order: longest edge capped at [`COMPACTION_IMAGE_MAX_EDGE`], then |
| 1065 | /// halved until [`COMPACTION_IMAGE_MIN_EDGE`]. Alpha-bearing images stay PNG |
| 1066 | /// (JPEG would flatten transparency); everything else becomes JPEG, which is |
| 1067 | /// what actually makes screenshots and artwork small. The smallest rung wins |
| 1068 | /// even when nothing fits `budget`, because a smaller-than-before payload is |
| 1069 | /// still progress for the retry. |
| 1070 | fn reencode_within_budget(image: DynamicImage, budget: usize) -> Option<(&'static str, Vec<u8>)> { |
| 1071 | let mut current = image; |
| 1072 | let (mut width, mut height) = current.dimensions(); |
| 1073 | if width.max(height) > COMPACTION_IMAGE_MAX_EDGE { |
| 1074 | let scale = f64::from(COMPACTION_IMAGE_MAX_EDGE) / f64::from(width.max(height)); |
| 1075 | let scaled_width = ((f64::from(width) * scale).round() as u32).max(1); |
| 1076 | let scaled_height = ((f64::from(height) * scale).round() as u32).max(1); |
| 1077 | current = current.resize(scaled_width, scaled_height, FilterType::Lanczos3); |
| 1078 | (width, height) = current.dimensions(); |
| 1079 | } |
| 1080 | let mut smallest: Option<(&'static str, Vec<u8>)> = None; |
| 1081 | loop { |
| 1082 | if let Some(candidate) = encode_inline_image(¤t) |
| 1083 | && smallest |
| 1084 | .as_ref() |
| 1085 | .is_none_or(|(_, best)| candidate.1.len() < best.len()) |
| 1086 | { |
| 1087 | smallest = Some(candidate); |
| 1088 | } |
| 1089 | if smallest |
| 1090 | .as_ref() |
| 1091 | .is_some_and(|(_, bytes)| bytes.len() <= budget) |
| 1092 | { |
| 1093 | break; |
| 1094 | } |
| 1095 | if width.max(height) <= COMPACTION_IMAGE_MIN_EDGE { |
| 1096 | break; |
| 1097 | } |
| 1098 | (width, height) = ((width / 2).max(1), (height / 2).max(1)); |
| 1099 | current = current.resize(width, height, FilterType::Lanczos3); |
| 1100 | } |
| 1101 | smallest |
| 1102 | } |
| 1103 | |
| 1104 | /// One encoder pass: PNG when alpha matters, JPEG otherwise. |
| 1105 | fn encode_inline_image(image: &DynamicImage) -> Option<(&'static str, Vec<u8>)> { |
| 1106 | let (width, height) = image.dimensions(); |
| 1107 | let mut bytes = Vec::new(); |
| 1108 | if image.color().has_alpha() { |
| 1109 | let rgba = image.to_rgba8(); |
| 1110 | PngEncoder::new_with_quality(&mut bytes, CompressionType::Best, PngFilter::Adaptive) |
| 1111 | .write_image(rgba.as_raw(), width, height, ExtendedColorType::Rgba8) |
| 1112 | .ok()?; |
| 1113 | Some(("image/png", bytes)) |
| 1114 | } else { |
| 1115 | let rgb = image.to_rgb8(); |
| 1116 | JpegEncoder::new_with_quality(&mut bytes, COMPACTION_IMAGE_JPEG_QUALITY) |
| 1117 | .write_image(rgb.as_raw(), width, height, ExtendedColorType::Rgb8) |
| 1118 | .ok()?; |
| 1119 | Some(("image/jpeg", bytes)) |
| 1120 | } |
| 1121 | } |
| 1122 | |
| 1123 | /// Render dropped-attachment notices as a block the model will read. |
| 1124 | /// |
| 1125 | /// Wrapped in a tag rather than appended as bare prose so the model can tell |
| 1126 | /// the difference between the user saying something and the harness reporting |
| 1127 | /// on itself. |
| 1128 | #[must_use] |
| 1129 | pub fn notice_block(notices: &[String]) -> Option<ContentBlock> { |
| 1130 | if notices.is_empty() { |
| 1131 | return None; |
| 1132 | } |
| 1133 | let body = notices.join("\n"); |
| 1134 | Some(ContentBlock::Text { |
| 1135 | text: format!( |
| 1136 | "<attachment_notice>\n{body}\nDo not describe these images from \ |
| 1137 | memory or from their filenames; ask the user to re-share them.\n\ |
| 1138 | </attachment_notice>" |
| 1139 | ), |
| 1140 | cache_control: None, |
| 1141 | }) |
| 1142 | } |
| 1143 | |
| 1144 | #[cfg(test)] |
| 1145 | #[path = "image_attach/tests.rs"] |
| 1146 | pub(crate) mod tests; |
| 1147 |