Skip to main content

vision_squeezer/
lib.rs

1use std::io::Cursor;
2
3use base64::{Engine, engine::general_purpose::STANDARD as B64};
4use chrono::Utc;
5use image::{DynamicImage, ImageBuffer, Luma, imageops::FilterType};
6use rusqlite::{Connection, params};
7use std::path::PathBuf;
8// ── Config ────────────────────────────────────────────────────────────────────
9
10/// Output encoding format.
11#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
12pub enum OutputFormat {
13    /// JPEG at configured quality (default).
14    #[default]
15    Jpeg,
16    /// WebP at configured quality — typically 30-50% smaller than JPEG at equal quality.
17    WebP,
18    /// AVIF at configured quality — typically 20-50% smaller than WebP at equal quality.
19    Avif,
20}
21
22/// All tuneable knobs for the pipeline.
23#[derive(Clone, Debug)]
24pub struct ProcessConfig {
25    /// Output quality 1–100 (default 75). Applies to both JPEG and WebP.
26    pub quality: u8,
27    /// LLM patch size in pixels. Overridden when `target_model` is set.
28    pub tile_size: u32,
29    /// Remove solid-color padding borders before resizing (default true).
30    pub crop: bool,
31    /// Max channel delta to treat a pixel as background (default 15).
32    pub bg_tolerance: u8,
33    /// Output encoding format (default: JPEG).
34    pub output_format: OutputFormat,
35    /// When set, resizing is model-aware (accounts for pre-scaling behavior).
36    pub target_model: Option<VisionModel>,
37    /// Limit the maximum number of tiles the output image can consume.
38    pub max_tiles: Option<u32>,
39    /// Use saliency (edge-energy) based crop instead of corner-tolerance crop.
40    pub smart_crop: bool,
41}
42
43impl Default for ProcessConfig {
44    fn default() -> Self {
45        Self {
46            quality: 75,
47            tile_size: 512,
48            crop: true,
49            bg_tolerance: 15,
50            output_format: OutputFormat::Jpeg,
51            target_model: None,
52            max_tiles: None,
53            smart_crop: false,
54        }
55    }
56}
57
58impl ProcessConfig {
59    pub fn builder() -> ProcessConfigBuilder {
60        ProcessConfigBuilder(Self::default())
61    }
62}
63
64pub struct ProcessConfigBuilder(ProcessConfig);
65
66impl ProcessConfigBuilder {
67    pub fn quality(mut self, q: u8) -> Self {
68        self.0.quality = q.clamp(1, 100);
69        self
70    }
71    pub fn tile_size(mut self, t: u32) -> Self {
72        self.0.tile_size = t.max(1);
73        self
74    }
75    pub fn crop(mut self, c: bool) -> Self {
76        self.0.crop = c;
77        self
78    }
79    pub fn bg_tolerance(mut self, t: u8) -> Self {
80        self.0.bg_tolerance = t;
81        self
82    }
83    pub fn output_format(mut self, f: OutputFormat) -> Self {
84        self.0.output_format = f;
85        self
86    }
87    pub fn target_model(mut self, m: VisionModel) -> Self {
88        self.0.target_model = Some(m);
89        self
90    }
91    pub fn max_tiles(mut self, m: u32) -> Self {
92        self.0.max_tiles = Some(m);
93        self
94    }
95    pub fn smart_crop(mut self, b: bool) -> Self {
96        self.0.smart_crop = b;
97        self
98    }
99    pub fn build(self) -> ProcessConfig {
100        self.0
101    }
102}
103
104// ── Token Estimation ──────────────────────────────────────────────────────────
105
106/// Supported vision model families with exact or explicitly advisory profiles.
107#[derive(Clone, Copy, Debug)]
108pub enum VisionModel {
109    /// Claude 4.7+ high-resolution vision: 28×28 patches, 2576px edge / 4784-token budget.
110    Claude,
111    /// Earlier Claude vision models: 28×28 patches, 1568px edge / 1568-token budget.
112    ClaudeStandard,
113    /// Current OpenAI vision models (GPT-6 / GPT-5.6): 32×32 patches with a 1.2 multiplier.
114    Gpt6,
115    /// GPT-4o / GPT-4.5 high detail: fits in 2048x2048, scales short side to 768, then 512x512 tiles.
116    Gpt4o,
117    /// Legacy GPT-5/5.1 high detail: 70 base tokens + 140 per 512×512 tile.
118    Gpt5,
119    /// Gemini 2.0/3.0: flat 258 tokens if ≤ 384x384, else 258 per 768x768 tile.
120    Gemini15,
121    /// Meta Llama 3.2/3.3 Vision (Mllama): 560×560 tiles, aspect-ratio canvas capped at 4 tiles.
122    /// (Llama 4 uses a different native-multimodal vision encoder and is not modeled by this arm.)
123    LlamaVision,
124    /// Alibaba Qwen2-VL / Qwen2.5-VL: 28px effective grid (14px ViT patch × 2×2 merge), token clamp.
125    QwenVl,
126    /// DeepSeek-VL2: 384×384 base + dynamic 384 local tiles (open-weights; value is local-context savings).
127    DeepseekVl,
128    /// DeepSeek Flash API: current multimodal endpoint, capped at 384 image tokens per image.
129    DeepseekFlash,
130    /// Kimi K2.5/K2.6/K3 native vision. Moonshot does not publish a fixed billing grid; estimate is advisory.
131    KimiVision,
132    /// Popular open/API vision families without a stable public billing grid.
133    /// Estimates are advisory; resizing remains useful and deterministic.
134    GenericVision,
135}
136
137impl VisionModel {
138    /// Parse the model aliases accepted by the CLI, MCP server, and Python binding.
139    pub fn parse(value: &str) -> Option<Self> {
140        match value.to_ascii_lowercase().as_str() {
141            "claude" | "claude-high" | "claude-4.7" => Some(Self::Claude),
142            "claude-standard" => Some(Self::ClaudeStandard),
143            "openai" | "gpt6" | "gpt-6" | "gpt6-astra" | "gpt-6-astra" | "gpt5.6" | "gpt-5.6"
144            | "gpt5.5" | "gpt-5.5" => Some(Self::Gpt6),
145            "gpt4o" | "gpt-4o" => Some(Self::Gpt4o),
146            "gpt5" | "gpt-5" | "gpt5.1" | "gpt-5.1" => Some(Self::Gpt5),
147            "gemini" | "gemini-3" | "gemini-3.8" => Some(Self::Gemini15),
148            "llama" | "llama-vision" => Some(Self::LlamaVision),
149            "qwen" | "qwen-vl" | "qwen3-vl" => Some(Self::QwenVl),
150            "deepseek-local" | "deepseek-vl" | "deepseek-vl2" => Some(Self::DeepseekVl),
151            "deepseek" | "deepseek-flash" | "deepseek-v4-flash-vision-exp" => {
152                Some(Self::DeepseekFlash)
153            }
154            "kimi" | "kimi-vision" | "kimi-k2.5" | "kimi-k2.6" | "kimi-k3" => {
155                Some(Self::KimiVision)
156            }
157            "glm"
158            | "glm-4v"
159            | "glm-4.5v"
160            | "glm-5.3-flash"
161            | "mistral"
162            | "pixtral"
163            | "pixtral-large"
164            | "pixtral-12b"
165            | "gemma"
166            | "gemma-3"
167            | "gemma-4"
168            | "gemma-4-31b"
169            | "internvl"
170            | "internvl2"
171            | "internvl2.5"
172            | "internvl3"
173            | "minicpm"
174            | "minicpm-v"
175            | "minicpm-o"
176            | "molmo"
177            | "molmo2"
178            | "aya"
179            | "aya-vision"
180            | "phi4"
181            | "phi-4"
182            | "phi-4-multimodal"
183            | "granite"
184            | "granite-vision"
185            | "llava"
186            | "llava-onevision"
187            | "llava-next"
188            | "falcon"
189            | "falcon-vision"
190            | "falcon-ocr"
191            | "minimax"
192            | "minimax-vl"
193            | "minimax-m3"
194            | "step"
195            | "step-3.7"
196            | "step-3.7-flash"
197            | "ling"
198            | "ling-vision"
199            | "ling-3.0-flash-vl"
200            | "voyage"
201            | "voyage-multimodal"
202            | "voyage-multimodal-3.5" => Some(Self::GenericVision),
203            _ => None,
204        }
205    }
206
207    pub fn display_name(self) -> &'static str {
208        match self {
209            Self::Claude => "Claude 4.7+",
210            Self::ClaudeStandard => "Claude (standard)",
211            Self::Gpt6 => "GPT-6 / GPT-5.6",
212            Self::Gpt4o => "GPT-4o",
213            Self::Gpt5 => "GPT-5 / 5.1 (legacy)",
214            Self::Gemini15 => "Gemini 3",
215            Self::LlamaVision => "Llama Vision",
216            Self::QwenVl => "Qwen-VL",
217            Self::DeepseekVl => "DeepSeek-VL",
218            Self::DeepseekFlash => "DeepSeek Flash",
219            Self::KimiVision => "Kimi Vision",
220            Self::GenericVision => "Generic vision (advisory)",
221        }
222    }
223}
224
225#[derive(Debug)]
226pub struct TokenEstimate {
227    pub model: VisionModel,
228    pub tokens: u32,
229    pub tiles: u32,
230}
231
232/// Estimate LLM vision tokens for an image of given dimensions.
233pub fn estimate_tokens(width: u32, height: u32, model: VisionModel) -> TokenEstimate {
234    match model {
235        VisionModel::Claude => {
236            let (w, h) = fit_within_patch_budget(width, height, 2576, 4784, 28);
237            let patches = patch_count(w, h, 28);
238            TokenEstimate {
239                model,
240                tiles: patches,
241                tokens: patches,
242            }
243        }
244        VisionModel::ClaudeStandard => {
245            let (w, h) = fit_within_patch_budget(width, height, 1568, 1568, 28);
246            let patches = patch_count(w, h, 28);
247            TokenEstimate {
248                model,
249                tiles: patches,
250                tokens: patches,
251            }
252        }
253        VisionModel::Gpt6 => {
254            let (w, h) = fit_within_patch_budget(width, height, 2048, 2500, 32);
255            let patches = patch_count(w, h, 32);
256            TokenEstimate {
257                model,
258                tiles: patches,
259                tokens: (patches * 12).div_ceil(10),
260            }
261        }
262        VisionModel::Gpt4o => {
263            // GPT-4o / 4.5: fit within 2048x2048, then short side scaled to 768px, then 512x512 tiles.
264            let (mut w, mut h) = fit_within(width, height, 2048);
265            let short_side = w.min(h);
266            if short_side > 768 {
267                let scale = 768.0 / short_side as f64;
268                w = (w as f64 * scale).round() as u32;
269                h = (h as f64 * scale).round() as u32;
270            }
271            let tiles = tile_count(w, 512) * tile_count(h, 512);
272            TokenEstimate {
273                model,
274                tiles,
275                tokens: 85 + tiles * 170,
276            }
277        }
278        VisionModel::Gpt5 => {
279            let (mut w, mut h) = fit_within(width, height, 2048);
280            let short_side = w.min(h);
281            if short_side > 768 {
282                let scale = 768.0 / short_side as f64;
283                w = (w as f64 * scale).round() as u32;
284                h = (h as f64 * scale).round() as u32;
285            }
286            let tiles = tile_count(w, 512) * tile_count(h, 512);
287            TokenEstimate {
288                model,
289                tiles,
290                tokens: 70 + tiles * 140,
291            }
292        }
293        VisionModel::Gemini15 => {
294            // Gemini 2026: flat 258 if <= 384x384, else 768x768 tiles.
295            if width <= 384 && height <= 384 {
296                TokenEstimate {
297                    model,
298                    tiles: 1,
299                    tokens: 258,
300                }
301            } else {
302                let tiles = tile_count(width, 768) * tile_count(height, 768);
303                TokenEstimate {
304                    model,
305                    tiles,
306                    tokens: tiles * 258,
307                }
308            }
309        }
310        VisionModel::LlamaVision => {
311            // Meta Llama 3.2 / 3.3 Vision (Mllama): aspect-ratio canvas of 560×560 tiles, capped at
312            // max_num_tiles = 4. No separate global-thumbnail tile — the canvas is the full
313            // representation. 14px ViT patch → 40×40 = 1600 patches + 1 CLS = 1601 tokens per tile.
314            // Source: transformers MllamaVisionConfig (image_size 560, patch_size 14, max_num_tiles 4).
315            let (w, h) = fit_within(width, height, 1120); // 2×2 max canvas (4 tiles)
316            let tiles = (tile_count(w, 560) * tile_count(h, 560)).clamp(1, 4);
317            TokenEstimate {
318                model,
319                tiles,
320                tokens: tiles * 1601,
321            }
322        }
323        VisionModel::QwenVl => {
324            // Alibaba Qwen2-VL / Qwen2.5-VL: 28px effective grid (image_patch_size 14 × spatial_merge 2).
325            // smart_resize rounds each side to a multiple of 28 and bounds total tokens to
326            // [IMAGE_MIN_TOKEN_NUM, IMAGE_MAX_TOKEN_NUM] = [4, 16384] (qwen_vl_utils defaults).
327            // tokens = (W/28)·(H/28). DashScope endpoints may cap lower via a per-request max_pixels.
328            let (w, h) = fit_within_pixels(width, height, u32::MAX, 16_384 * 28 * 28);
329            let patches = tile_count(w, 28) * tile_count(h, 28);
330            TokenEstimate {
331                model,
332                tiles: patches,
333                tokens: patches.clamp(4, 16_384),
334            }
335        }
336        VisionModel::DeepseekVl => {
337            // DeepSeek-VL2 (open weights — deepseek public API is text-only, so the win is local
338            // inference context, not API billing). SigLIP-SO400M-patch14-384 emits 27×27 patches per
339            // tile; a 2×2 pixel-shuffle compresses that to 14×14 = 196 tokens (h = 14). Token layout:
340            //   global thumbnail = 14·(14+1) = 210  (one <tile_newline> per row)
341            //   + 1 <view_separator>
342            //   local tiles      = (nh·14)·(nw·14 + 1) over the anyres canvas (m·384, n·384), m·n ≤ 9
343            // Sources: DeepSeek-VL2 paper §2 (arXiv:2412.10302) + processing_deepseek_vl_v2.py.
344            const H: u32 = 14;
345            let (nw, nh) = if width <= 384 && height <= 384 {
346                (1, 1)
347            } else {
348                let mut nw = tile_count(width, 384).max(1);
349                let mut nh = tile_count(height, 384).max(1);
350                while nw * nh > 9 {
351                    if nw >= nh {
352                        nw -= 1;
353                    } else {
354                        nh -= 1;
355                    }
356                }
357                (nw, nh)
358            };
359            let global = H * (H + 1) + 1; // global view + separator
360            let local = (nh * H) * (nw * H + 1);
361            TokenEstimate {
362                model,
363                tiles: nw * nh + 1, // local tiles + global view
364                tokens: global + local,
365            }
366        }
367        VisionModel::DeepseekFlash => TokenEstimate {
368            model,
369            tiles: 1,
370            tokens: 384,
371        },
372        VisionModel::KimiVision => {
373            // Moonshot exposes native-resolution vision but no public image billing grid.
374            // Keep this estimate advisory and use a conservative 28px effective grid.
375            let (w, h) = fit_within(width, height, 4096);
376            let patches = patch_count(w, h, 28);
377            TokenEstimate {
378                model,
379                tiles: patches,
380                tokens: patches,
381            }
382        }
383        VisionModel::GenericVision => {
384            // ponytail: one conservative profile for providers without a stable public grid;
385            // split into provider-specific formulas when billing docs become authoritative.
386            let (w, h) = fit_within(width, height, 2048);
387            let patches = patch_count(w, h, 28);
388            TokenEstimate {
389                model,
390                tiles: patches,
391                tokens: patches,
392            }
393        }
394    }
395}
396
397/// Scale dimensions to fit within `max_side` while preserving aspect ratio.
398pub fn fit_within(width: u32, height: u32, max_side: u32) -> (u32, u32) {
399    if width <= max_side && height <= max_side {
400        return (width, height);
401    }
402    let scale = max_side as f64 / width.max(height) as f64;
403    (
404        (width as f64 * scale) as u32,
405        (height as f64 * scale) as u32,
406    )
407}
408
409/// Scale dimensions to fit within both a max-side limit and a total-pixel limit.
410pub fn fit_within_pixels(width: u32, height: u32, max_side: u32, max_pixels: u64) -> (u32, u32) {
411    let (mut w, mut h) = fit_within(width, height, max_side);
412    let total = w as u64 * h as u64;
413    if total > max_pixels {
414        let scale = (max_pixels as f64 / total as f64).sqrt();
415        w = (w as f64 * scale) as u32;
416        h = (h as f64 * scale) as u32;
417    }
418    (w.max(1), h.max(1))
419}
420
421fn patch_count(width: u32, height: u32, patch: u32) -> u32 {
422    width.max(1).div_ceil(patch) * height.max(1).div_ceil(patch)
423}
424
425/// Apply a model's edge and patch limits while preserving the image aspect ratio.
426fn fit_within_patch_budget(
427    width: u32,
428    height: u32,
429    max_side: u32,
430    max_patches: u32,
431    patch: u32,
432) -> (u32, u32) {
433    let (mut w, mut h) = fit_within(width.max(1), height.max(1), max_side);
434    let patches = patch_count(w, h, patch);
435    if patches > max_patches {
436        let scale = (max_patches as f64 / patches as f64).sqrt();
437        w = ((w as f64 * scale) as u32 / patch * patch).max(patch);
438        h = ((h as f64 * scale) as u32 / patch * patch).max(patch);
439    }
440    (w, h)
441}
442
443/// Compute the optimal dimensions to *send* to a given model to minimize tiles.
444///
445/// For models that pre-scale images, we simulate their scaling,
446/// snap the scaled result to tile boundaries, then invert back to input space.
447pub fn optimal_send_dimensions(width: u32, height: u32, model: VisionModel) -> (u32, u32) {
448    match model {
449        VisionModel::Claude => optimal_for_patch_model(width, height, 2576, 4784, 28),
450        VisionModel::ClaudeStandard => optimal_for_patch_model(width, height, 1568, 1568, 28),
451        VisionModel::Gpt6 => optimal_for_patch_model(width, height, 2048, 2500, 32),
452        VisionModel::Gpt4o => optimal_for_prescaling_model(width, height, 2048, 512),
453        VisionModel::Gpt5 => optimal_for_prescaling_model(width, height, 2048, 512),
454        VisionModel::Gemini15 => {
455            // Gemini uses 768x768 tiles if > 384x384
456            if width <= 384 && height <= 384 {
457                (width, height)
458            } else {
459                optimal_for_prescaling_model(width, height, 4096, 768)
460            }
461        }
462        VisionModel::LlamaVision => {
463            // Snap to the 560px tile grid within the 2×2 (1120px) max canvas to avoid spill-over tiles.
464            optimal_for_prescaling_model(width, height, 1120, 560)
465        }
466        VisionModel::QwenVl => {
467            // Snap each side to the 28px patch grid, after fitting under the max-pixel budget.
468            let (fw, fh) = fit_within_pixels(width, height, u32::MAX, 16_384 * 28 * 28);
469            (
470                snap_to_tile_boundary(fw, 28).max(28),
471                snap_to_tile_boundary(fh, 28).max(28),
472            )
473        }
474        VisionModel::DeepseekVl => {
475            // ≤384 stays as a single tile; otherwise snap to the 384px tile grid.
476            if width <= 384 && height <= 384 {
477                (width, height)
478            } else {
479                optimal_for_prescaling_model(width, height, 1152, 384)
480            }
481        }
482        VisionModel::DeepseekFlash => fit_within(width, height, 2048),
483        VisionModel::KimiVision => {
484            let (w, h) = fit_within(width, height, 4096);
485            (snap_to_tile_boundary(w, 28), snap_to_tile_boundary(h, 28))
486        }
487        VisionModel::GenericVision => {
488            let (w, h) = fit_within(width, height, 2048);
489            (snap_to_tile_boundary(w, 28), snap_to_tile_boundary(h, 28))
490        }
491    }
492}
493
494fn optimal_for_patch_model(
495    width: u32,
496    height: u32,
497    max_side: u32,
498    max_patches: u32,
499    patch: u32,
500) -> (u32, u32) {
501    let (w, h) = fit_within_patch_budget(width, height, max_side, max_patches, patch);
502    (
503        snap_to_tile_boundary(w, patch),
504        snap_to_tile_boundary(h, patch),
505    )
506}
507
508/// For models that pre-scale (GPT-4o, Gemini), find the smallest input dimensions
509/// that, after the model's internal fit-within + tiling, produce the fewest tiles.
510///
511/// Strategy: enumerate candidate tile-grid dimensions (tw*tile, th*tile) that fit
512/// within max_side, compute the input size that would map to each, and pick the
513/// candidate that uses the fewest tiles while preserving the original aspect ratio
514/// as closely as possible.
515fn optimal_for_prescaling_model(width: u32, height: u32, max_side: u32, tile: u32) -> (u32, u32) {
516    let (fw, fh) = fit_within(width, height, max_side);
517
518    // Simply snap the fitted dimensions to the nearest tile boundary
519    let target_w = snap_to_tile_boundary(fw, tile).max(tile);
520    let target_h = snap_to_tile_boundary(fh, tile).max(tile);
521
522    // If image was larger than max_side, scale back to input space
523    if width > max_side || height > max_side {
524        let scale = width.max(height) as f64 / max_side as f64;
525        let opt_w = (target_w as f64 * scale).round() as u32;
526        let opt_h = (target_h as f64 * scale).round() as u32;
527        return (opt_w.max(1), opt_h.max(1));
528    }
529
530    (target_w, target_h)
531}
532
533/// Full token savings report for a before/after dimension pair across all models.
534pub struct TokenSavingsTable {
535    pub claude_before: TokenEstimate,
536    pub claude_after: TokenEstimate,
537    pub gpt6_before: TokenEstimate,
538    pub gpt6_after: TokenEstimate,
539    pub gpt4o_before: TokenEstimate,
540    pub gpt4o_after: TokenEstimate,
541    pub gpt5_before: TokenEstimate,
542    pub gpt5_after: TokenEstimate,
543    pub gemini_before: TokenEstimate,
544    pub gemini_after: TokenEstimate,
545}
546
547pub fn token_savings_table(orig_w: u32, orig_h: u32, opt_w: u32, opt_h: u32) -> TokenSavingsTable {
548    TokenSavingsTable {
549        claude_before: estimate_tokens(orig_w, orig_h, VisionModel::Claude),
550        claude_after: estimate_tokens(opt_w, opt_h, VisionModel::Claude),
551        gpt6_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt6),
552        gpt6_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt6),
553        gpt4o_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt4o),
554        gpt4o_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt4o),
555        gpt5_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt5),
556        gpt5_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt5),
557        gemini_before: estimate_tokens(orig_w, orig_h, VisionModel::Gemini15),
558        gemini_after: estimate_tokens(opt_w, opt_h, VisionModel::Gemini15),
559    }
560}
561
562impl TokenSavingsTable {
563    pub fn print(&self) {
564        println!(
565            "{:<12} {:>8} {:>8} {:>10}",
566            "Model", "Before", "After", "Saved"
567        );
568        println!("{}", "-".repeat(42));
569        self.print_row("Claude 4.7+", &self.claude_before, &self.claude_after);
570        self.print_row("GPT-6", &self.gpt6_before, &self.gpt6_after);
571        self.print_row("GPT-4o", &self.gpt4o_before, &self.gpt4o_after);
572        self.print_row("GPT-5", &self.gpt5_before, &self.gpt5_after);
573        self.print_row("Gemini", &self.gemini_before, &self.gemini_after);
574    }
575
576    fn print_row(&self, name: &str, before: &TokenEstimate, after: &TokenEstimate) {
577        let saved = before.tokens.saturating_sub(after.tokens);
578        let pct = if before.tokens > 0 {
579            saved as f64 / before.tokens as f64 * 100.0
580        } else {
581            0.0
582        };
583        println!(
584            "{:<12} {:>8} {:>8} {:>8} ({:.1}%)",
585            name, before.tokens, after.tokens, saved, pct
586        );
587    }
588}
589
590// ── Types ─────────────────────────────────────────────────────────────────────
591
592pub struct DimensionResult {
593    pub width: u32,
594    pub height: u32,
595    pub tiles_before: u32,
596    pub tiles_after: u32,
597}
598
599impl DimensionResult {
600    pub fn tokens_saved(&self) -> u32 {
601        self.tiles_before.saturating_sub(self.tiles_after)
602    }
603}
604
605#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
606pub enum ProcessMode {
607    /// General LLM vision — JPEG output at configured quality.
608    Standard,
609    /// Text extraction — high-contrast grayscale binarization (Otsu threshold).
610    Ocr,
611    /// Auto-detects if the image is mostly text (monochrome/grayscale).
612    #[default]
613    Auto,
614}
615
616pub fn detect_ocr_mode(img: &DynamicImage) -> bool {
617    let rgb = img.to_rgb8();
618    let mut colorful_count = 0;
619    let mut total_count = 0;
620    // Sample every 4th pixel for speed
621    for (x, y, p) in rgb.enumerate_pixels() {
622        if x % 4 == 0 && y % 4 == 0 {
623            total_count += 1;
624            let min = p[0].min(p[1]).min(p[2]);
625            let max = p[0].max(p[1]).max(p[2]);
626            if max.saturating_sub(min) > 25 {
627                colorful_count += 1;
628            }
629        }
630    }
631    let colorful_ratio = colorful_count as f64 / total_count.max(1) as f64;
632    colorful_ratio < 0.1 // if less than 10% of pixels are colorful, assume OCR
633}
634
635pub struct SavingsReport {
636    pub tiles_before: u32,
637    pub tiles_after: u32,
638    pub tiles_saved: u32,
639    pub bytes_before: Option<u64>,
640    pub bytes_after: Option<u64>,
641}
642
643impl SavingsReport {
644    pub fn size_reduction_pct(&self) -> Option<f64> {
645        match (self.bytes_before, self.bytes_after) {
646            (Some(b), Some(a)) if b > 0 => Some((1.0 - a as f64 / b as f64) * 100.0),
647            _ => None,
648        }
649    }
650
651    pub fn token_reduction_pct(&self) -> f64 {
652        if self.tiles_before == 0 {
653            return 0.0;
654        }
655        self.tiles_saved as f64 / self.tiles_before as f64 * 100.0
656    }
657}
658
659pub struct ProcessResult {
660    pub image: DynamicImage,
661    pub width: u32,
662    pub height: u32,
663    pub report: SavingsReport,
664}
665
666impl ProcessResult {
667    pub fn tokens_saved(&self) -> u32 {
668        self.report.tiles_saved
669    }
670}
671
672// ── Pipeline ──────────────────────────────────────────────────────────────────
673
674/// Full pipeline: [crop] → tile-snap resize → [OCR binarize].
675/// Pass `input_bytes = 0` if unknown (omits file-size from report).
676pub fn process(
677    img: DynamicImage,
678    mode: ProcessMode,
679    input_bytes: u64,
680    cfg: &ProcessConfig,
681) -> ProcessResult {
682    let (orig_w, orig_h) = (img.width(), img.height());
683    let tiles_before = match cfg.target_model {
684        Some(model) => estimate_tokens(orig_w, orig_h, model).tiles,
685        None => tile_count(orig_w, cfg.tile_size) * tile_count(orig_h, cfg.tile_size),
686    };
687
688    let after_crop = if cfg.crop {
689        if cfg.smart_crop {
690            saliency_crop(&img, 16)
691        } else {
692            crop_padding(img, cfg.bg_tolerance)
693        }
694    } else {
695        img
696    };
697    let (mut opt_w, mut opt_h) = match cfg.target_model {
698        Some(model) => optimal_send_dimensions(after_crop.width(), after_crop.height(), model),
699        None => {
700            let d = calculate_optimal_dimensions_with(
701                after_crop.width(),
702                after_crop.height(),
703                cfg.tile_size,
704            );
705            (d.width, d.height)
706        }
707    };
708
709    if let Some(max_t) = cfg.max_tiles {
710        let (nw, nh) = enforce_max_tiles(opt_w, opt_h, max_t, cfg.tile_size, cfg.target_model);
711        opt_w = nw;
712        opt_h = nh;
713    }
714
715    let tiles_after = match cfg.target_model {
716        Some(model) => {
717            let est = estimate_tokens(opt_w, opt_h, model);
718            est.tiles
719        }
720        None => tile_count(opt_w, cfg.tile_size) * tile_count(opt_h, cfg.tile_size),
721    };
722    let resized = after_crop.resize_exact(opt_w, opt_h, FilterType::Lanczos3);
723
724    let actual_mode = match mode {
725        ProcessMode::Auto => {
726            if detect_ocr_mode(&after_crop) {
727                ProcessMode::Ocr
728            } else {
729                ProcessMode::Standard
730            }
731        }
732        m => m,
733    };
734
735    let final_image = match actual_mode {
736        ProcessMode::Standard | ProcessMode::Auto => resized,
737        ProcessMode::Ocr => binarize(resized),
738    };
739
740    ProcessResult {
741        width: final_image.width(),
742        height: final_image.height(),
743        image: final_image,
744        report: SavingsReport {
745            tiles_before,
746            tiles_after,
747            tiles_saved: tiles_before.saturating_sub(tiles_after),
748            bytes_before: if input_bytes > 0 {
749                Some(input_bytes)
750            } else {
751                None
752            },
753            bytes_after: None,
754        },
755    }
756}
757
758fn enforce_max_tiles(
759    mut width: u32,
760    mut height: u32,
761    max_tiles: u32,
762    default_tile_size: u32,
763    model: Option<VisionModel>,
764) -> (u32, u32) {
765    if max_tiles == 0 {
766        return (width, height);
767    }
768
769    let mut scale = 1.0;
770    let orig_w = width;
771    let orig_h = height;
772
773    loop {
774        let (snapped_w, snapped_h) = match model {
775            Some(m) => optimal_send_dimensions(width, height, m),
776            None => {
777                let d = calculate_optimal_dimensions_with(width, height, default_tile_size);
778                (d.width, d.height)
779            }
780        };
781
782        let tiles = match model {
783            Some(m) => estimate_tokens(snapped_w, snapped_h, m).tiles,
784            None => {
785                tile_count(snapped_w, default_tile_size) * tile_count(snapped_h, default_tile_size)
786            }
787        };
788
789        if tiles <= max_tiles || scale < 0.1 {
790            return (snapped_w, snapped_h);
791        }
792
793        scale *= 0.95;
794        width = (orig_w as f64 * scale) as u32;
795        height = (orig_h as f64 * scale) as u32;
796        width = width.max(1);
797        height = height.max(1);
798    }
799}
800
801// ── Step 1: Tile-Aware Dimension Calculation ───────────────────────────────────
802
803/// Snap W×H to tile boundaries using default tile size (512).
804pub fn calculate_optimal_dimensions(width: u32, height: u32) -> DimensionResult {
805    calculate_optimal_dimensions_with(width, height, 512)
806}
807
808/// Snap W×H to tile boundaries using a custom tile size.
809pub fn calculate_optimal_dimensions_with(
810    width: u32,
811    height: u32,
812    tile_size: u32,
813) -> DimensionResult {
814    let opt_w = snap_to_tile_boundary(width, tile_size);
815    let opt_h = snap_to_tile_boundary(height, tile_size);
816
817    DimensionResult {
818        width: opt_w,
819        height: opt_h,
820        tiles_before: tile_count(width, tile_size) * tile_count(height, tile_size),
821        tiles_after: tile_count(opt_w, tile_size) * tile_count(opt_h, tile_size),
822    }
823}
824
825fn tile_count(dim: u32, tile_size: u32) -> u32 {
826    dim.div_ceil(tile_size)
827}
828
829fn snap_to_tile_boundary(dim: u32, tile_size: u32) -> u32 {
830    if dim.is_multiple_of(tile_size) {
831        return dim;
832    }
833    ((dim / tile_size) * tile_size).max(tile_size)
834}
835
836// ── Step 2: Semantic Crop (padding removal) ────────────────────────────────────
837
838/// Remove solid-color borders using corner sampling + configurable tolerance.
839pub fn crop_padding(img: DynamicImage, bg_tolerance: u8) -> DynamicImage {
840    let rgba = img.to_rgba8();
841    let (w, h) = rgba.dimensions();
842
843    let corners = [
844        *rgba.get_pixel(0, 0),
845        *rgba.get_pixel(w - 1, 0),
846        *rgba.get_pixel(0, h - 1),
847        *rgba.get_pixel(w - 1, h - 1),
848    ];
849    let bg = corners[0]; // first corner as background reference
850
851    let top = first_non_bg_row(&rgba, bg, bg_tolerance, true);
852    let bottom = first_non_bg_row(&rgba, bg, bg_tolerance, false);
853    let left = first_non_bg_col(&rgba, bg, bg_tolerance, true);
854    let right = first_non_bg_col(&rgba, bg, bg_tolerance, false);
855
856    if top >= bottom || left >= right {
857        return DynamicImage::ImageRgba8(rgba);
858    }
859
860    DynamicImage::ImageRgba8(
861        image::imageops::crop_imm(&rgba, left, top, right - left, bottom - top).to_image(),
862    )
863}
864
865fn is_bg(pixel: image::Rgba<u8>, bg: image::Rgba<u8>, tolerance: u8) -> bool {
866    pixel.0[3] < 10
867        || pixel.0[..3]
868            .iter()
869            .zip(bg.0[..3].iter())
870            .all(|(&a, &b)| a.abs_diff(b) <= tolerance)
871}
872
873fn first_non_bg_row(img: &image::RgbaImage, bg: image::Rgba<u8>, tol: u8, from_top: bool) -> u32 {
874    let (w, h) = img.dimensions();
875    let rows: Box<dyn Iterator<Item = u32>> = if from_top {
876        Box::new(0..h)
877    } else {
878        Box::new((0..h).rev())
879    };
880    for y in rows {
881        if (0..w).any(|x| !is_bg(*img.get_pixel(x, y), bg, tol)) {
882            return y;
883        }
884    }
885    0
886}
887
888fn first_non_bg_col(img: &image::RgbaImage, bg: image::Rgba<u8>, tol: u8, from_left: bool) -> u32 {
889    let (w, h) = img.dimensions();
890    let cols: Box<dyn Iterator<Item = u32>> = if from_left {
891        Box::new(0..w)
892    } else {
893        Box::new((0..w).rev())
894    };
895    for x in cols {
896        if (0..h).any(|y| !is_bg(*img.get_pixel(x, y), bg, tol)) {
897            return x;
898        }
899    }
900    0
901}
902
903// ── Saliency Crop (edge-energy based) ─────────────────────────────────────────
904
905/// Crop to the bounding box of high-energy (edge) pixels.
906/// Uses a Sobel-lite gradient magnitude per luma pixel. Pixels above 2× the
907/// mean gradient energy define the salient region; the bbox is expanded by
908/// `margin` pixels on every side.
909///
910/// Falls back to the input unchanged for uniform images (no salient region).
911pub fn saliency_crop(img: &DynamicImage, margin: u32) -> DynamicImage {
912    let gray = img.to_luma8();
913    let (w, h) = gray.dimensions();
914    if w < 3 || h < 3 {
915        return img.clone();
916    }
917
918    let mut energy = vec![0u32; (w * h) as usize];
919    let mut total: u64 = 0;
920    for y in 1..h - 1 {
921        for x in 1..w - 1 {
922            let l = gray.get_pixel(x - 1, y).0[0] as i32;
923            let r = gray.get_pixel(x + 1, y).0[0] as i32;
924            let t = gray.get_pixel(x, y - 1).0[0] as i32;
925            let b = gray.get_pixel(x, y + 1).0[0] as i32;
926            let e = ((r - l).abs() + (b - t).abs()) as u32;
927            energy[(y * w + x) as usize] = e;
928            total += e as u64;
929        }
930    }
931    let count = (w as u64) * (h as u64);
932    let mean = (total / count.max(1)) as u32;
933    let threshold = mean.saturating_mul(2).max(8);
934
935    let (mut min_x, mut min_y, mut max_x, mut max_y) = (w, h, 0u32, 0u32);
936    for y in 0..h {
937        for x in 0..w {
938            if energy[(y * w + x) as usize] > threshold {
939                if x < min_x {
940                    min_x = x;
941                }
942                if y < min_y {
943                    min_y = y;
944                }
945                if x > max_x {
946                    max_x = x;
947                }
948                if y > max_y {
949                    max_y = y;
950                }
951            }
952        }
953    }
954
955    if min_x >= max_x || min_y >= max_y {
956        return img.clone();
957    }
958
959    let x0 = min_x.saturating_sub(margin);
960    let y0 = min_y.saturating_sub(margin);
961    let x1 = (max_x + 1 + margin).min(w);
962    let y1 = (max_y + 1 + margin).min(h);
963    img.crop_imm(x0, y0, x1 - x0, y1 - y0)
964}
965
966// ── SSIM + Auto-Quality ───────────────────────────────────────────────────────
967
968/// Compute structural-similarity (single-window, luma) between two images.
969///
970/// Returns a value in [-1, 1]; 1.0 means identical. Both images are converted
971/// to grayscale; if dimensions differ, the smaller is rescaled to match the
972/// larger via Lanczos3.
973pub fn ssim(a: &DynamicImage, b: &DynamicImage) -> f64 {
974    let (aw, ah) = (a.width(), a.height());
975    let (bw, bh) = (b.width(), b.height());
976    let (target_w, target_h) = (aw.max(bw), ah.max(bh));
977
978    let resize_if_needed = |img: &DynamicImage| -> image::GrayImage {
979        if img.width() == target_w && img.height() == target_h {
980            img.to_luma8()
981        } else {
982            img.resize_exact(target_w, target_h, FilterType::Lanczos3)
983                .to_luma8()
984        }
985    };
986
987    let a_luma = resize_if_needed(a);
988    let b_luma = resize_if_needed(b);
989
990    let n = (target_w as u64 * target_h as u64).max(1) as f64;
991    let (mut sum_a, mut sum_b) = (0f64, 0f64);
992    for (pa, pb) in a_luma.pixels().zip(b_luma.pixels()) {
993        sum_a += pa.0[0] as f64;
994        sum_b += pb.0[0] as f64;
995    }
996    let mean_a = sum_a / n;
997    let mean_b = sum_b / n;
998
999    let (mut var_a, mut var_b, mut cov) = (0f64, 0f64, 0f64);
1000    for (pa, pb) in a_luma.pixels().zip(b_luma.pixels()) {
1001        let da = pa.0[0] as f64 - mean_a;
1002        let db = pb.0[0] as f64 - mean_b;
1003        var_a += da * da;
1004        var_b += db * db;
1005        cov += da * db;
1006    }
1007    var_a /= n;
1008    var_b /= n;
1009    cov /= n;
1010
1011    let c1 = (0.01f64 * 255.0).powi(2);
1012    let c2 = (0.03f64 * 255.0).powi(2);
1013    let num = (2.0 * mean_a * mean_b + c1) * (2.0 * cov + c2);
1014    let den = (mean_a.powi(2) + mean_b.powi(2) + c1) * (var_a + var_b + c2);
1015    if den.abs() < f64::EPSILON {
1016        1.0
1017    } else {
1018        num / den
1019    }
1020}
1021
1022/// Binary-search the lowest quality in [`min_q`, `max_q`] that meets a given
1023/// SSIM target against the original. Returns the encoded bytes and the quality used.
1024///
1025/// Used when callers want "automatic" quality: pick the smallest file that still
1026/// passes a perceptual threshold (typically 0.95).
1027pub fn encode_with_auto_quality(
1028    original: &DynamicImage,
1029    cfg: &ProcessConfig,
1030    target_ssim: f64,
1031    min_q: u8,
1032    max_q: u8,
1033) -> Result<(Vec<u8>, u8), String> {
1034    let mut lo = min_q.max(1);
1035    let mut hi = max_q.min(100).max(lo + 1);
1036    let mut best: Option<(Vec<u8>, u8)> = None;
1037
1038    while hi.saturating_sub(lo) > 2 {
1039        let mid = lo + (hi - lo) / 2;
1040        let trial = ProcessConfig {
1041            quality: mid,
1042            ..cfg.clone()
1043        };
1044        let bytes = encode_to_bytes(original, &trial)?;
1045        let decoded = image::load_from_memory(&bytes).map_err(|e| e.to_string())?;
1046        let score = ssim(original, &decoded);
1047        if score >= target_ssim {
1048            best = Some((bytes, mid));
1049            hi = mid;
1050        } else {
1051            lo = mid;
1052        }
1053    }
1054
1055    // If no quality met the target during the search, encode at max_q as fallback.
1056    if let Some((b, q)) = best {
1057        Ok((b, q))
1058    } else {
1059        let trial = ProcessConfig {
1060            quality: hi,
1061            ..cfg.clone()
1062        };
1063        let bytes = encode_to_bytes(original, &trial)?;
1064        Ok((bytes, hi))
1065    }
1066}
1067
1068// ── Step 3: OCR Binarization ───────────────────────────────────────────────────
1069
1070pub fn binarize(img: DynamicImage) -> DynamicImage {
1071    let gray = img.to_luma8();
1072    let (w, h) = gray.dimensions();
1073    let threshold = otsu_threshold(&gray);
1074    let binary: ImageBuffer<Luma<u8>, Vec<u8>> = ImageBuffer::from_fn(w, h, |x, y| {
1075        let p = gray.get_pixel(x, y).0[0];
1076        Luma([if p < threshold { 0u8 } else { 255u8 }])
1077    });
1078    DynamicImage::ImageLuma8(binary)
1079}
1080
1081fn otsu_threshold(img: &image::GrayImage) -> u8 {
1082    let mut histogram = [0u32; 256];
1083    for p in img.pixels() {
1084        histogram[p.0[0] as usize] += 1;
1085    }
1086    let total = img.width() * img.height();
1087    let (mut sum, mut sum_bg, mut weight_bg) = (0f64, 0f64, 0f64);
1088    for (i, &h) in histogram.iter().enumerate() {
1089        sum += i as f64 * h as f64;
1090    }
1091    let (mut best_thresh, mut best_var) = (0u8, 0f64);
1092    for (t, &h) in histogram.iter().enumerate() {
1093        weight_bg += h as f64;
1094        if weight_bg == 0.0 {
1095            continue;
1096        }
1097        let weight_fg = total as f64 - weight_bg;
1098        if weight_fg == 0.0 {
1099            break;
1100        }
1101        sum_bg += t as f64 * h as f64;
1102        let mean_bg = sum_bg / weight_bg;
1103        let mean_fg = (sum - sum_bg) / weight_fg;
1104        let var = weight_bg * weight_fg * (mean_bg - mean_fg).powi(2);
1105        if var > best_var {
1106            best_var = var;
1107            best_thresh = t as u8;
1108        }
1109    }
1110    best_thresh
1111}
1112
1113// ── Base64 I/O ────────────────────────────────────────────────────────────────
1114
1115/// Reject decode work that would blow up memory: an attacker-supplied base64
1116/// string can be tiny yet declare enormous dimensions (decompression bomb).
1117const MAX_B64_LEN: usize = 64 * 1024 * 1024; // ~48 MB of raw image bytes
1118const MAX_PIXELS: u64 = 100_000_000; // 100 MP
1119const MAX_DIM: u32 = 16_384;
1120
1121pub fn decode_base64_image(input: &str) -> Result<DynamicImage, String> {
1122    let data = if let Some(c) = input.find(',') {
1123        &input[c + 1..]
1124    } else {
1125        input
1126    };
1127    let data = data.trim();
1128    if data.len() > MAX_B64_LEN {
1129        return Err(format!(
1130            "image base64 exceeds {} MB limit",
1131            MAX_B64_LEN / 1_048_576
1132        ));
1133    }
1134    let bytes = B64.decode(data).map_err(|e| e.to_string())?;
1135
1136    let mut limits = image::Limits::default();
1137    limits.max_image_width = Some(MAX_DIM);
1138    limits.max_image_height = Some(MAX_DIM);
1139    limits.max_alloc = Some(MAX_PIXELS * 4); // RGBA worst case
1140
1141    let mut reader = image::ImageReader::new(std::io::Cursor::new(bytes))
1142        .with_guessed_format()
1143        .map_err(|e| e.to_string())?;
1144    reader.limits(limits);
1145    reader.decode().map_err(|e| e.to_string())
1146}
1147
1148pub fn encode_image_base64(img: &DynamicImage, cfg: &ProcessConfig) -> Result<String, String> {
1149    let bytes = encode_to_bytes(img, cfg)?;
1150    Ok(B64.encode(bytes))
1151}
1152
1153/// Encode image to raw bytes using the configured output format.
1154pub fn encode_to_bytes(img: &DynamicImage, cfg: &ProcessConfig) -> Result<Vec<u8>, String> {
1155    match cfg.output_format {
1156        OutputFormat::Jpeg => {
1157            use image::codecs::jpeg::JpegEncoder;
1158            let mut buf = Cursor::new(Vec::new());
1159            let rgb = img.to_rgb8();
1160            JpegEncoder::new_with_quality(&mut buf, cfg.quality)
1161                .encode_image(&DynamicImage::ImageRgb8(rgb))
1162                .map_err(|e| e.to_string())?;
1163            Ok(buf.into_inner())
1164        }
1165        OutputFormat::WebP => {
1166            let rgb = img.to_rgb8();
1167            let enc = webp::Encoder::from_rgb(rgb.as_raw(), rgb.width(), rgb.height());
1168            let mem = enc.encode(cfg.quality as f32);
1169            Ok(mem.to_vec())
1170        }
1171        OutputFormat::Avif => {
1172            use image::ImageEncoder;
1173            use image::codecs::avif::AvifEncoder;
1174            let mut buf = Cursor::new(Vec::new());
1175            let rgba = img.to_rgba8();
1176            // Speed 6 is a reasonable balance; lower = better compression but slower.
1177            AvifEncoder::new_with_speed_quality(&mut buf, 6, cfg.quality)
1178                .write_image(
1179                    rgba.as_raw(),
1180                    rgba.width(),
1181                    rgba.height(),
1182                    image::ExtendedColorType::Rgba8,
1183                )
1184                .map_err(|e| e.to_string())?;
1185            Ok(buf.into_inner())
1186        }
1187    }
1188}
1189
1190// ── MCP Tool: optimize_image ──────────────────────────────────────────────────
1191
1192pub struct OptimizeResult {
1193    pub optimized_base64: String,
1194    pub report: SavingsReport,
1195    pub original_width: u32,
1196    pub original_height: u32,
1197    pub width: u32,
1198    pub height: u32,
1199    pub optimized_bytes: usize,
1200}
1201
1202/// MCP entry point: base64 in → base64 JPEG out + savings report.
1203pub fn optimize_image(
1204    input_base64: &str,
1205    mode: ProcessMode,
1206    cfg: &ProcessConfig,
1207) -> Result<OptimizeResult, String> {
1208    let img = decode_base64_image(input_base64)?;
1209    let (orig_w, orig_h) = (img.width(), img.height());
1210    let input_bytes = {
1211        let data = if let Some(c) = input_base64.find(',') {
1212            &input_base64[c + 1..]
1213        } else {
1214            input_base64
1215        };
1216        B64.decode(data.trim()).map_err(|e| e.to_string())?.len() as u64
1217    };
1218
1219    let mut result = process(img, mode, input_bytes, cfg);
1220    let bytes = encode_to_bytes(&result.image, cfg)?;
1221    let encoded = B64.encode(&bytes);
1222    result.report.bytes_after = Some(bytes.len() as u64);
1223
1224    Ok(OptimizeResult {
1225        optimized_base64: encoded,
1226        report: result.report,
1227        original_width: orig_w,
1228        original_height: orig_h,
1229        width: result.width,
1230        height: result.height,
1231        optimized_bytes: bytes.len(),
1232    })
1233}
1234
1235// ── Tests ─────────────────────────────────────────────────────────────────────
1236
1237// ── Step 4: Sandbox (Think in Code) ──────────────────────────────────────────
1238
1239/// Atomic image operations for the Sandbox mode.
1240#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)]
1241#[serde(rename_all = "lowercase", tag = "op")]
1242pub enum ImageOp {
1243    /// Crop a specific region: { x, y, width, height }
1244    Crop {
1245        x: u32,
1246        y: u32,
1247        width: u32,
1248        height: u32,
1249    },
1250    /// Convert to grayscale.
1251    Grayscale,
1252    /// Binarize using Otsu's threshold (if threshold is None).
1253    Binarize { threshold: Option<u8> },
1254    /// Resize to exact dimensions.
1255    Resize { width: u32, height: u32 },
1256    /// Adjust contrast (e.g., 2.0 for double contrast).
1257    Contrast { amount: f32 },
1258    /// Adjust brightness (e.g., -20 to darken).
1259    Brightness { amount: f32 },
1260}
1261
1262/// Execute a sequence of operations on an image.
1263pub fn process_with_operations(mut img: DynamicImage, ops: Vec<ImageOp>) -> DynamicImage {
1264    for op in ops {
1265        img = match op {
1266            ImageOp::Crop {
1267                x,
1268                y,
1269                width,
1270                height,
1271            } => img.crop_imm(x, y, width, height),
1272            ImageOp::Grayscale => DynamicImage::ImageLuma8(img.to_luma8()),
1273            ImageOp::Binarize { threshold } => {
1274                let gray = img.to_luma8();
1275                let thr = threshold.unwrap_or(128);
1276                let mut binarized = ImageBuffer::new(gray.width(), gray.height());
1277                for (x, y, p) in gray.enumerate_pixels() {
1278                    let val = if p[0] > thr { 255 } else { 0 };
1279                    binarized.put_pixel(x, y, Luma([val]));
1280                }
1281                DynamicImage::ImageLuma8(binarized)
1282            }
1283            ImageOp::Resize { width, height } => {
1284                img.resize_exact(width, height, FilterType::Lanczos3)
1285            }
1286            ImageOp::Contrast { amount } => img.adjust_contrast(amount),
1287            ImageOp::Brightness { amount } => img.brighten(amount as i32),
1288        };
1289    }
1290    img
1291}
1292
1293#[cfg(test)]
1294mod tests {
1295    use super::*;
1296
1297    fn cfg() -> ProcessConfig {
1298        ProcessConfig::default()
1299    }
1300
1301    #[test]
1302    fn decode_round_trips_small_image() {
1303        let img = DynamicImage::ImageRgb8(ImageBuffer::from_fn(8, 8, |_, _| {
1304            image::Rgb([10u8, 20, 30])
1305        }));
1306        let b64 = encode_image_base64(&img, &cfg()).unwrap();
1307        let decoded = decode_base64_image(&b64).unwrap();
1308        assert_eq!((decoded.width(), decoded.height()), (8, 8));
1309    }
1310
1311    #[test]
1312    fn decode_rejects_oversized_dimensions() {
1313        // Valid, tiny-on-disk image whose width exceeds MAX_DIM — the decompression-bomb shape.
1314        let wide = DynamicImage::ImageRgb8(ImageBuffer::from_fn(MAX_DIM + 1, 1, |_, _| {
1315            image::Rgb([0u8, 0, 0])
1316        }));
1317        let b64 = encode_image_base64(&wide, &cfg()).unwrap();
1318        assert!(decode_base64_image(&b64).is_err());
1319    }
1320
1321    #[test]
1322    fn decode_rejects_garbage() {
1323        assert!(decode_base64_image("not valid base64 !!!").is_err());
1324    }
1325
1326    #[test]
1327    fn exact_boundary_unchanged() {
1328        let r = calculate_optimal_dimensions(1024, 512);
1329        assert_eq!((r.width, r.height), (1024, 512));
1330        assert_eq!(r.tokens_saved(), 0);
1331    }
1332
1333    #[test]
1334    fn one_pixel_over_saves_full_tile_row() {
1335        let r = calculate_optimal_dimensions(1025, 1025);
1336        assert_eq!((r.width, r.height), (1024, 1024));
1337        assert_eq!(r.tiles_before, 9);
1338        assert_eq!(r.tiles_after, 4);
1339        assert_eq!(r.tokens_saved(), 5);
1340    }
1341
1342    #[test]
1343    fn small_image_never_below_one_tile() {
1344        let r = calculate_optimal_dimensions(100, 200);
1345        assert_eq!((r.width, r.height), (512, 512));
1346    }
1347
1348    #[test]
1349    fn mid_boundary_snaps_down() {
1350        let r = calculate_optimal_dimensions(768, 512);
1351        assert_eq!(r.width, 512);
1352        assert_eq!(r.tiles_after, 1);
1353    }
1354
1355    #[test]
1356    fn custom_tile_size_256() {
1357        let r = calculate_optimal_dimensions_with(257, 512, 256);
1358        assert_eq!(r.width, 256); // 257 → snaps down to 256
1359        assert_eq!(r.tiles_before, 2 * 2); // ceil(257/256)*ceil(512/256) = 2*2
1360        assert_eq!(r.tiles_after, 1 * 2); // 256/256 * 512/256 = 1*2
1361    }
1362
1363    #[test]
1364    fn current_patch_models_match_provider_examples() {
1365        let claude = estimate_tokens(1000, 1000, VisionModel::Claude);
1366        assert_eq!(claude.tokens, 1296); // ceil(1000 / 28)^2
1367
1368        let gpt6 = estimate_tokens(1024, 1024, VisionModel::Gpt6);
1369        assert_eq!(gpt6.tokens, 1229); // 1024 patches × 1.2, rounded up
1370
1371        let large_gpt6 = estimate_tokens(2048, 2048, VisionModel::Gpt6);
1372        assert_eq!(large_gpt6.tiles, 2500); // resized to 1600×1600
1373        assert_eq!(large_gpt6.tokens, 3000);
1374    }
1375
1376    #[test]
1377    fn model_aliases_and_legacy_gpt5_pricing_are_stable() {
1378        assert!(matches!(
1379            VisionModel::parse("gpt-5.6"),
1380            Some(VisionModel::Gpt6)
1381        ));
1382        assert!(matches!(
1383            VisionModel::parse("gpt-5.5"),
1384            Some(VisionModel::Gpt6)
1385        ));
1386        assert!(matches!(
1387            VisionModel::parse("claude-standard"),
1388            Some(VisionModel::ClaudeStandard)
1389        ));
1390        assert!(matches!(
1391            VisionModel::parse("kimi-k2.6"),
1392            Some(VisionModel::KimiVision)
1393        ));
1394        assert!(matches!(
1395            VisionModel::parse("deepseek"),
1396            Some(VisionModel::DeepseekFlash)
1397        ));
1398        assert!(matches!(
1399            VisionModel::parse("pixtral"),
1400            Some(VisionModel::GenericVision)
1401        ));
1402        assert!(matches!(
1403            VisionModel::parse("glm-5.3-flash"),
1404            Some(VisionModel::GenericVision)
1405        ));
1406        assert_eq!(
1407            estimate_tokens(1024, 1024, VisionModel::GenericVision).tokens,
1408            1369
1409        );
1410        assert_eq!(
1411            estimate_tokens(4096, 4096, VisionModel::DeepseekFlash).tokens,
1412            384
1413        );
1414        assert_eq!(estimate_tokens(1024, 1024, VisionModel::Gpt5).tokens, 630);
1415    }
1416
1417    #[test]
1418    fn full_pipeline_reduces_tiles() {
1419        use image::{DynamicImage, Rgba, RgbaImage};
1420        let mut img = RgbaImage::from_pixel(1025, 1025, Rgba([255, 255, 255, 255]));
1421        for x in 400..600 {
1422            for y in 400..600 {
1423                img.put_pixel(x, y, Rgba([0, 0, 0, 255]));
1424            }
1425        }
1426        let result = process(
1427            DynamicImage::ImageRgba8(img),
1428            ProcessMode::Standard,
1429            0,
1430            &cfg(),
1431        );
1432        assert!(result.report.tiles_after < result.report.tiles_before);
1433    }
1434
1435    #[test]
1436    fn crop_disabled_preserves_size() {
1437        use image::{DynamicImage, Rgba, RgbaImage};
1438        let img = RgbaImage::from_pixel(1024, 1024, Rgba([255, 255, 255, 255]));
1439        let no_crop = ProcessConfig::builder().crop(false).build();
1440        let result = process(
1441            DynamicImage::ImageRgba8(img),
1442            ProcessMode::Standard,
1443            0,
1444            &no_crop,
1445        );
1446        assert_eq!(result.width, 1024);
1447    }
1448
1449    #[test]
1450    fn crop_removes_white_border() {
1451        use image::{Rgba, RgbaImage};
1452        let mut img = RgbaImage::from_pixel(100, 100, Rgba([255, 255, 255, 255]));
1453        for x in 45..55 {
1454            for y in 45..55 {
1455                img.put_pixel(x, y, Rgba([255, 0, 0, 255]));
1456            }
1457        }
1458        let cropped = crop_padding(DynamicImage::ImageRgba8(img), 15);
1459        assert!(cropped.width() < 100 && cropped.height() < 100);
1460    }
1461
1462    #[test]
1463    fn binarize_produces_only_black_white() {
1464        use image::{DynamicImage, GrayImage, Luma};
1465        let img = GrayImage::from_fn(64, 64, |x, _| Luma([if x < 32 { 50u8 } else { 200u8 }]));
1466        let result = binarize(DynamicImage::ImageLuma8(img)).to_luma8();
1467        for p in result.pixels() {
1468            assert!(p.0[0] == 0 || p.0[0] == 255);
1469        }
1470    }
1471
1472    #[test]
1473    fn ssim_identical_images_is_one() {
1474        use image::{DynamicImage, Rgba, RgbaImage};
1475        let img =
1476            DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([128, 128, 128, 255])));
1477        let s = ssim(&img, &img);
1478        assert!((s - 1.0).abs() < 1e-9);
1479    }
1480
1481    #[test]
1482    fn ssim_very_different_images_is_low() {
1483        use image::{DynamicImage, Rgba, RgbaImage};
1484        let black = DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([0, 0, 0, 255])));
1485        let white =
1486            DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([255, 255, 255, 255])));
1487        let s = ssim(&black, &white);
1488        assert!(s < 0.1, "expected low SSIM, got {s}");
1489    }
1490
1491    #[test]
1492    fn saliency_crop_tightens_around_high_energy_region() {
1493        use image::{DynamicImage, Rgba, RgbaImage};
1494        // Uniform white field with a 200x200 textured block in the middle of a 1000x1000 image.
1495        let mut img = RgbaImage::from_pixel(1000, 1000, Rgba([255, 255, 255, 255]));
1496        for x in 400..600 {
1497            for y in 400..600 {
1498                // checker pattern to generate edge energy
1499                let v = if (x + y) % 2 == 0 { 0 } else { 255 };
1500                img.put_pixel(x, y, Rgba([v, v, v, 255]));
1501            }
1502        }
1503        let dyn_img = DynamicImage::ImageRgba8(img);
1504        let cropped = saliency_crop(&dyn_img, 8);
1505        assert!(cropped.width() < 1000);
1506        assert!(cropped.height() < 1000);
1507        // expect to land near the 200×200 block plus margin
1508        assert!(cropped.width() < 400);
1509        assert!(cropped.height() < 400);
1510    }
1511
1512    #[test]
1513    fn auto_quality_returns_quality_in_range() {
1514        use image::{DynamicImage, Rgba, RgbaImage};
1515        let mut img = RgbaImage::from_pixel(256, 256, Rgba([100, 100, 100, 255]));
1516        for x in 0..256 {
1517            for y in 0..256 {
1518                img.put_pixel(x, y, Rgba([(x % 256) as u8, (y % 256) as u8, 128, 255]));
1519            }
1520        }
1521        let dyn_img = DynamicImage::ImageRgba8(img);
1522        let cfg = ProcessConfig::default();
1523        let (bytes, q) = encode_with_auto_quality(&dyn_img, &cfg, 0.95, 40, 95).expect("ok");
1524        assert!((40..=95).contains(&q));
1525        assert!(!bytes.is_empty());
1526    }
1527
1528    #[test]
1529    fn high_bg_tolerance_crops_more() {
1530        use image::{DynamicImage, Rgba, RgbaImage};
1531        // Corners: pure white [255,255,255]. Border: off-white [240,240,240]. Center: black.
1532        // diff = 15. strict(5): 15 > 5 → border NOT bg → no crop.
1533        // loose(20): 15 ≤ 20 → border IS bg → crops.
1534        let mut img = RgbaImage::from_pixel(100, 100, Rgba([240, 240, 240, 255]));
1535        for corner in [(0u32, 0u32), (99, 0), (0, 99), (99, 99)] {
1536            img.put_pixel(corner.0, corner.1, Rgba([255, 255, 255, 255]));
1537        }
1538        for x in 45..55 {
1539            for y in 45..55 {
1540                img.put_pixel(x, y, Rgba([0, 0, 0, 255]));
1541            }
1542        }
1543        let strict = crop_padding(DynamicImage::ImageRgba8(img.clone()), 5);
1544        let loose = crop_padding(DynamicImage::ImageRgba8(img), 20);
1545        assert!(loose.width() < strict.width());
1546    }
1547}
1548// ── Persistence & Analytics ───────────────────────────────────────────────────
1549
1550#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)]
1551pub struct OptimizationReport {
1552    pub timestamp: String,
1553    pub model: String,
1554    pub original_tokens: u32,
1555    pub optimized_tokens: u32,
1556    pub original_bytes: u64,
1557    pub optimized_bytes: u64,
1558    pub mode: String,
1559}
1560
1561#[derive(Debug, serde::Serialize, serde::Deserialize)]
1562pub struct SqueezerStats {
1563    pub total_optimizations: u64,
1564    pub total_original_tokens: u64,
1565    pub total_optimized_tokens: u64,
1566    pub total_original_bytes: u64,
1567    pub total_optimized_bytes: u64,
1568    pub history: Vec<OptimizationReport>,
1569}
1570
1571impl SqueezerStats {
1572    pub fn total_token_savings(&self) -> u64 {
1573        self.total_original_tokens
1574            .saturating_sub(self.total_optimized_tokens)
1575    }
1576
1577    pub fn total_byte_savings(&self) -> u64 {
1578        self.total_original_bytes
1579            .saturating_sub(self.total_optimized_bytes)
1580    }
1581
1582    pub fn estimated_usd_saved(&self) -> f64 {
1583        // Blended average: $2.50 per 1M tokens (Claude/GPT-4o blend)
1584        (self.total_token_savings() as f64 / 1_000_000.0) * 2.50
1585    }
1586}
1587
1588pub struct Persistence;
1589
1590impl Persistence {
1591    fn get_db_path() -> PathBuf {
1592        let mut path = dirs::home_dir().unwrap_or_else(|| PathBuf::from("."));
1593        path.push(".vision-squeezer");
1594        let _ = std::fs::create_dir_all(&path);
1595        path.push("stats.db");
1596        path
1597    }
1598
1599    pub fn init_db() -> Result<(), String> {
1600        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1601        conn.execute(
1602            "CREATE TABLE IF NOT EXISTS optimizations (
1603                id INTEGER PRIMARY KEY AUTOINCREMENT,
1604                timestamp TEXT NOT NULL,
1605                model TEXT NOT NULL,
1606                original_tokens INTEGER NOT NULL,
1607                optimized_tokens INTEGER NOT NULL,
1608                original_bytes INTEGER NOT NULL,
1609                optimized_bytes INTEGER NOT NULL,
1610                mode TEXT NOT NULL
1611            )",
1612            [],
1613        )
1614        .map_err(|e| e.to_string())?;
1615        Ok(())
1616    }
1617
1618    pub fn log_optimization(
1619        model: &str,
1620        orig_tokens: u32,
1621        opt_tokens: u32,
1622        orig_bytes: u64,
1623        opt_bytes: u64,
1624        mode: &str,
1625    ) -> Result<(), String> {
1626        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1627        conn.execute(
1628            "INSERT INTO optimizations (timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode)
1629             VALUES (?, ?, ?, ?, ?, ?, ?)",
1630            params![
1631                Utc::now().to_rfc3339(),
1632                model,
1633                orig_tokens,
1634                opt_tokens,
1635                orig_bytes as i64,
1636                opt_bytes as i64,
1637                mode,
1638            ],
1639        ).map_err(|e| e.to_string())?;
1640        Ok(())
1641    }
1642
1643    pub fn get_stats() -> Result<SqueezerStats, String> {
1644        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1645
1646        let mut stmt = conn
1647            .prepare(
1648                "SELECT 
1649                COUNT(*), 
1650                SUM(original_tokens), 
1651                SUM(optimized_tokens), 
1652                SUM(original_bytes), 
1653                SUM(optimized_bytes) 
1654             FROM optimizations",
1655            )
1656            .map_err(|e| e.to_string())?;
1657
1658        let (count, orig_t, opt_t, orig_b, opt_b) = stmt
1659            .query_row([], |row| {
1660                Ok((
1661                    row.get::<_, Option<i64>>(0)?.unwrap_or(0) as u64,
1662                    row.get::<_, Option<i64>>(1)?.unwrap_or(0) as u64,
1663                    row.get::<_, Option<i64>>(2)?.unwrap_or(0) as u64,
1664                    row.get::<_, Option<i64>>(3)?.unwrap_or(0) as u64,
1665                    row.get::<_, Option<i64>>(4)?.unwrap_or(0) as u64,
1666                ))
1667            })
1668            .map_err(|e| e.to_string())?;
1669
1670        let mut stmt = conn.prepare(
1671            "SELECT timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode 
1672             FROM optimizations ORDER BY timestamp DESC LIMIT 50"
1673        ).map_err(|e| e.to_string())?;
1674
1675        let history = stmt
1676            .query_map([], |row| {
1677                Ok(OptimizationReport {
1678                    timestamp: row.get(0)?,
1679                    model: row.get(1)?,
1680                    original_tokens: row.get(2)?,
1681                    optimized_tokens: row.get(3)?,
1682                    original_bytes: row.get::<_, i64>(4)? as u64,
1683                    optimized_bytes: row.get::<_, i64>(5)? as u64,
1684                    mode: row.get(6)?,
1685                })
1686            })
1687            .map_err(|e| e.to_string())?
1688            .collect::<Result<Vec<_>, _>>()
1689            .map_err(|e| e.to_string())?;
1690
1691        Ok(SqueezerStats {
1692            total_optimizations: count,
1693            total_original_tokens: orig_t,
1694            total_optimized_tokens: opt_t,
1695            total_original_bytes: orig_b,
1696            total_optimized_bytes: opt_b,
1697            history,
1698        })
1699    }
1700
1701    pub fn get_all_history() -> Result<Vec<OptimizationReport>, String> {
1702        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1703        let mut stmt = conn.prepare(
1704            "SELECT timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode
1705             FROM optimizations ORDER BY timestamp ASC"
1706        ).map_err(|e| e.to_string())?;
1707
1708        stmt.query_map([], |row| {
1709            Ok(OptimizationReport {
1710                timestamp: row.get(0)?,
1711                model: row.get(1)?,
1712                original_tokens: row.get(2)?,
1713                optimized_tokens: row.get(3)?,
1714                original_bytes: row.get::<_, i64>(4)? as u64,
1715                optimized_bytes: row.get::<_, i64>(5)? as u64,
1716                mode: row.get(6)?,
1717            })
1718        })
1719        .map_err(|e| e.to_string())?
1720        .collect::<Result<Vec<_>, _>>()
1721        .map_err(|e| e.to_string())
1722    }
1723}