Skip to main content

vision_squeezer/
lib.rs

1use std::io::Cursor;
2
3use base64::{Engine, engine::general_purpose::STANDARD as B64};
4use chrono::Utc;
5use image::{DynamicImage, ImageBuffer, Luma, imageops::FilterType};
6use rusqlite::{Connection, params};
7use std::path::PathBuf;
8// ── Config ────────────────────────────────────────────────────────────────────
9
10/// Output encoding format.
11#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
12pub enum OutputFormat {
13    /// JPEG at configured quality (default).
14    #[default]
15    Jpeg,
16    /// WebP at configured quality — typically 30-50% smaller than JPEG at equal quality.
17    WebP,
18    /// AVIF at configured quality — typically 20-50% smaller than WebP at equal quality.
19    Avif,
20}
21
22/// All tuneable knobs for the pipeline.
23#[derive(Clone, Debug)]
24pub struct ProcessConfig {
25    /// Output quality 1–100 (default 75). Applies to both JPEG and WebP.
26    pub quality: u8,
27    /// LLM patch size in pixels. Overridden when `target_model` is set.
28    pub tile_size: u32,
29    /// Remove solid-color padding borders before resizing (default true).
30    pub crop: bool,
31    /// Max channel delta to treat a pixel as background (default 15).
32    pub bg_tolerance: u8,
33    /// Output encoding format (default: JPEG).
34    pub output_format: OutputFormat,
35    /// When set, resizing is model-aware (accounts for pre-scaling behavior).
36    pub target_model: Option<VisionModel>,
37    /// Limit the maximum number of tiles the output image can consume.
38    pub max_tiles: Option<u32>,
39    /// Use saliency (edge-energy) based crop instead of corner-tolerance crop.
40    pub smart_crop: bool,
41}
42
43impl Default for ProcessConfig {
44    fn default() -> Self {
45        Self {
46            quality: 75,
47            tile_size: 512,
48            crop: true,
49            bg_tolerance: 15,
50            output_format: OutputFormat::Jpeg,
51            target_model: None,
52            max_tiles: None,
53            smart_crop: false,
54        }
55    }
56}
57
58impl ProcessConfig {
59    pub fn builder() -> ProcessConfigBuilder {
60        ProcessConfigBuilder(Self::default())
61    }
62}
63
64pub struct ProcessConfigBuilder(ProcessConfig);
65
66impl ProcessConfigBuilder {
67    pub fn quality(mut self, q: u8) -> Self {
68        self.0.quality = q.clamp(1, 100);
69        self
70    }
71    pub fn tile_size(mut self, t: u32) -> Self {
72        self.0.tile_size = t.max(1);
73        self
74    }
75    pub fn crop(mut self, c: bool) -> Self {
76        self.0.crop = c;
77        self
78    }
79    pub fn bg_tolerance(mut self, t: u8) -> Self {
80        self.0.bg_tolerance = t;
81        self
82    }
83    pub fn output_format(mut self, f: OutputFormat) -> Self {
84        self.0.output_format = f;
85        self
86    }
87    pub fn target_model(mut self, m: VisionModel) -> Self {
88        self.0.target_model = Some(m);
89        self
90    }
91    pub fn max_tiles(mut self, m: u32) -> Self {
92        self.0.max_tiles = Some(m);
93        self
94    }
95    pub fn smart_crop(mut self, b: bool) -> Self {
96        self.0.smart_crop = b;
97        self
98    }
99    pub fn build(self) -> ProcessConfig {
100        self.0
101    }
102}
103
104// ── Token Estimation ──────────────────────────────────────────────────────────
105
106/// Supported vision model families with their patch pricing.
107#[derive(Clone, Copy, Debug)]
108pub enum VisionModel {
109    /// Claude 3.5/4.5/4.6/4.7: Area-based calculation (Tokens ≈ width × height / 750).
110    Claude,
111    /// GPT-4o / GPT-4.5 high detail: fits in 2048x2048, scales short side to 768, then 512x512 tiles.
112    Gpt4o,
113    /// GPT-5/5.5: 6000px max dim, 10.24M max pixels, 512×512 tiles, 1536 token cap.
114    Gpt5,
115    /// Gemini 2.0/3.0: flat 258 tokens if ≤ 384x384, else 258 per 768x768 tile.
116    Gemini15,
117    /// Meta Llama 3.2/3.3 Vision (Mllama): 560×560 tiles, aspect-ratio canvas capped at 4 tiles.
118    /// (Llama 4 uses a different native-multimodal vision encoder and is not modeled by this arm.)
119    LlamaVision,
120    /// Alibaba Qwen2-VL / Qwen2.5-VL: 28px effective grid (14px ViT patch × 2×2 merge), token clamp.
121    QwenVl,
122    /// DeepSeek-VL2: 384×384 base + dynamic 384 local tiles (open-weights; value is local-context savings).
123    DeepseekVl,
124}
125
126#[derive(Debug)]
127pub struct TokenEstimate {
128    pub model: VisionModel,
129    pub tokens: u32,
130    pub tiles: u32,
131}
132
133/// Estimate LLM vision tokens for an image of given dimensions.
134pub fn estimate_tokens(width: u32, height: u32, model: VisionModel) -> TokenEstimate {
135    match model {
136        VisionModel::Claude => {
137            // 2026 area-based pricing for Claude
138            let tokens = ((width as u64 * height as u64) / 750) as u32;
139            TokenEstimate {
140                model,
141                tiles: 1,
142                tokens: tokens.max(85),
143            }
144        }
145        VisionModel::Gpt4o => {
146            // GPT-4o / 4.5: fit within 2048x2048, then short side scaled to 768px, then 512x512 tiles.
147            let (mut w, mut h) = fit_within(width, height, 2048);
148            let short_side = w.min(h);
149            if short_side > 768 {
150                let scale = 768.0 / short_side as f64;
151                w = (w as f64 * scale).round() as u32;
152                h = (h as f64 * scale).round() as u32;
153            }
154            let tiles = tile_count(w, 512) * tile_count(h, 512);
155            TokenEstimate {
156                model,
157                tiles,
158                tokens: 85 + tiles * 170,
159            }
160        }
161        VisionModel::Gpt5 => {
162            let (w, h) = fit_within_pixels(width, height, 6000, 10_240_000);
163            let tiles = tile_count(w, 512) * tile_count(h, 512);
164            TokenEstimate {
165                model,
166                tiles,
167                tokens: (85 + tiles * 170).min(1536),
168            }
169        }
170        VisionModel::Gemini15 => {
171            // Gemini 2026: flat 258 if <= 384x384, else 768x768 tiles.
172            if width <= 384 && height <= 384 {
173                TokenEstimate {
174                    model,
175                    tiles: 1,
176                    tokens: 258,
177                }
178            } else {
179                let tiles = tile_count(width, 768) * tile_count(height, 768);
180                TokenEstimate {
181                    model,
182                    tiles,
183                    tokens: tiles * 258,
184                }
185            }
186        }
187        VisionModel::LlamaVision => {
188            // Meta Llama 3.2 / 3.3 Vision (Mllama): aspect-ratio canvas of 560×560 tiles, capped at
189            // max_num_tiles = 4. No separate global-thumbnail tile — the canvas is the full
190            // representation. 14px ViT patch → 40×40 = 1600 patches + 1 CLS = 1601 tokens per tile.
191            // Source: transformers MllamaVisionConfig (image_size 560, patch_size 14, max_num_tiles 4).
192            let (w, h) = fit_within(width, height, 1120); // 2×2 max canvas (4 tiles)
193            let tiles = (tile_count(w, 560) * tile_count(h, 560)).clamp(1, 4);
194            TokenEstimate {
195                model,
196                tiles,
197                tokens: tiles * 1601,
198            }
199        }
200        VisionModel::QwenVl => {
201            // Alibaba Qwen2-VL / Qwen2.5-VL: 28px effective grid (image_patch_size 14 × spatial_merge 2).
202            // smart_resize rounds each side to a multiple of 28 and bounds total tokens to
203            // [IMAGE_MIN_TOKEN_NUM, IMAGE_MAX_TOKEN_NUM] = [4, 16384] (qwen_vl_utils defaults).
204            // tokens = (W/28)·(H/28). DashScope endpoints may cap lower via a per-request max_pixels.
205            let (w, h) = fit_within_pixels(width, height, u32::MAX, 16_384 * 28 * 28);
206            let patches = tile_count(w, 28) * tile_count(h, 28);
207            TokenEstimate {
208                model,
209                tiles: patches,
210                tokens: patches.clamp(4, 16_384),
211            }
212        }
213        VisionModel::DeepseekVl => {
214            // DeepSeek-VL2 (open weights — deepseek public API is text-only, so the win is local
215            // inference context, not API billing). SigLIP-SO400M-patch14-384 emits 27×27 patches per
216            // tile; a 2×2 pixel-shuffle compresses that to 14×14 = 196 tokens (h = 14). Token layout:
217            //   global thumbnail = 14·(14+1) = 210  (one <tile_newline> per row)
218            //   + 1 <view_separator>
219            //   local tiles      = (nh·14)·(nw·14 + 1) over the anyres canvas (m·384, n·384), m·n ≤ 9
220            // Sources: DeepSeek-VL2 paper §2 (arXiv:2412.10302) + processing_deepseek_vl_v2.py.
221            const H: u32 = 14;
222            let (nw, nh) = if width <= 384 && height <= 384 {
223                (1, 1)
224            } else {
225                let mut nw = tile_count(width, 384).max(1);
226                let mut nh = tile_count(height, 384).max(1);
227                while nw * nh > 9 {
228                    if nw >= nh {
229                        nw -= 1;
230                    } else {
231                        nh -= 1;
232                    }
233                }
234                (nw, nh)
235            };
236            let global = H * (H + 1) + 1; // global view + separator
237            let local = (nh * H) * (nw * H + 1);
238            TokenEstimate {
239                model,
240                tiles: nw * nh + 1, // local tiles + global view
241                tokens: global + local,
242            }
243        }
244    }
245}
246
247/// Scale dimensions to fit within `max_side` while preserving aspect ratio.
248pub fn fit_within(width: u32, height: u32, max_side: u32) -> (u32, u32) {
249    if width <= max_side && height <= max_side {
250        return (width, height);
251    }
252    let scale = max_side as f64 / width.max(height) as f64;
253    (
254        (width as f64 * scale) as u32,
255        (height as f64 * scale) as u32,
256    )
257}
258
259/// Scale dimensions to fit within both a max-side limit and a total-pixel limit.
260pub fn fit_within_pixels(width: u32, height: u32, max_side: u32, max_pixels: u64) -> (u32, u32) {
261    let (mut w, mut h) = fit_within(width, height, max_side);
262    let total = w as u64 * h as u64;
263    if total > max_pixels {
264        let scale = (max_pixels as f64 / total as f64).sqrt();
265        w = (w as f64 * scale) as u32;
266        h = (h as f64 * scale) as u32;
267    }
268    (w.max(1), h.max(1))
269}
270
271/// Compute the optimal dimensions to *send* to a given model to minimize tiles.
272///
273/// For models that pre-scale images (GPT-4o, Gemini), we simulate their scaling,
274/// snap the scaled result to tile boundaries, then invert back to input space.
275/// For Claude (no pre-scaling), we snap the input directly.
276pub fn optimal_send_dimensions(width: u32, height: u32, model: VisionModel) -> (u32, u32) {
277    match model {
278        VisionModel::Claude => {
279            // Claude is now area-based, so tiling doesn't dictate a specific rigid boundary.
280            // But we still snap to 256 or 512 so dimensions aren't completely arbitrary.
281            (
282                snap_to_tile_boundary(width, 256),
283                snap_to_tile_boundary(height, 256),
284            )
285        }
286        VisionModel::Gpt4o => optimal_for_prescaling_model(width, height, 2048, 512),
287        VisionModel::Gpt5 => {
288            let (fw, fh) = fit_within_pixels(width, height, 6000, 10_240_000);
289            (
290                snap_to_tile_boundary(fw, 512).max(512),
291                snap_to_tile_boundary(fh, 512).max(512),
292            )
293        }
294        VisionModel::Gemini15 => {
295            // Gemini uses 768x768 tiles if > 384x384
296            if width <= 384 && height <= 384 {
297                (width, height)
298            } else {
299                optimal_for_prescaling_model(width, height, 4096, 768)
300            }
301        }
302        VisionModel::LlamaVision => {
303            // Snap to the 560px tile grid within the 2×2 (1120px) max canvas to avoid spill-over tiles.
304            optimal_for_prescaling_model(width, height, 1120, 560)
305        }
306        VisionModel::QwenVl => {
307            // Snap each side to the 28px patch grid, after fitting under the max-pixel budget.
308            let (fw, fh) = fit_within_pixels(width, height, u32::MAX, 16_384 * 28 * 28);
309            (
310                snap_to_tile_boundary(fw, 28).max(28),
311                snap_to_tile_boundary(fh, 28).max(28),
312            )
313        }
314        VisionModel::DeepseekVl => {
315            // ≤384 stays as a single tile; otherwise snap to the 384px tile grid.
316            if width <= 384 && height <= 384 {
317                (width, height)
318            } else {
319                optimal_for_prescaling_model(width, height, 1152, 384)
320            }
321        }
322    }
323}
324
325/// For models that pre-scale (GPT-4o, Gemini), find the smallest input dimensions
326/// that, after the model's internal fit-within + tiling, produce the fewest tiles.
327///
328/// Strategy: enumerate candidate tile-grid dimensions (tw*tile, th*tile) that fit
329/// within max_side, compute the input size that would map to each, and pick the
330/// candidate that uses the fewest tiles while preserving the original aspect ratio
331/// as closely as possible.
332fn optimal_for_prescaling_model(width: u32, height: u32, max_side: u32, tile: u32) -> (u32, u32) {
333    let (fw, fh) = fit_within(width, height, max_side);
334
335    // Simply snap the fitted dimensions to the nearest tile boundary
336    let target_w = snap_to_tile_boundary(fw, tile).max(tile);
337    let target_h = snap_to_tile_boundary(fh, tile).max(tile);
338
339    // If image was larger than max_side, scale back to input space
340    if width > max_side || height > max_side {
341        let scale = width.max(height) as f64 / max_side as f64;
342        let opt_w = (target_w as f64 * scale).round() as u32;
343        let opt_h = (target_h as f64 * scale).round() as u32;
344        return (opt_w.max(1), opt_h.max(1));
345    }
346
347    (target_w, target_h)
348}
349
350/// Full token savings report for a before/after dimension pair across all models.
351pub struct TokenSavingsTable {
352    pub claude_before: TokenEstimate,
353    pub claude_after: TokenEstimate,
354    pub gpt4o_before: TokenEstimate,
355    pub gpt4o_after: TokenEstimate,
356    pub gpt5_before: TokenEstimate,
357    pub gpt5_after: TokenEstimate,
358    pub gemini_before: TokenEstimate,
359    pub gemini_after: TokenEstimate,
360}
361
362pub fn token_savings_table(orig_w: u32, orig_h: u32, opt_w: u32, opt_h: u32) -> TokenSavingsTable {
363    TokenSavingsTable {
364        claude_before: estimate_tokens(orig_w, orig_h, VisionModel::Claude),
365        claude_after: estimate_tokens(opt_w, opt_h, VisionModel::Claude),
366        gpt4o_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt4o),
367        gpt4o_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt4o),
368        gpt5_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt5),
369        gpt5_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt5),
370        gemini_before: estimate_tokens(orig_w, orig_h, VisionModel::Gemini15),
371        gemini_after: estimate_tokens(opt_w, opt_h, VisionModel::Gemini15),
372    }
373}
374
375impl TokenSavingsTable {
376    pub fn print(&self) {
377        println!(
378            "{:<12} {:>8} {:>8} {:>10}",
379            "Model", "Before", "After", "Saved"
380        );
381        println!("{}", "-".repeat(42));
382        self.print_row("Claude", &self.claude_before, &self.claude_after);
383        self.print_row("GPT-4o", &self.gpt4o_before, &self.gpt4o_after);
384        self.print_row("GPT-5", &self.gpt5_before, &self.gpt5_after);
385        self.print_row("Gemini", &self.gemini_before, &self.gemini_after);
386    }
387
388    fn print_row(&self, name: &str, before: &TokenEstimate, after: &TokenEstimate) {
389        let saved = before.tokens.saturating_sub(after.tokens);
390        let pct = if before.tokens > 0 {
391            saved as f64 / before.tokens as f64 * 100.0
392        } else {
393            0.0
394        };
395        println!(
396            "{:<12} {:>8} {:>8} {:>8} ({:.1}%)",
397            name, before.tokens, after.tokens, saved, pct
398        );
399    }
400}
401
402// ── Types ─────────────────────────────────────────────────────────────────────
403
404pub struct DimensionResult {
405    pub width: u32,
406    pub height: u32,
407    pub tiles_before: u32,
408    pub tiles_after: u32,
409}
410
411impl DimensionResult {
412    pub fn tokens_saved(&self) -> u32 {
413        self.tiles_before.saturating_sub(self.tiles_after)
414    }
415}
416
417#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
418pub enum ProcessMode {
419    /// General LLM vision — JPEG output at configured quality.
420    Standard,
421    /// Text extraction — high-contrast grayscale binarization (Otsu threshold).
422    Ocr,
423    /// Auto-detects if the image is mostly text (monochrome/grayscale).
424    #[default]
425    Auto,
426}
427
428pub fn detect_ocr_mode(img: &DynamicImage) -> bool {
429    let rgb = img.to_rgb8();
430    let mut colorful_count = 0;
431    let mut total_count = 0;
432    // Sample every 4th pixel for speed
433    for (x, y, p) in rgb.enumerate_pixels() {
434        if x % 4 == 0 && y % 4 == 0 {
435            total_count += 1;
436            let min = p[0].min(p[1]).min(p[2]);
437            let max = p[0].max(p[1]).max(p[2]);
438            if max.saturating_sub(min) > 25 {
439                colorful_count += 1;
440            }
441        }
442    }
443    let colorful_ratio = colorful_count as f64 / total_count.max(1) as f64;
444    colorful_ratio < 0.1 // if less than 10% of pixels are colorful, assume OCR
445}
446
447pub struct SavingsReport {
448    pub tiles_before: u32,
449    pub tiles_after: u32,
450    pub tiles_saved: u32,
451    pub bytes_before: Option<u64>,
452    pub bytes_after: Option<u64>,
453}
454
455impl SavingsReport {
456    pub fn size_reduction_pct(&self) -> Option<f64> {
457        match (self.bytes_before, self.bytes_after) {
458            (Some(b), Some(a)) if b > 0 => Some((1.0 - a as f64 / b as f64) * 100.0),
459            _ => None,
460        }
461    }
462
463    pub fn token_reduction_pct(&self) -> f64 {
464        if self.tiles_before == 0 {
465            return 0.0;
466        }
467        self.tiles_saved as f64 / self.tiles_before as f64 * 100.0
468    }
469}
470
471pub struct ProcessResult {
472    pub image: DynamicImage,
473    pub width: u32,
474    pub height: u32,
475    pub report: SavingsReport,
476}
477
478impl ProcessResult {
479    pub fn tokens_saved(&self) -> u32 {
480        self.report.tiles_saved
481    }
482}
483
484// ── Pipeline ──────────────────────────────────────────────────────────────────
485
486/// Full pipeline: [crop] → tile-snap resize → [OCR binarize].
487/// Pass `input_bytes = 0` if unknown (omits file-size from report).
488pub fn process(
489    img: DynamicImage,
490    mode: ProcessMode,
491    input_bytes: u64,
492    cfg: &ProcessConfig,
493) -> ProcessResult {
494    let (orig_w, orig_h) = (img.width(), img.height());
495    let tiles_before = match cfg.target_model {
496        Some(model) => estimate_tokens(orig_w, orig_h, model).tiles,
497        None => tile_count(orig_w, cfg.tile_size) * tile_count(orig_h, cfg.tile_size),
498    };
499
500    let after_crop = if cfg.crop {
501        if cfg.smart_crop {
502            saliency_crop(&img, 16)
503        } else {
504            crop_padding(img, cfg.bg_tolerance)
505        }
506    } else {
507        img
508    };
509    let (mut opt_w, mut opt_h) = match cfg.target_model {
510        Some(model) => optimal_send_dimensions(after_crop.width(), after_crop.height(), model),
511        None => {
512            let d = calculate_optimal_dimensions_with(
513                after_crop.width(),
514                after_crop.height(),
515                cfg.tile_size,
516            );
517            (d.width, d.height)
518        }
519    };
520
521    if let Some(max_t) = cfg.max_tiles {
522        let (nw, nh) = enforce_max_tiles(opt_w, opt_h, max_t, cfg.tile_size, cfg.target_model);
523        opt_w = nw;
524        opt_h = nh;
525    }
526
527    let tiles_after = match cfg.target_model {
528        Some(model) => {
529            let est = estimate_tokens(opt_w, opt_h, model);
530            est.tiles
531        }
532        None => tile_count(opt_w, cfg.tile_size) * tile_count(opt_h, cfg.tile_size),
533    };
534    let resized = after_crop.resize_exact(opt_w, opt_h, FilterType::Lanczos3);
535
536    let actual_mode = match mode {
537        ProcessMode::Auto => {
538            if detect_ocr_mode(&after_crop) {
539                ProcessMode::Ocr
540            } else {
541                ProcessMode::Standard
542            }
543        }
544        m => m,
545    };
546
547    let final_image = match actual_mode {
548        ProcessMode::Standard | ProcessMode::Auto => resized,
549        ProcessMode::Ocr => binarize(resized),
550    };
551
552    ProcessResult {
553        width: final_image.width(),
554        height: final_image.height(),
555        image: final_image,
556        report: SavingsReport {
557            tiles_before,
558            tiles_after,
559            tiles_saved: tiles_before.saturating_sub(tiles_after),
560            bytes_before: if input_bytes > 0 {
561                Some(input_bytes)
562            } else {
563                None
564            },
565            bytes_after: None,
566        },
567    }
568}
569
570fn enforce_max_tiles(
571    mut width: u32,
572    mut height: u32,
573    max_tiles: u32,
574    default_tile_size: u32,
575    model: Option<VisionModel>,
576) -> (u32, u32) {
577    if max_tiles == 0 {
578        return (width, height);
579    }
580
581    let mut scale = 1.0;
582    let orig_w = width;
583    let orig_h = height;
584
585    loop {
586        let (snapped_w, snapped_h) = match model {
587            Some(m) => optimal_send_dimensions(width, height, m),
588            None => {
589                let d = calculate_optimal_dimensions_with(width, height, default_tile_size);
590                (d.width, d.height)
591            }
592        };
593
594        let tiles = match model {
595            Some(m) => estimate_tokens(snapped_w, snapped_h, m).tiles,
596            None => {
597                tile_count(snapped_w, default_tile_size) * tile_count(snapped_h, default_tile_size)
598            }
599        };
600
601        if tiles <= max_tiles || scale < 0.1 {
602            return (snapped_w, snapped_h);
603        }
604
605        scale *= 0.95;
606        width = (orig_w as f64 * scale) as u32;
607        height = (orig_h as f64 * scale) as u32;
608        width = width.max(1);
609        height = height.max(1);
610    }
611}
612
613// ── Step 1: Tile-Aware Dimension Calculation ───────────────────────────────────
614
615/// Snap W×H to tile boundaries using default tile size (512).
616pub fn calculate_optimal_dimensions(width: u32, height: u32) -> DimensionResult {
617    calculate_optimal_dimensions_with(width, height, 512)
618}
619
620/// Snap W×H to tile boundaries using a custom tile size.
621pub fn calculate_optimal_dimensions_with(
622    width: u32,
623    height: u32,
624    tile_size: u32,
625) -> DimensionResult {
626    let opt_w = snap_to_tile_boundary(width, tile_size);
627    let opt_h = snap_to_tile_boundary(height, tile_size);
628
629    DimensionResult {
630        width: opt_w,
631        height: opt_h,
632        tiles_before: tile_count(width, tile_size) * tile_count(height, tile_size),
633        tiles_after: tile_count(opt_w, tile_size) * tile_count(opt_h, tile_size),
634    }
635}
636
637fn tile_count(dim: u32, tile_size: u32) -> u32 {
638    dim.div_ceil(tile_size)
639}
640
641fn snap_to_tile_boundary(dim: u32, tile_size: u32) -> u32 {
642    if dim.is_multiple_of(tile_size) {
643        return dim;
644    }
645    ((dim / tile_size) * tile_size).max(tile_size)
646}
647
648// ── Step 2: Semantic Crop (padding removal) ────────────────────────────────────
649
650/// Remove solid-color borders using corner sampling + configurable tolerance.
651pub fn crop_padding(img: DynamicImage, bg_tolerance: u8) -> DynamicImage {
652    let rgba = img.to_rgba8();
653    let (w, h) = rgba.dimensions();
654
655    let corners = [
656        *rgba.get_pixel(0, 0),
657        *rgba.get_pixel(w - 1, 0),
658        *rgba.get_pixel(0, h - 1),
659        *rgba.get_pixel(w - 1, h - 1),
660    ];
661    let bg = corners[0]; // first corner as background reference
662
663    let top = first_non_bg_row(&rgba, bg, bg_tolerance, true);
664    let bottom = first_non_bg_row(&rgba, bg, bg_tolerance, false);
665    let left = first_non_bg_col(&rgba, bg, bg_tolerance, true);
666    let right = first_non_bg_col(&rgba, bg, bg_tolerance, false);
667
668    if top >= bottom || left >= right {
669        return DynamicImage::ImageRgba8(rgba);
670    }
671
672    DynamicImage::ImageRgba8(
673        image::imageops::crop_imm(&rgba, left, top, right - left, bottom - top).to_image(),
674    )
675}
676
677fn is_bg(pixel: image::Rgba<u8>, bg: image::Rgba<u8>, tolerance: u8) -> bool {
678    pixel.0[3] < 10
679        || pixel.0[..3]
680            .iter()
681            .zip(bg.0[..3].iter())
682            .all(|(&a, &b)| a.abs_diff(b) <= tolerance)
683}
684
685fn first_non_bg_row(img: &image::RgbaImage, bg: image::Rgba<u8>, tol: u8, from_top: bool) -> u32 {
686    let (w, h) = img.dimensions();
687    let rows: Box<dyn Iterator<Item = u32>> = if from_top {
688        Box::new(0..h)
689    } else {
690        Box::new((0..h).rev())
691    };
692    for y in rows {
693        if (0..w).any(|x| !is_bg(*img.get_pixel(x, y), bg, tol)) {
694            return y;
695        }
696    }
697    0
698}
699
700fn first_non_bg_col(img: &image::RgbaImage, bg: image::Rgba<u8>, tol: u8, from_left: bool) -> u32 {
701    let (w, h) = img.dimensions();
702    let cols: Box<dyn Iterator<Item = u32>> = if from_left {
703        Box::new(0..w)
704    } else {
705        Box::new((0..w).rev())
706    };
707    for x in cols {
708        if (0..h).any(|y| !is_bg(*img.get_pixel(x, y), bg, tol)) {
709            return x;
710        }
711    }
712    0
713}
714
715// ── Saliency Crop (edge-energy based) ─────────────────────────────────────────
716
717/// Crop to the bounding box of high-energy (edge) pixels.
718/// Uses a Sobel-lite gradient magnitude per luma pixel. Pixels above 2× the
719/// mean gradient energy define the salient region; the bbox is expanded by
720/// `margin` pixels on every side.
721///
722/// Falls back to the input unchanged for uniform images (no salient region).
723pub fn saliency_crop(img: &DynamicImage, margin: u32) -> DynamicImage {
724    let gray = img.to_luma8();
725    let (w, h) = gray.dimensions();
726    if w < 3 || h < 3 {
727        return img.clone();
728    }
729
730    let mut energy = vec![0u32; (w * h) as usize];
731    let mut total: u64 = 0;
732    for y in 1..h - 1 {
733        for x in 1..w - 1 {
734            let l = gray.get_pixel(x - 1, y).0[0] as i32;
735            let r = gray.get_pixel(x + 1, y).0[0] as i32;
736            let t = gray.get_pixel(x, y - 1).0[0] as i32;
737            let b = gray.get_pixel(x, y + 1).0[0] as i32;
738            let e = ((r - l).abs() + (b - t).abs()) as u32;
739            energy[(y * w + x) as usize] = e;
740            total += e as u64;
741        }
742    }
743    let count = (w as u64) * (h as u64);
744    let mean = (total / count.max(1)) as u32;
745    let threshold = mean.saturating_mul(2).max(8);
746
747    let (mut min_x, mut min_y, mut max_x, mut max_y) = (w, h, 0u32, 0u32);
748    for y in 0..h {
749        for x in 0..w {
750            if energy[(y * w + x) as usize] > threshold {
751                if x < min_x {
752                    min_x = x;
753                }
754                if y < min_y {
755                    min_y = y;
756                }
757                if x > max_x {
758                    max_x = x;
759                }
760                if y > max_y {
761                    max_y = y;
762                }
763            }
764        }
765    }
766
767    if min_x >= max_x || min_y >= max_y {
768        return img.clone();
769    }
770
771    let x0 = min_x.saturating_sub(margin);
772    let y0 = min_y.saturating_sub(margin);
773    let x1 = (max_x + 1 + margin).min(w);
774    let y1 = (max_y + 1 + margin).min(h);
775    img.crop_imm(x0, y0, x1 - x0, y1 - y0)
776}
777
778// ── SSIM + Auto-Quality ───────────────────────────────────────────────────────
779
780/// Compute structural-similarity (single-window, luma) between two images.
781///
782/// Returns a value in [-1, 1]; 1.0 means identical. Both images are converted
783/// to grayscale; if dimensions differ, the smaller is rescaled to match the
784/// larger via Lanczos3.
785pub fn ssim(a: &DynamicImage, b: &DynamicImage) -> f64 {
786    let (aw, ah) = (a.width(), a.height());
787    let (bw, bh) = (b.width(), b.height());
788    let (target_w, target_h) = (aw.max(bw), ah.max(bh));
789
790    let resize_if_needed = |img: &DynamicImage| -> image::GrayImage {
791        if img.width() == target_w && img.height() == target_h {
792            img.to_luma8()
793        } else {
794            img.resize_exact(target_w, target_h, FilterType::Lanczos3)
795                .to_luma8()
796        }
797    };
798
799    let a_luma = resize_if_needed(a);
800    let b_luma = resize_if_needed(b);
801
802    let n = (target_w as u64 * target_h as u64).max(1) as f64;
803    let (mut sum_a, mut sum_b) = (0f64, 0f64);
804    for (pa, pb) in a_luma.pixels().zip(b_luma.pixels()) {
805        sum_a += pa.0[0] as f64;
806        sum_b += pb.0[0] as f64;
807    }
808    let mean_a = sum_a / n;
809    let mean_b = sum_b / n;
810
811    let (mut var_a, mut var_b, mut cov) = (0f64, 0f64, 0f64);
812    for (pa, pb) in a_luma.pixels().zip(b_luma.pixels()) {
813        let da = pa.0[0] as f64 - mean_a;
814        let db = pb.0[0] as f64 - mean_b;
815        var_a += da * da;
816        var_b += db * db;
817        cov += da * db;
818    }
819    var_a /= n;
820    var_b /= n;
821    cov /= n;
822
823    let c1 = (0.01f64 * 255.0).powi(2);
824    let c2 = (0.03f64 * 255.0).powi(2);
825    let num = (2.0 * mean_a * mean_b + c1) * (2.0 * cov + c2);
826    let den = (mean_a.powi(2) + mean_b.powi(2) + c1) * (var_a + var_b + c2);
827    if den.abs() < f64::EPSILON {
828        1.0
829    } else {
830        num / den
831    }
832}
833
834/// Binary-search the lowest quality in [`min_q`, `max_q`] that meets a given
835/// SSIM target against the original. Returns the encoded bytes and the quality used.
836///
837/// Used when callers want "automatic" quality: pick the smallest file that still
838/// passes a perceptual threshold (typically 0.95).
839pub fn encode_with_auto_quality(
840    original: &DynamicImage,
841    cfg: &ProcessConfig,
842    target_ssim: f64,
843    min_q: u8,
844    max_q: u8,
845) -> Result<(Vec<u8>, u8), String> {
846    let mut lo = min_q.max(1);
847    let mut hi = max_q.min(100).max(lo + 1);
848    let mut best: Option<(Vec<u8>, u8)> = None;
849
850    while hi.saturating_sub(lo) > 2 {
851        let mid = lo + (hi - lo) / 2;
852        let trial = ProcessConfig {
853            quality: mid,
854            ..cfg.clone()
855        };
856        let bytes = encode_to_bytes(original, &trial)?;
857        let decoded = image::load_from_memory(&bytes).map_err(|e| e.to_string())?;
858        let score = ssim(original, &decoded);
859        if score >= target_ssim {
860            best = Some((bytes, mid));
861            hi = mid;
862        } else {
863            lo = mid;
864        }
865    }
866
867    // If no quality met the target during the search, encode at max_q as fallback.
868    if let Some((b, q)) = best {
869        Ok((b, q))
870    } else {
871        let trial = ProcessConfig {
872            quality: hi,
873            ..cfg.clone()
874        };
875        let bytes = encode_to_bytes(original, &trial)?;
876        Ok((bytes, hi))
877    }
878}
879
880// ── Step 3: OCR Binarization ───────────────────────────────────────────────────
881
882pub fn binarize(img: DynamicImage) -> DynamicImage {
883    let gray = img.to_luma8();
884    let (w, h) = gray.dimensions();
885    let threshold = otsu_threshold(&gray);
886    let binary: ImageBuffer<Luma<u8>, Vec<u8>> = ImageBuffer::from_fn(w, h, |x, y| {
887        let p = gray.get_pixel(x, y).0[0];
888        Luma([if p < threshold { 0u8 } else { 255u8 }])
889    });
890    DynamicImage::ImageLuma8(binary)
891}
892
893fn otsu_threshold(img: &image::GrayImage) -> u8 {
894    let mut histogram = [0u32; 256];
895    for p in img.pixels() {
896        histogram[p.0[0] as usize] += 1;
897    }
898    let total = img.width() * img.height();
899    let (mut sum, mut sum_bg, mut weight_bg) = (0f64, 0f64, 0f64);
900    for (i, &h) in histogram.iter().enumerate() {
901        sum += i as f64 * h as f64;
902    }
903    let (mut best_thresh, mut best_var) = (0u8, 0f64);
904    for (t, &h) in histogram.iter().enumerate() {
905        weight_bg += h as f64;
906        if weight_bg == 0.0 {
907            continue;
908        }
909        let weight_fg = total as f64 - weight_bg;
910        if weight_fg == 0.0 {
911            break;
912        }
913        sum_bg += t as f64 * h as f64;
914        let mean_bg = sum_bg / weight_bg;
915        let mean_fg = (sum - sum_bg) / weight_fg;
916        let var = weight_bg * weight_fg * (mean_bg - mean_fg).powi(2);
917        if var > best_var {
918            best_var = var;
919            best_thresh = t as u8;
920        }
921    }
922    best_thresh
923}
924
925// ── Base64 I/O ────────────────────────────────────────────────────────────────
926
927/// Reject decode work that would blow up memory: an attacker-supplied base64
928/// string can be tiny yet declare enormous dimensions (decompression bomb).
929const MAX_B64_LEN: usize = 64 * 1024 * 1024; // ~48 MB of raw image bytes
930const MAX_PIXELS: u64 = 100_000_000; // 100 MP
931const MAX_DIM: u32 = 16_384;
932
933pub fn decode_base64_image(input: &str) -> Result<DynamicImage, String> {
934    let data = if let Some(c) = input.find(',') {
935        &input[c + 1..]
936    } else {
937        input
938    };
939    let data = data.trim();
940    if data.len() > MAX_B64_LEN {
941        return Err(format!(
942            "image base64 exceeds {} MB limit",
943            MAX_B64_LEN / 1_048_576
944        ));
945    }
946    let bytes = B64.decode(data).map_err(|e| e.to_string())?;
947
948    let mut limits = image::Limits::default();
949    limits.max_image_width = Some(MAX_DIM);
950    limits.max_image_height = Some(MAX_DIM);
951    limits.max_alloc = Some(MAX_PIXELS * 4); // RGBA worst case
952
953    let mut reader = image::ImageReader::new(std::io::Cursor::new(bytes))
954        .with_guessed_format()
955        .map_err(|e| e.to_string())?;
956    reader.limits(limits);
957    reader.decode().map_err(|e| e.to_string())
958}
959
960pub fn encode_image_base64(img: &DynamicImage, cfg: &ProcessConfig) -> Result<String, String> {
961    let bytes = encode_to_bytes(img, cfg)?;
962    Ok(B64.encode(bytes))
963}
964
965/// Encode image to raw bytes using the configured output format.
966pub fn encode_to_bytes(img: &DynamicImage, cfg: &ProcessConfig) -> Result<Vec<u8>, String> {
967    match cfg.output_format {
968        OutputFormat::Jpeg => {
969            use image::codecs::jpeg::JpegEncoder;
970            let mut buf = Cursor::new(Vec::new());
971            let rgb = img.to_rgb8();
972            JpegEncoder::new_with_quality(&mut buf, cfg.quality)
973                .encode_image(&DynamicImage::ImageRgb8(rgb))
974                .map_err(|e| e.to_string())?;
975            Ok(buf.into_inner())
976        }
977        OutputFormat::WebP => {
978            let rgb = img.to_rgb8();
979            let enc = webp::Encoder::from_rgb(rgb.as_raw(), rgb.width(), rgb.height());
980            let mem = enc.encode(cfg.quality as f32);
981            Ok(mem.to_vec())
982        }
983        OutputFormat::Avif => {
984            use image::ImageEncoder;
985            use image::codecs::avif::AvifEncoder;
986            let mut buf = Cursor::new(Vec::new());
987            let rgba = img.to_rgba8();
988            // Speed 6 is a reasonable balance; lower = better compression but slower.
989            AvifEncoder::new_with_speed_quality(&mut buf, 6, cfg.quality)
990                .write_image(
991                    rgba.as_raw(),
992                    rgba.width(),
993                    rgba.height(),
994                    image::ExtendedColorType::Rgba8,
995                )
996                .map_err(|e| e.to_string())?;
997            Ok(buf.into_inner())
998        }
999    }
1000}
1001
1002// ── MCP Tool: optimize_image ──────────────────────────────────────────────────
1003
1004pub struct OptimizeResult {
1005    pub optimized_base64: String,
1006    pub report: SavingsReport,
1007    pub original_width: u32,
1008    pub original_height: u32,
1009    pub width: u32,
1010    pub height: u32,
1011    pub optimized_bytes: usize,
1012}
1013
1014/// MCP entry point: base64 in → base64 JPEG out + savings report.
1015pub fn optimize_image(
1016    input_base64: &str,
1017    mode: ProcessMode,
1018    cfg: &ProcessConfig,
1019) -> Result<OptimizeResult, String> {
1020    let img = decode_base64_image(input_base64)?;
1021    let (orig_w, orig_h) = (img.width(), img.height());
1022    let input_bytes = {
1023        let data = if let Some(c) = input_base64.find(',') {
1024            &input_base64[c + 1..]
1025        } else {
1026            input_base64
1027        };
1028        B64.decode(data.trim()).map_err(|e| e.to_string())?.len() as u64
1029    };
1030
1031    let mut result = process(img, mode, input_bytes, cfg);
1032    let bytes = encode_to_bytes(&result.image, cfg)?;
1033    let encoded = B64.encode(&bytes);
1034    result.report.bytes_after = Some(bytes.len() as u64);
1035
1036    Ok(OptimizeResult {
1037        optimized_base64: encoded,
1038        report: result.report,
1039        original_width: orig_w,
1040        original_height: orig_h,
1041        width: result.width,
1042        height: result.height,
1043        optimized_bytes: bytes.len(),
1044    })
1045}
1046
1047// ── Tests ─────────────────────────────────────────────────────────────────────
1048
1049// ── Step 4: Sandbox (Think in Code) ──────────────────────────────────────────
1050
1051/// Atomic image operations for the Sandbox mode.
1052#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)]
1053#[serde(rename_all = "lowercase", tag = "op")]
1054pub enum ImageOp {
1055    /// Crop a specific region: { x, y, width, height }
1056    Crop {
1057        x: u32,
1058        y: u32,
1059        width: u32,
1060        height: u32,
1061    },
1062    /// Convert to grayscale.
1063    Grayscale,
1064    /// Binarize using Otsu's threshold (if threshold is None).
1065    Binarize { threshold: Option<u8> },
1066    /// Resize to exact dimensions.
1067    Resize { width: u32, height: u32 },
1068    /// Adjust contrast (e.g., 2.0 for double contrast).
1069    Contrast { amount: f32 },
1070    /// Adjust brightness (e.g., -20 to darken).
1071    Brightness { amount: f32 },
1072}
1073
1074/// Execute a sequence of operations on an image.
1075pub fn process_with_operations(mut img: DynamicImage, ops: Vec<ImageOp>) -> DynamicImage {
1076    for op in ops {
1077        img = match op {
1078            ImageOp::Crop {
1079                x,
1080                y,
1081                width,
1082                height,
1083            } => img.crop_imm(x, y, width, height),
1084            ImageOp::Grayscale => DynamicImage::ImageLuma8(img.to_luma8()),
1085            ImageOp::Binarize { threshold } => {
1086                let gray = img.to_luma8();
1087                let thr = threshold.unwrap_or(128);
1088                let mut binarized = ImageBuffer::new(gray.width(), gray.height());
1089                for (x, y, p) in gray.enumerate_pixels() {
1090                    let val = if p[0] > thr { 255 } else { 0 };
1091                    binarized.put_pixel(x, y, Luma([val]));
1092                }
1093                DynamicImage::ImageLuma8(binarized)
1094            }
1095            ImageOp::Resize { width, height } => {
1096                img.resize_exact(width, height, FilterType::Lanczos3)
1097            }
1098            ImageOp::Contrast { amount } => img.adjust_contrast(amount),
1099            ImageOp::Brightness { amount } => img.brighten(amount as i32),
1100        };
1101    }
1102    img
1103}
1104
1105#[cfg(test)]
1106mod tests {
1107    use super::*;
1108
1109    fn cfg() -> ProcessConfig {
1110        ProcessConfig::default()
1111    }
1112
1113    #[test]
1114    fn decode_round_trips_small_image() {
1115        let img = DynamicImage::ImageRgb8(ImageBuffer::from_fn(8, 8, |_, _| {
1116            image::Rgb([10u8, 20, 30])
1117        }));
1118        let b64 = encode_image_base64(&img, &cfg()).unwrap();
1119        let decoded = decode_base64_image(&b64).unwrap();
1120        assert_eq!((decoded.width(), decoded.height()), (8, 8));
1121    }
1122
1123    #[test]
1124    fn decode_rejects_oversized_dimensions() {
1125        // Valid, tiny-on-disk image whose width exceeds MAX_DIM — the decompression-bomb shape.
1126        let wide = DynamicImage::ImageRgb8(ImageBuffer::from_fn(MAX_DIM + 1, 1, |_, _| {
1127            image::Rgb([0u8, 0, 0])
1128        }));
1129        let b64 = encode_image_base64(&wide, &cfg()).unwrap();
1130        assert!(decode_base64_image(&b64).is_err());
1131    }
1132
1133    #[test]
1134    fn decode_rejects_garbage() {
1135        assert!(decode_base64_image("not valid base64 !!!").is_err());
1136    }
1137
1138    #[test]
1139    fn exact_boundary_unchanged() {
1140        let r = calculate_optimal_dimensions(1024, 512);
1141        assert_eq!((r.width, r.height), (1024, 512));
1142        assert_eq!(r.tokens_saved(), 0);
1143    }
1144
1145    #[test]
1146    fn one_pixel_over_saves_full_tile_row() {
1147        let r = calculate_optimal_dimensions(1025, 1025);
1148        assert_eq!((r.width, r.height), (1024, 1024));
1149        assert_eq!(r.tiles_before, 9);
1150        assert_eq!(r.tiles_after, 4);
1151        assert_eq!(r.tokens_saved(), 5);
1152    }
1153
1154    #[test]
1155    fn small_image_never_below_one_tile() {
1156        let r = calculate_optimal_dimensions(100, 200);
1157        assert_eq!((r.width, r.height), (512, 512));
1158    }
1159
1160    #[test]
1161    fn mid_boundary_snaps_down() {
1162        let r = calculate_optimal_dimensions(768, 512);
1163        assert_eq!(r.width, 512);
1164        assert_eq!(r.tiles_after, 1);
1165    }
1166
1167    #[test]
1168    fn custom_tile_size_256() {
1169        let r = calculate_optimal_dimensions_with(257, 512, 256);
1170        assert_eq!(r.width, 256); // 257 → snaps down to 256
1171        assert_eq!(r.tiles_before, 2 * 2); // ceil(257/256)*ceil(512/256) = 2*2
1172        assert_eq!(r.tiles_after, 1 * 2); // 256/256 * 512/256 = 1*2
1173    }
1174
1175    #[test]
1176    fn full_pipeline_reduces_tiles() {
1177        use image::{DynamicImage, Rgba, RgbaImage};
1178        let mut img = RgbaImage::from_pixel(1025, 1025, Rgba([255, 255, 255, 255]));
1179        for x in 400..600 {
1180            for y in 400..600 {
1181                img.put_pixel(x, y, Rgba([0, 0, 0, 255]));
1182            }
1183        }
1184        let result = process(
1185            DynamicImage::ImageRgba8(img),
1186            ProcessMode::Standard,
1187            0,
1188            &cfg(),
1189        );
1190        assert!(result.report.tiles_after < result.report.tiles_before);
1191    }
1192
1193    #[test]
1194    fn crop_disabled_preserves_size() {
1195        use image::{DynamicImage, Rgba, RgbaImage};
1196        let img = RgbaImage::from_pixel(1024, 1024, Rgba([255, 255, 255, 255]));
1197        let no_crop = ProcessConfig::builder().crop(false).build();
1198        let result = process(
1199            DynamicImage::ImageRgba8(img),
1200            ProcessMode::Standard,
1201            0,
1202            &no_crop,
1203        );
1204        assert_eq!(result.width, 1024);
1205    }
1206
1207    #[test]
1208    fn crop_removes_white_border() {
1209        use image::{Rgba, RgbaImage};
1210        let mut img = RgbaImage::from_pixel(100, 100, Rgba([255, 255, 255, 255]));
1211        for x in 45..55 {
1212            for y in 45..55 {
1213                img.put_pixel(x, y, Rgba([255, 0, 0, 255]));
1214            }
1215        }
1216        let cropped = crop_padding(DynamicImage::ImageRgba8(img), 15);
1217        assert!(cropped.width() < 100 && cropped.height() < 100);
1218    }
1219
1220    #[test]
1221    fn binarize_produces_only_black_white() {
1222        use image::{DynamicImage, GrayImage, Luma};
1223        let img = GrayImage::from_fn(64, 64, |x, _| Luma([if x < 32 { 50u8 } else { 200u8 }]));
1224        let result = binarize(DynamicImage::ImageLuma8(img)).to_luma8();
1225        for p in result.pixels() {
1226            assert!(p.0[0] == 0 || p.0[0] == 255);
1227        }
1228    }
1229
1230    #[test]
1231    fn ssim_identical_images_is_one() {
1232        use image::{DynamicImage, Rgba, RgbaImage};
1233        let img =
1234            DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([128, 128, 128, 255])));
1235        let s = ssim(&img, &img);
1236        assert!((s - 1.0).abs() < 1e-9);
1237    }
1238
1239    #[test]
1240    fn ssim_very_different_images_is_low() {
1241        use image::{DynamicImage, Rgba, RgbaImage};
1242        let black = DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([0, 0, 0, 255])));
1243        let white =
1244            DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([255, 255, 255, 255])));
1245        let s = ssim(&black, &white);
1246        assert!(s < 0.1, "expected low SSIM, got {s}");
1247    }
1248
1249    #[test]
1250    fn saliency_crop_tightens_around_high_energy_region() {
1251        use image::{DynamicImage, Rgba, RgbaImage};
1252        // Uniform white field with a 200x200 textured block in the middle of a 1000x1000 image.
1253        let mut img = RgbaImage::from_pixel(1000, 1000, Rgba([255, 255, 255, 255]));
1254        for x in 400..600 {
1255            for y in 400..600 {
1256                // checker pattern to generate edge energy
1257                let v = if (x + y) % 2 == 0 { 0 } else { 255 };
1258                img.put_pixel(x, y, Rgba([v, v, v, 255]));
1259            }
1260        }
1261        let dyn_img = DynamicImage::ImageRgba8(img);
1262        let cropped = saliency_crop(&dyn_img, 8);
1263        assert!(cropped.width() < 1000);
1264        assert!(cropped.height() < 1000);
1265        // expect to land near the 200×200 block plus margin
1266        assert!(cropped.width() < 400);
1267        assert!(cropped.height() < 400);
1268    }
1269
1270    #[test]
1271    fn auto_quality_returns_quality_in_range() {
1272        use image::{DynamicImage, Rgba, RgbaImage};
1273        let mut img = RgbaImage::from_pixel(256, 256, Rgba([100, 100, 100, 255]));
1274        for x in 0..256 {
1275            for y in 0..256 {
1276                img.put_pixel(x, y, Rgba([(x % 256) as u8, (y % 256) as u8, 128, 255]));
1277            }
1278        }
1279        let dyn_img = DynamicImage::ImageRgba8(img);
1280        let cfg = ProcessConfig::default();
1281        let (bytes, q) = encode_with_auto_quality(&dyn_img, &cfg, 0.95, 40, 95).expect("ok");
1282        assert!((40..=95).contains(&q));
1283        assert!(!bytes.is_empty());
1284    }
1285
1286    #[test]
1287    fn high_bg_tolerance_crops_more() {
1288        use image::{DynamicImage, Rgba, RgbaImage};
1289        // Corners: pure white [255,255,255]. Border: off-white [240,240,240]. Center: black.
1290        // diff = 15. strict(5): 15 > 5 → border NOT bg → no crop.
1291        // loose(20): 15 ≤ 20 → border IS bg → crops.
1292        let mut img = RgbaImage::from_pixel(100, 100, Rgba([240, 240, 240, 255]));
1293        for corner in [(0u32, 0u32), (99, 0), (0, 99), (99, 99)] {
1294            img.put_pixel(corner.0, corner.1, Rgba([255, 255, 255, 255]));
1295        }
1296        for x in 45..55 {
1297            for y in 45..55 {
1298                img.put_pixel(x, y, Rgba([0, 0, 0, 255]));
1299            }
1300        }
1301        let strict = crop_padding(DynamicImage::ImageRgba8(img.clone()), 5);
1302        let loose = crop_padding(DynamicImage::ImageRgba8(img), 20);
1303        assert!(loose.width() < strict.width());
1304    }
1305}
1306// ── Persistence & Analytics ───────────────────────────────────────────────────
1307
1308#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)]
1309pub struct OptimizationReport {
1310    pub timestamp: String,
1311    pub model: String,
1312    pub original_tokens: u32,
1313    pub optimized_tokens: u32,
1314    pub original_bytes: u64,
1315    pub optimized_bytes: u64,
1316    pub mode: String,
1317}
1318
1319#[derive(Debug, serde::Serialize, serde::Deserialize)]
1320pub struct SqueezerStats {
1321    pub total_optimizations: u64,
1322    pub total_original_tokens: u64,
1323    pub total_optimized_tokens: u64,
1324    pub total_original_bytes: u64,
1325    pub total_optimized_bytes: u64,
1326    pub history: Vec<OptimizationReport>,
1327}
1328
1329impl SqueezerStats {
1330    pub fn total_token_savings(&self) -> u64 {
1331        self.total_original_tokens
1332            .saturating_sub(self.total_optimized_tokens)
1333    }
1334
1335    pub fn total_byte_savings(&self) -> u64 {
1336        self.total_original_bytes
1337            .saturating_sub(self.total_optimized_bytes)
1338    }
1339
1340    pub fn estimated_usd_saved(&self) -> f64 {
1341        // Blended average: $2.50 per 1M tokens (Claude/GPT-4o blend)
1342        (self.total_token_savings() as f64 / 1_000_000.0) * 2.50
1343    }
1344}
1345
1346pub struct Persistence;
1347
1348impl Persistence {
1349    fn get_db_path() -> PathBuf {
1350        let mut path = dirs::home_dir().unwrap_or_else(|| PathBuf::from("."));
1351        path.push(".vision-squeezer");
1352        let _ = std::fs::create_dir_all(&path);
1353        path.push("stats.db");
1354        path
1355    }
1356
1357    pub fn init_db() -> Result<(), String> {
1358        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1359        conn.execute(
1360            "CREATE TABLE IF NOT EXISTS optimizations (
1361                id INTEGER PRIMARY KEY AUTOINCREMENT,
1362                timestamp TEXT NOT NULL,
1363                model TEXT NOT NULL,
1364                original_tokens INTEGER NOT NULL,
1365                optimized_tokens INTEGER NOT NULL,
1366                original_bytes INTEGER NOT NULL,
1367                optimized_bytes INTEGER NOT NULL,
1368                mode TEXT NOT NULL
1369            )",
1370            [],
1371        )
1372        .map_err(|e| e.to_string())?;
1373        Ok(())
1374    }
1375
1376    pub fn log_optimization(
1377        model: &str,
1378        orig_tokens: u32,
1379        opt_tokens: u32,
1380        orig_bytes: u64,
1381        opt_bytes: u64,
1382        mode: &str,
1383    ) -> Result<(), String> {
1384        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1385        conn.execute(
1386            "INSERT INTO optimizations (timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode)
1387             VALUES (?, ?, ?, ?, ?, ?, ?)",
1388            params![
1389                Utc::now().to_rfc3339(),
1390                model,
1391                orig_tokens,
1392                opt_tokens,
1393                orig_bytes as i64,
1394                opt_bytes as i64,
1395                mode,
1396            ],
1397        ).map_err(|e| e.to_string())?;
1398        Ok(())
1399    }
1400
1401    pub fn get_stats() -> Result<SqueezerStats, String> {
1402        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1403
1404        let mut stmt = conn
1405            .prepare(
1406                "SELECT 
1407                COUNT(*), 
1408                SUM(original_tokens), 
1409                SUM(optimized_tokens), 
1410                SUM(original_bytes), 
1411                SUM(optimized_bytes) 
1412             FROM optimizations",
1413            )
1414            .map_err(|e| e.to_string())?;
1415
1416        let (count, orig_t, opt_t, orig_b, opt_b) = stmt
1417            .query_row([], |row| {
1418                Ok((
1419                    row.get::<_, Option<i64>>(0)?.unwrap_or(0) as u64,
1420                    row.get::<_, Option<i64>>(1)?.unwrap_or(0) as u64,
1421                    row.get::<_, Option<i64>>(2)?.unwrap_or(0) as u64,
1422                    row.get::<_, Option<i64>>(3)?.unwrap_or(0) as u64,
1423                    row.get::<_, Option<i64>>(4)?.unwrap_or(0) as u64,
1424                ))
1425            })
1426            .map_err(|e| e.to_string())?;
1427
1428        let mut stmt = conn.prepare(
1429            "SELECT timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode 
1430             FROM optimizations ORDER BY timestamp DESC LIMIT 50"
1431        ).map_err(|e| e.to_string())?;
1432
1433        let history = stmt
1434            .query_map([], |row| {
1435                Ok(OptimizationReport {
1436                    timestamp: row.get(0)?,
1437                    model: row.get(1)?,
1438                    original_tokens: row.get(2)?,
1439                    optimized_tokens: row.get(3)?,
1440                    original_bytes: row.get::<_, i64>(4)? as u64,
1441                    optimized_bytes: row.get::<_, i64>(5)? as u64,
1442                    mode: row.get(6)?,
1443                })
1444            })
1445            .map_err(|e| e.to_string())?
1446            .collect::<Result<Vec<_>, _>>()
1447            .map_err(|e| e.to_string())?;
1448
1449        Ok(SqueezerStats {
1450            total_optimizations: count,
1451            total_original_tokens: orig_t,
1452            total_optimized_tokens: opt_t,
1453            total_original_bytes: orig_b,
1454            total_optimized_bytes: opt_b,
1455            history,
1456        })
1457    }
1458
1459    pub fn get_all_history() -> Result<Vec<OptimizationReport>, String> {
1460        let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
1461        let mut stmt = conn.prepare(
1462            "SELECT timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode
1463             FROM optimizations ORDER BY timestamp ASC"
1464        ).map_err(|e| e.to_string())?;
1465
1466        stmt.query_map([], |row| {
1467            Ok(OptimizationReport {
1468                timestamp: row.get(0)?,
1469                model: row.get(1)?,
1470                original_tokens: row.get(2)?,
1471                optimized_tokens: row.get(3)?,
1472                original_bytes: row.get::<_, i64>(4)? as u64,
1473                optimized_bytes: row.get::<_, i64>(5)? as u64,
1474                mode: row.get(6)?,
1475            })
1476        })
1477        .map_err(|e| e.to_string())?
1478        .collect::<Result<Vec<_>, _>>()
1479        .map_err(|e| e.to_string())
1480    }
1481}