use std::io::Cursor;
use base64::{Engine, engine::general_purpose::STANDARD as B64};
use chrono::Utc;
use image::{DynamicImage, ImageBuffer, Luma, imageops::FilterType};
use rusqlite::{Connection, params};
use std::path::PathBuf;
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub enum OutputFormat {
#[default]
Jpeg,
WebP,
Avif,
}
#[derive(Clone, Debug)]
pub struct ProcessConfig {
pub quality: u8,
pub tile_size: u32,
pub crop: bool,
pub bg_tolerance: u8,
pub output_format: OutputFormat,
pub target_model: Option<VisionModel>,
pub max_tiles: Option<u32>,
pub smart_crop: bool,
}
impl Default for ProcessConfig {
fn default() -> Self {
Self {
quality: 75,
tile_size: 512,
crop: true,
bg_tolerance: 15,
output_format: OutputFormat::Jpeg,
target_model: None,
max_tiles: None,
smart_crop: false,
}
}
}
impl ProcessConfig {
pub fn builder() -> ProcessConfigBuilder {
ProcessConfigBuilder(Self::default())
}
}
pub struct ProcessConfigBuilder(ProcessConfig);
impl ProcessConfigBuilder {
pub fn quality(mut self, q: u8) -> Self {
self.0.quality = q.clamp(1, 100);
self
}
pub fn tile_size(mut self, t: u32) -> Self {
self.0.tile_size = t.max(1);
self
}
pub fn crop(mut self, c: bool) -> Self {
self.0.crop = c;
self
}
pub fn bg_tolerance(mut self, t: u8) -> Self {
self.0.bg_tolerance = t;
self
}
pub fn output_format(mut self, f: OutputFormat) -> Self {
self.0.output_format = f;
self
}
pub fn target_model(mut self, m: VisionModel) -> Self {
self.0.target_model = Some(m);
self
}
pub fn max_tiles(mut self, m: u32) -> Self {
self.0.max_tiles = Some(m);
self
}
pub fn smart_crop(mut self, b: bool) -> Self {
self.0.smart_crop = b;
self
}
pub fn build(self) -> ProcessConfig {
self.0
}
}
#[derive(Clone, Copy, Debug)]
pub enum VisionModel {
Claude,
ClaudeStandard,
Gpt6,
Gpt4o,
Gpt5,
Gemini15,
LlamaVision,
QwenVl,
DeepseekVl,
DeepseekFlash,
KimiVision,
GenericVision,
}
impl VisionModel {
pub fn parse(value: &str) -> Option<Self> {
match value.to_ascii_lowercase().as_str() {
"claude" | "claude-high" | "claude-4.7" => Some(Self::Claude),
"claude-standard" => Some(Self::ClaudeStandard),
"openai" | "gpt6" | "gpt-6" | "gpt6-astra" | "gpt-6-astra" | "gpt5.6" | "gpt-5.6"
| "gpt5.5" | "gpt-5.5" => Some(Self::Gpt6),
"gpt4o" | "gpt-4o" => Some(Self::Gpt4o),
"gpt5" | "gpt-5" | "gpt5.1" | "gpt-5.1" => Some(Self::Gpt5),
"gemini" | "gemini-3" | "gemini-3.8" => Some(Self::Gemini15),
"llama" | "llama-vision" => Some(Self::LlamaVision),
"qwen" | "qwen-vl" | "qwen3-vl" => Some(Self::QwenVl),
"deepseek-local" | "deepseek-vl" | "deepseek-vl2" => Some(Self::DeepseekVl),
"deepseek" | "deepseek-flash" | "deepseek-v4-flash-vision-exp" => {
Some(Self::DeepseekFlash)
}
"kimi" | "kimi-vision" | "kimi-k2.5" | "kimi-k2.6" | "kimi-k3" => {
Some(Self::KimiVision)
}
"glm"
| "glm-4v"
| "glm-4.5v"
| "glm-5.3-flash"
| "mistral"
| "pixtral"
| "pixtral-large"
| "pixtral-12b"
| "gemma"
| "gemma-3"
| "gemma-4"
| "gemma-4-31b"
| "internvl"
| "internvl2"
| "internvl2.5"
| "internvl3"
| "minicpm"
| "minicpm-v"
| "minicpm-o"
| "molmo"
| "molmo2"
| "aya"
| "aya-vision"
| "phi4"
| "phi-4"
| "phi-4-multimodal"
| "granite"
| "granite-vision"
| "llava"
| "llava-onevision"
| "llava-next"
| "falcon"
| "falcon-vision"
| "falcon-ocr"
| "minimax"
| "minimax-vl"
| "minimax-m3"
| "step"
| "step-3.7"
| "step-3.7-flash"
| "ling"
| "ling-vision"
| "ling-3.0-flash-vl"
| "voyage"
| "voyage-multimodal"
| "voyage-multimodal-3.5" => Some(Self::GenericVision),
_ => None,
}
}
pub fn display_name(self) -> &'static str {
match self {
Self::Claude => "Claude 4.7+",
Self::ClaudeStandard => "Claude (standard)",
Self::Gpt6 => "GPT-6 / GPT-5.6",
Self::Gpt4o => "GPT-4o",
Self::Gpt5 => "GPT-5 / 5.1 (legacy)",
Self::Gemini15 => "Gemini 3",
Self::LlamaVision => "Llama Vision",
Self::QwenVl => "Qwen-VL",
Self::DeepseekVl => "DeepSeek-VL",
Self::DeepseekFlash => "DeepSeek Flash",
Self::KimiVision => "Kimi Vision",
Self::GenericVision => "Generic vision (advisory)",
}
}
}
#[derive(Debug)]
pub struct TokenEstimate {
pub model: VisionModel,
pub tokens: u32,
pub tiles: u32,
}
pub fn estimate_tokens(width: u32, height: u32, model: VisionModel) -> TokenEstimate {
match model {
VisionModel::Claude => {
let (w, h) = fit_within_patch_budget(width, height, 2576, 4784, 28);
let patches = patch_count(w, h, 28);
TokenEstimate {
model,
tiles: patches,
tokens: patches,
}
}
VisionModel::ClaudeStandard => {
let (w, h) = fit_within_patch_budget(width, height, 1568, 1568, 28);
let patches = patch_count(w, h, 28);
TokenEstimate {
model,
tiles: patches,
tokens: patches,
}
}
VisionModel::Gpt6 => {
let (w, h) = fit_within_patch_budget(width, height, 2048, 2500, 32);
let patches = patch_count(w, h, 32);
TokenEstimate {
model,
tiles: patches,
tokens: (patches * 12).div_ceil(10),
}
}
VisionModel::Gpt4o => {
let (mut w, mut h) = fit_within(width, height, 2048);
let short_side = w.min(h);
if short_side > 768 {
let scale = 768.0 / short_side as f64;
w = (w as f64 * scale).round() as u32;
h = (h as f64 * scale).round() as u32;
}
let tiles = tile_count(w, 512) * tile_count(h, 512);
TokenEstimate {
model,
tiles,
tokens: 85 + tiles * 170,
}
}
VisionModel::Gpt5 => {
let (mut w, mut h) = fit_within(width, height, 2048);
let short_side = w.min(h);
if short_side > 768 {
let scale = 768.0 / short_side as f64;
w = (w as f64 * scale).round() as u32;
h = (h as f64 * scale).round() as u32;
}
let tiles = tile_count(w, 512) * tile_count(h, 512);
TokenEstimate {
model,
tiles,
tokens: 70 + tiles * 140,
}
}
VisionModel::Gemini15 => {
if width <= 384 && height <= 384 {
TokenEstimate {
model,
tiles: 1,
tokens: 258,
}
} else {
let tiles = tile_count(width, 768) * tile_count(height, 768);
TokenEstimate {
model,
tiles,
tokens: tiles * 258,
}
}
}
VisionModel::LlamaVision => {
let (w, h) = fit_within(width, height, 1120); let tiles = (tile_count(w, 560) * tile_count(h, 560)).clamp(1, 4);
TokenEstimate {
model,
tiles,
tokens: tiles * 1601,
}
}
VisionModel::QwenVl => {
let (w, h) = fit_within_pixels(width, height, u32::MAX, 16_384 * 28 * 28);
let patches = tile_count(w, 28) * tile_count(h, 28);
TokenEstimate {
model,
tiles: patches,
tokens: patches.clamp(4, 16_384),
}
}
VisionModel::DeepseekVl => {
const H: u32 = 14;
let (nw, nh) = if width <= 384 && height <= 384 {
(1, 1)
} else {
let mut nw = tile_count(width, 384).max(1);
let mut nh = tile_count(height, 384).max(1);
while nw * nh > 9 {
if nw >= nh {
nw -= 1;
} else {
nh -= 1;
}
}
(nw, nh)
};
let global = H * (H + 1) + 1; let local = (nh * H) * (nw * H + 1);
TokenEstimate {
model,
tiles: nw * nh + 1, tokens: global + local,
}
}
VisionModel::DeepseekFlash => TokenEstimate {
model,
tiles: 1,
tokens: 384,
},
VisionModel::KimiVision => {
let (w, h) = fit_within(width, height, 4096);
let patches = patch_count(w, h, 28);
TokenEstimate {
model,
tiles: patches,
tokens: patches,
}
}
VisionModel::GenericVision => {
let (w, h) = fit_within(width, height, 2048);
let patches = patch_count(w, h, 28);
TokenEstimate {
model,
tiles: patches,
tokens: patches,
}
}
}
}
pub fn fit_within(width: u32, height: u32, max_side: u32) -> (u32, u32) {
if width <= max_side && height <= max_side {
return (width, height);
}
let scale = max_side as f64 / width.max(height) as f64;
(
(width as f64 * scale) as u32,
(height as f64 * scale) as u32,
)
}
pub fn fit_within_pixels(width: u32, height: u32, max_side: u32, max_pixels: u64) -> (u32, u32) {
let (mut w, mut h) = fit_within(width, height, max_side);
let total = w as u64 * h as u64;
if total > max_pixels {
let scale = (max_pixels as f64 / total as f64).sqrt();
w = (w as f64 * scale) as u32;
h = (h as f64 * scale) as u32;
}
(w.max(1), h.max(1))
}
fn patch_count(width: u32, height: u32, patch: u32) -> u32 {
width.max(1).div_ceil(patch) * height.max(1).div_ceil(patch)
}
fn fit_within_patch_budget(
width: u32,
height: u32,
max_side: u32,
max_patches: u32,
patch: u32,
) -> (u32, u32) {
let (mut w, mut h) = fit_within(width.max(1), height.max(1), max_side);
let patches = patch_count(w, h, patch);
if patches > max_patches {
let scale = (max_patches as f64 / patches as f64).sqrt();
w = ((w as f64 * scale) as u32 / patch * patch).max(patch);
h = ((h as f64 * scale) as u32 / patch * patch).max(patch);
}
(w, h)
}
pub fn optimal_send_dimensions(width: u32, height: u32, model: VisionModel) -> (u32, u32) {
match model {
VisionModel::Claude => optimal_for_patch_model(width, height, 2576, 4784, 28),
VisionModel::ClaudeStandard => optimal_for_patch_model(width, height, 1568, 1568, 28),
VisionModel::Gpt6 => optimal_for_patch_model(width, height, 2048, 2500, 32),
VisionModel::Gpt4o => optimal_for_prescaling_model(width, height, 2048, 512),
VisionModel::Gpt5 => optimal_for_prescaling_model(width, height, 2048, 512),
VisionModel::Gemini15 => {
if width <= 384 && height <= 384 {
(width, height)
} else {
optimal_for_prescaling_model(width, height, 4096, 768)
}
}
VisionModel::LlamaVision => {
optimal_for_prescaling_model(width, height, 1120, 560)
}
VisionModel::QwenVl => {
let (fw, fh) = fit_within_pixels(width, height, u32::MAX, 16_384 * 28 * 28);
(
snap_to_tile_boundary(fw, 28).max(28),
snap_to_tile_boundary(fh, 28).max(28),
)
}
VisionModel::DeepseekVl => {
if width <= 384 && height <= 384 {
(width, height)
} else {
optimal_for_prescaling_model(width, height, 1152, 384)
}
}
VisionModel::DeepseekFlash => fit_within(width, height, 2048),
VisionModel::KimiVision => {
let (w, h) = fit_within(width, height, 4096);
(snap_to_tile_boundary(w, 28), snap_to_tile_boundary(h, 28))
}
VisionModel::GenericVision => {
let (w, h) = fit_within(width, height, 2048);
(snap_to_tile_boundary(w, 28), snap_to_tile_boundary(h, 28))
}
}
}
fn optimal_for_patch_model(
width: u32,
height: u32,
max_side: u32,
max_patches: u32,
patch: u32,
) -> (u32, u32) {
let (w, h) = fit_within_patch_budget(width, height, max_side, max_patches, patch);
(
snap_to_tile_boundary(w, patch),
snap_to_tile_boundary(h, patch),
)
}
fn optimal_for_prescaling_model(width: u32, height: u32, max_side: u32, tile: u32) -> (u32, u32) {
let (fw, fh) = fit_within(width, height, max_side);
let target_w = snap_to_tile_boundary(fw, tile).max(tile);
let target_h = snap_to_tile_boundary(fh, tile).max(tile);
if width > max_side || height > max_side {
let scale = width.max(height) as f64 / max_side as f64;
let opt_w = (target_w as f64 * scale).round() as u32;
let opt_h = (target_h as f64 * scale).round() as u32;
return (opt_w.max(1), opt_h.max(1));
}
(target_w, target_h)
}
pub struct TokenSavingsTable {
pub claude_before: TokenEstimate,
pub claude_after: TokenEstimate,
pub gpt6_before: TokenEstimate,
pub gpt6_after: TokenEstimate,
pub gpt4o_before: TokenEstimate,
pub gpt4o_after: TokenEstimate,
pub gpt5_before: TokenEstimate,
pub gpt5_after: TokenEstimate,
pub gemini_before: TokenEstimate,
pub gemini_after: TokenEstimate,
}
pub fn token_savings_table(orig_w: u32, orig_h: u32, opt_w: u32, opt_h: u32) -> TokenSavingsTable {
TokenSavingsTable {
claude_before: estimate_tokens(orig_w, orig_h, VisionModel::Claude),
claude_after: estimate_tokens(opt_w, opt_h, VisionModel::Claude),
gpt6_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt6),
gpt6_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt6),
gpt4o_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt4o),
gpt4o_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt4o),
gpt5_before: estimate_tokens(orig_w, orig_h, VisionModel::Gpt5),
gpt5_after: estimate_tokens(opt_w, opt_h, VisionModel::Gpt5),
gemini_before: estimate_tokens(orig_w, orig_h, VisionModel::Gemini15),
gemini_after: estimate_tokens(opt_w, opt_h, VisionModel::Gemini15),
}
}
impl TokenSavingsTable {
pub fn print(&self) {
println!(
"{:<12} {:>8} {:>8} {:>10}",
"Model", "Before", "After", "Saved"
);
println!("{}", "-".repeat(42));
self.print_row("Claude 4.7+", &self.claude_before, &self.claude_after);
self.print_row("GPT-6", &self.gpt6_before, &self.gpt6_after);
self.print_row("GPT-4o", &self.gpt4o_before, &self.gpt4o_after);
self.print_row("GPT-5", &self.gpt5_before, &self.gpt5_after);
self.print_row("Gemini", &self.gemini_before, &self.gemini_after);
}
fn print_row(&self, name: &str, before: &TokenEstimate, after: &TokenEstimate) {
let saved = before.tokens.saturating_sub(after.tokens);
let pct = if before.tokens > 0 {
saved as f64 / before.tokens as f64 * 100.0
} else {
0.0
};
println!(
"{:<12} {:>8} {:>8} {:>8} ({:.1}%)",
name, before.tokens, after.tokens, saved, pct
);
}
}
pub struct DimensionResult {
pub width: u32,
pub height: u32,
pub tiles_before: u32,
pub tiles_after: u32,
}
impl DimensionResult {
pub fn tokens_saved(&self) -> u32 {
self.tiles_before.saturating_sub(self.tiles_after)
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)]
pub enum ProcessMode {
Standard,
Ocr,
#[default]
Auto,
}
pub fn detect_ocr_mode(img: &DynamicImage) -> bool {
let rgb = img.to_rgb8();
let mut colorful_count = 0;
let mut total_count = 0;
for (x, y, p) in rgb.enumerate_pixels() {
if x % 4 == 0 && y % 4 == 0 {
total_count += 1;
let min = p[0].min(p[1]).min(p[2]);
let max = p[0].max(p[1]).max(p[2]);
if max.saturating_sub(min) > 25 {
colorful_count += 1;
}
}
}
let colorful_ratio = colorful_count as f64 / total_count.max(1) as f64;
colorful_ratio < 0.1 }
pub struct SavingsReport {
pub tiles_before: u32,
pub tiles_after: u32,
pub tiles_saved: u32,
pub bytes_before: Option<u64>,
pub bytes_after: Option<u64>,
}
impl SavingsReport {
pub fn size_reduction_pct(&self) -> Option<f64> {
match (self.bytes_before, self.bytes_after) {
(Some(b), Some(a)) if b > 0 => Some((1.0 - a as f64 / b as f64) * 100.0),
_ => None,
}
}
pub fn token_reduction_pct(&self) -> f64 {
if self.tiles_before == 0 {
return 0.0;
}
self.tiles_saved as f64 / self.tiles_before as f64 * 100.0
}
}
pub struct ProcessResult {
pub image: DynamicImage,
pub width: u32,
pub height: u32,
pub report: SavingsReport,
}
impl ProcessResult {
pub fn tokens_saved(&self) -> u32 {
self.report.tiles_saved
}
}
pub fn process(
img: DynamicImage,
mode: ProcessMode,
input_bytes: u64,
cfg: &ProcessConfig,
) -> ProcessResult {
let (orig_w, orig_h) = (img.width(), img.height());
let tiles_before = match cfg.target_model {
Some(model) => estimate_tokens(orig_w, orig_h, model).tiles,
None => tile_count(orig_w, cfg.tile_size) * tile_count(orig_h, cfg.tile_size),
};
let after_crop = if cfg.crop {
if cfg.smart_crop {
saliency_crop(&img, 16)
} else {
crop_padding(img, cfg.bg_tolerance)
}
} else {
img
};
let (mut opt_w, mut opt_h) = match cfg.target_model {
Some(model) => optimal_send_dimensions(after_crop.width(), after_crop.height(), model),
None => {
let d = calculate_optimal_dimensions_with(
after_crop.width(),
after_crop.height(),
cfg.tile_size,
);
(d.width, d.height)
}
};
if let Some(max_t) = cfg.max_tiles {
let (nw, nh) = enforce_max_tiles(opt_w, opt_h, max_t, cfg.tile_size, cfg.target_model);
opt_w = nw;
opt_h = nh;
}
let tiles_after = match cfg.target_model {
Some(model) => {
let est = estimate_tokens(opt_w, opt_h, model);
est.tiles
}
None => tile_count(opt_w, cfg.tile_size) * tile_count(opt_h, cfg.tile_size),
};
let resized = after_crop.resize_exact(opt_w, opt_h, FilterType::Lanczos3);
let actual_mode = match mode {
ProcessMode::Auto => {
if detect_ocr_mode(&after_crop) {
ProcessMode::Ocr
} else {
ProcessMode::Standard
}
}
m => m,
};
let final_image = match actual_mode {
ProcessMode::Standard | ProcessMode::Auto => resized,
ProcessMode::Ocr => binarize(resized),
};
ProcessResult {
width: final_image.width(),
height: final_image.height(),
image: final_image,
report: SavingsReport {
tiles_before,
tiles_after,
tiles_saved: tiles_before.saturating_sub(tiles_after),
bytes_before: if input_bytes > 0 {
Some(input_bytes)
} else {
None
},
bytes_after: None,
},
}
}
fn enforce_max_tiles(
mut width: u32,
mut height: u32,
max_tiles: u32,
default_tile_size: u32,
model: Option<VisionModel>,
) -> (u32, u32) {
if max_tiles == 0 {
return (width, height);
}
let mut scale = 1.0;
let orig_w = width;
let orig_h = height;
loop {
let (snapped_w, snapped_h) = match model {
Some(m) => optimal_send_dimensions(width, height, m),
None => {
let d = calculate_optimal_dimensions_with(width, height, default_tile_size);
(d.width, d.height)
}
};
let tiles = match model {
Some(m) => estimate_tokens(snapped_w, snapped_h, m).tiles,
None => {
tile_count(snapped_w, default_tile_size) * tile_count(snapped_h, default_tile_size)
}
};
if tiles <= max_tiles || scale < 0.1 {
return (snapped_w, snapped_h);
}
scale *= 0.95;
width = (orig_w as f64 * scale) as u32;
height = (orig_h as f64 * scale) as u32;
width = width.max(1);
height = height.max(1);
}
}
pub fn calculate_optimal_dimensions(width: u32, height: u32) -> DimensionResult {
calculate_optimal_dimensions_with(width, height, 512)
}
pub fn calculate_optimal_dimensions_with(
width: u32,
height: u32,
tile_size: u32,
) -> DimensionResult {
let opt_w = snap_to_tile_boundary(width, tile_size);
let opt_h = snap_to_tile_boundary(height, tile_size);
DimensionResult {
width: opt_w,
height: opt_h,
tiles_before: tile_count(width, tile_size) * tile_count(height, tile_size),
tiles_after: tile_count(opt_w, tile_size) * tile_count(opt_h, tile_size),
}
}
fn tile_count(dim: u32, tile_size: u32) -> u32 {
dim.div_ceil(tile_size)
}
fn snap_to_tile_boundary(dim: u32, tile_size: u32) -> u32 {
if dim.is_multiple_of(tile_size) {
return dim;
}
((dim / tile_size) * tile_size).max(tile_size)
}
pub fn crop_padding(img: DynamicImage, bg_tolerance: u8) -> DynamicImage {
let rgba = img.to_rgba8();
let (w, h) = rgba.dimensions();
let corners = [
*rgba.get_pixel(0, 0),
*rgba.get_pixel(w - 1, 0),
*rgba.get_pixel(0, h - 1),
*rgba.get_pixel(w - 1, h - 1),
];
let bg = corners[0];
let top = first_non_bg_row(&rgba, bg, bg_tolerance, true);
let bottom = first_non_bg_row(&rgba, bg, bg_tolerance, false);
let left = first_non_bg_col(&rgba, bg, bg_tolerance, true);
let right = first_non_bg_col(&rgba, bg, bg_tolerance, false);
if top >= bottom || left >= right {
return DynamicImage::ImageRgba8(rgba);
}
DynamicImage::ImageRgba8(
image::imageops::crop_imm(&rgba, left, top, right - left, bottom - top).to_image(),
)
}
fn is_bg(pixel: image::Rgba<u8>, bg: image::Rgba<u8>, tolerance: u8) -> bool {
pixel.0[3] < 10
|| pixel.0[..3]
.iter()
.zip(bg.0[..3].iter())
.all(|(&a, &b)| a.abs_diff(b) <= tolerance)
}
fn first_non_bg_row(img: &image::RgbaImage, bg: image::Rgba<u8>, tol: u8, from_top: bool) -> u32 {
let (w, h) = img.dimensions();
let rows: Box<dyn Iterator<Item = u32>> = if from_top {
Box::new(0..h)
} else {
Box::new((0..h).rev())
};
for y in rows {
if (0..w).any(|x| !is_bg(*img.get_pixel(x, y), bg, tol)) {
return y;
}
}
0
}
fn first_non_bg_col(img: &image::RgbaImage, bg: image::Rgba<u8>, tol: u8, from_left: bool) -> u32 {
let (w, h) = img.dimensions();
let cols: Box<dyn Iterator<Item = u32>> = if from_left {
Box::new(0..w)
} else {
Box::new((0..w).rev())
};
for x in cols {
if (0..h).any(|y| !is_bg(*img.get_pixel(x, y), bg, tol)) {
return x;
}
}
0
}
pub fn saliency_crop(img: &DynamicImage, margin: u32) -> DynamicImage {
let gray = img.to_luma8();
let (w, h) = gray.dimensions();
if w < 3 || h < 3 {
return img.clone();
}
let mut energy = vec![0u32; (w * h) as usize];
let mut total: u64 = 0;
for y in 1..h - 1 {
for x in 1..w - 1 {
let l = gray.get_pixel(x - 1, y).0[0] as i32;
let r = gray.get_pixel(x + 1, y).0[0] as i32;
let t = gray.get_pixel(x, y - 1).0[0] as i32;
let b = gray.get_pixel(x, y + 1).0[0] as i32;
let e = ((r - l).abs() + (b - t).abs()) as u32;
energy[(y * w + x) as usize] = e;
total += e as u64;
}
}
let count = (w as u64) * (h as u64);
let mean = (total / count.max(1)) as u32;
let threshold = mean.saturating_mul(2).max(8);
let (mut min_x, mut min_y, mut max_x, mut max_y) = (w, h, 0u32, 0u32);
for y in 0..h {
for x in 0..w {
if energy[(y * w + x) as usize] > threshold {
if x < min_x {
min_x = x;
}
if y < min_y {
min_y = y;
}
if x > max_x {
max_x = x;
}
if y > max_y {
max_y = y;
}
}
}
}
if min_x >= max_x || min_y >= max_y {
return img.clone();
}
let x0 = min_x.saturating_sub(margin);
let y0 = min_y.saturating_sub(margin);
let x1 = (max_x + 1 + margin).min(w);
let y1 = (max_y + 1 + margin).min(h);
img.crop_imm(x0, y0, x1 - x0, y1 - y0)
}
pub fn ssim(a: &DynamicImage, b: &DynamicImage) -> f64 {
let (aw, ah) = (a.width(), a.height());
let (bw, bh) = (b.width(), b.height());
let (target_w, target_h) = (aw.max(bw), ah.max(bh));
let resize_if_needed = |img: &DynamicImage| -> image::GrayImage {
if img.width() == target_w && img.height() == target_h {
img.to_luma8()
} else {
img.resize_exact(target_w, target_h, FilterType::Lanczos3)
.to_luma8()
}
};
let a_luma = resize_if_needed(a);
let b_luma = resize_if_needed(b);
let n = (target_w as u64 * target_h as u64).max(1) as f64;
let (mut sum_a, mut sum_b) = (0f64, 0f64);
for (pa, pb) in a_luma.pixels().zip(b_luma.pixels()) {
sum_a += pa.0[0] as f64;
sum_b += pb.0[0] as f64;
}
let mean_a = sum_a / n;
let mean_b = sum_b / n;
let (mut var_a, mut var_b, mut cov) = (0f64, 0f64, 0f64);
for (pa, pb) in a_luma.pixels().zip(b_luma.pixels()) {
let da = pa.0[0] as f64 - mean_a;
let db = pb.0[0] as f64 - mean_b;
var_a += da * da;
var_b += db * db;
cov += da * db;
}
var_a /= n;
var_b /= n;
cov /= n;
let c1 = (0.01f64 * 255.0).powi(2);
let c2 = (0.03f64 * 255.0).powi(2);
let num = (2.0 * mean_a * mean_b + c1) * (2.0 * cov + c2);
let den = (mean_a.powi(2) + mean_b.powi(2) + c1) * (var_a + var_b + c2);
if den.abs() < f64::EPSILON {
1.0
} else {
num / den
}
}
pub fn encode_with_auto_quality(
original: &DynamicImage,
cfg: &ProcessConfig,
target_ssim: f64,
min_q: u8,
max_q: u8,
) -> Result<(Vec<u8>, u8), String> {
let mut lo = min_q.max(1);
let mut hi = max_q.min(100).max(lo + 1);
let mut best: Option<(Vec<u8>, u8)> = None;
while hi.saturating_sub(lo) > 2 {
let mid = lo + (hi - lo) / 2;
let trial = ProcessConfig {
quality: mid,
..cfg.clone()
};
let bytes = encode_to_bytes(original, &trial)?;
let decoded = image::load_from_memory(&bytes).map_err(|e| e.to_string())?;
let score = ssim(original, &decoded);
if score >= target_ssim {
best = Some((bytes, mid));
hi = mid;
} else {
lo = mid;
}
}
if let Some((b, q)) = best {
Ok((b, q))
} else {
let trial = ProcessConfig {
quality: hi,
..cfg.clone()
};
let bytes = encode_to_bytes(original, &trial)?;
Ok((bytes, hi))
}
}
pub fn binarize(img: DynamicImage) -> DynamicImage {
let gray = img.to_luma8();
let (w, h) = gray.dimensions();
let threshold = otsu_threshold(&gray);
let binary: ImageBuffer<Luma<u8>, Vec<u8>> = ImageBuffer::from_fn(w, h, |x, y| {
let p = gray.get_pixel(x, y).0[0];
Luma([if p < threshold { 0u8 } else { 255u8 }])
});
DynamicImage::ImageLuma8(binary)
}
fn otsu_threshold(img: &image::GrayImage) -> u8 {
let mut histogram = [0u32; 256];
for p in img.pixels() {
histogram[p.0[0] as usize] += 1;
}
let total = img.width() * img.height();
let (mut sum, mut sum_bg, mut weight_bg) = (0f64, 0f64, 0f64);
for (i, &h) in histogram.iter().enumerate() {
sum += i as f64 * h as f64;
}
let (mut best_thresh, mut best_var) = (0u8, 0f64);
for (t, &h) in histogram.iter().enumerate() {
weight_bg += h as f64;
if weight_bg == 0.0 {
continue;
}
let weight_fg = total as f64 - weight_bg;
if weight_fg == 0.0 {
break;
}
sum_bg += t as f64 * h as f64;
let mean_bg = sum_bg / weight_bg;
let mean_fg = (sum - sum_bg) / weight_fg;
let var = weight_bg * weight_fg * (mean_bg - mean_fg).powi(2);
if var > best_var {
best_var = var;
best_thresh = t as u8;
}
}
best_thresh
}
const MAX_B64_LEN: usize = 64 * 1024 * 1024; const MAX_PIXELS: u64 = 100_000_000; const MAX_DIM: u32 = 16_384;
pub fn decode_base64_image(input: &str) -> Result<DynamicImage, String> {
let data = if let Some(c) = input.find(',') {
&input[c + 1..]
} else {
input
};
let data = data.trim();
if data.len() > MAX_B64_LEN {
return Err(format!(
"image base64 exceeds {} MB limit",
MAX_B64_LEN / 1_048_576
));
}
let bytes = B64.decode(data).map_err(|e| e.to_string())?;
let mut limits = image::Limits::default();
limits.max_image_width = Some(MAX_DIM);
limits.max_image_height = Some(MAX_DIM);
limits.max_alloc = Some(MAX_PIXELS * 4);
let mut reader = image::ImageReader::new(std::io::Cursor::new(bytes))
.with_guessed_format()
.map_err(|e| e.to_string())?;
reader.limits(limits);
reader.decode().map_err(|e| e.to_string())
}
pub fn encode_image_base64(img: &DynamicImage, cfg: &ProcessConfig) -> Result<String, String> {
let bytes = encode_to_bytes(img, cfg)?;
Ok(B64.encode(bytes))
}
pub fn encode_to_bytes(img: &DynamicImage, cfg: &ProcessConfig) -> Result<Vec<u8>, String> {
match cfg.output_format {
OutputFormat::Jpeg => {
use image::codecs::jpeg::JpegEncoder;
let mut buf = Cursor::new(Vec::new());
let rgb = img.to_rgb8();
JpegEncoder::new_with_quality(&mut buf, cfg.quality)
.encode_image(&DynamicImage::ImageRgb8(rgb))
.map_err(|e| e.to_string())?;
Ok(buf.into_inner())
}
OutputFormat::WebP => {
let rgb = img.to_rgb8();
let enc = webp::Encoder::from_rgb(rgb.as_raw(), rgb.width(), rgb.height());
let mem = enc.encode(cfg.quality as f32);
Ok(mem.to_vec())
}
OutputFormat::Avif => {
use image::ImageEncoder;
use image::codecs::avif::AvifEncoder;
let mut buf = Cursor::new(Vec::new());
let rgba = img.to_rgba8();
AvifEncoder::new_with_speed_quality(&mut buf, 6, cfg.quality)
.write_image(
rgba.as_raw(),
rgba.width(),
rgba.height(),
image::ExtendedColorType::Rgba8,
)
.map_err(|e| e.to_string())?;
Ok(buf.into_inner())
}
}
}
pub struct OptimizeResult {
pub optimized_base64: String,
pub report: SavingsReport,
pub original_width: u32,
pub original_height: u32,
pub width: u32,
pub height: u32,
pub optimized_bytes: usize,
}
pub fn optimize_image(
input_base64: &str,
mode: ProcessMode,
cfg: &ProcessConfig,
) -> Result<OptimizeResult, String> {
let img = decode_base64_image(input_base64)?;
let (orig_w, orig_h) = (img.width(), img.height());
let input_bytes = {
let data = if let Some(c) = input_base64.find(',') {
&input_base64[c + 1..]
} else {
input_base64
};
B64.decode(data.trim()).map_err(|e| e.to_string())?.len() as u64
};
let mut result = process(img, mode, input_bytes, cfg);
let bytes = encode_to_bytes(&result.image, cfg)?;
let encoded = B64.encode(&bytes);
result.report.bytes_after = Some(bytes.len() as u64);
Ok(OptimizeResult {
optimized_base64: encoded,
report: result.report,
original_width: orig_w,
original_height: orig_h,
width: result.width,
height: result.height,
optimized_bytes: bytes.len(),
})
}
#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)]
#[serde(rename_all = "lowercase", tag = "op")]
pub enum ImageOp {
Crop {
x: u32,
y: u32,
width: u32,
height: u32,
},
Grayscale,
Binarize { threshold: Option<u8> },
Resize { width: u32, height: u32 },
Contrast { amount: f32 },
Brightness { amount: f32 },
}
pub fn process_with_operations(mut img: DynamicImage, ops: Vec<ImageOp>) -> DynamicImage {
for op in ops {
img = match op {
ImageOp::Crop {
x,
y,
width,
height,
} => img.crop_imm(x, y, width, height),
ImageOp::Grayscale => DynamicImage::ImageLuma8(img.to_luma8()),
ImageOp::Binarize { threshold } => {
let gray = img.to_luma8();
let thr = threshold.unwrap_or(128);
let mut binarized = ImageBuffer::new(gray.width(), gray.height());
for (x, y, p) in gray.enumerate_pixels() {
let val = if p[0] > thr { 255 } else { 0 };
binarized.put_pixel(x, y, Luma([val]));
}
DynamicImage::ImageLuma8(binarized)
}
ImageOp::Resize { width, height } => {
img.resize_exact(width, height, FilterType::Lanczos3)
}
ImageOp::Contrast { amount } => img.adjust_contrast(amount),
ImageOp::Brightness { amount } => img.brighten(amount as i32),
};
}
img
}
#[cfg(test)]
mod tests {
use super::*;
fn cfg() -> ProcessConfig {
ProcessConfig::default()
}
#[test]
fn decode_round_trips_small_image() {
let img = DynamicImage::ImageRgb8(ImageBuffer::from_fn(8, 8, |_, _| {
image::Rgb([10u8, 20, 30])
}));
let b64 = encode_image_base64(&img, &cfg()).unwrap();
let decoded = decode_base64_image(&b64).unwrap();
assert_eq!((decoded.width(), decoded.height()), (8, 8));
}
#[test]
fn decode_rejects_oversized_dimensions() {
let wide = DynamicImage::ImageRgb8(ImageBuffer::from_fn(MAX_DIM + 1, 1, |_, _| {
image::Rgb([0u8, 0, 0])
}));
let b64 = encode_image_base64(&wide, &cfg()).unwrap();
assert!(decode_base64_image(&b64).is_err());
}
#[test]
fn decode_rejects_garbage() {
assert!(decode_base64_image("not valid base64 !!!").is_err());
}
#[test]
fn exact_boundary_unchanged() {
let r = calculate_optimal_dimensions(1024, 512);
assert_eq!((r.width, r.height), (1024, 512));
assert_eq!(r.tokens_saved(), 0);
}
#[test]
fn one_pixel_over_saves_full_tile_row() {
let r = calculate_optimal_dimensions(1025, 1025);
assert_eq!((r.width, r.height), (1024, 1024));
assert_eq!(r.tiles_before, 9);
assert_eq!(r.tiles_after, 4);
assert_eq!(r.tokens_saved(), 5);
}
#[test]
fn small_image_never_below_one_tile() {
let r = calculate_optimal_dimensions(100, 200);
assert_eq!((r.width, r.height), (512, 512));
}
#[test]
fn mid_boundary_snaps_down() {
let r = calculate_optimal_dimensions(768, 512);
assert_eq!(r.width, 512);
assert_eq!(r.tiles_after, 1);
}
#[test]
fn custom_tile_size_256() {
let r = calculate_optimal_dimensions_with(257, 512, 256);
assert_eq!(r.width, 256); assert_eq!(r.tiles_before, 2 * 2); assert_eq!(r.tiles_after, 1 * 2); }
#[test]
fn current_patch_models_match_provider_examples() {
let claude = estimate_tokens(1000, 1000, VisionModel::Claude);
assert_eq!(claude.tokens, 1296);
let gpt6 = estimate_tokens(1024, 1024, VisionModel::Gpt6);
assert_eq!(gpt6.tokens, 1229);
let large_gpt6 = estimate_tokens(2048, 2048, VisionModel::Gpt6);
assert_eq!(large_gpt6.tiles, 2500); assert_eq!(large_gpt6.tokens, 3000);
}
#[test]
fn model_aliases_and_legacy_gpt5_pricing_are_stable() {
assert!(matches!(
VisionModel::parse("gpt-5.6"),
Some(VisionModel::Gpt6)
));
assert!(matches!(
VisionModel::parse("gpt-5.5"),
Some(VisionModel::Gpt6)
));
assert!(matches!(
VisionModel::parse("claude-standard"),
Some(VisionModel::ClaudeStandard)
));
assert!(matches!(
VisionModel::parse("kimi-k2.6"),
Some(VisionModel::KimiVision)
));
assert!(matches!(
VisionModel::parse("deepseek"),
Some(VisionModel::DeepseekFlash)
));
assert!(matches!(
VisionModel::parse("pixtral"),
Some(VisionModel::GenericVision)
));
assert!(matches!(
VisionModel::parse("glm-5.3-flash"),
Some(VisionModel::GenericVision)
));
assert_eq!(
estimate_tokens(1024, 1024, VisionModel::GenericVision).tokens,
1369
);
assert_eq!(
estimate_tokens(4096, 4096, VisionModel::DeepseekFlash).tokens,
384
);
assert_eq!(estimate_tokens(1024, 1024, VisionModel::Gpt5).tokens, 630);
}
#[test]
fn full_pipeline_reduces_tiles() {
use image::{DynamicImage, Rgba, RgbaImage};
let mut img = RgbaImage::from_pixel(1025, 1025, Rgba([255, 255, 255, 255]));
for x in 400..600 {
for y in 400..600 {
img.put_pixel(x, y, Rgba([0, 0, 0, 255]));
}
}
let result = process(
DynamicImage::ImageRgba8(img),
ProcessMode::Standard,
0,
&cfg(),
);
assert!(result.report.tiles_after < result.report.tiles_before);
}
#[test]
fn crop_disabled_preserves_size() {
use image::{DynamicImage, Rgba, RgbaImage};
let img = RgbaImage::from_pixel(1024, 1024, Rgba([255, 255, 255, 255]));
let no_crop = ProcessConfig::builder().crop(false).build();
let result = process(
DynamicImage::ImageRgba8(img),
ProcessMode::Standard,
0,
&no_crop,
);
assert_eq!(result.width, 1024);
}
#[test]
fn crop_removes_white_border() {
use image::{Rgba, RgbaImage};
let mut img = RgbaImage::from_pixel(100, 100, Rgba([255, 255, 255, 255]));
for x in 45..55 {
for y in 45..55 {
img.put_pixel(x, y, Rgba([255, 0, 0, 255]));
}
}
let cropped = crop_padding(DynamicImage::ImageRgba8(img), 15);
assert!(cropped.width() < 100 && cropped.height() < 100);
}
#[test]
fn binarize_produces_only_black_white() {
use image::{DynamicImage, GrayImage, Luma};
let img = GrayImage::from_fn(64, 64, |x, _| Luma([if x < 32 { 50u8 } else { 200u8 }]));
let result = binarize(DynamicImage::ImageLuma8(img)).to_luma8();
for p in result.pixels() {
assert!(p.0[0] == 0 || p.0[0] == 255);
}
}
#[test]
fn ssim_identical_images_is_one() {
use image::{DynamicImage, Rgba, RgbaImage};
let img =
DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([128, 128, 128, 255])));
let s = ssim(&img, &img);
assert!((s - 1.0).abs() < 1e-9);
}
#[test]
fn ssim_very_different_images_is_low() {
use image::{DynamicImage, Rgba, RgbaImage};
let black = DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([0, 0, 0, 255])));
let white =
DynamicImage::ImageRgba8(RgbaImage::from_pixel(64, 64, Rgba([255, 255, 255, 255])));
let s = ssim(&black, &white);
assert!(s < 0.1, "expected low SSIM, got {s}");
}
#[test]
fn saliency_crop_tightens_around_high_energy_region() {
use image::{DynamicImage, Rgba, RgbaImage};
let mut img = RgbaImage::from_pixel(1000, 1000, Rgba([255, 255, 255, 255]));
for x in 400..600 {
for y in 400..600 {
let v = if (x + y) % 2 == 0 { 0 } else { 255 };
img.put_pixel(x, y, Rgba([v, v, v, 255]));
}
}
let dyn_img = DynamicImage::ImageRgba8(img);
let cropped = saliency_crop(&dyn_img, 8);
assert!(cropped.width() < 1000);
assert!(cropped.height() < 1000);
assert!(cropped.width() < 400);
assert!(cropped.height() < 400);
}
#[test]
fn auto_quality_returns_quality_in_range() {
use image::{DynamicImage, Rgba, RgbaImage};
let mut img = RgbaImage::from_pixel(256, 256, Rgba([100, 100, 100, 255]));
for x in 0..256 {
for y in 0..256 {
img.put_pixel(x, y, Rgba([(x % 256) as u8, (y % 256) as u8, 128, 255]));
}
}
let dyn_img = DynamicImage::ImageRgba8(img);
let cfg = ProcessConfig::default();
let (bytes, q) = encode_with_auto_quality(&dyn_img, &cfg, 0.95, 40, 95).expect("ok");
assert!((40..=95).contains(&q));
assert!(!bytes.is_empty());
}
#[test]
fn high_bg_tolerance_crops_more() {
use image::{DynamicImage, Rgba, RgbaImage};
let mut img = RgbaImage::from_pixel(100, 100, Rgba([240, 240, 240, 255]));
for corner in [(0u32, 0u32), (99, 0), (0, 99), (99, 99)] {
img.put_pixel(corner.0, corner.1, Rgba([255, 255, 255, 255]));
}
for x in 45..55 {
for y in 45..55 {
img.put_pixel(x, y, Rgba([0, 0, 0, 255]));
}
}
let strict = crop_padding(DynamicImage::ImageRgba8(img.clone()), 5);
let loose = crop_padding(DynamicImage::ImageRgba8(img), 20);
assert!(loose.width() < strict.width());
}
}
#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)]
pub struct OptimizationReport {
pub timestamp: String,
pub model: String,
pub original_tokens: u32,
pub optimized_tokens: u32,
pub original_bytes: u64,
pub optimized_bytes: u64,
pub mode: String,
}
#[derive(Debug, serde::Serialize, serde::Deserialize)]
pub struct SqueezerStats {
pub total_optimizations: u64,
pub total_original_tokens: u64,
pub total_optimized_tokens: u64,
pub total_original_bytes: u64,
pub total_optimized_bytes: u64,
pub history: Vec<OptimizationReport>,
}
impl SqueezerStats {
pub fn total_token_savings(&self) -> u64 {
self.total_original_tokens
.saturating_sub(self.total_optimized_tokens)
}
pub fn total_byte_savings(&self) -> u64 {
self.total_original_bytes
.saturating_sub(self.total_optimized_bytes)
}
pub fn estimated_usd_saved(&self) -> f64 {
(self.total_token_savings() as f64 / 1_000_000.0) * 2.50
}
}
pub struct Persistence;
impl Persistence {
fn get_db_path() -> PathBuf {
let mut path = dirs::home_dir().unwrap_or_else(|| PathBuf::from("."));
path.push(".vision-squeezer");
let _ = std::fs::create_dir_all(&path);
path.push("stats.db");
path
}
pub fn init_db() -> Result<(), String> {
let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
conn.execute(
"CREATE TABLE IF NOT EXISTS optimizations (
id INTEGER PRIMARY KEY AUTOINCREMENT,
timestamp TEXT NOT NULL,
model TEXT NOT NULL,
original_tokens INTEGER NOT NULL,
optimized_tokens INTEGER NOT NULL,
original_bytes INTEGER NOT NULL,
optimized_bytes INTEGER NOT NULL,
mode TEXT NOT NULL
)",
[],
)
.map_err(|e| e.to_string())?;
Ok(())
}
pub fn log_optimization(
model: &str,
orig_tokens: u32,
opt_tokens: u32,
orig_bytes: u64,
opt_bytes: u64,
mode: &str,
) -> Result<(), String> {
let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
conn.execute(
"INSERT INTO optimizations (timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode)
VALUES (?, ?, ?, ?, ?, ?, ?)",
params![
Utc::now().to_rfc3339(),
model,
orig_tokens,
opt_tokens,
orig_bytes as i64,
opt_bytes as i64,
mode,
],
).map_err(|e| e.to_string())?;
Ok(())
}
pub fn get_stats() -> Result<SqueezerStats, String> {
let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
let mut stmt = conn
.prepare(
"SELECT
COUNT(*),
SUM(original_tokens),
SUM(optimized_tokens),
SUM(original_bytes),
SUM(optimized_bytes)
FROM optimizations",
)
.map_err(|e| e.to_string())?;
let (count, orig_t, opt_t, orig_b, opt_b) = stmt
.query_row([], |row| {
Ok((
row.get::<_, Option<i64>>(0)?.unwrap_or(0) as u64,
row.get::<_, Option<i64>>(1)?.unwrap_or(0) as u64,
row.get::<_, Option<i64>>(2)?.unwrap_or(0) as u64,
row.get::<_, Option<i64>>(3)?.unwrap_or(0) as u64,
row.get::<_, Option<i64>>(4)?.unwrap_or(0) as u64,
))
})
.map_err(|e| e.to_string())?;
let mut stmt = conn.prepare(
"SELECT timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode
FROM optimizations ORDER BY timestamp DESC LIMIT 50"
).map_err(|e| e.to_string())?;
let history = stmt
.query_map([], |row| {
Ok(OptimizationReport {
timestamp: row.get(0)?,
model: row.get(1)?,
original_tokens: row.get(2)?,
optimized_tokens: row.get(3)?,
original_bytes: row.get::<_, i64>(4)? as u64,
optimized_bytes: row.get::<_, i64>(5)? as u64,
mode: row.get(6)?,
})
})
.map_err(|e| e.to_string())?
.collect::<Result<Vec<_>, _>>()
.map_err(|e| e.to_string())?;
Ok(SqueezerStats {
total_optimizations: count,
total_original_tokens: orig_t,
total_optimized_tokens: opt_t,
total_original_bytes: orig_b,
total_optimized_bytes: opt_b,
history,
})
}
pub fn get_all_history() -> Result<Vec<OptimizationReport>, String> {
let conn = Connection::open(Self::get_db_path()).map_err(|e| e.to_string())?;
let mut stmt = conn.prepare(
"SELECT timestamp, model, original_tokens, optimized_tokens, original_bytes, optimized_bytes, mode
FROM optimizations ORDER BY timestamp ASC"
).map_err(|e| e.to_string())?;
stmt.query_map([], |row| {
Ok(OptimizationReport {
timestamp: row.get(0)?,
model: row.get(1)?,
original_tokens: row.get(2)?,
optimized_tokens: row.get(3)?,
original_bytes: row.get::<_, i64>(4)? as u64,
optimized_bytes: row.get::<_, i64>(5)? as u64,
mode: row.get(6)?,
})
})
.map_err(|e| e.to_string())?
.collect::<Result<Vec<_>, _>>()
.map_err(|e| e.to_string())
}
}