use async_trait::async_trait;
use std::path::PathBuf;
use crate::vlm_bench::scoring::{self, LevelScore, Rating};
use crate::vlm_bench::{BenchScenario, Difficulty, ExpectedAnswer, VlmBenchLevel};
const PASS_THRESHOLD: f64 = 0.80;
pub struct L1TuiState {
fixtures_dir: PathBuf,
}
impl Default for L1TuiState {
fn default() -> Self {
Self::new()
}
}
impl L1TuiState {
pub fn new() -> Self {
Self {
fixtures_dir: PathBuf::from("vlm_fixtures/l1_tui_state"),
}
}
}
#[async_trait]
impl VlmBenchLevel for L1TuiState {
fn name(&self) -> &str {
"L1 TUI State"
}
fn difficulty(&self) -> Difficulty {
Difficulty::Easy
}
fn description(&self) -> &str {
"Terminal state recognition: identify panels, read status text, \
detect color themes, and count visible widgets in TUI screenshots."
}
fn scenarios(&self) -> Vec<BenchScenario> {
vec![
BenchScenario {
id: "l1_dashboard_normal".into(),
description: "Identify the active panel and status in a normal dashboard view"
.into(),
image_path: self.fixtures_dir.join("dashboard_normal.png"),
prompt: "Analyze this terminal UI screenshot. Respond with a JSON object containing: \
{\"active_panel\": \"<panel name>\", \"status\": \"<status bar text>\", \
\"widget_count\": <number of visible widgets>, \"theme\": \"<dark or light>\"}".into(),
expected: ExpectedAnswer::JsonFields(serde_json::json!({
"active_panel": "dashboard",
"theme": "dark"
})),
},
BenchScenario {
id: "l1_dashboard_error".into(),
description: "Detect error state in a dashboard with error indicators".into(),
image_path: self.fixtures_dir.join("dashboard_error.png"),
prompt: "Analyze this terminal UI screenshot showing an error state. \
What error or warning is visible? Respond with JSON: \
{\"has_error\": true/false, \"error_type\": \"<type>\", \
\"active_panel\": \"<panel name>\"}".into(),
expected: ExpectedAnswer::Keywords(vec![
"error".into(),
"true".into(),
]),
},
BenchScenario {
id: "l1_help_panel".into(),
description: "Read content from a help/documentation panel".into(),
image_path: self.fixtures_dir.join("help_panel.png"),
prompt: "This screenshot shows a help panel in a terminal UI. \
List the keyboard shortcuts or commands visible. \
Respond with JSON: {\"panel_type\": \"help\", \
\"shortcuts\": [\"<shortcut1>\", ...]}".into(),
expected: ExpectedAnswer::Keywords(vec![
"help".into(),
"shortcut".into(),
]),
},
BenchScenario {
id: "l1_loading_state".into(),
description: "Identify a loading/spinner state".into(),
image_path: self.fixtures_dir.join("loading_state.png"),
prompt: "Analyze this terminal UI screenshot. Is the interface in a loading state? \
What visual indicators suggest loading? Respond with JSON: \
{\"is_loading\": true/false, \"indicators\": [\"<indicator>\", ...], \
\"active_panel\": \"<panel name>\"}".into(),
expected: ExpectedAnswer::Keywords(vec![
"loading".into(),
"true".into(),
]),
},
]
}
fn evaluate(&self, scenario: &BenchScenario, response: &str) -> LevelScore {
crate::vlm_bench::score_response(scenario, response, PASS_THRESHOLD)
}
}
#[cfg(test)]
#[path = "../../../tests/unit/vlm_bench/levels/l1_tui_state/l1_tui_state_test.rs"]
mod tests;