agent_works/guard/
default.rs1use agent_base::engine::react_loop_guard::{GuardCtx, GuardDecision, ReactLoopGuard};
2use agent_base::llm_trait::LlmProvider;
3use async_trait::async_trait;
4use std::sync::{Arc, Mutex};
5
6use super::config::{DefaultGuardConfig, ReasoningOnlyAction};
7use super::judge::call_completion_judge;
8
9pub struct DefaultGuard {
13 config: DefaultGuardConfig,
14 llm_client: Option<Arc<dyn LlmProvider>>,
15 notice: Mutex<Option<agent_base::NoticeHandle>>,
18}
19
20impl DefaultGuard {
21 pub fn new(config: DefaultGuardConfig) -> Self {
22 Self {
23 config,
24 llm_client: None,
25 notice: Mutex::new(None),
26 }
27 }
28
29 pub fn with_llm_client(config: DefaultGuardConfig, llm_client: Arc<dyn LlmProvider>) -> Self {
31 Self {
32 config,
33 llm_client: Some(llm_client),
34 notice: Mutex::new(None),
35 }
36 }
37
38 fn notice_handle(&self) -> Option<agent_base::NoticeHandle> {
40 self.notice.lock().unwrap().clone()
41 }
42
43 async fn handle_reasoning_only(&self, ctx: &GuardCtx) -> GuardDecision {
46 let strikes = ctx.reasoning_only_strikes;
47
48 match self.config.reasoning_only_action {
49 ReasoningOnlyAction::Fail => {
50 if strikes >= self.config.reasoning_only_max_strikes {
52 return GuardDecision::Fail {
53 error: "model produced only reasoning across multiple turns".to_string(),
54 };
55 }
56
57 GuardDecision::Continue {
58 nudge: Some(self.config.reasoning_only_nudge.clone()),
59 }
60 }
61 ReasoningOnlyAction::DisableThinking => {
62 if strikes >= self.config.reasoning_only_max_strikes {
64 if ctx.thinking_disabled {
66 return GuardDecision::Fail {
68 error: "model produced only reasoning even after thinking was disabled"
69 .to_string(),
70 };
71 }
72
73 return GuardDecision::DisableThinking {
75 nudge: self.config.disable_thinking_nudge.clone(),
76 };
77 }
78
79 GuardDecision::Continue {
80 nudge: Some(self.config.reasoning_only_nudge.clone()),
81 }
82 }
83 }
84 }
85
86 async fn handle_empty_response(&self, ctx: &GuardCtx) -> GuardDecision {
87 let strikes = ctx.empty_response_strikes;
88
89 if strikes >= self.config.empty_response_max_strikes {
90 return GuardDecision::Fail {
91 error: "model returned empty responses repeatedly".to_string(),
92 };
93 }
94
95 GuardDecision::Continue {
96 nudge: Some(self.config.empty_response_nudge.clone()),
97 }
98 }
99
100 async fn handle_text_only(&self, ctx: &GuardCtx) -> GuardDecision {
101 if ctx.last_tool_calls_invalid {
111 tracing::info!(
112 session_id = ctx.session_id.id,
113 turn = ctx.turn_count,
114 "text-only response after rejected tool calls — not completion, re-issuing"
115 );
116 return GuardDecision::Continue {
117 nudge: Some(
118 "Your previous tool call was NOT executed — its arguments \
119 were invalid or truncated. The report you just wrote \
120 describes work that never happened; do not narrate \
121 results. Re-issue the tool call with complete, valid \
122 JSON arguments."
123 .to_string(),
124 ),
125 };
126 }
127
128 let input_len = ctx.user_input.chars().count();
129 let output_len = ctx.model_response.chars().count();
130
131 let is_short_response = self.config.detect_short_response
134 && input_len > self.config.short_response_min_input
135 && output_len < self.config.short_response_max_output
136 && input_len > output_len;
137
138 if is_short_response {
139 tracing::info!(
140 input_chars = input_len,
141 output_chars = output_len,
142 min_input = self.config.short_response_min_input,
143 max_output = self.config.short_response_max_output,
144 run_has_tool_calls = ctx.run_has_tool_calls,
145 "short response detected in text-only branch"
146 );
147
148 if ctx.run_has_tool_calls && self.config.use_llm_judge {
149 const INPUT_LEN_LIMIT: usize = 10_000;
151 if input_len > INPUT_LEN_LIMIT {
152 tracing::info!(
153 input_chars = input_len,
154 input_limit = INPUT_LEN_LIMIT,
155 "skipping LLM judge — user input too large, trusting model"
156 );
157 return GuardDecision::Complete;
158 }
159 match call_completion_judge(
161 self.llm_client.as_ref(),
162 &ctx.user_input,
163 &ctx.model_response,
164 &ctx.all_user_inputs,
165 self.config.judge_fail_open,
166 self.config.judge_timeout_secs,
167 self.config.recent_user_count,
168 self.notice_handle().as_ref(),
169 )
170 .await
171 {
172 Ok(judge) => {
173 if judge.done {
174 GuardDecision::Complete
175 } else {
176 GuardDecision::Continue {
177 nudge: Some(format!(
178 "Your answer is incomplete: {}. Continue working on the task.",
179 judge.reason
180 )),
181 }
182 }
183 }
184 Err(e) => {
185 tracing::warn!("completion judge failed: {}", e);
187 if self.config.judge_fail_open {
188 GuardDecision::Complete
189 } else {
190 GuardDecision::Continue {
191 nudge: Some(
192 "Cannot verify task completion, please continue working."
193 .to_string(),
194 ),
195 }
196 }
197 }
198 }
199 } else {
200 GuardDecision::Continue {
202 nudge: Some(self.config.short_response_nudge.clone()),
203 }
204 }
205 } else if ctx.run_has_tool_calls && self.config.use_llm_judge {
206 if output_len >= self.config.judge_skip_threshold {
208 tracing::debug!(
209 response_chars = output_len,
210 threshold = self.config.judge_skip_threshold,
211 "text-only response long enough, skipping judge"
212 );
213 return GuardDecision::Complete;
214 }
215
216 const INPUT_LEN_LIMIT: usize = 10_000;
218 if input_len > INPUT_LEN_LIMIT {
219 tracing::info!(
220 input_chars = input_len,
221 input_limit = INPUT_LEN_LIMIT,
222 "skipping LLM judge — user input too large, trusting model"
223 );
224 return GuardDecision::Complete;
225 }
226
227 tracing::info!(
228 response_chars = output_len,
229 threshold = self.config.judge_skip_threshold,
230 "text-only response short, calling judge"
231 );
232 match call_completion_judge(
233 self.llm_client.as_ref(),
234 &ctx.user_input,
235 &ctx.model_response,
236 &ctx.all_user_inputs,
237 self.config.judge_fail_open,
238 self.config.judge_timeout_secs,
239 self.config.recent_user_count,
240 self.notice_handle().as_ref(),
241 )
242 .await
243 {
244 Ok(judge) => {
245 if judge.done {
246 GuardDecision::Complete
247 } else {
248 GuardDecision::Continue {
249 nudge: Some(format!(
250 "Your answer is incomplete: {}. Continue working on the task.",
251 judge.reason
252 )),
253 }
254 }
255 }
256 Err(e) => {
257 tracing::warn!("completion judge failed: {}", e);
259 if self.config.judge_fail_open {
260 GuardDecision::Complete
261 } else {
262 GuardDecision::Continue {
263 nudge: Some(
264 "Cannot verify task completion, please continue working."
265 .to_string(),
266 ),
267 }
268 }
269 }
270 }
271 } else {
272 GuardDecision::Complete
273 }
274 }
275}
276
277#[async_trait]
278impl ReactLoopGuard for DefaultGuard {
279 fn set_notice(&self, handle: agent_base::NoticeHandle) {
280 *self.notice.lock().unwrap() = Some(handle);
281 }
282
283 async fn on_turn(&self, ctx: &GuardCtx) -> GuardDecision {
284 if ctx.is_reasoning_only {
285 self.handle_reasoning_only(ctx).await
286 } else if ctx.is_empty_response {
287 self.handle_empty_response(ctx).await
288 } else if ctx.is_text_only {
289 self.handle_text_only(ctx).await
290 } else {
291 GuardDecision::Complete
292 }
293 }
294
295 async fn on_tool_call(&self, ctx: &GuardCtx) -> GuardDecision {
296 if ctx.thinking_disabled && ctx.original_thinking_enabled {
301 tracing::info!(
302 session_id = ctx.session_id.id,
303 turn = ctx.turn_count,
304 "tool call detected while thinking disabled, restoring thinking"
305 );
306 return GuardDecision::RestoreThinking;
307 }
308
309 GuardDecision::Complete
310 }
311}