1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
//! What the agent said to the conversation.
//!
//! Every surface that shows or returns an agent's answer (web chat, A2A, AG-UI,
//! MCP, FCP, the CLI, subagent results, evals) must agree on which assistant
//! output counts. Each used to decide on its own, and they disagreed: some
//! dropped commentary, some returned the first message, some the last. This
//! module is the one definition.
//!
//! Today an agent talks by writing assistant text, so "said" means a
//! non-commentary agent message with text. Explicit communication will add a
//! second source, messages sent through `send_message`; the readers below are
//! where that lands.
use crate::execution_phase::ExecutionPhase;
use super::message::{ContentPart, RuntimeMessage, RuntimeMessageRole};
/// Joins a message's non-empty text parts with newlines, dropping tool calls,
/// tool results, images, reasoning and provider-opaque parts.
pub fn spoken_text(parts: &[ContentPart]) -> String {
parts
.iter()
.filter_map(|part| match part {
ContentPart::Text(text) if !text.text.is_empty() => Some(text.text.as_str()),
_ => None,
})
.collect::<Vec<_>>()
.join("\n")
}
/// Whether an assistant message is working commentary rather than something
/// said to the person. Commentary stays in the transcript and never counts as
/// an answer.
pub fn is_commentary(phase: Option<ExecutionPhase>) -> bool {
matches!(phase, Some(ExecutionPhase::Commentary))
}
/// The text an agent message says to the conversation, or `None` when it is
/// not an agent message, is commentary, or carries no text.
pub fn said_text(message: &RuntimeMessage) -> Option<String> {
if message.role != RuntimeMessageRole::Agent {
return None;
}
said_text_with_phase(message.phase, &message.content)
}
/// [`said_text`] for callers that hold the parts and phase separately, such as
/// an `output.message.completed` payload already known to be from the agent.
pub fn said_text_with_phase(
phase: Option<ExecutionPhase>,
parts: &[ContentPart],
) -> Option<String> {
if is_commentary(phase) {
return None;
}
let text = spoken_text(parts);
(!text.trim().is_empty()).then_some(text)
}
/// [`said_text`] over the JSON of an `output.message.completed` payload, for
/// readers that hold raw event data. Lenient on purpose: it needs only
/// `message.content` (and `message.phase` when present), so partial fixtures
/// and older stored events read the same way typed messages do.
pub fn said_text_in_event(data: &serde_json::Value) -> Option<String> {
let message = data.get("message")?;
let phase = message
.get("phase")
.and_then(|phase| serde_json::from_value::<ExecutionPhase>(phase.clone()).ok());
if is_commentary(phase) {
return None;
}
let text = message
.get("content")?
.as_array()?
.iter()
.filter(|part| part.get("type").and_then(|t| t.as_str()) == Some("text"))
.filter_map(|part| part.get("text").and_then(|t| t.as_str()))
.filter(|text| !text.is_empty())
.collect::<Vec<_>>()
.join("\n");
(!text.trim().is_empty()).then_some(text)
}
/// One agent message as a reply candidate: what it said, and whether it also
/// carried tool calls (text next to a tool call is usually a preamble such as
/// "let me check", not the answer).
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct SaidMessage {
/// Text said to the conversation.
pub text: String,
/// Whether the same message also requested tool calls.
pub with_tool_calls: bool,
}
impl SaidMessage {
/// The candidate for an agent message, or `None` when it said nothing.
pub fn from_message(message: &RuntimeMessage) -> Option<Self> {
Some(Self {
text: said_text(message)?,
with_tool_calls: message.has_tool_calls(),
})
}
}
/// The reply of a turn or run: the last thing the agent said that was not a
/// preamble to a tool call, falling back to the last thing it said at all
/// (a turn cut short mid-tool still reports its last words).
pub fn final_reply<I>(said: I) -> Option<String>
where
I: IntoIterator<Item = SaidMessage>,
{
let mut last_any = None;
let mut last_plain = None;
for message in said {
if !message.with_tool_calls {
last_plain = Some(message.text.clone());
}
last_any = Some(message.text);
}
last_plain.or(last_any)
}
/// What an agent said over one turn, folded into the turn's reply with
/// [`final_reply`]. Hosts push as the turn runs and take the reply at the end.
#[derive(Debug, Clone, Default)]
pub struct TurnReply {
said: Vec<SaidMessage>,
}
impl TurnReply {
/// Record one reasoning step's text. Blank text says nothing.
pub fn push_text(&mut self, text: &str, with_tool_calls: bool) {
if !text.trim().is_empty() {
self.said.push(SaidMessage {
text: text.to_owned(),
with_tool_calls,
});
}
}
/// The turn's reply so far, empty when nothing was said. Resets the turn.
pub fn take(&mut self) -> String {
final_reply(std::mem::take(&mut self.said)).unwrap_or_default()
}
}
#[cfg(test)]
mod tests;