1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
//! Application configuration and (re)launching the inference/embedding servers through
//! [`ServerSupervisor`]. The orchestrator is the sole writer of `settings.json`.
use crate::app::events::AppEvent;
use crate::shared::config::{AppConfig, CloudProvider, ImpersonationMode, ServerMode};
use crate::shared::secrets::{ExternalSlot, SecretKey};
use crate::shared::server::ServerStatus;
use super::Orchestrator;
use super::engines::Server;
impl Orchestrator {
/// Applies configuration edits: saves, restarts the server/registry if
/// needed, and re-emits the settings. The sole writer of `settings.json`.
pub(super) fn handle_update_config(&mut self, config: AppConfig) {
let old = std::mem::replace(&mut self.config, config);
// The "last-open chat" is an orchestrator property, not a value editable on
// the settings screen. The config snapshot from the UI may carry a stale
// value (e.g. `None` from startup) — restore the actual one, so an edit to
// settings doesn't erase the memory of the chat.
self.config.last_active_chat = old.last_active_chat;
// Stored API keys are also an orchestrator property (`handle_set_api_key`):
// the settings screen sends the key itself as a separate command, and the config
// snapshot never carries them. Without restoring it, editing any setting would wipe the keys.
self.config.api_keys = old.api_keys.clone();
// MCP catalog TOFU pins are also an orchestrator property (`persist_mcp_pin`),
// not editable in the UI: inherit by server id when the UI snapshot doesn't
// carry them (a stale copy) — editing settings doesn't reset trust or
// trigger a false `config.mcp` diff (an extra server restart).
for srv in &mut self.config.mcp.servers {
if srv.pinned_catalog.is_none()
&& let Some(prev) = old.mcp.servers.iter().find(|s| s.id == srv.id)
{
srv.pinned_catalog = prev.pinned_catalog.clone();
}
}
if let Err(err) = self.storage.json().save_config(&self.config) {
let _ = self.evt_tx.send(AppEvent::Error(
self.ui_locale()
.tf("ui.err.save_settings_failed", &[("err", &err.to_string())]),
));
self.config = old; // roll back to the previous state
return;
}
// A chat-server settings change (model/mode/port/…) — a restart (spec
// §11.6), but deferred: the settings screen applies an edit on every field's
// commit, and the debounce coalesces a series of quick edits into one restart
// ([`super::restart_queue::RestartQueue`], flushed at the deadline in the loop).
if self.config.engine != old.engine {
self.restarts.mark_chat();
// A mode change alters the set of available sampling parameters
// (get_sampling/set_sampling) — rebuild the registry for the new
// provider right away (cheap, in-memory; the schema needs to be
// current starting from the very next turn).
if self.config.engine.mode.cloud_provider() != old.engine.mode.cloud_provider() {
self.rebuild_registry();
}
}
// An impersonation-server settings change — a deferred reconnect.
if self.config.impersonation_engine != old.impersonation_engine {
self.restarts.mark_impersonation();
}
// An embedding-server settings change — a deferred reconnect.
if self.config.embed != old.embed {
self.restarts.mark_embed();
}
// An MCP-servers settings change — a deferred re-raise (a debounce, like
// the engines): killing/spawning processes is an expensive operation.
if self.config.mcp != old.mcp {
self.restarts.mark_mcp();
}
// A tool-parameter change — a registry rebuild (python_path, limits).
// Video settings feed the same registry (the `youtube_watch` client is
// built there), so they rebuild it too.
if self.config.tools != old.tools || self.config.video != old.video {
self.rebuild_registry();
// The cap on background runs applies to the next start.
self.background_slots
.set_max(self.config.tools.subagent_background_max);
}
self.emit_settings();
}
/// Stores a secret entered in settings: encrypts it with the machine key
/// (`shared::secrets`) and puts it into `config.api_keys` as **this** machine's
/// entry. An empty value means removal. What has to happen afterwards depends
/// on the kind of secret, which is why the command carries a typed
/// [`SecretKey`] rather than a storage name.
///
/// The plaintext lives only in the argument and in the consumer (the HTTP
/// client, the archive, the MCP child process): what goes to disk is the
/// ciphertext, and the UI's config snapshot doesn't carry secrets at all (only
/// a "configured" flag). See docs/research/api-key-storage.md.
pub(super) fn handle_set_secret(&mut self, key: SecretKey, value: String) {
if !self.store_secret(&key.storage_name(), &value) {
return;
}
match key {
SecretKey::Provider(provider) => {
// Re-raise only the servers whose active provider had its key changed
// (the same debounce as for engine edits — see `flush_restarts`).
let p = Some(provider);
if self.config.engine.mode.cloud_provider() == p {
self.restarts.mark_chat();
}
if self.config.impersonation_engine.mode.cloud_provider() == p {
self.restarts.mark_impersonation();
}
if self.config.embed.mode.cloud_provider() == p {
self.restarts.mark_embed();
}
// The video slot uses the Gemini key too, and its client lives in the
// tool registry — so a Gemini key change has to rebuild it (cheap,
// in-memory), or `youtube_watch` would keep reporting itself
// unconfigured until the next unrelated settings edit.
if provider == CloudProvider::Gemini {
self.rebuild_registry();
}
}
// An external slot addresses exactly one server, so exactly one is
// re-raised — no mode check is needed (unlike a provider key, which
// several slots may or may not be pointing at). Speech is built per
// utterance from a settings snapshot, so like the backup password it has
// nothing to restart. See docs/history/external-api-key.md §5.4.
SecretKey::External(slot) => match slot {
ExternalSlot::Chat => self.restarts.mark_chat(),
ExternalSlot::Impersonation => self.restarts.mark_impersonation(),
ExternalSlot::Embed => self.restarts.mark_embed(),
ExternalSlot::Tts => {}
},
// The value is handed to a child process at spawn time, so it only
// takes effect on a re-apply — deferred like an engine edit. The
// config itself is unchanged, so `McpManager::is_current` has to
// compare the *resolved* environment, not just the settings (§9.1).
SecretKey::McpEnv { .. } => self.restarts.mark_mcp(),
// `WebSearch` takes its keyed backends at construction, so the key
// only reaches the tool through a registry rebuild — the same
// reason the Gemini branch above rebuilds for the video slot.
// Cheap and in-memory; without it `web_search` would keep using the
// keyless chain until the next unrelated settings edit.
SecretKey::Search(_) => self.rebuild_registry(),
// Read at backup/restore time — nothing to restart.
SecretKey::BackupPassword => {}
}
self.emit_settings();
}
/// Encrypts a secret into this machine's entry and persists the config.
/// `false` — it failed and the error was already reported to the user; the
/// previous state is restored, so a failed save never half-applies.
pub(super) fn store_secret(&mut self, name: &str, value: &str) -> bool {
let old = self.config.api_keys.clone();
let label = || {
format!(
"{} · {}",
crate::shared::secrets::machine_label(),
chrono::Local::now().format("%Y-%m-%d")
)
};
if let Err(err) =
crate::shared::secrets::put_key(&mut self.config.api_keys, name, value.trim(), label)
{
let _ = self.evt_tx.send(AppEvent::Error(
self.ui_locale()
.tf("ui.err.api_key_save_failed", &[("err", &err.to_string())]),
));
return false;
}
if let Err(err) = self.storage.json().save_config(&self.config) {
let _ = self.evt_tx.send(AppEvent::Error(
self.ui_locale()
.tf("ui.err.save_settings_failed", &[("err", &err.to_string())]),
));
self.config.api_keys = old; // roll back to the previous state
return false;
}
true
}
/// Applies (re)launches of servers that were deferred by the debounce (the deadline expired):
/// one `apply_*` per flagged server, and one shared status
/// snapshot. Reads the **final** `self.config` — the config is already replaced during
/// the edit, so a series of edits produces one restart with the final values.
pub(super) fn flush_restarts(&mut self) {
let (chat, embed, imp, mcp) = self.restarts.take();
let loc = self.ui_locale();
let keys = &self.config.api_keys;
// The flag only says "something was edited"; whether a restart is *worth doing*
// is decided here, against what each server is actually running. An edit and its
// undo (`Ctrl+Z`) both raise the flag, and the final config then equals the
// applied one — restarting would reload the server with the values it already
// has, and for a managed one that means killing and reloading a GGUF for
// nothing. See docs/history/settings-undo.md §5.1.
let mut applied_any = false;
if chat && !self.engines.chat_is_current(&self.config.engine, keys) {
self.engines.apply_chat(&self.config.engine, keys, loc);
applied_any = true;
}
if embed && !self.engines.embed_is_current(&self.config.embed, keys) {
self.engines.apply_embed(&self.config.embed, keys, loc);
applied_any = true;
}
if imp
&& !self
.engines
.impersonation_is_current(&self.config.impersonation_engine, keys)
{
self.engines
.apply_impersonation(&self.config.impersonation_engine, keys, loc);
applied_any = true;
}
if mcp && !self.mcp.is_current(&self.config.mcp, keys) {
self.apply_mcp_settings();
}
if applied_any {
self.emit_server_status();
}
}
/// (Re-)raises the chat server from `config.engine` and emits a status snapshot.
/// The immediate startup path (before the `run` loop); settings edits
/// go through the `restarts` debounce queue → [`Self::flush_restarts`].
pub(super) fn apply_chat_settings(&mut self) {
let loc = self.ui_locale();
self.engines
.apply_chat(&self.config.engine, &self.config.api_keys, loc);
// A different engine has a different context window, and an answer from
// the previous one must not be carried over — nor an in-flight one
// applied when it lands (spec §6.7). Asked again at once, not at the first
// turn: a command may read the answer before any turn runs.
self.refresh_engine_facts();
// The same for the model it is running: the previous server's name must
// not survive onto the new one's messages (spec §11.3).
self.refresh_model_name();
// And its slot count: the hint must not describe the previous server.
self.refresh_engine_slots();
self.emit_server_status();
}
/// The chat server's readiness flipped. A method rather than the loop's arm so
/// the order it keeps can be tested where it lives (docs/lessons.md §2): the
/// engine's facts are asked again **now**, not at the next turn — a gateway that
/// was unreachable at startup and came up later must not leave `/continue`
/// answering from silence (docs/history/gateway-images-and-continue.md §4, H2.1).
pub(super) fn handle_chat_status(&mut self, status: ServerStatus) {
self.engines.set_chat_status(status);
// Readiness flipped, so the engine may answer differently now: a server
// that was down could not report its context window or its catalogue, and
// one that just came up can. The channel only carries flips, so this is not
// a per-probe cost. See `ContextDiscovery`.
self.refresh_engine_facts();
// And a server that just came up can now say what it loaded, where a
// moment ago it could not (`ModelDiscovery`).
self.refresh_model_name();
// Likewise its slot count (the `sessions` hint).
self.refresh_engine_slots();
self.emit_server_status();
self.relaunch_dead_managed_servers();
}
/// (Re-)raises the embedding server from `config.embed` and emits a status snapshot
/// (the embeddings chip in the status line appears/disappears based on the setting).
pub(super) fn apply_embed_settings(&mut self) {
let loc = self.ui_locale();
self.engines
.apply_embed(&self.config.embed, &self.config.api_keys, loc);
// Two decorators, and the order is load-bearing:
//
// EmbedGuard { PrefixedEmbedder { real embedder } }
//
// The prefixer applies the model's input convention (`query:`/`passage:`
// and relatives). It goes *inside* the guard so the guard's own canary
// and calibration probes pass through it: that is what makes a
// convention switch read as the change of vector space it really is, and
// what keeps the similarity calibration measured in the same dressing
// the real text gets (docs/research/embedding-input-prefixes.md §3–§4).
//
// The guard itself: stored vectors are only comparable to a query from
// the same model, and dimensionality cannot establish that (see
// `embed_guard`). Rebuilding both here re-arms the check whenever the
// embedding settings change — exactly when the model is most likely to
// have been swapped. The check is lazy (embeddings have no readiness
// probe, ADR 0002).
let prefixed = std::sync::Arc::new(crate::shared::embed_prefix::PrefixedEmbedder::new(
self.engines.embedder.clone(),
self.config.embed.convention,
));
self.engines.embedder = std::sync::Arc::new(super::embed_guard::EmbedGuard::new(
prefixed,
self.storage.clone(),
self.config.embed.active_model_name(),
self.config.embed.convention,
self.ui_locale(),
self.evt_tx.clone(),
));
self.emit_server_status();
}
/// Revives a **managed** server whose process is gone.
///
/// Only managed servers are relaunched: we own the process, and a dead child
/// leaves a port that no amount of probing will revive. An external or cloud
/// server is someone else's to restart — its monitor keeps polling and picks the
/// recovery up on its own.
///
/// Called after every status update, so a relaunch that fails simply produces the
/// next `Disconnected` and the next attempt, until [`RestartBudget`] stops it. A
/// launch that fails *synchronously* (a missing model file — the preflight check)
/// posts no status at all and therefore never reaches this path: retrying it would
/// be pointless until the settings change. See docs/server-health-monitoring.md, F4.
///
/// [`RestartBudget`]: super::engines::RestartBudget
pub(super) fn relaunch_dead_managed_servers(&mut self) {
let loc = self.ui_locale();
let now = std::time::Instant::now();
let mut relaunched = false;
let chat_managed = self.config.engine.mode == ServerMode::Managed;
if chat_managed && self.needs_relaunch(Server::Chat, now) {
tracing::warn!("managed chat server is down — relaunching");
self.engines
.apply_chat(&self.config.engine, &self.config.api_keys, loc);
relaunched = true;
}
if self.config.embed.mode == ServerMode::Managed && self.needs_relaunch(Server::Embed, now)
{
tracing::warn!("managed embedding server is down — relaunching");
self.engines
.apply_embed(&self.config.embed, &self.config.api_keys, loc);
relaunched = true;
}
if self.config.impersonation_engine.mode == ImpersonationMode::Managed
&& self.needs_relaunch(Server::Impersonation, now)
{
tracing::warn!("managed impersonation server is down — relaunching");
self.engines.apply_impersonation(
&self.config.impersonation_engine,
&self.config.api_keys,
loc,
);
relaunched = true;
}
if relaunched {
self.emit_server_status(); // the chip returns to "connecting…"
}
}
/// Whether `server` is down and its crash-loop budget still allows a relaunch.
/// Consumes a budget slot when it answers `true`.
fn needs_relaunch(&mut self, server: Server, now: std::time::Instant) -> bool {
if !matches!(
self.engines.status_of(server),
ServerStatus::Disconnected(_)
) {
return false;
}
if self.engines.allow_relaunch(server, now) {
return true;
}
tracing::warn!(
?server,
"relaunch budget exhausted — leaving it disconnected"
);
false
}
/// (Re-)raises the impersonation server from `config.impersonation_engine`. In
/// `shared` mode a separate server isn't needed — the assistant's chat server is reused.
pub(super) fn apply_impersonation_settings(&mut self) {
let loc = self.ui_locale();
self.engines.apply_impersonation(
&self.config.impersonation_engine,
&self.config.api_keys,
loc,
);
self.emit_server_status();
}
/// (Re-)raises MCP servers from `config.mcp`: previous ones are killed, enabled
/// ones are spawned anew; their tools arrive via `Ready` events (see
/// [`super::mcp::McpManager`]). The registry is rebuilt right away — wrappers from the previous
/// generation (dead connections) leave it immediately.
pub(super) fn apply_mcp_settings(&mut self) {
let loc = self.ui_locale();
self.mcp.apply(&self.config.mcp, &self.config.api_keys, loc);
self.rebuild_registry();
}
}