trusty_memory/service/core.rs
1//! `MemoryService` — the pure business-logic facade over `AppState`.
2//!
3//! Why: lets the axum HTTP handlers stay thin one-liners and lets non-HTTP
4//! callers (chat tool dispatch, RPC bridges) reuse the same code paths without
5//! dragging axum types around (split out of the former monolithic `service.rs`,
6//! issue #607).
7//! What: the `MemoryService` struct + its full async method surface, moved
8//! verbatim. Each method returns `anyhow::Result<Value>` or a typed
9//! `ServiceResult`.
10//! Test: every method is covered by the corresponding handler test in
11//! `web::tests`.
12
13use crate::attribution::CreatorInfo;
14use crate::{ActivitySource, AppState, DaemonEvent};
15use anyhow::{anyhow, Context, Result};
16use serde_json::{json, Value};
17use std::sync::Arc;
18use trusty_common::memory_core::palace::{Palace, PalaceId, RoomType};
19use trusty_common::memory_core::retrieval::{
20 recall_across_palaces_with_default_embedder, recall_deep_with_default_embedder,
21 recall_with_default_embedder, RememberOptions,
22};
23use trusty_common::memory_core::store::PalaceStoreError;
24use trusty_common::memory_core::PalaceRegistry;
25use uuid::Uuid;
26
27use super::helpers::{
28 collect_palace_stats, drawer_content_preview, drawer_snippet, is_reserved_system_palace,
29 list_palaces_blocking, palace_info_blocking, palace_info_from, recall_entry_json,
30};
31use super::recall_stream::recall_streamed;
32use super::types::{
33 CreateDrawerBody, CreatePalaceBody, ListDrawersQuery, PalaceInfo, ServiceError, ServiceResult,
34 StatusPayload,
35};
36
37/// Hard cap on triples returned by the per-palace graph endpoint.
38pub(super) const KG_GRAPH_MAX_TRIPLES: usize = 5_000;
39
40// ---------------------------------------------------------------------------
41// MemoryService — pure business logic facade.
42// ---------------------------------------------------------------------------
43
44/// Wraps [`AppState`] and exposes one async method per logical operation.
45///
46/// Why: see module docs. Lets HTTP handlers stay thin and lets non-HTTP
47/// callers (chat tool dispatch, RPC bridges) reuse the same code paths.
48/// What: `Clone` (cheap — only the inner `AppState` is shared); construct
49/// with `MemoryService::new(state)`.
50/// Test: every method is covered by the corresponding handler test in
51/// `web::tests`.
52#[derive(Clone)]
53pub struct MemoryService {
54 pub(super) state: AppState,
55}
56
57impl MemoryService {
58 /// Construct a new service wrapper.
59 ///
60 /// Why: handlers cheaply re-wrap their `AppState` on every request; the
61 /// cost is just an `Arc` clone, so we don't bother caching the wrapper.
62 /// What: stores the `AppState` for later method calls.
63 /// Test: trivial — covered indirectly by every handler test.
64 pub fn new(state: AppState) -> Self {
65 Self { state }
66 }
67
68 /// Borrow the inner [`AppState`].
69 ///
70 /// Why: some handlers still need direct access (SSE broadcaster, session
71 /// store, etc.) while we incrementally extract code into the service.
72 /// What: returns a borrowed reference to the wrapped `AppState`.
73 /// Test: not directly tested; surface-level accessor.
74 pub fn state(&self) -> &AppState {
75 &self.state
76 }
77
78 // -----------------------------------------------------------------
79 // Status / config
80 // -----------------------------------------------------------------
81
82 /// Build the aggregate `/api/v1/status` payload.
83 ///
84 /// Why: dashboard widgets and the MCP `get_status` tool need the same
85 /// roll-up; centralising avoids drift between the two surfaces.
86 /// What: walks every persisted palace for `palace_count`, then sums
87 /// drawer/vector/triple counts across the cache-resident subset and
88 /// returns the [`StatusPayload`].
89 /// Why (issue #4637): this used to open every persisted palace to sum
90 /// those three counts. With 5,794 palaces on disk against a 64-slot LRU
91 /// that is ~5,730 cold opens of ~1s each — the endpoint measurably never
92 /// responded. `palace_count` still reflects the true on-disk total (the
93 /// directory walk is cheap and now runs on the blocking pool); the totals
94 /// cover only cache-resident palaces and say so via `cached_palace_count`.
95 /// Test: `status_endpoint_returns_payload`,
96 /// `status_does_not_open_uncached_palaces`.
97 pub async fn status(&self) -> StatusPayload {
98 // The `/status` endpoint is the one place we still want a disk view —
99 // an operator hitting this endpoint right after restart (before
100 // `load_palaces_from_disk` finishes) should still see every persisted
101 // palace counted, even if it isn't in the in-memory registry yet.
102 let palaces = list_palaces_blocking(&self.state).await.unwrap_or_default();
103 let palace_count = palaces.len();
104 // #4637: peek() not open_palace() — full-registry open is O(n) cold disk I/O
105 let stats = collect_palace_stats(&self.state, palaces.iter().map(|p| &p.id));
106 StatusPayload {
107 version: self.state.version.clone(),
108 palace_count,
109 default_palace: self.state.default_palace.clone(),
110 data_root: self.state.data_root.display().to_string(),
111 total_drawers: stats.total_drawers,
112 total_vectors: stats.total_vectors,
113 total_kg_triples: stats.total_kg_triples,
114 cached_palace_count: stats.cached_palace_count,
115 }
116 }
117
118 /// Compute the aggregate `StatusChanged` event used by SSE consumers.
119 ///
120 /// Why: mutating handlers — and the periodic status ticker — push a
121 /// refreshed status snapshot so dashboards stay in sync without an
122 /// extra `/api/v1/status` request.
123 /// Why (issue #228): this used to call `PalaceRegistry::list_palaces`
124 /// (a synchronous disk walk) + `open_palace` (more disk I/O on first
125 /// call) for every palace on every emit. Since every persisted palace
126 /// is already loaded into the in-memory registry by
127 /// `AppState::load_palaces_from_disk` at startup (and every `create_palace`
128 /// keeps it in sync), iterating the in-memory registry returns the same
129 /// counts without touching disk.
130 /// What: iterates `state.registry.list()` (a `DashMap` snapshot) and
131 /// sums the live handle stats via [`collect_palace_stats`]. Returns a
132 /// `DaemonEvent::StatusChanged`. Palaces that fail to resolve in the
133 /// registry (race during shutdown) are silently skipped — the next
134 /// emit will catch them.
135 /// Test: indirectly via SSE integration tests; the math is identical to
136 /// the disk-walk implementation and the `status_endpoint_returns_payload`
137 /// test still passes against `status()` (which keeps the disk view for
138 /// the dedicated endpoint).
139 pub fn aggregate_status_event(&self) -> DaemonEvent {
140 let ids: Vec<PalaceId> = self.state.registry.list();
141 let stats = collect_palace_stats(&self.state, ids.iter());
142 DaemonEvent::StatusChanged {
143 total_drawers: stats.total_drawers,
144 total_vectors: stats.total_vectors,
145 total_kg_triples: stats.total_kg_triples,
146 }
147 }
148
149 // -----------------------------------------------------------------
150 // Palaces
151 // -----------------------------------------------------------------
152
153 /// List every palace on disk, enriched with live handle stats.
154 ///
155 /// Why: shared between the HTTP handler and the chat tool dispatcher;
156 /// both want the same `PalaceInfo` shape. Issue #185 added the
157 /// reserved-prefix filter so internal "system" palaces (e.g. the
158 /// `__health_probe__` palace used by `/health`) never surface in the
159 /// admin UI, TUI, or any user-facing roster.
160 /// What: walks the registry, drops any palace whose id starts with the
161 /// reserved `__` prefix, and builds a `PalaceInfo` per remaining row.
162 /// Why (issue #4637): this used to call `open_palace` per row purely to
163 /// enrich it with counts. At 5,794 palaces against a 64-slot LRU that is
164 /// ~90 minutes of cold, blocking disk I/O inline on the async executor —
165 /// and it evicted the entire working set on every call. Rows now come
166 /// from `PalaceRegistry::peek` (zero I/O, no LRU promotion, mirroring the
167 /// #1924 fix in `console_metrics.rs`). Uncached rows carry `cached: false`
168 /// and zero counts; a client that needs live counts for one palace should
169 /// fetch `GET /api/v1/palaces/{id}`, which still opens it.
170 /// Test: `palace_list_includes_richer_counts`, `palace_list_includes_graph_counts`,
171 /// `health_probe_palace_is_invisible` (in `web::tests`),
172 /// `list_palaces_does_not_open_uncached_palaces`.
173 pub async fn list_palaces(&self) -> ServiceResult<Vec<PalaceInfo>> {
174 let palaces = list_palaces_blocking(&self.state)
175 .await
176 .map_err(|e| ServiceError::internal(format!("{e:#}")))?;
177 // #7106: the enrichment itself is blocking work — a resident palace's
178 // first community count after a write partitions its whole graph. One
179 // hop for the whole loop, not one per row: this list can be thousands
180 // of rows and a task each would cost more than the work.
181 let registry = Arc::clone(&self.state.registry);
182 let out = tokio::task::spawn_blocking(move || {
183 let mut out = Vec::with_capacity(palaces.len());
184 for p in palaces {
185 if is_reserved_system_palace(&p.id) {
186 continue;
187 }
188 // #4637: peek() not open_palace() — full-registry open is O(n) cold disk I/O
189 let handle = registry.peek(&p.id);
190 out.push(palace_info_from(&p, handle.as_ref()));
191 }
192 out
193 })
194 .await
195 .map_err(|e| ServiceError::internal(format!("join list_palaces enrichment: {e}")))?;
196 Ok(out)
197 }
198
199 /// Every non-system palace with REAL counts, keeping per-palace failures.
200 ///
201 /// Why (#6286): [`Self::list_palaces`] answers placeholder zeros plus
202 /// `cached: false` for any palace not already resident, which is why the
203 /// monitor could not use it and fanned out one [`Self::get_palace`] per id
204 /// instead. That fan-out then dropped a palace whose call failed at
205 /// `debug!`, so the panel could show "12 palaces" over 9 rows and nothing
206 /// said why. This is the one call that answers what the fan-out was
207 /// assembling, and it reports a failure as a failure rather than as an
208 /// absence.
209 ///
210 /// What: one entry per non-system palace, in registry order. `Ok` carries
211 /// the same [`PalaceInfo`] `get_palace` builds — the palace is opened, so
212 /// the counts are measurements. `Err` carries the open failure's message.
213 /// A palace never silently vanishes and never becomes a row of zeros.
214 ///
215 /// **This opens every palace, and that is the point.** #4637 removed
216 /// exactly this from `list_palaces` because at 5,794 palaces a cold open per
217 /// row is ~90 minutes of blocking disk I/O. The cost is unchanged from the
218 /// N-call fan-out this replaces — the same opens, one round trip instead of
219 /// N — and after the first poll the registry is warm. A caller that wants
220 /// cheap approximate rows still has `list_palaces`.
221 ///
222 /// # Errors
223 ///
224 /// Only when the registry itself cannot be walked. A palace that will not
225 /// open is an `Err` entry, not an error for the whole call.
226 ///
227 /// **Every open runs on the blocking pool (#6836).** The opens are cold
228 /// disk I/O, so running them inline on a tokio worker parked the executor
229 /// for the whole sweep — on a many-palace install that is minutes during
230 /// which nothing else the daemon serves makes progress. One
231 /// `spawn_blocking` per palace also gives the executor a yield point
232 /// between palaces rather than one uninterruptible block.
233 ///
234 /// Test: `rpc_palaces_list_reports_counts_per_palace`,
235 /// `rpc_palaces_list_reports_an_unreadable_palace_rather_than_dropping_it`,
236 /// `list_palaces_with_counts_opens_palaces_off_the_executor`.
237 pub async fn list_palaces_with_counts(
238 &self,
239 ) -> ServiceResult<Vec<(String, Result<PalaceInfo, String>)>> {
240 let palaces = list_palaces_blocking(&self.state)
241 .await
242 .map_err(|e| ServiceError::internal(format!("{e:#}")))?;
243 let mut out = Vec::with_capacity(palaces.len());
244 for p in palaces {
245 if is_reserved_system_palace(&p.id) {
246 continue;
247 }
248 let id = p.id.0.clone();
249 // #6836: the open is cold disk I/O — hop to the blocking pool so a
250 // full-estate sweep cannot park the async executor for its duration.
251 let registry = Arc::clone(&self.state.registry);
252 let root = self.state.data_root.clone();
253 let row =
254 tokio::task::spawn_blocking(move || match registry.open_palace(&root, &p.id) {
255 Ok(handle) => Ok(palace_info_from(&p, Some(&handle))),
256 Err(e) => Err(format!("{e:#}")),
257 })
258 .await
259 // A join failure is still a per-palace failure: the row says why
260 // rather than vanishing, exactly as an open failure does.
261 .unwrap_or_else(|e| Err(format!("join open palace: {e}")));
262 out.push((id, row));
263 }
264 Ok(out)
265 }
266
267 /// Create a new palace and emit the corresponding activity event.
268 ///
269 /// Why: trims duplicated work between the HTTP handler and any future
270 /// non-HTTP creation flow.
271 /// What: validates the name, builds the `Palace` row, calls
272 /// `PalaceRegistry::create_palace`, and emits `PalaceCreated`. Returns
273 /// the new palace id.
274 /// Test: covered indirectly by `palace_list_includes_richer_counts` (which
275 /// posts a palace through the HTTP layer then reads it back).
276 pub async fn create_palace(
277 &self,
278 body: CreatePalaceBody,
279 source: ActivitySource,
280 ) -> ServiceResult<String> {
281 let name = body.name.trim().to_string();
282 if name.is_empty() {
283 return Err(ServiceError::bad_request("name is required"));
284 }
285 // Issue #88 / Change 2: enforce palace = project mapping for
286 // HTTP-originated palace creation. The validation cwd is, in order of
287 // preference:
288 // a. `body.cwd` — the caller explicitly supplied their project path
289 // (correct for any client that is not the daemon itself).
290 // b. `std::env::current_dir()` — daemon's own cwd, the pre-Change-2
291 // fallback (rarely meaningful when the daemon is launched from ~).
292 // This keeps older clients that omit `cwd` working without a breaking
293 // change, while letting pin-file-aware clients get accurate validation.
294 // spec-001: `force=true` lets an application bypass the project-slug
295 // gate so it can create palaces under arbitrary slugs (e.g. one per
296 // app/tenant for chat-session storage). The env-var bypass remains for
297 // test contexts; both short-circuit the same validation call.
298 //
299 // Issue #1714: `force=true` bypasses slug validation entirely, so it
300 // is gated behind the minimal authz seam in `crate::authz` before any
301 // other check runs. In the default single-tenant mode this is a
302 // no-op (unchanged behaviour); in multi-tenant mode it fails closed
303 // until a real capability check lands. See `crate::authz` module
304 // docs for the full design rationale.
305 let skip_enforcement =
306 std::env::var("TRUSTY_SKIP_PALACE_ENFORCEMENT").as_deref() == Ok("1");
307 if body.force {
308 crate::authz::authorize_force_palace_create(&self.state)
309 .map_err(|e| ServiceError::forbidden(e.to_string()))?;
310 }
311 if !skip_enforcement && !body.force {
312 let cwd = body
313 .cwd
314 .as_deref()
315 .map(std::path::Path::new)
316 .map(|p| p.to_path_buf())
317 .or_else(|| std::env::current_dir().ok())
318 .unwrap_or_else(|| self.state.data_root.clone());
319 crate::project_root::validate_palace_name(&name, &cwd)
320 .map_err(|e| ServiceError::bad_request(e.to_string()))?;
321 }
322 let id = PalaceId::new(&name);
323 let palace = Palace {
324 id: id.clone(),
325 name: name.clone(),
326 description: body.description.filter(|s| !s.is_empty()),
327 created_at: chrono::Utc::now(),
328 data_dir: self.state.data_root.join(&name),
329 };
330 self.state
331 .registry
332 .create_palace(&self.state.data_root, palace)
333 .map_err(|e| ServiceError::internal(format!("create palace: {e:#}")))?;
334 // Issue #228: keep the in-memory palace-name cache in sync so writes
335 // to this palace can resolve `Palace.name` without a disk walk.
336 self.state.palace_names.insert(name.clone(), name.clone());
337 self.state.emit(DaemonEvent::PalaceCreated {
338 id: name.clone(),
339 name: name.clone(),
340 source,
341 });
342 Ok(name)
343 }
344
345 /// Delete a palace from disk, optionally rejecting non-empty palaces.
346 ///
347 /// Why: Issue #180 — operators need a way to drop an entire palace
348 /// without going through drawer-by-drawer deletion. Defaulting to a
349 /// "must be empty" guard prevents fat-finger destruction of populated
350 /// palaces; `force=true` is the explicit opt-in to the destructive path.
351 /// What: 1) confirms the palace exists on disk (else `NotFound`),
352 /// 2) when `!force`, lists drawers via the live handle and returns
353 /// `BadRequest("Palace has drawers; pass force=true to delete")` if
354 /// the palace is non-empty, 3) drops the in-memory registry entry so
355 /// future opens hit the (now-missing) disk state, 4) removes
356 /// `<data_root>/<palace_id>/` recursively via `tokio::fs::remove_dir_all`,
357 /// and 5) emits an aggregate `StatusChanged` so dashboards refresh.
358 /// Test: `delete_palace_removes_dir_when_empty`,
359 /// `delete_palace_refuses_when_drawers_present`,
360 /// `delete_palace_force_removes_populated_palace`,
361 /// `delete_palace_returns_not_found_for_missing_id` in `web::tests`.
362 pub async fn delete_palace(&self, palace_id: &str, force: bool) -> ServiceResult<()> {
363 let palaces = PalaceRegistry::list_palaces(&self.state.data_root)
364 .map_err(|e| ServiceError::internal(format!("list palaces: {e:#}")))?;
365 if !palaces.iter().any(|p| p.id.0 == palace_id) {
366 return Err(ServiceError::not_found(format!(
367 "palace not found: {palace_id}"
368 )));
369 }
370 if !force {
371 // Open the palace just long enough to count its drawers; we don't
372 // hold the handle past this check because the caller is about to
373 // delete the on-disk directory.
374 if let Ok(handle) = self
375 .state
376 .registry
377 .open_palace(&self.state.data_root, &PalaceId::new(palace_id))
378 {
379 if !handle.drawers.read().is_empty() {
380 return Err(ServiceError::conflict(
381 "Palace has drawers; pass force=true to delete",
382 ));
383 }
384 }
385 }
386 // Drop the cached `Arc<PalaceHandle>` and gap cache before unlinking
387 // the directory so subsequent reads can't be served from the stale
388 // in-memory state. The registry's `remove` is a no-op when the entry
389 // is absent (lazy-open palaces that no caller has touched yet).
390 self.state.registry.remove(&PalaceId::new(palace_id));
391 // #4639: drop the cached chat_sessions.redb handle too — otherwise the
392 // fd survives `remove_dir_all` and pins the deleted inode forever.
393 self.state.session_stores.remove(palace_id);
394 // Issue #228: drop the palace-name cache entry so future writes never
395 // resolve to a stale label.
396 self.state.palace_names.remove(palace_id);
397 let palace_dir = self.state.data_root.join(palace_id);
398 tokio::fs::remove_dir_all(&palace_dir).await.map_err(|e| {
399 ServiceError::internal(format!("remove palace dir {}: {e}", palace_dir.display()))
400 })?;
401 // Recompute aggregate totals so dashboards drop the deleted palace's
402 // counts. There's no dedicated `PalaceDeleted` event variant yet;
403 // `StatusChanged` is enough to keep the UI in sync.
404 self.state.emit(self.aggregate_status_event());
405 Ok(())
406 }
407
408 /// Rename a palace's display name without touching its data.
409 ///
410 /// Why: Operators need to fix typos and rebrand palaces without dropping
411 /// the underlying drawers / vectors / KG. The palace id (the directory
412 /// name on disk) is immutable — only the human-readable `name` field in
413 /// `palace.json` changes — so cached `PalaceHandle`s stay valid and no
414 /// registry invalidation is required.
415 /// What: 1) loads the palace via `PalaceStore::load_palace` (404 when the
416 /// directory or `palace.json` is genuinely missing; a probe that cannot
417 /// determine whether it is there is a 500, not a 404 — #5549), 2) trims the
418 /// new name and
419 /// returns `BadRequest` when empty, 3) mutates `palace.name` and writes
420 /// the metadata back through the atomic `PalaceStore::save_palace`
421 /// (tmp file + rename), 4) emits an aggregate `StatusChanged` so
422 /// dashboards re-render the relabelled palace, 5) returns the updated
423 /// palace as JSON (enriched with the live handle stats, so callers see
424 /// drawer/vector/KG counts in the same shape as `GET /palaces/{id}`).
425 /// Test: `update_palace_name_renames_palace`,
426 /// `update_palace_name_rejects_empty_name`,
427 /// `update_palace_name_returns_not_found_for_missing_id` in `web::tests`.
428 pub async fn update_palace_name(&self, palace_id: &str, name: &str) -> Result<Value> {
429 let trimmed = name.trim();
430 if trimmed.is_empty() {
431 return Err(anyhow!("name must be non-empty after trimming"));
432 }
433 let palace_dir = self.state.data_root.join(palace_id);
434 let mut palace = trusty_common::memory_core::store::PalaceStore::load_palace(&palace_dir)
435 .map_err(|e| {
436 // #5549: only a genuine absence may be reported as "not found".
437 if matches!(&e, PalaceStoreError::NotFound(_)) {
438 anyhow!("palace not found: {palace_id} ({e})")
439 } else {
440 anyhow!("cannot load palace {palace_id}: {e}")
441 }
442 })?;
443 palace.name = trimmed.to_string();
444 trusty_common::memory_core::store::PalaceStore::save_palace(&palace)
445 .with_context(|| format!("save palace metadata for {palace_id}"))?;
446 // Issue #228: refresh the in-memory name cache so subsequent writes
447 // surface the new label without a disk walk.
448 self.state
449 .palace_names
450 .insert(palace_id.to_string(), trimmed.to_string());
451 let handle = self
452 .state
453 .registry
454 .open_palace(&self.state.data_root, &palace.id)
455 .ok();
456 // #7106: enrichment runs on the blocking pool.
457 let info = palace_info_blocking(&palace, handle).await;
458 self.state.emit(self.aggregate_status_event());
459 serde_json::to_value(info).context("serialize palace info")
460 }
461
462 /// Typed variant of [`Self::update_palace_name`] used by the HTTP handler.
463 ///
464 /// Why: HTTP needs to distinguish 400 (empty name) from 404 (missing
465 /// palace) so the right status code is emitted; the chat / MCP tool
466 /// only cares about a `Result<Value>` because both errors are surfaced
467 /// as opaque MCP error strings. Keeping a typed variant alongside the
468 /// untyped one keeps the wire shape correct on both surfaces without
469 /// asking either caller to parse error strings.
470 /// What: same as [`Self::update_palace_name`] but returns
471 /// `ServiceError::BadRequest` for empty names and `ServiceError::NotFound`
472 /// for palace metadata that is genuinely absent. Metadata whose presence
473 /// cannot be determined — a denied or transient stat — is
474 /// `ServiceError::Internal`: a 404 would tell the client the palace does
475 /// not exist when nobody established that (#5549, ADR-0045).
476 /// Test: `update_palace_name_renames_palace`,
477 /// `update_palace_name_rejects_empty_name`,
478 /// `update_palace_name_returns_not_found_for_missing_id`,
479 /// `update_palace_name_reports_an_unstattable_palace_as_internal`.
480 pub async fn update_palace_name_typed(
481 &self,
482 palace_id: &str,
483 name: &str,
484 ) -> ServiceResult<Value> {
485 let trimmed = name.trim();
486 if trimmed.is_empty() {
487 return Err(ServiceError::bad_request(
488 "name must be non-empty after trimming",
489 ));
490 }
491 let palace_dir = self.state.data_root.join(palace_id);
492 let mut palace = trusty_common::memory_core::store::PalaceStore::load_palace(&palace_dir)
493 .map_err(|e| {
494 // #5549: `not_found` on every variant told the client the palace
495 // does not exist for a stat we were merely denied.
496 if matches!(&e, PalaceStoreError::NotFound(_)) {
497 ServiceError::not_found(format!("palace not found: {palace_id} ({e})"))
498 } else {
499 ServiceError::internal(format!("cannot load palace {palace_id}: {e}"))
500 }
501 })?;
502 palace.name = trimmed.to_string();
503 trusty_common::memory_core::store::PalaceStore::save_palace(&palace).map_err(|e| {
504 ServiceError::internal(format!("save palace metadata for {palace_id}: {e}"))
505 })?;
506 // Issue #228: refresh the in-memory name cache so subsequent writes
507 // surface the new label without a disk walk.
508 self.state
509 .palace_names
510 .insert(palace_id.to_string(), trimmed.to_string());
511 let handle = self
512 .state
513 .registry
514 .open_palace(&self.state.data_root, &palace.id)
515 .ok();
516 // #7106: enrichment runs on the blocking pool.
517 let info = palace_info_blocking(&palace, handle).await;
518 self.state.emit(self.aggregate_status_event());
519 serde_json::to_value(info)
520 .map_err(|e| ServiceError::internal(format!("serialize palace info: {e}")))
521 }
522
523 /// Look up a single palace by id and enrich with live handle stats.
524 ///
525 /// Why: distinct 404 vs. 500 path is needed by both HTTP and chat callers.
526 /// What: returns `NotFound` when the id is unknown, otherwise a fully
527 /// populated `PalaceInfo`.
528 /// Test: indirectly via `health_endpoint_round_trip_with_palace_is_ok`.
529 pub async fn get_palace(&self, id: &str) -> ServiceResult<PalaceInfo> {
530 let palaces = PalaceRegistry::list_palaces(&self.state.data_root)
531 .map_err(|e| ServiceError::internal(format!("list palaces: {e:#}")))?;
532 let palace = palaces
533 .into_iter()
534 .find(|p| p.id.0 == id)
535 .ok_or_else(|| ServiceError::not_found(format!("palace not found: {id}")))?;
536 let handle = self
537 .state
538 .registry
539 .open_palace(&self.state.data_root, &palace.id)
540 .ok();
541 // #7106: enrichment runs on the blocking pool.
542 Ok(palace_info_blocking(&palace, handle).await)
543 }
544
545 // -----------------------------------------------------------------
546 // Drawers
547 // -----------------------------------------------------------------
548
549 /// List drawers in a palace with optional room/tag filters and pagination.
550 ///
551 /// Why: deduplicates the open-handle + listing path between HTTP and chat,
552 /// and (issue #184) lets the TUI activity panel page through drawers in
553 /// creation-date order without breaking the importance-sorted default the
554 /// legacy callers rely on.
555 /// What: opens the palace handle, fetches a window of drawers, optionally
556 /// re-sorts by `created_at` descending when `sort = "created_desc"`
557 /// (leaving the importance-desc default untouched), then drops the
558 /// leading `offset` rows and keeps `limit`. For `created_desc` the
559 /// window must cover the full filtered set (otherwise the importance
560 /// pre-sort hides truly-recent low-importance drawers), so the window
561 /// is widened to a sane ceiling (`MAX_DRAWER_WINDOW`); the default
562 /// importance path keeps a tight `limit+offset` window.
563 /// Returns the serialised JSON array.
564 /// Test: `service::tests::list_drawers_creates_desc_paginates`.
565 pub async fn list_drawers(&self, id: &str, q: ListDrawersQuery) -> ServiceResult<Value> {
566 const MAX_DRAWER_WINDOW: usize = 10_000;
567 let handle = self.open_handle(id)?;
568 let room = q.room.as_deref().map(RoomType::parse);
569 let limit = q.limit.unwrap_or(50);
570 let offset = q.offset.unwrap_or(0);
571 let by_created = matches!(q.sort.as_deref(), Some("created_desc"));
572 // For created_desc the importance pre-sort would hide low-importance
573 // drawers that happen to be the most recent, so we need to fetch the
574 // full filtered set (capped at MAX_DRAWER_WINDOW). For importance
575 // ordering the legacy `limit + offset` window is sufficient.
576 let window = if by_created {
577 MAX_DRAWER_WINDOW
578 } else {
579 limit.saturating_add(offset).min(MAX_DRAWER_WINDOW)
580 };
581 let mut drawers = handle.list_drawers(room, q.tag.clone(), window);
582 if by_created {
583 drawers.sort_by_key(|d| std::cmp::Reverse(d.created_at));
584 }
585 let page: Vec<_> = drawers.into_iter().skip(offset).take(limit).collect();
586 // Issue #202: enrich every row with a short `snippet` derived from
587 // the drawer's content so the TUI activity panel can render a
588 // glanceable summary without re-parsing the full body. The
589 // snippet is whitespace-collapsed and bounded at
590 // `DRAWER_SNIPPET_MAX_CHARS` (60) — shorter than the SSE preview
591 // because the activity panel renders it on a single narrow row.
592 let payload: Vec<Value> = page
593 .into_iter()
594 .map(|drawer| {
595 let snippet = drawer_snippet(drawer.content());
596 let mut value = serde_json::to_value(&drawer).unwrap_or_else(|_| json!({}));
597 if let Value::Object(ref mut map) = value {
598 // `null` when the drawer has no usable content so
599 // clients can distinguish "no body" from "empty body
600 // after whitespace collapse".
601 let snippet_value = if snippet.is_empty() {
602 Value::Null
603 } else {
604 Value::String(snippet)
605 };
606 map.insert("snippet".to_string(), snippet_value);
607 }
608 value
609 })
610 .collect();
611 Ok(Value::Array(payload))
612 }
613
614 /// Store a new drawer and emit the matching activity events.
615 ///
616 /// Why: HTTP and chat both need the auto-KG-extraction follow-up; this
617 /// method keeps that side-effect chain in one place.
618 /// What: opens the palace, stores the drawer via
619 /// `PalaceHandle::remember_with_options` (issue #3225: `body.force`
620 /// threads through as `RememberOptions::force`, letting a caller bypass
621 /// the QUALITY gates only — `allow_secret_like` is left at its default
622 /// `false`, so secret detection always still runs, `force` or not),
623 /// emits `DrawerAdded` + `StatusChanged`, then triggers
624 /// `tools::auto_extract_and_assert`. Returns the new drawer id.
625 /// Test: `http_create_drawer_runs_auto_kg_extraction`,
626 /// `create_drawer_rejects_json_content_without_force`,
627 /// `create_drawer_force_bypasses_quality_gate_for_json_content`.
628 pub async fn create_drawer(
629 &self,
630 id: &str,
631 body: CreateDrawerBody,
632 creator: CreatorInfo,
633 source: ActivitySource,
634 ) -> ServiceResult<Uuid> {
635 let handle = self.open_handle(id)?;
636 let room = body
637 .room
638 .as_deref()
639 .map(RoomType::parse)
640 .unwrap_or(RoomType::General);
641 let importance = body.importance.unwrap_or(0.5);
642 let force = body.force.unwrap_or(false);
643 let content_preview = drawer_content_preview(&body.content);
644 let mut tags_with_creator = body.tags;
645 // Issue #202: project a bare-UUID session tag (when the caller
646 // passed one in the request body) into the reserved
647 // `creator:session=<first-8>` slot so the activity panel can
648 // surface session attribution without bespoke parsing.
649 if let Some(session_tag) = crate::attribution::session_tag_from_tags(&tags_with_creator) {
650 tags_with_creator.push(session_tag);
651 }
652 creator.merge_into(&mut tags_with_creator);
653 let content_for_kg = body.content.clone();
654 let tags_for_kg = tags_with_creator.clone();
655 let room_label_for_kg = crate::tools::room_label(&room);
656 let drawer_id = handle
657 .remember_with_options(
658 body.content,
659 room,
660 tags_with_creator,
661 importance,
662 RememberOptions {
663 force,
664 ..Default::default()
665 },
666 )
667 .await
668 .map_err(|e| ServiceError::internal(format!("remember: {e:#}")))?;
669 let drawer_count = handle.drawers.read().len();
670 // Issue #228: resolve from the in-memory cache instead of re-walking
671 // the data root on every HTTP `create_drawer` call. Same cache the
672 // MCP `lookup_palace_name` helper consults.
673 let palace_name = self
674 .state
675 .palace_names
676 .get(id)
677 .map(|entry| entry.value().clone())
678 .unwrap_or_else(|| id.to_string());
679 self.state.emit(DaemonEvent::DrawerAdded {
680 palace_id: id.to_string(),
681 palace_name,
682 drawer_count,
683 timestamp: chrono::Utc::now(),
684 content_preview,
685 source,
686 });
687 // Issue #228: do NOT emit `StatusChanged` on every drawer create —
688 // the periodic ticker (`run_http_on`) refreshes aggregate totals on
689 // a fixed cadence so dashboards stay current without an O(N palaces)
690 // recompute on the write hot path.
691 crate::tools::auto_extract_and_assert(
692 &handle,
693 drawer_id,
694 &content_for_kg,
695 &tags_for_kg,
696 room_label_for_kg.as_deref(),
697 )
698 .await;
699 Ok(drawer_id)
700 }
701
702 /// Forget (delete) a drawer and emit the matching events.
703 ///
704 /// Why: same dedup story as `create_drawer`. #5231: `DELETE` on a drawer id
705 /// that was never stored used to answer `204 No Content`, the same as a
706 /// real delete — this now 404s, matching `delete_palace`.
707 /// What: parses the drawer UUID, calls `PalaceHandle::forget`, deletes the
708 /// drawer's BM25 document, maps `ForgetOutcome::NotFound` to
709 /// `ServiceError::not_found`, and emits `DrawerDeleted` only when a drawer
710 /// was actually removed. #5053: the lexical delete runs on this path for
711 /// the same reason it runs on the MCP one — `HTTP DELETE` and
712 /// `memory_forget` remove the same drawer, and the backfill indexes it
713 /// whichever way it was written, so a lexical copy left here is the same
714 /// stale document.
715 /// Test: `delete_drawer_404s_for_an_unknown_drawer_id`;
716 /// `tests/bm25_forget_delete.rs` covers the deletion contract itself.
717 pub async fn delete_drawer(
718 &self,
719 id: &str,
720 drawer_id: &str,
721 source: ActivitySource,
722 ) -> ServiceResult<()> {
723 let handle = self.open_handle(id)?;
724 let uuid = Uuid::parse_str(drawer_id)
725 .map_err(|_| ServiceError::bad_request("drawer_id must be a UUID"))?;
726 let outcome = handle
727 .forget(uuid)
728 .await
729 .map_err(|e| ServiceError::internal(format!("forget: {e:#}")))?;
730 // #5053: a drawer the user deleted must stop matching lexical queries.
731 crate::tools::bm25::bm25_delete_document(&self.state, handle.id.as_str(), uuid)
732 .await
733 .map_err(|e| ServiceError::internal(format!("{e:#}")))?;
734 if !outcome.is_deleted() {
735 return Err(ServiceError::not_found(format!(
736 "drawer '{drawer_id}' not found in palace '{id}'"
737 )));
738 }
739 let drawer_count = handle.drawers.read().len();
740 self.state.emit(DaemonEvent::DrawerDeleted {
741 palace_id: id.to_string(),
742 drawer_count,
743 source,
744 });
745 // Issue #228: skip the per-write `StatusChanged` emit — the
746 // periodic ticker handles aggregate roll-ups.
747 Ok(())
748 }
749
750 // -----------------------------------------------------------------
751 // Recall
752 // -----------------------------------------------------------------
753
754 /// Per-palace recall (semantic search), optionally with deep retrieval.
755 ///
756 /// Why: HTTP and chat tools both perform the same fan-out logic.
757 /// What: opens the palace handle and dispatches to the shallow or deep
758 /// recall helper. Returns a JSON array of flattened drawer rows (the
759 /// `recall_entry_json` shape from issue #69).
760 /// Test: `recall_entry_json_hoists_drawer_fields`.
761 pub async fn recall(
762 &self,
763 id: &str,
764 query: &str,
765 top_k: usize,
766 deep: bool,
767 ) -> ServiceResult<Value> {
768 let handle = self.open_handle(id)?;
769 let mut results = if deep {
770 recall_deep_with_default_embedder(&handle, query, top_k).await
771 } else {
772 recall_with_default_embedder(&handle, query, top_k).await
773 }
774 .map_err(|e| ServiceError::internal(format!("recall: {e:#}")))?;
775 // #5036: the lexical lane, on the path the UserPromptSubmit hook
776 // actually takes. `handle_memory_recall` has run vector and BM25 in
777 // parallel and RRF-fused them since #156; this route reached
778 // `retrieval::layers` directly and was vector-only, so a prompt with no
779 // lexical counterweight retrieved by vector centroid alone.
780 //
781 // Keyed on the RESOLVED palace id, never the caller's slug —
782 // `open_handle` follows aliases, and the corpus the backfill wrote
783 // belongs to the resolved palace.
784 //
785 // Reuses `fuse_bm25_into_recall` rather than deriving a second scorer:
786 // it only BOOSTS drawers the vector lane already returned and never
787 // promotes a BM25-only hit, so it has no scaling constant that can
788 // degenerate when the surviving set is empty — the failure that folded
789 // three earlier attempts at this wiring.
790 if let Some(hits) =
791 crate::tools::bm25::bm25_search_optional(&self.state, handle.id.as_str(), query, top_k)
792 .await
793 {
794 crate::tools::bm25::fuse_bm25_into_recall(&mut results, &hits, top_k);
795 }
796 let payload: Vec<Value> = results.into_iter().map(recall_entry_json).collect();
797 Ok(json!(payload))
798 }
799
800 /// Cross-palace recall.
801 ///
802 /// Why: shared between `/api/v1/recall` and the `memory_recall_all` chat
803 /// tool. Encapsulating the open-everything-fanout-merge dance avoids
804 /// drift.
805 /// What: lists every palace, then streams them through `recall_streamed`
806 /// in bounded batches, delegating each batch to
807 /// `recall_across_palaces_with_default_embedder`. Returns a JSON array.
808 /// Why (issue #4637): unlike `list_palaces`/`status`, this route is NOT
809 /// converted to `peek()`. A cross-palace recall that answered from
810 /// cache-resident palaces only would silently omit ~98.9% of the corpus —
811 /// a wrong answer that looks like a right one, which is strictly worse
812 /// than a slow correct one. Every palace is still opened and still
813 /// searched; what changed is when.
814 /// Why (issue #7125): opening all of them AT ONCE made peak residency and
815 /// the post-call LRU residue both scale with the palace count. The batch
816 /// walk bounds the peak at `RECALL_PALACE_BATCH` and hands back everything
817 /// the query itself brought in, so the daemon's steady state after a
818 /// recall-all matches its steady state before one.
819 /// Test: indirectly via `recall_across_palaces_merges_results` and the
820 /// MCP `memory_recall_all` integration paths;
821 /// `open_palaces_blocking_opens_every_palace` pins that uncached palaces
822 /// are still searched; `recall_all_returns_open_palaces_to_baseline` pins
823 /// the residency bound.
824 pub async fn recall_all(&self, query: &str, top_k: usize, deep: bool) -> Value {
825 let palaces = match list_palaces_blocking(&self.state).await {
826 Ok(v) => v,
827 Err(e) => return json!({ "error": format!("{e:#}") }),
828 };
829 // #7125: stream the estate in batches instead of opening all of it.
830 let streamed = recall_streamed(
831 &self.state,
832 &palaces,
833 "recall_all",
834 top_k,
835 |handles| async move {
836 recall_across_palaces_with_default_embedder(&handles, query, top_k, deep).await
837 },
838 )
839 .await;
840 match streamed {
841 Ok(results) => json!(results
842 .into_iter()
843 .map(|r| json!({
844 "palace_id": r.palace_id,
845 "drawer_id": r.result.drawer.id.to_string(),
846 "content": r.result.drawer.content(),
847 "importance": r.result.drawer.importance,
848 "tags": r.result.drawer.tags,
849 "score": r.result.score,
850 "layer": r.result.layer,
851 }))
852 .collect::<Vec<_>>()),
853 Err(e) => json!({ "error": format!("recall_across_palaces: {e:#}") }),
854 }
855 }
856}