Skip to main content

trusty_memory/service/
core.rs

1//! `MemoryService` — the pure business-logic facade over `AppState`.
2//!
3//! Why: lets the axum HTTP handlers stay thin one-liners and lets non-HTTP
4//! callers (chat tool dispatch, RPC bridges) reuse the same code paths without
5//! dragging axum types around (split out of the former monolithic `service.rs`,
6//! issue #607).
7//! What: the `MemoryService` struct + its full async method surface, moved
8//! verbatim. Each method returns `anyhow::Result<Value>` or a typed
9//! `ServiceResult`.
10//! Test: every method is covered by the corresponding handler test in
11//! `web::tests`.
12
13use crate::attribution::CreatorInfo;
14use crate::{ActivitySource, AppState, DaemonEvent};
15use anyhow::{anyhow, Context, Result};
16use serde_json::{json, Value};
17use std::sync::Arc;
18use trusty_common::memory_core::palace::{Palace, PalaceId, RoomType};
19use trusty_common::memory_core::retrieval::{
20    recall_across_palaces_with_default_embedder, recall_deep_with_default_embedder,
21    recall_with_default_embedder, RememberOptions,
22};
23use trusty_common::memory_core::store::PalaceStoreError;
24use trusty_common::memory_core::PalaceRegistry;
25use uuid::Uuid;
26
27use super::helpers::{
28    collect_palace_stats, drawer_content_preview, drawer_snippet, is_reserved_system_palace,
29    list_palaces_blocking, palace_info_blocking, palace_info_from, recall_entry_json,
30};
31use super::recall_stream::recall_streamed;
32use super::types::{
33    CreateDrawerBody, CreatePalaceBody, ListDrawersQuery, PalaceInfo, ServiceError, ServiceResult,
34    StatusPayload,
35};
36
37/// Hard cap on triples returned by the per-palace graph endpoint.
38pub(super) const KG_GRAPH_MAX_TRIPLES: usize = 5_000;
39
40// ---------------------------------------------------------------------------
41// MemoryService — pure business logic facade.
42// ---------------------------------------------------------------------------
43
44/// Wraps [`AppState`] and exposes one async method per logical operation.
45///
46/// Why: see module docs. Lets HTTP handlers stay thin and lets non-HTTP
47/// callers (chat tool dispatch, RPC bridges) reuse the same code paths.
48/// What: `Clone` (cheap — only the inner `AppState` is shared); construct
49/// with `MemoryService::new(state)`.
50/// Test: every method is covered by the corresponding handler test in
51/// `web::tests`.
52#[derive(Clone)]
53pub struct MemoryService {
54    pub(super) state: AppState,
55}
56
57impl MemoryService {
58    /// Construct a new service wrapper.
59    ///
60    /// Why: handlers cheaply re-wrap their `AppState` on every request; the
61    /// cost is just an `Arc` clone, so we don't bother caching the wrapper.
62    /// What: stores the `AppState` for later method calls.
63    /// Test: trivial — covered indirectly by every handler test.
64    pub fn new(state: AppState) -> Self {
65        Self { state }
66    }
67
68    /// Borrow the inner [`AppState`].
69    ///
70    /// Why: some handlers still need direct access (SSE broadcaster, session
71    /// store, etc.) while we incrementally extract code into the service.
72    /// What: returns a borrowed reference to the wrapped `AppState`.
73    /// Test: not directly tested; surface-level accessor.
74    pub fn state(&self) -> &AppState {
75        &self.state
76    }
77
78    // -----------------------------------------------------------------
79    // Status / config
80    // -----------------------------------------------------------------
81
82    /// Build the aggregate `/api/v1/status` payload.
83    ///
84    /// Why: dashboard widgets and the MCP `get_status` tool need the same
85    /// roll-up; centralising avoids drift between the two surfaces.
86    /// What: walks every persisted palace for `palace_count`, then sums
87    /// drawer/vector/triple counts across the cache-resident subset and
88    /// returns the [`StatusPayload`].
89    /// Why (issue #4637): this used to open every persisted palace to sum
90    /// those three counts. With 5,794 palaces on disk against a 64-slot LRU
91    /// that is ~5,730 cold opens of ~1s each — the endpoint measurably never
92    /// responded. `palace_count` still reflects the true on-disk total (the
93    /// directory walk is cheap and now runs on the blocking pool); the totals
94    /// cover only cache-resident palaces and say so via `cached_palace_count`.
95    /// Test: `status_endpoint_returns_payload`,
96    /// `status_does_not_open_uncached_palaces`.
97    pub async fn status(&self) -> StatusPayload {
98        // The `/status` endpoint is the one place we still want a disk view —
99        // an operator hitting this endpoint right after restart (before
100        // `load_palaces_from_disk` finishes) should still see every persisted
101        // palace counted, even if it isn't in the in-memory registry yet.
102        let palaces = list_palaces_blocking(&self.state).await.unwrap_or_default();
103        let palace_count = palaces.len();
104        // #4637: peek() not open_palace() — full-registry open is O(n) cold disk I/O
105        let stats = collect_palace_stats(&self.state, palaces.iter().map(|p| &p.id));
106        StatusPayload {
107            version: self.state.version.clone(),
108            palace_count,
109            default_palace: self.state.default_palace.clone(),
110            data_root: self.state.data_root.display().to_string(),
111            total_drawers: stats.total_drawers,
112            total_vectors: stats.total_vectors,
113            total_kg_triples: stats.total_kg_triples,
114            cached_palace_count: stats.cached_palace_count,
115        }
116    }
117
118    /// Compute the aggregate `StatusChanged` event used by SSE consumers.
119    ///
120    /// Why: mutating handlers — and the periodic status ticker — push a
121    /// refreshed status snapshot so dashboards stay in sync without an
122    /// extra `/api/v1/status` request.
123    /// Why (issue #228): this used to call `PalaceRegistry::list_palaces`
124    /// (a synchronous disk walk) + `open_palace` (more disk I/O on first
125    /// call) for every palace on every emit. Since every persisted palace
126    /// is already loaded into the in-memory registry by
127    /// `AppState::load_palaces_from_disk` at startup (and every `create_palace`
128    /// keeps it in sync), iterating the in-memory registry returns the same
129    /// counts without touching disk.
130    /// What: iterates `state.registry.list()` (a `DashMap` snapshot) and
131    /// sums the live handle stats via [`collect_palace_stats`]. Returns a
132    /// `DaemonEvent::StatusChanged`. Palaces that fail to resolve in the
133    /// registry (race during shutdown) are silently skipped — the next
134    /// emit will catch them.
135    /// Test: indirectly via SSE integration tests; the math is identical to
136    /// the disk-walk implementation and the `status_endpoint_returns_payload`
137    /// test still passes against `status()` (which keeps the disk view for
138    /// the dedicated endpoint).
139    pub fn aggregate_status_event(&self) -> DaemonEvent {
140        let ids: Vec<PalaceId> = self.state.registry.list();
141        let stats = collect_palace_stats(&self.state, ids.iter());
142        DaemonEvent::StatusChanged {
143            total_drawers: stats.total_drawers,
144            total_vectors: stats.total_vectors,
145            total_kg_triples: stats.total_kg_triples,
146        }
147    }
148
149    // -----------------------------------------------------------------
150    // Palaces
151    // -----------------------------------------------------------------
152
153    /// List every palace on disk, enriched with live handle stats.
154    ///
155    /// Why: shared between the HTTP handler and the chat tool dispatcher;
156    /// both want the same `PalaceInfo` shape. Issue #185 added the
157    /// reserved-prefix filter so internal "system" palaces (e.g. the
158    /// `__health_probe__` palace used by `/health`) never surface in the
159    /// admin UI, TUI, or any user-facing roster.
160    /// What: walks the registry, drops any palace whose id starts with the
161    /// reserved `__` prefix, and builds a `PalaceInfo` per remaining row.
162    /// Why (issue #4637): this used to call `open_palace` per row purely to
163    /// enrich it with counts. At 5,794 palaces against a 64-slot LRU that is
164    /// ~90 minutes of cold, blocking disk I/O inline on the async executor —
165    /// and it evicted the entire working set on every call. Rows now come
166    /// from `PalaceRegistry::peek` (zero I/O, no LRU promotion, mirroring the
167    /// #1924 fix in `console_metrics.rs`). Uncached rows carry `cached: false`
168    /// and zero counts; a client that needs live counts for one palace should
169    /// fetch `GET /api/v1/palaces/{id}`, which still opens it.
170    /// Test: `palace_list_includes_richer_counts`, `palace_list_includes_graph_counts`,
171    /// `health_probe_palace_is_invisible` (in `web::tests`),
172    /// `list_palaces_does_not_open_uncached_palaces`.
173    pub async fn list_palaces(&self) -> ServiceResult<Vec<PalaceInfo>> {
174        let palaces = list_palaces_blocking(&self.state)
175            .await
176            .map_err(|e| ServiceError::internal(format!("{e:#}")))?;
177        // #7106: the enrichment itself is blocking work — a resident palace's
178        // first community count after a write partitions its whole graph. One
179        // hop for the whole loop, not one per row: this list can be thousands
180        // of rows and a task each would cost more than the work.
181        let registry = Arc::clone(&self.state.registry);
182        let out = tokio::task::spawn_blocking(move || {
183            let mut out = Vec::with_capacity(palaces.len());
184            for p in palaces {
185                if is_reserved_system_palace(&p.id) {
186                    continue;
187                }
188                // #4637: peek() not open_palace() — full-registry open is O(n) cold disk I/O
189                let handle = registry.peek(&p.id);
190                out.push(palace_info_from(&p, handle.as_ref()));
191            }
192            out
193        })
194        .await
195        .map_err(|e| ServiceError::internal(format!("join list_palaces enrichment: {e}")))?;
196        Ok(out)
197    }
198
199    /// Every non-system palace with REAL counts, keeping per-palace failures.
200    ///
201    /// Why (#6286): [`Self::list_palaces`] answers placeholder zeros plus
202    /// `cached: false` for any palace not already resident, which is why the
203    /// monitor could not use it and fanned out one [`Self::get_palace`] per id
204    /// instead. That fan-out then dropped a palace whose call failed at
205    /// `debug!`, so the panel could show "12 palaces" over 9 rows and nothing
206    /// said why. This is the one call that answers what the fan-out was
207    /// assembling, and it reports a failure as a failure rather than as an
208    /// absence.
209    ///
210    /// What: one entry per non-system palace, in registry order. `Ok` carries
211    /// the same [`PalaceInfo`] `get_palace` builds — the palace is opened, so
212    /// the counts are measurements. `Err` carries the open failure's message.
213    /// A palace never silently vanishes and never becomes a row of zeros.
214    ///
215    /// **This opens every palace, and that is the point.** #4637 removed
216    /// exactly this from `list_palaces` because at 5,794 palaces a cold open per
217    /// row is ~90 minutes of blocking disk I/O. The cost is unchanged from the
218    /// N-call fan-out this replaces — the same opens, one round trip instead of
219    /// N — and after the first poll the registry is warm. A caller that wants
220    /// cheap approximate rows still has `list_palaces`.
221    ///
222    /// # Errors
223    ///
224    /// Only when the registry itself cannot be walked. A palace that will not
225    /// open is an `Err` entry, not an error for the whole call.
226    ///
227    /// **Every open runs on the blocking pool (#6836).** The opens are cold
228    /// disk I/O, so running them inline on a tokio worker parked the executor
229    /// for the whole sweep — on a many-palace install that is minutes during
230    /// which nothing else the daemon serves makes progress. One
231    /// `spawn_blocking` per palace also gives the executor a yield point
232    /// between palaces rather than one uninterruptible block.
233    ///
234    /// Test: `rpc_palaces_list_reports_counts_per_palace`,
235    /// `rpc_palaces_list_reports_an_unreadable_palace_rather_than_dropping_it`,
236    /// `list_palaces_with_counts_opens_palaces_off_the_executor`.
237    pub async fn list_palaces_with_counts(
238        &self,
239    ) -> ServiceResult<Vec<(String, Result<PalaceInfo, String>)>> {
240        let palaces = list_palaces_blocking(&self.state)
241            .await
242            .map_err(|e| ServiceError::internal(format!("{e:#}")))?;
243        let mut out = Vec::with_capacity(palaces.len());
244        for p in palaces {
245            if is_reserved_system_palace(&p.id) {
246                continue;
247            }
248            let id = p.id.0.clone();
249            // #6836: the open is cold disk I/O — hop to the blocking pool so a
250            // full-estate sweep cannot park the async executor for its duration.
251            let registry = Arc::clone(&self.state.registry);
252            let root = self.state.data_root.clone();
253            let row =
254                tokio::task::spawn_blocking(move || match registry.open_palace(&root, &p.id) {
255                    Ok(handle) => Ok(palace_info_from(&p, Some(&handle))),
256                    Err(e) => Err(format!("{e:#}")),
257                })
258                .await
259                // A join failure is still a per-palace failure: the row says why
260                // rather than vanishing, exactly as an open failure does.
261                .unwrap_or_else(|e| Err(format!("join open palace: {e}")));
262            out.push((id, row));
263        }
264        Ok(out)
265    }
266
267    /// Create a new palace and emit the corresponding activity event.
268    ///
269    /// Why: trims duplicated work between the HTTP handler and any future
270    /// non-HTTP creation flow.
271    /// What: validates the name, builds the `Palace` row, calls
272    /// `PalaceRegistry::create_palace`, and emits `PalaceCreated`. Returns
273    /// the new palace id.
274    /// Test: covered indirectly by `palace_list_includes_richer_counts` (which
275    /// posts a palace through the HTTP layer then reads it back).
276    pub async fn create_palace(
277        &self,
278        body: CreatePalaceBody,
279        source: ActivitySource,
280    ) -> ServiceResult<String> {
281        let name = body.name.trim().to_string();
282        if name.is_empty() {
283            return Err(ServiceError::bad_request("name is required"));
284        }
285        // Issue #88 / Change 2: enforce palace = project mapping for
286        // HTTP-originated palace creation. The validation cwd is, in order of
287        // preference:
288        //   a. `body.cwd` — the caller explicitly supplied their project path
289        //      (correct for any client that is not the daemon itself).
290        //   b. `std::env::current_dir()` — daemon's own cwd, the pre-Change-2
291        //      fallback (rarely meaningful when the daemon is launched from ~).
292        // This keeps older clients that omit `cwd` working without a breaking
293        // change, while letting pin-file-aware clients get accurate validation.
294        // spec-001: `force=true` lets an application bypass the project-slug
295        // gate so it can create palaces under arbitrary slugs (e.g. one per
296        // app/tenant for chat-session storage). The env-var bypass remains for
297        // test contexts; both short-circuit the same validation call.
298        //
299        // Issue #1714: `force=true` bypasses slug validation entirely, so it
300        // is gated behind the minimal authz seam in `crate::authz` before any
301        // other check runs. In the default single-tenant mode this is a
302        // no-op (unchanged behaviour); in multi-tenant mode it fails closed
303        // until a real capability check lands. See `crate::authz` module
304        // docs for the full design rationale.
305        let skip_enforcement =
306            std::env::var("TRUSTY_SKIP_PALACE_ENFORCEMENT").as_deref() == Ok("1");
307        if body.force {
308            crate::authz::authorize_force_palace_create(&self.state)
309                .map_err(|e| ServiceError::forbidden(e.to_string()))?;
310        }
311        if !skip_enforcement && !body.force {
312            let cwd = body
313                .cwd
314                .as_deref()
315                .map(std::path::Path::new)
316                .map(|p| p.to_path_buf())
317                .or_else(|| std::env::current_dir().ok())
318                .unwrap_or_else(|| self.state.data_root.clone());
319            crate::project_root::validate_palace_name(&name, &cwd)
320                .map_err(|e| ServiceError::bad_request(e.to_string()))?;
321        }
322        let id = PalaceId::new(&name);
323        let palace = Palace {
324            id: id.clone(),
325            name: name.clone(),
326            description: body.description.filter(|s| !s.is_empty()),
327            created_at: chrono::Utc::now(),
328            data_dir: self.state.data_root.join(&name),
329        };
330        self.state
331            .registry
332            .create_palace(&self.state.data_root, palace)
333            .map_err(|e| ServiceError::internal(format!("create palace: {e:#}")))?;
334        // Issue #228: keep the in-memory palace-name cache in sync so writes
335        // to this palace can resolve `Palace.name` without a disk walk.
336        self.state.palace_names.insert(name.clone(), name.clone());
337        self.state.emit(DaemonEvent::PalaceCreated {
338            id: name.clone(),
339            name: name.clone(),
340            source,
341        });
342        Ok(name)
343    }
344
345    /// Delete a palace from disk, optionally rejecting non-empty palaces.
346    ///
347    /// Why: Issue #180 — operators need a way to drop an entire palace
348    /// without going through drawer-by-drawer deletion. Defaulting to a
349    /// "must be empty" guard prevents fat-finger destruction of populated
350    /// palaces; `force=true` is the explicit opt-in to the destructive path.
351    /// What: 1) confirms the palace exists on disk (else `NotFound`),
352    /// 2) when `!force`, lists drawers via the live handle and returns
353    /// `BadRequest("Palace has drawers; pass force=true to delete")` if
354    /// the palace is non-empty, 3) drops the in-memory registry entry so
355    /// future opens hit the (now-missing) disk state, 4) removes
356    /// `<data_root>/<palace_id>/` recursively via `tokio::fs::remove_dir_all`,
357    /// and 5) emits an aggregate `StatusChanged` so dashboards refresh.
358    /// Test: `delete_palace_removes_dir_when_empty`,
359    /// `delete_palace_refuses_when_drawers_present`,
360    /// `delete_palace_force_removes_populated_palace`,
361    /// `delete_palace_returns_not_found_for_missing_id` in `web::tests`.
362    pub async fn delete_palace(&self, palace_id: &str, force: bool) -> ServiceResult<()> {
363        let palaces = PalaceRegistry::list_palaces(&self.state.data_root)
364            .map_err(|e| ServiceError::internal(format!("list palaces: {e:#}")))?;
365        if !palaces.iter().any(|p| p.id.0 == palace_id) {
366            return Err(ServiceError::not_found(format!(
367                "palace not found: {palace_id}"
368            )));
369        }
370        if !force {
371            // Open the palace just long enough to count its drawers; we don't
372            // hold the handle past this check because the caller is about to
373            // delete the on-disk directory.
374            if let Ok(handle) = self
375                .state
376                .registry
377                .open_palace(&self.state.data_root, &PalaceId::new(palace_id))
378            {
379                if !handle.drawers.read().is_empty() {
380                    return Err(ServiceError::conflict(
381                        "Palace has drawers; pass force=true to delete",
382                    ));
383                }
384            }
385        }
386        // Drop the cached `Arc<PalaceHandle>` and gap cache before unlinking
387        // the directory so subsequent reads can't be served from the stale
388        // in-memory state. The registry's `remove` is a no-op when the entry
389        // is absent (lazy-open palaces that no caller has touched yet).
390        self.state.registry.remove(&PalaceId::new(palace_id));
391        // #4639: drop the cached chat_sessions.redb handle too — otherwise the
392        // fd survives `remove_dir_all` and pins the deleted inode forever.
393        self.state.session_stores.remove(palace_id);
394        // Issue #228: drop the palace-name cache entry so future writes never
395        // resolve to a stale label.
396        self.state.palace_names.remove(palace_id);
397        let palace_dir = self.state.data_root.join(palace_id);
398        tokio::fs::remove_dir_all(&palace_dir).await.map_err(|e| {
399            ServiceError::internal(format!("remove palace dir {}: {e}", palace_dir.display()))
400        })?;
401        // Recompute aggregate totals so dashboards drop the deleted palace's
402        // counts. There's no dedicated `PalaceDeleted` event variant yet;
403        // `StatusChanged` is enough to keep the UI in sync.
404        self.state.emit(self.aggregate_status_event());
405        Ok(())
406    }
407
408    /// Rename a palace's display name without touching its data.
409    ///
410    /// Why: Operators need to fix typos and rebrand palaces without dropping
411    /// the underlying drawers / vectors / KG. The palace id (the directory
412    /// name on disk) is immutable — only the human-readable `name` field in
413    /// `palace.json` changes — so cached `PalaceHandle`s stay valid and no
414    /// registry invalidation is required.
415    /// What: 1) loads the palace via `PalaceStore::load_palace` (404 when the
416    /// directory or `palace.json` is genuinely missing; a probe that cannot
417    /// determine whether it is there is a 500, not a 404 — #5549), 2) trims the
418    /// new name and
419    /// returns `BadRequest` when empty, 3) mutates `palace.name` and writes
420    /// the metadata back through the atomic `PalaceStore::save_palace`
421    /// (tmp file + rename), 4) emits an aggregate `StatusChanged` so
422    /// dashboards re-render the relabelled palace, 5) returns the updated
423    /// palace as JSON (enriched with the live handle stats, so callers see
424    /// drawer/vector/KG counts in the same shape as `GET /palaces/{id}`).
425    /// Test: `update_palace_name_renames_palace`,
426    /// `update_palace_name_rejects_empty_name`,
427    /// `update_palace_name_returns_not_found_for_missing_id` in `web::tests`.
428    pub async fn update_palace_name(&self, palace_id: &str, name: &str) -> Result<Value> {
429        let trimmed = name.trim();
430        if trimmed.is_empty() {
431            return Err(anyhow!("name must be non-empty after trimming"));
432        }
433        let palace_dir = self.state.data_root.join(palace_id);
434        let mut palace = trusty_common::memory_core::store::PalaceStore::load_palace(&palace_dir)
435            .map_err(|e| {
436            // #5549: only a genuine absence may be reported as "not found".
437            if matches!(&e, PalaceStoreError::NotFound(_)) {
438                anyhow!("palace not found: {palace_id} ({e})")
439            } else {
440                anyhow!("cannot load palace {palace_id}: {e}")
441            }
442        })?;
443        palace.name = trimmed.to_string();
444        trusty_common::memory_core::store::PalaceStore::save_palace(&palace)
445            .with_context(|| format!("save palace metadata for {palace_id}"))?;
446        // Issue #228: refresh the in-memory name cache so subsequent writes
447        // surface the new label without a disk walk.
448        self.state
449            .palace_names
450            .insert(palace_id.to_string(), trimmed.to_string());
451        let handle = self
452            .state
453            .registry
454            .open_palace(&self.state.data_root, &palace.id)
455            .ok();
456        // #7106: enrichment runs on the blocking pool.
457        let info = palace_info_blocking(&palace, handle).await;
458        self.state.emit(self.aggregate_status_event());
459        serde_json::to_value(info).context("serialize palace info")
460    }
461
462    /// Typed variant of [`Self::update_palace_name`] used by the HTTP handler.
463    ///
464    /// Why: HTTP needs to distinguish 400 (empty name) from 404 (missing
465    /// palace) so the right status code is emitted; the chat / MCP tool
466    /// only cares about a `Result<Value>` because both errors are surfaced
467    /// as opaque MCP error strings. Keeping a typed variant alongside the
468    /// untyped one keeps the wire shape correct on both surfaces without
469    /// asking either caller to parse error strings.
470    /// What: same as [`Self::update_palace_name`] but returns
471    /// `ServiceError::BadRequest` for empty names and `ServiceError::NotFound`
472    /// for palace metadata that is genuinely absent. Metadata whose presence
473    /// cannot be determined — a denied or transient stat — is
474    /// `ServiceError::Internal`: a 404 would tell the client the palace does
475    /// not exist when nobody established that (#5549, ADR-0045).
476    /// Test: `update_palace_name_renames_palace`,
477    /// `update_palace_name_rejects_empty_name`,
478    /// `update_palace_name_returns_not_found_for_missing_id`,
479    /// `update_palace_name_reports_an_unstattable_palace_as_internal`.
480    pub async fn update_palace_name_typed(
481        &self,
482        palace_id: &str,
483        name: &str,
484    ) -> ServiceResult<Value> {
485        let trimmed = name.trim();
486        if trimmed.is_empty() {
487            return Err(ServiceError::bad_request(
488                "name must be non-empty after trimming",
489            ));
490        }
491        let palace_dir = self.state.data_root.join(palace_id);
492        let mut palace = trusty_common::memory_core::store::PalaceStore::load_palace(&palace_dir)
493            .map_err(|e| {
494            // #5549: `not_found` on every variant told the client the palace
495            // does not exist for a stat we were merely denied.
496            if matches!(&e, PalaceStoreError::NotFound(_)) {
497                ServiceError::not_found(format!("palace not found: {palace_id} ({e})"))
498            } else {
499                ServiceError::internal(format!("cannot load palace {palace_id}: {e}"))
500            }
501        })?;
502        palace.name = trimmed.to_string();
503        trusty_common::memory_core::store::PalaceStore::save_palace(&palace).map_err(|e| {
504            ServiceError::internal(format!("save palace metadata for {palace_id}: {e}"))
505        })?;
506        // Issue #228: refresh the in-memory name cache so subsequent writes
507        // surface the new label without a disk walk.
508        self.state
509            .palace_names
510            .insert(palace_id.to_string(), trimmed.to_string());
511        let handle = self
512            .state
513            .registry
514            .open_palace(&self.state.data_root, &palace.id)
515            .ok();
516        // #7106: enrichment runs on the blocking pool.
517        let info = palace_info_blocking(&palace, handle).await;
518        self.state.emit(self.aggregate_status_event());
519        serde_json::to_value(info)
520            .map_err(|e| ServiceError::internal(format!("serialize palace info: {e}")))
521    }
522
523    /// Look up a single palace by id and enrich with live handle stats.
524    ///
525    /// Why: distinct 404 vs. 500 path is needed by both HTTP and chat callers.
526    /// What: returns `NotFound` when the id is unknown, otherwise a fully
527    /// populated `PalaceInfo`.
528    /// Test: indirectly via `health_endpoint_round_trip_with_palace_is_ok`.
529    pub async fn get_palace(&self, id: &str) -> ServiceResult<PalaceInfo> {
530        let palaces = PalaceRegistry::list_palaces(&self.state.data_root)
531            .map_err(|e| ServiceError::internal(format!("list palaces: {e:#}")))?;
532        let palace = palaces
533            .into_iter()
534            .find(|p| p.id.0 == id)
535            .ok_or_else(|| ServiceError::not_found(format!("palace not found: {id}")))?;
536        let handle = self
537            .state
538            .registry
539            .open_palace(&self.state.data_root, &palace.id)
540            .ok();
541        // #7106: enrichment runs on the blocking pool.
542        Ok(palace_info_blocking(&palace, handle).await)
543    }
544
545    // -----------------------------------------------------------------
546    // Drawers
547    // -----------------------------------------------------------------
548
549    /// List drawers in a palace with optional room/tag filters and pagination.
550    ///
551    /// Why: deduplicates the open-handle + listing path between HTTP and chat,
552    /// and (issue #184) lets the TUI activity panel page through drawers in
553    /// creation-date order without breaking the importance-sorted default the
554    /// legacy callers rely on.
555    /// What: opens the palace handle, fetches a window of drawers, optionally
556    /// re-sorts by `created_at` descending when `sort = "created_desc"`
557    /// (leaving the importance-desc default untouched), then drops the
558    /// leading `offset` rows and keeps `limit`. For `created_desc` the
559    /// window must cover the full filtered set (otherwise the importance
560    /// pre-sort hides truly-recent low-importance drawers), so the window
561    /// is widened to a sane ceiling (`MAX_DRAWER_WINDOW`); the default
562    /// importance path keeps a tight `limit+offset` window.
563    /// Returns the serialised JSON array.
564    /// Test: `service::tests::list_drawers_creates_desc_paginates`.
565    pub async fn list_drawers(&self, id: &str, q: ListDrawersQuery) -> ServiceResult<Value> {
566        const MAX_DRAWER_WINDOW: usize = 10_000;
567        let handle = self.open_handle(id)?;
568        let room = q.room.as_deref().map(RoomType::parse);
569        let limit = q.limit.unwrap_or(50);
570        let offset = q.offset.unwrap_or(0);
571        let by_created = matches!(q.sort.as_deref(), Some("created_desc"));
572        // For created_desc the importance pre-sort would hide low-importance
573        // drawers that happen to be the most recent, so we need to fetch the
574        // full filtered set (capped at MAX_DRAWER_WINDOW). For importance
575        // ordering the legacy `limit + offset` window is sufficient.
576        let window = if by_created {
577            MAX_DRAWER_WINDOW
578        } else {
579            limit.saturating_add(offset).min(MAX_DRAWER_WINDOW)
580        };
581        let mut drawers = handle.list_drawers(room, q.tag.clone(), window);
582        if by_created {
583            drawers.sort_by_key(|d| std::cmp::Reverse(d.created_at));
584        }
585        let page: Vec<_> = drawers.into_iter().skip(offset).take(limit).collect();
586        // Issue #202: enrich every row with a short `snippet` derived from
587        // the drawer's content so the TUI activity panel can render a
588        // glanceable summary without re-parsing the full body. The
589        // snippet is whitespace-collapsed and bounded at
590        // `DRAWER_SNIPPET_MAX_CHARS` (60) — shorter than the SSE preview
591        // because the activity panel renders it on a single narrow row.
592        let payload: Vec<Value> = page
593            .into_iter()
594            .map(|drawer| {
595                let snippet = drawer_snippet(drawer.content());
596                let mut value = serde_json::to_value(&drawer).unwrap_or_else(|_| json!({}));
597                if let Value::Object(ref mut map) = value {
598                    // `null` when the drawer has no usable content so
599                    // clients can distinguish "no body" from "empty body
600                    // after whitespace collapse".
601                    let snippet_value = if snippet.is_empty() {
602                        Value::Null
603                    } else {
604                        Value::String(snippet)
605                    };
606                    map.insert("snippet".to_string(), snippet_value);
607                }
608                value
609            })
610            .collect();
611        Ok(Value::Array(payload))
612    }
613
614    /// Store a new drawer and emit the matching activity events.
615    ///
616    /// Why: HTTP and chat both need the auto-KG-extraction follow-up; this
617    /// method keeps that side-effect chain in one place.
618    /// What: opens the palace, stores the drawer via
619    /// `PalaceHandle::remember_with_options` (issue #3225: `body.force`
620    /// threads through as `RememberOptions::force`, letting a caller bypass
621    /// the QUALITY gates only — `allow_secret_like` is left at its default
622    /// `false`, so secret detection always still runs, `force` or not),
623    /// emits `DrawerAdded` + `StatusChanged`, then triggers
624    /// `tools::auto_extract_and_assert`. Returns the new drawer id.
625    /// Test: `http_create_drawer_runs_auto_kg_extraction`,
626    /// `create_drawer_rejects_json_content_without_force`,
627    /// `create_drawer_force_bypasses_quality_gate_for_json_content`.
628    pub async fn create_drawer(
629        &self,
630        id: &str,
631        body: CreateDrawerBody,
632        creator: CreatorInfo,
633        source: ActivitySource,
634    ) -> ServiceResult<Uuid> {
635        let handle = self.open_handle(id)?;
636        let room = body
637            .room
638            .as_deref()
639            .map(RoomType::parse)
640            .unwrap_or(RoomType::General);
641        let importance = body.importance.unwrap_or(0.5);
642        let force = body.force.unwrap_or(false);
643        let content_preview = drawer_content_preview(&body.content);
644        let mut tags_with_creator = body.tags;
645        // Issue #202: project a bare-UUID session tag (when the caller
646        // passed one in the request body) into the reserved
647        // `creator:session=<first-8>` slot so the activity panel can
648        // surface session attribution without bespoke parsing.
649        if let Some(session_tag) = crate::attribution::session_tag_from_tags(&tags_with_creator) {
650            tags_with_creator.push(session_tag);
651        }
652        creator.merge_into(&mut tags_with_creator);
653        let content_for_kg = body.content.clone();
654        let tags_for_kg = tags_with_creator.clone();
655        let room_label_for_kg = crate::tools::room_label(&room);
656        let drawer_id = handle
657            .remember_with_options(
658                body.content,
659                room,
660                tags_with_creator,
661                importance,
662                RememberOptions {
663                    force,
664                    ..Default::default()
665                },
666            )
667            .await
668            .map_err(|e| ServiceError::internal(format!("remember: {e:#}")))?;
669        let drawer_count = handle.drawers.read().len();
670        // Issue #228: resolve from the in-memory cache instead of re-walking
671        // the data root on every HTTP `create_drawer` call. Same cache the
672        // MCP `lookup_palace_name` helper consults.
673        let palace_name = self
674            .state
675            .palace_names
676            .get(id)
677            .map(|entry| entry.value().clone())
678            .unwrap_or_else(|| id.to_string());
679        self.state.emit(DaemonEvent::DrawerAdded {
680            palace_id: id.to_string(),
681            palace_name,
682            drawer_count,
683            timestamp: chrono::Utc::now(),
684            content_preview,
685            source,
686        });
687        // Issue #228: do NOT emit `StatusChanged` on every drawer create —
688        // the periodic ticker (`run_http_on`) refreshes aggregate totals on
689        // a fixed cadence so dashboards stay current without an O(N palaces)
690        // recompute on the write hot path.
691        crate::tools::auto_extract_and_assert(
692            &handle,
693            drawer_id,
694            &content_for_kg,
695            &tags_for_kg,
696            room_label_for_kg.as_deref(),
697        )
698        .await;
699        Ok(drawer_id)
700    }
701
702    /// Forget (delete) a drawer and emit the matching events.
703    ///
704    /// Why: same dedup story as `create_drawer`. #5231: `DELETE` on a drawer id
705    /// that was never stored used to answer `204 No Content`, the same as a
706    /// real delete — this now 404s, matching `delete_palace`.
707    /// What: parses the drawer UUID, calls `PalaceHandle::forget`, deletes the
708    /// drawer's BM25 document, maps `ForgetOutcome::NotFound` to
709    /// `ServiceError::not_found`, and emits `DrawerDeleted` only when a drawer
710    /// was actually removed. #5053: the lexical delete runs on this path for
711    /// the same reason it runs on the MCP one — `HTTP DELETE` and
712    /// `memory_forget` remove the same drawer, and the backfill indexes it
713    /// whichever way it was written, so a lexical copy left here is the same
714    /// stale document.
715    /// Test: `delete_drawer_404s_for_an_unknown_drawer_id`;
716    /// `tests/bm25_forget_delete.rs` covers the deletion contract itself.
717    pub async fn delete_drawer(
718        &self,
719        id: &str,
720        drawer_id: &str,
721        source: ActivitySource,
722    ) -> ServiceResult<()> {
723        let handle = self.open_handle(id)?;
724        let uuid = Uuid::parse_str(drawer_id)
725            .map_err(|_| ServiceError::bad_request("drawer_id must be a UUID"))?;
726        let outcome = handle
727            .forget(uuid)
728            .await
729            .map_err(|e| ServiceError::internal(format!("forget: {e:#}")))?;
730        // #5053: a drawer the user deleted must stop matching lexical queries.
731        crate::tools::bm25::bm25_delete_document(&self.state, handle.id.as_str(), uuid)
732            .await
733            .map_err(|e| ServiceError::internal(format!("{e:#}")))?;
734        if !outcome.is_deleted() {
735            return Err(ServiceError::not_found(format!(
736                "drawer '{drawer_id}' not found in palace '{id}'"
737            )));
738        }
739        let drawer_count = handle.drawers.read().len();
740        self.state.emit(DaemonEvent::DrawerDeleted {
741            palace_id: id.to_string(),
742            drawer_count,
743            source,
744        });
745        // Issue #228: skip the per-write `StatusChanged` emit — the
746        // periodic ticker handles aggregate roll-ups.
747        Ok(())
748    }
749
750    // -----------------------------------------------------------------
751    // Recall
752    // -----------------------------------------------------------------
753
754    /// Per-palace recall (semantic search), optionally with deep retrieval.
755    ///
756    /// Why: HTTP and chat tools both perform the same fan-out logic.
757    /// What: opens the palace handle and dispatches to the shallow or deep
758    /// recall helper. Returns a JSON array of flattened drawer rows (the
759    /// `recall_entry_json` shape from issue #69).
760    /// Test: `recall_entry_json_hoists_drawer_fields`.
761    pub async fn recall(
762        &self,
763        id: &str,
764        query: &str,
765        top_k: usize,
766        deep: bool,
767    ) -> ServiceResult<Value> {
768        let handle = self.open_handle(id)?;
769        let mut results = if deep {
770            recall_deep_with_default_embedder(&handle, query, top_k).await
771        } else {
772            recall_with_default_embedder(&handle, query, top_k).await
773        }
774        .map_err(|e| ServiceError::internal(format!("recall: {e:#}")))?;
775        // #5036: the lexical lane, on the path the UserPromptSubmit hook
776        // actually takes. `handle_memory_recall` has run vector and BM25 in
777        // parallel and RRF-fused them since #156; this route reached
778        // `retrieval::layers` directly and was vector-only, so a prompt with no
779        // lexical counterweight retrieved by vector centroid alone.
780        //
781        // Keyed on the RESOLVED palace id, never the caller's slug —
782        // `open_handle` follows aliases, and the corpus the backfill wrote
783        // belongs to the resolved palace.
784        //
785        // Reuses `fuse_bm25_into_recall` rather than deriving a second scorer:
786        // it only BOOSTS drawers the vector lane already returned and never
787        // promotes a BM25-only hit, so it has no scaling constant that can
788        // degenerate when the surviving set is empty — the failure that folded
789        // three earlier attempts at this wiring.
790        if let Some(hits) =
791            crate::tools::bm25::bm25_search_optional(&self.state, handle.id.as_str(), query, top_k)
792                .await
793        {
794            crate::tools::bm25::fuse_bm25_into_recall(&mut results, &hits, top_k);
795        }
796        let payload: Vec<Value> = results.into_iter().map(recall_entry_json).collect();
797        Ok(json!(payload))
798    }
799
800    /// Cross-palace recall.
801    ///
802    /// Why: shared between `/api/v1/recall` and the `memory_recall_all` chat
803    /// tool. Encapsulating the open-everything-fanout-merge dance avoids
804    /// drift.
805    /// What: lists every palace, then streams them through `recall_streamed`
806    /// in bounded batches, delegating each batch to
807    /// `recall_across_palaces_with_default_embedder`. Returns a JSON array.
808    /// Why (issue #4637): unlike `list_palaces`/`status`, this route is NOT
809    /// converted to `peek()`. A cross-palace recall that answered from
810    /// cache-resident palaces only would silently omit ~98.9% of the corpus —
811    /// a wrong answer that looks like a right one, which is strictly worse
812    /// than a slow correct one. Every palace is still opened and still
813    /// searched; what changed is when.
814    /// Why (issue #7125): opening all of them AT ONCE made peak residency and
815    /// the post-call LRU residue both scale with the palace count. The batch
816    /// walk bounds the peak at `RECALL_PALACE_BATCH` and hands back everything
817    /// the query itself brought in, so the daemon's steady state after a
818    /// recall-all matches its steady state before one.
819    /// Test: indirectly via `recall_across_palaces_merges_results` and the
820    /// MCP `memory_recall_all` integration paths;
821    /// `open_palaces_blocking_opens_every_palace` pins that uncached palaces
822    /// are still searched; `recall_all_returns_open_palaces_to_baseline` pins
823    /// the residency bound.
824    pub async fn recall_all(&self, query: &str, top_k: usize, deep: bool) -> Value {
825        let palaces = match list_palaces_blocking(&self.state).await {
826            Ok(v) => v,
827            Err(e) => return json!({ "error": format!("{e:#}") }),
828        };
829        // #7125: stream the estate in batches instead of opening all of it.
830        let streamed = recall_streamed(
831            &self.state,
832            &palaces,
833            "recall_all",
834            top_k,
835            |handles| async move {
836                recall_across_palaces_with_default_embedder(&handles, query, top_k, deep).await
837            },
838        )
839        .await;
840        match streamed {
841            Ok(results) => json!(results
842                .into_iter()
843                .map(|r| json!({
844                    "palace_id": r.palace_id,
845                    "drawer_id": r.result.drawer.id.to_string(),
846                    "content": r.result.drawer.content(),
847                    "importance": r.result.drawer.importance,
848                    "tags": r.result.drawer.tags,
849                    "score": r.result.score,
850                    "layer": r.result.layer,
851                }))
852                .collect::<Vec<_>>()),
853            Err(e) => json!({ "error": format!("recall_across_palaces: {e:#}") }),
854        }
855    }
856}