trusty-memory 0.27.0

MCP server (stdio + Unix socket) for trusty-memory
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
//! Palace-tool handlers for the trusty-memory MCP surface.
//!
//! Why: the `palace_*` tool handlers (create/list/delete/update/info/compact)
//! form one cohesive group split out of the former monolithic `tools.rs`
//! (issue #607).
//! What: per-tool `pub(crate) async fn handle_*` functions moved verbatim;
//! visibility widened to `pub(crate)` so the dispatcher (in `tools::mod`) and
//! the test module can reach them.
//! Test: `dispatch_palace_create_persists`, `dispatch_palace_delete_*`,
//! `dispatch_palace_update_*` in `tools::tests`.

use crate::{ActivitySource, AppState, DaemonEvent};
use anyhow::{anyhow, Context, Result};
use serde_json::{json, Value};
use trusty_common::memory_core::palace::{Palace, PalaceId};
use uuid::Uuid;

use super::helpers::{open_palace_handle, resolve_palace};
// #6318: `palace_info` reads, so it falls back to a palace index; `palace_compact`,
// `palace_reembed` and `palace_unalias` can write, so they keep the error.
use super::palace_index::{resolve_palace_or_index, PalaceScope};

/// Validate that a palace slug is a safe, well-formed filesystem name.
///
/// Why: `force=true` bypasses the project-slug enforcement gate but must not
/// allow arbitrary strings that could cause path traversal, redb table-name
/// collisions, or filesystem issues. This guard runs unconditionally.
///
/// The rule itself is [`trusty_common::palace_id::palace_id_is_valid`], not a
/// copy of it. This function used to restate the shape independently, and the
/// deriving side in `trusty_common::palace_id` stated a different one — it had
/// no length cap — so a long repo or directory name derived an id this gate
/// refused, and trusty-code's turn recorder stayed fail-open for that project
/// (#2443). One statement of the rule is what stops the two drifting again.
/// What: delegates the accept/reject decision, then names which half failed —
/// length or character shape — in the error text callers already match on.
/// Test: `force_flag_rejects_unsafe_slugs`, `a_derived_palace_id_is_accepted`
/// (both in `tests/palace_force.rs`).
fn validate_slug_format(slug: &str) -> Result<()> {
    use trusty_common::palace_id::{palace_id_is_valid, PALACE_ID_MAX_LEN};

    if palace_id_is_valid(slug) {
        return Ok(());
    }
    if slug.is_empty() || slug.len() > PALACE_ID_MAX_LEN {
        return Err(anyhow!(
            "palace slug must be 1–{PALACE_ID_MAX_LEN} characters (got {}): {slug:?}",
            slug.len()
        ));
    }
    Err(anyhow!(
        "palace slug must match [a-z0-9][a-z0-9-]{{0,62}} \
         (lowercase letters, digits, hyphens only): {slug:?}"
    ))
}

pub(crate) async fn handle_palace_create(state: &AppState, args: Value) -> Result<Value> {
    let palace_name = args
        .get("name")
        .and_then(|v| v.as_str())
        .ok_or_else(|| anyhow!("palace_create: missing 'name'"))?;

    // Issue #88 / Change 2: enforce palace = project mapping. New palaces must
    // be named after the current project slug (derived by walking up from CWD)
    // or the special `personal` sentinel. Existing palaces are unaffected —
    // this gate only applies to NEW creation requests.
    //
    // The validation cwd is, in order of preference:
    //   a. `args["cwd"]` — the MCP caller's project path. When present and the
    //      project has a `.trusty-tools/trusty-memory.yaml` pin file, the
    //      pinned slug is used for validation (correct even after a drive reorg).
    //   b. `std::env::current_dir()` — daemon's own cwd, pre-Change-2 fallback.
    //
    // Skip enforcement when invoked from a test context (tests use arbitrary
    // names against tempdir roots that are not real projects). The bypass is
    // keyed on an env var (`TRUSTY_SKIP_PALACE_ENFORCEMENT=1`) that tests set
    // locally; production deployments never set it.
    // spec-001 / Phase 1: `force=true` bypasses slug validation so an
    // application can create a palace under an arbitrary slug (e.g. one palace
    // per app/tenant for chat-session storage). The env-var bypass remains for
    // test contexts; either short-circuits the same validation call.
    let force = args.get("force").and_then(|v| v.as_bool()).unwrap_or(false);
    let skip_enforcement = std::env::var("TRUSTY_SKIP_PALACE_ENFORCEMENT").as_deref() == Ok("1");
    // Issue #1714: `force=true` is a privileged bypass; gate it behind the
    // minimal authz seam (no-op in the default single-tenant mode, fails
    // closed in multi-tenant mode). See `crate::authz` for the full design.
    if force {
        crate::authz::authorize_force_palace_create(state)?;
    }
    // Even when `force=true`, validate that the slug is a safe filesystem name:
    // lowercase letters, digits, and hyphens only; must start with a letter or
    // digit; max 63 chars. This prevents path traversal and redb table-name
    // collisions regardless of the project-slug enforcement bypass.
    // The test-context bypass (TRUSTY_SKIP_PALACE_ENFORCEMENT=1) also skips
    // the format gate so unit tests that use historical slug names keep passing.
    if !skip_enforcement {
        validate_slug_format(palace_name)?;
    }
    if !skip_enforcement && !force {
        let cwd = args
            .get("cwd")
            .and_then(|v| v.as_str())
            .filter(|s| !s.is_empty())
            .map(std::path::Path::new)
            .map(|p| p.to_path_buf())
            .or_else(|| std::env::current_dir().ok())
            .unwrap_or_else(|| state.data_root.clone());
        crate::project_root::validate_palace_name(palace_name, &cwd)?;
    }

    let description = args
        .get("description")
        .and_then(|v| v.as_str())
        .map(|s| s.to_string());
    let palace = Palace {
        id: PalaceId::new(palace_name),
        name: palace_name.to_string(),
        description,
        created_at: chrono::Utc::now(),
        data_dir: state.data_root.join(palace_name),
    };
    let _handle = state
        .registry
        .create_palace(&state.data_root, palace)
        .context("create_palace")?;
    // Issue #228: keep the in-memory palace-name cache in sync so
    // subsequent writes can resolve the friendly name without a disk
    // walk. The id == name pairing matches what the registry persisted.
    state
        .palace_names
        .insert(palace_name.to_string(), palace_name.to_string());
    // Issue #96: emit so MCP-driven palace creation lands in the
    // dashboard activity feed alongside HTTP-origin creates.
    state.emit(DaemonEvent::PalaceCreated {
        id: palace_name.to_string(),
        name: palace_name.to_string(),
        source: ActivitySource::Mcp,
    });
    // Issue #60: auto-seed the KG with temporal metadata so every
    // new palace has at least `created_at` + `bootstrapped_at`
    // triples anchored to the palace name. We deliberately do NOT
    // pass a project_path here — that requires an explicit user
    // decision (which directory belongs to this palace?). Failures
    // are non-fatal: the palace was already created, and the user
    // can re-run `kg_bootstrap` manually if needed.
    let bootstrap_summary = match crate::bootstrap::bootstrap_palace(state, palace_name, None).await
    {
        Ok(r) => Some(serde_json::json!({
            "triples_asserted": r.triples_asserted,
            "project_subject": r.project_subject,
        })),
        Err(e) => {
            tracing::warn!(
                palace = %palace_name,
                "auto-bootstrap on palace_create failed: {e:#}",
            );
            None
        }
    };
    Ok(json!({
        "palace_id": palace_name,
        "status": "created",
        "bootstrap": bootstrap_summary,
    }))
}

pub(crate) async fn handle_palace_list(state: &AppState, _args: Value) -> Result<Value> {
    let root = state.data_root.clone();
    let palaces = tokio::task::spawn_blocking(move || {
        trusty_common::memory_core::PalaceRegistry::list_palaces(&root)
    })
    .await
    .context("join list_palaces")??;
    let ids: Vec<String> = palaces.iter().map(|p| p.id.as_str().to_string()).collect();
    Ok(json!({ "palaces": ids }))
}

pub(crate) async fn handle_palace_delete(state: &AppState, args: Value) -> Result<Value> {
    // Issue #180: full palace teardown. The HTTP layer is the
    // canonical implementation; we just delegate to the same
    // `MemoryService::delete_palace` method to keep behaviour
    // (and the conflict / not-found / 204 split) identical
    // across surfaces. ServiceError variants are folded into
    // anyhow here so the MCP wire shape matches every other
    // tool's error contract.
    let palace_id = args
        .get("palace_id")
        .and_then(|v| v.as_str())
        .ok_or_else(|| anyhow!("palace_delete: missing 'palace_id'"))?
        .to_string();
    let force = args.get("force").and_then(|v| v.as_bool()).unwrap_or(false);
    use crate::service::{MemoryService, ServiceError};
    let svc = MemoryService::new(state.clone());
    match svc.delete_palace(&palace_id, force).await {
        Ok(()) => Ok(json!({ "deleted": palace_id })),
        Err(ServiceError::NotFound(_)) => Err(anyhow!("Palace not found: {palace_id}")),
        Err(ServiceError::Conflict(msg)) => Err(anyhow!(msg)),
        Err(e) => Err(anyhow!("palace_delete: {e}")),
    }
}

pub(crate) async fn handle_palace_update(state: &AppState, args: Value) -> Result<Value> {
    // Issue #180 follow-up: rename a palace's display name. The HTTP
    // layer is the canonical implementation; we delegate to the
    // same `MemoryService::update_palace_name` so the
    // load-mutate-save-emit chain stays consistent across surfaces.
    // The MCP wire shape is the minimal acknowledgement payload —
    // callers needing the enriched palace info should use
    // `palace_info` (or the HTTP endpoint, which returns the full
    // shape).
    let palace_id = args
        .get("palace_id")
        .and_then(|v| v.as_str())
        .ok_or_else(|| anyhow!("palace_update: missing 'palace_id'"))?
        .to_string();
    let name = args
        .get("name")
        .and_then(|v| v.as_str())
        .ok_or_else(|| anyhow!("palace_update: missing 'name'"))?
        .to_string();
    use crate::service::MemoryService;
    let svc = MemoryService::new(state.clone());
    match svc.update_palace_name(&palace_id, &name).await {
        Ok(_info) => Ok(json!({ "updated": palace_id, "name": name.trim() })),
        Err(e) => Err(anyhow!("palace_update: {e}")),
    }
}

pub(crate) async fn handle_palace_info(state: &AppState, args: Value) -> Result<Value> {
    // #6318: no palace and no default is answered with an index, not an error.
    let palace = match resolve_palace_or_index(state, &args, "palace_info").await? {
        PalaceScope::Palace(p) => p,
        PalaceScope::Index(index) => return Ok(index),
    };
    let handle = open_palace_handle(state, &palace)?;
    let drawer_count = handle.list_drawers(None, None, usize::MAX).len();
    let data_dir = handle
        .data_dir
        .as_ref()
        .map(|p| p.to_string_lossy().to_string());
    // ADR-0027 D6 / #4807: report how many rooms the palace has. #4809 (T9)
    // adds `wing_count` alongside it — #4807 left it out rather than report a
    // constant, and the Wing entity it was waiting for is now here, so this is
    // a real count from the `WINGS` registry.
    let store = handle.kg.store();
    let (room_count, wing_count) = tokio::task::spawn_blocking(move || {
        anyhow::Ok((store.list_rooms()?.len(), store.list_wings()?.len()))
    })
    .await
    .context("join palace_info counts")?
    .context("count rooms and wings")?;
    // #6424: the durable last-used stamp, null for a palace never used since
    // the stamp shipped. Additive — no existing field moves.
    let last_used_unix = handle
        .data_dir
        .as_ref()
        .and_then(|d| crate::palace_last_used::read(d));
    Ok(json!({
        "id": handle.id.as_str(),
        "name": handle.id.as_str(),
        "drawer_count": drawer_count,
        "room_count": room_count,
        "wing_count": wing_count,
        "data_dir": data_dir,
        "last_used_unix": last_used_unix,
    }))
}

/// `palace_reembed` — report, and optionally repair, drawers with no vector.
///
/// Why (#4906): the deferred-embed lane used to drop failures silently, leaving
/// drawers durable in redb and permanently invisible to vector recall — 39 of
/// 1,241 on the live `trusty-tools` palace. Fixing the write path forward
/// repairs none of those, and the repair has to run INSIDE the daemon: the
/// daemon holds the palace's writer lock, so a CLI would only ever get a
/// read-only snapshot it cannot write vectors into.
/// What: `dry_run` (the default) returns the exact set of vectorless drawer ids
/// without touching the embedder; `dry_run: false` re-embeds them through the
/// same primitive the write path uses. Idempotent — a second run over a
/// repaired palace reports zero missing and does no work.
/// Test: `dispatch_palace_reembed_dry_run_reports_counts` in `tools::tests`.
pub(crate) async fn handle_palace_reembed(state: &AppState, args: Value) -> Result<Value> {
    use trusty_common::memory_core::retrieval::VectorBackfillOptions;
    let palace = resolve_palace(state, &args, "palace_reembed")?;
    let handle = open_palace_handle(state, &palace)?;
    // Defaults to a dry run on purpose: the first thing an operator wants is
    // the number, and #4834 deletes source files on the strength of it.
    let dry_run = args
        .get("dry_run")
        .and_then(|v| v.as_bool())
        .unwrap_or(true);
    let limit = args
        .get("limit")
        .and_then(|v| v.as_u64())
        .map(|n| n as usize);
    let report = handle
        .backfill_missing_vectors(VectorBackfillOptions {
            dry_run,
            limit,
            ..Default::default()
        })
        .await?;
    let health = handle.embed_health();
    Ok(json!({
        "palace": report.palace_id,
        "dry_run": report.dry_run,
        "drawer_count": report.drawer_count,
        "vector_count": report.vector_count,
        "missing": report.missing,
        "attempted": report.attempted,
        "repaired": report.repaired,
        "still_failing": report.still_failing,
        "still_missing_ids": report.still_missing_ids
            .iter().map(|i| i.to_string()).collect::<Vec<_>>(),
        // Without this a shortfall is unexplained: "no embedder on this host"
        // reads identically to "the embedder is dropping writes".
        "embedder_ready": health.embedder_ready,
        "recorded_failures": health.recorded_failures.len(),
        // #5005 / #5000: `missing` counts drawers with no vector key. An
        // aliased drawer HAS a key and is still unretrievable, so `missing: 0`
        // was a false all-clear on the palace that lost four of them. Gate
        // deletions on `alias_audit == "clean"` as well as `missing == 0`:
        // `"unavailable"` means the scan failed and nothing is known, which is
        // a block, not a pass. `vector_key_rows` / `distinct_vector_ids` are
        // null in that case rather than 0, so no zero can be misread as clean.
        "alias_audit": alias_audit_state(&report.alias_audit),
        "alias_audit_error": report.alias_audit.unavailable_reason(),
        "vector_key_rows": report.alias_audit.counts().map(|(rows, _)| rows),
        "distinct_vector_ids": report.alias_audit.counts().map(|(_, ids)| ids),
        // `aliased` reported 0 for an unreadable audit in the first cut, while
        // the two fields above it correctly reported null. Every count-shaped
        // field in this object is now absent rather than zero when nothing was
        // read — a lone zero is exactly the misreading #5005 documents.
        "aliased": report.alias_audit.aliased_drawer_ids().map(<[Uuid]>::len),
        "aliased_ids": report.alias_audit.aliased_drawer_ids()
            .map(|ids| ids.iter().map(|i| i.to_string()).collect::<Vec<_>>()),
    }))
}

/// `palace_unalias` — free drawers destroyed by a vector-id collision so a
/// re-embed can repair them.
///
/// Why (#5005): the allocator fix stops NEW aliasing and `palace_reembed` now
/// makes existing aliasing visible, but neither repairs it — `unalias` had no
/// caller at all, so an operator could see the damage and not act on it. The
/// three drawers still blocking #4834 need this surface. It runs inside the
/// daemon for the same reason `palace_reembed` does: the daemon holds the
/// palace's writer lock, so a CLI would only get a read-only snapshot.
/// What: `dry_run` (the default) names the exact drawer ids it would free and
/// writes nothing. `dry_run: false` frees the whole collision group, then
/// re-audits — `outcome` is `"repaired"` only when that verification ran and
/// came back clean. Idempotent: a second run reports `"clean"` and frees
/// nothing. The freed drawers still need a `palace_reembed` run to become
/// findable, which `reembed_required` says outright.
///
/// 🔴 `outcome` is the field to branch on, never `freed_ids.len()`. `"partial"`
/// and `"unavailable"` both carry ids and neither is a success.
/// Test: `dispatch_palace_unalias_dry_run_names_ids_and_writes_nothing`, and
/// `dispatch_palace_unalias_frees_a_real_collision_and_is_idempotent` for the
/// write path (#5005 review: the success path had only ever run empty).
pub(crate) async fn handle_palace_unalias(state: &AppState, args: Value) -> Result<Value> {
    use trusty_common::memory_core::retrieval::{AliasRepairOptions, AliasRepairOutcome};
    let palace = resolve_palace(state, &args, "palace_unalias")?;
    let handle = open_palace_handle(state, &palace)?;
    // Defaults to a dry run for the same reason `palace_reembed` does, and with
    // more at stake: this one deletes vector keys.
    let dry_run = args
        .get("dry_run")
        .and_then(|v| v.as_bool())
        .unwrap_or(true);
    let report =
        tokio::task::spawn_blocking(move || handle.repair_aliases(AliasRepairOptions { dry_run }))
            .await
            .context("join palace_unalias")??;

    let ids = |v: &[Uuid]| v.iter().map(|i| i.to_string()).collect::<Vec<_>>();
    let (still_aliased_ids, not_freed_ids, unparsed_keys) = match &report.outcome {
        AliasRepairOutcome::Partial {
            still_aliased,
            not_freed,
            unparsed_keys,
        } => (
            Some(ids(still_aliased)),
            Some(ids(not_freed)),
            Some(unparsed_keys.clone()),
        ),
        _ => (None, None, None),
    };
    let error = match &report.outcome {
        AliasRepairOutcome::Unavailable { reason } => Some(reason.as_str()),
        _ => None,
    };
    Ok(json!({
        "palace": report.palace_id,
        "dry_run": report.dry_run,
        // Branch on this. `"clean"` and `"repaired"` are the only successes.
        "outcome": report.outcome.as_str(),
        "success": report.outcome.is_success(),
        // The id SET, never a bare count: #5005 was a count reporting all-clear
        // over real loss, and these ids are also the re-embed worklist.
        "freed_ids": ids(&report.freed_ids),
        "aliased_before_ids": report.before.aliased_drawer_ids().map(ids),
        // #5005 review HIGH: a collision group whose keys are not uuids names
        // no drawer, so `aliased_before_ids` can be EMPTY over a real
        // collision. Non-empty here means that id list is short — read
        // `vector_key_rows` vs `distinct_vector_ids` off `palace_reembed`, and
        // branch on `outcome`, never on the id counts.
        "unnameable_keys": report.unnameable_keys.clone(),
        // Present only on a partial repair, which is exactly when a caller must
        // not read the run as done.
        "still_aliased_ids": still_aliased_ids,
        "not_freed_ids": not_freed_ids,
        "unparsed_keys": unparsed_keys,
        "error": error,
        // Freeing a group turns an invisible drawer into an ordinary missing
        // one; only `palace_reembed` makes it retrievable again.
        "reembed_required": report.reembed_required(),
    }))
}

/// One word for how the #5005 alias audit went, for the `palace_reembed` payload.
///
/// Why: a caller has to be able to tell "no drawer is aliased" from "the scan
/// failed and nothing is known" without inspecting counts — the second must
/// never read as the first.
/// What: `"clean"`, `"aliased"`, or `"unavailable"`.
/// Test: `dispatch_palace_reembed_dry_run_reports_counts` in `tools::tests`.
fn alias_audit_state(audit: &trusty_common::memory_core::retrieval::AliasAudit) -> &'static str {
    if audit.unavailable_reason().is_some() {
        "unavailable"
    } else if audit.is_clean() {
        "clean"
    } else {
        "aliased"
    }
}

/// Reclaim orphaned HNSW vector rows. VECTOR INDEX ONLY.
///
/// Why (#6652): the name invites the reading that this shrinks the palace on
/// disk. It does not. This tool works on `index.usearch.redb` and never opens
/// `kg.redb`, which is the file that grows without bound — 342 MB on
/// `trusty-tools` against a 7 MB vector index. Overloading it to also rewrite
/// `kg.redb` would widen a narrowly-documented tool's blast radius with nothing
/// in its name or schema to warn a caller, so the KG rewrite lives behind
/// `palace_dream { compact: true }` and `trusty-memory palace compact` instead.
/// What: delegates to `PalaceHandle::compact_vector_orphans`, unchanged.
/// Test: `palace_compact_description_says_vector_index_only`.
pub(crate) async fn handle_palace_compact(state: &AppState, args: Value) -> Result<Value> {
    let palace = resolve_palace(state, &args, "palace_compact")?;
    let handle = open_palace_handle(state, &palace)?;
    // #6208: route through the handle's locked reclamation. It snapshots the
    // valid-id set and reclaims orphans while holding the palace write mutex,
    // so a concurrent `remember` (vector upserted, drawer not yet registered)
    // cannot have its brand-new vector reclaimed as a false orphan. Snapshotting
    // the valid-ids here without that lock is exactly the window #6208 closes.
    let res = handle.compact_vector_orphans().await?;
    Ok(json!({
        "palace": palace,
        "total_checked": res.total_checked,
        "orphans_removed": res.orphans_removed,
        "index_size_before": res.index_size_before,
        "index_size_after": res.index_size_after,
    }))
}