basis 0.4.3

The basis SDK: workspace discovery, run lifecycle, one event stream, and the two seams. No protocol, no transport, no TTY.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
//! The wrapper: one declaration, presented to mentra as an `ExecutableTool`.
//!
//! What the model sees is an ordinary tool — a name, a description, a JSON
//! schema. What happens when it calls one is that basis writes the input object
//! to a program's stdin and reads its stdout back as the result. Nothing is
//! quoted, escaped, or handed to a shell anywhere on that path, which is the
//! whole point of the binding (see [`super`]).

use std::{
    path::{Path, PathBuf},
    sync::Arc,
    time::Duration,
};

use async_trait::async_trait;
use mentra::tool::{
    ParallelToolContext, RuntimeToolDescriptor, ToolApprovalCategory, ToolAuthorizationPreview,
    ToolCapability, ToolDefinition, ToolDurability, ToolExecutionCategory, ToolExecutor,
    ToolResult,
};
use serde_json::{Value, json};

use crate::subprocess::{self, Completion};

use super::manifest::DeclaredToolSpec;

/// How much later than the tool's own deadline mentra's is set.
///
/// basis's deadline is the one that matters, because it is the one that kills
/// the process: mentra enforces `execution_timeout` by dropping the future, and
/// a `spawn_blocking` future dropped mid-wait abandons the thread rather than
/// stopping the program it is waiting on. So the descriptor still carries a
/// deadline — a turn must not hang on a saturated blocking pool — and it is set
/// deliberately later, so the message the model reads is the one that names the
/// tool and says it was stopped, not the generic one from outside.
const TIMEOUT_BACKSTOP: Duration = Duration::from_secs(5);

/// One tool a workspace declared, wrapped as the thing mentra can run.
///
/// Cheap to clone in the sense that matters: the declaration sits behind an
/// `Arc`, so the per-call clone `spawn_blocking` needs costs a refcount rather
/// than a copy of the schema.
#[derive(Clone)]
pub struct DeclaredTool {
    spec: Arc<DeclaredToolSpec>,
    /// The workspace root the manifest was discovered from — what a relative
    /// `cwd` and a relative program path are resolved against.
    workspace: PathBuf,
    /// The runtime's fixed command environment, which this program receives on
    /// top of the one it inherits. Empty unless the host said otherwise.
    environment: Arc<Vec<(String, String)>>,
}

/// Hand-written for [`DeclaredToolSpec`]'s reason, now with a second field
/// that needs it: the runtime's command environment can hold whatever the host
/// put there, and a derived impl would put every value of it in every `{:?}`
/// of anything holding one. Names survive, because naming a variable is what
/// makes a misconfiguration fixable and repeats nothing that was read.
impl std::fmt::Debug for DeclaredTool {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        f.debug_struct("DeclaredTool")
            .field("spec", &self.spec)
            .field("workspace", &self.workspace)
            .field(
                "environment",
                &self
                    .environment
                    .iter()
                    .map(|(name, _)| (name, "<redacted>"))
                    .collect::<std::collections::BTreeMap<_, _>>(),
            )
            .finish()
    }
}

impl DeclaredTool {
    pub fn new(spec: DeclaredToolSpec, workspace: impl Into<PathBuf>) -> Self {
        Self {
            spec: Arc::new(spec),
            workspace: workspace.into(),
            environment: Arc::new(Vec::new()),
        }
    }

    /// Adds the runtime's fixed command environment to what this program is
    /// spawned with.
    ///
    /// The gap this closes: a host calls
    /// [`RuntimeBuilder::with_command_environment`](crate::RuntimeBuilder::with_command_environment)
    /// to say where its service lives, and reasonably expects every process the
    /// runtime spawns to be told. Commands through
    /// [`spawn`](crate::tools::spawn) were; a declared tool's program was not,
    /// and failed at the far end complaining about a variable the runtime had
    /// been given.
    ///
    /// Separate from [`new`](Self::new) rather than a third argument to it,
    /// because it is the *runtime's* contribution and `new` takes what the
    /// manifest said. [`crate::Workspace`] applies it at open, from the runtime
    /// the workspace borrows.
    #[must_use]
    pub fn with_command_environment(
        self,
        environment: impl IntoIterator<Item = (String, String)>,
    ) -> Self {
        Self {
            environment: Arc::new(environment.into_iter().collect()),
            ..self
        }
    }

    /// The name the model calls and an operator writes in a rule.
    pub fn name(&self) -> &str {
        &self.spec.name
    }

    /// The declaration this was built from.
    pub fn spec(&self) -> &DeclaredToolSpec {
        &self.spec
    }

    /// The runtime pairs this program is spawned with, before the manifest's
    /// own `env` is layered over them.
    pub fn command_environment(&self) -> &[(String, String)] {
        &self.environment
    }
}

impl ToolDefinition for DeclaredTool {
    fn descriptor(&self) -> RuntimeToolDescriptor {
        RuntimeToolDescriptor::builder(&self.spec.name)
            .description(&self.spec.description)
            .input_schema(self.spec.input_schema.clone())
            // `ProcessExec` and nothing else: basis knows a program will run and
            // knows nothing about what it touches, and a capability list that
            // guessed would be worse than one that is merely coarse.
            .capabilities(vec![ToolCapability::ProcessExec])
            .side_effect_level(self.spec.side_effect.level())
            .durability(ToolDurability::Ephemeral)
            // Never batched with anything. basis cannot know what somebody's
            // program writes, so it never lets one run beside another call.
            .execution_category(ToolExecutionCategory::ExclusiveLocalMutation)
            .approval_category(ToolApprovalCategory::Process)
            .execution_timeout(self.spec.timeout() + TIMEOUT_BACKSTOP)
            .build()
    }
}

#[async_trait]
impl ToolExecutor for DeclaredTool {
    /// What the approver sees, and it is deliberately not the default.
    ///
    /// The default preview restates the static descriptor and passes the raw
    /// input through as the structured input, which for this binding would show
    /// an approver the arguments and leave out the only thing they actually
    /// need: *which program is about to run*. A declared tool's name is chosen
    /// by the same file that chooses its command, so the name is not evidence.
    ///
    /// So `structured_input` carries `{tool, command, cwd, input}` — what an
    /// approver renders, what mentra globs a remembered rule's pattern against
    /// (`RuleStore::matching_rule`), and what the audit trail keeps. **`env` is
    /// not in it**, and that is the one asymmetry worth stating: the command
    /// and its arguments are how a spawn is understood, while the environment
    /// is where the credential is, and a preview travels further than a glance.
    fn authorization_preview(
        &self,
        _ctx: &ParallelToolContext,
        input: &Value,
    ) -> Result<ToolAuthorizationPreview, String> {
        preview(&self.spec, &self.workspace, &self.descriptor(), input)
    }

    async fn execute(&self, _ctx: ParallelToolContext, input: Value) -> ToolResult {
        run(
            Arc::clone(&self.spec),
            self.workspace.clone(),
            &self.environment,
            input,
        )
        .await
    }
}

/// Assembles the per-call preview.
///
/// A free function rather than the method's body so the shape an approver is
/// shown can be asserted without a runtime to build a context from — the
/// context is unused here anyway, because everything this presents comes from
/// the declaration rather than from the call's surroundings.
fn preview(
    spec: &DeclaredToolSpec,
    workspace: &Path,
    descriptor: &RuntimeToolDescriptor,
    input: &Value,
) -> Result<ToolAuthorizationPreview, String> {
    // Refused here, ahead of the approver: a call that cannot run is not worth
    // a person's attention, and asking about one teaches them that approving is
    // what makes errors go away.
    check_input(spec, input)?;

    let cwd = spec.working_directory(workspace);

    Ok(ToolAuthorizationPreview {
        capabilities: descriptor.capabilities.clone(),
        side_effect_level: descriptor.side_effect_level,
        durability: descriptor.durability,
        execution_category: descriptor.execution_category,
        approval_category: descriptor.approval_category,
        raw_input: input.clone(),
        structured_input: json!({
            "tool": spec.name,
            "command": spec.command,
            "cwd": cwd,
            "input": input,
        }),
        working_directory: cwd,
    })
}

/// Runs the program and turns what it did into the call's result.
async fn run(
    spec: Arc<DeclaredToolSpec>,
    workspace: PathBuf,
    runtime_environment: &[(String, String)],
    input: Value,
) -> ToolResult {
    // Asked again rather than trusted from the preview: the preview is only
    // reached when an authorizer is installed, and a check that a missing
    // authorizer removes is not a check.
    check_input(&spec, &input)?;

    let payload = serde_json::to_string(&input).map_err(|error| {
        format!(
            "{} was called with input basis could not serialize: {error}",
            spec.name
        )
    })?;

    let running = Arc::clone(&spec);
    let environment = environment(runtime_environment, &spec.env);

    // Spawning a process and waiting for it is genuinely blocking work, so it
    // goes to a thread meant for it rather than onto a runtime worker — which
    // holds on every runtime flavor, including the `current_thread` an embedder
    // inside an editor is likely to have.
    let completion = tokio::task::spawn_blocking(move || {
        subprocess::execute(
            &running.command,
            &running.working_directory(&workspace),
            &environment,
            &payload,
            running.timeout(),
        )
    })
    .await
    .map_err(|error| format!("{} could not be run: {error}", spec.name))?;

    answer(&spec, completion)
}

/// What one declared program is spawned with, over the environment it
/// inherits.
///
/// **The manifest wins.** The runtime's pairs are the host's statement about
/// every process this runtime spawns; the manifest's are this one tool's own,
/// and between two statements about the same name the more specific one holds —
/// the same direction the workspace's `tools.json` already beats the global
/// one. A host that wants the opposite is asking for a value a repository
/// cannot override, which is a different feature and not this one.
///
/// Neither set clears the inherited environment, and that is deliberate:
/// `PATH`, `HOME` and the rest come from there, and a program that lost them
/// would be a manifest that used to work and now does not.
fn environment(
    runtime: &[(String, String)],
    manifest: &[(String, String)],
) -> Vec<(String, String)> {
    runtime
        .iter()
        .filter(|(name, _)| !manifest.iter().any(|(declared, _)| declared == name))
        .chain(manifest.iter())
        .cloned()
        .collect()
}

/// Turns how the program ended into what the model reads.
///
/// Every failure is an `Err`, which reaches the model as that call's result and
/// is the only thing telling it what to do next — so each one says what
/// happened rather than that something did. The program's own stderr is quoted
/// because it is the program's own explanation; the manifest's `env` is not,
/// anywhere.
fn answer(spec: &DeclaredToolSpec, completion: std::io::Result<Completion>) -> ToolResult {
    let completion = completion.map_err(|error| {
        // The io error names the failure, never the command: a `${VAR}` in a
        // program path was resolved before it got here.
        format!("{} could not be started: {error}", spec.name)
    })?;

    let (code, stdout, stderr) = match completion {
        Completion::TimedOut => {
            return Err(format!(
                "{} did not finish within {} seconds and was stopped",
                spec.name,
                spec.timeout().as_secs()
            ));
        }
        Completion::Exited {
            code,
            stdout,
            stderr,
        } => (code, stdout, stderr),
    };

    match code {
        Some(0) => Ok(succeeded(spec, stdout)),
        Some(code) => Err(failed(spec, code, &stdout, &stderr)),
        None => Err(format!(
            "{} was killed by a signal before it answered",
            spec.name
        )),
    }
}

/// stdout, verbatim but for the trailing newline every program prints.
///
/// How much of it survives is mentra's `ToolOutputLimiter`, which bounds and
/// spills every tool result on the runtime; a second cap here would only make
/// the two disagree.
fn succeeded(spec: &DeclaredToolSpec, stdout: String) -> String {
    if stdout.trim().is_empty() {
        // A result that is empty or only whitespace reads to a model as a tool
        // that did nothing, which is a different thing from one that succeeded
        // quietly.
        return format!("{} finished and printed nothing", spec.name);
    }

    // Only the trailing newline, and only from the end: what a program printed
    // in between is the answer, indentation and blank lines included.
    stdout.trim_end_matches('\n').to_string()
}

/// Names the tool, the exit code, and whatever the program said about itself.
///
/// stderr first because that is where a program explains a failure, and stdout
/// as the fallback because plenty of them do not — a failure that quotes
/// neither leaves the model with "it failed" and nothing to act on.
fn failed(spec: &DeclaredToolSpec, code: i32, stdout: &str, stderr: &str) -> String {
    let explanation = if stderr.trim().is_empty() {
        subprocess::truncated_output(stdout)
    } else {
        stderr.trim().to_string()
    };

    if explanation.is_empty() {
        return format!("{} exited {code} and said nothing", spec.name);
    }

    format!("{} exited {code}: {explanation}", spec.name)
}

/// The half of "schema-checked" basis can do without a JSON Schema
/// implementation.
///
/// mentra does not validate a call against the `input_schema` a tool declares —
/// the schema goes to the provider and the raw `Value` comes back — so a tool
/// that wants the guarantee its own descriptor advertises has to check. This
/// checks the two things that make the difference between a legible error and
/// an inscrutable one: the input is an object, and the properties the schema
/// calls `required` are there. Anything deeper is the program's own business,
/// and real validation belongs upstream where every binding would get it
/// (recorded as an upstream candidate in `docs/REDESIGN.md` §2).
fn check_input(spec: &DeclaredToolSpec, input: &Value) -> Result<(), String> {
    if !input.is_object() {
        return Err(format!(
            "{} takes a JSON object matching its input schema",
            spec.name
        ));
    }

    let missing: Vec<&str> = spec
        .input_schema
        .get("required")
        .and_then(Value::as_array)
        .map(|required| {
            required
                .iter()
                .filter_map(Value::as_str)
                .filter(|field| input.get(*field).is_none())
                .collect()
        })
        .unwrap_or_default();

    if missing.is_empty() {
        return Ok(());
    }

    Err(format!(
        "{} was called without {}, which its input schema requires",
        spec.name,
        missing
            .iter()
            .map(|field| format!("`{field}`"))
            .collect::<Vec<_>>()
            .join(", ")
    ))
}

#[cfg(test)]
mod tests;