1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
//! The `ostraka` command.
mod adapters;
mod banner;
mod bench;
mod check;
mod discover;
mod drain;
mod fix;
mod init;
mod init_cmd;
mod offer;
mod promote;
mod prune;
mod remedy;
mod replay;
mod run;
mod runs;
mod task_cmd;
mod tasks;
mod tui;
mod update;
mod workspace;
use clap::{Parser, Subcommand};
use std::path::PathBuf;
use std::process::ExitCode;
#[derive(Parser)]
#[command(
name = "ostraka",
version,
about = "Run agent fleets you can actually review."
)]
struct Cli {
/// Workspace directory. Defaults to the current directory.
#[arg(long, global = true)]
workspace: Option<PathBuf>,
/// Which repository in the repositories directory to work in. Only needed
/// where the workspace holds more than one.
#[arg(long, global = true)]
repository: Option<String>,
/// Where this workspace's repositories live, instead of `repositories/`.
///
/// Read against the current directory. `[workspace] repositories` in
/// `.ostraka/ostraka.toml` sets it for good; this wins over that for one
/// command, and `init --repositories` writes it down.
#[arg(long, global = true, value_name = "DIR")]
repositories: Option<PathBuf>,
/// Emit machine-readable output.
#[arg(long, global = true)]
json: bool,
/// Absent when the binary is run by name and nothing else, which is how
/// somebody asks a program what it is. clap's answer to a missing required
/// subcommand is a usage error on stderr and a non-zero exit; that is a
/// true sentence about the parser and the wrong one to be met by.
#[command(subcommand)]
command: Option<Commands>,
}
#[derive(Subcommand)]
enum TaskCommand {
/// Write a task down. It waits until a run takes it.
Add {
/// What the agent should do.
prompt: String,
/// Which repository it belongs in.
#[arg(long)]
repository: Option<String>,
/// The profile that should write it.
#[arg(long)]
adapter: Option<String>,
},
/// Put a claimed task back, for a run that never reported.
Release {
/// The task id, as `ostraka tasks` shows it.
id: String,
},
}
#[derive(Subcommand)]
enum Commands {
/// Validate the project config and every adapter profile.
Check {
/// Walk the steps out of what it found, one at a time, asking first.
#[arg(long)]
fix: bool,
},
/// Write the files a project needs to be run by Ostraka.
Init {
/// Rewrite files that already exist. Off by default: what is there is
/// somebody's, and a setup command should not be how they lose it.
#[arg(long)]
force: bool,
},
/// List adapter profiles and whether each one can run here.
Adapters,
/// Run the same tasks through every candidate and tabulate what the gate
/// said.
///
/// Declared in `.ostraka/bench.toml`: the tasks, the candidates — a
/// profile and the models to try it on — and the reviewer profiles held
/// constant across them. Every cell is a full run, so a matrix costs what
/// its size says it does; `--dry-run` prints that size without spending it.
Bench {
/// Print the matrix and stop. Nothing is run and no vendor is called.
#[arg(long)]
dry_run: bool,
},
/// Run one task: isolate, execute, gate, review, record.
Run {
/// What the agent should do. Omitted with `--next`, which takes it
/// from the list.
prompt: Option<String>,
/// Identity accountable for the change.
#[arg(long, default_value = run::AUTHOR)]
author: String,
/// Identity that reviews it. Must differ from the author.
#[arg(long, default_value = run::REVIEWER)]
reviewer: String,
/// Adapter profile that writes the change.
#[arg(long)]
adapter: Option<String>,
/// Adapter profile that reviews it. Must differ from `--adapter`.
#[arg(long)]
review_adapter: Option<String>,
/// Git ref the worktree branches from.
#[arg(long, default_value = run::BASE_REF)]
base_ref: String,
/// Continue a finished run: branch from what it left, not from HEAD.
/// Takes a run id. Only an approved run can be continued.
#[arg(long, conflicts_with = "base_ref", value_name = "RUN_ID")]
from: Option<String>,
/// Take the oldest task from the list instead of being given one.
#[arg(long, conflicts_with = "prompt")]
next: bool,
/// Model hint passed through to the adapter.
#[arg(long)]
model: Option<String>,
},
/// Write a shell completion script to stdout.
///
/// Generated from this parser, so it names the commands this binary
/// actually has. Install it the way your shell wants:
///
/// ostraka completion bash > ~/.local/share/bash-completion/completions/ostraka
/// ostraka completion zsh > "${fpath[1]}/_ostraka"
/// ostraka completion fish > ~/.config/fish/completions/ostraka.fish
Completion {
/// Which shell to write for.
shell: clap_complete::Shell,
},
/// Work written down before somebody is free to do it.
Task {
#[command(subcommand)]
what: TaskCommand,
},
/// Everything on the task list, in every state.
Tasks,
/// Take the task list down, several at a time.
Drain {
/// How many runs at once.
#[arg(long, default_value_t = drain::WORKERS)]
workers: usize,
/// Identity accountable for the changes.
#[arg(long, default_value = run::AUTHOR)]
author: String,
/// Identity that reviews them. Must differ from the author.
#[arg(long, default_value = run::REVIEWER)]
reviewer: String,
/// Adapter profile that writes, where a task has not named one.
#[arg(long)]
adapter: Option<String>,
/// Adapter profile that reviews. Must differ from `--adapter`.
#[arg(long)]
review_adapter: Option<String>,
},
/// List every run this project has recorded.
Runs,
/// Browse runs in the terminal.
Tui,
/// Give an approved run a branch of its own. Merges nothing.
Promote {
/// Run id, as printed by `run`.
run_id: String,
/// Branch to create. Defaults to `promoted/<run-id>`.
#[arg(long)]
branch: Option<String>,
},
/// Remove worktrees left by finished runs. Branches are untouched.
Prune {
/// Actually remove them. Without this it only reports what it would do.
#[arg(long)]
apply: bool,
},
/// Replace this binary with the current release.
///
/// Refuses to overwrite a copy Homebrew, cargo or npm installed — those
/// track which version is on this machine, and writing over the file
/// behind them leaves that record wrong. It names whose it is instead.
Update {
/// Report what is available and change nothing.
#[arg(long)]
check: bool,
},
/// Read a finished run back from its record.
Replay {
/// Run id, as printed by `run`.
run_id: String,
},
}
fn main() -> ExitCode {
let cli = Cli::parse();
let Some(command) = &cli.command else {
banner::print();
return ExitCode::SUCCESS;
};
// Before a workspace is looked for. Completions are about the command line,
// not about a project, and a shell asking for them in someone's home
// directory should not be told that their home directory is not a
// workspace.
// Matched on a reference. It compiles either way today — `Shell` is `Copy`,
// so binding it takes a copy and `cli` is not moved — but that is a fact
// about a dependency's derives, and the day one field of this variant stops
// being `Copy` the error lands here rather than in the change that caused
// it. Locked once, because a completion script is one write.
if let Commands::Completion { shell } = command {
write_completion(*shell, &mut std::io::stdout().lock());
return ExitCode::SUCCESS;
}
// Also before a workspace is looked for, and for the same reason: which
// version of this binary is installed is a fact about the machine, not
// about a project. Somebody running it in their home directory should not
// be told their home directory is not a workspace.
if let Commands::Update { check } = command {
return match update::run(*check, cli.json) {
Ok(_) => ExitCode::SUCCESS,
Err(e) => {
eprintln!("ostraka: {e}");
ExitCode::FAILURE
}
};
}
let here = cli.workspace.clone().unwrap_or_else(|| PathBuf::from("."));
let workspace = match command {
// Not checked: `init` is what makes a workspace, and it may be asked to
// make the very directory `--repositories` names.
Commands::Init { .. } => workspace::Workspace::at(&here),
// Checked before anything runs, so a repositories directory somebody
// chose and that is not there is said by name — not discovered as a
// workspace that looks empty.
_ => match workspace::Workspace::open(&here, cli.repositories.as_deref()) {
Ok(workspace) => workspace,
Err(e) => {
eprintln!("error: {e}");
return ExitCode::FAILURE;
}
},
};
let result = match command {
Commands::Check { fix } => check::run(&workspace, *fix, cli.json),
Commands::Init { force } => {
init_cmd::run(&here, cli.repositories.as_deref(), *force, cli.json)
}
Commands::Adapters => adapters::run(&workspace, cli.json),
Commands::Bench { dry_run } => bench::run(&workspace, *dry_run, cli.json),
Commands::Replay { run_id } => replay::run(&workspace, run_id, cli.json),
Commands::Task { what } => match what {
TaskCommand::Add {
prompt,
repository,
adapter,
} => task_cmd::add(
&workspace,
prompt,
repository.as_deref().or(cli.repository.as_deref()),
adapter.as_deref(),
cli.json,
),
TaskCommand::Release { id } => task_cmd::release(&workspace, id, cli.json),
},
Commands::Tasks => task_cmd::list(&workspace, cli.json),
Commands::Drain {
workers,
author,
reviewer,
adapter,
review_adapter,
} => {
let mut args = run::Args::for_task(String::new());
args.repository = cli.repository.clone();
args.author = author.clone();
args.reviewer = reviewer.clone();
args.adapter = adapter.clone();
args.review_adapter = review_adapter.clone();
drain::run(&workspace, args, *workers, cli.json)
}
Commands::Runs => runs::run(&workspace, cli.json),
Commands::Completion { .. } | Commands::Update { .. } => {
unreachable!("handled before a workspace is resolved")
}
Commands::Prune { apply } => prune::run(&workspace, *apply, cli.json),
Commands::Tui => tui::run(&workspace),
Commands::Promote { run_id, branch } => {
promote::run(&workspace, run_id, branch.as_deref(), cli.json)
}
Commands::Run {
prompt,
next,
author,
reviewer,
adapter,
review_adapter,
base_ref,
from,
model,
} => {
let args = run::Args {
prompt: prompt.clone().unwrap_or_default(),
repository: cli.repository.clone(),
author: author.clone(),
reviewer: reviewer.clone(),
adapter: adapter.clone(),
review_adapter: review_adapter.clone(),
base_ref: base_ref.clone(),
from: from.clone(),
model: model.clone(),
};
if *next {
task_cmd::run_next(&workspace, args, cli.json)
} else if args.prompt.is_empty() {
Err("say what the agent should do, or `--next` to take it from the list".into())
} else {
run::run(&workspace, &args, cli.json)
}
}
};
match result {
// A refused run is a correct outcome reported correctly, but the exit
// code has to distinguish it: CI treats a non-zero exit as "not ready".
//
// Both of these get the notice, and it is the last thing printed. A
// refused run is an answer; somebody who has just been told their
// change was rejected has been told something true and can hear one
// more line.
Ok(approved) => {
update::notice::offer(cli.json);
if approved {
ExitCode::SUCCESS
} else {
ExitCode::FAILURE
}
}
// An error gets none. Somebody staring at a failure does not need to
// hear about a version, and appending to it would put the notice
// between them and the thing that went wrong. Raised in review: the
// call used to be above this block, which printed the notice *before*
// the error — the opposite of what its comment claimed.
Err(e) => {
eprintln!("error: {e}");
ExitCode::FAILURE
}
}
}
/// Writes the completion script for one shell.
///
/// Taken from `Cli::command()` rather than written out, so it describes the
/// commands and flags this binary has — including the ones added after the
/// script was installed, once it is regenerated.
fn write_completion(shell: clap_complete::Shell, out: &mut impl std::io::Write) {
let mut command = <Cli as clap::CommandFactory>::command();
clap_complete::generate(shell, &mut command, "ostraka", out);
}
#[cfg(test)]
mod tests {
use super::*;
use std::path::Path;
fn parse(args: &[&str]) -> Result<Cli, clap::Error> {
Cli::try_parse_from(std::iter::once("ostraka").chain(args.iter().copied()))
}
#[test]
fn a_completion_script_is_written_for_every_shell_clap_knows() {
// Generated from the parser, so the assertion worth making is that the
// parser is what reached the script: a command added later appears
// without anybody editing anything, and one renamed stops appearing.
// Asked of clap rather than listed here. `Shell` is `#[non_exhaustive]`
// — upstream adds one without it being a breaking change — and a list
// written out is the second description of something this whole change
// exists to argue against keeping.
let shells = <clap_complete::Shell as clap::ValueEnum>::value_variants();
assert!(!shells.is_empty(), "clap knows no shells");
for shell in shells.iter().copied() {
let mut out: Vec<u8> = Vec::new();
write_completion(shell, &mut out);
let text = String::from_utf8(out).expect("utf-8");
assert!(!text.is_empty(), "{shell} wrote nothing");
assert!(text.contains("ostraka"), "{shell} did not name the binary");
for command in ["run", "promote", "prune", "adapters", "replay"] {
assert!(
text.contains(command),
"{shell} did not carry {command}:\n{text}"
);
}
// And the flags, which are the half a hand-written script forgets.
// Asked by name rather than by spelling: fish writes a long option
// as `-l from` and the rest write `--from`, so looking for the
// dashes tests which shell this is and not whether the option
// arrived.
assert!(
text.contains("--from") || text.contains("-l from"),
"{shell} did not carry the --from option:\n{text}"
);
}
}
#[test]
fn where_the_repositories_are_is_a_flag_every_command_takes() {
// Global, because it describes the workspace rather than a command:
// `ostraka --repositories ~/code runs` and `ostraka runs --repositories
// ~/code` have to mean the same thing.
for args in [
&["--repositories", "/code", "runs"][..],
&["runs", "--repositories", "/code"][..],
] {
let cli = parse(args).expect("parses");
assert_eq!(cli.repositories.as_deref(), Some(Path::new("/code")));
}
assert!(parse(&["runs"]).expect("parses").repositories.is_none());
}
#[test]
fn next_and_a_prompt_are_alternatives_and_a_bare_run_is_still_refused() {
// `--next` made `prompt` optional, which is the change worth pinning:
// an optional positional cannot be required by clap any more, so
// "say what the agent should do" moved out of the parser and into the
// command. Three shapes, and the parser only decides the first two.
let cli = parse(&["run", "--next"]).expect("--next alone parses");
let Some(Commands::Run { prompt, next, .. }) = &cli.command else {
panic!("not the run command")
};
assert!(*next);
assert_eq!(prompt.as_deref(), None);
let cli = parse(&["run", "a task"]).expect("a prompt alone parses");
let Some(Commands::Run { prompt, next, .. }) = &cli.command else {
panic!("not the run command")
};
assert!(!*next);
assert_eq!(prompt.as_deref(), Some("a task"));
// Both is refused: one says take the next thing on the list and the
// other says do this instead, and a run given both would ignore one.
let err = match parse(&["run", "a task", "--next"]) {
Ok(_) => panic!("--next and a prompt were accepted together"),
Err(e) => e,
};
assert_eq!(err.kind(), clap::error::ErrorKind::ArgumentConflict);
// And neither still parses, because the positional is optional now.
// What used to be a parse error is a sentence the command prints, and
// this is the assertion that says so out loud.
let cli = parse(&["run"]).expect("a bare run parses, and is refused later");
let Some(Commands::Run { prompt, next, .. }) = &cli.command else {
panic!("not the run command")
};
assert!(!*next);
assert_eq!(prompt.as_deref(), None);
}
#[test]
fn from_and_base_ref_are_alternatives_and_only_one_of_them_may_be_given() {
// The pair carries a real hazard: `--base-ref` has a default, and an
// argument declared `conflicts_with` a defaulted one would reject every
// invocation if clap counted the default as having been supplied. It
// does not — but "it does not" is a fact about a dependency, and the
// failure if it changed is that `--from` stops working entirely rather
// than misbehaving somewhere visible. So it is pinned here.
let cli = parse(&["run", "a task", "--from", "t1"]).expect("--from alone parses");
let Some(Commands::Run { from, base_ref, .. }) = &cli.command else {
panic!("not the run command")
};
assert_eq!(from.as_deref(), Some("t1"));
assert_eq!(base_ref, run::BASE_REF, "the default still applies");
// And naming both is refused, because one says which run and the other
// says which ref: a run given both would have to ignore one of them.
// `Cli` is not `Debug`, so the error is taken by hand.
let err = match parse(&["run", "a task", "--from", "t1", "--base-ref", "other"]) {
Ok(_) => panic!("--from and --base-ref were accepted together"),
Err(e) => e,
};
assert_eq!(err.kind(), clap::error::ErrorKind::ArgumentConflict);
// Neither is required, and the default is what a run gets.
let cli = parse(&["run", "a task"]).expect("a bare run parses");
let Some(Commands::Run { from, base_ref, .. }) = &cli.command else {
panic!("not the run command")
};
assert_eq!(from.as_deref(), None);
assert_eq!(base_ref, run::BASE_REF);
}
}