1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
// T42: `main.rs` + `cli.rs` are their own crate root, so `lib.rs`'s panic
// lints did not reach them — a future unjustified unwrap on the CLI path
// would have compiled clean.
#![warn(clippy::unwrap_used, clippy::panic)]
use std::sync::Arc;
use clap::Parser;
use orion::config;
mod cli;
mod package_cli;
use orion::bootstrap;
#[derive(Parser)]
#[command(
name = "orion-server",
version,
long_version = concat!(
env!("CARGO_PKG_VERSION"),
"\ngit hash: ", env!("GIT_HASH"),
"\nbuilt: ", env!("BUILD_TIMESTAMP"),
),
about = "Orion — Declarative Services Runtime",
long_about = "Orion — Declarative Services Runtime\n\n\
A workflow engine that processes data through configurable channels \
and workflows. Supports REST, HTTP, Kafka, and async processing modes.\n\
Ships as a single binary with an embedded SQLite database.",
after_help = "\
EXAMPLES:\n \
orion-server Start with default config\n \
orion-server -c config.toml Start with a config file\n \
orion-server validate-config Validate + dump effective config (TOML)\n \
orion-server validate-config --format summary Short human summary instead\n \
orion-server -c config.toml migrate Run pending database migrations\n \
orion-server migrate --dry-run Preview pending migrations\n \
orion-server lint workflow.json Validate a workflow JSON file\n \
orion-server dry-run -w wf.json -i x.json Dry-run a workflow against an input\n \
orion-server dry-run -w wf.json -i x.json --stubs s.json ... with canned connector replies\n \
orion-server test examples/workflow-tests Run a directory of workflow test cases\n \
orion-server test-connectivity Probe DB (and Kafka if enabled)\n \
orion-server preflight Scan stored channels/workflows before upgrading\n \
orion-server dump-openapi > spec.json Write the OpenAPI 3.1 spec to a file\n \
orion-server package export -s <url> --tag payments --name payments --version 1.0.0 -o pkg.json\n \
Export a promotion package from an instance\n \
orion-server package apply -s <url> -f pkg.json Stage, activate and reload the package on a target\n\n\
ENVIRONMENT VARIABLES:\n \
All settings can be overridden via ORION_SECTION__KEY env vars:\n\n \
ORION_SERVER__PORT=9090 Override server port\n \
ORION_STORAGE__URL=sqlite:app.db Override database URL\n \
ORION_LOGGING__LEVEL=debug Override log level\n \
ORION_ENVIRONMENT=production Set deployment environment\n\n \
See config.toml.example for all available settings."
)]
struct Cli {
/// Path to TOML configuration file
#[arg(short, long, global = true)]
config: Option<String>,
#[command(subcommand)]
command: Option<Command>,
}
#[derive(clap::Subcommand)]
enum Command {
/// Validate configuration without starting the server, then print the
/// full effective config (defaults + file + ORION_* env overrides) with
/// secrets masked. `--format summary` prints a short human summary
/// instead.
ValidateConfig {
/// Output format for the effective config.
#[arg(long, value_enum, default_value = "toml")]
format: cli::ConfigFormat,
},
/// Run database migrations without starting the server.
Migrate {
/// Preview pending migrations without applying them.
#[arg(long)]
dry_run: bool,
},
/// Statically validate a workflow JSON file (A6).
///
/// Runs the same checks the admin POST /workflows endpoint performs:
/// name/id/description, task uniqueness, and the A1 function-input
/// schema registry. Exits non-zero with field-pathed errors on
/// failure — wire into CI to catch broken workflows before deploy.
Lint {
/// Path to a workflow JSON file (matches the CreateWorkflowRequest shape).
workflow: String,
},
/// Dry-run a workflow against a JSON input file (A6).
///
/// Boots an in-process engine with just the supplied workflow, then prints
/// the per-task execution trace from dataflow_rs.
///
/// Connector-backed tasks (`http_call`, `db_read`, `data_query`,
/// `channel_call`, …) are answered from `--stubs`; nothing reaches a real
/// backend. Without a stub file such a task fails naming the stub that
/// would satisfy it, so a workflow is never silently half-run.
DryRun {
/// Path to a workflow JSON file.
#[arg(short, long)]
workflow: String,
/// Path to a JSON file used as the message payload.
#[arg(short, long)]
input: String,
/// Path to a JSON file of canned connector responses:
/// `{"http_call": {"crm": {...}}, "db_read": {"*": [...]}}`.
/// The inner key is the task's `connector` (or `channel` for
/// `channel_call`); `"*"` matches any.
#[arg(short, long)]
stubs: Option<String>,
},
/// Run a directory of workflow test cases (A6).
///
/// Each `*.case.json` case names a workflow, an input, optional connector stubs
/// and the values expected in the output:
///
/// {"name": "flags high-value orders", "workflow": "wf.json",
/// "input": {...}, "stubs": {...},
/// "expect": {"data.order.flagged": true}}
///
/// Paths inside a case are resolved relative to the case file. Prints a
/// per-case diff and exits non-zero on any failure, so it gates CI the way
/// `lint`, `validate-config` and `preflight` already do.
Test {
/// Directory of case files, or a single case file.
path: String,
},
/// Probe configured backends for reachability (A6).
///
/// Opens the configured database pool (using the same `storage.url`)
/// and runs a no-op query. Catches "DB credentials wrong / file
/// unreadable" before the server tries to start.
TestConnectivity,
/// Print the public HTTP API's OpenAPI 3.1 spec as JSON to stdout.
///
/// Needs no config, database, or running server. Redirect it to refresh
/// the checked-in copy: `orion-server dump-openapi > docs/openapi.json`.
DumpOpenapi,
/// Package a set of channels + their workflows and connectors, and
/// promote the artifact through environments (the K-stream design).
///
/// The artifact is one JSON document; git is the registry. `export`
/// computes the dependency closure from a running instance; `lint` checks
/// an artifact offline; `plan` pre-flights it against a target with zero
/// writes; `apply` stages, activates in dependency order, reloads once
/// and records the package receipt; `diff` reports drift between the
/// artifact and a running instance. Server calls authenticate with the
/// ORION_ADMIN_TOKEN environment variable and are stamped with an
/// `X-Orion-Change-Context: package=<name>@<version>` audit context.
Package {
#[command(subcommand)]
command: PackageCommand,
},
/// Scan the stored channels and workflows for anything the 1.0 rules will
/// refuse, before the upgrade rather than during it.
///
/// Answers the database-backed rows of the 0.3.0 -> 1.0.0 upgrade
/// checklist: channel configs that no longer parse (the pre-1.0 `cors` and
/// `backpressure.max_concurrent` spellings, and typos that were always
/// silently ignored), workflows whose tasks the create validator would
/// reject, and `data_query`/`data_write` tasks with no `schema` — the one
/// change that surfaces on live traffic rather than at startup.
///
/// Read-only, and exits non-zero when it finds anything, so it can gate a
/// deploy. Config-file and ORION_* problems are reported by
/// `validate-config`; this reads what only the database knows.
Preflight,
}
#[derive(clap::Subcommand)]
enum PackageCommand {
/// Export a package artifact from a running instance: the selected
/// channels, their workflows, and every connector those workflows
/// reference (closure computed via GET /workflows/{id}/dependencies).
/// channel_call targets outside the selection land in `requires`.
Export {
/// Base URL of the source instance, e.g. https://dev.orion.internal
#[arg(short, long)]
server: String,
/// Select every channel carrying this tag.
#[arg(long)]
tag: Option<String>,
/// Select channels by id (comma-separated or repeated).
#[arg(long, value_delimiter = ',')]
channels: Vec<String>,
/// Package name, e.g. payments.
#[arg(long)]
name: String,
/// Package version, e.g. 1.4.0. Applied versions are immutable —
/// any content change needs a bump.
#[arg(long)]
version: String,
/// Write the artifact here instead of stdout.
#[arg(short, long)]
output: Option<String>,
},
/// Validate an artifact offline: entity shapes (the same validators the
/// POST endpoints run), closure completeness against `requires`, and the
/// content hash. Exits non-zero on findings — the CI gate that needs no
/// server and no secrets.
Lint {
/// Path to the artifact file.
#[arg(short, long)]
file: String,
},
/// Pre-flight an artifact against a target with zero writes: the receipt
/// immutability check, per-entity would-be import actions, `requires`
/// verification, and every activation gate.
Plan {
/// Base URL of the target instance.
#[arg(short, long)]
server: String,
/// Path to the artifact file.
#[arg(short, long)]
file: String,
},
/// Apply an artifact: claim the receipt as staged, stage all entities
/// (connectors → workflows → channels), activate in dependency order
/// with one engine reload at the end, then flip the receipt to applied.
/// Idempotent — re-running an identical artifact is a no-op.
Apply {
/// Base URL of the target instance.
#[arg(short, long)]
server: String,
/// Path to the artifact file.
#[arg(short, long)]
file: String,
},
/// Report drift between an artifact and a running instance, comparing
/// the server's content hashes against the artifact's. Exits non-zero
/// when anything differs.
Diff {
/// Base URL of the instance to compare against.
#[arg(short, long)]
server: String,
/// Path to the artifact file.
#[arg(short, long)]
file: String,
},
}
#[tokio::main]
async fn main() {
if let Err(err) = run().await {
eprintln!("Error: {err}");
let mut source = std::error::Error::source(&*err);
while let Some(cause) = source {
eprintln!(" Caused by: {cause}");
source = std::error::Error::source(cause);
}
std::process::exit(1);
}
}
async fn run() -> Result<(), Box<dyn std::error::Error>> {
let cli = Cli::parse();
// Load configuration
let mut config = config::load_config(cli.config.as_deref())?;
// Resolve the instance identity once, up front, so the tracing resource,
// cluster runtime, and Kafka static membership all agree on it.
config.cluster.instance_id = config.cluster.effective_instance_id();
let config = config;
if cli.config.is_none() {
eprintln!(
"Note: no config file specified (-c <path>). Using defaults + ORION_* env overrides."
);
}
// Handle subcommands that exit early (before starting the server)
match cli.command {
Some(Command::ValidateConfig { format }) => {
return cli::handle_validate_config(&config, format);
}
Some(Command::Migrate { dry_run }) => return cli::handle_migrate(&config, dry_run).await,
Some(Command::Lint { workflow }) => return cli::run_lint(&workflow),
Some(Command::DryRun {
workflow,
input,
stubs,
}) => {
return cli::run_dry_run(&workflow, &input, stubs.as_deref()).await;
}
Some(Command::Test { path }) => return cli::run_test(&path).await,
Some(Command::TestConnectivity) => return cli::run_test_connectivity(&config).await,
Some(Command::DumpOpenapi) => return cli::run_dump_openapi(),
Some(Command::Preflight) => return cli::run_preflight(&config).await,
Some(Command::Package { command }) => {
return match command {
PackageCommand::Export {
server,
tag,
channels,
name,
version,
output,
} => {
package_cli::run_export(
&server,
tag.as_deref(),
&channels,
&name,
&version,
output.as_deref(),
)
.await
}
PackageCommand::Lint { file } => package_cli::run_lint(&file),
PackageCommand::Plan { server, file } => {
package_cli::run_plan(&server, &file).await
}
PackageCommand::Apply { server, file } => {
package_cli::run_apply(&server, &file).await
}
PackageCommand::Diff { server, file } => {
package_cli::run_diff(&server, &file).await
}
};
}
None => {} // Continue to start the server
}
// Init tracing subscriber with optional OpenTelemetry layer (see
// `bootstrap::init_observability`). The provider is flushed at shutdown.
let _otel_provider = bootstrap::init_observability(&config)?;
tracing::info!(
version = env!("CARGO_PKG_VERSION"),
git_hash = env!("GIT_HASH"),
build_timestamp = env!("BUILD_TIMESTAMP"),
environment = %config.environment,
"Starting Orion — Declarative Services Runtime"
);
// Init metrics (gated by config)
let metrics_handle = bootstrap::init_metrics_handle(&config);
// Install sqlx Any drivers for external connector pools (must be before any pool creation)
sqlx::any::install_default_drivers();
// Init database. With auto_migrate = false (multi-replica deploys) a
// stale schema is a hard startup error — a replica must never serve
// against pending migrations; `orion-server migrate` is the deploy step.
let pool = orion::storage::init_pool_for_startup(&config.storage).await?;
// S20: the DSN can embed `user:password@` credentials — never log it raw.
tracing::info!(
storage = %orion::connector::redact_url_secrets_or_raw(&config.storage.url),
"Database initialized"
);
// C7: in production this pairing is refused by `validate_config` before
// anything opens a connection, so only a development cluster reaches here
// — the Helm `devStack` shape, whose database is a release resource and so
// has no pre-install migrate Job to run instead.
if config.cluster.enabled && config.storage.auto_migrate {
tracing::warn!(
"cluster.enabled with storage.auto_migrate = true: replicas race \
migrations at boot. Tolerated outside production and refused in it — \
use auto_migrate = false plus an `orion-server migrate` deploy step"
);
}
// Cluster runtime: instance identity + shared Redis (fails fast when
// enabled and Redis is unreachable).
let cluster = orion::cluster::init_cluster_runtime(&config.cluster, &pool).await?;
tracing::info!(
instance_id = %cluster.instance_id,
cluster_enabled = cluster.enabled,
"Instance identity"
);
// Create repositories
let repos = bootstrap::Repositories::new(&pool, &config.storage)?;
// Channel registry
let channel_registry = Arc::new(if config.cluster.enabled {
orion::channel::ChannelRegistry::with_cluster((&*cluster).into())
} else {
orion::channel::ChannelRegistry::new()
});
// Connector registry, shared HTTP client, engine lock, cache pools,
// custom function handlers, and the Kafka producer (see
// `bootstrap::build_engine_components`).
let components =
bootstrap::build_engine_components(&config, &repos, channel_registry.clone()).await?;
// Readiness flag — set after engine is fully initialized
let ready = Arc::new(std::sync::atomic::AtomicBool::new(false));
// Load active channels and workflows, build the engine, and populate the
// pre-created engine lock. Channels that fail to load are quarantined —
// refused at every ingress until fixed. Consumes `components`: the
// handler map goes into the engine, and what comes back is the half that
// backs `AppState` (F55).
let (components, channels, active_workflow_count) = components
.load_channels_and_build_engine(&config, &repos, &channel_registry)
.await?;
// Mark the service as ready now that the engine and channel registry are loaded
ready.store(true, std::sync::atomic::Ordering::Release);
let kafka_consumer_handle = bootstrap::start_kafka_ingest(
&config.kafka,
&channels,
components.engine.clone(),
channel_registry.clone(),
components.datalogic.clone(),
components.kafka_producer.clone(),
cluster.enabled.then(|| cluster.instance_id.as_str()),
)?;
// Start the background tasks: trace persistence queue, trace queue
// worker pool (with DLQ for failed async traces), trace + audit-log
// cleanup, and the DLQ retry consumer.
let (trace_persistence_queue, trace_queue, audit_queue, mut task_handles) =
bootstrap::start_background_tasks(
&config,
components.engine.clone(),
&repos,
channel_registry.clone(),
&cluster,
);
// Set initial active rules gauge
orion::metrics::set_active_workflows(active_workflow_count as f64);
// Build rate limiter (if enabled)
let rate_limit_state = bootstrap::build_rate_limit_state(&config);
// Build state and router
let config = Arc::new(config);
let state = bootstrap::build_app_state(bootstrap::AppStateParams {
config: config.clone(),
pool,
repos,
components,
channel_registry,
trace_queue,
trace_persistence_queue,
audit_queue,
rate_limit_state,
metrics_handle,
ready: ready.clone(),
kafka_consumer_handle,
cluster,
});
// Cluster background tasks (epoch watcher). Empty when disabled.
task_handles.cluster_task_handles = orion::cluster::start_cluster_tasks(&state);
let router = orion::server::build_router(state.clone());
// Optional dedicated metrics listener (O12), bound before the main server
// starts (see `bootstrap::start_metrics_listener`).
let metrics_server = bootstrap::start_metrics_listener(&config, &state)?;
if config.server.tls.enabled {
let handle = axum_server::Handle::new();
orion::server::serve::serve_tls(
config.clone(),
ready.clone(),
router,
handle,
orion::server::shutdown_signal(),
)
.await?;
} else {
let addr = format!("{}:{}", config.server.host, config.server.port);
let listener = orion::server::serve::create_tcp_listener(&addr)?;
orion::server::serve::serve_plain_http(
listener,
config.clone(),
ready.clone(),
router,
orion::server::shutdown_signal(),
)
.await?;
}
bootstrap::join_metrics_listener(metrics_server).await;
// Graceful shutdown
if let Some(handle) = state.kafka.consumer_handle.lock().await.take() {
tracing::info!("Shutting down Kafka consumer...");
handle.shutdown().await;
}
// Release the state's trace-queue sender before draining the workers —
// they exit when the last sender closes, and holding `state` here would
// stall the drain until its timeout.
drop(state);
task_handles.shutdown().await;
// Flush pending OTel spans before exit
if let Some(provider) = _otel_provider {
tracing::info!("Flushing OpenTelemetry spans...");
if let Err(e) = provider.shutdown() {
tracing::warn!(error = %e, "Error shutting down OTel tracer provider");
}
}
tracing::info!("Orion shut down cleanly");
Ok(())
}