surrealdb-server 3.3.1

A scalable, distributed, collaborative, document-graph database, for the realtime web
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
use std::net::SocketAddr;
use std::path::PathBuf;
use std::sync::Arc;
use std::sync::atomic::{AtomicBool, Ordering};
use std::time::Duration;

use anyhow::Result;
use clap::Args;
use surrealdb::engine::any;
use surrealdb_core::kvs::TransactionBuilderFactory;
use surrealdb_core::options::EngineOptions;
use surrealdb_observe::{ExecutionObserver, FanOutObserver};
use tokio::sync::oneshot;
use tokio_util::sync::CancellationToken;

use super::config::Config;
use crate::cli::ConfigCheck;
use crate::cnf::{LOGO, METRICS_ENABLED, PROCESS_METRICS_REFRESH_INTERVAL};
use crate::dbs::StartCommandDbsOptions;
use crate::ntw::RouterFactory;
use crate::ntw::client_ip::ClientIp;
use crate::observe::instruments::scope;
use crate::observe::{MetricsObserver, MetricsState, ObservabilityProvider, ObservabilityRuntime};
use crate::telemetry::metrics::otlp_metrics_active;
use crate::{dbs, env, ntw};

/// How long the HTTP bind waits for the deferred startup work to finish when
/// there is no startup import.
///
/// Bounded from both sides, and the gap between them is wide:
///
/// - It must exceed that work, which against local storage is milliseconds per step except the root
///   credential hash — deliberately expensive, but bounded CPU work at a little over a second on a
///   loaded CI runner. Binding before it finishes costs the guarantee that a server answering
///   `/health` can serve, which clients and test harnesses rely on far more than they poll
///   `/ready`.
/// - It must stay under the shortest deadline a client gives the listener to appear, since it is
///   also the longest the listener can take. The tightest known is 10s, in the JS SDK's own
///   harness.
///
/// Anything past this is a datastore that is not accepting transactions, which
/// no amount of further waiting fixes, so the listener binds and the work
/// continues behind the readiness gate.
const STARTUP_BIND_GRACE: Duration = Duration::from_secs(5);

/// How long shutdown waits for the deferred startup work to stop before it
/// closes the storage engine.
///
/// That work opens transactions of its own — the version stamp, the defaults,
/// this node's registration, the import — and is not joined the way the
/// datastore's own maintenance tasks are, because it is the task that starts
/// the maintenance schedule: there is nothing to register with until it has
/// already run. So the join is made here instead, on the same terms
/// `Datastore::shutdown` gives the maintenance tasks: bounded at the same
/// budget, and a task still running when it expires is abandoned exactly as a
/// maintenance pass would be.
const STARTUP_SHUTDOWN_JOIN: Duration = Duration::from_secs(30);

#[derive(Args, Debug)]
pub struct StartCommandArguments {
	#[arg(help = "Database path used for storing data")]
	#[arg(env = "SURREAL_PATH", index = 1)]
	#[arg(default_value = "memory")]
	path: String,
	#[arg(help = "Whether to hide the startup banner")]
	#[arg(env = "SURREAL_NO_BANNER", long)]
	#[arg(default_value_t = false)]
	no_banner: bool,
	#[arg(help = "Encryption key to use for on-disk encryption")]
	#[arg(env = "SURREAL_KEY", short = 'k', long = "key")]
	#[arg(value_parser = super::validator::key_valid)]
	#[arg(hide = true)] // Not currently in use
	key: Option<String>,
	//
	// Tasks
	#[arg(
		help = "The interval at which to refresh node registration information",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_NODE_MEMBERSHIP_REFRESH_INTERVAL", long = "node-membership-refresh-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "3s")]
	node_membership_refresh_interval: Duration,
	#[arg(
		help = "The interval at which to process and archive inactive nodes",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_NODE_MEMBERSHIP_CHECK_INTERVAL", long = "node-membership-check-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "15s")]
	node_membership_check_interval: Duration,
	#[arg(
		help = "The interval at which to process and cleanup archived nodes",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_NODE_MEMBERSHIP_CLEANUP_INTERVAL", long = "node-membership-cleanup-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "300s")]
	node_membership_cleanup_interval: Duration,
	#[arg(
		help = "The interval at which to perform changefeed garbage collection",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_CHANGEFEED_GC_INTERVAL", long = "changefeed-gc-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "30s")]
	changefeed_gc_interval: Duration,
	#[arg(env = "SURREAL_INDEX_COMPACTION_INTERVAL", long = "index-compaction-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "5s")]
	index_compaction_interval: Duration,
	#[arg(
		help = "The interval at which to fold graph adjacency keys into packed blocks",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_GRAPH_FOLD_INTERVAL", long = "graph-fold-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "5s")]
	graph_fold_interval: Duration,
	#[arg(
		help = "The interval at which to resume index builds left unfinished by a crashed node (0 to disable)",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_INDEX_BUILD_RESUME_INTERVAL", long = "index-build-resume-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "30s")]
	index_build_resume_interval: Duration,
	#[arg(env = "SURREAL_ASYNC_EVENT_PROCESSING_INTERVAL", long = "async-event-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "5s")]
	event_processing_interval: Duration,
	#[arg(env = "SURREAL_RECLAIM_INTERVAL", long = "reclaim-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "60s")]
	reclaim_interval: Duration,
	#[arg(
		help = "Minimum age a removed namespace/database/index must reach before its data is physically reclaimed (snapshot-safety grace; effective value is max(this, --tikv-gc-lifetime))",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_RECLAIM_GRACE", long = "reclaim-grace", value_parser = super::validator::duration)]
	#[arg(default_value = "10m")]
	reclaim_grace: Duration,
	#[arg(
		help = "The interval at which the TiKV MVCC garbage collector runs",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_TIKV_GC_INTERVAL", long = "tikv-gc-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "10m")]
	tikv_gc_interval: Duration,
	#[arg(
		help = "How far behind the current TSO the TiKV GC safepoint is allowed to sit",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_TIKV_GC_LIFETIME", long = "tikv-gc-lifetime", value_parser = super::validator::duration)]
	#[arg(default_value = "10m")]
	tikv_gc_lifetime: Duration,
	#[arg(
		help = "The interval at which TiKV stale transactional locks are resolved",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_TIKV_LOCK_CLEANUP_INTERVAL", long = "tikv-lock-cleanup-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "60s")]
	tikv_lock_cleanup_interval: Duration,
	#[arg(
		help = "Whether to persist client-attached HTTP RPC sessions in the datastore so they survive server restarts and can be resumed on any cluster node. Intended for deployments that route a given session to one node at a time (sticky routing / one runtime per session); a session used concurrently from multiple nodes is best-effort. The durable copy contains the session's authentication state, stored unencrypted in the datastore",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_DURABLE_SESSIONS", long = "durable-sessions")]
	#[arg(default_value_t = false)]
	durable_sessions: bool,
	#[arg(
		help = "How long a persisted RPC session survives without being used; each use refreshes the expiry",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_DURABLE_SESSION_TTL", long = "durable-session-ttl", value_parser = super::validator::duration)]
	#[arg(default_value = "24h")]
	durable_session_ttl: Duration,
	#[arg(
		help = "The interval at which expired persisted RPC sessions are purged (0 to disable)",
		help_heading = "Database"
	)]
	#[arg(env = "SURREAL_DURABLE_SESSION_GC_INTERVAL", long = "durable-session-gc-interval", value_parser = super::validator::duration)]
	#[arg(default_value = "60s")]
	durable_session_gc_interval: Duration,
	//
	// Authentication
	#[arg(
		help = "The username for the initial database root user. Only if no other root user exists",
		help_heading = "Authentication"
	)]
	#[arg(
		env = "SURREAL_USER",
		short = 'u',
		long = "username",
		visible_alias = "user",
		requires = "password"
	)]
	username: Option<String>,
	#[arg(
		help = "The password for the initial database root user. Only if no other root user exists",
		help_heading = "Authentication"
	)]
	#[arg(
		env = "SURREAL_PASS",
		short = 'p',
		long = "password",
		visible_alias = "pass",
		requires = "username"
	)]
	password: Option<String>,
	//
	// Datastore connection
	#[command(next_help_heading = "Datastore connection")]
	#[command(flatten)]
	kvs: Option<StartCommandRemoteTlsOptions>,
	//
	// HTTP Server
	#[command(next_help_heading = "HTTP server")]
	#[command(flatten)]
	web: Option<StartCommandWebTlsOptions>,
	#[arg(help = "The method of detecting the client's IP address")]
	#[arg(env = "SURREAL_CLIENT_IP", long)]
	#[arg(default_value = "socket", value_enum)]
	client_ip: ClientIp,
	#[arg(help = "The hostname or IP address to listen for connections on")]
	#[arg(env = "SURREAL_BIND", short = 'b', long = "bind")]
	#[arg(default_value = "127.0.0.1:8000")]
	listen_addresses: Vec<SocketAddr>,
	#[arg(help = "Whether to suppress the server name and version headers")]
	#[arg(env = "SURREAL_NO_IDENTIFICATION_HEADERS", long)]
	#[arg(default_value_t = false)]
	no_identification_headers: bool,
	#[arg(help = "The allowed origins for CORS requests. Defaults to allow all origins")]
	#[arg(env = "SURREAL_ALLOW_ORIGIN", long = "allow-origin")]
	#[arg(value_delimiter = ',', value_parser = super::validator::cors_origin)]
	allow_origin: Vec<String>,
	//
	// Postgres server
	#[arg(
		help = "The hostname or IP address to listen for Postgres wire protocol connections on",
		help_heading = "Postgres server"
	)]
	#[arg(env = "SURREAL_POSTGRES_BIND", long = "postgres-bind")]
	postgres_bind: Option<SocketAddr>,
	//
	// Database options
	#[command(flatten)]
	#[command(next_help_heading = "Database")]
	dbs: StartCommandDbsOptions,
}

#[derive(Args, Debug)]
#[group(requires_all = ["kvs_ca", "kvs_crt", "kvs_key"], multiple = true)]
struct StartCommandRemoteTlsOptions {
	#[arg(help = "Path to the CA file used when connecting to the remote KV store")]
	#[arg(env = "SURREAL_KVS_CA", long = "kvs-ca", value_parser = super::validator::file_exists)]
	kvs_ca: Option<PathBuf>,
	#[arg(help = "Path to the certificate file used when connecting to the remote KV store")]
	#[arg(env = "SURREAL_KVS_CRT", long = "kvs-crt", value_parser = super::validator::file_exists)]
	kvs_crt: Option<PathBuf>,
	#[arg(help = "Path to the private key file used when connecting to the remote KV store")]
	#[arg(env = "SURREAL_KVS_KEY", long = "kvs-key", value_parser = super::validator::file_exists)]
	kvs_key: Option<PathBuf>,
}

#[derive(Args, Debug)]
#[group(requires_all = ["web_crt", "web_key"], multiple = true)]
struct StartCommandWebTlsOptions {
	#[arg(help = "Path to the certificate file for encrypted client connections")]
	#[arg(env = "SURREAL_WEB_CRT", long = "web-crt", value_parser = super::validator::file_exists)]
	web_crt: Option<PathBuf>,
	#[arg(help = "Path to the private key file for encrypted client connections")]
	#[arg(env = "SURREAL_WEB_KEY", long = "web-key", value_parser = super::validator::file_exists)]
	web_key: Option<PathBuf>,
}

/// Start the server.
///
/// Initializes and starts the SurrealDB server with the provided configuration.
///
/// # Parameters
/// - `composer`: A composer implementing the required traits for dependency injection
///
/// # Generic parameters
/// - `C`: A composer type that implements:
///   - `TransactionBuilderFactory` (datastore transaction builder for storage/backend selection)
///   - `RouterFactory` (HTTP router factory for route/middleware customization)
///   - `ConfigCheck` (validates configuration before initialization)
pub async fn init<
	C: TransactionBuilderFactory + RouterFactory + ConfigCheck + ObservabilityProvider,
>(
	mut composer: C,
	StartCommandArguments {
		path,
		username: user,
		password: pass,
		client_ip,
		listen_addresses,
		dbs,
		web,
		node_membership_refresh_interval,
		node_membership_check_interval,
		node_membership_cleanup_interval,
		changefeed_gc_interval,
		index_compaction_interval,
		graph_fold_interval,
		index_build_resume_interval,
		event_processing_interval,
		reclaim_interval,
		reclaim_grace,
		tikv_gc_interval,
		tikv_gc_lifetime,
		tikv_lock_cleanup_interval,
		durable_sessions,
		durable_session_ttl,
		durable_session_gc_interval,
		no_banner,
		no_identification_headers,
		allow_origin,
		postgres_bind,
		..
	}: StartCommandArguments,
	runtime: ObservabilityRuntime,
) -> Result<()> {
	// Check the path is valid
	composer.path_valid(&path)?;
	// Persisted sessions must have a finite expiration
	if durable_sessions && durable_session_ttl.is_zero() {
		return Err(anyhow::anyhow!("The durable session TTL must be greater than zero"));
	}
	// Check if we should output a banner
	if !no_banner {
		println!("{LOGO}");
	}
	// Clean the path
	let endpoint = any::__into_endpoint(path)?;
	let path = if endpoint.path.is_empty() {
		endpoint.url.to_string()
	} else {
		endpoint.path
	};
	// Extract the certificate and key
	let (crt, key) = if let Some(val) = web {
		(val.web_crt, val.web_key)
	} else {
		(None, None)
	};
	// Configure the engine
	let engine = EngineOptions::default()
		.with_node_membership_refresh_interval(node_membership_refresh_interval)
		.with_node_membership_check_interval(node_membership_check_interval)
		.with_node_membership_cleanup_interval(node_membership_cleanup_interval)
		.with_changefeed_gc_interval(changefeed_gc_interval)
		.with_index_compaction_interval(index_compaction_interval)
		.with_graph_fold_interval(graph_fold_interval)
		.with_index_build_resume_interval(index_build_resume_interval)
		.with_event_processing_interval(event_processing_interval)
		.with_reclaim_interval(reclaim_interval)
		.with_reclaim_grace(reclaim_grace)
		.with_tikv_gc_interval(tikv_gc_interval)
		.with_tikv_gc_lifetime(tikv_gc_lifetime)
		.with_tikv_lock_cleanup_interval(tikv_lock_cleanup_interval)
		.with_rpc_session_gc_interval(durable_session_gc_interval)
		// Floor the configured value at 1s so a misconfigured zero neither
		// produces a tight refresh loop nor unregisters the job.
		.with_system_metrics_refresh_interval(Duration::from_secs(
			(*PROCESS_METRICS_REFRESH_INTERVAL).max(1),
		));
	// The memory threshold is read once per process, so this is the only place
	// its effective value is observable.
	match surrealdb_core::mem::memory_threshold() {
		0 => info!("Memory threshold is disabled; queries will not be refused on memory pressure"),
		bytes => info!("Memory threshold is {bytes} bytes"),
	}
	// Configure the config
	let Some(bind) = listen_addresses.first().copied() else {
		return Err(anyhow::anyhow!("No listen address provided"));
	};
	let config = Config {
		bind,
		postgres_bind,
		client_ip,
		path,
		user,
		pass,
		no_identification_headers,
		allow_origin,
		engine,
		crt,
		key,
		durable_session_ttl: durable_sessions.then_some(durable_session_ttl),
	};
	composer.check_config(&config).await?;
	// Setup the command-line options
	// Initiate environment
	env::init()?;

	// Build the observability pipeline before starting the datastore so the
	// observer can be installed at construction time. The community metrics
	// observer is only instantiated when /metrics is enabled; composer
	// extensions may contribute an additional audit observer regardless.
	let (metrics_state, combined_observer) = build_observability::<C>(&composer, &runtime)?;

	// Create a token to cancel tasks
	let canceller = CancellationToken::new();

	// Build the datastore. Every startup step that opens a transaction — the
	// version check, the default namespace and database, this node's cluster
	// registration, the import and the root credentials — is returned rather than
	// run here, so the bind waits for none of them. It still waits for the
	// datastore itself: a composer that does not return its builder until its
	// engine is ready to serve holds the bind for that long regardless.
	let (datastore, recv, router_state, pending_startup) =
		dbs::init::<C>(composer, &config, canceller.clone(), combined_observer, dbs).await?;
	// Tracks whether the instance has finished starting up and is ready to serve
	// user-facing queries. Until it flips to `true`, query/auth endpoints return
	// 503 and `/ready` reports not-ready.
	let ready = Arc::new(AtomicBool::new(false));
	// Carries a failure of the steps the instance cannot serve without back to
	// this function, so the process reports it as its exit status. The token is
	// what the rest of this function watches: it is tripped only by such a
	// failure, and only after the error itself has been sent, so a tripped token
	// means the channel holds the error. A dropped sender — which a clean
	// startup and a cancelled one both leave behind — never trips it, and
	// awaiting or testing a token is safe any number of times, which reading the
	// channel is not.
	let (failed_tx, mut failed_rx) = oneshot::channel::<anyhow::Error>();
	let fatal = CancellationToken::new();
	// Whether the deferred work includes a startup import, which decides whether
	// the bind waits for it (see the grace period below).
	let has_import = pending_startup.has_import();
	let mut startup = Some({
		// Run the deferred startup work as a task, so the listener can bind
		// without waiting for it when it turns out to be slow.
		let ds = Arc::clone(&datastore);
		let ready = Arc::clone(&ready);
		let startup_canceller = canceller.clone();
		let fatal_canceller = canceller.clone();
		let fatal = fatal.clone();
		// A plain spawn: this is the task that starts the maintenance schedule,
		// so until it has run there is no task set to register it with. Shutdown
		// joins this handle directly instead — see [`STARTUP_SHUTDOWN_JOIN`] —
		// which keeps the same guarantee: the storage engine does not close
		// under the transactions this task opens.
		tokio::spawn(async move {
			tokio::select! {
				biased;
				// Stop promptly if the server is shutting down; leave `ready`
				// unset so we never flip to ready mid-shutdown.
				_ = startup_canceller.cancelled() => {
					debug!("Startup aborted before completion due to shutdown");
				}
				() = async move {
					// Nothing this instance serves works without these steps, so a
					// failure ends the process rather than leaving a server that
					// can never become ready. An operator facing a datastore from
					// another version, or storage that never came up, has to act
					// on it, and an exit status is both what reaches them and
					// what restarts the pod.
					if let Err(err) = dbs::initialise_datastore(&ds, &pending_startup).await {
						error!("Failed to initialise the datastore: {err}");
						// The error goes first, so a tripped `fatal` always has one
						// to report.
						let _ = failed_tx.send(err);
						fatal.cancel();
						fatal_canceller.cancel();
						return;
					}
					// Eagerly load surrealism modules in the background unless
					// opted out. Ordered behind the initialisation because the
					// load reads the catalog once and gives up with a warning: an
					// engine that refuses transactions until it has joined a
					// cluster would spend that one attempt against a datastore
					// that cannot answer, leaving every module to load lazily on
					// first use instead.
					#[cfg(feature = "surrealism")]
					if !ds.is_lazy_surrealism() {
						let ds = Arc::clone(&ds);
						tokio::spawn(async move {
							ds.eager_load_surrealism_modules().await;
						});
					}
					// The import and the root credentials, by contrast, leave a
					// running instance that is simply never ready: it keeps
					// serving `/status`, `/metrics` and the logs an operator needs
					// to see what the import did.
					match dbs::finish_startup(&ds, &pending_startup).await {
						Ok(()) => {
							ready.store(true, Ordering::SeqCst);
							info!("Startup complete; instance is now ready to serve");
						}
						Err(err) => {
							error!("Startup failed; instance will not become ready: {err}");
						}
					}
				} => {}
			}
		})
	});
	// With no startup import, everything deferred is bounded work against local
	// storage — the version check, the defaults, this node's registration and the
	// root credential hash — so the bind waits for it. A client that connects the
	// instant the port opens is then served, instead of being turned away with a
	// 503 it never asked to poll for; `/ready` answers 200 from the first probe.
	//
	// The wait is bounded because that work is only bounded when the datastore is
	// accepting transactions. One that is not — an engine still joining a
	// cluster, storage that has not come up — runs past this, and the listener
	// binds anyway so `/status` and `/ready` can answer for the rest of it
	// instead of leaving an operator with a refused connection and no way to tell
	// starting from dead.
	//
	// An import is never waited for: it is unbounded, and not starving the probes
	// for the length of one is why it is deferred at all.
	if !has_import && let Some(handle) = startup.as_mut() {
		if tokio::time::timeout(STARTUP_BIND_GRACE, handle).await.is_ok() {
			// The work finished inside the grace period, so the handle has already
			// yielded its output. Drop it here: a `JoinHandle` panics when polled
			// again after completion, and the shutdown join below is that second
			// poll. What that join is there for — not closing the storage engine
			// under this task's transactions — is satisfied either way, because
			// the task is done.
			startup = None;
		} else {
			info!(
				"Startup is still running after {STARTUP_BIND_GRACE:?}; binding the listener now and finishing in the background"
			);
		}
	}
	// A failure already known here has not been served against, so report it
	// without binding at all.
	if fatal.is_cancelled() {
		canceller.cancel();
		datastore.shutdown().await?;
		return Err(startup_failure(&mut failed_rx));
	}
	// Register datastore metrics against the unified meter provider. The
	// instruments flow to both the Prometheus text exporter (rendered by
	// `/metrics`) and the OTLP push exporter (when configured), so
	// operators get the same storage-engine gauges via either path.
	// Storage-backend metric names are not in `PUBLIC_METRICS`, so
	// unauthenticated `/metrics` scrapers never see them.
	if let Err(err) = crate::observe::register_storage_metrics(&datastore, &runtime) {
		warn!("failed to register storage metrics: {err}");
	}
	// The `/ready` probe treats the node as unhealthy if its cluster heartbeat
	// hasn't refreshed within a few cycles of the node-membership refresh task
	// (which also confirms the storage read and write paths are working).
	const READINESS_HEARTBEAT_STALENESS_FACTOR: u32 = 3;
	// `checked_mul` guards against overflow from an extreme configured interval;
	// `Duration::MAX` degrades to "heartbeat never considered stale".
	let max_heartbeat_age = config
		.engine
		.node_membership_refresh_interval
		.checked_mul(READINESS_HEARTBEAT_STALENESS_FACTOR)
		.unwrap_or(Duration::MAX);
	let readiness = ntw::Readiness {
		ready: Arc::clone(&ready),
		// The heartbeat freshness check applies on the server path, where the
		// node-membership refresh task keeps the heartbeat current.
		max_heartbeat_age: Some(max_heartbeat_age),
	};
	// Start the Postgres wire protocol listener when configured
	#[cfg(feature = "postgres")]
	if let Some(postgres_addr) = config.postgres_bind {
		crate::pg::start(
			postgres_addr,
			Arc::clone(&datastore),
			Arc::clone(&ready),
			canceller.clone(),
			config.crt.clone(),
			config.key.clone(),
		)
		.await?;
	}
	#[cfg(not(feature = "postgres"))]
	if config.postgres_bind.is_some() {
		return Err(anyhow::anyhow!(
			"The --postgres-bind option requires a binary built with the 'postgres' feature"
		));
	}
	// Build and run the HTTP server using the provided RouterFactory implementation
	let serve = ntw::init_with_metrics::<C>(
		&config,
		Arc::clone(&datastore),
		recv,
		canceller.clone(),
		router_state,
		metrics_state,
		readiness,
	);
	// The web server ends when an OS signal reaches its shutdown handler or its
	// listener fails. Cancelling the token does not end it: that handler waits
	// on a signal first and only then watches the token. So a startup failure
	// the instance cannot serve without has to stop the server from here, by
	// dropping it — otherwise the process would answer every probe with a 503
	// for as long as the pod lives instead of reporting the condition an
	// operator has to act on.
	let failed = tokio::select! {
		res = serve => {
			res?;
			// The server stopped on its own. A failure that landed while it was
			// stopping is still this process's exit status.
			fatal.is_cancelled()
		}
		_ = fatal.cancelled() => true,
	};
	// Shutdown and stop closed tasks
	canceller.cancel();
	// Join the deferred startup work before the engine closes under the
	// transactions it opens. The wait is short in practice because the task
	// selects on the token cancelled just above.
	if let Some(handle) = startup
		&& tokio::time::timeout(STARTUP_SHUTDOWN_JOIN, handle).await.is_err()
	{
		warn!(
			"Deferred startup work did not stop within {STARTUP_SHUTDOWN_JOIN:?}; continuing shutdown without it"
		);
	}
	// Wait for background tasks to finish
	// Shutdown the datastore
	let shutdown = datastore.shutdown().await;
	// Report a startup failure as this process's exit status, in preference to
	// whatever shutdown reports: a datastore that could not be initialised is
	// also the likeliest to fail on the way out, and the startup error is the
	// one that names what an operator has to act on.
	if failed {
		return Err(startup_failure(&mut failed_rx));
	}
	shutdown?;
	// All ok
	Ok(())
}

/// Takes the error a fatal startup step reported.
///
/// Only call this with the fatal signal tripped, and only once: the error is
/// sent before the signal, so a tripped signal means it is there to take. The
/// fallback stands in for a channel that has already been drained rather than
/// masking a different failure — the log line carrying the real error is written
/// where it happens, before the signal is tripped.
fn startup_failure(failed_rx: &mut oneshot::Receiver<anyhow::Error>) -> anyhow::Error {
	failed_rx
		.try_recv()
		.unwrap_or_else(|_| anyhow::anyhow!("the datastore could not be initialised"))
}

/// Build the observer that will be installed on the datastore along with the
/// `/metrics` state that will be attached to the HTTP router.
///
/// Behaviour:
///
/// - Every metric is recorded through the unified [`opentelemetry_sdk::metrics::SdkMeterProvider`]
///   built in [`crate::telemetry::metrics::init`]. The provider routes instruments to both the
///   Prometheus text exporter (rendered by `/metrics`) and the OTLP push exporter (when
///   configured).
/// - The process / pipeline observable gauges (`surrealdb.build.info`, `surrealdb.process.*`, audit
///   / slow-query self-metrics) are registered whenever any reader is attached -- either Prometheus
///   pull or OTLP push. This keeps OTLP-only deployments (`SURREAL_METRICS_ENABLED=false` +
///   `SURREAL_TELEMETRY_PROVIDER=otlp`) wired up to the same gauge surface as Prometheus scrapers.
///   The snapshot these gauges read is refreshed by the engine's maintenance scheduler regardless
///   of whether any reader is attached, because `INFO FOR ROOT` reads the same cache.
/// - When [`METRICS_ENABLED`] is `true`, a [`MetricsObserver`] is constructed and returned as part
///   of the [`MetricsState`] so the `/metrics` handler can reach the Prometheus text exporter; it
///   also lands in the fan-out so the labelled `surrealdb.*` family is recorded on every emit.
/// - The composer's [`ObservabilityProvider::create_observer`] is always invoked; composer
///   extensions use this hook to install per-tenant rollups, SurrealDS cluster, and audit /
///   slow-query observers under their own signal-domain scopes.
/// - The resulting fan-out is `[MetricsObserver?, composer]`: one labelled recording site for the
///   primary surface, plus whatever the composer contributes.
///
/// The returned [`MetricsState`] is `None` when metrics are disabled, which
/// keeps the `/metrics` route from being mounted at all.
#[allow(clippy::clone_on_ref_ptr)]
fn build_observability<C: ObservabilityProvider>(
	composer: &C,
	runtime: &ObservabilityRuntime,
) -> Result<(Option<MetricsState>, Arc<dyn ExecutionObserver>)> {
	let extra = composer.create_observer_with_runtime(runtime);

	// Register the process snapshot and pipeline self-metric gauges
	// against the unified meter provider whenever any reader -- Prometheus
	// pull or OTLP push -- is configured. Without this hoist OTLP-only
	// deployments would build the `SdkMeterProvider` (see
	// `telemetry::metrics::init`) and keep the process snapshot fresh but
	// never expose any gauges that read from the cache, leaving OTLP
	// collectors with zero `surrealdb.build.info` / `surrealdb.process.*` /
	// `surrealdb_audit_*` / `surrealdb_slow_query_*` samples.
	if *METRICS_ENABLED || otlp_metrics_active() {
		crate::observe::metrics::register_process_metrics(runtime);
		if let Some(counters) = composer.audit_counters() {
			MetricsObserver::register_pipeline_self_metrics(
				runtime,
				scope::AUDIT,
				"audit",
				counters,
			)?;
		}
		if let Some(counters) = composer.slow_query_counters() {
			MetricsObserver::register_pipeline_self_metrics(
				runtime,
				scope::SLOW_QUERY,
				"slow-query",
				counters,
			)?;
		}
	}

	if !*METRICS_ENABLED {
		// `/metrics` is disabled; keep the composer observers attached
		// so OTLP push and audit pipelines still receive events. The
		// gauge registrations above already wired up the observable
		// instruments against the OTLP reader.
		return Ok((None, extra));
	}

	// Build the unified labelled observer that exposes the
	// SurrealDB-native families via `/metrics`. The shared process /
	// pipeline gauge registrations above already ran on this branch.
	let metrics_observer = Arc::new(MetricsObserver::new(runtime)?);
	let metrics_obs: Arc<dyn ExecutionObserver> = metrics_observer.clone();
	let combined: Arc<dyn ExecutionObserver> = Arc::new(FanOutObserver::new([metrics_obs, extra]));
	// `/metrics` is mounted only when the runtime carries a Prometheus
	// exporter; otherwise the route is disabled but the labelled
	// observer still records to whichever reader (OTLP push, in-test
	// `ManualReader`, ...) the runtime carries.
	let metrics_state = runtime.prometheus_exporter().map(|exporter| MetricsState {
		exporter,
		observer: metrics_observer,
	});
	Ok((metrics_state, combined))
}