1use std::sync::OnceLock;
8
9use metrics::{SharedString, counter, gauge, histogram};
10use metrics_exporter_prometheus::{Matcher, PrometheusBuilder, PrometheusHandle};
11
12const HTTP_REQUESTS_TOTAL: &str = "praxis_http_requests_total";
18
19const HTTP_REQUEST_DURATION_SECONDS: &str = "praxis_http_request_duration_seconds";
21
22const HTTP_REQUEST_BODY_BYTES: &str = "praxis_http_request_body_bytes";
24
25const HTTP_RESPONSE_BODY_BYTES: &str = "praxis_http_response_body_bytes";
27
28const CONNECTIONS_ACTIVE: &str = "praxis_connections_active";
38
39const OVERLOAD_REJECTS_TOTAL: &str = "praxis_overload_rejects_total";
41
42const UPSTREAM_CONNECT_DURATION_SECONDS: &str = "praxis_upstream_connect_duration_seconds";
44
45const UPSTREAM_CONNECT_FAILURES_TOTAL: &str = "praxis_upstream_connect_failures_total";
47
48const UPSTREAM_RETRIES_TOTAL: &str = "praxis_upstream_retries_total";
50
51const UPSTREAM_HEALTHY_ENDPOINTS: &str = "praxis_upstream_healthy_endpoints";
53
54const UPSTREAM_TOTAL_ENDPOINTS: &str = "praxis_upstream_total_endpoints";
56
57const UPSTREAM_HEALTH_TRANSITIONS_TOTAL: &str = "praxis_upstream_health_transitions_total";
59
60const CONFIG_RELOAD_TOTAL: &str = "praxis_config_reload_total";
62
63const CONFIG_RELOAD_LAST_SUCCESS_TIMESTAMP: &str = "praxis_config_reload_last_success_timestamp";
65
66pub(crate) const OVERLOAD_REASON_MEMORY: &str = "memory";
68
69pub(crate) const OVERLOAD_REASON_GLOBAL_CONNECTIONS: &str = "global_connections";
71
72pub(crate) const OVERLOAD_REASON_LISTENER_CONNECTIONS: &str = "listener_connections";
74
75pub(crate) const RETRY_RESULT_SUCCESS: &str = "success";
77
78pub(crate) const RETRY_RESULT_EXHAUSTED: &str = "exhausted";
80
81pub(crate) const HEALTH_RESULT_HEALTHY: &str = "healthy";
83
84pub(crate) const HEALTH_RESULT_UNHEALTHY: &str = "unhealthy";
86
87pub(crate) const RELOAD_RESULT_SUCCESS: &str = "success";
89
90pub(crate) const RELOAD_RESULT_FAILURE: &str = "failure";
92
93const BODY_SIZE_BUCKETS_BYTES: &[f64] = &[
98 64.0,
99 256.0,
100 1_024.0,
101 4_096.0,
102 16_384.0,
103 65_536.0,
104 262_144.0,
105 1_048_576.0,
106 10_485_760.0,
107];
108
109static PROMETHEUS_HANDLE: OnceLock<PrometheusHandle> = OnceLock::new();
115
116pub fn install_prometheus_recorder() -> &'static PrometheusHandle {
126 #[expect(
127 clippy::expect_used,
128 reason = "recorder installation is a one-time startup operation"
129 )]
130 PROMETHEUS_HANDLE.get_or_init(|| {
131 PrometheusBuilder::new()
132 .set_buckets_for_metric(
133 Matcher::Full(HTTP_REQUEST_BODY_BYTES.to_owned()),
134 BODY_SIZE_BUCKETS_BYTES,
135 )
136 .expect("body request histogram buckets must be non-empty")
137 .set_buckets_for_metric(
138 Matcher::Full(HTTP_RESPONSE_BODY_BYTES.to_owned()),
139 BODY_SIZE_BUCKETS_BYTES,
140 )
141 .expect("body response histogram buckets must be non-empty")
142 .install_recorder()
143 .expect("failed to install Prometheus recorder")
144 })
145}
146
147pub fn render_prometheus() -> Option<String> {
151 PROMETHEUS_HANDLE.get().map(PrometheusHandle::render)
152}
153
154pub(crate) fn is_recorder_installed() -> bool {
156 PROMETHEUS_HANDLE.get().is_some()
157}
158
159pub fn status_class(code: u16) -> &'static str {
176 match code {
177 100..=199 => "1xx",
178 200..=299 => "2xx",
179 300..=399 => "3xx",
180 400..=499 => "4xx",
181 500..=599 => "5xx",
182 _ => "unknown",
183 }
184}
185
186pub fn method_label(method: &str) -> &'static str {
200 match method {
201 "GET" => "GET",
202 "POST" => "POST",
203 "PUT" => "PUT",
204 "DELETE" => "DELETE",
205 "PATCH" => "PATCH",
206 "HEAD" => "HEAD",
207 "OPTIONS" => "OPTIONS",
208 "TRACE" => "TRACE",
209 "CONNECT" => "CONNECT",
210 _ => "OTHER",
211 }
212}
213
214pub(crate) struct RequestMetricLabels {
226 pub cluster: SharedString,
228 pub method: &'static str,
230 pub route: SharedString,
232 pub status_class: &'static str,
234}
235
236pub(crate) fn record_request_metrics(labels: RequestMetricLabels, duration_secs: f64) {
238 let cluster = labels.cluster;
239 let route = labels.route;
240 counter!(
241 HTTP_REQUESTS_TOTAL,
242 "method" => labels.method,
243 "status_class" => labels.status_class,
244 "route" => route.clone(),
245 "cluster" => cluster.clone()
246 )
247 .increment(1);
248 histogram!(
249 HTTP_REQUEST_DURATION_SECONDS,
250 "method" => labels.method,
251 "status_class" => labels.status_class,
252 "route" => route,
253 "cluster" => cluster
254 )
255 .record(duration_secs);
256}
257
258pub(crate) fn record_body_size_metrics(
260 method: &'static str,
261 status_class: &'static str,
262 cluster: SharedString,
263 request_body_bytes: u64,
264 response_body_bytes: u64,
265) {
266 #[expect(
267 clippy::cast_precision_loss,
268 reason = "body byte counts as histogram observations; exact integer precision not required"
269 )]
270 {
271 histogram!(
272 HTTP_REQUEST_BODY_BYTES,
273 "method" => method,
274 "status_class" => status_class,
275 "cluster" => cluster.clone()
276 )
277 .record(request_body_bytes as f64);
278 histogram!(
279 HTTP_RESPONSE_BODY_BYTES,
280 "method" => method,
281 "status_class" => status_class,
282 "cluster" => cluster
283 )
284 .record(response_body_bytes as f64);
285 }
286}
287
288pub(crate) fn inc_connections_active(listener: SharedString) {
290 if !is_recorder_installed() {
291 return;
292 }
293 gauge!(CONNECTIONS_ACTIVE, "listener" => listener).increment(1.0);
294}
295
296pub(crate) fn dec_connections_active(listener: SharedString) {
298 if !is_recorder_installed() {
299 return;
300 }
301 gauge!(CONNECTIONS_ACTIVE, "listener" => listener).decrement(1.0);
302}
303
304pub struct ActiveConnectionGuard {
310 listener: SharedString,
312}
313
314impl ActiveConnectionGuard {
315 pub(crate) fn acquire(listener: SharedString) -> Self {
317 inc_connections_active(listener.clone());
318 Self { listener }
319 }
320}
321
322impl Drop for ActiveConnectionGuard {
323 fn drop(&mut self) {
324 dec_connections_active(self.listener.clone());
325 }
326}
327
328pub(crate) fn record_overload_reject(reason: &'static str) {
330 if !is_recorder_installed() {
331 return;
332 }
333 counter!(OVERLOAD_REJECTS_TOTAL, "reason" => reason).increment(1);
334}
335
336pub(crate) fn record_upstream_connect_duration(cluster: SharedString, duration_secs: f64) {
338 if !is_recorder_installed() {
339 return;
340 }
341 histogram!(UPSTREAM_CONNECT_DURATION_SECONDS, "cluster" => cluster).record(duration_secs);
342}
343
344pub(crate) fn record_upstream_connect_failure(cluster: SharedString) {
346 if !is_recorder_installed() {
347 return;
348 }
349 counter!(UPSTREAM_CONNECT_FAILURES_TOTAL, "cluster" => cluster).increment(1);
350}
351
352pub(crate) fn record_upstream_retry(cluster: SharedString, result: &'static str) {
354 if !is_recorder_installed() {
355 return;
356 }
357 counter!(UPSTREAM_RETRIES_TOTAL, "cluster" => cluster, "result" => result).increment(1);
358}
359
360pub(crate) fn set_upstream_endpoint_gauges(cluster: SharedString, healthy: usize, total: usize) {
362 if !is_recorder_installed() {
363 return;
364 }
365 #[expect(clippy::cast_precision_loss, reason = "endpoint counts fit f64 exactly below 2^53")]
366 {
367 gauge!(UPSTREAM_HEALTHY_ENDPOINTS, "cluster" => cluster.clone()).set(healthy as f64);
368 gauge!(UPSTREAM_TOTAL_ENDPOINTS, "cluster" => cluster).set(total as f64);
369 }
370}
371
372pub fn clear_stale_upstream_health_gauges<'a>(
377 previous_health_clusters: impl IntoIterator<Item = &'a str>,
378 current_health_clusters: impl IntoIterator<Item = &'a str>,
379) {
380 if !is_recorder_installed() {
381 return;
382 }
383 let current: std::collections::HashSet<&str> = current_health_clusters.into_iter().collect();
384 for name in previous_health_clusters {
385 if !current.contains(name) {
386 set_upstream_endpoint_gauges(SharedString::from(name.to_owned()), 0, 0);
387 }
388 }
389}
390
391pub fn seed_upstream_health_gauges(registry: &praxis_core::health::HealthRegistry) {
396 if !is_recorder_installed() {
397 return;
398 }
399 for (name, state) in registry.iter() {
400 let (healthy, total) = state.endpoint_counts();
401 set_upstream_endpoint_gauges(SharedString::from(name.as_ref().to_owned()), healthy, total);
402 }
403}
404
405pub(crate) fn record_health_transition(cluster: SharedString, result: &'static str, healthy: usize, total: usize) {
407 if !is_recorder_installed() {
408 return;
409 }
410 counter!(
411 UPSTREAM_HEALTH_TRANSITIONS_TOTAL,
412 "cluster" => cluster.clone(),
413 "result" => result
414 )
415 .increment(1);
416 set_upstream_endpoint_gauges(cluster, healthy, total);
417}
418
419pub(crate) fn count_healthy_endpoints(health: &praxis_core::health::ClusterHealthEntry) -> (usize, usize) {
421 health.endpoint_counts()
422}
423
424pub fn record_config_reload_success() {
426 if !is_recorder_installed() {
427 return;
428 }
429 counter!(CONFIG_RELOAD_TOTAL, "result" => RELOAD_RESULT_SUCCESS).increment(1);
430 let ts = std::time::SystemTime::now()
431 .duration_since(std::time::UNIX_EPOCH)
432 .map_or(0.0, |d| d.as_secs_f64());
433 gauge!(CONFIG_RELOAD_LAST_SUCCESS_TIMESTAMP).set(ts);
434}
435
436pub fn record_config_reload_failure() {
438 if !is_recorder_installed() {
439 return;
440 }
441 counter!(CONFIG_RELOAD_TOTAL, "result" => RELOAD_RESULT_FAILURE).increment(1);
442}
443
444pub(crate) fn cluster_none() -> SharedString {
448 SharedString::const_str("none")
449}
450
451pub(crate) fn route_unknown() -> SharedString {
455 SharedString::const_str("unknown")
456}
457
458#[cfg(test)]
463#[expect(clippy::allow_attributes, reason = "blanket test suppressions")]
464#[allow(clippy::unwrap_used, clippy::expect_used, clippy::indexing_slicing, reason = "tests")]
465mod tests {
466 use super::*;
467
468 #[test]
469 fn status_class_1xx() {
470 assert_eq!(status_class(100), "1xx", "100 should be 1xx");
471 assert_eq!(status_class(199), "1xx", "199 should be 1xx");
472 }
473
474 #[test]
475 fn status_class_2xx() {
476 assert_eq!(status_class(200), "2xx", "200 should be 2xx");
477 assert_eq!(status_class(204), "2xx", "204 should be 2xx");
478 assert_eq!(status_class(299), "2xx", "299 should be 2xx");
479 }
480
481 #[test]
482 fn status_class_3xx() {
483 assert_eq!(status_class(301), "3xx", "301 should be 3xx");
484 assert_eq!(status_class(399), "3xx", "399 should be 3xx");
485 }
486
487 #[test]
488 fn status_class_4xx() {
489 assert_eq!(status_class(400), "4xx", "400 should be 4xx");
490 assert_eq!(status_class(404), "4xx", "404 should be 4xx");
491 assert_eq!(status_class(499), "4xx", "499 should be 4xx");
492 }
493
494 #[test]
495 fn status_class_5xx() {
496 assert_eq!(status_class(500), "5xx", "500 should be 5xx");
497 assert_eq!(status_class(503), "5xx", "503 should be 5xx");
498 assert_eq!(status_class(599), "5xx", "599 should be 5xx");
499 }
500
501 #[test]
502 fn status_class_zero_is_unknown() {
503 assert_eq!(status_class(0), "unknown", "0 should be unknown");
504 }
505
506 #[test]
507 fn status_class_out_of_range_is_unknown() {
508 assert_eq!(status_class(600), "unknown", "600 should be unknown");
509 assert_eq!(status_class(99), "unknown", "99 should be unknown");
510 }
511
512 #[test]
513 fn method_label_standard_methods() {
514 for m in [
515 "GET", "POST", "PUT", "DELETE", "PATCH", "HEAD", "OPTIONS", "TRACE", "CONNECT",
516 ] {
517 assert_eq!(method_label(m), m, "{m} should pass through");
518 }
519 }
520
521 #[test]
522 fn method_label_custom_methods_collapse_to_other() {
523 assert_eq!(method_label("PURGE"), "OTHER", "PURGE should be OTHER");
524 assert_eq!(method_label("FOOBAR"), "OTHER", "FOOBAR should be OTHER");
525 assert_eq!(method_label(""), "OTHER", "empty should be OTHER");
526 }
527
528 #[test]
529 fn record_helpers_noop_without_recorder() {
530 record_overload_reject(OVERLOAD_REASON_MEMORY);
532 record_upstream_connect_failure(cluster_none());
533 record_upstream_retry(cluster_none(), RETRY_RESULT_SUCCESS);
534 record_upstream_connect_duration(cluster_none(), 0.01);
535 set_upstream_endpoint_gauges(cluster_none(), 1, 2);
536 record_health_transition(cluster_none(), HEALTH_RESULT_HEALTHY, 1, 2);
537 record_config_reload_success();
538 record_config_reload_failure();
539 clear_stale_upstream_health_gauges(["gone"], std::iter::empty::<&str>());
540 let _guard = ActiveConnectionGuard::acquire(SharedString::const_str("test"));
541 }
542
543 #[test]
544 fn overload_reject_reasons_appear_in_scrape() {
545 install_prometheus_recorder();
546 record_overload_reject(OVERLOAD_REASON_MEMORY);
547 record_overload_reject(OVERLOAD_REASON_GLOBAL_CONNECTIONS);
548 record_overload_reject(OVERLOAD_REASON_LISTENER_CONNECTIONS);
549 let body = render_prometheus().expect("recorder should render");
550 for reason in [
551 OVERLOAD_REASON_MEMORY,
552 OVERLOAD_REASON_GLOBAL_CONNECTIONS,
553 OVERLOAD_REASON_LISTENER_CONNECTIONS,
554 ] {
555 let needle = format!("praxis_overload_rejects_total{{reason=\"{reason}\"}}");
556 assert!(body.contains(&needle), "expected `{needle}` in scrape:\n{body}");
557 }
558 }
559
560 #[test]
561 fn body_size_histograms_use_byte_buckets() {
562 install_prometheus_recorder();
563 record_body_size_metrics("GET", "2xx", cluster_none(), 500, 4_000);
564 let body = render_prometheus().expect("recorder should render");
565 assert!(
566 body.contains("praxis_http_request_body_bytes_bucket") && body.contains("le=\"1024\""),
567 "request body histogram should use byte buckets, not duration defaults:\n{body}"
568 );
569 assert!(
570 body.contains("praxis_http_response_body_bytes_bucket") && body.contains("le=\"4096\""),
571 "response body histogram should use byte buckets:\n{body}"
572 );
573 assert!(
574 !body.contains("praxis_http_request_body_bytes_bucket{le=\"0.005\"}")
575 && !body.contains("praxis_http_request_body_bytes_bucket{method=\"GET\",status_class=\"2xx\",cluster=\"\",le=\"0.005\"}"),
576 "request body histogram must not use duration default buckets:\n{body}"
577 );
578 }
579
580 #[test]
581 fn clear_stale_upstream_health_gauges_zeros_removed_clusters() {
582 install_prometheus_recorder();
583 set_upstream_endpoint_gauges(SharedString::from("old-cluster".to_owned()), 2, 3);
584 set_upstream_endpoint_gauges(SharedString::from("kept-cluster".to_owned()), 1, 1);
585 clear_stale_upstream_health_gauges(["old-cluster", "kept-cluster"], ["kept-cluster"]);
586 let body = render_prometheus().expect("recorder should render");
587 assert!(
588 body.contains("praxis_upstream_healthy_endpoints{cluster=\"old-cluster\"} 0"),
589 "removed cluster healthy gauge should be zeroed:\n{body}"
590 );
591 assert!(
592 body.contains("praxis_upstream_total_endpoints{cluster=\"old-cluster\"} 0"),
593 "removed cluster total gauge should be zeroed:\n{body}"
594 );
595 assert!(
596 body.contains("praxis_upstream_healthy_endpoints{cluster=\"kept-cluster\"} 1"),
597 "kept cluster should retain its value:\n{body}"
598 );
599 }
600
601 #[test]
602 fn seed_upstream_health_gauges_publishes_registry_counts() {
603 use std::sync::Arc;
604
605 use praxis_core::health::{ClusterHealthEntry, EndpointHealth};
606
607 install_prometheus_recorder();
608 let endpoints = vec![EndpointHealth::new(), EndpointHealth::new()];
609 endpoints[0].mark_unhealthy();
610 let entry = Arc::new(ClusterHealthEntry::new(
611 endpoints,
612 vec![Arc::from("a:1"), Arc::from("b:1")],
613 None,
614 None,
615 ));
616 let registry = Arc::new([(Arc::from("backend"), entry)].into_iter().collect());
617 seed_upstream_health_gauges(®istry);
618 let body = render_prometheus().expect("recorder should render");
619 assert!(
620 body.contains("praxis_upstream_healthy_endpoints{cluster=\"backend\"} 1"),
621 "seed should publish healthy count:\n{body}"
622 );
623 assert!(
624 body.contains("praxis_upstream_total_endpoints{cluster=\"backend\"} 2"),
625 "seed should publish total count:\n{body}"
626 );
627 }
628
629 #[test]
630 fn count_healthy_endpoints_counts_correctly() {
631 use std::sync::Arc;
632
633 use praxis_core::health::{ClusterHealthEntry, EndpointHealth};
634
635 let endpoints = vec![EndpointHealth::new(), EndpointHealth::new(), EndpointHealth::new()];
636 endpoints[1].mark_unhealthy();
637 let entry = ClusterHealthEntry::new(
638 endpoints,
639 vec![Arc::from("a:1"), Arc::from("b:1"), Arc::from("c:1")],
640 None,
641 None,
642 );
643 let (healthy, total) = count_healthy_endpoints(&entry);
644 assert_eq!(total, 3, "total should be 3");
645 assert_eq!(healthy, 2, "two endpoints should be healthy");
646 }
647}