Skip to main content

praxis_protocol/http/pingora/
metrics.rs

1// SPDX-License-Identifier: Apache-2.0
2// Copyright (c) 2026 Praxis Contributors
3
4//! Prometheus metrics: recorder installation, HTTP/upstream/config metric
5//! recording, and scrape rendering.
6
7use std::sync::OnceLock;
8
9use metrics::{SharedString, counter, gauge, histogram};
10use metrics_exporter_prometheus::{Matcher, PrometheusBuilder, PrometheusHandle};
11
12// -----------------------------------------------------------------------------
13// Constants
14// -----------------------------------------------------------------------------
15
16/// Counter for completed HTTP requests.
17const HTTP_REQUESTS_TOTAL: &str = "praxis_http_requests_total";
18
19/// Histogram for HTTP request duration in seconds.
20const HTTP_REQUEST_DURATION_SECONDS: &str = "praxis_http_request_duration_seconds";
21
22/// Histogram for HTTP request body size in bytes.
23const HTTP_REQUEST_BODY_BYTES: &str = "praxis_http_request_body_bytes";
24
25/// Histogram for HTTP response body size in bytes.
26const HTTP_RESPONSE_BODY_BYTES: &str = "praxis_http_response_body_bytes";
27
28/// Gauge for concurrent in-flight proxy sessions per listener.
29///
30/// For HTTP, each active request holds a guard for its lifetime (so the
31/// series tracks in-flight requests more than raw TCP sockets). For TCP,
32/// each accepted connection holds a guard while the session is open.
33///
34/// When multiple TCP listeners share one Pingora service group, the
35/// `listener` label is resolved from the connection's local bind address
36/// (see TCP proxy listener-name map).
37const CONNECTIONS_ACTIVE: &str = "praxis_connections_active";
38
39/// Counter for connections rejected by overload protection.
40const OVERLOAD_REJECTS_TOTAL: &str = "praxis_overload_rejects_total";
41
42/// Histogram for upstream connect duration in seconds.
43const UPSTREAM_CONNECT_DURATION_SECONDS: &str = "praxis_upstream_connect_duration_seconds";
44
45/// Counter for upstream connect failures.
46const UPSTREAM_CONNECT_FAILURES_TOTAL: &str = "praxis_upstream_connect_failures_total";
47
48/// Counter for upstream connect-failure retries.
49const UPSTREAM_RETRIES_TOTAL: &str = "praxis_upstream_retries_total";
50
51/// Gauge for healthy endpoints per cluster.
52const UPSTREAM_HEALTHY_ENDPOINTS: &str = "praxis_upstream_healthy_endpoints";
53
54/// Gauge for total endpoints per cluster.
55const UPSTREAM_TOTAL_ENDPOINTS: &str = "praxis_upstream_total_endpoints";
56
57/// Counter for endpoint health state transitions.
58const UPSTREAM_HEALTH_TRANSITIONS_TOTAL: &str = "praxis_upstream_health_transitions_total";
59
60/// Counter for config reload attempts.
61const CONFIG_RELOAD_TOTAL: &str = "praxis_config_reload_total";
62
63/// Gauge for unix timestamp of last successful config reload.
64const CONFIG_RELOAD_LAST_SUCCESS_TIMESTAMP: &str = "praxis_config_reload_last_success_timestamp";
65
66/// Overload reject reason: process memory pressure.
67pub(crate) const OVERLOAD_REASON_MEMORY: &str = "memory";
68
69/// Overload reject reason: process-wide connection limit.
70pub(crate) const OVERLOAD_REASON_GLOBAL_CONNECTIONS: &str = "global_connections";
71
72/// Overload reject reason: per-listener connection limit.
73pub(crate) const OVERLOAD_REASON_LISTENER_CONNECTIONS: &str = "listener_connections";
74
75/// Retry result: connect eventually succeeded after at least one retry.
76pub(crate) const RETRY_RESULT_SUCCESS: &str = "success";
77
78/// Retry result: retries gave up or were exhausted.
79pub(crate) const RETRY_RESULT_EXHAUSTED: &str = "exhausted";
80
81/// Health transition result: endpoint became healthy.
82pub(crate) const HEALTH_RESULT_HEALTHY: &str = "healthy";
83
84/// Health transition result: endpoint became unhealthy.
85pub(crate) const HEALTH_RESULT_UNHEALTHY: &str = "unhealthy";
86
87/// Config reload result: success.
88pub(crate) const RELOAD_RESULT_SUCCESS: &str = "success";
89
90/// Config reload result: failure.
91pub(crate) const RELOAD_RESULT_FAILURE: &str = "failure";
92
93/// Histogram bucket upper bounds for HTTP body sizes in bytes.
94///
95/// Defaults from `PrometheusBuilder` target request durations in seconds
96/// (`0.005`…`10`), which collapse almost every body into `+Inf`.
97const BODY_SIZE_BUCKETS_BYTES: &[f64] = &[
98    64.0,
99    256.0,
100    1_024.0,
101    4_096.0,
102    16_384.0,
103    65_536.0,
104    262_144.0,
105    1_048_576.0,
106    10_485_760.0,
107];
108
109// -----------------------------------------------------------------------------
110// Recorder Installation
111// -----------------------------------------------------------------------------
112
113/// Global handle to the Prometheus exporter.
114static PROMETHEUS_HANDLE: OnceLock<PrometheusHandle> = OnceLock::new();
115
116/// Install the global Prometheus metrics recorder.
117///
118/// Must be called exactly once during server startup. Subsequent
119/// calls are no-ops and return the existing handle.
120///
121/// # Panics
122///
123/// Panics if the global recorder cannot be installed (another
124/// recorder was already set by a different subsystem).
125pub fn install_prometheus_recorder() -> &'static PrometheusHandle {
126    #[expect(
127        clippy::expect_used,
128        reason = "recorder installation is a one-time startup operation"
129    )]
130    PROMETHEUS_HANDLE.get_or_init(|| {
131        PrometheusBuilder::new()
132            .set_buckets_for_metric(
133                Matcher::Full(HTTP_REQUEST_BODY_BYTES.to_owned()),
134                BODY_SIZE_BUCKETS_BYTES,
135            )
136            .expect("body request histogram buckets must be non-empty")
137            .set_buckets_for_metric(
138                Matcher::Full(HTTP_RESPONSE_BODY_BYTES.to_owned()),
139                BODY_SIZE_BUCKETS_BYTES,
140            )
141            .expect("body response histogram buckets must be non-empty")
142            .install_recorder()
143            .expect("failed to install Prometheus recorder")
144    })
145}
146
147/// Render all collected metrics in Prometheus text exposition format.
148///
149/// Returns `None` if the recorder has not been installed.
150pub fn render_prometheus() -> Option<String> {
151    PROMETHEUS_HANDLE.get().map(PrometheusHandle::render)
152}
153
154/// Returns `true` if the Prometheus recorder has been installed.
155pub(crate) fn is_recorder_installed() -> bool {
156    PROMETHEUS_HANDLE.get().is_some()
157}
158
159// -----------------------------------------------------------------------------
160// Status Class
161// -----------------------------------------------------------------------------
162
163/// Map an HTTP status code to its class label (`"1xx"`, `"2xx"`, etc.).
164///
165/// Returns `"unknown"` for zero (no response written) or codes
166/// outside the 100–599 range.
167///
168/// ```
169/// use praxis_protocol::http::pingora::metrics::status_class;
170///
171/// assert_eq!(status_class(200), "2xx");
172/// assert_eq!(status_class(404), "4xx");
173/// assert_eq!(status_class(0), "unknown");
174/// ```
175pub fn status_class(code: u16) -> &'static str {
176    match code {
177        100..=199 => "1xx",
178        200..=299 => "2xx",
179        300..=399 => "3xx",
180        400..=499 => "4xx",
181        500..=599 => "5xx",
182        _ => "unknown",
183    }
184}
185
186/// Map an HTTP method to a bounded label value.
187///
188/// Returns the method string for the nine standard methods
189/// defined in [RFC 9110]; all others collapse to `"OTHER"`.
190///
191/// ```
192/// use praxis_protocol::http::pingora::metrics::method_label;
193///
194/// assert_eq!(method_label("GET"), "GET");
195/// assert_eq!(method_label("PURGE"), "OTHER");
196/// ```
197///
198/// [RFC 9110]: https://datatracker.ietf.org/doc/html/rfc9110#section-9.1
199pub fn method_label(method: &str) -> &'static str {
200    match method {
201        "GET" => "GET",
202        "POST" => "POST",
203        "PUT" => "PUT",
204        "DELETE" => "DELETE",
205        "PATCH" => "PATCH",
206        "HEAD" => "HEAD",
207        "OPTIONS" => "OPTIONS",
208        "TRACE" => "TRACE",
209        "CONNECT" => "CONNECT",
210        _ => "OTHER",
211    }
212}
213
214// -----------------------------------------------------------------------------
215// Metric Recording
216// -----------------------------------------------------------------------------
217
218/// Labels for a completed HTTP request.
219///
220/// Static labels (`method`, `status_class`) use `&'static str`
221/// so the metrics facade can intern them without per-request allocation.
222/// `cluster` and `route` are dynamic [`SharedString`] values.
223///
224/// [`SharedString`]: ::metrics::SharedString
225pub(crate) struct RequestMetricLabels {
226    /// Cluster name or `"none"`.
227    pub cluster: SharedString,
228    /// HTTP method (e.g. `"GET"`).
229    pub method: &'static str,
230    /// Route path-match pattern or `"unknown"`.
231    pub route: SharedString,
232    /// Status class (e.g. `"2xx"`).
233    pub status_class: &'static str,
234}
235
236/// Record HTTP request metrics for a completed request.
237pub(crate) fn record_request_metrics(labels: RequestMetricLabels, duration_secs: f64) {
238    if !is_recorder_installed() {
239        return;
240    }
241    let cluster = labels.cluster;
242    let route = labels.route;
243    counter!(
244        HTTP_REQUESTS_TOTAL,
245        "method" => labels.method,
246        "status_class" => labels.status_class,
247        "route" => route.clone(),
248        "cluster" => cluster.clone()
249    )
250    .increment(1);
251    histogram!(
252        HTTP_REQUEST_DURATION_SECONDS,
253        "method" => labels.method,
254        "status_class" => labels.status_class,
255        "route" => route,
256        "cluster" => cluster
257    )
258    .record(duration_secs);
259}
260
261/// Record HTTP request and response body size histograms.
262pub(crate) fn record_body_size_metrics(
263    method: &'static str,
264    status_class: &'static str,
265    cluster: SharedString,
266    request_body_bytes: u64,
267    response_body_bytes: u64,
268) {
269    if !is_recorder_installed() {
270        return;
271    }
272    #[expect(
273        clippy::cast_precision_loss,
274        reason = "body byte counts as histogram observations; exact integer precision not required"
275    )]
276    {
277        histogram!(
278            HTTP_REQUEST_BODY_BYTES,
279            "method" => method,
280            "status_class" => status_class,
281            "cluster" => cluster.clone()
282        )
283        .record(request_body_bytes as f64);
284        histogram!(
285            HTTP_RESPONSE_BODY_BYTES,
286            "method" => method,
287            "status_class" => status_class,
288            "cluster" => cluster
289        )
290        .record(response_body_bytes as f64);
291    }
292}
293
294/// Increment the active-connections gauge for a listener.
295pub(crate) fn inc_connections_active(listener: SharedString) {
296    if !is_recorder_installed() {
297        return;
298    }
299    gauge!(CONNECTIONS_ACTIVE, "listener" => listener).increment(1.0);
300}
301
302/// Decrement the active-connections gauge for a listener.
303pub(crate) fn dec_connections_active(listener: SharedString) {
304    if !is_recorder_installed() {
305        return;
306    }
307    gauge!(CONNECTIONS_ACTIVE, "listener" => listener).decrement(1.0);
308}
309
310/// RAII guard that decrements `praxis_connections_active` on drop.
311///
312/// Acquired once per HTTP request / TCP session. See the
313/// `CONNECTIONS_ACTIVE` constant docs for HTTP vs TCP semantics and
314/// grouped TCP listener labeling.
315pub struct ActiveConnectionGuard {
316    /// Listener name label.
317    listener: SharedString,
318}
319
320impl ActiveConnectionGuard {
321    /// Increment the gauge and return a guard that decrements on drop.
322    pub(crate) fn acquire(listener: SharedString) -> Self {
323        inc_connections_active(listener.clone());
324        Self { listener }
325    }
326}
327
328impl Drop for ActiveConnectionGuard {
329    fn drop(&mut self) {
330        dec_connections_active(self.listener.clone());
331    }
332}
333
334/// Record an overload rejection.
335pub(crate) fn record_overload_reject(reason: &'static str) {
336    if !is_recorder_installed() {
337        return;
338    }
339    counter!(OVERLOAD_REJECTS_TOTAL, "reason" => reason).increment(1);
340}
341
342/// Record upstream connect duration for a cluster.
343pub(crate) fn record_upstream_connect_duration(cluster: SharedString, duration_secs: f64) {
344    if !is_recorder_installed() {
345        return;
346    }
347    histogram!(UPSTREAM_CONNECT_DURATION_SECONDS, "cluster" => cluster).record(duration_secs);
348}
349
350/// Record an upstream connect failure.
351pub(crate) fn record_upstream_connect_failure(cluster: SharedString) {
352    if !is_recorder_installed() {
353        return;
354    }
355    counter!(UPSTREAM_CONNECT_FAILURES_TOTAL, "cluster" => cluster).increment(1);
356}
357
358/// Record an upstream connect-failure retry outcome.
359pub(crate) fn record_upstream_retry(cluster: SharedString, result: &'static str) {
360    if !is_recorder_installed() {
361        return;
362    }
363    counter!(UPSTREAM_RETRIES_TOTAL, "cluster" => cluster, "result" => result).increment(1);
364}
365
366/// Refresh cluster endpoint health gauges.
367pub(crate) fn set_upstream_endpoint_gauges(cluster: SharedString, healthy: usize, total: usize) {
368    if !is_recorder_installed() {
369        return;
370    }
371    #[expect(clippy::cast_precision_loss, reason = "endpoint counts fit f64 exactly below 2^53")]
372    {
373        gauge!(UPSTREAM_HEALTHY_ENDPOINTS, "cluster" => cluster.clone()).set(healthy as f64);
374        gauge!(UPSTREAM_TOTAL_ENDPOINTS, "cluster" => cluster).set(total as f64);
375    }
376}
377
378/// Zero health gauges for clusters that lost active health checks on reload.
379///
380/// Prometheus does not drop series automatically; clearing removed clusters
381/// prevents stale `healthy`/`total` values from lingering after config change.
382pub fn clear_stale_upstream_health_gauges<'a, P: IntoIterator<Item = &'a str>, C: IntoIterator<Item = &'a str>>(
383    previous_health_clusters: P,
384    current_health_clusters: C,
385) {
386    if !is_recorder_installed() {
387        return;
388    }
389    let current: std::collections::HashSet<&str> = current_health_clusters.into_iter().collect();
390    for name in previous_health_clusters {
391        if !current.contains(name) {
392            set_upstream_endpoint_gauges(SharedString::from(name.to_owned()), 0, 0);
393        }
394    }
395}
396
397/// Publish current healthy/total gauges for every cluster in a health registry.
398///
399/// Called on reload so scrapes reflect the new registry immediately instead of
400/// waiting for the first probe round.
401pub fn seed_upstream_health_gauges(registry: &praxis_core::health::HealthRegistry) {
402    if !is_recorder_installed() {
403        return;
404    }
405    for (name, state) in registry.iter() {
406        let (healthy, total) = state.endpoint_counts();
407        set_upstream_endpoint_gauges(SharedString::from(name.as_ref().to_owned()), healthy, total);
408    }
409}
410
411/// Record an endpoint health state transition and refresh gauges.
412pub(crate) fn record_health_transition(cluster: SharedString, result: &'static str, healthy: usize, total: usize) {
413    if !is_recorder_installed() {
414        return;
415    }
416    counter!(
417        UPSTREAM_HEALTH_TRANSITIONS_TOTAL,
418        "cluster" => cluster.clone(),
419        "result" => result
420    )
421    .increment(1);
422    set_upstream_endpoint_gauges(cluster, healthy, total);
423}
424
425/// Count healthy endpoints in a cluster health entry.
426pub(crate) fn count_healthy_endpoints(health: &praxis_core::health::ClusterHealthEntry) -> (usize, usize) {
427    health.endpoint_counts()
428}
429
430/// Record a successful config reload.
431pub fn record_config_reload_success() {
432    if !is_recorder_installed() {
433        return;
434    }
435    counter!(CONFIG_RELOAD_TOTAL, "result" => RELOAD_RESULT_SUCCESS).increment(1);
436    let ts = std::time::SystemTime::now()
437        .duration_since(std::time::UNIX_EPOCH)
438        .map_or(0.0, |d| d.as_secs_f64());
439    gauge!(CONFIG_RELOAD_LAST_SUCCESS_TIMESTAMP).set(ts);
440}
441
442/// Record a failed config reload.
443pub fn record_config_reload_failure() {
444    if !is_recorder_installed() {
445        return;
446    }
447    counter!(CONFIG_RELOAD_TOTAL, "result" => RELOAD_RESULT_FAILURE).increment(1);
448}
449
450/// [`SharedString`] for the `"none"` cluster label.
451///
452/// [`SharedString`]: ::metrics::SharedString
453pub(crate) fn cluster_none() -> SharedString {
454    SharedString::const_str("none")
455}
456
457/// [`SharedString`] for the `"unknown"` route label.
458///
459/// [`SharedString`]: ::metrics::SharedString
460pub(crate) fn route_unknown() -> SharedString {
461    SharedString::const_str("unknown")
462}
463
464// -----------------------------------------------------------------------------
465// Tests
466// -----------------------------------------------------------------------------
467
468#[cfg(test)]
469#[expect(clippy::allow_attributes, reason = "blanket test suppressions")]
470#[allow(clippy::unwrap_used, clippy::expect_used, clippy::indexing_slicing, reason = "tests")]
471mod tests {
472    use super::*;
473
474    #[test]
475    fn status_class_1xx() {
476        assert_eq!(status_class(100), "1xx", "100 should be 1xx");
477        assert_eq!(status_class(199), "1xx", "199 should be 1xx");
478    }
479
480    #[test]
481    fn status_class_2xx() {
482        assert_eq!(status_class(200), "2xx", "200 should be 2xx");
483        assert_eq!(status_class(204), "2xx", "204 should be 2xx");
484        assert_eq!(status_class(299), "2xx", "299 should be 2xx");
485    }
486
487    #[test]
488    fn status_class_3xx() {
489        assert_eq!(status_class(301), "3xx", "301 should be 3xx");
490        assert_eq!(status_class(399), "3xx", "399 should be 3xx");
491    }
492
493    #[test]
494    fn status_class_4xx() {
495        assert_eq!(status_class(400), "4xx", "400 should be 4xx");
496        assert_eq!(status_class(404), "4xx", "404 should be 4xx");
497        assert_eq!(status_class(499), "4xx", "499 should be 4xx");
498    }
499
500    #[test]
501    fn status_class_5xx() {
502        assert_eq!(status_class(500), "5xx", "500 should be 5xx");
503        assert_eq!(status_class(503), "5xx", "503 should be 5xx");
504        assert_eq!(status_class(599), "5xx", "599 should be 5xx");
505    }
506
507    #[test]
508    fn status_class_zero_is_unknown() {
509        assert_eq!(status_class(0), "unknown", "0 should be unknown");
510    }
511
512    #[test]
513    fn status_class_out_of_range_is_unknown() {
514        assert_eq!(status_class(600), "unknown", "600 should be unknown");
515        assert_eq!(status_class(99), "unknown", "99 should be unknown");
516    }
517
518    #[test]
519    fn method_label_standard_methods() {
520        for m in [
521            "GET", "POST", "PUT", "DELETE", "PATCH", "HEAD", "OPTIONS", "TRACE", "CONNECT",
522        ] {
523            assert_eq!(method_label(m), m, "{m} should pass through");
524        }
525    }
526
527    #[test]
528    fn method_label_custom_methods_collapse_to_other() {
529        assert_eq!(method_label("PURGE"), "OTHER", "PURGE should be OTHER");
530        assert_eq!(method_label("FOOBAR"), "OTHER", "FOOBAR should be OTHER");
531        assert_eq!(method_label(""), "OTHER", "empty should be OTHER");
532    }
533
534    #[test]
535    fn record_helpers_noop_without_recorder() {
536        // Must not panic when the Prometheus recorder is absent.
537        record_overload_reject(OVERLOAD_REASON_MEMORY);
538        record_upstream_connect_failure(cluster_none());
539        record_upstream_retry(cluster_none(), RETRY_RESULT_SUCCESS);
540        record_upstream_connect_duration(cluster_none(), 0.01);
541        set_upstream_endpoint_gauges(cluster_none(), 1, 2);
542        record_health_transition(cluster_none(), HEALTH_RESULT_HEALTHY, 1, 2);
543        record_config_reload_success();
544        record_config_reload_failure();
545        clear_stale_upstream_health_gauges(["gone"], std::iter::empty::<&str>());
546        let _guard = ActiveConnectionGuard::acquire(SharedString::const_str("test"));
547    }
548
549    #[test]
550    fn overload_reject_reasons_appear_in_scrape() {
551        install_prometheus_recorder();
552        record_overload_reject(OVERLOAD_REASON_MEMORY);
553        record_overload_reject(OVERLOAD_REASON_GLOBAL_CONNECTIONS);
554        record_overload_reject(OVERLOAD_REASON_LISTENER_CONNECTIONS);
555        let body = render_prometheus().expect("recorder should render");
556        for reason in [
557            OVERLOAD_REASON_MEMORY,
558            OVERLOAD_REASON_GLOBAL_CONNECTIONS,
559            OVERLOAD_REASON_LISTENER_CONNECTIONS,
560        ] {
561            let needle = format!("praxis_overload_rejects_total{{reason=\"{reason}\"}}");
562            assert!(body.contains(&needle), "expected `{needle}` in scrape:\n{body}");
563        }
564    }
565
566    #[test]
567    fn body_size_histograms_use_byte_buckets() {
568        install_prometheus_recorder();
569        record_body_size_metrics("GET", "2xx", cluster_none(), 500, 4_000);
570        let body = render_prometheus().expect("recorder should render");
571        assert!(
572            body.contains("praxis_http_request_body_bytes_bucket") && body.contains("le=\"1024\""),
573            "request body histogram should use byte buckets, not duration defaults:\n{body}"
574        );
575        assert!(
576            body.contains("praxis_http_response_body_bytes_bucket") && body.contains("le=\"4096\""),
577            "response body histogram should use byte buckets:\n{body}"
578        );
579        assert!(
580            !body.contains("praxis_http_request_body_bytes_bucket{le=\"0.005\"}")
581                && !body.contains("praxis_http_request_body_bytes_bucket{method=\"GET\",status_class=\"2xx\",cluster=\"\",le=\"0.005\"}"),
582            "request body histogram must not use duration default buckets:\n{body}"
583        );
584    }
585
586    #[test]
587    fn clear_stale_upstream_health_gauges_zeros_removed_clusters() {
588        install_prometheus_recorder();
589        set_upstream_endpoint_gauges(SharedString::from("old-cluster".to_owned()), 2, 3);
590        set_upstream_endpoint_gauges(SharedString::from("kept-cluster".to_owned()), 1, 1);
591        clear_stale_upstream_health_gauges(["old-cluster", "kept-cluster"], ["kept-cluster"]);
592        let body = render_prometheus().expect("recorder should render");
593        assert!(
594            body.contains("praxis_upstream_healthy_endpoints{cluster=\"old-cluster\"} 0"),
595            "removed cluster healthy gauge should be zeroed:\n{body}"
596        );
597        assert!(
598            body.contains("praxis_upstream_total_endpoints{cluster=\"old-cluster\"} 0"),
599            "removed cluster total gauge should be zeroed:\n{body}"
600        );
601        assert!(
602            body.contains("praxis_upstream_healthy_endpoints{cluster=\"kept-cluster\"} 1"),
603            "kept cluster should retain its value:\n{body}"
604        );
605    }
606
607    #[test]
608    fn seed_upstream_health_gauges_publishes_registry_counts() {
609        use std::sync::Arc;
610
611        use praxis_core::health::{ClusterHealthEntry, EndpointHealth};
612
613        install_prometheus_recorder();
614        let endpoints = vec![EndpointHealth::new(), EndpointHealth::new()];
615        endpoints[0].mark_unhealthy();
616        let entry = Arc::new(ClusterHealthEntry::new(
617            endpoints,
618            vec![Arc::from("a:1"), Arc::from("b:1")],
619            None,
620            None,
621        ));
622        let registry = Arc::new([(Arc::from("backend"), entry)].into_iter().collect());
623        seed_upstream_health_gauges(&registry);
624        let body = render_prometheus().expect("recorder should render");
625        assert!(
626            body.contains("praxis_upstream_healthy_endpoints{cluster=\"backend\"} 1"),
627            "seed should publish healthy count:\n{body}"
628        );
629        assert!(
630            body.contains("praxis_upstream_total_endpoints{cluster=\"backend\"} 2"),
631            "seed should publish total count:\n{body}"
632        );
633    }
634
635    #[test]
636    fn count_healthy_endpoints_counts_correctly() {
637        use std::sync::Arc;
638
639        use praxis_core::health::{ClusterHealthEntry, EndpointHealth};
640
641        let endpoints = vec![EndpointHealth::new(), EndpointHealth::new(), EndpointHealth::new()];
642        endpoints[1].mark_unhealthy();
643        let entry = ClusterHealthEntry::new(
644            endpoints,
645            vec![Arc::from("a:1"), Arc::from("b:1"), Arc::from("c:1")],
646            None,
647            None,
648        );
649        let (healthy, total) = count_healthy_endpoints(&entry);
650        assert_eq!(total, 3, "total should be 3");
651        assert_eq!(healthy, 2, "two endpoints should be healthy");
652    }
653}