turnframe-telemetry 0.1.0

Tracing, metrics and optional OpenTelemetry helpers for Turnframe
Documentation
//! The attribute keys this crate emits, in one place.
//!
//! Provider adapters, applications and the OpenTelemetry bridge all spell an
//! attribute the same way only if they spell it once. Every key a span or an
//! OTel instrument carries is a constant here.
//!
//! Two families live in this module:
//!
//! * **GenAI semantic conventions** (`gen_ai.*`) on a provider-call span. These
//!   are the OpenTelemetry conventions every LLM observability backend already
//!   reads — Langfuse, Datadog LLM Observability, Phoenix, Braintrust — so a
//!   Turnframe application is legible to them without an adapter.
//! * **Trace grouping**, vendor-neutral: which session a span belongs to, which
//!   (hashed) end user, which tags, which environment and which release. These
//!   use the general OpenTelemetry resource and session conventions rather than
//!   any vendor's names.
//!
//! A backend that wants its own spelling — `langfuse.session.id` instead of
//! `session.id`, say — is a five-line renaming function in the adopter's code,
//! mapping [`crate::tracing::TraceGrouping::fields`] onto its own keys. This
//! crate deliberately hardcodes no vendor.

// ---------------------------------------------------------------------------
// GenAI semantic conventions
// ---------------------------------------------------------------------------

/// The provider the call went to, e.g. `openai`. Our provider key.
pub const GEN_AI_SYSTEM: &str = "gen_ai.system";

/// The operation the call performed. Turnframe records the normalized request
/// purpose here, e.g. `extract` (spec §20.2).
pub const GEN_AI_OPERATION_NAME: &str = "gen_ai.operation.name";

/// The model asked for, e.g. `gpt-x`. Our model key.
pub const GEN_AI_REQUEST_MODEL: &str = "gen_ai.request.model";

/// The sampling temperature requested.
pub const GEN_AI_REQUEST_TEMPERATURE: &str = "gen_ai.request.temperature";

/// The provider's own identifier for the response, for support tickets.
pub const GEN_AI_RESPONSE_ID: &str = "gen_ai.response.id";

/// The model that actually answered, which is not always the one asked for.
pub const GEN_AI_RESPONSE_MODEL: &str = "gen_ai.response.model";

/// Why generation stopped, e.g. `stop`, `length`, `content_filter`.
pub const GEN_AI_RESPONSE_FINISH_REASONS: &str = "gen_ai.response.finish_reasons";

/// Input tokens billed for this call, **net of cached tokens**.
///
/// A consumer that adds [`GEN_AI_USAGE_INPUT_TOKENS`] and
/// [`GEN_AI_USAGE_INPUT_CACHED_TOKENS`] gets the total prompt size, and one
/// that reads only this key gets the uncached cost. Neither double counts.
pub const GEN_AI_USAGE_INPUT_TOKENS: &str = "gen_ai.usage.input_tokens";

/// Output tokens generated by the call.
pub const GEN_AI_USAGE_OUTPUT_TOKENS: &str = "gen_ai.usage.output_tokens";

/// Input tokens served from the provider's prompt cache. Excluded from
/// [`GEN_AI_USAGE_INPUT_TOKENS`].
pub const GEN_AI_USAGE_INPUT_CACHED_TOKENS: &str = "gen_ai.usage.input_cached_tokens";

/// The prompt sent to the model. **User data**: recorded only when content
/// recording is explicitly enabled and only through a redaction hook.
pub const GEN_AI_INPUT_MESSAGES: &str = "gen_ai.input.messages";

/// The completion returned by the model. **User data**: recorded only when
/// content recording is explicitly enabled and only through a redaction hook.
pub const GEN_AI_OUTPUT_MESSAGES: &str = "gen_ai.output.messages";

/// Every GenAI attribute key, for a consumer that wants to enumerate them.
pub const GEN_AI_KEYS: [&str; 12] = [
    GEN_AI_SYSTEM,
    GEN_AI_OPERATION_NAME,
    GEN_AI_REQUEST_MODEL,
    GEN_AI_REQUEST_TEMPERATURE,
    GEN_AI_RESPONSE_ID,
    GEN_AI_RESPONSE_MODEL,
    GEN_AI_RESPONSE_FINISH_REASONS,
    GEN_AI_USAGE_INPUT_TOKENS,
    GEN_AI_USAGE_OUTPUT_TOKENS,
    GEN_AI_USAGE_INPUT_CACHED_TOKENS,
    GEN_AI_INPUT_MESSAGES,
    GEN_AI_OUTPUT_MESSAGES,
];

// ---------------------------------------------------------------------------
// Trace grouping
// ---------------------------------------------------------------------------

/// The session a span belongs to. Turnframe uses the conversation id, which is
/// what makes a multi-turn thread one thing in a backend's session view.
pub const SESSION_ID: &str = "session.id";

/// A reference to the end user, **always a digest**. The raw account or user
/// identifier never leaves the process (spec §25.5).
pub const USER_ID: &str = "user.id";

/// Free-form grouping labels the application chose, comma separated.
pub const TAGS: &str = "tags";

/// Deployment environment, e.g. `production`, `staging`.
pub const DEPLOYMENT_ENVIRONMENT: &str = "deployment.environment.name";

/// Release or build of the running service.
pub const SERVICE_VERSION: &str = "service.version";

/// Every trace-grouping key, in the order they are emitted.
pub const GROUPING_KEYS: [&str; 5] = [
    SESSION_ID,
    USER_ID,
    TAGS,
    DEPLOYMENT_ENVIRONMENT,
    SERVICE_VERSION,
];

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn every_key_is_distinct_and_conventionally_named() {
        let mut keys: Vec<&str> = GEN_AI_KEYS
            .iter()
            .chain(GROUPING_KEYS.iter())
            .copied()
            .collect();
        let total = keys.len();
        keys.sort_unstable();
        keys.dedup();
        assert_eq!(keys.len(), total, "an attribute key is declared twice");
        assert!(GEN_AI_KEYS.iter().all(|key| key.starts_with("gen_ai.")));
        assert!(
            GROUPING_KEYS.iter().all(|key| !key.starts_with("gen_ai.")),
            "grouping keys must stay vendor- and domain-neutral"
        );
    }

    #[test]
    fn cached_tokens_are_a_separate_key_from_input_tokens() {
        assert_ne!(GEN_AI_USAGE_INPUT_TOKENS, GEN_AI_USAGE_INPUT_CACHED_TOKENS);
    }
}