Skip to main content

ironflow_store/
store.rs

1//! The [`RunStore`] trait — async storage abstraction for runs and steps.
2//!
3//! Implement this trait to plug in any backing store. Built-in implementations:
4//!
5//! - [`InMemoryStore`](crate::memory::InMemoryStore) — development and testing.
6//! - `PostgresStore` — production (behind the `store-postgres` feature).
7
8use std::future::Future;
9use std::pin::Pin;
10
11use chrono::{DateTime, Utc};
12use uuid::Uuid;
13
14use crate::api_key_store::ApiKeyStore;
15use crate::approval_delegation_store::ApprovalDelegationStore;
16use crate::artifact_store::ArtifactStore;
17use crate::audit_log_store::AuditLogStore;
18use crate::entities::{
19    ConcurrencyGroupBacklog, LeaseRequest, NewRun, NewStep, NewStepDependency, Page, PurgePolicy,
20    PurgeableRun, ReapedRun, Run, RunCreation, RunFilter, RunStats, RunStatus, RunUpdate,
21    StatsHistoryBucket, StatsHistoryFilter, Step, StepApproval, StepDependency, StepUpdate,
22    WorkerCapabilities,
23};
24use crate::error::StoreError;
25use crate::log_store::LogStore;
26use crate::provider_account_store::ProviderAccountStore;
27use crate::schedule_store::ScheduleStore;
28use crate::secret_store::SecretStore;
29use crate::signal_store::SignalStore;
30use crate::user_store::UserStore;
31
32/// Boxed future for [`RunStore`] methods — ensures object safety for `dyn RunStore`.
33pub type StoreFuture<'a, T> = Pin<Box<dyn Future<Output = Result<T, StoreError>> + Send + 'a>>;
34
35/// Error recorded on a run that exhausted its retries through lease expiries.
36///
37/// Set by [`RunStore::reap_expired_leases`] when a run has been recovered more
38/// than `max_retries` times.
39pub const LEASE_EXPIRED_ERROR: &str = "worker lease expired";
40
41/// Error recorded on a step that was running when its worker lost the lease.
42///
43/// Set by the reaper on the `Running` steps of a run that
44/// [`RunStore::reap_expired_leases`] requeued. The engine executes such a step
45/// again at the same position when the run is picked up, keeping the
46/// interrupted record in the step history.
47///
48/// # Examples
49///
50/// ```
51/// use ironflow_store::store::{LEASE_EXPIRED_ERROR, STEP_INTERRUPTED_ERROR};
52///
53/// assert_eq!(STEP_INTERRUPTED_ERROR, "interrupted: worker lease lost");
54/// assert_ne!(STEP_INTERRUPTED_ERROR, LEASE_EXPIRED_ERROR);
55/// ```
56pub const STEP_INTERRUPTED_ERROR: &str = "interrupted: worker lease lost";
57
58/// Async storage abstraction for workflow runs and steps.
59///
60/// All methods return a [`StoreFuture`] (boxed future) to maintain object safety,
61/// allowing the store to be used as `Arc<dyn RunStore>`.
62///
63/// # Examples
64///
65/// ```no_run
66/// use std::collections::HashMap;
67/// use ironflow_store::prelude::*;
68/// use serde_json::json;
69/// use uuid::Uuid;
70///
71/// # async fn example() -> Result<(), ironflow_store::error::StoreError> {
72/// let store = InMemoryStore::new();
73///
74/// let run = store.create_run(NewRun {
75///     workflow_name: "deploy".to_string(),
76///     trigger: TriggerKind::Manual,
77///     payload: json!({}),
78///     max_retries: 3,
79///     handler_version: None,
80///     labels: HashMap::new(),
81///     scheduled_at: None,
82///     created_by: None,
83///     idempotency_key: None,
84///     concurrency_key: None,
85///     concurrency_limits: Vec::new(),
86///     max_cost_usd: None,
87///     worker_tags: Vec::new(),
88/// }).await?.into_run();
89///
90/// let fetched = store.get_run(run.id).await?;
91/// assert!(fetched.is_some());
92/// # Ok(())
93/// # }
94/// ```
95pub trait RunStore: Send + Sync {
96    /// Create a new run in `Pending` status.
97    ///
98    /// When [`NewRun::idempotency_key`] is set and already bound to a run created
99    /// within [`IDEMPOTENCY_WINDOW`](crate::entities::IDEMPOTENCY_WINDOW), nothing is
100    /// inserted and that run is returned as [`RunCreation::Existing`]. A key bound to
101    /// an older run is released and reused for the new one.
102    ///
103    /// Concurrent calls sharing the same key resolve to a single run: exactly one
104    /// receives [`RunCreation::Created`], the others [`RunCreation::Existing`].
105    ///
106    /// When [`NewRun::concurrency_key`] is set, the idempotency lookup runs first,
107    /// then the key is checked: concurrent calls sharing it are serialized, and
108    /// at most one non-terminal run holds it at a time.
109    ///
110    /// [`NewRun::concurrency_limits`] is validated before anything is written.
111    ///
112    /// # Errors
113    ///
114    /// Returns [`StoreError::ConcurrencyConflict`](crate::error::StoreError::ConcurrencyConflict)
115    /// when a run that is not Completed, Failed, Warning or Cancelled already
116    /// holds [`NewRun::concurrency_key`],
117    /// [`StoreError::InvalidConcurrencyLimit`](crate::error::StoreError::InvalidConcurrencyLimit)
118    /// when [`NewRun::concurrency_limits`] holds an empty or too long group, a
119    /// zero limit or a duplicated group, and a database error when the backing
120    /// store fails.
121    fn create_run(&self, req: NewRun) -> StoreFuture<'_, RunCreation>;
122
123    /// Look up the run bound to an idempotency key.
124    ///
125    /// Returns `None` when the key is unknown, or when the run holding it is older
126    /// than [`IDEMPOTENCY_WINDOW`](crate::entities::IDEMPOTENCY_WINDOW).
127    fn find_run_by_idempotency_key(&self, key: &str) -> StoreFuture<'_, Option<Run>>;
128
129    /// Get a run by ID. Returns `None` if not found.
130    fn get_run(&self, id: Uuid) -> StoreFuture<'_, Option<Run>>;
131
132    /// List runs matching the given filter, with pagination.
133    ///
134    /// Results are ordered by `created_at` descending (newest first).
135    fn list_runs(&self, filter: RunFilter, page: u32, per_page: u32) -> StoreFuture<'_, Page<Run>>;
136
137    /// Update a run's status with FSM validation.
138    ///
139    /// # Errors
140    ///
141    /// Returns [`StoreError::InvalidTransition`] if the transition is not allowed.
142    /// Returns [`StoreError::RunNotFound`] if the run does not exist.
143    fn update_run_status(&self, id: Uuid, new_status: RunStatus) -> StoreFuture<'_, ()>;
144
145    /// Apply a partial update to a run.
146    ///
147    /// [`RunUpdate::lease`] is applied in the same transaction as the status
148    /// transition, after it: `status: Running` with
149    /// [`LeaseUpdate::Set`](crate::entities::LeaseUpdate::Set) leaves the run
150    /// `Running` and owned by that worker, so it is never `Running` without a
151    /// lease in between. [`LeaseUpdate::Release`](crate::entities::LeaseUpdate::Release)
152    /// drops the lease without touching the status. An explicit lease change
153    /// wins over the clearing that a transition out of `Running` does.
154    ///
155    /// # Errors
156    ///
157    /// Returns [`StoreError::RunNotFound`] if the run does not exist.
158    fn update_run(&self, id: Uuid, update: RunUpdate) -> StoreFuture<'_, ()>;
159
160    /// List the non-terminal descendants of a run, oldest first.
161    ///
162    /// A descendant is a sub-workflow run
163    /// ([`TriggerKind::Workflow`](crate::entities::TriggerKind::Workflow))
164    /// reached from `run_id` through
165    /// [`PARENT_RUN_ID_LABEL`](crate::entities::PARENT_RUN_ID_LABEL), at any
166    /// depth. Terminal runs are not returned, but their own descendants are:
167    /// a child left running under a finished parent is still found. Labels are
168    /// data, so a chain that loops back on itself is followed once and never
169    /// returns `run_id` itself.
170    ///
171    /// Returns an empty list for an unknown run or a run without children.
172    ///
173    /// # Errors
174    ///
175    /// Returns a database error when the backing store fails.
176    fn list_active_descendants(&self, run_id: Uuid) -> StoreFuture<'_, Vec<Run>>;
177
178    /// Atomically pick the oldest pending run and transition it to `Running`.
179    ///
180    /// In PostgreSQL, this uses `SELECT FOR UPDATE SKIP LOCKED` for safe
181    /// multi-worker concurrency. The in-memory implementation uses a write lock.
182    ///
183    /// When `lease` is `Some`, the worker lease is attached in the same
184    /// transaction as the status change, so a run is never `Running` without an
185    /// owner. Pass `None` for callers that execute runs in-process and cannot
186    /// refresh a lease (inline execution, API-side resume): those runs are never
187    /// recovered by [`reap_expired_leases`](Self::reap_expired_leases).
188    ///
189    /// Concurrency groups gate the pick: a run carrying
190    /// [`Run::concurrency_limits`] is skipped while, for any of its groups, the
191    /// number of root runs in state `Running` carrying that group is already at
192    /// or above the run's own limit for it. Sleeping, awaiting approval,
193    /// retrying and pending runs do not count, and sub-workflow runs
194    /// ([`TriggerKind::Workflow`](crate::entities::TriggerKind::Workflow)) are
195    /// never counted. A held-back run does not block the queue: the oldest
196    /// eligible run wins. The check is atomic across concurrent callers, so a
197    /// group never exceeds its limit.
198    ///
199    /// Returns `None` if no pending runs are available.
200    ///
201    /// Equivalent to [`pick_next_pending_for`](Self::pick_next_pending_for)
202    /// with no worker capabilities: every run is eligible.
203    fn pick_next_pending(&self, lease: Option<LeaseRequest>) -> StoreFuture<'_, Option<Run>> {
204        self.pick_next_pending_for(lease, None)
205    }
206
207    /// Atomically pick the oldest pending run the worker can take and
208    /// transition it to `Running`.
209    ///
210    /// Same contract as [`pick_next_pending`](Self::pick_next_pending), with
211    /// worker routing on top: when `capabilities` is `Some`, a run is only
212    /// eligible when [`WorkerCapabilities::can_take`] accepts its workflow name
213    /// and its [`Run::worker_tags`]. An ineligible run is skipped and never
214    /// blocks younger runs. `None` keeps the legacy behavior of a worker that
215    /// sends no capabilities: every run is eligible.
216    ///
217    /// Returns `None` if no eligible pending run is available.
218    fn pick_next_pending_for(
219        &self,
220        lease: Option<LeaseRequest>,
221        capabilities: Option<WorkerCapabilities>,
222    ) -> StoreFuture<'_, Option<Run>>;
223
224    /// Extend the worker lease on a run and return the new expiry.
225    ///
226    /// # Errors
227    ///
228    /// Returns [`StoreError::RunNotFound`] if the run does not exist.
229    /// Returns [`StoreError::LeaseLost`] if the run is no longer `Running` or if
230    /// the lease belongs to another worker — the caller must stop executing it.
231    fn renew_lease(&self, id: Uuid, lease: LeaseRequest) -> StoreFuture<'_, DateTime<Utc>>;
232
233    /// Count, for each concurrency group, the due runs it currently holds back.
234    ///
235    /// A run is counted when it is pending or retrying, due (no
236    /// `scheduled_at` in the future) and not pickable because the group is
237    /// saturated for its own limit (see [`pick_next_pending`](Self::pick_next_pending)).
238    /// A run held back by two groups counts in both. Groups holding back no
239    /// run are omitted. Results are sorted by group name.
240    ///
241    /// # Errors
242    ///
243    /// Returns a database error when the backing store fails.
244    ///
245    /// # Examples
246    ///
247    /// ```no_run
248    /// use ironflow_store::store::RunStore;
249    ///
250    /// # async fn example(store: &dyn RunStore) -> Result<(), ironflow_store::error::StoreError> {
251    /// for backlog in store.count_blocked_runs_by_group().await? {
252    ///     println!("{}: {} runs held back", backlog.group, backlog.blocked_runs);
253    /// }
254    /// # Ok(())
255    /// # }
256    /// ```
257    fn count_blocked_runs_by_group(&self) -> StoreFuture<'_, Vec<ConcurrencyGroupBacklog>>;
258
259    /// Recover runs whose worker lease expired, at most `limit` per call.
260    ///
261    /// Each recovered run has [`Run::lease_recoveries`] incremented and its lease
262    /// cleared, then goes back to `Pending` — or to `Failed` with
263    /// [`LEASE_EXPIRED_ERROR`] once more than `max_retries` recoveries happened.
264    /// Runs without a lease are never touched.
265    /// A root run resumed through its sub-workflow child carries the lease the
266    /// child held (see [`RunUpdate::lease`]), so it is recovered like any run.
267    ///
268    /// [`Run::retry_count`], and so the attempt number of the steps created
269    /// afterwards, is left unchanged: a requeued run resumes in the same attempt
270    /// and replays the steps it already finished.
271    ///
272    /// The whole batch is atomic per run (`FOR UPDATE SKIP LOCKED` in
273    /// PostgreSQL), so concurrent reapers never recover the same run twice.
274    ///
275    /// Callers are responsible for the side effects that follow a recovery:
276    /// failing orphaned steps and publishing status-change events.
277    fn reap_expired_leases(&self, limit: u32) -> StoreFuture<'_, Vec<ReapedRun>>;
278
279    /// Atomically claim approval steps whose SLA deadline has passed.
280    ///
281    /// Returns the claimed steps with their *pre-claim* `approval_deadline_at`
282    /// still populated, so the caller can report which deadline fired. The
283    /// timer is cleared in the same transaction, so a deadline fires at most
284    /// once even with several API instances running the escalator (the
285    /// PostgreSQL implementation uses `FOR UPDATE SKIP LOCKED`).
286    ///
287    /// Only steps still in [`StepStatus::AwaitingApproval`](crate::entities::StepStatus::AwaitingApproval)
288    /// are returned.
289    ///
290    /// Delivery is at most once: a caller that crashes between the claim and
291    /// the escalation leaves the gate open with no timer, the same trade-off
292    /// [`reap_expired_leases`](Self::reap_expired_leases) accepts.
293    fn claim_due_approval_deadlines(&self, limit: u32) -> StoreFuture<'_, Vec<Step>>;
294
295    /// Atomically wake the `Sleeping` runs whose `scheduled_at` has passed, at
296    /// most `limit` per call.
297    ///
298    /// Each claimed run goes `Sleeping -> Pending` (`delay_elapsed`) and has
299    /// its `scheduled_at` cleared in the same transaction, so a run is woken
300    /// exactly once even with several API instances running the waker (the
301    /// PostgreSQL implementation uses `FOR UPDATE SKIP LOCKED`). Runs are
302    /// claimed oldest `scheduled_at` first.
303    ///
304    /// Returns the runs as they are after the transition. Callers decide how
305    /// the requeued runs resume: a worker picks them up, or the API resumes
306    /// them in-process when it has no worker.
307    ///
308    /// # Errors
309    ///
310    /// Returns [`StoreError`] on storage failure.
311    fn claim_due_sleeping_runs(&self, limit: u32) -> StoreFuture<'_, Vec<Run>>;
312
313    /// Create a new step for a run.
314    ///
315    /// # Errors
316    ///
317    /// Returns [`StoreError::RunNotFound`] if the parent run does not exist.
318    fn create_step(&self, step: NewStep) -> StoreFuture<'_, Step>;
319
320    /// Apply a partial update to a step after execution.
321    ///
322    /// # Errors
323    ///
324    /// Returns [`StoreError::StepNotFound`] if the step does not exist.
325    fn update_step(&self, id: Uuid, update: StepUpdate) -> StoreFuture<'_, ()>;
326
327    /// Get a single step by ID. Returns `None` if not found.
328    fn get_step(&self, id: Uuid) -> StoreFuture<'_, Option<Step>>;
329
330    /// List all steps for a run, ordered by position ascending.
331    fn list_steps(&self, run_id: Uuid) -> StoreFuture<'_, Vec<Step>>;
332
333    /// Record a vote on an approval gate and return the updated step.
334    ///
335    /// The vote is appended atomically to [`Step::approvals`] unless the same
336    /// [`StepApproval::user_id`] already voted, in which case the step is
337    /// returned unchanged. Recording a vote never resolves the gate: the
338    /// caller compares the vote count against the step's
339    /// [`approval_requirement`](Step::approval_requirement).
340    ///
341    /// # Errors
342    ///
343    /// Returns [`StoreError::StepNotFound`] if the step does not exist.
344    fn record_step_approval(&self, step_id: Uuid, approval: StepApproval) -> StoreFuture<'_, Step>;
345
346    /// Get aggregated statistics across runs matching the filter.
347    ///
348    /// Returns counts of runs by terminal state, counts of active runs
349    /// (`Pending`, `Running`, `Retrying`, `AwaitingApproval` or `Sleeping`),
350    /// the number of runs awaiting approval, and totals for cost and duration.
351    /// Computed efficiently by the store implementation (single SQL query in
352    /// PostgreSQL).
353    ///
354    /// Pass [`RunFilter::default()`] to get stats across all runs.
355    fn get_stats(&self, filter: RunFilter) -> StoreFuture<'_, RunStats>;
356
357    /// Get time-bucketed historical statistics for trend charts.
358    ///
359    /// Aggregates runs created during the filter's period into time buckets
360    /// based on its granularity, counting every run status and computing
361    /// duration percentiles. Applies the same run filters as
362    /// [`get_stats`](Self::get_stats) (workflow substring, status, labels,
363    /// steps, author). Bucket boundaries are UTC and weeks start on Monday
364    /// (see [`HistoryGranularity::bucket_start`](crate::entities::HistoryGranularity::bucket_start)).
365    /// Returns buckets ordered by time ascending; empty buckets are omitted.
366    ///
367    /// # Errors
368    ///
369    /// Returns [`StoreError::Database`] on underlying store failures.
370    ///
371    /// # Examples
372    ///
373    /// ```no_run
374    /// use ironflow_store::entities::{StatsHistoryFilter, HistoryPeriod, HistoryGranularity};
375    /// use ironflow_store::store::RunStore;
376    ///
377    /// # async fn example(store: &dyn RunStore) -> Result<(), ironflow_store::error::StoreError> {
378    /// let filter = StatsHistoryFilter {
379    ///     period: HistoryPeriod::SevenDays,
380    ///     granularity: HistoryGranularity::OneDay,
381    ///     ..StatsHistoryFilter::default()
382    /// };
383    /// let buckets = store.get_stats_history(filter).await?;
384    /// for b in &buckets {
385    ///     println!("{}: {} completed, {} failed", b.time, b.completed, b.failed);
386    /// }
387    /// # Ok(())
388    /// # }
389    /// ```
390    fn get_stats_history(
391        &self,
392        filter: StatsHistoryFilter,
393    ) -> StoreFuture<'_, Vec<StatsHistoryBucket>>;
394
395    /// Create step dependency edges in batch.
396    ///
397    /// Each entry records that `step_id` depends on `depends_on`.
398    /// Duplicate edges are silently ignored.
399    ///
400    /// # Errors
401    ///
402    /// Returns [`StoreError`] if a referenced step does not exist.
403    fn create_step_dependencies(&self, deps: Vec<NewStepDependency>) -> StoreFuture<'_, ()>;
404
405    /// List all step dependencies for a given run.
406    ///
407    /// Returns every edge where either `step_id` or `depends_on` belongs
408    /// to the run. Ordered by `created_at` ascending.
409    fn list_step_dependencies(&self, run_id: Uuid) -> StoreFuture<'_, Vec<StepDependency>>;
410
411    /// List runs eligible for purging according to the given policy.
412    ///
413    /// A run is eligible when it is in a terminal state ([`RunStatus::is_terminal`])
414    /// **and** either older than `policy.max_age_days` or exceeding
415    /// `policy.max_runs_per_workflow` for its workflow (oldest first).
416    ///
417    /// Runs in non-terminal states (`Pending`, `Running`, `Retrying`,
418    /// `AwaitingApproval`) are never returned.
419    ///
420    /// # Examples
421    ///
422    /// ```no_run
423    /// use ironflow_store::entities::PurgePolicy;
424    /// use ironflow_store::store::RunStore;
425    ///
426    /// # async fn example(store: &dyn RunStore) -> Result<(), ironflow_store::error::StoreError> {
427    /// let policy = PurgePolicy { max_age_days: 90, max_runs_per_workflow: 1000, dry_run: false };
428    /// let purgeable = store.list_purgeable_runs(&policy, 100).await?;
429    /// for p in &purgeable {
430    ///     println!("purge {} ({}): {}", p.run_id, p.workflow_name, p.reason);
431    /// }
432    /// # Ok(())
433    /// # }
434    /// ```
435    fn list_purgeable_runs(
436        &self,
437        policy: &PurgePolicy,
438        batch_size: u32,
439    ) -> StoreFuture<'_, Vec<PurgeableRun>>;
440
441    /// Delete a run and all its associated data (steps, step dependencies).
442    ///
443    /// Returns the `storage_key` of every artifact that belonged to the run,
444    /// so the caller can delete the corresponding blobs from the blob store.
445    ///
446    /// # Errors
447    ///
448    /// Returns [`StoreError::RunNotFound`] if the run does not exist.
449    ///
450    /// # Examples
451    ///
452    /// ```no_run
453    /// use ironflow_store::store::RunStore;
454    /// use uuid::Uuid;
455    ///
456    /// # async fn example(store: &dyn RunStore, run_id: Uuid) -> Result<(), ironflow_store::error::StoreError> {
457    /// let storage_keys = store.delete_run(run_id).await?;
458    /// // Caller deletes blobs from the blob store using these keys.
459    /// # Ok(())
460    /// # }
461    /// ```
462    fn delete_run(&self, id: Uuid) -> StoreFuture<'_, Vec<String>>;
463
464    /// Apply a partial update to a run and return the updated run.
465    ///
466    /// Combines [`update_run`](Self::update_run) and [`get_run`](Self::get_run) in
467    /// a single operation to avoid an extra round-trip. Store implementations
468    /// may override this for efficiency (e.g. reading within the same transaction).
469    ///
470    /// The default implementation calls `update_run` followed by `get_run`.
471    ///
472    /// # Errors
473    ///
474    /// Returns [`StoreError::RunNotFound`] if the run does not exist.
475    /// Returns [`StoreError::InvalidTransition`] if the status transition is not allowed.
476    fn update_run_returning(&self, id: Uuid, update: RunUpdate) -> StoreFuture<'_, Run> {
477        Box::pin(async move {
478            self.update_run(id, update).await?;
479            self.get_run(id).await?.ok_or(StoreError::RunNotFound(id))
480        })
481    }
482}
483
484/// Unified storage abstraction combining all store capabilities.
485///
486/// Implementors provide runs, steps, users, API keys, and secrets
487/// through a single type. Pick one backend (in-memory or PostgreSQL)
488/// and it handles everything.
489///
490/// Both [`InMemoryStore`](crate::memory::InMemoryStore) and
491/// [`PostgresStore`](crate::postgres::PostgresStore) implement this trait.
492///
493/// # Examples
494///
495/// ```no_run
496/// use std::collections::HashMap;
497/// use std::sync::Arc;
498/// use ironflow_store::prelude::*;
499///
500/// # async fn example() -> Result<(), ironflow_store::error::StoreError> {
501/// let store: Arc<dyn Store> = Arc::new(InMemoryStore::new());
502///
503/// // All capabilities through one reference
504/// let _run = store.create_run(NewRun {
505///     workflow_name: "deploy".to_string(),
506///     trigger: TriggerKind::Manual,
507///     payload: serde_json::json!({}),
508///     max_retries: 3,
509///     handler_version: None,
510///     labels: HashMap::new(),
511///     scheduled_at: None,
512///     created_by: None,
513///     idempotency_key: None,
514///     concurrency_key: None,
515///     concurrency_limits: Vec::new(),
516///     max_cost_usd: None,
517///     worker_tags: Vec::new(),
518/// }).await?.into_run();
519/// let _users = store.count_users().await?;
520/// # Ok(())
521/// # }
522/// ```
523pub trait Store:
524    RunStore
525    + UserStore
526    + ApiKeyStore
527    + SecretStore
528    + AuditLogStore
529    + ArtifactStore
530    + LogStore
531    + ScheduleStore
532    + ApprovalDelegationStore
533    + ProviderAccountStore
534    + SignalStore
535{
536}
537
538impl<
539    T: RunStore
540        + UserStore
541        + ApiKeyStore
542        + SecretStore
543        + AuditLogStore
544        + ArtifactStore
545        + LogStore
546        + ScheduleStore
547        + ApprovalDelegationStore
548        + ProviderAccountStore
549        + SignalStore,
550> Store for T
551{
552}