ironflow_store/store.rs
1//! The [`RunStore`] trait — async storage abstraction for runs and steps.
2//!
3//! Implement this trait to plug in any backing store. Built-in implementations:
4//!
5//! - [`InMemoryStore`](crate::memory::InMemoryStore) — development and testing.
6//! - `PostgresStore` — production (behind the `store-postgres` feature).
7
8use std::future::Future;
9use std::pin::Pin;
10
11use chrono::{DateTime, Utc};
12use uuid::Uuid;
13
14use crate::api_key_store::ApiKeyStore;
15use crate::approval_delegation_store::ApprovalDelegationStore;
16use crate::artifact_store::ArtifactStore;
17use crate::audit_log_store::AuditLogStore;
18use crate::entities::{
19 ConcurrencyGroupBacklog, LeaseRequest, NewRun, NewStep, NewStepDependency, Page, PurgePolicy,
20 PurgeableRun, ReapedRun, Run, RunCreation, RunFilter, RunStats, RunStatus, RunUpdate,
21 StatsHistoryBucket, StatsHistoryFilter, Step, StepApproval, StepDependency, StepUpdate,
22};
23use crate::error::StoreError;
24use crate::log_store::LogStore;
25use crate::provider_account_store::ProviderAccountStore;
26use crate::schedule_store::ScheduleStore;
27use crate::secret_store::SecretStore;
28use crate::signal_store::SignalStore;
29use crate::user_store::UserStore;
30
31/// Boxed future for [`RunStore`] methods — ensures object safety for `dyn RunStore`.
32pub type StoreFuture<'a, T> = Pin<Box<dyn Future<Output = Result<T, StoreError>> + Send + 'a>>;
33
34/// Error recorded on a run that exhausted its retries through lease expiries.
35///
36/// Set by [`RunStore::reap_expired_leases`] when a run has been recovered more
37/// than `max_retries` times.
38pub const LEASE_EXPIRED_ERROR: &str = "worker lease expired";
39
40/// Error recorded on a step that was running when its worker lost the lease.
41///
42/// Set by the reaper on the `Running` steps of a run that
43/// [`RunStore::reap_expired_leases`] requeued. The engine executes such a step
44/// again at the same position when the run is picked up, keeping the
45/// interrupted record in the step history.
46///
47/// # Examples
48///
49/// ```
50/// use ironflow_store::store::{LEASE_EXPIRED_ERROR, STEP_INTERRUPTED_ERROR};
51///
52/// assert_eq!(STEP_INTERRUPTED_ERROR, "interrupted: worker lease lost");
53/// assert_ne!(STEP_INTERRUPTED_ERROR, LEASE_EXPIRED_ERROR);
54/// ```
55pub const STEP_INTERRUPTED_ERROR: &str = "interrupted: worker lease lost";
56
57/// Async storage abstraction for workflow runs and steps.
58///
59/// All methods return a [`StoreFuture`] (boxed future) to maintain object safety,
60/// allowing the store to be used as `Arc<dyn RunStore>`.
61///
62/// # Examples
63///
64/// ```no_run
65/// use std::collections::HashMap;
66/// use ironflow_store::prelude::*;
67/// use serde_json::json;
68/// use uuid::Uuid;
69///
70/// # async fn example() -> Result<(), ironflow_store::error::StoreError> {
71/// let store = InMemoryStore::new();
72///
73/// let run = store.create_run(NewRun {
74/// workflow_name: "deploy".to_string(),
75/// trigger: TriggerKind::Manual,
76/// payload: json!({}),
77/// max_retries: 3,
78/// handler_version: None,
79/// labels: HashMap::new(),
80/// scheduled_at: None,
81/// created_by: None,
82/// idempotency_key: None,
83/// concurrency_key: None,
84/// concurrency_limits: Vec::new(),
85/// max_cost_usd: None,
86/// }).await?.into_run();
87///
88/// let fetched = store.get_run(run.id).await?;
89/// assert!(fetched.is_some());
90/// # Ok(())
91/// # }
92/// ```
93pub trait RunStore: Send + Sync {
94 /// Create a new run in `Pending` status.
95 ///
96 /// When [`NewRun::idempotency_key`] is set and already bound to a run created
97 /// within [`IDEMPOTENCY_WINDOW`](crate::entities::IDEMPOTENCY_WINDOW), nothing is
98 /// inserted and that run is returned as [`RunCreation::Existing`]. A key bound to
99 /// an older run is released and reused for the new one.
100 ///
101 /// Concurrent calls sharing the same key resolve to a single run: exactly one
102 /// receives [`RunCreation::Created`], the others [`RunCreation::Existing`].
103 ///
104 /// When [`NewRun::concurrency_key`] is set, the idempotency lookup runs first,
105 /// then the key is checked: concurrent calls sharing it are serialized, and
106 /// at most one non-terminal run holds it at a time.
107 ///
108 /// [`NewRun::concurrency_limits`] is validated before anything is written.
109 ///
110 /// # Errors
111 ///
112 /// Returns [`StoreError::ConcurrencyConflict`](crate::error::StoreError::ConcurrencyConflict)
113 /// when a run that is not Completed, Failed, Warning or Cancelled already
114 /// holds [`NewRun::concurrency_key`],
115 /// [`StoreError::InvalidConcurrencyLimit`](crate::error::StoreError::InvalidConcurrencyLimit)
116 /// when [`NewRun::concurrency_limits`] holds an empty or too long group, a
117 /// zero limit or a duplicated group, and a database error when the backing
118 /// store fails.
119 fn create_run(&self, req: NewRun) -> StoreFuture<'_, RunCreation>;
120
121 /// Look up the run bound to an idempotency key.
122 ///
123 /// Returns `None` when the key is unknown, or when the run holding it is older
124 /// than [`IDEMPOTENCY_WINDOW`](crate::entities::IDEMPOTENCY_WINDOW).
125 fn find_run_by_idempotency_key(&self, key: &str) -> StoreFuture<'_, Option<Run>>;
126
127 /// Get a run by ID. Returns `None` if not found.
128 fn get_run(&self, id: Uuid) -> StoreFuture<'_, Option<Run>>;
129
130 /// List runs matching the given filter, with pagination.
131 ///
132 /// Results are ordered by `created_at` descending (newest first).
133 fn list_runs(&self, filter: RunFilter, page: u32, per_page: u32) -> StoreFuture<'_, Page<Run>>;
134
135 /// Update a run's status with FSM validation.
136 ///
137 /// # Errors
138 ///
139 /// Returns [`StoreError::InvalidTransition`] if the transition is not allowed.
140 /// Returns [`StoreError::RunNotFound`] if the run does not exist.
141 fn update_run_status(&self, id: Uuid, new_status: RunStatus) -> StoreFuture<'_, ()>;
142
143 /// Apply a partial update to a run.
144 ///
145 /// [`RunUpdate::lease`] is applied in the same transaction as the status
146 /// transition, after it: `status: Running` with
147 /// [`LeaseUpdate::Set`](crate::entities::LeaseUpdate::Set) leaves the run
148 /// `Running` and owned by that worker, so it is never `Running` without a
149 /// lease in between. [`LeaseUpdate::Release`](crate::entities::LeaseUpdate::Release)
150 /// drops the lease without touching the status. An explicit lease change
151 /// wins over the clearing that a transition out of `Running` does.
152 ///
153 /// # Errors
154 ///
155 /// Returns [`StoreError::RunNotFound`] if the run does not exist.
156 fn update_run(&self, id: Uuid, update: RunUpdate) -> StoreFuture<'_, ()>;
157
158 /// List the non-terminal descendants of a run, oldest first.
159 ///
160 /// A descendant is a sub-workflow run
161 /// ([`TriggerKind::Workflow`](crate::entities::TriggerKind::Workflow))
162 /// reached from `run_id` through
163 /// [`PARENT_RUN_ID_LABEL`](crate::entities::PARENT_RUN_ID_LABEL), at any
164 /// depth. Terminal runs are not returned, but their own descendants are:
165 /// a child left running under a finished parent is still found. Labels are
166 /// data, so a chain that loops back on itself is followed once and never
167 /// returns `run_id` itself.
168 ///
169 /// Returns an empty list for an unknown run or a run without children.
170 ///
171 /// # Errors
172 ///
173 /// Returns a database error when the backing store fails.
174 fn list_active_descendants(&self, run_id: Uuid) -> StoreFuture<'_, Vec<Run>>;
175
176 /// Atomically pick the oldest pending run and transition it to `Running`.
177 ///
178 /// In PostgreSQL, this uses `SELECT FOR UPDATE SKIP LOCKED` for safe
179 /// multi-worker concurrency. The in-memory implementation uses a write lock.
180 ///
181 /// When `lease` is `Some`, the worker lease is attached in the same
182 /// transaction as the status change, so a run is never `Running` without an
183 /// owner. Pass `None` for callers that execute runs in-process and cannot
184 /// refresh a lease (inline execution, API-side resume): those runs are never
185 /// recovered by [`reap_expired_leases`](Self::reap_expired_leases).
186 ///
187 /// Concurrency groups gate the pick: a run carrying
188 /// [`Run::concurrency_limits`] is skipped while, for any of its groups, the
189 /// number of root runs in state `Running` carrying that group is already at
190 /// or above the run's own limit for it. Sleeping, awaiting approval,
191 /// retrying and pending runs do not count, and sub-workflow runs
192 /// ([`TriggerKind::Workflow`](crate::entities::TriggerKind::Workflow)) are
193 /// never counted. A held-back run does not block the queue: the oldest
194 /// eligible run wins. The check is atomic across concurrent callers, so a
195 /// group never exceeds its limit.
196 ///
197 /// Returns `None` if no pending runs are available.
198 fn pick_next_pending(&self, lease: Option<LeaseRequest>) -> StoreFuture<'_, Option<Run>>;
199
200 /// Extend the worker lease on a run and return the new expiry.
201 ///
202 /// # Errors
203 ///
204 /// Returns [`StoreError::RunNotFound`] if the run does not exist.
205 /// Returns [`StoreError::LeaseLost`] if the run is no longer `Running` or if
206 /// the lease belongs to another worker — the caller must stop executing it.
207 fn renew_lease(&self, id: Uuid, lease: LeaseRequest) -> StoreFuture<'_, DateTime<Utc>>;
208
209 /// Count, for each concurrency group, the due runs it currently holds back.
210 ///
211 /// A run is counted when it is pending or retrying, due (no
212 /// `scheduled_at` in the future) and not pickable because the group is
213 /// saturated for its own limit (see [`pick_next_pending`](Self::pick_next_pending)).
214 /// A run held back by two groups counts in both. Groups holding back no
215 /// run are omitted. Results are sorted by group name.
216 ///
217 /// # Errors
218 ///
219 /// Returns a database error when the backing store fails.
220 ///
221 /// # Examples
222 ///
223 /// ```no_run
224 /// use ironflow_store::store::RunStore;
225 ///
226 /// # async fn example(store: &dyn RunStore) -> Result<(), ironflow_store::error::StoreError> {
227 /// for backlog in store.count_blocked_runs_by_group().await? {
228 /// println!("{}: {} runs held back", backlog.group, backlog.blocked_runs);
229 /// }
230 /// # Ok(())
231 /// # }
232 /// ```
233 fn count_blocked_runs_by_group(&self) -> StoreFuture<'_, Vec<ConcurrencyGroupBacklog>>;
234
235 /// Recover runs whose worker lease expired, at most `limit` per call.
236 ///
237 /// Each recovered run has [`Run::lease_recoveries`] incremented and its lease
238 /// cleared, then goes back to `Pending` — or to `Failed` with
239 /// [`LEASE_EXPIRED_ERROR`] once more than `max_retries` recoveries happened.
240 /// Runs without a lease are never touched.
241 /// A root run resumed through its sub-workflow child carries the lease the
242 /// child held (see [`RunUpdate::lease`]), so it is recovered like any run.
243 ///
244 /// [`Run::retry_count`], and so the attempt number of the steps created
245 /// afterwards, is left unchanged: a requeued run resumes in the same attempt
246 /// and replays the steps it already finished.
247 ///
248 /// The whole batch is atomic per run (`FOR UPDATE SKIP LOCKED` in
249 /// PostgreSQL), so concurrent reapers never recover the same run twice.
250 ///
251 /// Callers are responsible for the side effects that follow a recovery:
252 /// failing orphaned steps and publishing status-change events.
253 fn reap_expired_leases(&self, limit: u32) -> StoreFuture<'_, Vec<ReapedRun>>;
254
255 /// Atomically claim approval steps whose SLA deadline has passed.
256 ///
257 /// Returns the claimed steps with their *pre-claim* `approval_deadline_at`
258 /// still populated, so the caller can report which deadline fired. The
259 /// timer is cleared in the same transaction, so a deadline fires at most
260 /// once even with several API instances running the escalator (the
261 /// PostgreSQL implementation uses `FOR UPDATE SKIP LOCKED`).
262 ///
263 /// Only steps still in [`StepStatus::AwaitingApproval`](crate::entities::StepStatus::AwaitingApproval)
264 /// are returned.
265 ///
266 /// Delivery is at most once: a caller that crashes between the claim and
267 /// the escalation leaves the gate open with no timer, the same trade-off
268 /// [`reap_expired_leases`](Self::reap_expired_leases) accepts.
269 fn claim_due_approval_deadlines(&self, limit: u32) -> StoreFuture<'_, Vec<Step>>;
270
271 /// Atomically wake the `Sleeping` runs whose `scheduled_at` has passed, at
272 /// most `limit` per call.
273 ///
274 /// Each claimed run goes `Sleeping -> Pending` (`delay_elapsed`) and has
275 /// its `scheduled_at` cleared in the same transaction, so a run is woken
276 /// exactly once even with several API instances running the waker (the
277 /// PostgreSQL implementation uses `FOR UPDATE SKIP LOCKED`). Runs are
278 /// claimed oldest `scheduled_at` first.
279 ///
280 /// Returns the runs as they are after the transition. Callers decide how
281 /// the requeued runs resume: a worker picks them up, or the API resumes
282 /// them in-process when it has no worker.
283 ///
284 /// # Errors
285 ///
286 /// Returns [`StoreError`] on storage failure.
287 fn claim_due_sleeping_runs(&self, limit: u32) -> StoreFuture<'_, Vec<Run>>;
288
289 /// Create a new step for a run.
290 ///
291 /// # Errors
292 ///
293 /// Returns [`StoreError::RunNotFound`] if the parent run does not exist.
294 fn create_step(&self, step: NewStep) -> StoreFuture<'_, Step>;
295
296 /// Apply a partial update to a step after execution.
297 ///
298 /// # Errors
299 ///
300 /// Returns [`StoreError::StepNotFound`] if the step does not exist.
301 fn update_step(&self, id: Uuid, update: StepUpdate) -> StoreFuture<'_, ()>;
302
303 /// Get a single step by ID. Returns `None` if not found.
304 fn get_step(&self, id: Uuid) -> StoreFuture<'_, Option<Step>>;
305
306 /// List all steps for a run, ordered by position ascending.
307 fn list_steps(&self, run_id: Uuid) -> StoreFuture<'_, Vec<Step>>;
308
309 /// Record a vote on an approval gate and return the updated step.
310 ///
311 /// The vote is appended atomically to [`Step::approvals`] unless the same
312 /// [`StepApproval::user_id`] already voted, in which case the step is
313 /// returned unchanged. Recording a vote never resolves the gate: the
314 /// caller compares the vote count against the step's
315 /// [`approval_requirement`](Step::approval_requirement).
316 ///
317 /// # Errors
318 ///
319 /// Returns [`StoreError::StepNotFound`] if the step does not exist.
320 fn record_step_approval(&self, step_id: Uuid, approval: StepApproval) -> StoreFuture<'_, Step>;
321
322 /// Get aggregated statistics across runs matching the filter.
323 ///
324 /// Returns counts of runs by terminal state, counts of active runs
325 /// (`Pending`, `Running`, `Retrying`, `AwaitingApproval` or `Sleeping`),
326 /// the number of runs awaiting approval, and totals for cost and duration.
327 /// Computed efficiently by the store implementation (single SQL query in
328 /// PostgreSQL).
329 ///
330 /// Pass [`RunFilter::default()`] to get stats across all runs.
331 fn get_stats(&self, filter: RunFilter) -> StoreFuture<'_, RunStats>;
332
333 /// Get time-bucketed historical statistics for trend charts.
334 ///
335 /// Aggregates runs created during the filter's period into time buckets
336 /// based on its granularity, counting every run status and computing
337 /// duration percentiles. Applies the same run filters as
338 /// [`get_stats`](Self::get_stats) (workflow substring, status, labels,
339 /// steps, author). Bucket boundaries are UTC and weeks start on Monday
340 /// (see [`HistoryGranularity::bucket_start`](crate::entities::HistoryGranularity::bucket_start)).
341 /// Returns buckets ordered by time ascending; empty buckets are omitted.
342 ///
343 /// # Errors
344 ///
345 /// Returns [`StoreError::Database`] on underlying store failures.
346 ///
347 /// # Examples
348 ///
349 /// ```no_run
350 /// use ironflow_store::entities::{StatsHistoryFilter, HistoryPeriod, HistoryGranularity};
351 /// use ironflow_store::store::RunStore;
352 ///
353 /// # async fn example(store: &dyn RunStore) -> Result<(), ironflow_store::error::StoreError> {
354 /// let filter = StatsHistoryFilter {
355 /// period: HistoryPeriod::SevenDays,
356 /// granularity: HistoryGranularity::OneDay,
357 /// ..StatsHistoryFilter::default()
358 /// };
359 /// let buckets = store.get_stats_history(filter).await?;
360 /// for b in &buckets {
361 /// println!("{}: {} completed, {} failed", b.time, b.completed, b.failed);
362 /// }
363 /// # Ok(())
364 /// # }
365 /// ```
366 fn get_stats_history(
367 &self,
368 filter: StatsHistoryFilter,
369 ) -> StoreFuture<'_, Vec<StatsHistoryBucket>>;
370
371 /// Create step dependency edges in batch.
372 ///
373 /// Each entry records that `step_id` depends on `depends_on`.
374 /// Duplicate edges are silently ignored.
375 ///
376 /// # Errors
377 ///
378 /// Returns [`StoreError`] if a referenced step does not exist.
379 fn create_step_dependencies(&self, deps: Vec<NewStepDependency>) -> StoreFuture<'_, ()>;
380
381 /// List all step dependencies for a given run.
382 ///
383 /// Returns every edge where either `step_id` or `depends_on` belongs
384 /// to the run. Ordered by `created_at` ascending.
385 fn list_step_dependencies(&self, run_id: Uuid) -> StoreFuture<'_, Vec<StepDependency>>;
386
387 /// List runs eligible for purging according to the given policy.
388 ///
389 /// A run is eligible when it is in a terminal state ([`RunStatus::is_terminal`])
390 /// **and** either older than `policy.max_age_days` or exceeding
391 /// `policy.max_runs_per_workflow` for its workflow (oldest first).
392 ///
393 /// Runs in non-terminal states (`Pending`, `Running`, `Retrying`,
394 /// `AwaitingApproval`) are never returned.
395 ///
396 /// # Examples
397 ///
398 /// ```no_run
399 /// use ironflow_store::entities::PurgePolicy;
400 /// use ironflow_store::store::RunStore;
401 ///
402 /// # async fn example(store: &dyn RunStore) -> Result<(), ironflow_store::error::StoreError> {
403 /// let policy = PurgePolicy { max_age_days: 90, max_runs_per_workflow: 1000, dry_run: false };
404 /// let purgeable = store.list_purgeable_runs(&policy, 100).await?;
405 /// for p in &purgeable {
406 /// println!("purge {} ({}): {}", p.run_id, p.workflow_name, p.reason);
407 /// }
408 /// # Ok(())
409 /// # }
410 /// ```
411 fn list_purgeable_runs(
412 &self,
413 policy: &PurgePolicy,
414 batch_size: u32,
415 ) -> StoreFuture<'_, Vec<PurgeableRun>>;
416
417 /// Delete a run and all its associated data (steps, step dependencies).
418 ///
419 /// Returns the `storage_key` of every artifact that belonged to the run,
420 /// so the caller can delete the corresponding blobs from the blob store.
421 ///
422 /// # Errors
423 ///
424 /// Returns [`StoreError::RunNotFound`] if the run does not exist.
425 ///
426 /// # Examples
427 ///
428 /// ```no_run
429 /// use ironflow_store::store::RunStore;
430 /// use uuid::Uuid;
431 ///
432 /// # async fn example(store: &dyn RunStore, run_id: Uuid) -> Result<(), ironflow_store::error::StoreError> {
433 /// let storage_keys = store.delete_run(run_id).await?;
434 /// // Caller deletes blobs from the blob store using these keys.
435 /// # Ok(())
436 /// # }
437 /// ```
438 fn delete_run(&self, id: Uuid) -> StoreFuture<'_, Vec<String>>;
439
440 /// Apply a partial update to a run and return the updated run.
441 ///
442 /// Combines [`update_run`](Self::update_run) and [`get_run`](Self::get_run) in
443 /// a single operation to avoid an extra round-trip. Store implementations
444 /// may override this for efficiency (e.g. reading within the same transaction).
445 ///
446 /// The default implementation calls `update_run` followed by `get_run`.
447 ///
448 /// # Errors
449 ///
450 /// Returns [`StoreError::RunNotFound`] if the run does not exist.
451 /// Returns [`StoreError::InvalidTransition`] if the status transition is not allowed.
452 fn update_run_returning(&self, id: Uuid, update: RunUpdate) -> StoreFuture<'_, Run> {
453 Box::pin(async move {
454 self.update_run(id, update).await?;
455 self.get_run(id).await?.ok_or(StoreError::RunNotFound(id))
456 })
457 }
458}
459
460/// Unified storage abstraction combining all store capabilities.
461///
462/// Implementors provide runs, steps, users, API keys, and secrets
463/// through a single type. Pick one backend (in-memory or PostgreSQL)
464/// and it handles everything.
465///
466/// Both [`InMemoryStore`](crate::memory::InMemoryStore) and
467/// [`PostgresStore`](crate::postgres::PostgresStore) implement this trait.
468///
469/// # Examples
470///
471/// ```no_run
472/// use std::collections::HashMap;
473/// use std::sync::Arc;
474/// use ironflow_store::prelude::*;
475///
476/// # async fn example() -> Result<(), ironflow_store::error::StoreError> {
477/// let store: Arc<dyn Store> = Arc::new(InMemoryStore::new());
478///
479/// // All capabilities through one reference
480/// let _run = store.create_run(NewRun {
481/// workflow_name: "deploy".to_string(),
482/// trigger: TriggerKind::Manual,
483/// payload: serde_json::json!({}),
484/// max_retries: 3,
485/// handler_version: None,
486/// labels: HashMap::new(),
487/// scheduled_at: None,
488/// created_by: None,
489/// idempotency_key: None,
490/// concurrency_key: None,
491/// concurrency_limits: Vec::new(),
492/// max_cost_usd: None,
493/// }).await?.into_run();
494/// let _users = store.count_users().await?;
495/// # Ok(())
496/// # }
497/// ```
498pub trait Store:
499 RunStore
500 + UserStore
501 + ApiKeyStore
502 + SecretStore
503 + AuditLogStore
504 + ArtifactStore
505 + LogStore
506 + ScheduleStore
507 + ApprovalDelegationStore
508 + ProviderAccountStore
509 + SignalStore
510{
511}
512
513impl<
514 T: RunStore
515 + UserStore
516 + ApiKeyStore
517 + SecretStore
518 + AuditLogStore
519 + ArtifactStore
520 + LogStore
521 + ScheduleStore
522 + ApprovalDelegationStore
523 + ProviderAccountStore
524 + SignalStore,
525> Store for T
526{
527}