Skip to main content

cloud/reconciler/
container.rs

1//! `kind = "container"` reconciler — the local build+run path (R602-T1).
2//!
3//! A `container` component is a service packaged as a Docker image built from
4//! a Dockerfile that lives next to the component (`<path>/Dockerfile`). This
5//! reconciler drives the operator-local tier: detect the workspace's
6//! `local-container` runtime (orbstack/colima/docker, same provider the pond
7//! primitives use), `docker build` the image, then `docker run` it with the
8//! declared ports — adopt-idempotent, so a re-reconcile rebuilds (layer-cached)
9//! and replaces the running container in place.
10//!
11//! Config lives in the component's `workload.toml`:
12//!
13//! ```toml
14//! kind = "container"
15//!
16//! [build]
17//! # Dockerfile path, relative to the component dir. Default "Dockerfile".
18//! dockerfile = "Dockerfile"
19//! # Build context, relative to the workspace root. Default: the component
20//! # dir. Workspace crates set "." so their path-dependency sources resolve.
21//! context = "."
22//! # Image tag to build + run. Default: yah-local/<service>-<component>:dev.
23//! image = "yah-local/yah-cloud-admin:dev"
24//!
25//! [run]
26//! # Container port the process listens on.
27//! port = 4325
28//! # Host port to publish it on. Default: same as `port`.
29//! host_port = 4325
30//! # Environment passed into the container.
31//! [run.env]
32//! YAH_CLOUD_ADMIN_ADDR = "0.0.0.0:4325"
33//!
34//! # Bind mounts. `host` is relative to the workspace root (absolute paths are
35//! # taken as-is); `read_only` defaults to true.
36//! [[run.mounts]]
37//! host = ".yah/infra"
38//! container = "/workspace/.yah/infra"
39//! ```
40//!
41//! Mounts exist because a runtime image is a *binary*, not a checkout: the
42//! multi-stage build that produces it deliberately drops the workspace after
43//! the compile, so a service whose job is to read workspace config
44//! (yah-cloud-admin reads `.yah/infra/machines/*.toml`) came up rendering an
45//! empty fleet — running, healthy, and describing a fleet of zero machines,
46//! which is the worst possible failure for a monitor (R568-T7).
47//!
48//! Scope (R602-T1): the **local** tier only. Non-`local` mirror shapes bail
49//! with a pointer to `yah cloud workload deploy` (the yubaba-mediated cloud
50//! tier), which is a separate surface.
51
52use std::collections::BTreeMap;
53use std::future::Future;
54use std::pin::Pin;
55use std::time::Duration;
56
57use anyhow::{bail, Context, Result};
58use async_trait::async_trait;
59use local_driver::{canonical_name, ContainerRunSpec, ContainerState, LocalRuntime};
60use serde::Deserialize;
61
62use super::{ReconcileCtx, Reconciler, RunningWorkload};
63use crate::config::{CloudConfig, Provider};
64use crate::local_container_spec_from_provider;
65use crate::MirrorShape;
66
67/// The slot role a container component occupies on its mirror.
68const SLOT: &str = "compute";
69
70/// Options controlling the container reconciler's local path.
71#[derive(Debug, Clone, Default)]
72pub struct ContainerOptions {
73    /// When true, skip build+run and only adopt an already-running container
74    /// (parity with `PondOptions::adopt_only`). Errors when none is running.
75    pub adopt_only: bool,
76}
77
78/// Reconciler for `kind = "container"` components.
79#[derive(Debug, Default)]
80pub struct ContainerReconciler {
81    opts: ContainerOptions,
82}
83
84impl ContainerReconciler {
85    pub fn new() -> Self {
86        Self::default()
87    }
88
89    pub fn with_options(mut self, opts: ContainerOptions) -> Self {
90        self.opts = opts;
91        self
92    }
93}
94
95/// On-disk `workload.toml` shape for a `kind = "container"` component. Only
96/// the `[build]` + `[run]` sections this reconciler drives are parsed; other
97/// keys (name, kind, …) are ignored.
98#[derive(Debug, Default, Deserialize)]
99struct ContainerComponent {
100    #[serde(default)]
101    build: BuildSpec,
102    #[serde(default)]
103    run: RunSpec,
104}
105
106#[derive(Debug, Deserialize)]
107struct BuildSpec {
108    /// Dockerfile path, relative to the component dir.
109    #[serde(default = "default_dockerfile")]
110    dockerfile: String,
111    /// Build context, relative to the workspace root. `None` → component dir.
112    #[serde(default)]
113    context: Option<String>,
114    /// Image tag to build + run. `None` → derived from (service, component).
115    #[serde(default)]
116    image: Option<String>,
117}
118
119impl Default for BuildSpec {
120    fn default() -> Self {
121        Self {
122            dockerfile: default_dockerfile(),
123            context: None,
124            image: None,
125        }
126    }
127}
128
129fn default_dockerfile() -> String {
130    "Dockerfile".to_string()
131}
132
133#[derive(Debug, Default, Deserialize)]
134struct RunSpec {
135    /// Container port the process listens on.
136    #[serde(default)]
137    port: Option<u16>,
138    /// Host port to publish. `None` → same as `port`.
139    #[serde(default)]
140    host_port: Option<u16>,
141    /// Environment variables passed into the container.
142    #[serde(default)]
143    env: BTreeMap<String, String>,
144    /// Bind mounts from the workspace into the container.
145    #[serde(default)]
146    mounts: Vec<MountSpec>,
147}
148
149/// One `[[run.mounts]]` entry.
150#[derive(Debug, Deserialize)]
151struct MountSpec {
152    /// Host path. Relative paths resolve against the workspace root — the
153    /// declaration lives in the repo, so it should read like a repo path and
154    /// stay valid on whichever machine the operator runs it from.
155    host: String,
156    /// Absolute path inside the container.
157    container: String,
158    /// Default `true`. A workspace mount is config the service *reads*; a
159    /// writable default would let a container mutate the operator's checkout
160    /// as a side effect of running, so opting into that has to be explicit.
161    #[serde(default = "default_read_only")]
162    read_only: bool,
163}
164
165fn default_read_only() -> bool {
166    true
167}
168
169#[async_trait]
170impl Reconciler for ContainerReconciler {
171    fn kind(&self) -> &'static str {
172        "container"
173    }
174
175    async fn up(&self, ctx: ReconcileCtx<'_>) -> Result<RunningWorkload> {
176        // git-sourced components: clone/update the local checkout first
177        // (no-op for in-tree components).
178        ctx.materialize().await?;
179
180        // Scope guard: T1 is the operator-local tier. The cloud tier is
181        // yubaba-mediated (`yah cloud workload deploy`), a separate surface.
182        if !matches!(ctx.mirror.shape, MirrorShape::Local) {
183            bail!(
184                "component {}: kind \"container\" has only a local reconciler — mirror \
185                 shape is {:?}, not `local`. Deploy the cloud tier via \
186                 `yah cloud workload deploy` against a yubaba machine.",
187                ctx.component.id,
188                ctx.mirror.shape,
189            );
190        }
191
192        let spec = load_container_component(&ctx)?;
193        let container_port = spec.run.port.with_context(|| {
194            format!(
195                "component {}: workload.toml is missing [run].port — the container reconciler \
196                 needs the port the process listens on",
197                ctx.component.id,
198            )
199        })?;
200        let host_port = spec.run.host_port.unwrap_or(container_port);
201
202        let runtime = detect_local_runtime(&ctx)
203            .await
204            .context("detecting local container runtime (orbstack/colima/docker)")?;
205
206        let name = canonical_name(&ctx.service.name, ctx.env, &ctx.component.id);
207
208        // adopt-only: don't build/run, just report an already-running container.
209        if self.opts.adopt_only {
210            return match runtime.container_state(&name).await? {
211                Some(ContainerState::Running) => {
212                    let hp = runtime
213                        .container_host_port(&name, container_port)
214                        .await
215                        .unwrap_or(host_port);
216                    Ok(RunningWorkload::adopted(
217                        "container",
218                        SLOT,
219                        Some(format!("http://127.0.0.1:{hp}")),
220                    )
221                    .with_teardown(teardown_for(&ctx, name.clone())))
222                }
223                other => bail!(
224                    "adopt_only: no running container named {name} for component {} \
225                     (state: {other:?}) — nothing to adopt",
226                    ctx.component.id,
227                ),
228            };
229        }
230
231        // Build the image from the component's Dockerfile.
232        let image = spec
233            .build
234            .image
235            .clone()
236            .unwrap_or_else(|| default_image_tag(&ctx.service.name, &ctx.component.id));
237        let dockerfile = ctx.workload_dir().join(&spec.build.dockerfile);
238        let context = match &spec.build.context {
239            Some(rel) => ctx.workspace_root.join(rel),
240            None => ctx.workload_dir(),
241        };
242        runtime
243            .build_image(&image, &dockerfile, &context)
244            .await
245            .with_context(|| {
246                format!(
247                    "building image {image} for component {} (dockerfile {}, context {})",
248                    ctx.component.id,
249                    dockerfile.display(),
250                    context.display(),
251                )
252            })?;
253
254        // Run it with the declared ports + env. `run` clears any prior
255        // container of the same name first, so re-reconcile is idempotent.
256        let mut run_spec =
257            ContainerRunSpec::new(&ctx.service.name, ctx.env, &ctx.component.id, image);
258        run_spec.ports = vec![(host_port, container_port)];
259        run_spec.env = spec.run.env.clone();
260        run_spec.volumes = resolve_mounts(&spec.run.mounts, ctx.workspace_root)?;
261        runtime
262            .run(&run_spec)
263            .await
264            .with_context(|| format!("running container for component {}", ctx.component.id))?;
265
266        // Read the actual host port (host_port=0 requests an ephemeral one).
267        let actual = runtime
268            .container_host_port(&name, container_port)
269            .await
270            .unwrap_or(host_port);
271
272        Ok(RunningWorkload::adopted(
273            "container",
274            SLOT,
275            Some(format!("http://127.0.0.1:{actual}")),
276        )
277        .with_teardown(teardown_for(&ctx, name.clone())))
278    }
279}
280
281/// Grace period for the stop half of the teardown before the container is
282/// removed. Matches what an operator expects from a ■ button: long enough for
283/// a well-behaved server to close listeners, short enough not to look hung.
284const TEARDOWN_GRACE: Duration = Duration::from_secs(5);
285
286/// Build the explicit teardown for a container-kind workload (R714-B1).
287///
288/// Before this, `up()` handed back a bare `RunningWorkload::adopted()`, whose
289/// `shutdown()` is a documented no-op. Nothing else covered the gap either:
290/// the desktop's pond teardown half gates on a `Provider::MiniflareContainer`
291/// static slot, and a plain `kind = container` mirror declares no static slot,
292/// so its ident list came back empty and the loop body never ran. The ■ button
293/// removed the registry entry, called the no-op, and returned success with the
294/// container still up.
295///
296/// The runtime is re-detected inside the hook rather than captured: the hook
297/// outlives the borrowed [`ReconcileCtx`], and re-running the same lookup
298/// `up()` did means both halves agree on which daemon they mean even if the
299/// operator switched runtimes in between.
300/// The `Sync` in the return bound is required by [`RunningWorkload::with_teardown`]
301/// — see the note on its `TeardownFn` alias for why a non-`Sync` hook breaks
302/// every desktop Tauri command that touches the mirror registry.
303fn teardown_for(
304    ctx: &ReconcileCtx<'_>,
305    name: String,
306) -> impl FnOnce() -> Pin<Box<dyn Future<Output = Result<()>> + Send>> + Send + Sync + 'static {
307    let workspace_root = ctx.workspace_root.to_path_buf();
308    move || {
309        Box::pin(async move {
310            let runtime = detect_local_runtime_at(&workspace_root)
311                .await
312                .context("detecting local container runtime for teardown")?;
313            runtime
314                .stop_and_remove(&name, TEARDOWN_GRACE)
315                .await
316                .with_context(|| format!("stopping container {name}"))
317        })
318    }
319}
320
321/// Read `<workload_dir>/workload.toml` and parse the container `[build]` +
322/// `[run]` sections.
323fn load_container_component(ctx: &ReconcileCtx<'_>) -> Result<ContainerComponent> {
324    let path = ctx.workload_dir().join("workload.toml");
325    let src =
326        std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
327    toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))
328}
329
330/// Default image tag when `workload.toml` doesn't pin one.
331fn default_image_tag(service: &str, component: &str) -> String {
332    format!("yah-local/{service}-{component}:dev")
333}
334
335/// Turn `[[run.mounts]]` into `docker run -v` pairs, resolved against the
336/// workspace root.
337///
338/// A missing host path is a hard error rather than a skipped mount. Docker
339/// would happily create an empty directory in its place, and the container
340/// would then start, pass health checks, and serve whatever "no config found"
341/// means for that service — a deploy that looks green while the thing it was
342/// supposed to read isn't there.
343///
344/// The `:ro` suffix rides on the container-path string because
345/// `ContainerRunSpec::docker_run_args` emits `-v <host>:<container>` verbatim;
346/// that is docker's own mount-option syntax, not a hack around the type.
347fn resolve_mounts(
348    mounts: &[MountSpec],
349    workspace_root: &std::path::Path,
350) -> Result<Vec<(std::path::PathBuf, String)>> {
351    mounts
352        .iter()
353        .map(|m| {
354            let host = std::path::Path::new(&m.host);
355            let host = if host.is_absolute() {
356                host.to_path_buf()
357            } else {
358                workspace_root.join(host)
359            };
360            if !host.exists() {
361                bail!(
362                    "mount source {} does not exist (declared as `{}` in workload.toml \
363                     [[run.mounts]]) — the container would silently get an empty directory",
364                    host.display(),
365                    m.host,
366                );
367            }
368            let target = if m.read_only {
369                format!("{}:ro", m.container)
370            } else {
371                m.container.clone()
372            };
373            Ok((host, target))
374        })
375        .collect()
376}
377
378/// Detect the workspace's `local-container` runtime, the same way the pond
379/// primitives do — find the `kind = "local-container"` provider (orbstack.toml
380/// et al.) and probe its sockets. `ReconcileCtx` doesn't carry `CloudConfig`,
381/// so we reload it from the workspace root.
382async fn detect_local_runtime(ctx: &ReconcileCtx<'_>) -> Result<LocalRuntime> {
383    detect_local_runtime_at(ctx.workspace_root).await
384}
385
386/// Same lookup keyed on the workspace root alone, so the R714-B1 teardown hook
387/// — which outlives the borrowed [`ReconcileCtx`] — can re-detect the runtime
388/// without capturing it.
389async fn detect_local_runtime_at(workspace_root: &std::path::Path) -> Result<LocalRuntime> {
390    let cfg = CloudConfig::load(workspace_root)
391        .context("loading CloudConfig for local-container provider lookup")?;
392    let provider = cfg
393        .providers
394        .iter()
395        .find(|p| matches!(p.kind, Provider::LocalContainer))
396        .with_context(|| {
397            format!(
398                "no `kind = \"local-container\"` provider declared in {}/.yah/infra/providers/ — \
399                 the container reconciler needs orbstack.toml or equivalent",
400                workspace_root.display(),
401            )
402        })?;
403    let local_spec = local_container_spec_from_provider(provider)?;
404    LocalRuntime::detect(&local_spec).await
405}
406
407#[cfg(test)]
408mod tests {
409    use super::*;
410
411    #[test]
412    fn parses_build_and_run_sections() {
413        let src = r#"
414schema_version = 1
415name = "yah-cloud-admin"
416kind = "container"
417
418[build]
419dockerfile = "Dockerfile"
420context = "."
421image = "yah-local/yah-cloud-admin:dev"
422
423[run]
424port = 4325
425host_port = 4325
426
427[run.env]
428YAH_CLOUD_ADMIN_ADDR = "0.0.0.0:4325"
429YAH_CLOUD_ADMIN_DEV_ANON = "1"
430"#;
431        let c: ContainerComponent = toml::from_str(src).unwrap();
432        assert_eq!(c.build.dockerfile, "Dockerfile");
433        assert_eq!(c.build.context.as_deref(), Some("."));
434        assert_eq!(
435            c.build.image.as_deref(),
436            Some("yah-local/yah-cloud-admin:dev")
437        );
438        assert_eq!(c.run.port, Some(4325));
439        assert_eq!(c.run.host_port, Some(4325));
440        assert_eq!(
441            c.run.env.get("YAH_CLOUD_ADMIN_ADDR").map(String::as_str),
442            Some("0.0.0.0:4325")
443        );
444        assert_eq!(
445            c.run
446                .env
447                .get("YAH_CLOUD_ADMIN_DEV_ANON")
448                .map(String::as_str),
449            Some("1")
450        );
451    }
452
453    #[test]
454    fn mounts_resolve_against_the_workspace_root_and_default_read_only() {
455        let tmp = tempfile::tempdir().unwrap();
456        std::fs::create_dir_all(tmp.path().join(".yah/infra")).unwrap();
457        let src = r#"
458[run]
459port = 4325
460[[run.mounts]]
461host = ".yah/infra"
462container = "/workspace/.yah/infra"
463"#;
464        let c: ContainerComponent = toml::from_str(src).unwrap();
465        let out = resolve_mounts(&c.run.mounts, tmp.path()).unwrap();
466        assert_eq!(out.len(), 1);
467        assert_eq!(out[0].0, tmp.path().join(".yah/infra"));
468        assert_eq!(out[0].1, "/workspace/.yah/infra:ro");
469    }
470
471    #[test]
472    fn a_writable_mount_must_be_asked_for() {
473        let tmp = tempfile::tempdir().unwrap();
474        std::fs::create_dir_all(tmp.path().join("state")).unwrap();
475        let src = r#"
476[run]
477port = 1
478[[run.mounts]]
479host = "state"
480container = "/var/lib/state"
481read_only = false
482"#;
483        let c: ContainerComponent = toml::from_str(src).unwrap();
484        let out = resolve_mounts(&c.run.mounts, tmp.path()).unwrap();
485        assert_eq!(out[0].1, "/var/lib/state");
486    }
487
488    /// Docker would invent an empty directory here; a monitor mounting a
489    /// non-existent inventory must fail the deploy, not render zero machines.
490    #[test]
491    fn a_missing_mount_source_fails_the_reconcile() {
492        let tmp = tempfile::tempdir().unwrap();
493        let src = r#"
494[run]
495port = 1
496[[run.mounts]]
497host = "nope"
498container = "/nope"
499"#;
500        let c: ContainerComponent = toml::from_str(src).unwrap();
501        let err = resolve_mounts(&c.run.mounts, tmp.path()).unwrap_err();
502        assert!(err.to_string().contains("does not exist"), "{err}");
503    }
504
505    #[test]
506    fn no_mounts_declared_is_no_volumes() {
507        let c: ContainerComponent = toml::from_str("[run]\nport = 1\n").unwrap();
508        assert!(c.run.mounts.is_empty());
509        assert!(resolve_mounts(&c.run.mounts, std::path::Path::new("/"))
510            .unwrap()
511            .is_empty());
512    }
513
514    #[test]
515    fn build_defaults_when_section_absent() {
516        // A workload.toml with only [run] still parses; build takes defaults.
517        let src = r#"
518kind = "container"
519[run]
520port = 8080
521"#;
522        let c: ContainerComponent = toml::from_str(src).unwrap();
523        assert_eq!(c.build.dockerfile, "Dockerfile");
524        assert!(c.build.context.is_none());
525        assert!(c.build.image.is_none());
526        assert_eq!(c.run.port, Some(8080));
527        assert!(c.run.host_port.is_none());
528        assert!(c.run.env.is_empty());
529    }
530
531    /// R714-B1: the teardown must target the container `run()` actually
532    /// created. `LocalRuntime::stop_and_remove` is a documented no-op on a
533    /// container that doesn't exist, so if these two names ever drift apart
534    /// the ■ button goes back to reporting success while the container runs —
535    /// and it does so silently, with no error to surface. `up()` feeds the
536    /// same `(service, env, component.id)` triple to both; this pins that.
537    #[test]
538    fn the_teardown_name_matches_the_name_run_created() {
539        let (service, env, component) = ("yah-cloud-admin", "pond", "cloud-admin");
540        let run_spec = ContainerRunSpec::new(service, env, component, "img:dev");
541        let teardown_target = canonical_name(service, env, component);
542        assert_eq!(
543            run_spec.name, teardown_target,
544            "teardown would docker-stop a name that was never created"
545        );
546    }
547
548    #[test]
549    fn default_image_tag_derives_from_service_and_component() {
550        assert_eq!(
551            default_image_tag("yah-cloud-admin", "cloud-admin"),
552            "yah-local/yah-cloud-admin-cloud-admin:dev"
553        );
554    }
555
556    #[test]
557    fn run_spec_publishes_declared_ports_and_env() {
558        // The docker_run_args wiring a live reconcile would emit, exercised
559        // without a docker socket.
560        let mut run_spec =
561            ContainerRunSpec::new("yah-cloud-admin", "dev", "cloud-admin", "img:dev");
562        run_spec.ports = vec![(4325, 4325)];
563        run_spec.env.insert("K".into(), "V".into());
564        let args = run_spec.docker_run_args();
565        // -p 4325:4325 present.
566        let joined = args.join(" ");
567        assert!(joined.contains("-p 4325:4325"), "args: {joined}");
568        assert!(joined.contains("-e K=V"), "args: {joined}");
569        assert!(
570            joined.ends_with("img:dev"),
571            "image is the final arg: {joined}"
572        );
573    }
574}