Skip to main content

lfsx_server/
lib.rs

1pub(crate) mod audit;
2pub mod auth;
3pub mod config;
4pub mod console;
5pub mod dashboard;
6pub mod error;
7#[cfg(feature = "fuzzing")]
8pub mod fuzzing;
9pub mod locks;
10pub mod metrics;
11pub mod model;
12pub mod namespace;
13pub mod oid;
14pub mod page;
15pub mod range;
16pub mod routes;
17pub mod state;
18pub mod storage;
19pub mod telemetry;
20pub mod tls;
21
22use std::sync::Arc;
23
24use axum::Router;
25
26use crate::auth::Authorizer;
27use crate::config::Config;
28use crate::locks::LockStore;
29use crate::metrics::Metrics;
30use crate::state::AppState;
31use crate::storage::s3::{
32    AzureConfig, AzureKeys, GcsConfig, GcsKeys, Keyspace, S3Config, S3Keys, S3Store,
33};
34use crate::storage::{LocalStore, Store};
35
36pub fn app(config: Config) -> Router {
37    // Here rather than in `backends`, which `reclaim` also calls: a reclaim pass
38    // hands nobody a URL, and saying this twice at every boot teaches an operator
39    // to skim it.
40    //
41    // The hrefs in a batch answer are where a client sends the object, and it
42    // sends its credential with them. Unset, they are built from the `Host` and
43    // `X-Forwarded-Proto` of whoever asked, which is a deployment fact only for
44    // as long as something in front is rewriting both.
45    //
46    // Warned rather than refused. Every deployment that works today works without
47    // it, and taking those down to close a hole most of them do not have is the
48    // wrong trade.
49    if config.public_url.is_none() && !matches!(config.auth, crate::config::Auth::Disabled) {
50        tracing::warn!(
51            "LFSX_PUBLIC_URL is not set, so the URLs handed to clients are built from the Host and \
52             X-Forwarded-Proto headers of whoever asked. Behind a proxy that does not rewrite them, \
53             a caller chooses where the next request goes and takes its token there. Set it to the \
54             address clients actually use"
55        );
56    }
57
58    announce_access(&config);
59    announce_forges(&config);
60    announce_storage(&config);
61    let (store, locks) = backends(&config);
62    let authorizer = Authorizer::new(&config.auth);
63    let forges = config
64        .forges
65        .iter()
66        .map(|forge| (forge.name.clone(), Authorizer::new(&forge.auth)))
67        .collect();
68    let transfers = (config.max_concurrent_transfers > 0)
69        .then(|| Arc::new(tokio::sync::Semaphore::new(config.max_concurrent_transfers)));
70
71    let state = Arc::new(AppState {
72        store,
73        locks,
74        config,
75        authorizer,
76        forges,
77        metrics: Metrics::new(),
78        transfers,
79        started: std::time::Instant::now(),
80        session_key: tokio::sync::OnceCell::new(),
81    });
82
83    if state.config.dashboard.is_some() && tokio::runtime::Handle::try_current().is_ok() {
84        console::start(state.clone());
85    }
86
87    routes::router(state)
88}
89
90// Everything an interrupted upload left behind, wherever it left it: a staging
91// file on the volume, or bytes under an upload key nobody ever reported. Built
92// from the same construction the server uses, so a bucket deployment does not
93// end up sweeping only half of itself.
94pub async fn reclaim(config: &Config) {
95    let reclaimed = backends(config).0.reclaim(config.staging_max_age).await;
96
97    if reclaimed.files > 0 {
98        tracing::info!(
99            files = reclaimed.files,
100            bytes = reclaimed.bytes,
101            "reclaimed what interrupted uploads left behind"
102        );
103    }
104}
105
106// Ask the bucket, once, whether it really refuses an upload whose body does not
107// match the checksum its URL was signed for, and give up pre-signing if it does
108// not say yes.
109//
110// Handing a client a write URL is safe only because of that refusal. Without it,
111// anyone with push rights to any repository can put chosen bytes under a chosen
112// digest, and objects are shared: bytes live once at `.content/{oid}`, so every
113// repository that later pushes that digest gets a marker pointing at them and
114// uploads nothing. One store that ignores the header decides what an object is
115// for everybody.
116//
117// Losing pre-signing costs throughput and nothing else, because transfers fall
118// back to coming through this server, which hashes what it is sent. That is why
119// a store which cannot be asked loses it too: the question guards data, and an
120// unanswered question is not a yes.
121pub async fn verify_presign(config: &mut Config) {
122    use crate::storage::s3::probe::{Checksums, checksums};
123
124    let crate::config::Storage::Bucket { presign: true, .. } = &config.storage else {
125        return;
126    };
127
128    let Some(keys) = keyspace(config) else {
129        return;
130    };
131
132    let keys = match keys {
133        Keyspace::S3(keys) => keys,
134        Keyspace::Azure(keys) => {
135            if keys.signed_download("").is_none() {
136                tracing::warn!(
137                    "LFSX_S3_PRESIGN is set, and only an account key can sign a download URL on \
138                     Azure, so downloads keep coming through this server"
139                );
140            }
141            return;
142        }
143        Keyspace::Gcs(keys) => {
144            if keys.signed_download("probe").is_none() {
145                tracing::warn!(
146                    "LFSX_S3_PRESIGN is set, and only a service account key can sign a download \
147                     URL on Google Cloud Storage, so downloads keep coming through this server"
148                );
149            }
150            return;
151        }
152    };
153
154    let refusal = match checksums(&keys).await {
155        Checksums::Enforced => return,
156        Checksums::Ignored => {
157            "this object store accepted an upload whose body did not match the checksum its own \
158             signature named. A store that does not verify that header lets a client with push \
159             rights put chosen bytes under a chosen digest, and every repository that later pushes \
160             that digest would get a marker pointing at them"
161        }
162        Checksums::Unknown => {
163            "this object store could not be asked whether it verifies upload checksums. Handing out \
164             a write URL is only safe if the store refuses a body that does not match it, and that \
165             has not been established"
166        }
167    };
168
169    tracing::error!(
170        "{refusal}, so LFSX_S3_PRESIGN is being ignored and uploads keep coming through this server"
171    );
172
173    if let crate::config::Storage::Bucket { presign, .. } = &mut config.storage {
174        *presign = false;
175    }
176}
177
178// Ask the bucket, once, whether it refuses the second of two conditional writes,
179// and give up locking if it will not say yes.
180//
181// That refusal is the entirety of lock uniqueness here. Two clients race for the
182// same path, both write, and the store is the only thing that can say one of them
183// arrived second. A store that accepts `If-None-Match: *` without implementing it
184// performs both writes and reports success twice, so both are told the lock is
185// theirs, and nothing anywhere notices.
186//
187// There is no safe degraded mode for that, so taking a lock becomes a `501`
188// instead. It is the loudest honest answer: a client sees a refusal at the moment
189// it asks, rather than a lock somebody else also holds. Everything else about the
190// deployment is untouched, objects included, because a team that never takes a
191// lock should not lose a working server over this.
192pub async fn verify_locking(config: &mut Config) {
193    use crate::storage::s3::probe::{Conditional, conditional_writes};
194
195    let Some(keys) = keyspace(config) else {
196        return;
197    };
198
199    let refusal = match conditional_writes(&keys).await {
200        Conditional::Enforced => return,
201        Conditional::Ignored => {
202            "this object store wrote the same key twice under a condition that should have refused \
203             the second, so it cannot say which of two clients racing for a lock arrived first"
204        }
205        Conditional::Unknown => {
206            "this object store could not be asked whether it refuses a conditional write, and lock \
207             uniqueness is exactly that refusal"
208        }
209    };
210
211    tracing::error!(
212        "{refusal}, so taking a lock here answers 501. Objects are unaffected, and so is everything \
213         else this server does"
214    );
215
216    if let crate::config::Storage::Bucket { locking, .. } = &mut config.storage {
217        *locking = false;
218    }
219}
220
221fn keyspace(config: &Config) -> Option<Keyspace> {
222    let crate::config::Storage::Bucket { dialect, .. } = &config.storage else {
223        return None;
224    };
225
226    let lifetime = std::time::Duration::from_secs(config.action_lifetime.into());
227
228    Some(match dialect {
229        crate::config::Dialect::S3 {
230            endpoint,
231            bucket,
232            region,
233            access_key,
234            secret_key,
235            path_style,
236        } => Keyspace::S3(
237            S3Keys::new(&S3Config {
238                endpoint: endpoint.clone(),
239                bucket: bucket.clone(),
240                region: region.clone(),
241                access_key: access_key.clone(),
242                secret_key: secret_key.clone(),
243                path_style: *path_style,
244                lifetime,
245            })
246            .expect("the bucket configuration is not usable"),
247        ),
248        crate::config::Dialect::Azure {
249            endpoint,
250            account,
251            container,
252            credential,
253        } => Keyspace::Azure(
254            AzureKeys::new(&AzureConfig {
255                endpoint: endpoint.clone(),
256                account: account.clone(),
257                container: container.clone(),
258                credential: credential.clone(),
259                lifetime,
260            })
261            .expect("the Azure container configuration is not usable"),
262        ),
263        crate::config::Dialect::Gcs {
264            endpoint,
265            bucket,
266            credential,
267        } => Keyspace::Gcs(
268            GcsKeys::new(&GcsConfig {
269                endpoint: endpoint.clone(),
270                bucket: bucket.clone(),
271                credential: credential.clone(),
272                lifetime,
273            })
274            .expect("the Google Cloud Storage configuration is not usable"),
275        ),
276    })
277}
278
279pub fn store(config: &Config) -> Store {
280    backends(config).0
281}
282
283fn announce_access(config: &Config) {
284    // Said out loud because it decides who can read the objects. It is off unless
285    // asked for, so this line means somebody asked: it belongs in the log so a
286    // deployment that inherited the flag from an older chart sees it rather than
287    // discovers it.
288    if let crate::config::Auth::Forge {
289        anonymous_read: true,
290        ..
291    } = config.auth
292    {
293        tracing::info!(
294            "anonymous read is on: a request with no credentials is resolved against the forge, so \
295             objects in a repository the forge serves publicly can be read by anybody, and the \
296             bandwidth is yours. Unset LFSX_ANONYMOUS_READ to require a token whatever the \
297             repository's visibility"
298        );
299    }
300
301    // Same discipline: a list inherited from a chart is worth seeing at boot
302    // rather than discovering when somebody reports a repository they can clone
303    // and cannot pull.
304    if let crate::config::Auth::Forge { restricted, .. } = &config.auth
305        && !restricted.is_empty()
306    {
307        tracing::info!(
308            "restricted namespaces are configured: objects in a listed repository take write \
309             access to read, so a caller the forge grants pull is refused. Unset LFSX_RESTRICTED \
310             to serve every repository the permissions the forge gives it"
311        );
312    }
313
314    if let crate::config::Auth::Forge { allowed, .. } = &config.auth {
315        match allowed {
316            None => tracing::warn!(
317                "LFSX_ALLOWED is unset, so this server stores objects for any repository on the \
318                 forge whose caller can push to it, including a repository a stranger creates \
319                 for the purpose. List the organisations or repositories it is for"
320            ),
321            Some(allowed) if allowed.is_empty() => tracing::warn!(
322                "LFSX_ALLOWED is set and none of its entries is org/repo, so this server serves no \
323                 repository at all"
324            ),
325            Some(_) => tracing::info!(
326                "an allow-list is configured: a repository outside LFSX_ALLOWED is answered 404 \
327                 without asking the forge"
328            ),
329        }
330    }
331}
332
333fn announce_forges(config: &Config) {
334    for forge in &config.forges {
335        let crate::config::Auth::Forge {
336            provider,
337            api_url,
338            allowed,
339            ..
340        } = &forge.auth
341        else {
342            continue;
343        };
344        tracing::info!(
345            forge = %forge.name,
346            ?provider,
347            %api_url,
348            "repositories on this forge are served under /-/{}/", forge.name
349        );
350        if allowed.is_none() {
351            tracing::warn!(
352                forge = %forge.name,
353                "this forge has no allow-list, so it stores objects for any repository on it whose \
354                 caller can push to it. List the organisations or repositories it is for"
355            );
356        }
357    }
358}
359
360fn announce_storage(config: &Config) {
361    let crate::config::Storage::Bucket { presign, cache, .. } = &config.storage else {
362        return;
363    };
364
365    tracing::warn!(
366        "objects and locks are stored in a bucket: deduplication, rewriting and \
367     verification answer 501, and the lfsx_objects_stored and lfsx_store_bytes \
368     gauges are not measured: read capacity from the bucket itself"
369    );
370
371    if *presign {
372        if config.encryption_key.is_some() || config.compression.is_some() {
373            tracing::warn!(
374                "LFSX_S3_PRESIGN=true, but a codec is configured, so downloads keep \
375     streaming through this server: what sits in the bucket is a frame under \
376     the plaintext digest, and a client handed that directly would hash it \
377     and reject the object"
378            );
379        } else {
380            tracing::warn!(
381                "LFSX_S3_PRESIGN=true, downloads are redirected to the bucket, so \
382     lfsx_downloaded_bytes stops counting them and the bucket serves the ranges"
383            );
384        }
385
386        if config.encryption_key.is_some() {
387            tracing::warn!(
388                "an encryption key is configured, so uploads keep coming through this \
389     server rather than going straight to the bucket: an object a client \
390     writes itself would arrive unencrypted"
391            );
392        } else if config.compression.is_some() {
393            tracing::warn!(
394                "LFSX_COMPRESSION is set, and objects clients upload straight to the \
395     bucket arrive uncompressed: only what passes through this server is \
396     compressed"
397            );
398        }
399    }
400
401    if cache.is_some() && *presign {
402        tracing::warn!(
403            "LFSX_S3_CACHE_DIR is set with LFSX_S3_PRESIGN=true, so downloads go straight to the \
404     bucket and the cache never sees them: the two settings pull in opposite directions"
405        );
406    }
407}
408
409fn backends(config: &Config) -> (Store, LockStore) {
410    // Refusing to start beats starting without it. A server that silently wrote
411    // plaintext because a Secret failed to mount is the one failure this feature
412    // must never have: nothing downstream would notice, and the objects written
413    // in the meantime are the ones the operator believed were covered.
414    let keys = config.encryption_key.as_ref().map(|source| {
415        std::sync::Arc::new(
416            crate::storage::crypt::Keyring::from_source(source)
417                .expect("the encryption key source is not usable"),
418        )
419    });
420
421    let local = LocalStore::new(config.storage_root.clone())
422        .with_max_object_size(config.max_object_size)
423        .with_compression(config.compression)
424        .with_encryption(keys);
425
426    // The two backends are chosen together and the lock policy is applied once,
427    // to both. Deciding it per arm is how `LFSX_LOCK_MAX_AGE` came to be silently
428    // ignored in bucket mode: the arms are far apart, only one of them had it,
429    // and nothing failed.
430    let (store, lock_backend) = match &config.storage {
431        crate::config::Storage::Local => (
432            Store::local(local),
433            LockStore::local(config.storage_root.clone()),
434        ),
435        crate::config::Storage::Bucket {
436            presign,
437            locking,
438            cache,
439            ..
440        } => {
441            // Built once and shared: the objects and the locks are two ways of
442            // using the same bucket, not two buckets. Signing, the connection
443            // pool and the retry policy are settled here, and neither layer
444            // reaches into the other to get at them.
445            let keys = keyspace(config).expect("a bucket keyspace for a bucket store");
446
447            // The locks go with the objects. Left on the volume they would make
448            // the bucket a half measure: capacity would be shared and the one
449            // piece of state a second replica must agree on would not be.
450            // A cache the server cannot create is a misconfiguration worth
451            // stopping for: the alternative is a deployment that silently keeps
452            // paying the round trips the operator thought they had bought out of.
453            let disk = cache.as_ref().map(|disk| {
454                crate::storage::cache::Cache::new(disk.dir.clone(), disk.max_bytes)
455                    .expect("the cache directory is not usable")
456            });
457
458            (
459                Store::bucket(S3Store::new(keys.clone(), *presign), local).with_cache(disk),
460                LockStore::bucket(keys).with_conditional_writes(*locking),
461            )
462        }
463    };
464    (store, lock_backend.with_max_age(config.lock_max_age))
465}