Skip to main content

webserver_base/sitemap/
mod.rs

1//! Sitemap generation: always an index, always at least one url set.
2//!
3//! Which URLs a site has comes from [`Pages`](crate::webserver::Pages), so the
4//! sitemap is derived from the route table rather than kept beside it.
5//!
6//! The shape is deliberately uniform. `/sitemap.xml` is a `<sitemapindex>` on a
7//! three-page site and on a three-million-page one, so `robots.txt` and any
8//! Search Console submission point at one URL forever. A site that grows past
9//! the protocol's 50,000-URL limit simply gains another `/sitemap-N.xml` inside
10//! the index; nothing outside the server changes, and nothing has to be
11//! resubmitted. The alternative — a bare `<urlset>` that silently becomes an
12//! index at the boundary — invalidates that submission at exactly the moment
13//! the site matters most.
14
15use chrono::{DateTime, Utc};
16use sitemap_rs::image::Image;
17use sitemap_rs::sitemap::Sitemap;
18use sitemap_rs::sitemap_index::SitemapIndex;
19use sitemap_rs::url::Url;
20use sitemap_rs::url_builder::UrlBuilder;
21use sitemap_rs::url_set::UrlSet;
22use tracing::{debug, instrument};
23
24/// The route the sitemap index is served at.
25pub const SITEMAP_INDEX_PATH: &str = "/sitemap.xml";
26
27/// URLs per url set. The protocol allows 50,000; the margin absorbs the
28/// difference between an entry's typical and worst-case size.
29pub const MAX_URLS_PER_SITEMAP: usize = 45_000;
30
31/// Bytes per url set. The protocol allows 50 MB uncompressed. `sitemap-rs`
32/// enforces only the URL count, so a set of image-heavy entries can pass that
33/// check and still be rejected — this is the guard for that case.
34pub const MAX_BYTES_PER_SITEMAP: usize = 45 * 1024 * 1024;
35
36/// Why a sitemap could not be produced.
37#[derive(Debug, thiserror::Error)]
38pub enum SitemapError {
39    /// A single `<url>` entry was rejected.
40    #[error("invalid sitemap entry for `{location}`")]
41    Url {
42        location: String,
43        #[source]
44        source: sitemap_rs::url_error::UrlError,
45    },
46
47    /// A `<urlset>` was rejected.
48    #[error("invalid sitemap url set ({count} urls)")]
49    UrlSet {
50        count: usize,
51        #[source]
52        source: sitemap_rs::url_set_error::UrlSetError,
53    },
54
55    /// The `<sitemapindex>` was rejected — more than 50,000 url sets, which is
56    /// 2.25 billion URLs.
57    #[error("invalid sitemap index ({count} sitemaps)")]
58    Index {
59        count: usize,
60        #[source]
61        source: sitemap_rs::sitemap_index_error::SitemapIndexError,
62    },
63
64    /// The XML could not be serialized.
65    #[error("failed to serialize the sitemap")]
66    Write {
67        #[source]
68        source: xml_builder::XMLError,
69    },
70
71    /// Serialized XML was not valid UTF-8, which cannot happen for input this
72    /// crate produces.
73    #[error("the serialized sitemap was not valid UTF-8")]
74    Encoding,
75}
76
77/// One entry in a sitemap. Holds a site-relative path; the origin is prepended
78/// at build time, so a URL cannot be listed under the wrong host.
79///
80/// Deliberately absent: `<changefreq>` and `<priority>`. Google ignores both —
81/// priority is subjective and change frequency is guessed — so carrying them
82/// would be markup with no consumer.
83#[derive(Debug, Clone)]
84pub struct SitemapUrl {
85    path: String,
86    last_modified: Option<DateTime<Utc>>,
87    images: Vec<String>,
88}
89
90impl SitemapUrl {
91    /// An entry for a site-relative `path`, e.g. `/blog/rust-tips`.
92    #[must_use]
93    pub fn new(path: impl Into<String>) -> Self {
94        Self {
95            path: path.into(),
96            last_modified: None,
97            images: Vec::new(),
98        }
99    }
100
101    /// Sets `<lastmod>` for this page specifically.
102    ///
103    /// Prefer a real modification date. Google uses `lastmod` only when it is
104    /// "consistently and verifiably accurate" — it fetches the page and checks
105    /// — and one bad pattern discredits the whole file, so a made-up date is
106    /// worse than none.
107    #[must_use]
108    pub const fn with_last_modified(mut self, last_modified: DateTime<Utc>) -> Self {
109        self.last_modified = Some(last_modified);
110        self
111    }
112
113    /// Adds `<image:image>` entries, as site-relative or absolute URLs.
114    #[must_use]
115    pub fn extend_images<I, S>(mut self, images: I) -> Self
116    where
117        I: IntoIterator<Item = S>,
118        S: Into<String>,
119    {
120        self.images.extend(images.into_iter().map(Into::into));
121        self
122    }
123
124    /// The site-relative path.
125    #[must_use]
126    pub fn path(&self) -> &str {
127        &self.path
128    }
129
130    /// The `<lastmod>` override, if one was set.
131    #[must_use]
132    pub const fn last_modified(&self) -> Option<DateTime<Utc>> {
133        self.last_modified
134    }
135
136    /// The `<image:image>` entries, as supplied.
137    #[must_use]
138    pub fn images(&self) -> &[String] {
139        &self.images
140    }
141
142    /// Rewrites every image path through `resolve`.
143    ///
144    /// Images are declared by their logical path but served only at their
145    /// content-hashed one, so without this the sitemap advertises URLs that
146    /// 404 — and an image sitemap full of dead links is worse than none.
147    #[must_use]
148    pub fn map_images<F>(mut self, resolve: F) -> Self
149    where
150        F: Fn(&str) -> String,
151    {
152        self.images = self.images.iter().map(|image| resolve(image)).collect();
153        self
154    }
155
156    /// Builds the `sitemap-rs` entry, resolving `path` against `base_url`.
157    fn build(
158        &self,
159        base_url: &str,
160        fallback_last_modified: Option<DateTime<Utc>>,
161    ) -> Result<Url, SitemapError> {
162        let location: String = absolute(base_url, &self.path);
163
164        let mut builder: UrlBuilder = Url::builder(location.clone());
165        if let Some(last_modified) = self.last_modified.or(fallback_last_modified) {
166            builder.last_modified(DateTime::from(last_modified));
167        }
168
169        if !self.images.is_empty() {
170            builder.images(
171                self.images
172                    .iter()
173                    .map(|image| Image::new(absolute(base_url, image)))
174                    .collect(),
175            );
176        }
177
178        builder
179            .build()
180            .map_err(|source| SitemapError::Url { location, source })
181    }
182}
183
184/// A complete sitemap set: one index naming one or more url sets.
185///
186/// Held in memory and served from there. Nothing is written to disk, so a
187/// production image needs no writable filesystem, and there is no generated
188/// file to hash, cache, or accidentally commit.
189#[derive(Debug, Clone, PartialEq, Eq)]
190pub struct SitemapSet {
191    index: String,
192    chunks: Vec<String>,
193}
194
195impl SitemapSet {
196    /// The `<sitemapindex>` document, served at [`SITEMAP_INDEX_PATH`].
197    #[must_use]
198    pub fn index(&self) -> &str {
199        &self.index
200    }
201
202    /// The `<urlset>` documents, in order. Chunk `n` is served at
203    /// `/sitemap-{n+1}.xml`.
204    #[must_use]
205    pub fn chunks(&self) -> &[String] {
206        &self.chunks
207    }
208
209    /// Every path this set is served at, index first.
210    ///
211    /// Index first is load-bearing for `robots.txt`: Google discards anything
212    /// past 500 KiB, and truncation is positional, so the one line that must
213    /// survive goes at the top.
214    #[must_use]
215    pub fn paths(&self) -> Vec<String> {
216        let mut paths: Vec<String> = vec![String::from(SITEMAP_INDEX_PATH)];
217        paths.extend((1..=self.chunks.len()).map(|n| format!("/sitemap-{n}.xml")));
218        paths
219    }
220}
221
222/// Builds the sitemap index and its url sets for `urls`.
223///
224/// `fallback_last_modified` is used for entries that name no date of their own;
225/// pass `None` to omit `<lastmod>` rather than assert something unverifiable.
226///
227/// # Errors
228///
229/// [`SitemapError`] if an entry, a url set, or the index is invalid, or if the
230/// XML cannot be serialized.
231#[instrument(skip_all)]
232pub fn build_sitemaps(
233    base_url: &str,
234    urls: &[SitemapUrl],
235    fallback_last_modified: Option<DateTime<Utc>>,
236) -> Result<SitemapSet, SitemapError> {
237    let mut chunks: Vec<String> = Vec::new();
238
239    // An empty site still gets one (empty) url set, so the index is never a
240    // dangling reference and the served shape never varies.
241    for batch in urls.chunks(MAX_URLS_PER_SITEMAP).chain(if urls.is_empty() {
242        Some([].as_slice())
243    } else {
244        None
245    }) {
246        chunks.extend(render_url_set(base_url, batch, fallback_last_modified)?);
247    }
248
249    let sitemaps: Vec<Sitemap> = (1..=chunks.len())
250        .map(|n| {
251            Sitemap::new(
252                absolute(base_url, &format!("/sitemap-{n}.xml")),
253                fallback_last_modified.map(DateTime::from),
254            )
255        })
256        .collect();
257
258    let count: usize = sitemaps.len();
259    let index: SitemapIndex =
260        SitemapIndex::new(sitemaps).map_err(|source| SitemapError::Index { count, source })?;
261
262    let mut buffer: Vec<u8> = Vec::new();
263    index
264        .write(&mut buffer)
265        .map_err(|source| SitemapError::Write { source })?;
266    let index: String = String::from_utf8(buffer).map_err(|_| SitemapError::Encoding)?;
267
268    debug!("built {} url(s) across {count} sitemap(s)", urls.len());
269    Ok(SitemapSet { index, chunks })
270}
271
272/// Serializes one batch, splitting it further if the result exceeds the
273/// protocol's byte limit.
274fn render_url_set(
275    base_url: &str,
276    batch: &[SitemapUrl],
277    fallback_last_modified: Option<DateTime<Utc>>,
278) -> Result<Vec<String>, SitemapError> {
279    let built: Vec<Url> = batch
280        .iter()
281        .map(|url| url.build(base_url, fallback_last_modified))
282        .collect::<Result<Vec<Url>, SitemapError>>()?;
283
284    let count: usize = built.len();
285    let url_set: UrlSet =
286        UrlSet::new(built).map_err(|source| SitemapError::UrlSet { count, source })?;
287
288    let mut buffer: Vec<u8> = Vec::new();
289    url_set
290        .write(&mut buffer)
291        .map_err(|source| SitemapError::Write { source })?;
292
293    if buffer.len() <= MAX_BYTES_PER_SITEMAP || batch.len() < 2 {
294        return Ok(vec![
295            String::from_utf8(buffer).map_err(|_| SitemapError::Encoding)?,
296        ]);
297    }
298
299    // Too large despite an acceptable URL count — image-heavy entries. Halve
300    // and retry; each half is re-measured, so this terminates.
301    let (left, right) = batch.split_at(batch.len() / 2);
302    let mut rendered: Vec<String> = render_url_set(base_url, left, fallback_last_modified)?;
303    rendered.extend(render_url_set(base_url, right, fallback_last_modified)?);
304    Ok(rendered)
305}
306
307/// Joins a site origin and a path, tolerating a slash on either side or both.
308fn absolute(base_url: &str, path: &str) -> String {
309    if path.starts_with("http://") || path.starts_with("https://") {
310        return path.to_string();
311    }
312    let base: &str = base_url.trim_end_matches('/');
313    let path: &str = path.trim_start_matches('/');
314    if path.is_empty() {
315        format!("{base}/")
316    } else {
317        format!("{base}/{path}")
318    }
319}
320
321#[cfg(test)]
322mod tests {
323    use chrono::{DateTime, TimeZone, Utc};
324
325    use super::{SitemapSet, SitemapUrl, absolute, build_sitemaps};
326
327    #[test]
328    fn joining_never_doubles_or_drops_a_slash() {
329        let cases: [(&str, &str, &str); 5] = [
330            ("https://a.com", "/blog", "https://a.com/blog"),
331            ("https://a.com/", "/blog", "https://a.com/blog"),
332            ("https://a.com", "blog", "https://a.com/blog"),
333            ("https://a.com", "/", "https://a.com/"),
334            ("https://a.com", "", "https://a.com/"),
335        ];
336        for (base, path, expected) in cases {
337            let actual: String = absolute(base, path);
338            assert_eq!(expected, actual, "base {base:?} path {path:?}");
339        }
340    }
341
342    #[test]
343    fn an_absolute_image_url_is_left_alone() {
344        let expected: String = String::from("https://cdn.example.com/card.png");
345        let actual: String = absolute("https://a.com", "https://cdn.example.com/card.png");
346        assert_eq!(expected, actual);
347    }
348
349    #[test]
350    fn a_small_site_still_gets_an_index_so_the_shape_never_changes() {
351        let urls: Vec<SitemapUrl> = vec![SitemapUrl::new("/"), SitemapUrl::new("/blog")];
352        let set: SitemapSet =
353            build_sitemaps("https://www.example.com", &urls, None).expect("builds");
354
355        assert!(set.index().contains("<sitemapindex"));
356        assert!(
357            set.index()
358                .contains("https://www.example.com/sitemap-1.xml")
359        );
360
361        let expected: usize = 1;
362        let actual: usize = set.chunks().len();
363        assert_eq!(expected, actual);
364
365        assert!(set.chunks()[0].contains("<loc>https://www.example.com/</loc>"));
366        assert!(set.chunks()[0].contains("<loc>https://www.example.com/blog</loc>"));
367    }
368
369    #[test]
370    fn a_site_with_no_pages_still_produces_a_well_formed_pair() {
371        let set: SitemapSet = build_sitemaps("https://www.example.com", &[], None).expect("builds");
372
373        let expected: usize = 1;
374        let actual: usize = set.chunks().len();
375        assert_eq!(expected, actual);
376        assert!(set.index().contains("<sitemapindex"));
377    }
378
379    #[test]
380    fn urls_beyond_the_limit_spill_into_another_url_set() {
381        let urls: Vec<SitemapUrl> = (0..super::MAX_URLS_PER_SITEMAP + 10)
382            .map(|n| SitemapUrl::new(format!("/page/{n}")))
383            .collect();
384        let set: SitemapSet =
385            build_sitemaps("https://www.example.com", &urls, None).expect("builds");
386
387        let expected: usize = 2;
388        let actual: usize = set.chunks().len();
389        assert_eq!(expected, actual);
390        assert!(
391            set.index()
392                .contains("https://www.example.com/sitemap-2.xml")
393        );
394    }
395
396    #[test]
397    fn the_index_path_comes_first_so_it_survives_truncation() {
398        let urls: Vec<SitemapUrl> = (0..=super::MAX_URLS_PER_SITEMAP)
399            .map(|n| SitemapUrl::new(format!("/page/{n}")))
400            .collect();
401        let set: SitemapSet =
402            build_sitemaps("https://www.example.com", &urls, None).expect("builds");
403
404        let expected: Vec<String> = vec![
405            String::from("/sitemap.xml"),
406            String::from("/sitemap-1.xml"),
407            String::from("/sitemap-2.xml"),
408        ];
409        let actual: Vec<String> = set.paths();
410        assert_eq!(expected, actual);
411    }
412
413    #[test]
414    fn a_page_without_a_date_gets_no_lastmod_rather_than_a_false_one() {
415        let urls: Vec<SitemapUrl> = vec![SitemapUrl::new("/")];
416        let set: SitemapSet =
417            build_sitemaps("https://www.example.com", &urls, None).expect("builds");
418
419        assert!(!set.chunks()[0].contains("<lastmod>"));
420    }
421
422    #[test]
423    fn a_page_date_wins_over_the_site_wide_fallback() {
424        let page_date: DateTime<Utc> = Utc.with_ymd_and_hms(2026, 1, 2, 3, 4, 5).unwrap();
425        let site_date: DateTime<Utc> = Utc.with_ymd_and_hms(2020, 1, 1, 0, 0, 0).unwrap();
426
427        let urls: Vec<SitemapUrl> = vec![
428            SitemapUrl::new("/blog").with_last_modified(page_date),
429            SitemapUrl::new("/"),
430        ];
431        let set: SitemapSet =
432            build_sitemaps("https://www.example.com", &urls, Some(site_date)).expect("builds");
433
434        assert!(set.chunks()[0].contains("2026-01-02"));
435        assert!(set.chunks()[0].contains("2020-01-01"));
436    }
437
438    #[test]
439    fn images_are_rewritten_to_the_paths_they_are_actually_served_at() {
440        let url: SitemapUrl = SitemapUrl::new("/")
441            .extend_images(["/static/image/social/card.webp"])
442            .map_images(|image| image.replace("card.webp", "card.abc123.webp"));
443
444        let expected: Vec<String> = vec![String::from("/static/image/social/card.abc123.webp")];
445        let actual: Vec<String> = url.images().to_vec();
446        assert_eq!(expected, actual);
447    }
448
449    #[test]
450    fn images_are_resolved_against_the_site_origin() {
451        let urls: Vec<SitemapUrl> =
452            vec![SitemapUrl::new("/").extend_images(["/static/image/social/card.webp"])];
453        let set: SitemapSet =
454            build_sitemaps("https://www.example.com", &urls, None).expect("builds");
455
456        assert!(set.chunks()[0].contains("https://www.example.com/static/image/social/card.webp"));
457    }
458}