Skip to main content

feather_reader/
lexicon.rs

1//! Serde types for the `community.lexicon.rss.*` atproto record schemas.
2//!
3//! FeatherReader's defining bet is that a user's feed subscriptions, folders,
4//! saved items, and batched read-state live as records in their own atproto PDS
5//! under an **open, vendor-neutral community lexicon** (`community.lexicon.rss.*`)
6//! rather than in the app's database — portable across any reader that adopts
7//! the standard, not merely across FeatherReader instances.
8//!
9//! These types mirror the `community.lexicon.rss.*` schemas, authored in
10//! the Lexicon Community idiom (`createdAt`/`updatedAt` as ISO-8601 datetimes,
11//! `url`/`siteUrl`/`feedUrl` as URIs, `folder` as an `at://` strong ref). Each
12//! record carries its `$type` NSID so it round-trips against the atproto record
13//! shape returned by `com.atproto.repo.getRecord` / `listRecords`.
14//!
15//! Storage rules (never write these authoritatively to local SQLite):
16//! - [`Subscription`] — one followed feed. `com.atproto.repo.createRecord` on
17//!   subscribe; `deleteRecord` on unsubscribe. Source of truth for the follow list.
18//! - [`Folder`] — a lightweight named grouping (a feed lives in one folder).
19//! - [`Saved`] — a starred / save-for-later entry.
20//! - [`ReadState`] — the **batched** per-feed read cursor (one record per feed,
21//!   at a feed-derived rkey — never one record per article). Written by the
22//!   read-state flusher; see the caveats on that flush path in
23//!   [`crate::atproto`].
24
25use serde::{Deserialize, Serialize};
26
27/// NSID `$type` constants for the `community.lexicon.rss.*` record collections.
28///
29/// These double as the atproto **collection** NSIDs for `listRecords` /
30/// `createRecord` / `putRecord` calls.
31pub mod nsid {
32    /// `community.lexicon.rss.subscription` — one followed feed.
33    pub const SUBSCRIPTION: &str = "community.lexicon.rss.subscription";
34    /// `community.lexicon.rss.folder` — a named grouping of subscriptions.
35    pub const FOLDER: &str = "community.lexicon.rss.folder";
36    /// `community.lexicon.rss.saved` — a starred / save-for-later entry.
37    pub const SAVED: &str = "community.lexicon.rss.saved";
38    /// `community.lexicon.rss.readState` — batched per-feed read cursor.
39    pub const READ_STATE: &str = "community.lexicon.rss.readState";
40
41    /// `site.standard.publication` — a standard.site publication. NOT one of
42    /// ours: it is another project's lexicon, named here because it is the only
43    /// foreign collection this reader will accept as a subscribable feed.
44    pub const STANDARD_PUBLICATION: &str = "site.standard.publication";
45
46    /// `site.standard.document` — one standard.site article. Also not ours.
47    pub const STANDARD_DOCUMENT: &str = "site.standard.document";
48}
49
50/// Optional polling-cadence hint on a [`Subscription`]. Readers MAY honor or
51/// ignore it. Mirrors the lexicon's `knownValues` for `fetchHint`.
52///
53/// `knownValues` in atproto is an *open* enum — an unrecognized value MUST NOT
54/// break deserialization — so [`FetchHint::Other`] captures forward-compatible
55/// values a future reader might write.
56#[derive(Serialize, Deserialize, Clone, Debug, PartialEq, Eq)]
57#[serde(rename_all = "lowercase")]
58pub enum FetchHint {
59    /// Poll as close to realtime as the reader supports.
60    Realtime,
61    /// Poll roughly hourly.
62    Hourly,
63    /// Poll roughly daily.
64    Daily,
65    /// Poll roughly weekly.
66    Weekly,
67    /// An unrecognized (forward-compatible) hint value.
68    #[serde(untagged)]
69    Other(String),
70}
71
72/// Drop a `siteUrl` this reader would refuse to render, at the point a record
73/// crosses into the process.
74///
75/// Absent stays absent and a good URL is passed through trimmed, matching
76/// [`crate::net::safe_link`]'s handling of entry links — the same allow-list, so
77/// the two URL fields on a record cannot disagree about what a link is.
78fn de_scheme_checked<'de, D>(deserializer: D) -> Result<Option<String>, D::Error>
79where
80    D: serde::Deserializer<'de>,
81{
82    let raw = Option::<String>::deserialize(deserializer)?;
83    Ok(raw.as_deref().and_then(crate::net::safe_link))
84}
85
86/// `community.lexicon.rss.subscription` — a subscription to a syndication feed
87/// (RSS / Atom / JSON Feed). Record key: `tid`.
88///
89/// `url` + `createdAt` are required; everything else is optional.
90///
91/// ## Public feeds only (and the reserved `private` marker)
92///
93/// atproto PDS records are **public**: anyone can read them via unauthenticated
94/// `getRecord` / `listRecords` and off the firehose, and they are retained even
95/// after `deleteRecord`. A **private feed** (a Substack `…/feed/private/<token>`,
96/// a Patreon `?auth=…` feed, a Ghost members `?uuid=` feed, a private-podcast
97/// token feed, or any URL that carries a secret token / key / auth credential)
98/// has its *secret in the URL*, so writing that URL here would leak paid /
99/// members-only access to the whole network.
100///
101/// **Current decision: FeatherReader supports PUBLIC feeds only.** A private
102/// feed is *refused* at the add / import boundary (see
103/// [`crate::feed::classify_feed_privacy`]) — it is never fetched, never stored,
104/// and no record (redacted or otherwise) is ever written. The server therefore
105/// holds NO private secret, which keeps "your data lives in your public PDS"
106/// 100% honest. Consequently every [`Subscription`] record actually written
107/// carries a real, public feed `url`, and [`Subscription::private`] is **always omitted**.
108///
109/// The [`Subscription::private`] field is retained ONLY as a documented, forward-compatible
110/// **reserved marker** for the eventual migration once atproto ships
111/// **permissioned data / permission-sets** (early-proposal as of mid-2026,
112/// bluesky-social/proposals#94). At that point a private feed's secret can live
113/// in an owner-scoped, permission-gated collection and this record can reference
114/// it with `private: true`. Until then the field has **no runtime behavior** —
115/// nothing sets it and nothing branches on it.
116#[derive(Serialize, Deserialize, Clone, Debug, PartialEq, Eq)]
117pub struct Subscription {
118    /// The `$type` NSID discriminator; always [`nsid::SUBSCRIPTION`].
119    #[serde(rename = "$type", default = "subscription_type")]
120    pub r#type: String,
121
122    /// Canonical feed URL (the RSS/Atom/JSON Feed document). Required.
123    ///
124    /// Always a real, PUBLIC feed URL: private/secret-bearing feeds are refused
125    /// at the add boundary (see the type-level docs), so no record with a
126    /// withheld or redacted `url` is ever written.
127    pub url: String,
128
129    /// Display title; a reader MAY override from feed metadata.
130    #[serde(skip_serializing_if = "Option::is_none", default)]
131    pub title: Option<String>,
132
133    /// Human-facing site the feed belongs to.
134    ///
135    /// **Scheme-checked on the way in.** Any atproto client can write this field
136    /// into the user's repo, and the lexicon invites readers to render it as a
137    /// link, so a record fetched from the PDS is attacker-controlled input. The
138    /// `deserialize_with` below is the read-side counterpart to the write-side
139    /// vet in [`crate::repo`]: together they mean a `Subscription` that entered
140    /// this process from outside cannot be carrying a `javascript:` URL, whatever
141    /// it is later rendered into — an `href`, or an OPML `htmlUrl` we hand back
142    /// to the user as a file.
143    ///
144    /// **The READ side only.** A record built in-process rather than
145    /// deserialised does not pass through here — OPML import parses `htmlUrl`
146    /// out of XML by hand, and the manage form assigns the field directly.
147    /// Those are the write boundary's to vet, which is why both guards exist
148    /// rather than either one being sufficient.
149    ///
150    /// A rejected value becomes `None`, so it is omitted rather than emitted
151    /// empty; a consumer renders no link instead of a broken one.
152    ///
153    /// **Round-trip fidelity is deliberately lost.** Read a record holding a
154    /// hostile `siteUrl`, re-put it, and we write it back cleaned rather than
155    /// preserving what another client stored. That heals the user's repo
156    /// instead of propagating someone else's script URL — but it does mean a
157    /// `putRecord` following a read is not byte-identical to what was there,
158    /// and that is a decision, not an accident.
159    #[serde(
160        rename = "siteUrl",
161        skip_serializing_if = "Option::is_none",
162        default,
163        deserialize_with = "de_scheme_checked"
164    )]
165    pub site_url: Option<String>,
166
167    /// Optional `at://` strong ref to a [`Folder`] record.
168    #[serde(skip_serializing_if = "Option::is_none", default)]
169    pub folder: Option<String>,
170
171    /// Optional polling-cadence hint; readers MAY honor or ignore it.
172    #[serde(rename = "fetchHint", skip_serializing_if = "Option::is_none", default)]
173    pub fetch_hint: Option<FetchHint>,
174
175    /// **Reserved** — no runtime behavior today.
176    ///
177    /// FeatherReader currently supports public feeds only (private/secret-bearing
178    /// feeds are refused at the add boundary), so nothing sets this and every
179    /// written record omits it (`None`). It is kept as a documented,
180    /// forward-compatible seam for the eventual migration once atproto ships
181    /// permissioned data: at that point a private feed's secret can live in an
182    /// owner-scoped, permission-gated collection and this record can reference it
183    /// with `private: true`. See the type-level docs.
184    #[serde(skip_serializing_if = "Option::is_none", default)]
185    pub private: Option<bool>,
186
187    /// Record creation time (ISO-8601 datetime). Required.
188    #[serde(rename = "createdAt")]
189    pub created_at: String,
190}
191
192fn subscription_type() -> String {
193    nsid::SUBSCRIPTION.to_string()
194}
195
196impl Subscription {
197    /// Construct a minimal subscription with only the required fields.
198    pub fn new(url: impl Into<String>, created_at: impl Into<String>) -> Self {
199        Self {
200            r#type: nsid::SUBSCRIPTION.to_string(),
201            url: url.into(),
202            title: None,
203            site_url: None,
204            folder: None,
205            fetch_hint: None,
206            private: None,
207            created_at: created_at.into(),
208        }
209    }
210}
211
212/// `community.lexicon.rss.folder` — a named folder/grouping for subscriptions.
213/// Record key: `tid`.
214///
215/// `name` + `createdAt` are required; `position` is an optional sort hint.
216#[derive(Serialize, Deserialize, Clone, Debug, PartialEq, Eq)]
217pub struct Folder {
218    /// The `$type` NSID discriminator; always [`nsid::FOLDER`].
219    #[serde(rename = "$type", default = "folder_type")]
220    pub r#type: String,
221
222    /// Folder display name. Required.
223    pub name: String,
224
225    /// Optional sort hint among sibling folders (>= 0).
226    #[serde(skip_serializing_if = "Option::is_none", default)]
227    pub position: Option<u64>,
228
229    /// Record creation time (ISO-8601 datetime). Required.
230    #[serde(rename = "createdAt")]
231    pub created_at: String,
232}
233
234fn folder_type() -> String {
235    nsid::FOLDER.to_string()
236}
237
238impl Folder {
239    /// Construct a minimal folder with only the required fields.
240    pub fn new(name: impl Into<String>, created_at: impl Into<String>) -> Self {
241        Self {
242            r#type: nsid::FOLDER.to_string(),
243            name: name.into(),
244            position: None,
245            created_at: created_at.into(),
246        }
247    }
248}
249
250/// `community.lexicon.rss.saved` — an article kept for later (the reader's
251/// "star"). Record key: `tid`.
252///
253/// `url` + `createdAt` are required; the rest aid cross-reader dedup.
254#[derive(Serialize, Deserialize, Clone, Debug, PartialEq, Eq)]
255pub struct Saved {
256    /// The `$type` NSID discriminator; always [`nsid::SAVED`].
257    #[serde(rename = "$type", default = "saved_type")]
258    pub r#type: String,
259
260    /// The article/entry permalink. Required.
261    pub url: String,
262
263    /// Display title of the saved entry.
264    #[serde(skip_serializing_if = "Option::is_none", default)]
265    pub title: Option<String>,
266
267    /// Feed the entry came from (soft ref; may outlive the subscription).
268    #[serde(rename = "feedUrl", skip_serializing_if = "Option::is_none", default)]
269    pub feed_url: Option<String>,
270
271    /// Feed-native guid/id when present, for cross-reader dedup.
272    #[serde(rename = "entryId", skip_serializing_if = "Option::is_none", default)]
273    pub entry_id: Option<String>,
274
275    /// Record creation time (ISO-8601 datetime). Required.
276    #[serde(rename = "createdAt")]
277    pub created_at: String,
278}
279
280fn saved_type() -> String {
281    nsid::SAVED.to_string()
282}
283
284impl Saved {
285    /// Construct a minimal saved entry with only the required fields.
286    pub fn new(url: impl Into<String>, created_at: impl Into<String>) -> Self {
287        Self {
288            r#type: nsid::SAVED.to_string(),
289            url: url.into(),
290            title: None,
291            feed_url: None,
292            entry_id: None,
293            created_at: created_at.into(),
294        }
295    }
296}
297
298/// `community.lexicon.rss.readState` — a batched read high-water-mark for a
299/// single feed. Record key: `any`; the rkey is derived deterministically from the
300/// feed (a hash of the feed URL), so there is one record per feed with a stable
301/// key, NOT one record per article.
302///
303/// `feedUrl` + `updatedAt` are required; `readThrough` is OPTIONAL — it is a
304/// water-mark ("every entry seen/published `<=` this is read"), so it is written
305/// only once a real high-water-mark exists. Omitting it (rather than synthesizing
306/// a flush-time value) means a brand-new cursor asserts nothing about the backlog:
307/// only the explicit `readIds` mark entries read. The two capped id-sets carry
308/// out-of-order reads and explicit mark-unread exceptions.
309#[derive(Serialize, Deserialize, Clone, Debug, PartialEq, Eq)]
310pub struct ReadState {
311    /// The `$type` NSID discriminator; always [`nsid::READ_STATE`].
312    #[serde(rename = "$type", default = "read_state_type")]
313    pub r#type: String,
314
315    /// The feed this cursor covers. Required.
316    #[serde(rename = "feedUrl")]
317    pub feed_url: String,
318
319    /// High-water-mark: every entry with seen/published time <= this is READ.
320    /// **Optional** — omitted from the record when no local high-water-mark
321    /// exists yet, so a fresh cursor never implicitly marks the backlog read.
322    #[serde(
323        rename = "readThrough",
324        skip_serializing_if = "Option::is_none",
325        default
326    )]
327    pub read_through: Option<String>,
328
329    /// Entries newer than `readThrough` that are ALSO read (out-of-order reads).
330    /// Capped at 1000 by the lexicon; empty sets are omitted from the record.
331    #[serde(rename = "readIds", skip_serializing_if = "Vec::is_empty", default)]
332    pub read_ids: Vec<String>,
333
334    /// Entries older than `readThrough` explicitly kept UNREAD (mark-unread).
335    /// Capped at 1000 by the lexicon; empty sets are omitted from the record.
336    #[serde(rename = "unreadIds", skip_serializing_if = "Vec::is_empty", default)]
337    pub unread_ids: Vec<String>,
338
339    /// Last time this cursor was flushed (ISO-8601 datetime). Required. Intended
340    /// as the tie-breaker for cross-device merges (newest `updatedAt` wins);
341    /// a login-time reconcile that uses it is not implemented yet.
342    #[serde(rename = "updatedAt")]
343    pub updated_at: String,
344}
345
346fn read_state_type() -> String {
347    nsid::READ_STATE.to_string()
348}
349
350impl ReadState {
351    /// Maximum length of the `readIds` / `unreadIds` exception sets, per the
352    /// lexicon. The flusher enforces this cap before writing (see
353    /// `scheduler::cap`).
354    pub const MAX_IDS: usize = 1000;
355
356    /// Construct a minimal read cursor with only the required fields.
357    ///
358    /// `read_through` is optional: pass `None` for a cursor that has no local
359    /// high-water-mark yet, so the record omits `readThrough` entirely rather than
360    /// synthesizing a flush-time value that would mark the backlog read.
361    pub fn new(
362        feed_url: impl Into<String>,
363        read_through: Option<String>,
364        updated_at: impl Into<String>,
365    ) -> Self {
366        Self {
367            r#type: nsid::READ_STATE.to_string(),
368            feed_url: feed_url.into(),
369            read_through,
370            read_ids: Vec::new(),
371            unread_ids: Vec::new(),
372            updated_at: updated_at.into(),
373        }
374    }
375}
376
377#[cfg(test)]
378mod tests {
379    use super::*;
380    use serde_json::json;
381
382    /// A record written by some other client is attacker-controlled input.
383    ///
384    /// Asserted through `serde_json::from_str` rather than by calling the
385    /// deserialiser directly: the production path is a PDS fetch, and a test that
386    /// calls the helper would pass just as happily with `deserialize_with`
387    /// removed from the field.
388    #[test]
389    fn a_hostile_site_url_does_not_survive_deserialisation() {
390        for hostile in [
391            "javascript:alert(1)",
392            "data:text/html;base64,PHNjcmlwdD4=",
393            "vbscript:msgbox(1)",
394            "  javascript:alert(1)  ",
395            "not a url at all",
396        ] {
397            let json = serde_json::json!({
398                "$type": "community.lexicon.rss.subscription",
399                "url": "https://example.com/feed.xml",
400                "siteUrl": hostile,
401                "createdAt": "2026-01-01T00:00:00.000Z",
402            })
403            .to_string();
404            let sub: Subscription = serde_json::from_str(&json).expect("record should parse");
405            assert_eq!(
406                sub.site_url, None,
407                "{hostile:?} survived into a record this reader will re-publish and export"
408            );
409            assert_eq!(
410                sub.url, "https://example.com/feed.xml",
411                "the feed URL is not the field under test and must be untouched"
412            );
413        }
414    }
415
416    /// The check must not eat an ordinary record, and must normalise the way the
417    /// entry-link path already does.
418    #[test]
419    fn a_legitimate_site_url_survives_deserialisation() {
420        for (stored, expected) in [
421            ("https://example.com/blog", "https://example.com/blog"),
422            ("http://example.com/blog", "http://example.com/blog"),
423            ("  https://example.com/blog  ", "https://example.com/blog"),
424        ] {
425            let json = serde_json::json!({
426                "$type": "community.lexicon.rss.subscription",
427                "url": "https://example.com/feed.xml",
428                "siteUrl": stored,
429                "createdAt": "2026-01-01T00:00:00.000Z",
430            })
431            .to_string();
432            let sub: Subscription = serde_json::from_str(&json).expect("record should parse");
433            assert_eq!(sub.site_url.as_deref(), Some(expected));
434        }
435    }
436
437    /// An absent `siteUrl` stays absent — no empty string is invented, and the
438    /// `default` path must not trip over the custom deserialiser.
439    #[test]
440    fn an_absent_site_url_stays_absent() {
441        let json = serde_json::json!({
442            "$type": "community.lexicon.rss.subscription",
443            "url": "https://example.com/feed.xml",
444            "createdAt": "2026-01-01T00:00:00.000Z",
445        })
446        .to_string();
447        let sub: Subscription = serde_json::from_str(&json).expect("record should parse");
448        assert_eq!(sub.site_url, None);
449
450        // Explicit null is the same as absent, not an error.
451        let json = serde_json::json!({
452            "$type": "community.lexicon.rss.subscription",
453            "url": "https://example.com/feed.xml",
454            "siteUrl": serde_json::Value::Null,
455            "createdAt": "2026-01-01T00:00:00.000Z",
456        })
457        .to_string();
458        let sub: Subscription = serde_json::from_str(&json).expect("explicit null should parse");
459        assert_eq!(sub.site_url, None);
460    }
461
462    #[test]
463    fn subscription_round_trips_full_record() {
464        // Matches the atproto record shape returned by getRecord's `value`.
465        let value = json!({
466            "$type": "community.lexicon.rss.subscription",
467            "url": "https://example.com/feed.xml",
468            "title": "Example Blog",
469            "siteUrl": "https://example.com/",
470            "folder": "at://did:plc:abc123/community.lexicon.rss.folder/3kfolderrkey",
471            "fetchHint": "hourly",
472            "createdAt": "2026-07-12T00:00:00.000Z"
473        });
474
475        let sub: Subscription = serde_json::from_value(value.clone()).expect("deserialize");
476        assert_eq!(sub.r#type, nsid::SUBSCRIPTION);
477        assert_eq!(sub.url, "https://example.com/feed.xml");
478        assert_eq!(sub.title.as_deref(), Some("Example Blog"));
479        assert_eq!(sub.site_url.as_deref(), Some("https://example.com/"));
480        assert_eq!(sub.fetch_hint, Some(FetchHint::Hourly));
481
482        let back = serde_json::to_value(&sub).expect("serialize");
483        assert_eq!(back, value);
484    }
485
486    #[test]
487    fn subscription_minimal_omits_optional_fields() {
488        let sub = Subscription::new("https://example.com/feed.xml", "2026-07-12T00:00:00.000Z");
489        let back = serde_json::to_value(&sub).expect("serialize");
490        assert_eq!(
491            back,
492            json!({
493                "$type": "community.lexicon.rss.subscription",
494                "url": "https://example.com/feed.xml",
495                "createdAt": "2026-07-12T00:00:00.000Z"
496            })
497        );
498    }
499
500    #[test]
501    fn subscription_reserved_private_marker_omitted_by_default_but_round_trips() {
502        // Default construction never sets `private`; a public record omits it
503        // entirely (byte-for-byte unchanged from before the reserved field).
504        let public = Subscription::new("https://example.com/feed.xml", "2026-07-12T00:00:00.000Z");
505        assert_eq!(public.private, None);
506        let public_body = serde_json::to_value(&public).expect("serialize");
507        assert!(public_body.get("private").is_none());
508
509        // The reserved field is forward-compatible: if a future record ever
510        // carries `private: true`, it (de)serializes cleanly. Nothing in the
511        // current codebase sets it, but the seam must round-trip.
512        let mut future =
513            Subscription::new("https://example.com/feed.xml", "2026-07-12T00:00:00.000Z");
514        future.private = Some(true);
515        let back = serde_json::to_value(&future).expect("serialize");
516        assert_eq!(back["private"], serde_json::json!(true));
517        let parsed: Subscription = serde_json::from_value(back).expect("deserialize");
518        assert_eq!(parsed.private, Some(true));
519    }
520
521    #[test]
522    fn fetch_hint_open_enum_accepts_unknown() {
523        let sub: Subscription = serde_json::from_value(json!({
524            "url": "https://example.com/feed.xml",
525            "fetchHint": "every-15-min",
526            "createdAt": "2026-07-12T00:00:00.000Z"
527        }))
528        .expect("deserialize");
529        assert_eq!(
530            sub.fetch_hint,
531            Some(FetchHint::Other("every-15-min".to_string()))
532        );
533        // $type defaults in when the record value omits it.
534        assert_eq!(sub.r#type, nsid::SUBSCRIPTION);
535    }
536
537    #[test]
538    fn folder_round_trips() {
539        let value = json!({
540            "$type": "community.lexicon.rss.folder",
541            "name": "Tech",
542            "position": 2,
543            "createdAt": "2026-07-12T00:00:00.000Z"
544        });
545        let folder: Folder = serde_json::from_value(value.clone()).expect("deserialize");
546        assert_eq!(folder.name, "Tech");
547        assert_eq!(folder.position, Some(2));
548        assert_eq!(serde_json::to_value(&folder).expect("serialize"), value);
549    }
550
551    #[test]
552    fn saved_round_trips() {
553        let value = json!({
554            "$type": "community.lexicon.rss.saved",
555            "url": "https://example.com/post/1",
556            "title": "A kept post",
557            "feedUrl": "https://example.com/feed.xml",
558            "entryId": "tag:example.com,2026:1",
559            "createdAt": "2026-07-12T00:00:00.000Z"
560        });
561        let saved: Saved = serde_json::from_value(value.clone()).expect("deserialize");
562        assert_eq!(saved.url, "https://example.com/post/1");
563        assert_eq!(
564            saved.feed_url.as_deref(),
565            Some("https://example.com/feed.xml")
566        );
567        assert_eq!(saved.entry_id.as_deref(), Some("tag:example.com,2026:1"));
568        assert_eq!(serde_json::to_value(&saved).expect("serialize"), value);
569    }
570
571    #[test]
572    fn read_state_round_trips_with_id_sets() {
573        let value = json!({
574            "$type": "community.lexicon.rss.readState",
575            "feedUrl": "https://example.com/feed.xml",
576            "readThrough": "2026-07-12T00:00:00.000Z",
577            "readIds": ["entry-a", "entry-b"],
578            "unreadIds": ["entry-c"],
579            "updatedAt": "2026-07-12T01:00:00.000Z"
580        });
581        let rs: ReadState = serde_json::from_value(value.clone()).expect("deserialize");
582        assert_eq!(rs.feed_url, "https://example.com/feed.xml");
583        assert_eq!(rs.read_through.as_deref(), Some("2026-07-12T00:00:00.000Z"));
584        assert_eq!(rs.read_ids, vec!["entry-a", "entry-b"]);
585        assert_eq!(rs.unread_ids, vec!["entry-c"]);
586        assert_eq!(serde_json::to_value(&rs).expect("serialize"), value);
587    }
588
589    #[test]
590    fn read_state_minimal_omits_empty_id_sets() {
591        let rs = ReadState::new(
592            "https://example.com/feed.xml",
593            Some("2026-07-12T00:00:00.000Z".to_string()),
594            "2026-07-12T01:00:00.000Z",
595        );
596        let back = serde_json::to_value(&rs).expect("serialize");
597        assert_eq!(
598            back,
599            json!({
600                "$type": "community.lexicon.rss.readState",
601                "feedUrl": "https://example.com/feed.xml",
602                "readThrough": "2026-07-12T00:00:00.000Z",
603                "updatedAt": "2026-07-12T01:00:00.000Z"
604            })
605        );
606    }
607
608    #[test]
609    fn read_state_omits_read_through_when_none() {
610        // A brand-new cursor with no high-water-mark must NOT synthesize one:
611        // `readThrough` is absent entirely so the backlog is not implicitly read.
612        let rs = ReadState::new(
613            "https://example.com/feed.xml",
614            None,
615            "2026-07-12T01:00:00.000Z",
616        );
617        let back = serde_json::to_value(&rs).expect("serialize");
618        assert!(back.get("readThrough").is_none());
619        assert_eq!(
620            back,
621            json!({
622                "$type": "community.lexicon.rss.readState",
623                "feedUrl": "https://example.com/feed.xml",
624                "updatedAt": "2026-07-12T01:00:00.000Z"
625            })
626        );
627        // And a record without readThrough round-trips back to None.
628        let parsed: ReadState = serde_json::from_value(back).expect("deserialize");
629        assert_eq!(parsed.read_through, None);
630    }
631}
632
633/// Deterministic orderings for the reader's record lists.
634///
635/// These live here, beside the types, and are used by **both** the sidecar
636/// client and the Rust-native one. That is deliberate: the two clients coexist
637/// until cutover, and a divergence in ordering would not be a subtle bug — it
638/// would reorder the user's feed list the moment the implementation swapped, in
639/// a way no test comparing the clients' *data* would catch.
640pub mod sort {
641    use super::{Folder, Saved, Subscription};
642    use std::cmp::Ordering;
643
644    /// Subscriptions: display title (case-insensitive), then URL, then rkey.
645    ///
646    /// An untitled feed sorts by its URL, so it lands where a reader would look
647    /// for it rather than at one end of the list.
648    pub fn subscriptions(
649        (a_key, a): &(String, Subscription),
650        (b_key, b): &(String, Subscription),
651    ) -> Ordering {
652        let a_title = a.title.as_deref().unwrap_or(&a.url).to_lowercase();
653        let b_title = b.title.as_deref().unwrap_or(&b.url).to_lowercase();
654        a_title
655            .cmp(&b_title)
656            .then_with(|| a.url.cmp(&b.url))
657            .then_with(|| a_key.cmp(b_key))
658    }
659
660    /// Folders: `position` (the lexicon's sort hint; unset sorts LAST), then
661    /// name (case-insensitive), then rkey.
662    pub fn folders((a_key, a): &(String, Folder), (b_key, b): &(String, Folder)) -> Ordering {
663        a.position
664            .unwrap_or(u64::MAX)
665            .cmp(&b.position.unwrap_or(u64::MAX))
666            .then_with(|| a.name.to_lowercase().cmp(&b.name.to_lowercase()))
667            .then_with(|| a_key.cmp(b_key))
668    }
669
670    /// Saved entries: newest first by `createdAt` (RFC 3339 sorts
671    /// lexicographically), then rkey ascending.
672    pub fn saved((a_key, a): &(String, Saved), (b_key, b): &(String, Saved)) -> Ordering {
673        b.created_at
674            .cmp(&a.created_at)
675            .then_with(|| a_key.cmp(b_key))
676    }
677}
678
679#[cfg(test)]
680mod sort_tests {
681    use super::sort;
682    use super::{Folder, Saved, Subscription};
683
684    fn sub(rkey: &str, url: &str, title: Option<&str>) -> (String, Subscription) {
685        let mut s = Subscription::new(url, "2026-01-01T00:00:00Z");
686        s.title = title.map(str::to_string);
687        (rkey.to_string(), s)
688    }
689
690    fn folder(rkey: &str, name: &str, position: Option<u64>) -> (String, Folder) {
691        let mut f = Folder::new(name, "2026-01-01T00:00:00Z");
692        f.position = position;
693        (rkey.to_string(), f)
694    }
695
696    fn saved(rkey: &str, url: &str, created_at: &str) -> (String, Saved) {
697        (rkey.to_string(), Saved::new(url, created_at))
698    }
699
700    fn order<T>(
701        mut items: Vec<(String, T)>,
702        cmp: fn(&(String, T), &(String, T)) -> std::cmp::Ordering,
703    ) -> Vec<String> {
704        items.sort_by(cmp);
705        items.into_iter().map(|(k, _)| k).collect()
706    }
707
708    /// Title first, and case must NOT split the alphabet.
709    #[test]
710    fn subscriptions_sort_by_title_case_insensitively() {
711        let items = vec![
712            sub("r1", "https://z.example/f", Some("banana")),
713            sub("r2", "https://a.example/f", Some("Apple")),
714            sub("r3", "https://m.example/f", Some("cherry")),
715        ];
716        assert_eq!(order(items, sort::subscriptions), ["r2", "r1", "r3"]);
717    }
718
719    /// An UNTITLED feed sorts by its URL, so it lands where a reader would look
720    /// rather than being bunched at one end.
721    #[test]
722    fn an_untitled_subscription_sorts_by_its_url() {
723        let items = vec![
724            sub("r1", "https://zebra.example/f", Some("aardvark")),
725            sub("r2", "https://bison.example/f", None),
726        ];
727        assert_eq!(order(items, sort::subscriptions), ["r1", "r2"]);
728    }
729
730    /// Equal titles fall to URL, then to rkey — so the order is TOTAL and a
731    /// re-read cannot shuffle the list.
732    #[test]
733    fn subscriptions_break_ties_by_url_then_rkey() {
734        let items = vec![
735            sub("r2", "https://b.example/f", Some("same")),
736            sub("r1", "https://b.example/f", Some("same")),
737            sub("r3", "https://a.example/f", Some("same")),
738        ];
739        assert_eq!(order(items, sort::subscriptions), ["r3", "r1", "r2"]);
740    }
741
742    /// `position` is the lexicon's sort hint; an UNSET one sorts last rather
743    /// than first, which `unwrap_or(0)` would have got backwards.
744    #[test]
745    fn folders_sort_by_position_with_unset_last() {
746        let items = vec![
747            folder("r1", "zulu", None),
748            folder("r2", "alpha", Some(10)),
749            folder("r3", "bravo", Some(2)),
750        ];
751        assert_eq!(order(items, sort::folders), ["r3", "r2", "r1"]);
752    }
753
754    #[test]
755    fn folders_break_ties_by_name_then_rkey() {
756        let items = vec![
757            folder("r2", "Beta", Some(1)),
758            folder("r1", "alpha", Some(1)),
759            folder("r3", "alpha", Some(1)),
760        ];
761        assert_eq!(order(items, sort::folders), ["r1", "r3", "r2"]);
762    }
763
764    /// Saved entries read NEWEST FIRST -- the one ordering here that is
765    /// descending, and the easiest to get backwards.
766    #[test]
767    fn saved_entries_are_newest_first() {
768        let items = vec![
769            saved("r1", "https://a.example/x", "2026-01-01T00:00:00Z"),
770            saved("r2", "https://b.example/x", "2026-06-01T00:00:00Z"),
771            saved("r3", "https://c.example/x", "2026-03-01T00:00:00Z"),
772        ];
773        assert_eq!(order(items, sort::saved), ["r2", "r3", "r1"]);
774    }
775
776    /// Same instant: rkey ASCENDING, even though the timestamp is descending.
777    #[test]
778    fn saved_entries_break_ties_by_ascending_rkey() {
779        let items = vec![
780            saved("r3", "https://c.example/x", "2026-01-01T00:00:00Z"),
781            saved("r1", "https://a.example/x", "2026-01-01T00:00:00Z"),
782            saved("r2", "https://b.example/x", "2026-01-01T00:00:00Z"),
783        ];
784        assert_eq!(order(items, sort::saved), ["r1", "r2", "r3"]);
785    }
786}