Skip to main content

data_beans/aux/
feature_names.rs

1//! Feature-name kind + canonicalizer hooks for multi-file data
2//! alignment.
3//!
4//! NOT to be confused with the sibling [`crate::aux::feature_rows`]. That module
5//! defines the row-name **grammar** faba's producers emit
6//! (`{unit}/{modality}/{subunit}/{channel}`); this one **canonicalizes** a name
7//! that already exists, so the same gene or locus matches across files that
8//! spell it differently. Reach for `feature_rows` to build or split a row, and
9//! for this module to decide whether two spellings are the same feature.
10//!
11//! Loaders that union rows across multiple sparse backends
12//! ([`crate::aux::data_loading::read_data_on_shared_rows`]) can opt into
13//! generous matching by passing a [`FeatureNameKind`] — same row-name
14//! canonicalization machinery used by `senna marker` /
15//! `FeaturePairGraph::from_edge_list` (via
16//! [`legume_numeric::matrix::membership::GeneIndexResolver`]) but plumbed at the
17//! [`crate::sparse_io_vector::SparseIoVec`] level so the row
18//! intersection itself sees aligned names.
19//!
20//! The two flavors cover what biology pipelines see in practice:
21//!
22//! - [`FeatureNameKind::Gene`] for gene rows in scRNA / spatial-RNA
23//!   data, where the same gene shows up as `TGFB1`, `ENSG00000105329`,
24//!   or `ENSG00000105329_TGFB1` across cohorts;
25//! - [`FeatureNameKind::Locus`] for chromosome-coordinate rows in ATAC
26//!   / chickpea-style data, where `chr1:1000-2000`, `chr1_1000_2000`,
27//!   and `1:1000-2000` should all resolve to the same peak.
28
29use std::sync::Arc;
30
31use crate::sparse_io_vector::RowNameCanonicalizer;
32use genomic_data::coordinates::{self, chr_stripped, PeakCoord};
33use rustc_hash::FxHashMap as HashMap;
34
35/// Per-name canonicalization rule for cross-backend row alignment.
36/// Concrete strategy only — no "request" variants. Callers that want
37/// auto-detection pass [`None`] (or whatever wrapping enum they choose)
38/// and call [`FeatureNameKind::auto_detect`] once row names are in hand.
39#[derive(Clone, Debug, Default, PartialEq, Eq)]
40pub enum FeatureNameKind {
41    /// Strict string match — no canonicalization. Default.
42    #[default]
43    Exact,
44    /// Gene-symbol rule: register every `delim`-split component as an
45    /// alias of the full name. `ENSG00000105329_TGFB1` and `TGFB1`
46    /// resolve to the same row.
47    Gene { delim: char },
48    /// Genomic-locus rule. Normalizes formats (`chr1:1000-2000`,
49    /// `1:1000-2000`, `chr1_1000_2000` → `1_1000_2000`). If
50    /// `merge_overlapping`, intervals that overlap on the same
51    /// chromosome additionally collapse into one cluster
52    /// (`chr1:1-20` ∪ `chr1:15-30` → `1_1_30`). Useful for ATAC peak
53    /// sets called independently across datasets.
54    Locus { merge_overlapping: bool },
55    /// Heterogeneous axis: dispatch per row name. Names that parse as
56    /// loci go through [`FeatureNameKind::Locus`] with overlap merging;
57    /// names with `_` use [`FeatureNameKind::Gene`]; the rest pass
58    /// through. Picked automatically when [`auto_detect`] finds both
59    /// signatures in the same axis (e.g. paired RNA + ATAC union).
60    Mixed,
61}
62
63impl FeatureNameKind {
64    /// Canonicalize a single name under this kind's per-name rule.
65    /// [`Locus { merge_overlapping: true }`] and [`Mixed`] only describe
66    /// the per-name part here (format normalization for loci, last-token
67    /// split for gene-style); the global cluster step lives in
68    /// [`build_locus_overlap_canonical_map`] and is installed by
69    /// [`build_canonicalizer`].
70    pub fn canonicalize(&self, name: &str) -> Box<str> {
71        match self {
72            FeatureNameKind::Exact => name.into(),
73            FeatureNameKind::Gene { delim } => gene_canonicalize(name, *delim),
74            FeatureNameKind::Locus { .. } => locus_key(name).unwrap_or_else(|| name.into()),
75            FeatureNameKind::Mixed => mixed_canonicalize(name),
76        }
77    }
78
79    /// True iff this kind is [`Exact`] — no canonicalizer needed.
80    pub fn is_exact(&self) -> bool {
81        matches!(self, FeatureNameKind::Exact)
82    }
83
84    /// True iff installing the canonicalizer requires peeking every row
85    /// name across all backends first (to build the locus-overlap cluster
86    /// map). Loaders branch on this.
87    pub fn needs_global_pass(&self) -> bool {
88        matches!(
89            self,
90            FeatureNameKind::Locus {
91                merge_overlapping: true
92            } | FeatureNameKind::Mixed
93        )
94    }
95
96    /// Sniff `names` and pick the right kind. Tallies:
97    /// `n_locus` = count where `parse_locus` matches; `n_gene_like` =
98    /// count of remaining names that contain `_` (loci with `_`
99    /// separators are not double-counted as gene-like). Decision:
100    /// both ≥ 10% → [`Mixed`]; else loci ≥ 50% →
101    /// `Locus { merge_overlapping: true }`; else gene-like ≥ 50% →
102    /// `Gene { delim: '_' }`; else [`Exact`].
103    pub fn auto_detect(names: &[Box<str>]) -> Self {
104        let n = names.len();
105        if n == 0 {
106            return Self::Exact;
107        }
108        let mut n_locus = 0usize;
109        let mut n_gene_like = 0usize;
110        for name in names {
111            if parse_locus(name).is_some() {
112                n_locus += 1;
113            } else if name.contains('_') {
114                n_gene_like += 1;
115            }
116        }
117        let pct_locus = n_locus as f32 / n as f32;
118        let pct_gene = n_gene_like as f32 / n as f32;
119        if pct_locus >= 0.10 && pct_gene >= 0.10 {
120            Self::Mixed
121        } else if pct_locus >= 0.50 {
122            Self::Locus {
123                merge_overlapping: true,
124            }
125        } else if pct_gene >= 0.50 {
126            Self::Gene { delim: '_' }
127        } else {
128            Self::Exact
129        }
130    }
131
132    /// The one kind to install for a set of files that were each sniffed
133    /// with [`auto_detect`](Self::auto_detect) on their own.
134    ///
135    /// Sniffing the POOLED names does not work: the signature usually lives
136    /// on one side only. A raw `ENSG_SYM` cohort pooled with a reference
137    /// already on the bare-symbol axis (a carried `pb_reference`, a
138    /// symbol-keyed panel) leaves the gene-like share under half, the pair
139    /// sniffs as `Exact`, and every gene becomes two rows. Canonicalizing
140    /// under `Gene` is a no-op for names lacking the delimiter, so adopting
141    /// the informative side is safe for both.
142    ///
143    /// Gene-style and locus-style files together dispatch per name
144    /// (`Mixed`), which is what `auto_detect` would pick on one axis holding
145    /// both; `Mixed` anywhere stays `Mixed`; all-`Exact` stays `Exact`.
146    #[must_use]
147    pub fn reconcile(kinds: &[FeatureNameKind]) -> FeatureNameKind {
148        if kinds.iter().any(|k| matches!(k, FeatureNameKind::Mixed)) {
149            return FeatureNameKind::Mixed;
150        }
151        let gene = kinds
152            .iter()
153            .find(|k| matches!(k, FeatureNameKind::Gene { .. }));
154        let locus = kinds
155            .iter()
156            .find(|k| matches!(k, FeatureNameKind::Locus { .. }));
157        match (gene, locus) {
158            (Some(_), Some(_)) => FeatureNameKind::Mixed,
159            _ => gene.or(locus).cloned().unwrap_or(FeatureNameKind::Exact),
160        }
161    }
162
163    /// Build a `RowNameCanonicalizer` suitable for
164    /// [`SparseIoVec::with_row_canonicalizer`]. Returns `None` for
165    /// [`FeatureNameKind::Exact`] so callers don't install a no-op
166    /// closure. **Does not** handle the LocusOverlap global step — for
167    /// that, use [`build_locus_overlap_canonicalizer`].
168    pub fn into_canonicalizer(self) -> Option<RowNameCanonicalizer> {
169        if self.is_exact() {
170            return None;
171        }
172        Some(Arc::new(move |name: &str| self.canonicalize(name)))
173    }
174}
175
176/// Parse a row name as `(chr, start, end)` under the shared locus grammar
177/// ([`coordinates::parse_interval`]): `chr1:1000-2000`, `chr1_1000_2000`,
178/// `chr1-1000-2000`, `1_1000_2000`. The chromosome comes back with its
179/// `chr` prefix dropped and its case kept (`chrX` and `X` match, `x` does
180/// not), and contig names carrying `_` or `-` stay whole. Returns `None`
181/// for anything that doesn't match; those names pass through the overlap
182/// pass untouched.
183pub fn parse_locus(name: &str) -> Option<(Box<str>, u64, u64)> {
184    let l = coordinates::parse_interval(name)?;
185    Some((chr_stripped(&l.chr).into(), l.start as u64, l.end as u64))
186}
187
188/// Canonical key of one locus (`chrX:0-100` becomes `X_0_100`), or
189/// `None` when `name` is not a locus.
190fn locus_key(name: &str) -> Option<Box<str>> {
191    coordinates::parse_interval(name).map(|l| l.locus_key())
192}
193
194/// Per-name rule of [`FeatureNameKind::Mixed`]: locus key, else the gene
195/// rule for `_` names, else the name unchanged.
196fn mixed_canonicalize(name: &str) -> Box<str> {
197    if let Some(key) = locus_key(name) {
198        key
199    } else if name.contains('_') {
200        gene_canonicalize(name, '_')
201    } else {
202        name.into()
203    }
204}
205
206/// Build the overlap-merge canonical map from a flat list of row names
207/// across all input backends. Names that parse as `(chr, start, end)`
208/// are grouped per chromosome, sorted by start, and clustered by
209/// transitive overlap (any interval whose start falls before the
210/// running cluster's max end). The cluster canonical is
211/// `{chr}_{min_start}_{max_end}` so every member name maps to a single
212/// well-defined string.
213///
214/// Names that fail to parse are not entered into the map; the caller
215/// falls back to the per-name rule for those.
216pub fn build_locus_overlap_canonical_map(names: &[Box<str>]) -> HashMap<Box<str>, Box<str>> {
217    let n = names.len();
218    let parsed: Vec<Option<(Box<str>, u64, u64)>> = names.iter().map(|n| parse_locus(n)).collect();
219
220    // Bucket valid indices by chromosome.
221    let mut by_chr: HashMap<Box<str>, Vec<usize>> = HashMap::default();
222    for (i, p) in parsed.iter().enumerate() {
223        if let Some((chr, _, _)) = p {
224            by_chr.entry(chr.clone()).or_default().push(i);
225        }
226    }
227
228    // Union-find with path compression.
229    let mut parent: Vec<usize> = (0..n).collect();
230    fn find(p: &mut [usize], mut x: usize) -> usize {
231        while p[x] != x {
232            let g = p[p[x]];
233            p[x] = g;
234            x = g;
235        }
236        x
237    }
238
239    // Per chr: sort by start, sweep, union anything overlapping the running cluster.
240    let mut cluster_extent: HashMap<usize, (u64, u64)> = HashMap::default();
241    for (_, mut idxs) in by_chr {
242        idxs.sort_by_key(|&i| parsed[i].as_ref().map(|p| p.1).unwrap_or(0));
243        let mut current_root: Option<usize> = None;
244        let mut current_min_start: u64 = 0;
245        let mut current_max_end: u64 = 0;
246        for i in idxs {
247            let (_, s, e) = parsed[i].as_ref().unwrap();
248            match current_root {
249                Some(root) if *s < current_max_end => {
250                    let ra = find(&mut parent, root);
251                    let rb = find(&mut parent, i);
252                    if ra != rb {
253                        parent[rb] = ra;
254                    }
255                    current_max_end = current_max_end.max(*e);
256                    cluster_extent
257                        .insert(find(&mut parent, i), (current_min_start, current_max_end));
258                }
259                _ => {
260                    current_root = Some(i);
261                    current_min_start = *s;
262                    current_max_end = *e;
263                    cluster_extent.insert(i, (*s, *e));
264                }
265            }
266        }
267    }
268
269    // Build name → canonical map: the cluster extent under `locus_key`.
270    let mut out: HashMap<Box<str>, Box<str>> = HashMap::default();
271    for (i, p) in parsed.iter().enumerate() {
272        if let Some((chr, _, _)) = p {
273            let root = find(&mut parent, i);
274            let (mn, mx) = cluster_extent.get(&root).copied().unwrap_or((0, 0));
275            let cluster = PeakCoord {
276                chr: chr.clone(),
277                start: mn as i64,
278                end: mx as i64,
279            };
280            out.insert(names[i].clone(), cluster.locus_key());
281        }
282    }
283    out
284}
285
286/// Build a `RowNameCanonicalizer` for [`FeatureNameKind::LocusOverlap`].
287/// `names` should be the concatenation of every input backend's row
288/// names (in any order). The returned canonicalizer does
289/// `map.get(name).cloned()` first, falling back to per-name
290/// locus canonical for names outside the map.
291pub fn build_locus_overlap_canonicalizer(names: &[Box<str>]) -> RowNameCanonicalizer {
292    let map = Arc::new(build_locus_overlap_canonical_map(names));
293    Arc::new(move |name: &str| {
294        map.get(name)
295            .cloned()
296            .or_else(|| locus_key(name))
297            .unwrap_or_else(|| name.into())
298    })
299}
300
301/// Per-name dispatcher for **mixed-kind** axes (e.g. multiome with peaks
302/// ∪ genes in one feature axis). For each name:
303///   • parses as `(chr, start, end)` → LocusOverlap canonical
304///     (cluster representative from `names`).
305///   • contains `_` → gene rule: last token after the rightmost `_`.
306///   • else → passthrough.
307///
308/// Use this when the auto-detector sees significant evidence of BOTH
309/// loci and gene-style names in the same axis.
310pub fn build_mixed_kind_canonicalizer(names: &[Box<str>]) -> RowNameCanonicalizer {
311    let map = Arc::new(build_locus_overlap_canonical_map(names));
312    Arc::new(move |name: &str| {
313        map.get(name)
314            .cloned()
315            .unwrap_or_else(|| mixed_canonicalize(name))
316    })
317}
318
319/// Gene-symbol canonicalization with Cell Ranger feature-type suffix
320/// awareness. 10x Cell Ranger HDF5 row names commonly arrive as
321/// `ENSG..._SYMBOL_<feature_type>` where the third component is a
322/// sanitized `feature_type` tag (e.g. `Gene` for `Gene Expression`).
323/// A naive `rsplit(delim).next()` would return that constant tag,
324/// canonicalizing *every* row to the same string and collapsing the
325/// row intersection to one global key. Strip the known tag suffix
326/// first so the actual symbol becomes the rsplit target.
327fn gene_canonicalize(name: &str, delim: char) -> Box<str> {
328    let stripped = strip_feature_type_suffix(name, delim);
329    stripped.rsplit(delim).next().unwrap_or(stripped).into()
330}
331
332/// Cell Ranger sanitizes `features/feature_type` into the row name as
333/// the trailing component. Strip the known tags so the actual gene
334/// symbol becomes the rsplit target. Conservative list — only the
335/// shapes we've actually seen in the wild — so an unknown tag falls
336/// through untouched rather than corrupting a real symbol.
337fn strip_feature_type_suffix(name: &str, delim: char) -> &str {
338    // Names come pre-sanitized in different ways depending on the
339    // producer (Cell Ranger's own h5, scanpy/anndata exports, R-side
340    // tools), so accept both `_Gene` and `_Gene_Expression` plus the
341    // common companion tags.
342    const TAGS: &[&str] = &[
343        "Gene_Expression",
344        "Gene",
345        "Antibody_Capture",
346        "CRISPR_Guide_Capture",
347        "Multiplexing_Capture",
348        "Custom",
349        "Peaks",
350    ];
351    for tag in TAGS {
352        // Only strip if the suffix sits behind `delim` (otherwise we'd
353        // mangle a real symbol that happens to end in "Gene").
354        let mut suffix = String::with_capacity(tag.len() + 1);
355        suffix.push(delim);
356        suffix.push_str(tag);
357        if let Some(rest) = name.strip_suffix(suffix.as_str()) {
358            return rest;
359        }
360    }
361    name
362}
363
364/// Clap-facing spelling of [`FeatureNameKind`].
365///
366/// The rule and the flag that selects it belong together: every crate that
367/// aligns feature names across files exposes the same `--feature-name-kind`
368/// vocabulary, so `senna` and anything after it agree on what
369/// `gene` or `locus` means without each inventing a local rule.
370#[derive(clap::ValueEnum, Clone, Debug, Default, serde::Serialize, serde::Deserialize)]
371#[serde(rename_all = "kebab-case")]
372pub enum FeatureNameKindArg {
373    #[default]
374    Auto,
375    Exact,
376    Gene,
377    Locus,
378    LocusOverlap,
379    Mixed,
380}
381
382impl FeatureNameKindArg {
383    /// Resolve to a concrete [`FeatureNameKind`], defaulting `Auto` to
384    /// `Gene { delim: '_' }` — the standard for gene-keyed pre-train
385    /// inputs (bge / fne / topic-family dictionaries).
386    pub fn resolve_or_gene(&self) -> FeatureNameKind {
387        Option::<FeatureNameKind>::from(self.clone())
388            .unwrap_or(FeatureNameKind::Gene { delim: '_' })
389    }
390}
391
392impl From<FeatureNameKindArg> for Option<FeatureNameKind> {
393    fn from(arg: FeatureNameKindArg) -> Self {
394        match arg {
395            FeatureNameKindArg::Auto => None,
396            FeatureNameKindArg::Exact => Some(FeatureNameKind::Exact),
397            FeatureNameKindArg::Gene => Some(FeatureNameKind::Gene { delim: '_' }),
398            FeatureNameKindArg::Locus => Some(FeatureNameKind::Locus {
399                merge_overlapping: false,
400            }),
401            FeatureNameKindArg::LocusOverlap => Some(FeatureNameKind::Locus {
402                merge_overlapping: true,
403            }),
404            FeatureNameKindArg::Mixed => Some(FeatureNameKind::Mixed),
405        }
406    }
407}
408
409#[cfg(test)]
410#[path = "feature_names_tests.rs"]
411mod feature_names_tests;
412
413#[cfg(test)]
414mod tests {
415    use super::*;
416
417    #[test]
418    fn exact_passthrough() {
419        let k = FeatureNameKind::Exact;
420        assert_eq!(
421            k.canonicalize("ENSG00000000003_TSPAN6").as_ref(),
422            "ENSG00000000003_TSPAN6"
423        );
424        assert!(k.is_exact());
425        assert!(k.into_canonicalizer().is_none());
426    }
427
428    #[test]
429    fn gene_takes_last_underscore_component() {
430        let k = FeatureNameKind::Gene { delim: '_' };
431        assert_eq!(k.canonicalize("ENSG00000000003_TSPAN6").as_ref(), "TSPAN6");
432        // Symbol-only inputs survive unchanged.
433        assert_eq!(k.canonicalize("TSPAN6").as_ref(), "TSPAN6");
434        // Unknown trailing tokens still get rsplit — caller's responsibility
435        // to pick a sensible delim if a non-feature-type trailing token matters.
436        assert_eq!(k.canonicalize("A_B_C").as_ref(), "C");
437        assert!(!k.is_exact());
438        assert!(k.into_canonicalizer().is_some());
439    }
440
441    #[test]
442    fn gene_strips_cell_ranger_feature_type_suffix() {
443        let k = FeatureNameKind::Gene { delim: '_' };
444        // 10x Cell Ranger HDF5: `ENSG..._SYMBOL_Gene`. Trailing `_Gene`
445        // would otherwise collapse every gene to the literal "Gene".
446        assert_eq!(
447            k.canonicalize("ENSG00000187634_SAMD11_Gene").as_ref(),
448            "SAMD11"
449        );
450        // Full `Gene_Expression` tag variant.
451        assert_eq!(
452            k.canonicalize("ENSG00000187634_SAMD11_Gene_Expression")
453                .as_ref(),
454            "SAMD11"
455        );
456        // A real gene whose name happens to end in "Gene" *without* the
457        // delimiter shouldn't be stripped (no underscore in front of "Gene").
458        assert_eq!(k.canonicalize("FakeGene").as_ref(), "FakeGene");
459    }
460
461    #[test]
462    fn locus_strips_chr_and_folds_separators() {
463        let k = FeatureNameKind::Locus {
464            merge_overlapping: false,
465        };
466        assert_eq!(k.canonicalize("chr1:1000-2000").as_ref(), "1_1000_2000");
467        assert_eq!(k.canonicalize("1_1000_2000").as_ref(), "1_1000_2000");
468        assert_eq!(k.canonicalize("ChrX:5000-6000").as_ref(), "X_5000_6000");
469    }
470
471    // -- Genomic region parsing edge cases ----------------------------------
472
473    #[test]
474    fn parse_locus_accepts_common_formats() {
475        // colon-dash, bare chromosome, underscore-separated, chrX caps.
476        assert_eq!(
477            parse_locus("chr1:1000-2000"),
478            Some(("1".into(), 1000, 2000))
479        );
480        assert_eq!(parse_locus("1:1000-2000"), Some(("1".into(), 1000, 2000)));
481        assert_eq!(
482            parse_locus("chr1_1000_2000"),
483            Some(("1".into(), 1000, 2000))
484        );
485        assert_eq!(
486            parse_locus("CHR1:1000-2000"),
487            Some(("1".into(), 1000, 2000))
488        );
489        assert_eq!(
490            parse_locus("chrX:5000-6000"),
491            Some(("X".into(), 5000, 6000))
492        );
493        assert_eq!(parse_locus("chrMT:1-100"), Some(("MT".into(), 1, 100)));
494        assert_eq!(
495            parse_locus("chr1-1000-2000"),
496            Some(("1".into(), 1000, 2000))
497        );
498    }
499
500    #[test]
501    fn parse_locus_keeps_contig_names_with_separators() {
502        assert_eq!(
503            parse_locus("chrUn_CTG1v1:0-100"),
504            Some(("Un_CTG1v1".into(), 0, 100))
505        );
506        assert_eq!(
507            parse_locus("chr1_CTG2_random_5_10"),
508            Some(("1_CTG2_random".into(), 5, 10))
509        );
510        // The canonical key parses back to the same locus.
511        assert_eq!(
512            parse_locus("Un_CTG1v1_0_100"),
513            Some(("Un_CTG1v1".into(), 0, 100))
514        );
515    }
516
517    #[test]
518    fn contig_peaks_stay_loci_on_a_mixed_axis() {
519        let names: Vec<Box<str>> = vec![
520            "chr1_CTG1v1_random:5-10".into(),
521            "chr4_CTG2v2_random:5-10".into(),
522            "ENSG000_GENE1".into(),
523        ];
524        let canon = build_mixed_kind_canonicalizer(&names);
525        assert_eq!(canon(&names[0]).as_ref(), "1_CTG1v1_random_5_10");
526        assert_eq!(canon(&names[1]).as_ref(), "4_CTG2v2_random_5_10");
527        assert_eq!(canon(&names[2]).as_ref(), "GENE1");
528    }
529
530    #[test]
531    fn locus_canonical_keeps_case_on_every_path() {
532        let names: Vec<Box<str>> = vec!["chrX:0-100".into(), "chr1:0-100".into()];
533        let map = build_locus_overlap_canonical_map(&names);
534        assert_eq!(map.get(&names[0]).unwrap().as_ref(), "X_0_100");
535        assert_eq!(map.get(&names[1]).unwrap().as_ref(), "1_0_100");
536        // Names outside the map land on the same key.
537        let canon = build_locus_overlap_canonicalizer(&names);
538        assert_eq!(canon("chrX:200-300").as_ref(), "X_200_300");
539        let mixed = build_mixed_kind_canonicalizer(&names);
540        assert_eq!(mixed("chrX:0-100").as_ref(), "X_0_100");
541        assert_eq!(mixed("chrM:200-300").as_ref(), "M_200_300");
542        let k = FeatureNameKind::Locus {
543            merge_overlapping: false,
544        };
545        assert_eq!(k.canonicalize("chrM:0-100").as_ref(), "M_0_100");
546    }
547
548    #[test]
549    fn parse_locus_rejects_non_loci() {
550        assert!(parse_locus("TGFB1").is_none()); // gene symbol
551        assert!(parse_locus("ENSG00000105329").is_none()); // ensembl
552        assert!(parse_locus("chr1:bad-2000").is_none()); // non-numeric start
553        assert!(parse_locus("chr1:1000").is_none()); // missing end
554        assert!(parse_locus("chr1:2000-1000").is_none()); // end < start
555        assert!(parse_locus("").is_none()); // empty
556        assert!(parse_locus("chr1").is_none()); // chr-only
557        assert!(parse_locus("chr:1-2").is_none()); // empty chromosome
558        assert!(parse_locus("ENSG000_GENE1").is_none()); // gene-style
559        assert!(parse_locus("GENE1-AS1").is_none()); // antisense symbol
560    }
561
562    #[test]
563    fn overlap_map_merges_two_overlapping_intervals() {
564        // User's motivating example: chr1:1-20 and chr1:15-30 → same cluster.
565        let names = vec![
566            "chr1:1-20".to_string().into_boxed_str(),
567            "chr1:15-30".to_string().into_boxed_str(),
568        ];
569        let map = build_locus_overlap_canonical_map(&names);
570        let c0 = map.get(&names[0]).unwrap();
571        let c1 = map.get(&names[1]).unwrap();
572        assert_eq!(c0, c1, "both inputs should map to the same canonical");
573        assert_eq!(c0.as_ref(), "1_1_30"); // union-range canonical
574    }
575
576    #[test]
577    fn overlap_map_keeps_non_overlapping_separate() {
578        let names = vec![
579            "chr1:1-20".to_string().into_boxed_str(),
580            "chr1:100-200".to_string().into_boxed_str(),
581            "chr2:1-20".to_string().into_boxed_str(),
582        ];
583        let map = build_locus_overlap_canonical_map(&names);
584        assert_eq!(map.get(&names[0]).unwrap().as_ref(), "1_1_20");
585        assert_eq!(map.get(&names[1]).unwrap().as_ref(), "1_100_200");
586        // different chromosome — separate cluster even if start overlaps
587        assert_eq!(map.get(&names[2]).unwrap().as_ref(), "2_1_20");
588    }
589
590    #[test]
591    fn overlap_map_handles_transitive_chain() {
592        // A overlaps B (1-20 vs 15-30), B overlaps C (15-30 vs 25-40),
593        // A does NOT overlap C directly — but they should still cluster
594        // via transitive closure through B.
595        let names = vec![
596            "chr1:1-20".to_string().into_boxed_str(),
597            "chr1:15-30".to_string().into_boxed_str(),
598            "chr1:25-40".to_string().into_boxed_str(),
599        ];
600        let map = build_locus_overlap_canonical_map(&names);
601        let c0 = map.get(&names[0]).unwrap();
602        let c1 = map.get(&names[1]).unwrap();
603        let c2 = map.get(&names[2]).unwrap();
604        assert_eq!(c0, c1);
605        assert_eq!(c1, c2);
606        assert_eq!(c0.as_ref(), "1_1_40"); // union of full chain
607    }
608
609    #[test]
610    fn overlap_map_handles_full_containment() {
611        // chr1:1-100 contains chr1:30-50 — should cluster.
612        let names = vec![
613            "chr1:1-100".to_string().into_boxed_str(),
614            "chr1:30-50".to_string().into_boxed_str(),
615        ];
616        let map = build_locus_overlap_canonical_map(&names);
617        let c0 = map.get(&names[0]).unwrap();
618        let c1 = map.get(&names[1]).unwrap();
619        assert_eq!(c0, c1);
620        assert_eq!(c0.as_ref(), "1_1_100");
621    }
622
623    #[test]
624    fn overlap_map_treats_adjacent_as_separate() {
625        // chr1:1-20 and chr1:20-30 are *touching* but not overlapping
626        // (end of first == start of second, exclusive end convention).
627        let names = vec![
628            "chr1:1-20".to_string().into_boxed_str(),
629            "chr1:20-30".to_string().into_boxed_str(),
630        ];
631        let map = build_locus_overlap_canonical_map(&names);
632        assert_ne!(map.get(&names[0]).unwrap(), map.get(&names[1]).unwrap());
633    }
634
635    #[test]
636    fn overlap_map_normalizes_chr_prefix_within_cluster() {
637        // chr1:1-20 and 1:15-30 (no chr prefix) should still cluster
638        // because parse_locus normalizes both to chr="1".
639        let names = vec![
640            "chr1:1-20".to_string().into_boxed_str(),
641            "1:15-30".to_string().into_boxed_str(),
642        ];
643        let map = build_locus_overlap_canonical_map(&names);
644        let c0 = map.get(&names[0]).unwrap();
645        let c1 = map.get(&names[1]).unwrap();
646        assert_eq!(c0, c1);
647        assert_eq!(c0.as_ref(), "1_1_30");
648    }
649
650    #[test]
651    fn overlap_map_normalizes_separators_within_cluster() {
652        // chr1:1-20 and chr1_15_30 (different separators) → same cluster.
653        let names = vec![
654            "chr1:1-20".to_string().into_boxed_str(),
655            "chr1_15_30".to_string().into_boxed_str(),
656        ];
657        let map = build_locus_overlap_canonical_map(&names);
658        assert_eq!(map.get(&names[0]).unwrap(), map.get(&names[1]).unwrap());
659    }
660
661    #[test]
662    fn overlap_map_ignores_non_locus_names() {
663        // Non-locus names should not appear in the map; caller falls
664        // back to the per-name rule.
665        let names = vec![
666            "TGFB1".to_string().into_boxed_str(),
667            "chr1:1-20".to_string().into_boxed_str(),
668        ];
669        let map = build_locus_overlap_canonical_map(&names);
670        assert!(!map.contains_key(&names[0]));
671        assert!(map.contains_key(&names[1]));
672    }
673
674    #[test]
675    fn overlap_map_skips_empty_intervals() {
676        // chr1:1000-1000 holds no base, so it is not a locus.
677        let names = vec!["chr1:1000-1000".to_string().into_boxed_str()];
678        let map = build_locus_overlap_canonical_map(&names);
679        assert!(map.is_empty());
680    }
681
682    #[test]
683    fn overlap_canonicalizer_falls_back_for_unmatched() {
684        let names = vec!["chr1:1-20".to_string().into_boxed_str()];
685        let canon = build_locus_overlap_canonicalizer(&names);
686        // In-cluster name → cluster canonical.
687        assert_eq!(canon("chr1:1-20").as_ref(), "1_1_20");
688        // Unrelated locus not in the map → its own key.
689        assert_eq!(canon("chr2:500-600").as_ref(), "2_500_600");
690        // Non-locus → unchanged.
691        assert_eq!(canon("GENE1").as_ref(), "GENE1");
692    }
693
694    // -- Auto-detect & Mixed dispatcher -------------------------------------
695
696    #[test]
697    fn auto_detect_pure_locus_axis() {
698        let names: Vec<Box<str>> = (0..100)
699            .map(|i| format!("chr1:{}-{}", i * 100, i * 100 + 50).into_boxed_str())
700            .collect();
701        assert!(matches!(
702            FeatureNameKind::auto_detect(&names),
703            FeatureNameKind::Locus {
704                merge_overlapping: true
705            }
706        ));
707    }
708
709    #[test]
710    fn auto_detect_pure_gene_axis() {
711        let names: Vec<Box<str>> = (0..100)
712            .map(|i| format!("ENSG000_GENE{}", i).into_boxed_str())
713            .collect();
714        assert!(matches!(
715            FeatureNameKind::auto_detect(&names),
716            FeatureNameKind::Gene { delim: '_' }
717        ));
718    }
719
720    #[test]
721    fn auto_detect_mixed_axis() {
722        // 80 loci + 20 gene-style → both fractions ≥ 10% → Mixed.
723        let mut names: Vec<Box<str>> = (0..80)
724            .map(|i| format!("chr1:{}-{}", i * 1000, i * 1000 + 500).into_boxed_str())
725            .collect();
726        names.extend((0..20).map(|i| format!("ENSG000_GENE{}", i).into_boxed_str()));
727        assert!(matches!(
728            FeatureNameKind::auto_detect(&names),
729            FeatureNameKind::Mixed
730        ));
731    }
732
733    #[test]
734    fn auto_detect_empty_or_exact() {
735        assert!(matches!(
736            FeatureNameKind::auto_detect(&[]),
737            FeatureNameKind::Exact
738        ));
739        let names = vec!["TGFB1".into(), "CD4".into(), "IL2".into(), "GAPDH".into()];
740        assert!(matches!(
741            FeatureNameKind::auto_detect(&names),
742            FeatureNameKind::Exact
743        ));
744    }
745
746    #[test]
747    fn mixed_dispatcher_canonicalizes_each_name_by_kind() {
748        let names: Vec<Box<str>> = vec![
749            "chr1:1-20".into(),     // locus → cluster canonical
750            "chr1:15-30".into(),    // locus, overlaps above → same cluster
751            "ENSG000_TGFB1".into(), // gene-style → "TGFB1"
752            "CD4".into(),           // plain symbol → passthrough
753        ];
754        let canon = build_mixed_kind_canonicalizer(&names);
755        assert_eq!(canon("chr1:1-20").as_ref(), "1_1_30");
756        assert_eq!(canon("chr1:15-30").as_ref(), "1_1_30");
757        assert_eq!(canon("ENSG000_TGFB1").as_ref(), "TGFB1");
758        assert_eq!(canon("CD4").as_ref(), "CD4");
759    }
760}