Skip to main content

data_beans/aux/
feature_names.rs

1//! Feature-name kind + canonicalizer hooks for multi-file data
2//! alignment.
3//!
4//! NOT to be confused with the sibling [`crate::aux::feature_rows`]. That module
5//! defines the row-name **grammar** faba's producers emit
6//! (`{unit}/{modality}/{subunit}/{channel}`); this one **canonicalizes** a name
7//! that already exists, so the same gene or locus matches across files that
8//! spell it differently. Reach for `feature_rows` to build or split a row, and
9//! for this module to decide whether two spellings are the same feature.
10//!
11//! Loaders that union rows across multiple sparse backends
12//! ([`crate::aux::data_loading::read_data_on_shared_rows`]) can opt into
13//! generous matching by passing a [`FeatureNameKind`] — same row-name
14//! canonicalization machinery used by `senna marker` /
15//! `FeaturePairGraph::from_edge_list` (via
16//! [`legume_numeric::matrix::membership::GeneIndexResolver`]) but plumbed at the
17//! [`crate::sparse_io_vector::SparseIoVec`] level so the row
18//! intersection itself sees aligned names.
19//!
20//! The two flavors cover what biology pipelines see in practice:
21//!
22//! - [`FeatureNameKind::Gene`] for gene rows in scRNA / spatial-RNA
23//!   data, where the same gene shows up as `TGFB1`, `ENSG00000105329`,
24//!   or `ENSG00000105329_TGFB1` across cohorts;
25//! - [`FeatureNameKind::Locus`] for chromosome-coordinate rows in ATAC
26//!   / chickpea-style data, where `chr1:1000-2000` and `1:1000-2000`
27//!   should resolve to the same peak. Only the colon form is a locus;
28//!   `chr1_1000_2000` and `chr1-1000-2000` are not.
29
30use std::sync::Arc;
31
32use crate::sparse_io_vector::RowNameCanonicalizer;
33use genomic_data::coordinates::{self, chr_stripped, PeakCoord};
34use rustc_hash::FxHashMap as HashMap;
35
36/// Per-name canonicalization rule for cross-backend row alignment.
37/// Concrete strategy only — no "request" variants. Callers that want
38/// auto-detection pass [`None`] (or whatever wrapping enum they choose)
39/// and call [`FeatureNameKind::auto_detect`] once row names are in hand.
40#[derive(Clone, Debug, Default, PartialEq, Eq)]
41pub enum FeatureNameKind {
42    /// Strict string match — no canonicalization. Default.
43    #[default]
44    Exact,
45    /// Gene-symbol rule: register every `delim`-split component as an
46    /// alias of the full name. `ENSG00000105329_TGFB1` and `TGFB1`
47    /// resolve to the same row.
48    Gene { delim: char },
49    /// Genomic-locus rule. A colon-form locus becomes its key
50    /// (`chr1:1000-2000` and `1:1000-2000` → `1:1000-2000`); other names
51    /// pass through. If `merge_overlapping`, intervals that overlap on
52    /// the same chromosome additionally collapse into one cluster
53    /// (`chr1:1-20` ∪ `chr1:15-30` → `1:1-30`). Useful for ATAC peak
54    /// sets called independently across datasets.
55    Locus { merge_overlapping: bool },
56    /// Heterogeneous axis: dispatch per row name. Names that parse as
57    /// loci go through [`FeatureNameKind::Locus`] with overlap merging;
58    /// gene-style names (see [`FeatureNameKind::Gene`]) take the gene
59    /// rule; the rest pass through. Picked automatically when [`auto_detect`] finds both
60    /// signatures in the same axis (e.g. paired RNA + ATAC union).
61    Mixed,
62}
63
64impl FeatureNameKind {
65    /// Canonicalize a single name under this kind's per-name rule.
66    /// [`Locus { merge_overlapping: true }`] and [`Mixed`] only describe
67    /// the per-name part here (the locus key for loci, last-token split
68    /// for gene-style); the global cluster step lives in
69    /// [`build_locus_overlap_canonical_map`] and is installed by
70    /// [`build_canonicalizer`].
71    pub fn canonicalize(&self, name: &str) -> Box<str> {
72        match self {
73            FeatureNameKind::Exact => name.into(),
74            FeatureNameKind::Gene { delim } => gene_canonicalize(name, *delim),
75            FeatureNameKind::Locus { .. } => locus_key(name).unwrap_or_else(|| name.into()),
76            FeatureNameKind::Mixed => mixed_canonicalize(name),
77        }
78    }
79
80    /// True iff this kind is [`Exact`] — no canonicalizer needed.
81    pub fn is_exact(&self) -> bool {
82        matches!(self, FeatureNameKind::Exact)
83    }
84
85    /// True iff installing the canonicalizer requires peeking every row
86    /// name across all backends first (to build the locus-overlap cluster
87    /// map). Loaders branch on this.
88    pub fn needs_global_pass(&self) -> bool {
89        matches!(
90            self,
91            FeatureNameKind::Locus {
92                merge_overlapping: true
93            } | FeatureNameKind::Mixed
94        )
95    }
96
97    /// Sniff `names` and pick the right kind. Tallies:
98    /// `n_locus` = count where `parse_locus` matches; `n_gene_like` =
99    /// count of remaining names the gene rule applies to. Decision:
100    /// both ≥ 10% → [`Mixed`]; else loci ≥ 50% →
101    /// `Locus { merge_overlapping: true }`; else gene-like ≥ 50% →
102    /// `Gene { delim: '_' }`; else [`Exact`].
103    pub fn auto_detect(names: &[Box<str>]) -> Self {
104        let n = names.len();
105        if n == 0 {
106            return Self::Exact;
107        }
108        let mut n_locus = 0usize;
109        let mut n_gene_like = 0usize;
110        for name in names {
111            if coordinates::is_locus(name) {
112                n_locus += 1;
113            } else if is_gene_like(name, '_') {
114                n_gene_like += 1;
115            }
116        }
117        let pct_locus = n_locus as f32 / n as f32;
118        let pct_gene = n_gene_like as f32 / n as f32;
119        if pct_locus < 0.50 {
120            let n_spelled = names
121                .iter()
122                .filter(|name| {
123                    !coordinates::is_locus(name) && coordinates::import_interval(name).is_some()
124                })
125                .count();
126            if n_spelled * 2 >= n {
127                log::warn!(
128                    "{n_spelled} of {n} row names read as intervals only in a non-colon \
129                     spelling (e.g. `chr1-100-200`); they are not loci here. Re-import them so \
130                     peaks are named `chr:start-end`."
131                );
132            }
133        }
134        if pct_locus >= 0.10 && pct_gene >= 0.10 {
135            Self::Mixed
136        } else if pct_locus >= 0.50 {
137            Self::Locus {
138                merge_overlapping: true,
139            }
140        } else if pct_gene >= 0.50 {
141            Self::Gene { delim: '_' }
142        } else {
143            Self::Exact
144        }
145    }
146
147    /// The one kind to install for a set of files that were each sniffed
148    /// with [`auto_detect`](Self::auto_detect) on their own.
149    ///
150    /// Sniffing the POOLED names does not work: the signature usually lives
151    /// on one side only. A raw `ENSG_SYM` cohort pooled with a reference
152    /// already on the bare-symbol axis (a carried `pb_reference`, a
153    /// symbol-keyed panel) leaves the gene-like share under half, the pair
154    /// sniffs as `Exact`, and every gene becomes two rows. Canonicalizing
155    /// under `Gene` is a no-op for names lacking the delimiter, so adopting
156    /// the informative side is safe for both.
157    ///
158    /// Gene-style and locus-style files together dispatch per name
159    /// (`Mixed`), which is what `auto_detect` would pick on one axis holding
160    /// both; `Mixed` anywhere stays `Mixed`; all-`Exact` stays `Exact`.
161    #[must_use]
162    pub fn reconcile(kinds: &[FeatureNameKind]) -> FeatureNameKind {
163        if kinds.iter().any(|k| matches!(k, FeatureNameKind::Mixed)) {
164            return FeatureNameKind::Mixed;
165        }
166        let gene = kinds
167            .iter()
168            .find(|k| matches!(k, FeatureNameKind::Gene { .. }));
169        let locus = kinds
170            .iter()
171            .find(|k| matches!(k, FeatureNameKind::Locus { .. }));
172        match (gene, locus) {
173            (Some(_), Some(_)) => FeatureNameKind::Mixed,
174            _ => gene.or(locus).cloned().unwrap_or(FeatureNameKind::Exact),
175        }
176    }
177
178    /// Build a `RowNameCanonicalizer` suitable for
179    /// [`SparseIoVec::with_row_canonicalizer`]. Returns `None` for
180    /// [`FeatureNameKind::Exact`] so callers don't install a no-op
181    /// closure. **Does not** handle the LocusOverlap global step — for
182    /// that, use [`build_locus_overlap_canonicalizer`].
183    pub fn into_canonicalizer(self) -> Option<RowNameCanonicalizer> {
184        if self.is_exact() {
185            return None;
186        }
187        Some(Arc::new(move |name: &str| self.canonicalize(name)))
188    }
189}
190
191/// Parse a row name as `(chr, start, end)` under the shared locus grammar
192/// ([`coordinates::parse_interval`]): colon form only, `chr1:1000-2000` or
193/// `1:1000-2000`. The chromosome comes back with its `chr` prefix dropped
194/// and its case kept (`chrX` and `X` match, `x` does not). Returns `None`
195/// for anything that doesn't match; those names pass through the overlap
196/// pass untouched.
197pub fn parse_locus(name: &str) -> Option<(Box<str>, u64, u64)> {
198    let (chr, start, end) = coordinates::split_interval(name)?;
199    Some((chr_stripped(chr).into(), start as u64, end as u64))
200}
201
202/// Canonical key of one locus (`chrX:0-100` and `CHRX:0-100` become
203/// `X:0-100`), or `None` when `name` is not a locus: the one answer to "is
204/// this row a locus, and under which key".
205pub use genomic_data::coordinates::locus_key;
206
207/// Per-name rule of [`FeatureNameKind::Mixed`]: locus key, else the gene
208/// rule for gene-style names, else the name unchanged.
209fn mixed_canonicalize(name: &str) -> Box<str> {
210    locus_key(name).unwrap_or_else(|| gene_canonicalize(name, '_'))
211}
212
213/// Build the overlap-merge canonical map from a flat list of row names
214/// across all input backends. Names that parse as `(chr, start, end)`
215/// are grouped per chromosome, sorted by start, and clustered by
216/// transitive overlap (any interval whose start falls before the
217/// running cluster's max end). The cluster canonical is
218/// `{chr}:{min_start}-{max_end}` so every member name maps to a single
219/// well-defined string.
220///
221/// Names that fail to parse are not entered into the map; the caller
222/// falls back to the per-name rule for those.
223pub fn build_locus_overlap_canonical_map(names: &[Box<str>]) -> HashMap<Box<str>, Box<str>> {
224    let n = names.len();
225    let parsed: Vec<Option<PeakCoord>> = names
226        .iter()
227        .map(|n| coordinates::parse_interval(n))
228        .collect();
229
230    // Bucket valid indices by chromosome (`chr1` and `1` together).
231    let mut by_chr: HashMap<&str, Vec<usize>> = HashMap::default();
232    for (i, p) in parsed.iter().enumerate() {
233        if let Some(p) = p {
234            by_chr.entry(chr_stripped(&p.chr)).or_default().push(i);
235        }
236    }
237
238    // Union-find with path compression.
239    let mut parent: Vec<usize> = (0..n).collect();
240    fn find(p: &mut [usize], mut x: usize) -> usize {
241        while p[x] != x {
242            let g = p[p[x]];
243            p[x] = g;
244            x = g;
245        }
246        x
247    }
248
249    // Per chr: sort by start, sweep, union anything overlapping the running cluster.
250    let mut cluster_extent: HashMap<usize, (i64, i64)> = HashMap::default();
251    for (_, mut idxs) in by_chr {
252        idxs.sort_by_key(|&i| parsed[i].as_ref().map_or(0, |p| p.start));
253        let mut current_root: Option<usize> = None;
254        let mut current_min_start: i64 = 0;
255        let mut current_max_end: i64 = 0;
256        for i in idxs {
257            let PeakCoord {
258                start: s, end: e, ..
259            } = parsed[i].as_ref().unwrap();
260            match current_root {
261                Some(root) if *s < current_max_end => {
262                    let ra = find(&mut parent, root);
263                    let rb = find(&mut parent, i);
264                    if ra != rb {
265                        parent[rb] = ra;
266                    }
267                    current_max_end = current_max_end.max(*e);
268                    cluster_extent
269                        .insert(find(&mut parent, i), (current_min_start, current_max_end));
270                }
271                _ => {
272                    current_root = Some(i);
273                    current_min_start = *s;
274                    current_max_end = *e;
275                    cluster_extent.insert(i, (*s, *e));
276                }
277            }
278        }
279    }
280
281    // Build name → canonical map: the cluster extent under `locus_key`.
282    let mut out: HashMap<Box<str>, Box<str>> = HashMap::default();
283    for (i, p) in parsed.iter().enumerate() {
284        if let Some(p) = p {
285            let root = find(&mut parent, i);
286            let (start, end) = cluster_extent.get(&root).copied().unwrap_or((0, 0));
287            // The member's own chromosome spelling: `locus_key` strips it
288            // once, as on every per-name path.
289            let cluster = PeakCoord {
290                chr: p.chr.clone(),
291                start,
292                end,
293            };
294            out.insert(names[i].clone(), cluster.locus_key());
295        }
296    }
297    out
298}
299
300/// Build a `RowNameCanonicalizer` for [`FeatureNameKind::LocusOverlap`].
301/// `names` should be the concatenation of every input backend's row
302/// names (in any order). The returned canonicalizer does
303/// `map.get(name).cloned()` first, falling back to per-name
304/// locus canonical for names outside the map.
305pub fn build_locus_overlap_canonicalizer(names: &[Box<str>]) -> RowNameCanonicalizer {
306    let map = Arc::new(build_locus_overlap_canonical_map(names));
307    Arc::new(move |name: &str| {
308        map.get(name)
309            .cloned()
310            .or_else(|| locus_key(name))
311            .unwrap_or_else(|| name.into())
312    })
313}
314
315/// Per-name dispatcher for **mixed-kind** axes (e.g. multiome with peaks
316/// ∪ genes in one feature axis). For each name:
317///   • parses as `(chr, start, end)` → LocusOverlap canonical
318///     (cluster representative from `names`).
319///   • gene-style (see [`FeatureNameKind::Gene`]) → gene rule: last token
320///     after the rightmost `_`.
321///   • else → passthrough.
322///
323/// Use this when the auto-detector sees significant evidence of BOTH
324/// loci and gene-style names in the same axis.
325pub fn build_mixed_kind_canonicalizer(names: &[Box<str>]) -> RowNameCanonicalizer {
326    let map = Arc::new(build_locus_overlap_canonical_map(names));
327    Arc::new(move |name: &str| {
328        map.get(name)
329            .cloned()
330            .unwrap_or_else(|| mixed_canonicalize(name))
331    })
332}
333
334/// Gene-symbol canonicalization with Cell Ranger feature-type suffix
335/// awareness. 10x Cell Ranger HDF5 row names commonly arrive as
336/// `ENSG..._SYMBOL_<feature_type>` where the third component is a
337/// sanitized `feature_type` tag (e.g. `Gene` for `Gene Expression`).
338/// A naive `rsplit(delim).next()` would return that constant tag,
339/// canonicalizing *every* row to the same string and collapsing the
340/// row intersection to one global key. Strip the known tag suffix
341/// first so the actual symbol becomes the rsplit target.
342fn gene_canonicalize(name: &str, delim: char) -> Box<str> {
343    let (head, rest) = gene_part(name);
344    match gene_symbol(head, delim) {
345        Some(symbol) if rest.is_empty() => symbol.into(),
346        Some(symbol) => format!("{symbol}{rest}").into_boxed_str(),
347        None => name.into(),
348    }
349}
350
351/// The gene rule applies: see [`gene_symbol`].
352fn is_gene_like(name: &str, delim: char) -> bool {
353    gene_symbol(gene_part(name).0, delim).is_some()
354}
355
356/// A row name's gene part, its first `/`-segment, and the rest from the
357/// first `/` on (`ENSG_GENE1/m6a/chr1:100/methylated` gives `ENSG_GENE1`
358/// and `/m6a/chr1:100/methylated`). The gene rule reads only the first;
359/// the rest is kept as written.
360fn gene_part(name: &str) -> (&str, &str) {
361    name.find('/').map_or((name, ""), |i| name.split_at(i))
362}
363
364/// The symbol the gene rule keys a gene part on: its last `delim` component
365/// once a Cell Ranger feature-type tag is stripped. `None` (the name is kept
366/// whole) when there is no `delim`, when the part is a coordinate (a locus
367/// `chr:start-end` or a position `chr:pos`, see [`coordinates::is_region`]),
368/// or when the symbol is all digits: cutting `chrUn_CTG1v1:0-100` or
369/// `chr1_CTG1v1_random:12345` would key it on its contig tail, and
370/// `chr1_100_200` or `chr1_100_200_Peaks` (not colon form, so not loci)
371/// would collapse onto `200`.
372fn gene_symbol(head: &str, delim: char) -> Option<&str> {
373    if !head.contains(delim) {
374        return None;
375    }
376    let stripped = strip_feature_type_suffix(head, delim);
377    if coordinates::is_region(stripped) {
378        return None;
379    }
380    let symbol = stripped.rsplit(delim).next().unwrap_or(stripped);
381    (!symbol.bytes().all(|b| b.is_ascii_digit())).then_some(symbol)
382}
383
384/// Cell Ranger sanitizes `features/feature_type` into the row name as
385/// the trailing component. Strip the known tags so the actual gene
386/// symbol becomes the rsplit target. Conservative list — only the
387/// shapes we've actually seen in the wild — so an unknown tag falls
388/// through untouched rather than corrupting a real symbol.
389fn strip_feature_type_suffix(name: &str, delim: char) -> &str {
390    // Names come pre-sanitized in different ways depending on the
391    // producer (Cell Ranger's own h5, scanpy/anndata exports, R-side
392    // tools), so accept both `_Gene` and `_Gene_Expression` plus the
393    // common companion tags.
394    const TAGS: &[&str] = &[
395        "Gene_Expression",
396        "Gene",
397        "Antibody_Capture",
398        "CRISPR_Guide_Capture",
399        "Multiplexing_Capture",
400        "Custom",
401        "Peaks",
402    ];
403    for tag in TAGS {
404        // Only strip if the suffix sits behind `delim` (otherwise we'd
405        // mangle a real symbol that happens to end in "Gene").
406        if let Some(rest) = name.strip_suffix(tag).and_then(|r| r.strip_suffix(delim)) {
407            return rest;
408        }
409    }
410    name
411}
412
413/// Clap-facing spelling of [`FeatureNameKind`].
414///
415/// The rule and the flag that selects it belong together: every crate that
416/// aligns feature names across files exposes the same `--feature-name-kind`
417/// vocabulary, so `senna` and anything after it agree on what
418/// `gene` or `locus` means without each inventing a local rule.
419#[derive(clap::ValueEnum, Clone, Debug, Default, serde::Serialize, serde::Deserialize)]
420#[serde(rename_all = "kebab-case")]
421pub enum FeatureNameKindArg {
422    #[default]
423    Auto,
424    Exact,
425    Gene,
426    Locus,
427    LocusOverlap,
428    Mixed,
429}
430
431impl FeatureNameKindArg {
432    /// Resolve to a concrete [`FeatureNameKind`], defaulting `Auto` to
433    /// `Gene { delim: '_' }` — the standard for gene-keyed pre-train
434    /// inputs (bge / fne / topic-family dictionaries).
435    pub fn resolve_or_gene(&self) -> FeatureNameKind {
436        Option::<FeatureNameKind>::from(self.clone())
437            .unwrap_or(FeatureNameKind::Gene { delim: '_' })
438    }
439}
440
441impl From<FeatureNameKindArg> for Option<FeatureNameKind> {
442    fn from(arg: FeatureNameKindArg) -> Self {
443        match arg {
444            FeatureNameKindArg::Auto => None,
445            FeatureNameKindArg::Exact => Some(FeatureNameKind::Exact),
446            FeatureNameKindArg::Gene => Some(FeatureNameKind::Gene { delim: '_' }),
447            FeatureNameKindArg::Locus => Some(FeatureNameKind::Locus {
448                merge_overlapping: false,
449            }),
450            FeatureNameKindArg::LocusOverlap => Some(FeatureNameKind::Locus {
451                merge_overlapping: true,
452            }),
453            FeatureNameKindArg::Mixed => Some(FeatureNameKind::Mixed),
454        }
455    }
456}
457
458#[cfg(test)]
459#[path = "feature_names_tests.rs"]
460mod feature_names_tests;
461
462#[cfg(test)]
463mod tests {
464    use super::*;
465
466    #[test]
467    fn exact_passthrough() {
468        let k = FeatureNameKind::Exact;
469        assert_eq!(
470            k.canonicalize("ENSG00000000003_TSPAN6").as_ref(),
471            "ENSG00000000003_TSPAN6"
472        );
473        assert!(k.is_exact());
474        assert!(k.into_canonicalizer().is_none());
475    }
476
477    #[test]
478    fn gene_takes_last_underscore_component() {
479        let k = FeatureNameKind::Gene { delim: '_' };
480        assert_eq!(k.canonicalize("ENSG00000000003_TSPAN6").as_ref(), "TSPAN6");
481        // Symbol-only inputs survive unchanged.
482        assert_eq!(k.canonicalize("TSPAN6").as_ref(), "TSPAN6");
483        // Unknown trailing tokens still get rsplit — caller's responsibility
484        // to pick a sensible delim if a non-feature-type trailing token matters.
485        assert_eq!(k.canonicalize("A_B_C").as_ref(), "C");
486        assert!(!k.is_exact());
487        assert!(k.into_canonicalizer().is_some());
488    }
489
490    #[test]
491    fn gene_strips_cell_ranger_feature_type_suffix() {
492        let k = FeatureNameKind::Gene { delim: '_' };
493        // 10x Cell Ranger HDF5: `ENSG..._SYMBOL_Gene`. Trailing `_Gene`
494        // would otherwise collapse every gene to the literal "Gene".
495        assert_eq!(
496            k.canonicalize("ENSG00000187634_SAMD11_Gene").as_ref(),
497            "SAMD11"
498        );
499        // Full `Gene_Expression` tag variant.
500        assert_eq!(
501            k.canonicalize("ENSG00000187634_SAMD11_Gene_Expression")
502                .as_ref(),
503            "SAMD11"
504        );
505        // A real gene whose name happens to end in "Gene" *without* the
506        // delimiter shouldn't be stripped (no underscore in front of "Gene").
507        assert_eq!(k.canonicalize("FakeGene").as_ref(), "FakeGene");
508    }
509
510    #[test]
511    fn locus_strips_chr_and_leaves_other_spellings_alone() {
512        let k = FeatureNameKind::Locus {
513            merge_overlapping: false,
514        };
515        assert_eq!(k.canonicalize("chr1:1000-2000").as_ref(), "1:1000-2000");
516        assert_eq!(k.canonicalize("ChrX:5000-6000").as_ref(), "X:5000-6000");
517        // Not colon form: not a locus, so the name passes through.
518        assert_eq!(k.canonicalize("1_1000_2000").as_ref(), "1_1000_2000");
519    }
520
521    // -- Genomic region parsing edge cases ----------------------------------
522
523    #[test]
524    fn parse_locus_accepts_common_formats() {
525        // colon-dash, bare chromosome, chr prefix in any case, chrX caps.
526        assert_eq!(
527            parse_locus("chr1:1000-2000"),
528            Some(("1".into(), 1000, 2000))
529        );
530        assert_eq!(parse_locus("1:1000-2000"), Some(("1".into(), 1000, 2000)));
531        assert_eq!(
532            parse_locus("CHR1:1000-2000"),
533            Some(("1".into(), 1000, 2000))
534        );
535        assert_eq!(
536            parse_locus("chrX:5000-6000"),
537            Some(("X".into(), 5000, 6000))
538        );
539        assert_eq!(parse_locus("chrMT:1-100"), Some(("MT".into(), 1, 100)));
540    }
541
542    #[test]
543    fn parse_locus_keeps_contig_names_with_separators() {
544        assert_eq!(
545            parse_locus("chrUn_CTG1v1:0-100"),
546            Some(("Un_CTG1v1".into(), 0, 100))
547        );
548        // The canonical key parses back to the same locus.
549        assert_eq!(
550            parse_locus("Un_CTG1v1:0-100"),
551            Some(("Un_CTG1v1".into(), 0, 100))
552        );
553    }
554
555    #[test]
556    fn contig_peaks_stay_loci_on_a_mixed_axis() {
557        let names: Vec<Box<str>> = vec![
558            "chr1_CTG1v1_random:5-10".into(),
559            "chr4_CTG2v2_random:5-10".into(),
560            "ENSG000_GENE1".into(),
561        ];
562        let canon = build_mixed_kind_canonicalizer(&names);
563        assert_eq!(canon(&names[0]).as_ref(), "1_CTG1v1_random:5-10");
564        assert_eq!(canon(&names[1]).as_ref(), "4_CTG2v2_random:5-10");
565        assert_eq!(canon(&names[2]).as_ref(), "GENE1");
566    }
567
568    #[test]
569    fn every_locus_path_gives_one_key() {
570        let names: Vec<Box<str>> = vec!["chrChr1:0-100".into(), "chr1:0-100".into()];
571        let map = build_locus_overlap_canonical_map(&names);
572        let k = FeatureNameKind::Locus {
573            merge_overlapping: false,
574        };
575        for name in &names {
576            assert_eq!(map.get(name).unwrap(), &k.canonicalize(name));
577        }
578    }
579
580    #[test]
581    fn locus_canonical_keeps_case_on_every_path() {
582        let names: Vec<Box<str>> = vec!["chrX:0-100".into(), "chr1:0-100".into()];
583        let map = build_locus_overlap_canonical_map(&names);
584        assert_eq!(map.get(&names[0]).unwrap().as_ref(), "X:0-100");
585        assert_eq!(map.get(&names[1]).unwrap().as_ref(), "1:0-100");
586        // Names outside the map land on the same key.
587        let canon = build_locus_overlap_canonicalizer(&names);
588        assert_eq!(canon("chrX:200-300").as_ref(), "X:200-300");
589        let mixed = build_mixed_kind_canonicalizer(&names);
590        assert_eq!(mixed("chrX:0-100").as_ref(), "X:0-100");
591        assert_eq!(mixed("chrM:200-300").as_ref(), "M:200-300");
592        let k = FeatureNameKind::Locus {
593            merge_overlapping: false,
594        };
595        assert_eq!(k.canonicalize("chrM:0-100").as_ref(), "M:0-100");
596    }
597
598    #[test]
599    fn parse_locus_rejects_non_loci() {
600        assert!(parse_locus("TGFB1").is_none()); // gene symbol
601        assert!(parse_locus("ENSG00000105329").is_none()); // ensembl
602        assert!(parse_locus("chr1:bad-2000").is_none()); // non-numeric start
603        assert!(parse_locus("chr1:1000").is_none()); // missing end
604        assert!(parse_locus("chr1:2000-1000").is_none()); // end < start
605        assert!(parse_locus("").is_none()); // empty
606        assert!(parse_locus("chr1").is_none()); // chr-only
607        assert!(parse_locus("chr:1-2").is_none()); // empty chromosome
608        assert!(parse_locus("ENSG000_GENE1").is_none()); // gene-style
609        assert!(parse_locus("GENE1-AS1").is_none()); // antisense symbol
610        assert!(parse_locus("chr1_1000_2000").is_none()); // underscore form
611        assert!(parse_locus("chr1-1000-2000").is_none()); // dash form
612    }
613
614    #[test]
615    fn overlap_map_merges_two_overlapping_intervals() {
616        // User's motivating example: chr1:1-20 and chr1:15-30 → same cluster.
617        let names = vec![
618            "chr1:1-20".to_string().into_boxed_str(),
619            "chr1:15-30".to_string().into_boxed_str(),
620        ];
621        let map = build_locus_overlap_canonical_map(&names);
622        let c0 = map.get(&names[0]).unwrap();
623        let c1 = map.get(&names[1]).unwrap();
624        assert_eq!(c0, c1, "both inputs should map to the same canonical");
625        assert_eq!(c0.as_ref(), "1:1-30"); // union-range canonical
626    }
627
628    #[test]
629    fn overlap_map_keeps_non_overlapping_separate() {
630        let names = vec![
631            "chr1:1-20".to_string().into_boxed_str(),
632            "chr1:100-200".to_string().into_boxed_str(),
633            "chr2:1-20".to_string().into_boxed_str(),
634        ];
635        let map = build_locus_overlap_canonical_map(&names);
636        assert_eq!(map.get(&names[0]).unwrap().as_ref(), "1:1-20");
637        assert_eq!(map.get(&names[1]).unwrap().as_ref(), "1:100-200");
638        // different chromosome — separate cluster even if start overlaps
639        assert_eq!(map.get(&names[2]).unwrap().as_ref(), "2:1-20");
640    }
641
642    #[test]
643    fn overlap_map_handles_transitive_chain() {
644        // A overlaps B (1-20 vs 15-30), B overlaps C (15-30 vs 25-40),
645        // A does NOT overlap C directly — but they should still cluster
646        // via transitive closure through B.
647        let names = vec![
648            "chr1:1-20".to_string().into_boxed_str(),
649            "chr1:15-30".to_string().into_boxed_str(),
650            "chr1:25-40".to_string().into_boxed_str(),
651        ];
652        let map = build_locus_overlap_canonical_map(&names);
653        let c0 = map.get(&names[0]).unwrap();
654        let c1 = map.get(&names[1]).unwrap();
655        let c2 = map.get(&names[2]).unwrap();
656        assert_eq!(c0, c1);
657        assert_eq!(c1, c2);
658        assert_eq!(c0.as_ref(), "1:1-40"); // union of full chain
659    }
660
661    #[test]
662    fn overlap_map_handles_full_containment() {
663        // chr1:1-100 contains chr1:30-50 — should cluster.
664        let names = vec![
665            "chr1:1-100".to_string().into_boxed_str(),
666            "chr1:30-50".to_string().into_boxed_str(),
667        ];
668        let map = build_locus_overlap_canonical_map(&names);
669        let c0 = map.get(&names[0]).unwrap();
670        let c1 = map.get(&names[1]).unwrap();
671        assert_eq!(c0, c1);
672        assert_eq!(c0.as_ref(), "1:1-100");
673    }
674
675    #[test]
676    fn overlap_map_treats_adjacent_as_separate() {
677        // chr1:1-20 and chr1:20-30 are *touching* but not overlapping
678        // (end of first == start of second, exclusive end convention).
679        let names = vec![
680            "chr1:1-20".to_string().into_boxed_str(),
681            "chr1:20-30".to_string().into_boxed_str(),
682        ];
683        let map = build_locus_overlap_canonical_map(&names);
684        assert_ne!(map.get(&names[0]).unwrap(), map.get(&names[1]).unwrap());
685    }
686
687    #[test]
688    fn overlap_map_normalizes_chr_prefix_within_cluster() {
689        // chr1:1-20 and 1:15-30 (no chr prefix) should still cluster
690        // because parse_locus normalizes both to chr="1".
691        let names = vec![
692            "chr1:1-20".to_string().into_boxed_str(),
693            "1:15-30".to_string().into_boxed_str(),
694        ];
695        let map = build_locus_overlap_canonical_map(&names);
696        let c0 = map.get(&names[0]).unwrap();
697        let c1 = map.get(&names[1]).unwrap();
698        assert_eq!(c0, c1);
699        assert_eq!(c0.as_ref(), "1:1-30");
700    }
701
702    #[test]
703    fn overlap_map_leaves_non_colon_spellings_out() {
704        // chr1_15_30 is not a locus, so it neither joins nor extends the cluster.
705        let names = vec![
706            "chr1:1-20".to_string().into_boxed_str(),
707            "chr1_15_30".to_string().into_boxed_str(),
708        ];
709        let map = build_locus_overlap_canonical_map(&names);
710        assert_eq!(map.get(&names[0]).unwrap().as_ref(), "1:1-20");
711        assert!(!map.contains_key(&names[1]));
712    }
713
714    #[test]
715    fn underscore_peaks_are_not_collapsed_by_the_gene_rule() {
716        // Neither loci nor genes: an axis of these is Exact, and the gene
717        // rule leaves them whole rather than keying every row on `200`.
718        let names: Vec<Box<str>> = (1..=20)
719            .map(|i| format!("chr{i}_100_200").into_boxed_str())
720            .collect();
721        assert_eq!(FeatureNameKind::auto_detect(&names), FeatureNameKind::Exact);
722        let mixed = build_mixed_kind_canonicalizer(&names);
723        assert_eq!(mixed("chr2_100_200").as_ref(), "chr2_100_200");
724        let gene = FeatureNameKind::Gene { delim: '_' };
725        assert_eq!(gene.canonicalize("chr2_100_200").as_ref(), "chr2_100_200");
726        assert_eq!(gene.canonicalize("ENSG000_GENE1").as_ref(), "GENE1");
727        assert_eq!(gene.canonicalize("GENE1_Gene").as_ref(), "GENE1");
728        assert_eq!(
729            gene.canonicalize("chr2_100_200_Peaks").as_ref(),
730            "chr2_100_200_Peaks"
731        );
732        // Only the gene part (first `/`-segment) is read, and a position
733        // there is a coordinate, not a gene.
734        assert_eq!(
735            gene.canonicalize("chr1_CTG1v1_random:12345/baf/alt")
736                .as_ref(),
737            "chr1_CTG1v1_random:12345/baf/alt"
738        );
739        assert_eq!(
740            gene.canonicalize("ENSG000_GENE1/m6a/chr1_CTG1v1_random:123/methylated")
741                .as_ref(),
742            "GENE1/m6a/chr1_CTG1v1_random:123/methylated"
743        );
744        assert_eq!(
745            gene.canonicalize("ENSG000_GENE1/count/spliced").as_ref(),
746            "GENE1/count/spliced"
747        );
748        // A locus is never cut at `_`, even when its contig name has one or
749        // it carries a feature-type tag.
750        assert_eq!(
751            gene.canonicalize("chrUn_CTG1v1:0-100_Peaks").as_ref(),
752            "chrUn_CTG1v1:0-100_Peaks"
753        );
754        assert_eq!(
755            gene.canonicalize("chrUn_CTG1v1:0-100").as_ref(),
756            "chrUn_CTG1v1:0-100"
757        );
758    }
759
760    #[test]
761    fn overlap_map_ignores_non_locus_names() {
762        // Non-locus names should not appear in the map; caller falls
763        // back to the per-name rule.
764        let names = vec![
765            "TGFB1".to_string().into_boxed_str(),
766            "chr1:1-20".to_string().into_boxed_str(),
767        ];
768        let map = build_locus_overlap_canonical_map(&names);
769        assert!(!map.contains_key(&names[0]));
770        assert!(map.contains_key(&names[1]));
771    }
772
773    #[test]
774    fn overlap_map_skips_empty_intervals() {
775        // chr1:1000-1000 holds no base, so it is not a locus.
776        let names = vec!["chr1:1000-1000".to_string().into_boxed_str()];
777        let map = build_locus_overlap_canonical_map(&names);
778        assert!(map.is_empty());
779    }
780
781    #[test]
782    fn overlap_canonicalizer_falls_back_for_unmatched() {
783        let names = vec!["chr1:1-20".to_string().into_boxed_str()];
784        let canon = build_locus_overlap_canonicalizer(&names);
785        // In-cluster name → cluster canonical.
786        assert_eq!(canon("chr1:1-20").as_ref(), "1:1-20");
787        // Unrelated locus not in the map → its own key.
788        assert_eq!(canon("chr2:500-600").as_ref(), "2:500-600");
789        // Non-locus → unchanged.
790        assert_eq!(canon("GENE1").as_ref(), "GENE1");
791    }
792
793    // -- Auto-detect & Mixed dispatcher -------------------------------------
794
795    #[test]
796    fn auto_detect_pure_locus_axis() {
797        let names: Vec<Box<str>> = (0..100)
798            .map(|i| format!("chr1:{}-{}", i * 100, i * 100 + 50).into_boxed_str())
799            .collect();
800        assert!(matches!(
801            FeatureNameKind::auto_detect(&names),
802            FeatureNameKind::Locus {
803                merge_overlapping: true
804            }
805        ));
806    }
807
808    #[test]
809    fn auto_detect_pure_gene_axis() {
810        let names: Vec<Box<str>> = (0..100)
811            .map(|i| format!("ENSG000_GENE{}", i).into_boxed_str())
812            .collect();
813        assert!(matches!(
814            FeatureNameKind::auto_detect(&names),
815            FeatureNameKind::Gene { delim: '_' }
816        ));
817    }
818
819    #[test]
820    fn auto_detect_mixed_axis() {
821        // 80 loci + 20 gene-style → both fractions ≥ 10% → Mixed.
822        let mut names: Vec<Box<str>> = (0..80)
823            .map(|i| format!("chr1:{}-{}", i * 1000, i * 1000 + 500).into_boxed_str())
824            .collect();
825        names.extend((0..20).map(|i| format!("ENSG000_GENE{}", i).into_boxed_str()));
826        assert!(matches!(
827            FeatureNameKind::auto_detect(&names),
828            FeatureNameKind::Mixed
829        ));
830    }
831
832    #[test]
833    fn auto_detect_empty_or_exact() {
834        assert!(matches!(
835            FeatureNameKind::auto_detect(&[]),
836            FeatureNameKind::Exact
837        ));
838        let names = vec!["TGFB1".into(), "CD4".into(), "IL2".into(), "GAPDH".into()];
839        assert!(matches!(
840            FeatureNameKind::auto_detect(&names),
841            FeatureNameKind::Exact
842        ));
843    }
844
845    #[test]
846    fn mixed_dispatcher_canonicalizes_each_name_by_kind() {
847        let names: Vec<Box<str>> = vec![
848            "chr1:1-20".into(),     // locus → cluster canonical
849            "chr1:15-30".into(),    // locus, overlaps above → same cluster
850            "ENSG000_TGFB1".into(), // gene-style → "TGFB1"
851            "CD4".into(),           // plain symbol → passthrough
852        ];
853        let canon = build_mixed_kind_canonicalizer(&names);
854        assert_eq!(canon("chr1:1-20").as_ref(), "1:1-30");
855        assert_eq!(canon("chr1:15-30").as_ref(), "1:1-30");
856        assert_eq!(canon("ENSG000_TGFB1").as_ref(), "TGFB1");
857        assert_eq!(canon("CD4").as_ref(), "CD4");
858    }
859}