Skip to main content

scc_context/
rank.rs

1//! Candidate ranking (docs/CONTEXT_COMPILER.md §4–§6).
2//!
3//! Score = lexical overlap + FTS rank + graph/flow/ownership expansion.
4//! Deterministic and dependency-free (no embeddings in MVP).
5
6use scc_core::kinds;
7use scc_graph::TrustedGraphView;
8use scc_store::Store;
9use std::collections::BTreeSet;
10
11#[derive(Debug, Clone)]
12pub struct ScoredEntity {
13    pub id: String,
14    pub kind: String,
15    pub name: String,
16    pub score: f64,
17    /// Why this entity was selected (for evidence/debugging).
18    pub reason: String,
19}
20
21/// Tokenize free text into lowercase alphanumeric terms (>= 3 chars, no
22/// stopwords). Identifiers like `street-name` / `raw_text` split into their
23/// pieces so goal terms match them.
24pub fn terms(text: &str) -> BTreeSet<String> {
25    const STOPWORDS: &[&str] = &[
26        "the", "and", "for", "are", "was", "were", "with", "from", "into", "that", "this",
27        "not", "but", "you", "our", "all", "can", "has", "had", "its", "who", "what",
28    ];
29    text.split(|c: char| !c.is_alphanumeric())
30        .filter(|t| t.len() >= 3)
31        .map(|t| t.to_ascii_lowercase())
32        .filter(|t| !STOPWORDS.contains(&t.as_str()))
33        .collect()
34}
35
36/// Light morphological variants of a term so goal words match identifier
37/// stems (`normalization` → `normalize`, `transcripts` → `transcript`).
38pub fn stem_variants(term: &str) -> Vec<String> {
39    let mut out = vec![term.to_string()];
40    if term.len() < 5 {
41        return out;
42    }
43    if let Some(base) = term.strip_suffix("ation") {
44        out.push(format!("{base}e")); // normalization -> normalize
45    }
46    if let Some(base) = term.strip_suffix("ations") {
47        out.push(format!("{base}e")); // normalizations -> normalize
48    }
49    for suf in ["ing", "ings", "es", "ed", "s"] {
50        if let Some(base) = term.strip_suffix(suf) {
51            if base.len() >= 4 {
52                out.push(base.to_string());
53            }
54            // drop doubled consonants: running -> run
55            if base.len() >= 3 && base.ends_with(base.chars().last().unwrap()) {
56                let trimmed: String = base.chars().take(base.len() - 1).collect();
57                if trimmed.len() >= 3 {
58                    out.push(trimmed);
59                }
60            }
61        }
62    }
63    out.sort();
64    out.dedup();
65    out
66}
67
68/// Prefix-aware term match (light stemming): `normalize` matches
69/// `normalizes`/`normalizer`; `transcript` matches `transcripts`.
70pub fn term_match(a: &str, b: &str) -> bool {
71    if a.len() < 4 || b.len() < 4 {
72        return a == b;
73    }
74    a.starts_with(b) || b.starts_with(a)
75}
76
77/// Lexical similarity of an entity to the goal terms: name matches weighted
78/// double; prefix matches count.
79pub fn entity_similarity(e: &scc_core::Entity, goal_terms: &BTreeSet<String>) -> f64 {
80    if goal_terms.is_empty() {
81        return 0.0;
82    }
83    let name_terms = terms(&e.name);
84    let mut attr_text = String::new();
85    for (k, v) in &e.attributes {
86        if k == "docstring" || k == "signature" || k == "responsibility" || k == "path" {
87            if let Some(s) = v.as_str() {
88                attr_text.push_str(s);
89                attr_text.push(' ');
90            }
91        }
92    }
93    let attr_terms = terms(&attr_text);
94    let name_hits: usize = goal_terms
95        .iter()
96        .filter(|g| name_terms.iter().any(|n| term_match(g, n)))
97        .count();
98    let attr_hits: usize = goal_terms
99        .iter()
100        .filter(|g| attr_terms.iter().any(|n| term_match(g, n)))
101        .count();
102    (name_hits * 2 + attr_hits) as f64
103}
104
105/// Collect candidate entities for a task:
106/// 1. lexical: FTS over entities + symbols
107/// 2. explicit: named files/symbols
108/// 3. graph expansion: containing component, flows, upstream/downstream,
109///    ownership, invariants, tests — handled by the pack builder.
110///
111/// Pluggable semantic scorer (SCC-071): an optional provider (e.g. an
112/// embedding model) contributes a relevance signal fused into the candidate
113/// score. Providers are opt-in (`inference.enabled`); the default is the
114/// lexical/graph ranker alone. The provider may label evidence, never invent
115/// topology.
116pub trait SemanticScorer: Send + Sync {
117    fn score(&self, goal: &str, entity: &scc_core::Entity) -> f64;
118}
119
120/// Optional second-stage reranker (e.g. a separate cross-encoder model):
121/// reorders the collected candidates after lexical/semantic collection.
122/// `rerank` may reorder `candidates` in place; failures degrade gracefully
123/// (the trait method should treat errors as no-op).
124pub trait Reranker: Send + Sync {
125    fn rerank(&self, goal: &str, candidates: &mut Vec<ScoredEntity>);
126}
127
128pub fn collect_lexical_candidates(
129    store: &Store,
130    view: &TrustedGraphView,
131    goal: &str,
132    symbols: &[String],
133    limit: usize,
134) -> Vec<ScoredEntity> {
135    collect_lexical_candidates_with(store, view, goal, symbols, limit, None)
136}
137
138/// `collect_lexical_candidates` with an optional semantic scorer fused in.
139pub fn collect_lexical_candidates_with(
140    store: &Store,
141    view: &TrustedGraphView,
142    goal: &str,
143    symbols: &[String],
144    limit: usize,
145    scorer: Option<&dyn SemanticScorer>,
146) -> Vec<ScoredEntity> {
147    collect_lexical_candidates_full(store, view, goal, symbols, limit, scorer, None)
148}
149// trace:v1 id=impl.scc.rank work=WORK-SCC-001 satisfies=REQ-SCC-CTX
150
151/// `collect_lexical_candidates` with both a semantic scorer and a reranker.
152pub fn collect_lexical_candidates_full(
153    store: &Store,
154    view: &TrustedGraphView,
155    goal: &str,
156    symbols: &[String],
157    limit: usize,
158    scorer: Option<&dyn SemanticScorer>,
159    reranker: Option<&dyn Reranker>,
160) -> Vec<ScoredEntity> {
161    let goal_terms = terms(goal);
162    let sem = |e: &scc_core::Entity| -> f64 { scorer.map(|s| s.score(goal, e)).unwrap_or(0.0) };
163    let mut out: Vec<ScoredEntity> = Vec::new();
164    let mut seen: BTreeSet<String> = BTreeSet::new();
165
166    let push = |e: &scc_core::Entity, base: f64, reason: &str,
167                    out: &mut Vec<ScoredEntity>, seen: &mut BTreeSet<String>| {
168        if !seen.insert(e.id.clone()) {
169            return;
170        }
171        let sim = entity_similarity(e, &goal_terms);
172        let score = base + sim + sem(e);
173        out.push(ScoredEntity {
174            id: e.id.clone(),
175            kind: e.kind.clone(),
176            name: e.name.clone(),
177            score,
178            reason: reason.to_string(),
179        });
180    };
181
182    // FTS entities
183    if !goal_terms.is_empty() {
184        let joined: Vec<&str> = goal_terms.iter().map(|s| s.as_str()).collect();
185        let query = joined.join(" ");
186        if let Ok(hits) = store.search_entities(&query, limit) {
187            let rank = limit as f64;
188            for (i, e) in hits.iter().enumerate() {
189                push(e, (rank - i as f64) / rank, "lexical", &mut out, &mut seen);
190            }
191        }
192        if let Ok(hits) = store.search_symbols(&query, limit) {
193            let rank = limit as f64;
194            for (i, (name, sig, _kind, file, _line)) in hits.iter().enumerate() {
195                let id = scc_core::symbol_id(&view.graph.repo_id, file, name);
196                let mut e = scc_core::Entity::new(id.clone(), kinds::SYMBOL, name.clone());
197                e.attr("file", serde_json::json!(file));
198                if !sig.is_empty() {
199                    e.attr("signature", serde_json::json!(sig));
200                }
201                push(&e, (rank - i as f64) / rank, "lexical", &mut out, &mut seen);
202            }
203        }
204        // substring fallback with stem variants (FTS prefix matching cannot
205        // catch `normalization` → `normalize`)
206        for term in &goal_terms {
207            for variant in crate::rank::stem_variants(term) {
208                if let Ok(hits) = store.search_entities_like(&variant, 6) {
209                    for e in hits.iter() {
210                        push(e, 0.6, "substring", &mut out, &mut seen);
211                    }
212                }
213                if let Ok(hits) = store.search_symbols_like(&variant, 6) {
214                    for (name, sig, _kind, file, _line) in hits.iter() {
215                        let id = scc_core::symbol_id(&view.graph.repo_id, file, name);
216                        let mut e = scc_core::Entity::new(id.clone(), kinds::SYMBOL, name.clone());
217                        e.attr("file", serde_json::json!(file));
218                        if !sig.is_empty() {
219                            e.attr("signature", serde_json::json!(sig));
220                        }
221                        push(&e, 0.6, "substring", &mut out, &mut seen);
222                    }
223                }
224            }
225        }
226    }
227
228    // semantic proposal pass: with a scorer present, entities with a
229    // positive semantic signal surface even when no lexical term matches
230    // (docs/CONTEXT_COMPILER.md §4 — embeddings are never truth, but they
231    // may propose candidates)
232    if scorer.is_some() {
233        for e in view.entities() {
234            let sem_score = sem(e);
235            if sem_score > 0.0 {
236                push(e, 0.0, "semantic", &mut out, &mut seen);
237            }
238        }
239    }
240
241    // explicit symbols
242    for s in symbols {
243        let matches: Vec<String> = view
244            .entities_of_kind(kinds::SYMBOL)
245            .into_iter()
246            .filter(|e| e.name == *s)
247            .map(|e| e.id.clone())
248            .collect();
249        for id in matches {
250            if let Some(e) = view.entity(&id) {
251                push(e, 2.0, "explicit-symbol", &mut out, &mut seen);
252            }
253        }
254        // fallback: entity id directly
255        if view.entity(s).is_some() {
256            push(view.entity(s).unwrap(), 2.0, "explicit-id", &mut out, &mut seen);
257        }
258    }
259
260    // graph expansion: callers (upstream) and callees (downstream) of
261    // candidate symbols (docs/CONTEXT_COMPILER.md §6)
262    let symbol_candidates: Vec<String> = out
263        .iter()
264        .filter(|c| c.kind == kinds::SYMBOL)
265        .map(|c| c.id.clone())
266        .collect();
267    for sid in &symbol_candidates {
268        for r in view.in_pred(sid, "calls") {
269            if let Some(e) = view.entity(&r.subject) {
270                push(e, 1.0, "upstream", &mut out, &mut seen);
271            }
272        }
273        for r in view.out_pred(sid, "calls") {
274            if let Some(e) = view.entity(&r.object) {
275                push(e, 0.8, "downstream", &mut out, &mut seen);
276            }
277        }
278    }
279
280    // Routes are contracts — they must always surface when they match the
281    // goal, regardless of lexical crowding from the fact layer (exports/
282    // contracts/tests compete for the same candidate budget). Ground-truth
283    // recall for route items depends on these surviving truncation.
284    for r in view.entities_of_kind(kinds::ROUTE) {
285        let name_l = r.name.to_ascii_lowercase();
286        if goal_terms.iter().any(|t| name_l.contains(t)) {
287            push(r, 3.0, "route-contract", &mut out, &mut seen);
288        }
289    }
290
291    out.sort_by(|a, b| b.score.partial_cmp(&a.score).unwrap_or(std::cmp::Ordering::Equal));
292    if let Some(rr) = reranker {
293        rr.rerank(goal, &mut out);
294    }
295    out.truncate((limit + 8).max(20));
296    out
297}
298
299// ---------------------------------------------------------------------------
300// Startup-atlas confidence ranking (Wave 11 — GENERALIZATION II)
301// ---------------------------------------------------------------------------
302//
303// `rank_startup_atlas` orders the startup atlas' three fact sections
304// (components, entrypoints, contracts) so the highest-confidence entries
305// render first. The rule is precision via ordering, NOT deletion: every
306// entry survives, but evidence-backed facts (graph-derived edges, explicit
307// surface kinds, producer/consumer contracts) rank above heuristic ones
308// (bare-name components, generic flow-compiler entrypoints, schema/reactive
309// inferences), so an agent under a token budget reads the strongest facts
310// first. The scoring is a pure function of the `SystemAtlas` fields — no
311// store access, no heuristics invented from a bare get() — and is fully
312// deterministic: ties break on the entry id (lexicographic).
313
314use scc_core::{AtlasComponent, AtlasEntrypoint, Contract, ContractSubclass, SystemAtlas};
315
316/// Confidence (0..1) of one component entry: how much of it is graph-derived
317/// evidence vs a bare-name heuristic.
318///
319/// A component named only (no purpose, no implementation, no edges) is the
320/// heuristic floor (`0.2`) — the component compiler can still emit a name
321/// from clustering alone. Every attribute that carries an EXTRACTED/
322/// graph-resolved fact raises the score: a responsibility claim (purpose),
323/// implementation paths + member symbols, consumes/produces and
324/// upstream/downstream edges, ownership claims, failure behavior, and a
325/// hierarchy layer assigned by the clusterer. The strongest components
326/// (all evidence present) reach `1.0`.
327pub fn component_confidence(c: &AtlasComponent) -> f64 {
328    let mut score: f64 = 0.2; // bare-name heuristic floor
329    if !c.purpose.is_empty() {
330        score += 0.15;
331    }
332    if !c.implementation.is_empty() {
333        score += 0.10;
334    }
335    if !c.symbols.is_empty() {
336        score += 0.10;
337    }
338    if !c.consumes.is_empty() {
339        score += 0.10;
340    }
341    if !c.produces.is_empty() {
342        score += 0.10;
343    }
344    if !c.upstream.is_empty() {
345        score += 0.10;
346    }
347    if !c.downstream.is_empty() {
348        score += 0.10;
349    }
350    if !c.owns.is_empty() {
351        score += 0.10;
352    }
353    if !c.failure_behavior.is_empty() {
354        score += 0.05;
355    }
356    if c.layer != "component" {
357        score += 0.05; // hierarchy clusterer assigned a real layer
358    }
359    score.min(1.0)
360}
361
362/// Confidence (0..1) of one entrypoint entry by surface kind: extractor
363/// evidence-backed kinds outrank the flow compiler's generic heuristic.
364///
365/// `route`/`http` (ROUTE entities), `cli`/`cli-subcommand` (CLI flag
366/// facts), `public_api` (EXPORTS evidence), `event`/`queue` (topic
367/// publish/subscribe), and the other invocation-surface kinds all come from
368/// explicit extractor facts. The generic `entrypoint` kind is the flow
369/// compiler's heuristic marker and ranks lower; an unknown/empty kind is the
370/// floor. A resolved symbol id is a weak positive signal on top.
371pub fn entrypoint_confidence(e: &AtlasEntrypoint) -> f64 {
372    let base: f64 = match e.kind.as_str() {
373        "route" | "http" => 1.0,
374        "cli" | "cli-subcommand" => 0.95,
375        "public_api" | "public-api" => 0.9,
376        "event" => 0.85,
377        "queue" => 0.85,
378        "schedule" => 0.8,
379        "process" => 0.8,
380        "plugin" => 0.75,
381        "framework_callback" => 0.75,
382        "lifecycle" => 0.7,
383        "entrypoint" => 0.4, // heuristic, not extractor evidence
384        _ => 0.3,
385    };
386    let symbol_bonus = if e.symbol.is_empty() { 0.0 } else { 0.05 };
387    (base + symbol_bonus).min(1.0)
388}
389
390/// Confidence (0..1) of one contract entry by subclass evidence: concrete
391/// surfaces with a producer + consumers outrank schema/reactive inferences.
392///
393/// `http`/`cli`/`event`/`config` contracts come from ROUTE / CLI-flag /
394/// TOPIC / CONFIGURATION entities — the strongest evidence. A real producer
395/// symbol, non-empty consumers, and preserved evidence ids each add to the
396/// score. `schema` (SchemaDefinition, derived from validation/model
397/// schemas) and the other inferred subclasses rank lowest: the contract
398/// ontology knows the surface is a schema, but there is no producer or
399/// consumer wiring to make it an observed contract. The gap is engineered:
400/// even a bare http contract (`0.95`) outranks a fully-wired schema
401/// (`0.35 + 0.10 + 0.05 + 0.05 = 0.55`).
402pub fn contract_confidence(c: &Contract) -> f64 {
403    let base: f64 = match c.subclass {
404        ContractSubclass::Http => 0.95,
405        ContractSubclass::Cli => 0.90,
406        ContractSubclass::Event => 0.85,
407        ContractSubclass::Configuration => 0.80,
408        ContractSubclass::PublicApi => 0.75,
409        ContractSubclass::Rpc => 0.75,
410        ContractSubclass::Message => 0.70,
411        ContractSubclass::Plugin => 0.70,
412        ContractSubclass::Extension => 0.70,
413        ContractSubclass::Serialization => 0.65,
414        ContractSubclass::CallContract => 0.60,
415        ContractSubclass::Schema => 0.35, // schema/reactive: inferred, not observed
416    };
417    let producer_bonus = if c.producer.is_empty() || c.producer == c.id {
418        0.0
419    } else {
420        0.10
421    };
422    let consumers_bonus = if c.consumers.is_empty() { 0.0 } else { 0.05 };
423    let evidence_bonus = if c.evidence.is_empty() { 0.0 } else { 0.05 };
424    (base + producer_bonus + consumers_bonus + evidence_bonus).min(1.0)
425}
426
427/// Order the startup atlas sections by evidence-backed confidence, highest
428/// first. Every entry is kept — this is precision via ordering, never
429/// deletion. Deterministic: ties break on the entry id (lexicographic; for
430/// entrypoints, which carry no id, on name then kind then symbol).
431pub fn rank_startup_atlas(atlas: &mut SystemAtlas) {
432    atlas.components.sort_by(|a, b| {
433        component_confidence(b)
434            .partial_cmp(&component_confidence(a))
435            .unwrap_or(std::cmp::Ordering::Equal)
436            .then_with(|| a.name.cmp(&b.name))
437    });
438    atlas.entrypoints.sort_by(|a, b| {
439        entrypoint_confidence(b)
440            .partial_cmp(&entrypoint_confidence(a))
441            .unwrap_or(std::cmp::Ordering::Equal)
442            .then_with(|| a.name.cmp(&b.name))
443            .then_with(|| a.kind.cmp(&b.kind))
444            .then_with(|| a.symbol.cmp(&b.symbol))
445    });
446    atlas.contracts.sort_by(|a, b| {
447        contract_confidence(b)
448            .partial_cmp(&contract_confidence(a))
449            .unwrap_or(std::cmp::Ordering::Equal)
450            .then_with(|| a.id.cmp(&b.id))
451    });
452}
453
454#[cfg(test)]
455mod tests {
456    use super::*;
457    use std::collections::BTreeMap;
458
459    #[test]
460    fn term_tokenization() {
461        let t = terms("change street-name normalization for transcripts");
462        assert!(t.contains("street"));
463        assert!(t.contains("name"));
464        assert!(t.contains("transcripts"));
465        assert!(!t.contains("for"));
466        let t2 = terms("raw_text");
467        assert!(t2.contains("raw"));
468        assert!(t2.contains("text"));
469    }
470
471    #[test]
472    fn prefix_match() {
473        assert!(term_match("normalize", "normalizes"));
474        assert!(term_match("transcript", "transcripts"));
475        assert!(term_match("normalizer", "normalize"));
476        assert!(!term_match("cat", "category"));
477        assert!(!term_match("run", "running"));
478    }
479
480    // trace:v1 id=impl.crates-scc-context-src-rank.boost-scorer work=WORK-SI-MMMJA4G6 satisfies=REQ-SI-503JSBGP
481    struct BoostScorer;
482    impl SemanticScorer for BoostScorer {
483        fn score(&self, _goal: &str, e: &scc_core::Entity) -> f64 {
484            if e.name.to_ascii_lowercase().contains("boost") {
485                5.0
486            } else {
487                0.0
488            }
489        }
490    }
491
492    #[test]
493    fn semantic_scorer_fuses_into_candidates() {
494        let dir = tempfile::TempDir::new().unwrap();
495        let root = dir.path().join("repo");
496        std::fs::create_dir_all(&root).unwrap();
497        let store = scc_store::Store::open(&dir.path().join("scc.db"), &root).unwrap();
498        let mut e1 = scc_core::Entity::new("repo://r/symbol/a.py/boosted", "symbol", "boosted");
499        e1.attr("file", serde_json::json!("a.py"));
500        store.insert_entity(&e1, &["a.py".into()]).unwrap();
501        let mut e2 = scc_core::Entity::new("repo://r/symbol/a.py/plain", "symbol", "plain");
502        e2.attr("file", serde_json::json!("a.py"));
503        store.insert_entity(&e2, &["a.py".into()]).unwrap();
504        let graph = scc_graph::RealityGraph::load(&store).unwrap();
505        let view = scc_graph::TrustedGraphView::new(&graph, &store, &[], scc_graph::TrustPolicy::default());
506        let plain = collect_lexical_candidates(&store, &view, "nothing matches", &[], 10);
507        assert_eq!(plain.len(), 0, "lexical-only finds nothing");
508        let fused = collect_lexical_candidates_with(
509            &store, &view, "nothing matches", &[], 10, Some(&BoostScorer),
510        );
511        assert!(
512            fused.iter().any(|c| c.name == "boosted" && c.score >= 5.0),
513            "semantic signal must surface boosted: {fused:?}"
514        );
515        assert!(fused[0].name == "boosted", "boosted ranks first: {fused:?}");
516    }
517
518    #[test]
519    fn similarity_weights_names() {
520        let mut e = scc_core::Entity::new("repo://r/component/x", "component", "transcript-normalizer");
521        e.attr("responsibility", serde_json::json!("normalizes radio transcripts"));
522        let t = terms("normalize transcripts");
523        let s = entity_similarity(&e, &t);
524        assert!(s > 0.0, "got {s}");
525        assert!(s >= 4.0, "expected name+attr hits, got {s}");
526    }
527
528    // ---- Wave 11: startup-atlas confidence ranking ----
529
530    // trace:exempt reason=test-helper
531    fn component(name: &str) -> AtlasComponent {
532        AtlasComponent {
533            name: name.to_string(),
534            purpose: String::new(),
535            implementation: Vec::new(),
536            implementation_paths: Vec::new(),
537            symbols: Vec::new(),
538            consumes: Vec::new(),
539            produces: Vec::new(),
540            upstream: Vec::new(),
541            downstream: Vec::new(),
542            failure_behavior: Vec::new(),
543            owns: Vec::new(),
544            layer: "component".into(),
545            parent: None,
546            role: String::new(),
547        }
548    }
549
550    fn rich_component(name: &str) -> AtlasComponent {
551        let mut c = component(name);
552        c.purpose = "owns the billing pipeline".into();
553        c.implementation = vec!["src/billing".into(), "BillingService".into()];
554        c.symbols = vec!["BillingService".into()];
555        c.consumes = vec!["db.ledger".into()];
556        c.produces = vec!["Invoice".into()];
557        c.upstream = vec!["api".into()];
558        c.downstream = vec!["notifier".into()];
559        c.owns = vec![scc_core::AtlasOwnershipClaim {
560            target: "db.ledger".into(),
561            provenance: "write-edge".into(),
562        }];
563        c.layer = "subsystem".into();
564        c
565    }
566
567    fn empty_atlas() -> SystemAtlas {
568        SystemAtlas {
569            repository: String::new(),
570            revision: String::new(),
571            indexed_at: String::new(),
572            freshness: String::new(),
573            purpose: String::new(),
574            components: Vec::new(),
575            entrypoints: Vec::new(),
576            contracts: Vec::new(),
577            coverage: BTreeMap::new(),
578            flows: Vec::new(),
579            invariants: Vec::new(),
580            deployment_units: Vec::new(),
581            external_systems: Vec::new(),
582            trust_boundaries: Vec::new(),
583            async_boundaries: Vec::new(),
584            implementation_map: BTreeMap::new(),
585            data_stores: Vec::new(),
586            archetype: None,
587            state_authority: BTreeMap::new(),
588            hierarchy: Vec::new(),
589            evidence_summary: BTreeMap::new(),
590            warnings: Vec::new(),
591            public_api: BTreeMap::new(),
592            framework_semantics: BTreeMap::new(),
593            pipeline: Vec::new(),
594            landmarks: Vec::new(),
595        }
596    }
597
598    #[test]
599    fn component_confidence_orders_evidence_over_heuristic() {
600        // a bare-name component is the heuristic floor
601        assert!((component_confidence(&component("zzz_bare")) - 0.2).abs() < 1e-9);
602        // a fully-evidenced component is capped at 1.0, never above
603        assert_eq!(component_confidence(&rich_component("billing")), 1.0);
604        // every graph attribute contributes
605        let mut mid = component("mid");
606        mid.purpose = "p".into();
607        mid.implementation = vec!["src/mid".into()];
608        mid.owns = vec![scc_core::AtlasOwnershipClaim {
609            target: "db.x".into(),
610            provenance: "write-edge".into(),
611        }];
612        let bare = component_confidence(&component("bare"));
613        let mid_score = component_confidence(&mid);
614        let rich_score = component_confidence(&rich_component("billing"));
615        assert!(mid_score > bare, "{mid_score} > {bare}");
616        assert!(rich_score > mid_score, "{rich_score} > {mid_score}");
617        assert!((0.0..=1.0).contains(&mid_score));
618    }
619
620    #[test]
621    fn entrypoint_confidence_ranks_surface_kinds() {
622        let ep = |kind: &str, symbol: &str| AtlasEntrypoint {
623            name: "e".into(),
624            kind: kind.into(),
625            trigger: "t".into(),
626            symbol: symbol.into(),
627        };
628        // http/cli/public_api evidence-backed > generic heuristic
629        assert!(entrypoint_confidence(&ep("route", "s")) > entrypoint_confidence(&ep("entrypoint", "s")));
630        assert!(entrypoint_confidence(&ep("http", "s")) > entrypoint_confidence(&ep("entrypoint", "s")));
631        assert!(entrypoint_confidence(&ep("cli-subcommand", "s")) > entrypoint_confidence(&ep("entrypoint", "s")));
632        assert!(entrypoint_confidence(&ep("public_api", "s")) > entrypoint_confidence(&ep("entrypoint", "s")));
633        // unknown kind is the floor
634        assert!(entrypoint_confidence(&ep("entrypoint", "s")) > entrypoint_confidence(&ep("", "s")));
635        // a resolved symbol id is a weak positive signal (on a kind below
636        // the 1.0 cap, so the bonus is observable)
637        assert!(entrypoint_confidence(&ep("entrypoint", "repo://x/symbol/h")) > entrypoint_confidence(&ep("entrypoint", "")));
638        assert!((0.0..=1.0).contains(&entrypoint_confidence(&ep("route", "s"))));
639    }
640
641    #[test]
642    fn contract_confidence_ranks_surface_subclasses_over_schema() {
643        let contract = |subclass: ContractSubclass, producer: &str, consumers: usize, evidence: usize| Contract {
644            id: "id".into(),
645            kind: subclass.as_str().into(),
646            subclass,
647            producer: producer.into(),
648            consumers: (0..consumers).map(|i| format!("c{i}")).collect(),
649            operations: vec!["op".into()],
650            evidence: (0..evidence).map(|i| format!("ev{i}")).collect(),
651        };
652        let schema = contract(ContractSubclass::Schema, "", 0, 0);
653        let schema_wired = contract(ContractSubclass::Schema, "producer", 2, 2);
654        // http/cli/event/config with producer+consumers beat schema (even a
655        // fully-wired schema): 0.55 max vs 0.95 bare http
656        for subclass in [
657            ContractSubclass::Http,
658            ContractSubclass::Cli,
659            ContractSubclass::Event,
660            ContractSubclass::Configuration,
661        ] {
662            let surface = contract(subclass, "producer", 2, 2);
663            assert!(
664                contract_confidence(&surface) > contract_confidence(&schema_wired),
665                "{:?} must beat schema",
666                subclass
667            );
668        }
669        // wiring raises a contract's confidence
670        assert!(contract_confidence(&schema_wired) > contract_confidence(&schema));
671        assert!((contract_confidence(&schema) - 0.35).abs() < 1e-9);
672        assert!((0.0..=1.0).contains(&contract_confidence(&schema_wired)));
673    }
674
675    #[test]
676    fn rank_startup_atlas_sorts_mixed_evidence_deterministically() {
677        let mut atlas = empty_atlas();
678        atlas.components = vec![
679            rich_component("billing"),
680            component("zzz_bare"),
681            rich_component("aaa_billing_dup"), // same confidence as billing
682            component("aaa_bare"),
683        ];
684        atlas.entrypoints = vec![
685            AtlasEntrypoint { name: "zzz".into(), kind: "entrypoint".into(), trigger: "t".into(), symbol: "".into() },
686            AtlasEntrypoint { name: "api".into(), kind: "route".into(), trigger: "GET /x".into(), symbol: "repo://r/symbol/h".into() },
687            AtlasEntrypoint { name: "cli".into(), kind: "cli-subcommand".into(), trigger: "t".into(), symbol: "".into() },
688        ];
689        atlas.contracts = vec![
690            Contract::new("c-schema", "schema", "").with_subclass(ContractSubclass::Schema),
691            Contract::new("c-http", "http", "repo://r/symbol/h").with_subclass(ContractSubclass::Http),
692            Contract::new("c-cli", "cli", "").with_subclass(ContractSubclass::Cli),
693        ];
694
695        rank_startup_atlas(&mut atlas);
696
697        // components: evidence first, then bare; ties by name lexicographic
698        assert_eq!(atlas.components[0].name, "aaa_billing_dup");
699        assert_eq!(atlas.components[1].name, "billing");
700        assert_eq!(atlas.components[2].name, "aaa_bare");
701        assert_eq!(atlas.components[3].name, "zzz_bare");
702        // every entry kept — precision via ordering, not deletion
703        assert_eq!(atlas.components.len(), 4);
704
705        // entrypoints: evidence-backed surfaces first, heuristic last
706        assert_eq!(atlas.entrypoints[0].kind, "route");
707        assert_eq!(atlas.entrypoints[1].kind, "cli-subcommand");
708        assert_eq!(atlas.entrypoints[2].kind, "entrypoint");
709        assert_eq!(atlas.entrypoints.len(), 3);
710
711        // contracts: http > cli > schema, ties by id
712        assert_eq!(atlas.contracts[0].id, "c-http");
713        assert_eq!(atlas.contracts[1].id, "c-cli");
714        assert_eq!(atlas.contracts[2].id, "c-schema");
715        assert_eq!(atlas.contracts.len(), 3);
716
717        // idempotent + deterministic: re-ranking changes nothing
718        let snapshot = format!("{:?}", atlas);
719        rank_startup_atlas(&mut atlas);
720        assert_eq!(format!("{:?}", atlas), snapshot);
721    }
722}