1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
//! #2023 Increment 5 — Tier-0 shared mean is NATIVE to the ONE fit entry.
//!
//! [`crate::manifold::tests_tier0_shared_mean_2023`] proves the reconstruction /
//! survival mechanics for a HAND-BUILT term with the mean installed by hand. This
//! module proves the missing production half: the single fit entry
//! [`run_sae_manifold_fit`] itself peels the shared column mean off a raw target,
//! runs the whole fit on the de-meaned frame, and hands back a fitted term that
//! SELF-CONTAINS μ — so the tiered schedule's Tier-0 "seed policy" is realized by
//! the primary path, not a separate surface. Two properties, exercised through the
//! exact typed seed→fit pipeline the bindings drive (`examples/sae_fit.rs`):
//!
//! 1. A raw circle target carrying a large global DC gets μ installed on the
//! returned term equal to the column mean, and the reported reconstruction is
//! lifted back to raw-target space (its column mean matches the target's).
//! 2. An already-centered target yields μ ≈ 0 — always-on Tier-0 is a no-op on
//! centered data (μ is computed from THIS target, so there is no
//! double-subtraction hazard), the safety the C4 module documents.
#[cfg(test)]
mod tests {
use crate::manifold::{
SaeFitAssignmentKind, SaeFitConfig, SaeFitRequest, SaeFitSeedReport, SaeFitSeedRequest,
SaeMinimalSeedReport, SaeMinimalSeedRequest, build_sae_fit_seed, build_sae_minimal_seed,
run_sae_manifold_fit,
};
use gam_terms::analytic_penalties::AnalyticPenaltyRegistry;
use ndarray::{Array2, Axis};
/// `N_CIRCLE` evenly-spaced points on the unit circle, plus a per-dim constant
/// `offset` (the global DC Tier-0 must carry) plus small deterministic iid
/// observation noise.
///
/// The noise is load-bearing for identifiability, NOT decoration. A K=1
/// Periodic atom with harmonics reconstructs a clean circle `[cosθ, sinθ]` to
/// machine precision, so the primary fit's pass-0 residual would be ≈0. The
/// always-on structured-residual pass (`fit_entry.rs`, magic-by-default) then
/// fits a residual-covariance model on that ≈0 residual: its idiosyncratic
/// diagonal collapses to the denormal floor, whitening by ~1/D is near-singular,
/// and the whitened-residual REML over the smoothing ρ has NO interior
/// stationary point — so the outer optimizer CORRECTLY refuses to certify (a
/// genuinely non-identifiable structured problem, confirmed by the outer-cert
/// owner). Genuine above-floor residual energy makes that pass well-posed. The
/// evenly-spaced angles sum to zero, so μ stays ≈ OFFSET up to the noise's
/// finite-sample column mean (std `σ/√N`), which the tolerances below reflect.
const N_CIRCLE: usize = 64;
const NOISE_SIGMA: f64 = 0.05;
fn lcg(state: &mut u64) -> f64 {
*state = state
.wrapping_mul(6364136223846793005)
.wrapping_add(1442695040888963407);
((*state >> 11) as f64) / ((1u64 << 53) as f64)
}
fn lcg_normal(state: &mut u64) -> f64 {
let u1 = lcg(state).max(1e-12);
let u2 = lcg(state);
(-2.0 * u1.ln()).sqrt() * (std::f64::consts::TAU * u2).cos()
}
fn circle_target(offset: f64) -> Array2<f64> {
let mut state = 0x2023_0000_0000_0007u64 ^ offset.to_bits();
Array2::from_shape_fn((N_CIRCLE, 2), |(i, j)| {
let theta = std::f64::consts::TAU * (i as f64) / (N_CIRCLE as f64);
let clean = if j == 0 { theta.cos() } else { theta.sin() };
clean + offset + NOISE_SIGMA * lcg_normal(&mut state)
})
}
/// Drive the full typed pipeline (`build_sae_minimal_seed` →
/// `build_sae_fit_seed` → `run_sae_manifold_fit`) on `target`, returning the fit
/// report. Mirrors `examples/sae_fit.rs` with a single periodic atom.
fn run_primary(target: Array2<f64>) -> crate::manifold::SaeFitReport {
let assignment_kind = SaeFitAssignmentKind::Softmax;
let minimal = build_sae_minimal_seed(SaeMinimalSeedRequest {
target: target.view(),
atom_basis: vec!["periodic".to_string()],
atom_dim: vec![1],
assignment_kind,
alpha: 1.0,
tau: 1.0,
threshold: 0.0,
top_k: None,
ibp_alpha_override: None,
random_state: 0,
initial_logits: None,
initial_coords: None,
})
.expect("minimal seed");
let SaeMinimalSeedReport {
atom_basis,
effective_atom_dim,
atom_centers,
basis_values,
basis_jacobian,
basis_sizes,
decoder_coefficients,
smooth_penalties,
initial_logits,
initial_coords,
refine_routing,
} = minimal;
let registry = AnalyticPenaltyRegistry::new();
let seed = build_sae_fit_seed(SaeFitSeedRequest {
target: target.view(),
atom_basis: &atom_basis,
atom_dim: &effective_atom_dim,
atom_centers: &atom_centers,
basis_values: basis_values.view(),
basis_jacobian: basis_jacobian.view(),
basis_sizes: &basis_sizes,
decoder_coefficients: decoder_coefficients.view(),
smooth_penalties: smooth_penalties.view(),
initial_logits: initial_logits.view(),
initial_coords: initial_coords.view(),
alpha: 1.0,
tau: 1.0,
learnable_alpha: false,
assignment_kind,
sparsity_strength: 1.0,
smoothness: 1.0,
max_iter: 4,
learning_rate: 1.0,
ridge_ext_coord: 1.0e-6,
ridge_beta: 1.0e-6,
top_k: None,
threshold: 0.0,
native_ard_enabled: true,
seed_refine_routing: refine_routing,
seed_refine_random_state: 0,
data_row_reseed: false,
fit_config: SaeFitConfig::default(),
temperature_schedule: None,
fisher_metric: None,
row_loss_weights: None,
registry: ®istry,
})
.expect("fit seed");
let SaeFitSeedReport {
base_term,
initial_rho,
isometry_pin_active,
metric_provenance,
} = seed;
run_sae_manifold_fit(SaeFitRequest {
base_term,
target,
registry,
initial_rho,
max_iter: 4,
learning_rate: 1.0,
ridge_ext_coord: 1.0e-6,
ridge_beta: 1.0e-6,
assignment_kind,
alpha: 1.0,
top_k: None,
isometry_pin_active,
metric_provenance,
promote_from_residual: false,
run_structure_search: false,
run_outer_rho_search: false,
cancel: None,
})
.expect("primary fit runs")
}
/// The primary entry peels the shared mean off a RAW target: μ is installed on
/// the returned term (≈ the column mean = OFFSET), and the reported
/// reconstruction is lifted back to raw-target space (its column mean tracks the
/// target's, i.e. the DC is present in `report.fitted`, carried by Tier-0).
#[test]
fn primary_entry_installs_tier0_mean_on_raw_target() {
const OFFSET: f64 = 7.0;
let target = circle_target(OFFSET);
let target_col_mean = target.mean_axis(Axis(0)).unwrap();
let report = run_primary(target.clone());
let mu = report
.term
.tier0_mean()
.expect("primary entry must install a Tier-0 mean on a raw target")
.clone();
assert_eq!(mu.len(), 2, "μ must have one entry per output dim");
for j in 0..2 {
// μ IS the target's column mean by construction (both are `mean_axis`),
// so this stays exact regardless of the noise.
assert!(
(mu[j] - target_col_mean[j]).abs() < 1e-9,
"μ[{j}]={} must equal the target column mean {}",
mu[j],
target_col_mean[j]
);
// The DC dominates the column mean (the evenly-spaced circle coords sum
// to 0), so μ ≈ OFFSET up to the noise's finite-sample column mean
// (std σ/√N = 0.05/8 ≈ 6.3e-3); 0.1 is a comfortable ~16σ/√N band.
assert!(
(mu[j] - OFFSET).abs() < 0.1,
"μ[{j}]={} must be ≈ OFFSET={OFFSET} within the finite-sample noise mean",
mu[j]
);
}
// The reported reconstruction lives in RAW-target space: its column mean
// tracks the target's (the DC is added back by Tier-0), not the de-meaned
// frame the atoms were fit in. With observation noise the fit is no longer
// exact, so the add-back is checked within a finite-sample band rather than
// at machine precision.
let recon_col_mean = report.fitted.mean_axis(Axis(0)).unwrap();
for j in 0..2 {
assert!(
(recon_col_mean[j] - target_col_mean[j]).abs() < 5e-2,
"reconstruction column mean[{j}]={} must track the raw target mean {} \
(Tier-0 add-back), not the de-meaned 0",
recon_col_mean[j],
target_col_mean[j]
);
}
assert!(
report.reconstruction_r2.is_finite(),
"reconstruction R² must be finite"
);
}
/// Always-on Tier-0 is a NO-OP on already-centered data: μ is computed from THIS
/// target, so a mean-zero target yields μ ≈ 0 (no double-subtraction hazard).
#[test]
fn primary_entry_tier0_is_a_noop_on_centered_target() {
let target = circle_target(0.0); // the circle coords are already mean-zero
let report = run_primary(target);
let mu = report
.term
.tier0_mean()
.expect("Tier-0 is installed unconditionally (value ≈ 0 here)")
.clone();
for (j, &m) in mu.iter().enumerate() {
// μ is the target's column mean; with mean-zero circle coords it is just
// the noise's finite-sample column mean (std σ/√N ≈ 6.3e-3) — small, not
// a phantom DC. 0.1 is a comfortable ~16σ/√N band.
assert!(
m.abs() < 0.1,
"μ[{j}]={m} must be ≈ 0 on already-centered data (no phantom mean)"
);
}
}
}