#[path = "support/workspace_root.rs"]
#[allow(dead_code)]
mod workspace_root;
use std::{
collections::{BTreeMap, BTreeSet},
env, fs,
io::{self, Read, Seek, SeekFrom},
path::{Path, PathBuf},
};
const FP16_MIN_SUBNORMAL: f64 = 5.960_464_477_539_063e-8;
const FP16_MIN_NORMAL: f64 = 6.103_515_625e-5;
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum GuardBand {
Normal,
SubnormalOnly,
Inert,
}
impl GuardBand {
fn of(effective: f64) -> Self {
if effective >= FP16_MIN_NORMAL {
GuardBand::Normal
} else if effective >= FP16_MIN_SUBNORMAL {
GuardBand::SubnormalOnly
} else {
GuardBand::Inert
}
}
fn label(self) -> &'static str {
match self {
GuardBand::Normal => "normal",
GuardBand::SubnormalOnly => "subnormal-only",
GuardBand::Inert => "inert",
}
}
}
const MIL_SCALAR_TYPES: &[&str] = &[
"bool", "string", "fp16", "fp32", "fp64", "int8", "int16", "int32", "int64", "uint8", "uint16",
"uint32", "uint64",
];
fn bare_scalar_head(line: &str) -> Option<(&str, &str)> {
let (ty, rest) = line.split_once(' ')?;
MIL_SCALAR_TYPES.contains(&ty).then_some((ty, rest))
}
struct Stmt {
dtype: String,
op: String,
args: String,
attrs: String,
}
struct Graph {
consts: BTreeMap<String, f64>,
producers: BTreeMap<String, Stmt>,
unresolved: Vec<String>,
}
fn parse_hex_float(s: &str) -> Option<f64> {
let s = s.trim();
let (neg, s) = match s.strip_prefix('-') {
Some(rest) => (true, rest),
None => (false, s.strip_prefix('+').unwrap_or(s)),
};
let s = s.strip_prefix("0x").or_else(|| s.strip_prefix("0X"))?;
let (mantissa, exponent) = s.split_once(['p', 'P'])?;
let (int_part, frac_part) = mantissa.split_once('.').unwrap_or((mantissa, ""));
if int_part.is_empty() && frac_part.is_empty() {
return None;
}
let mut value = 0.0_f64;
for c in int_part.chars() {
value = value * 16.0 + f64::from(c.to_digit(16)?);
}
let mut scale = 1.0 / 16.0;
for c in frac_part.chars() {
value += f64::from(c.to_digit(16)?) * scale;
scale /= 16.0;
}
let exp: i32 = exponent.parse().ok()?;
value *= 2.0_f64.powi(exp);
Some(if neg { -value } else { value })
}
fn parse_scalar(tok: &str) -> Option<f64> {
let tok = tok.trim();
if tok.contains("0x") || tok.contains("0X") {
return parse_hex_float(tok);
}
tok.parse::<f64>().ok()
}
fn split_args(args: &str) -> Vec<&str> {
let (mut out, mut depth, mut start) = (Vec::new(), 0_i32, 0_usize);
for (i, c) in args.char_indices() {
match c {
'(' | '[' | '<' => depth += 1,
')' | ']' | '>' => depth -= 1,
',' if depth == 0 => {
out.push(args[start..i].trim());
start = i + 1;
}
_ => {}
}
}
let tail = args[start..].trim();
if !tail.is_empty() {
out.push(tail);
}
out
}
fn arg<'a>(args: &'a str, key: &str) -> Option<&'a str> {
split_args(args).into_iter().find_map(|pair| {
let (k, v) = pair.split_once('=')?;
(k.trim() == key).then(|| v.trim())
})
}
const GUARD_LOOKING_OPS: &[&str] = &[
"log",
"rsqrt",
"sqrt",
"real_div",
"instance_norm",
"layer_norm",
"batch_norm",
"l2_norm",
"add",
"clip",
"maximum",
"softmax",
"exp",
];
const EPSILON_KWARGS: &[&str] = &["epsilon"];
fn is_ident_byte(b: u8) -> bool {
b.is_ascii_alphanumeric() || b == b'_'
}
fn guard_op_in(line: &str) -> Option<&'static str> {
let bytes = line.as_bytes();
GUARD_LOOKING_OPS.iter().copied().find(|&op| {
let mut from = 0;
while let Some(rel) = line[from..].find(op) {
let i = from + rel;
let after = i + op.len();
let before_ok = i == 0 || !is_ident_byte(bytes[i - 1]);
if before_ok && bytes.get(after) == Some(&b'(') {
return true;
}
from = i + 1;
}
false
})
}
struct Parsed {
var: String,
stmt: Stmt,
const_val: Option<f64>,
}
enum ParseOutcome {
NotStatement,
Parsed(Parsed),
Unparsed(Option<&'static str>),
}
fn parse_stmt_line(line: &str) -> ParseOutcome {
let unparsed = || ParseOutcome::Unparsed(guard_op_in(line));
let (dtype, rest) = if let Some(rest) = line.strip_prefix("tensor<") {
let Some((ty, rest)) = rest.split_once('>') else {
return unparsed();
};
(ty.split(',').next().unwrap_or("").trim().to_string(), rest)
} else if let Some((ty, rest)) = bare_scalar_head(line) {
(ty.to_string(), rest)
} else {
return ParseOutcome::NotStatement;
};
let Some((var, rest)) = rest.split_once('=') else {
return unparsed();
};
let var = var.trim();
if var.is_empty() || !var.chars().all(|c| c.is_alphanumeric() || c == '_') {
return unparsed();
}
let rest = rest.trim();
let Some(open) = rest.find('(') else {
return unparsed();
};
let op = rest[..open].trim().to_string();
let mut depth = 0_i32;
let mut close = None;
for (i, c) in rest[open..].char_indices() {
match c {
'(' => depth += 1,
')' => {
depth -= 1;
if depth == 0 {
close = Some(open + i);
break;
}
}
_ => {}
}
}
let Some(close) = close else {
let guard = GUARD_LOOKING_OPS
.iter()
.copied()
.find(|&g| g == op.as_str());
return ParseOutcome::Unparsed(guard);
};
let args = rest[open + 1..close].to_string();
let attrs = rest[close + 1..].trim().to_string();
let const_val = (op == "const").then(|| const_scalar(&attrs)).flatten();
ParseOutcome::Parsed(Parsed {
var: var.to_string(),
stmt: Stmt {
dtype,
op,
args,
attrs,
},
const_val,
})
}
fn parse_mil(text: &str) -> Graph {
let mut consts = BTreeMap::new();
let mut producers = BTreeMap::new();
let mut unresolved = Vec::new();
for line in text.lines() {
let line = line.trim();
match parse_stmt_line(line) {
ParseOutcome::NotStatement => {}
ParseOutcome::Parsed(Parsed {
var,
stmt,
const_val,
}) => {
if let Some(value) = const_val {
consts.insert(var.clone(), value);
}
producers.insert(var, stmt);
}
ParseOutcome::Unparsed(Some(op)) => {
unresolved.push(format!("unreadable `{op}` statement: {line}"));
}
ParseOutcome::Unparsed(None) => {}
}
}
Graph {
consts,
producers,
unresolved,
}
}
fn const_scalar(attrs: &str) -> Option<f64> {
let val = attrs.find("val")?;
let open = attrs[val..].find("(")? + val;
let head = attrs[val..open].replace(' ', "");
let scalar_tensor = head.contains(",[]>");
let bare_scalar = head
.rsplit_once("val=")
.is_some_and(|(_, ty)| MIL_SCALAR_TYPES.contains(&ty));
if !(scalar_tensor || bare_scalar) {
return None;
}
let close = attrs[open..].find(')')? + open;
parse_scalar(&attrs[open + 1..close])
}
const BLOB_SENTINEL: u32 = 0xdead_beef;
const BLOB_DTYPE_FP16: u32 = 1;
const BLOB_DTYPE_FP32: u32 = 2;
const BLOB_MAX_BYTES: u64 = 512 * 1024 * 1024;
struct BlobRef {
path: String,
offset: u64,
}
fn blob_ref(attrs: &str) -> Option<BlobRef> {
let rest = &attrs[attrs.find("BLOBFILE(")? + "BLOBFILE(".len()..];
let open_quote = rest.find('"')?;
let close_quote = open_quote + 1 + rest[open_quote + 1..].find('"')?;
let path = rest[open_quote + 1..close_quote].to_string();
let after = &rest[close_quote..];
let offset_key = after.find("offset")?;
let open = offset_key + after[offset_key..].find('(')?;
let close = open + after[open..].find(')')?;
let offset = after[open + 1..close].trim().parse().ok()?;
Some(BlobRef { path, offset })
}
fn read_blob_as_fp16(bundle: &Path, blob: &BlobRef) -> Result<Vec<half::f16>, String> {
let rel = blob.path.strip_prefix("@model_path/").unwrap_or(&blob.path);
let path = bundle.join(rel);
let mut file = fs::File::open(&path).map_err(|e| format!("open {}: {e}", path.display()))?;
let mut meta = [0_u8; 24];
file
.seek(SeekFrom::Start(blob.offset))
.and_then(|_| file.read_exact(&mut meta))
.map_err(|e| {
format!(
"read blob metadata at offset {} of {}: {e}",
blob.offset,
path.display()
)
})?;
let word = |at: usize| u32::from_le_bytes(meta[at..at + 4].try_into().expect("4 bytes"));
let long = |at: usize| u64::from_le_bytes(meta[at..at + 8].try_into().expect("8 bytes"));
let sentinel = word(0);
if sentinel != BLOB_SENTINEL {
return Err(format!(
"blob metadata at offset {} of {} begins {sentinel:#010x}, not the {BLOB_SENTINEL:#010x} \
sentinel — the offset or the blob layout is not what this reader assumes",
blob.offset,
path.display()
));
}
let dtype = word(4);
let width = match dtype {
BLOB_DTYPE_FP16 => 2_u64,
BLOB_DTYPE_FP32 => 4,
other => {
return Err(format!(
"blob at offset {} of {} declares dtype {other}, which this reader does not narrow to \
fp16 (it knows {BLOB_DTYPE_FP16} = fp16 and {BLOB_DTYPE_FP32} = fp32)",
blob.offset,
path.display()
));
}
};
let size = long(8);
if size > BLOB_MAX_BYTES || size % width != 0 {
return Err(format!(
"blob at offset {} of {} declares a {size}-byte payload, which is not a sane multiple of \
its {width}-byte element",
blob.offset,
path.display()
));
}
let mut raw = vec![0_u8; usize::try_from(size).map_err(|e| format!("{size} bytes: {e}"))?];
file
.seek(SeekFrom::Start(long(16)))
.and_then(|_| file.read_exact(&mut raw))
.map_err(|e| {
format!(
"read {size} bytes of payload at offset {} of {}: {e}",
long(16),
path.display()
)
})?;
Ok(match dtype {
BLOB_DTYPE_FP16 => raw
.as_chunks::<2>()
.0
.iter()
.map(|c| half::f16::from_le_bytes(*c))
.collect(),
_ => raw
.as_chunks::<4>()
.0
.iter()
.map(|c| half::f16::from_f32(f32::from_le_bytes(*c)))
.collect(),
})
}
struct ConstVarianceNorm {
var: String,
dtype: String,
eps: f64,
blob: BlobRef,
}
impl Graph {
fn value(&self, tok: Option<&str>) -> Option<f64> {
let tok = tok?;
parse_scalar(tok).or_else(|| self.consts.get(tok).copied())
}
fn const_through_cast(&self, tok: Option<&str>, depth: u8) -> Option<f64> {
let tok = tok?;
if let Some(v) = self.value(Some(tok)) {
return Some(v);
}
if depth > 6 {
return None;
}
match self.producers.get(tok) {
Some(stmt) if stmt.op == "cast" => self.const_through_cast(arg(&stmt.args, "x"), depth + 1),
_ => None,
}
}
fn producer_through_cast(&self, tok: Option<&str>, depth: u8) -> Option<&Stmt> {
let tok = tok?;
if depth > 6 {
return None;
}
match self.producers.get(tok) {
Some(stmt) if stmt.op == "cast" => {
self.producer_through_cast(arg(&stmt.args, "x"), depth + 1)
}
other => other,
}
}
fn unreadable_floor_guard(&self, operand: Option<&str>) -> bool {
self
.producer_through_cast(operand, 0)
.is_some_and(|stmt| matches!(stmt.op.as_str(), "add" | "maximum" | "clip"))
}
fn floor(&self, var: Option<&str>, depth: u8) -> Option<(f64, String)> {
let var = var?;
if depth > 6 {
return None;
}
let stmt = self.producers.get(var)?;
match stmt.op.as_str() {
"const" => self.consts.get(var).map(|v| (*v, format!("const({v:e})"))),
"add" => ["y", "x"]
.iter()
.find_map(|k| self.const_through_cast(arg(&stmt.args, k), 0))
.map(|c| (c, format!("add(+{c:e})"))),
"clip" => self
.const_through_cast(arg(&stmt.args, "alpha"), 0)
.map(|lo| (lo, format!("clip(alpha={lo:e})"))),
"maximum" => ["y", "x"]
.iter()
.find_map(|k| self.const_through_cast(arg(&stmt.args, k), 0))
.map(|c| (c, format!("maximum({c:e})"))),
"softmax" => Some((0.0, "softmax->log".to_string())),
"cast" => self.floor(arg(&stmt.args, "x"), depth + 1),
_ => None,
}
}
fn audit(&self) -> Audit {
let mut found = Vec::new();
let mut unresolved = self.unresolved.clone();
let mut const_variance_norms = Vec::new();
for (var, stmt) in &self.producers {
let eps_kwarg = self.value(arg(&stmt.args, "epsilon"));
let variance_const = self
.producer_through_cast(arg(&stmt.args, "variance"), 0)
.filter(|producer| producer.op == "const");
if let ("batch_norm", Some(eps), Some(producer)) =
(stmt.op.as_str(), eps_kwarg, variance_const)
{
match blob_ref(&producer.attrs) {
Some(blob) => const_variance_norms.push(ConstVarianceNorm {
var: var.clone(),
dtype: stmt.dtype.clone(),
eps,
blob,
}),
None if const_scalar(&producer.attrs).is_none() => {
unresolved.push(format!(
"unreadable constant `variance` on batch_norm/{} {var}: {}",
stmt.dtype, producer.attrs
));
}
None => {}
}
}
let site = match stmt.op.as_str() {
"log" | "rsqrt" => match eps_kwarg {
Some(eps) => {
let (floor, guard) = self
.floor(arg(&stmt.args, "x"), 0)
.unwrap_or((0.0, "-".into()));
Some((eps, floor, guard))
}
None => {
unresolved.push(unresolved_site(var, stmt));
None
}
},
"instance_norm" | "layer_norm" | "batch_norm" | "l2_norm" => match eps_kwarg {
Some(e) => Some((e, 0.0, "norm".into())),
None => {
unresolved.push(unresolved_site(var, stmt));
None
}
},
"sqrt" => match self.floor(arg(&stmt.args, "x"), 0) {
Some((f, g)) => Some((0.0, f, g)),
None => {
if self.unreadable_floor_guard(arg(&stmt.args, "x")) {
unresolved.push(unresolved_site(var, stmt));
}
None
}
},
"real_div" => match self.floor(arg(&stmt.args, "y"), 0) {
Some((f, g)) => Some((0.0, f, format!("denom:{g}"))),
None => {
if self.unreadable_floor_guard(arg(&stmt.args, "y")) {
unresolved.push(unresolved_site(var, stmt));
}
None
}
},
_ if EPSILON_KWARGS.iter().any(|k| arg(&stmt.args, k).is_some()) => {
unresolved.push(unresolved_site(var, stmt));
None
}
_ => None,
};
if let Some((eps, floor, guard)) = site {
found.push(Finding {
op: stmt.op.clone(),
dtype: stmt.dtype.clone(),
var: var.clone(),
eps,
floor,
guard,
});
}
}
Audit {
findings: found,
unresolved,
const_variance_norms,
}
}
}
struct Audit {
findings: Vec<Finding>,
unresolved: Vec<String>,
const_variance_norms: Vec<ConstVarianceNorm>,
}
fn unresolved_site(var: &str, stmt: &Stmt) -> String {
format!(
"unresolvable epsilon on {}/{} {var}: {}({})",
stmt.op, stmt.dtype, stmt.op, stmt.args
)
}
struct Finding {
op: String,
dtype: String,
var: String,
eps: f64,
floor: f64,
guard: String,
}
impl Finding {
fn effective(&self) -> f64 {
self.eps.max(self.floor)
}
fn band(&self) -> GuardBand {
GuardBand::of(self.effective())
}
fn survives_fp16(&self) -> bool {
self.band() != GuardBand::Inert
}
fn is_decomposed_log_softmax(&self) -> bool {
self.op == "log" && self.guard == "softmax->log"
}
fn render(&self) -> String {
format!(
"{}/{} guard={} eff={:e}",
self.op,
self.dtype,
self.guard,
self.effective()
)
}
}
struct KnownDefect {
path: &'static str,
sites: &'static [&'static str],
note: &'static str,
}
const KNOWN_DEFECTS: &[KnownDefect] = &[
KnownDefect {
path: "alignkit/base960h_aligner.mlmodelc",
sites: &["log/fp16 guard=softmax->log eff=1.401298464324817e-45"],
note: "Decomposed log-softmax; eps 0x1p-149 rounds to 0 in fp16. `emissions` IS the log \
output, so ANE log(0) -> ~-45440 lands directly in the shipped tensor: 16.7% of \
output cells corrupted, word timings shifted up to 881 ms.",
},
KnownDefect {
path: "speakerkit/Segmentation.mlmodelc",
sites: &["log/fp32 guard=softmax->log eff=1.401298464324817e-45"],
note: "Same source model and same coremltools default epsilon as pyannote_segmentation; the \
fp32 artifact merely KEEPS 0x1p-149 rather than folding it to zero. That is fp32's \
smallest subnormal — it survives fp32 arithmetic and nothing else. Loaded under the \
default ComputeUnits::All it is demoted to fp16 on the ANE and vanishes exactly like \
its fp16 sibling.",
},
KnownDefect {
path: "speakerkit/wespeaker_v2.mlmodelc",
sites: &[
"real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9",
"real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9",
"real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9",
],
note: "Attentive-stat pooling divides by `count + 1e-8` at THREE sites (the weighted mean \
and the two divisions behind the weighted variance/std). 1e-8 is 0.168x fp16's \
smallest subnormal, so on the ANE all three divisor guards are zero. This is the int8 \
embedder issue #15 RETIRED from shipping (its per-tensor palettization silently \
collapses 8-speaker audio — tests/speaker/model_io.rs's DECISION); it stays on disk \
as the tested sibling the factorial record runs on, guards unrepaired. The published \
re-conversion is NOT adopted for it: that artifact is also a RE-PALETTIZATION \
(different LUTs), and it moves clip 14's int8 ANE arm from 0.8178 % to 1.4860 % DER. \
The fp32 `wespeaker.mlmodelc` the pipeline ships instead carries these SAME pooling \
guards repaired to 0x1p-24 — which is why the shipping embedder has no entry here.",
},
KnownDefect {
path: "speakerkit/wespeaker_int8.mlmodelc",
sites: &[
"real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9",
"real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9",
"real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9",
],
note: "Same three-site pooling epsilon as wespeaker_v2.mlmodelc (byte-identical artifact).",
},
KnownDefect {
path: "speakerkit/PLDA.mlmodelc",
sites: &[
"sqrt/fp32 guard=clip(alpha=9.999999960041972e-13) eff=9.999999960041972e-13",
"sqrt/fp32 guard=clip(alpha=9.999999960041972e-13) eff=9.999999960041972e-13",
],
note: "Normalization clips to 1e-12 before `sqrt` at TWO sites, then divides by it. 1e-12 is \
1.7e-5x fp16's smallest subnormal, so IF these ops are lowered to fp16 the clip floor \
becomes zero, giving sqrt(0) and a divide by zero. That premise is UNTESTED: this sweep \
reads the MIL text STATICALLY, nothing loads these graphs, and no run has observed \
their actual placement or whether ComputeUnits::All demotes them. THIS finding is why \
the runtime projects with diaric's f64 `PldaTransform` (weights `include_bytes!`d, host \
arithmetic, no compute-unit demotion) instead of the CoreML PLDA the same model repo \
ships — see `Extraction::into_offline_input`. The graph arrives with the vendor tree \
whatever this crate decides, so the pin stays: it is what forces a re-conversion that \
repairs the epsilon to be SEEN, rather than leaving the rejection as folklore a future \
contributor would have to rediscover.",
},
KnownDefect {
path: "speakerkit/PldaRho.mlmodelc",
sites: &[
"sqrt/fp32 guard=clip(alpha=9.999999960041972e-13) eff=9.999999960041972e-13",
"sqrt/fp32 guard=clip(alpha=9.999999960041972e-13) eff=9.999999960041972e-13",
],
note: "Same two-site 1e-12 clip floor as PLDA.mlmodelc, unloaded for the same reason.",
},
KnownDefect {
path: "argmax-speakerkit/speaker_segmenter/pyannote-v3/W32A32/SpeakerSegmenter.mlmodelc",
sites: &["log/fp16 guard=softmax->log eff=0e0"],
note: "Vendored from argmax. Epsilon already folded to 0x0p+0, and the graph is fp16 \
DESPITE the W32A32 directory name. Contained, not silent-clean: the saturated log \
feeds an `exp` that maps it back toward 0 before any shipped output, and the winning \
powerset class never underflows, so `speaker_probs`/`speaker_ids` survive. The guard \
is still inert — pinned so a re-vendored graph cannot widen the blast radius unseen.",
},
KnownDefect {
path: "argmax-speakerkit/speaker_segmenter/pyannote-v3/W8A16/SpeakerSegmenter.mlmodelc",
sites: &["log/fp16 guard=softmax->log eff=1.401298464324817e-45"],
note: "Same graph as the W32A32 variant with the epsilon left at 0x1p-149 instead of folded \
to zero — identically inert in fp16, identically contained by the downstream `exp`.",
},
KnownDefect {
path: "lid/SpeechBrainECAPAVoxLingua107.mlmodelc",
sites: &["log/fp16 guard=softmax->log eff=1.401298464324817e-45"],
note: "The THIRD instance of this class, after alignkit/base960h_aligner and \
speakerkit/Segmentation: a decomposed log-softmax at the classifier tail, guarded by \
fp32's smallest subnormal (0x1p-149) against an fp16 log whose floor is 2^-24 ~ 6e-8. \
TODAY IT COSTS NOTHING, AND THE REASON IS THAT THE EPSILON IS NEVER CONSULTED. The \
shipped graph is FLEXIBLE-SHAPE, and on every current backend — All, ANE, GPU, CpuOnly \
— the softmax->log pair executes as one fused, higher-precision log-softmax. Measured \
on both census clips, all four arms: 107 of 107 rows finite, no saturation, the tail \
reaching -37.27 (a probability of 6.5e-17, NINE orders below fp16's smallest subnormal \
— a number the fp16 grid cannot hold, which is the proof the log is not being taken in \
fp16), and sum(exp) = 0.994-1.000. `Error::NonFiniteOutput` cannot fire. The epsilon is \
dead code. THE TRIGGER THAT ARMS IT IS STATIC SHAPES — exactly what a re-conversion \
would do to put this tail on the ANE. Recompiled with fixed input dims the fusion is \
gone, and every entry whose true log-prob is below ln(2^-24) = -16.635 becomes ANE \
log(0) = -45440.0. On the 13 s anchor clip that is 105 of 107 entries, starting at RANK \
2, and the fp64 reference agrees exactly on WHICH 105. Top-1 survives essentially \
intact (-0.009842 against an fp64 truth of -0.010195, delta 3.5e-4) and so does top-2, \
so the blast radius is confidence-gated: on a flat 3 s clip it is 0 of 107. -45440.0 is \
FINITE, so nothing upstack notices — the door's non-finite guard stays silent, \
`LanguageScore::probability` reads 0.0, and `identify(k >= 3)` fills ranks 2+ by \
ascending model column, the order `RankedScore` breaks ties in. TWO OBVIOUS REPAIRS ARE \
ALREADY KNOWN NOT TO WORK — measured on that static ANE arm, and recorded here because \
without it the first of them looks correct AND appears to succeed. (1) A `HEALTHY` fp16 \
EPSILON — an explicit +6e-8, the whisper-mel discipline the issue-#15 \
speakerkit re-conversions used — does NOT survive: the ANE flushes subnormals and the output is -45440.0, \
unchanged. A floor that does survive must be at least 6.1e-5, fp16's smallest NORMAL, \
which clamps the whole tail at -9.7 and is uselessly coarse. (2) The PYANNOTE-STYLE `x \
- logsumexp(x)` rewrite also fails on the ANE at this graph's logit range: exp(~23) \
overflows fp16 inside the reduce and the output comes back as raw logits (max +22.86, \
sum(exp) 8.6e9) — worse than the defect. The ONLY verified ANE-safe tail is an fp32 \
ISLAND: the final softmax+log excluded from the fp16 pass (8 CPU ops, 2 unit \
transitions, latency unchanged at ~23.6 ms), which reproduces the flexible graph's full \
finite tail (-37.28) on the ANE. Pinned unrepaired because the shipping artifact is \
flexible-shape and the guard is therefore inert; the pin is what makes the arming event \
impossible to miss, since a re-conversion to static shapes cannot land without this \
entry changing.",
},
];
struct ZeroVarianceSite {
dtype: String,
eps: f64,
channels: usize,
zeros: usize,
}
impl ZeroVarianceSite {
fn render(&self) -> String {
format!(
"batch_norm/{} eps={:e} band={} zero={}/{}",
self.dtype,
self.eps,
GuardBand::of(self.eps).label(),
self.zeros,
self.channels
)
}
}
struct LoadBearingNorm {
path: &'static str,
sites: &'static [&'static str],
channels: usize,
zero_channels: usize,
note: &'static str,
}
const LOAD_BEARING_NORMS: &[LoadBearingNorm] = &[LoadBearingNorm {
path: "lid/SpeechBrainECAPAVoxLingua107.mlmodelc",
sites: &[
"batch_norm/fp16 eps=1.0013580322265625e-5 band=subnormal-only zero=1/1024",
"batch_norm/fp16 eps=1.0013580322265625e-5 band=subnormal-only zero=144/6144",
"batch_norm/fp16 eps=1.0013580322265625e-5 band=subnormal-only zero=4/1024",
"batch_norm/fp16 eps=1.0013580322265625e-5 band=subnormal-only zero=4/3072",
"batch_norm/fp16 eps=1.0013580322265625e-5 band=subnormal-only zero=6/1024",
],
channels: 19_968,
zero_channels: 159,
note: "SpeechBrain's ECAPA carries PyTorch's BatchNorm1d default epsilon, 1e-5, rounded to \
fp16 as 0x1.5p-17 = 1.0014e-5 — 168x fp16's smallest subnormal, but 0.164x its \
smallest NORMAL, so it exists only as a subnormal. Across the 33 `batch_norm` sites \
the artifact's own `running_var` blobs hold 19 968 channels, of which 159 are exactly \
0.0 in fp16 (0.80 %), spread over five layers — 144 of them in `asp_bn`, the 6 144-wide \
statistics-pooling norm feeding the embedding. Those 159 channels are guarded by the \
epsilon and by nothing else. MEASURED, on this exact artifact (SHA-pinned by \
tests/lid/common/mod.rs), by re-emitting the graph with the two epsilon constants \
changed and running all four ComputeUnits arms on three inputs: at 0x0p+0 every arm \
returns 107 of 107 NaN — the falsifier, red, so the guard is genuinely load-bearing; \
at the SHIPPING 0x1.5p-17, and at 0x1p-24 (fp16's SMALLEST subnormal, the gate's own \
floor), every arm returns 107 of 107 finite log-probabilities with the same top-1, and \
0x1p-24 / 0x1p-23 / 0x1p-20 / 0x1p-15 / 0x1p-14 each give a DISTINCT output, so the \
value is being consulted at full fp16 resolution rather than flushed. MLComputePlan \
places 29 of the 33 `batch_norm` ops on the ANE under the default `All`. So the \
subnormal is honoured here, and the reason is structural: `variance` and `epsilon` are \
BOTH constants, so `variance + epsilon` is folded before any fp16 kernel runs — unlike \
the `log` epsilon pinned for this same artifact in KNOWN_DEFECTS, whose guarded operand \
is a runtime tensor and which the ANE was measured to flush. Pinned unrepaired because \
the guard holds; the pin is what makes a re-conversion that drops, folds or re-rounds \
this epsilon — or one that clamps `running_var` and removes the exposure — impossible \
to land unseen.",
}];
fn pinned_paths() -> BTreeSet<&'static str> {
KNOWN_DEFECTS
.iter()
.map(|d| d.path)
.chain(LOAD_BEARING_NORMS.iter().map(|d| d.path))
.collect()
}
const ALIGNKIT_LOG_SOFTMAX: &str = r#"
tensor<int32, []> var_847 = const()[name = tensor<string, []>("op_847"), val = tensor<int32, []>(-1)];
tensor<fp16, [1, 2999, 29]> var_849_softmax_cast_fp16 = softmax(axis = var_847, x = linear_73_cast_fp16)[name = tensor<string, []>("op_849_softmax_cast_fp16")];
tensor<fp32, []> var_849_epsilon_0 = const()[name = tensor<string, []>("op_849_epsilon_0"), val = tensor<fp32, []>(0x1p-149)];
tensor<fp16, [1, 2999, 29]> var_849_cast_fp16 = log(epsilon = var_849_epsilon_0, x = var_849_softmax_cast_fp16)[name = tensor<string, []>("op_849_cast_fp16")];
"#;
const SPEAKERKIT_SEG_FP16: &str = r#"
tensor<fp16, [1, 589, 7]> var_231_softmax_cast_fp16 = softmax(axis = var_230, x = linear_2_cast_fp16)[name = tensor<string, []>("op_231_softmax_cast_fp16")];
tensor<fp16, []> var_231_epsilon_0_to_fp16 = const()[name = tensor<string, []>("op_231_epsilon_0_to_fp16"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [1, 589, 7]> var_231_cast_fp16 = log(epsilon = var_231_epsilon_0_to_fp16, x = var_231_softmax_cast_fp16)[name = tensor<string, []>("op_231_cast_fp16")];
"#;
const WESPEAKER_POOLING: &str = r#"
tensor<fp32, []> var_5790 = const()[name = tensor<string, []>("op_5790"), val = tensor<fp32, []>(0x1.5798eep-27)];
tensor<fp32, [3, 1]> v1 = add(x = var_5789, y = var_5790)[name = tensor<string, []>("v1")];
tensor<fp32, [3, 2560]> mean = real_div(x = var_5794, y = v1)[name = tensor<string, []>("mean")];
"#;
const WHISPER_MEL: &str = r#"
tensor<fp16, []> var_41_to_fp16 = const()[name = tensor<string, []>("op_41_to_fp16"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, [80, 3000]> mel_spec_cast_fp16 = add(x = mel_spec_1_cast_fp16, y = var_41_to_fp16)[name = tensor<string, []>("mel_spec_cast_fp16")];
tensor<fp16, []> log_0_epsilon_0_to_fp16 = const()[name = tensor<string, []>("log_0_epsilon_0_to_fp16"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [80, 3000]> log_0_cast_fp16 = log(epsilon = log_0_epsilon_0_to_fp16, x = mel_spec_cast_fp16)[name = tensor<string, []>("log_0_cast_fp16")];
"#;
const VADKIT_STFT_SQRT: &str = r#"
tensor<fp16, []> var_201_promoted_to_fp16 = const()[name = tensor<string, []>("op_201_promoted_to_fp16"), val = tensor<fp16, []>(0x1p+1)];
tensor<fp16, [1, 129, 4]> var_227_cast_fp16 = pow(x = var_222_cast_fp16, y = var_201_promoted_to_fp16)[name = tensor<string, []>("op_227_cast_fp16")];
tensor<fp16, []> var_201_promoted_1_to_fp16 = const()[name = tensor<string, []>("op_201_promoted_1_to_fp16"), val = tensor<fp16, []>(0x1p+1)];
tensor<fp16, [1, 129, 4]> var_228_cast_fp16 = pow(x = var_225_cast_fp16, y = var_201_promoted_1_to_fp16)[name = tensor<string, []>("op_228_cast_fp16")];
tensor<fp16, [1, 129, 4]> var_229_cast_fp16 = add(x = var_227_cast_fp16, y = var_228_cast_fp16)[name = tensor<string, []>("op_229_cast_fp16")];
tensor<fp16, []> var_230_to_fp16 = const()[name = tensor<string, []>("op_230_to_fp16"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, [1, 129, 4]> var_231_cast_fp16 = add(x = var_229_cast_fp16, y = var_230_to_fp16)[name = tensor<string, []>("op_231_cast_fp16")];
tensor<fp16, [1, 129, 4]> input_3_cast_fp16 = sqrt(x = var_231_cast_fp16)[name = tensor<string, []>("input_3_cast_fp16")];
"#;
const CLAPKIT_NORMS_CLEAN: &str = r#"
tensor<fp16, []> var_7_to_fp16 = const()[name = tensor<string, []>("op_7_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 64, 1001, 1]> normalized_input_features_cast_fp16 = batch_norm(beta = audio_model_audio_encoder_batch_norm_bias_to_fp16, epsilon = var_7_to_fp16, gamma = audio_model_audio_encoder_batch_norm_weight_to_fp16, mean = audio_model_audio_encoder_batch_norm_running_mean_to_fp16, variance = audio_model_audio_encoder_batch_norm_running_var_to_fp16, x = input_1_cast_fp16)[name = tensor<string, []>("normalized_input_features_cast_fp16")];
tensor<fp16, []> var_23_to_fp16 = const()[name = tensor<string, []>("op_23_to_fp16"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, [1, 512, 768]> input_5_cast_fp16 = layer_norm(axes = input_5_axes_0, beta = text_model_embeddings_LayerNorm_bias_to_fp16, epsilon = var_23_to_fp16, gamma = text_model_embeddings_LayerNorm_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("input_5_cast_fp16")];
"#;
const CLAPKIT_NORM_VANISHING_MUTANT: &str = r#"
tensor<fp16, []> var_23_to_fp16 = const()[name = tensor<string, []>("op_23_to_fp16"), val = tensor<fp16, []>(0x1p-25)];
tensor<fp16, [1, 512, 768]> input_5_cast_fp16 = layer_norm(axes = input_5_axes_0, beta = text_model_embeddings_LayerNorm_bias_to_fp16, epsilon = var_23_to_fp16, gamma = text_model_embeddings_LayerNorm_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("input_5_cast_fp16")];
"#;
const GRANITE_NORMS_CLEAN: &str = r#"
tensor<fp16, []> var_42_to_fp16 = const()[name = tensor<string, []>("op_42_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 512, 384]> input_5_cast_fp16 = layer_norm(axes = input_5_axes_0, epsilon = var_42_to_fp16, gamma = embeddings_norm_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("input_5_cast_fp16")];
tensor<fp16, []> var_95_to_fp16 = const()[name = tensor<string, []>("op_95_to_fp16"), val = tensor<fp16, []>(0x1.5p-17)];
tensor<fp16, [1, 512, 384]> input_15_cast_fp16 = layer_norm(axes = input_15_axes_0, epsilon = var_95_to_fp16, gamma = layers_0_mlp_norm_weight_to_fp16, x = input_13_cast_fp16)[name = tensor<string, []>("input_15_cast_fp16")];
"#;
const GRANITE_NORM_VANISHING_MUTANT: &str = r#"
tensor<fp16, []> var_42_to_fp16 = const()[name = tensor<string, []>("op_42_to_fp16"), val = tensor<fp16, []>(0x1p-25)];
tensor<fp16, [1, 512, 384]> input_5_cast_fp16 = layer_norm(axes = input_5_axes_0, epsilon = var_42_to_fp16, gamma = embeddings_norm_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("input_5_cast_fp16")];
"#;
const SIGLIP_NORMS_CLEAN: &str = r#"
tensor<fp16, []> var_63_to_fp16 = const()[name = tensor<string, []>("op_63_to_fp16"), val = tensor<fp16, []>(0x1.1p-20)];
tensor<fp16, [1, 512, 768]> input_7_cast_fp16 = layer_norm(axes = input_7_axes_0, beta = encoder_layers_0_layer_norm2_bias_to_fp16, epsilon = var_63_to_fp16, gamma = encoder_layers_0_layer_norm2_weight_to_fp16, x = input_5_cast_fp16)[name = tensor<string, []>("input_7_cast_fp16")];
tensor<fp16, []> var_11_to_fp16 = const()[name = tensor<string, []>("op_11_to_fp16"), val = tensor<fp16, []>(0x1.1p-20)];
tensor<fp16, [1, 64, 768]> hidden_states_1_cast_fp16 = layer_norm(axes = hidden_states_1_axes_0, beta = text_model_encoder_layers_0_layer_norm1_bias_to_fp16, epsilon = var_11_to_fp16, gamma = text_model_encoder_layers_0_layer_norm1_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("hidden_states_1_cast_fp16")];
"#;
const SIGLIP_NORM_VANISHING_MUTANT: &str = r#"
tensor<fp16, []> var_11_to_fp16 = const()[name = tensor<string, []>("op_11_to_fp16"), val = tensor<fp16, []>(0x1p-25)];
tensor<fp16, [1, 64, 768]> hidden_states_1_cast_fp16 = layer_norm(axes = hidden_states_1_axes_0, beta = text_model_encoder_layers_0_layer_norm1_bias_to_fp16, epsilon = var_11_to_fp16, gamma = text_model_encoder_layers_0_layer_norm1_weight_to_fp16, x = input_3_cast_fp16)[name = tensor<string, []>("hidden_states_1_cast_fp16")];
"#;
const TWO_HALF_SUBNORMAL_GUARDS: &str = r#"
tensor<fp16, []> half_add_c = const()[name = tensor<string, []>("half_add_c"), val = tensor<fp16, []>(0x1p-25)];
tensor<fp16, [80, 3000]> half_guarded = add(x = feat_in, y = half_add_c)[name = tensor<string, []>("half_guarded")];
tensor<fp16, []> half_log_eps = const()[name = tensor<string, []>("half_log_eps"), val = tensor<fp16, []>(0x1p-25)];
tensor<fp16, [80, 3000]> half_log = log(epsilon = half_log_eps, x = half_guarded)[name = tensor<string, []>("half_log")];
"#;
const TWO_SITE_LOG_SOFTMAX: &str = r#"
tensor<fp16, [1, 589, 7]> a_softmax_cast_fp16 = softmax(axis = a_axis, x = a_linear)[name = tensor<string, []>("a_softmax")];
tensor<fp16, []> a_epsilon = const()[name = tensor<string, []>("a_epsilon"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [1, 589, 7]> a_cast_fp16 = log(epsilon = a_epsilon, x = a_softmax_cast_fp16)[name = tensor<string, []>("a_log")];
tensor<fp16, [1, 589, 7]> b_softmax_cast_fp16 = softmax(axis = b_axis, x = b_linear)[name = tensor<string, []>("b_softmax")];
tensor<fp16, []> b_epsilon = const()[name = tensor<string, []>("b_epsilon"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [1, 589, 7]> b_cast_fp16 = log(epsilon = b_epsilon, x = b_softmax_cast_fp16)[name = tensor<string, []>("b_log")];
"#;
const VALID_GUARD_PLUS_UNREADABLE_GUARD: &str = r#"
tensor<fp16, []> ok_eps = const()[name = tensor<string, []>("ok_eps"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, [80, 3000]> ok_mel = add(x = ok_mel_1, y = ok_eps)[name = tensor<string, []>("ok_mel")];
tensor<fp16, []> ok_log_eps = const()[name = tensor<string, []>("ok_log_eps"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [80, 3000]> ok_log = log(epsilon = ok_log_eps, x = ok_mel)[name = tensor<string, []>("ok_log")];
tensor<fp16, [1, 589, 7]> bad_softmax = softmax(axis = bad_axis, x = bad_linear)[name = tensor<string, []>("bad_softmax")];
tensor<fp16, [1, 589, 7]> bad_log = log(epsilon = bad_eps, x = bad_softmax [name = tensor<string, []>("bad_log")];
"#;
const NORM_WITH_UNRESOLVABLE_EPSILON: &str = r#"
tensor<fp16, [1, 384, 1, 1500]> n_out = batch_norm(beta = n_beta, epsilon = n_eps_missing, gamma = n_gamma, mean = n_mean, variance = n_var, x = n_in)[name = tensor<string, []>("n_out")];
"#;
const CAST_WRAPPED_POOLING_DIVISOR: &str = r#"
tensor<fp32, []> eps_fp32 = const()[name = tensor<string, []>("eps_fp32"), val = tensor<fp32, []>(0x1.5798eep-27)];
tensor<fp16, []> eps_fp16 = cast(dtype = fp16, x = eps_fp32)[name = tensor<string, []>("eps_fp16")];
tensor<fp16, [3, 1]> v1 = add(x = count_cast_fp16, y = eps_fp16)[name = tensor<string, []>("v1")];
tensor<fp16, [3, 2560]> mean = real_div(x = numer_cast_fp16, y = v1)[name = tensor<string, []>("mean")];
"#;
const DYNAMIC_UNRESOLVABLE_DIVISOR: &str = r#"
tensor<fp16, [3, 1]> dyn_eps = mul(x = a_cast_fp16, y = b_cast_fp16)[name = tensor<string, []>("dyn_eps")];
tensor<fp16, [3, 1]> v1 = add(x = count_cast_fp16, y = dyn_eps)[name = tensor<string, []>("v1")];
tensor<fp16, [3, 2560]> mean = real_div(x = numer_cast_fp16, y = v1)[name = tensor<string, []>("mean")];
"#;
const CAST_WRAPPED_DYNAMIC_DIVISOR: &str = r#"
tensor<fp16, []> ok_eps = const()[name = tensor<string, []>("ok_eps"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, [80, 3000]> ok_mel = add(x = ok_mel_1, y = ok_eps)[name = tensor<string, []>("ok_mel")];
tensor<fp16, []> ok_log_eps = const()[name = tensor<string, []>("ok_log_eps"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [80, 3000]> ok_log = log(epsilon = ok_log_eps, x = ok_mel)[name = tensor<string, []>("ok_log")];
tensor<fp16, [3, 1]> dyn_eps = mul(x = a_cast_fp16, y = b_cast_fp16)[name = tensor<string, []>("dyn_eps")];
tensor<fp16, [3, 1]> v1 = add(x = count_cast_fp16, y = dyn_eps)[name = tensor<string, []>("v1")];
tensor<fp16, [3, 1]> v1_fp16 = cast(dtype = fp16, x = v1)[name = tensor<string, []>("v1_fp16")];
tensor<fp16, [3, 2560]> mean = real_div(x = numer_cast_fp16, y = v1_fp16)[name = tensor<string, []>("mean")];
"#;
const L2_NORM_VANISHING: &str = r#"
tensor<fp16, []> ok_eps = const()[name = tensor<string, []>("ok_eps"), val = tensor<fp16, []>(0x1p-24)];
tensor<fp16, [80, 3000]> ok_mel = add(x = ok_mel_1, y = ok_eps)[name = tensor<string, []>("ok_mel")];
tensor<fp16, []> ok_log_eps = const()[name = tensor<string, []>("ok_log_eps"), val = tensor<fp16, []>(0x0p+0)];
tensor<fp16, [80, 3000]> ok_log = log(epsilon = ok_log_eps, x = ok_mel)[name = tensor<string, []>("ok_log")];
tensor<fp32, []> l2_eps = const()[name = tensor<string, []>("l2_eps"), val = tensor<fp32, []>(0x1.5798eep-27)];
tensor<fp16, [1, 256]> l2_out = l2_norm(epsilon = l2_eps, x = embed_cast_fp16)[name = tensor<string, []>("l2_out")];
"#;
const EPSILON_BEARING_UNKNOWN_OP: &str = r#"
tensor<fp16, []> mystery_eps = const()[name = tensor<string, []>("mystery_eps"), val = tensor<fp16, []>(0x1p-30)];
tensor<fp16, [1, 256]> mystery_out = some_future_norm(epsilon = mystery_eps, x = in_cast_fp16)[name = tensor<string, []>("mystery_out")];
"#;
const LID_ECAPA_SCALAR_DIALECT: &str = r#"
fp16 var_23_to_fp16 = const()[name = string("op_23_to_fp16"), val = fp16(0x1.5p-17)];
tensor<fp16, [1, 1024, ?]> input_7_cast_fp16 = batch_norm(beta = embedding_model_blocks_0_norm_norm_bias_to_fp16, epsilon = var_23_to_fp16, gamma = embedding_model_blocks_0_norm_norm_weight_to_fp16, mean = embedding_model_blocks_0_norm_norm_running_mean_to_fp16, variance = embedding_model_blocks_0_norm_norm_running_var_to_fp16, x = x_3_cast_fp16)[name = string("input_7_cast_fp16")];
fp16 var_13_to_fp16 = const()[name = string("op_13_to_fp16"), val = fp16(0x1p-24)];
fp16 const_37_to_fp16 = const()[name = string("const_37_to_fp16"), val = fp16(inf)];
tensor<fp16, [?, 3072]> clip_0_cast_fp16 = clip(alpha = var_13_to_fp16, beta = const_37_to_fp16, x = var_835_cast_fp16)[name = string("clip_0_cast_fp16")];
tensor<fp16, [?, 3072]> std_1_cast_fp16 = sqrt(x = clip_0_cast_fp16)[name = string("std_1_cast_fp16")];
tensor<fp16, [1, 107]> x_act_softmax_cast_fp16 = softmax(axis = var_914, x = input_cast_fp16)[name = string("x_act_softmax_cast_fp16")];
fp32 x_act_epsilon_0 = const()[name = string("x_act_epsilon_0"), val = fp32(0x1p-149)];
tensor<fp16, [1, 107]> x_act_cast_fp16 = log(epsilon = x_act_epsilon_0, x = x_act_softmax_cast_fp16)[name = string("x_act_cast_fp16")];
"#;
const LID_SCALAR_NORM_VANISHING_MUTANT: &str = r#"
fp16 var_23_to_fp16 = const()[name = string("op_23_to_fp16"), val = fp16(0x1p-25)];
tensor<fp16, [1, 1024, ?]> input_7_cast_fp16 = batch_norm(beta = embedding_model_blocks_0_norm_norm_bias_to_fp16, epsilon = var_23_to_fp16, gamma = embedding_model_blocks_0_norm_norm_weight_to_fp16, mean = embedding_model_blocks_0_norm_norm_running_mean_to_fp16, variance = embedding_model_blocks_0_norm_norm_running_var_to_fp16, x = x_3_cast_fp16)[name = string("input_7_cast_fp16")];
"#;
const LID_NON_STATEMENT_LINES: &[&str] = &[
"program(1.3)",
r#"[buildInfo = dict<string, string>({{"coremltools-version", "9.0"}})]"#,
"{",
r#" func main<ios18>(tensor<fp32, [1, ?, 60]> mel_features) [FlexibleShapeInformation = tuple<tuple<string, dict<string, tensor<int32, [?]>>>, tuple<string, dict<string, list<tensor<int32, [2]>, ?>>>>((("DefaultShapes", {{"mel_features", [1, 301, 60]}}), ("RangeDims", {{"mel_features", [[1, 1], [10, 3001], [60, 60]]}})))] {"#,
"} -> (log_probabilities);",
"}",
];
const LID_BATCH_NORM: &str = r#"
tensor<fp16, [1024]> embedding_model_blocks_0_norm_norm_running_mean_to_fp16 = const()[name = string("embedding_model_blocks_0_norm_norm_running_mean_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(616640)))];
tensor<fp16, [1024]> embedding_model_blocks_0_norm_norm_running_var_to_fp16 = const()[name = string("embedding_model_blocks_0_norm_norm_running_var_to_fp16"), val = tensor<fp16, [1024]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(618752)))];
fp16 var_23_to_fp16 = const()[name = string("op_23_to_fp16"), val = fp16(0x1.5p-17)];
tensor<fp16, [1, 1024, ?]> input_7_cast_fp16 = batch_norm(beta = embedding_model_blocks_0_norm_norm_bias_to_fp16, epsilon = var_23_to_fp16, gamma = embedding_model_blocks_0_norm_norm_weight_to_fp16, mean = embedding_model_blocks_0_norm_norm_running_mean_to_fp16, variance = embedding_model_blocks_0_norm_norm_running_var_to_fp16, x = x_3_cast_fp16)[name = string("input_7_cast_fp16")];
"#;
const LID_VARIANCE_BLOB_OFFSET: u64 = 618_752;
fn vanishing_sites(findings: &[Finding]) -> Vec<String> {
let mut sites: Vec<String> = findings
.iter()
.filter(|f| !f.survives_fp16())
.map(Finding::render)
.collect();
sites.sort();
sites
}
fn vanishing(mil: &str) -> Vec<String> {
let audit = parse_mil(mil).audit();
assert!(
audit.unresolved.is_empty(),
"audit left guard-looking statement(s) unresolved: {:?}",
audit.unresolved
);
vanishing_sites(&audit.findings)
}
#[test]
fn threshold_is_fp16s_smallest_subnormal() {
assert_eq!(
FP16_MIN_SUBNORMAL,
2.0_f64.powi(-24),
"the gate's threshold must be exactly 2^-24"
);
assert_eq!(FP16_MIN_NORMAL, 2.0_f64.powi(-14));
assert_eq!(
half::f16::from_f64(2.0_f64.powi(-149)),
half::f16::from_f64(0.0),
"0x1p-149 must round to zero in fp16"
);
assert_eq!(half::f16::from_f64(1e-8), half::f16::from_f64(0.0));
assert_eq!(half::f16::from_f64(1e-12), half::f16::from_f64(0.0));
assert!(
half::f16::from_f64(FP16_MIN_SUBNORMAL) > half::f16::from_f64(0.0),
"2^-24 must be representable in fp16"
);
}
#[test]
fn hex_float_literals_parse_exactly() {
assert_eq!(parse_hex_float("0x1p-149"), Some(2.0_f64.powi(-149)));
assert_eq!(parse_hex_float("0x0p+0"), Some(0.0));
assert_eq!(parse_hex_float("0x1p-24"), Some(FP16_MIN_SUBNORMAL));
let eight = parse_hex_float("0x1.5798eep-27").expect("parses");
assert!(
(eight - 1e-8).abs() < 1e-15,
"0x1.5798eep-27 ~= 1e-8, got {eight:e}"
);
let twelve = parse_hex_float("0x1.197998p-40").expect("parses");
assert!(
(twelve - 1e-12).abs() < 1e-19,
"0x1.197998p-40 ~= 1e-12, got {twelve:e}"
);
}
#[test]
fn detects_the_alignkit_log_softmax_defect() {
assert_eq!(
vanishing(ALIGNKIT_LOG_SOFTMAX),
["log/fp16 guard=softmax->log eff=1.401298464324817e-45"],
"alignkit's fp16 log(eps = 0x1p-149) must be caught"
);
let graph = parse_mil(ALIGNKIT_LOG_SOFTMAX);
let audit = graph.audit();
assert!(
audit.unresolved.is_empty(),
"the real alignkit excerpt must parse completely: {:?}",
audit.unresolved
);
let log = audit
.findings
.iter()
.find(|f| f.op == "log")
.expect("a log site");
assert!(!log.survives_fp16());
assert!(
log.is_decomposed_log_softmax(),
"and it must be recognized as a decomposed log_softmax"
);
}
#[test]
fn detects_the_wespeaker_pooling_defect() {
assert_eq!(
vanishing(WESPEAKER_POOLING),
["real_div/fp32 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9"],
"wespeaker's `count + 1e-8` divisor guard must be caught even though \
the graph declares fp32 — the ANE demotes it to fp16 regardless"
);
}
#[test]
fn detects_the_speakerkit_segmentation_defect() {
assert_eq!(
vanishing(SPEAKERKIT_SEG_FP16),
["log/fp16 guard=softmax->log eff=0e0"],
"an epsilon coremltools already folded to 0x0p+0 must be caught"
);
}
#[test]
fn accepts_whisperkits_mel_guard() {
assert_eq!(
vanishing(WHISPER_MEL),
Vec::<String>::new(),
"whisper's mel add(x, 0x1p-24) is exactly at the fp16 floor and survives"
);
let graph = parse_mil(WHISPER_MEL);
let audit = graph.audit();
assert!(
audit.unresolved.is_empty(),
"the real whisper-mel excerpt must parse completely: {:?}",
audit.unresolved
);
let log = audit
.findings
.iter()
.find(|f| f.op == "log")
.expect("a log site");
assert_eq!(log.eps, 0.0, "the log's OWN epsilon is 0x0p+0 here");
assert_eq!(
log.floor, FP16_MIN_SUBNORMAL,
"the guard is the preceding add, not the log's epsilon"
);
assert!(log.survives_fp16());
assert!(
!log.is_decomposed_log_softmax(),
"it logs a mel spectrogram, not a softmax"
);
}
#[test]
fn reads_the_coremltools_9_bare_scalar_dialect() {
let graph = parse_mil(LID_ECAPA_SCALAR_DIALECT);
assert_eq!(
graph.consts.get("var_23_to_fp16").copied(),
Some(1.0013580322265625e-5),
"the batch_norm epsilon is declared `fp16 …`, not `tensor<fp16, []> …`"
);
assert_eq!(
graph.consts.get("var_13_to_fp16").copied(),
Some(FP16_MIN_SUBNORMAL)
);
assert_eq!(
graph.consts.get("const_37_to_fp16").copied(),
Some(f64::INFINITY)
);
assert_eq!(
graph.consts.get("x_act_epsilon_0").copied(),
Some(2.0_f64.powi(-149)),
"and an fp32 scalar is spelled bare too, in the same graph as `tensor<fp16, [1, 107]>`"
);
assert_eq!(
vanishing(LID_ECAPA_SCALAR_DIALECT),
["log/fp16 guard=softmax->log eff=1.401298464324817e-45"],
"exactly one of the three sites vanishes: the classifier tail's decomposed \
log-softmax. The batch_norm's 1.001e-5 and the clip's 0x1p-24 both clear the floor."
);
let audit = graph.audit();
assert_eq!(
audit.findings.len(),
3,
"batch_norm, sqrt and log — a reader that saw only `tensor<` statements found NONE of \
them and reported three unresolved holes instead: {:?}",
audit
.findings
.iter()
.map(Finding::render)
.collect::<Vec<_>>()
);
let sqrt = audit
.findings
.iter()
.find(|f| f.op == "sqrt")
.expect("a sqrt site");
assert_eq!(
sqrt.floor, FP16_MIN_SUBNORMAL,
"the attentive-stat sqrt is floored by clip(alpha = 0x1p-24), read through a bare scalar"
);
assert!(sqrt.survives_fp16());
}
#[test]
fn a_vanishing_bare_scalar_epsilon_is_caught() {
assert_eq!(
vanishing(LID_SCALAR_NORM_VANISHING_MUTANT),
["batch_norm/fp16 guard=norm eff=2.9802322387695313e-8"],
"a bare-scalar epsilon below the fp16 floor must fail exactly like a `tensor<fp16, []>` one"
);
}
#[test]
fn the_bare_scalar_arm_does_not_swallow_the_program_header() {
for line in LID_NON_STATEMENT_LINES {
assert!(
matches!(parse_stmt_line(line.trim()), ParseOutcome::NotStatement),
"not a declaration, and must not be read as one: {line}"
);
}
assert_eq!(
bare_scalar_head("fp16 var_23_to_fp16 = const()[]"),
Some(("fp16", "var_23_to_fp16 = const()[]")),
"…while a real scalar declaration still is one"
);
assert_eq!(
bare_scalar_head("float16 v = const()[]"),
None,
"the vocabulary is closed: a near-miss type name is not a declaration"
);
}
#[test]
fn accepts_vadkits_stft_sqrt_guard() {
assert_eq!(
vanishing(VADKIT_STFT_SQRT),
Vec::<String>::new(),
"vadkit's STFT add(x, 0x1p-24) -> sqrt is exactly at the fp16 floor and survives"
);
let graph = parse_mil(VADKIT_STFT_SQRT);
let audit = graph.audit();
assert!(
audit.unresolved.is_empty(),
"the real vadkit STFT excerpt must parse completely: {:?}",
audit.unresolved
);
let sqrt = audit
.findings
.iter()
.find(|f| f.op == "sqrt")
.expect("a sqrt site");
assert_eq!(sqrt.eps, 0.0, "sqrt carries no epsilon of its own");
assert_eq!(
sqrt.floor, FP16_MIN_SUBNORMAL,
"the guard is the preceding add(0x1p-24), not any epsilon on the sqrt"
);
assert!(sqrt.survives_fp16());
}
#[test]
fn accepts_clapkit_conversion_norm_guards() {
assert_eq!(
vanishing(CLAPKIT_NORMS_CLEAN),
Vec::<String>::new(),
"clapkit's audio 0x1.5p-17 and text 0x1p-24 norm guards both survive fp16"
);
let audit = parse_mil(CLAPKIT_NORMS_CLEAN).audit();
assert!(
audit.unresolved.is_empty(),
"the real clapkit norm excerpts must parse completely: {:?}",
audit.unresolved
);
assert_eq!(
audit.findings.len(),
2,
"one batch_norm + one layer_norm site"
);
assert!(audit.findings.iter().all(|f| f.survives_fp16()));
assert_eq!(
vanishing(CLAPKIT_NORM_VANISHING_MUTANT),
["layer_norm/fp16 guard=norm eff=2.9802322387695313e-8"],
"a clapkit norm eps halved to 0x1p-25 (below the fp16 floor) must be flagged"
);
}
#[test]
fn accepts_granite_conversion_norm_guards() {
assert_eq!(
vanishing(GRANITE_NORMS_CLEAN),
Vec::<String>::new(),
"granite's 0x1.5p-17 layer_norm guards all survive fp16"
);
let audit = parse_mil(GRANITE_NORMS_CLEAN).audit();
assert!(
audit.unresolved.is_empty(),
"the real granite norm excerpts must parse completely: {:?}",
audit.unresolved
);
assert_eq!(audit.findings.len(), 2, "two layer_norm sites");
assert!(audit.findings.iter().all(|f| f.survives_fp16()));
assert!(
audit.findings.iter().all(|f| f.op == "layer_norm"),
"granite carries only layer_norm guard sites"
);
assert_eq!(
vanishing(GRANITE_NORM_VANISHING_MUTANT),
["layer_norm/fp16 guard=norm eff=2.9802322387695313e-8"],
"a granite norm eps halved to 0x1p-25 (below the fp16 floor) must be flagged"
);
}
#[test]
fn accepts_siglip_conversion_norm_guards() {
assert_eq!(
vanishing(SIGLIP_NORMS_CLEAN),
Vec::<String>::new(),
"siglip's 0x1.1p-20 vision+text layer_norm guards all survive fp16"
);
let audit = parse_mil(SIGLIP_NORMS_CLEAN).audit();
assert!(
audit.unresolved.is_empty(),
"the real siglip norm excerpts must parse completely: {:?}",
audit.unresolved
);
assert_eq!(
audit.findings.len(),
2,
"one vision + one text layer_norm site"
);
assert!(audit.findings.iter().all(|f| f.survives_fp16()));
assert!(
audit.findings.iter().all(|f| f.op == "layer_norm"),
"siglip carries only layer_norm guard sites"
);
assert_eq!(
vanishing(SIGLIP_NORM_VANISHING_MUTANT),
["layer_norm/fp16 guard=norm eff=2.9802322387695313e-8"],
"a siglip norm eps halved to 0x1p-25 (below the fp16 floor) must be flagged"
);
}
#[test]
fn two_sub_threshold_guards_do_not_sum_into_survival() {
assert_eq!(
vanishing(TWO_HALF_SUBNORMAL_GUARDS),
["log/fp16 guard=add(+2.9802322387695313e-8) eff=2.9802322387695313e-8"],
"add(0x1p-25) feeding log(eps=0x1p-25): each constant is half the fp16 floor \
and rounds to zero on its own, so the guard vanishes — the two must not sum \
past the floor"
);
assert_eq!(
vanishing(WHISPER_MEL),
Vec::<String>::new(),
"the MAX rule must keep whisper-mel clean (add's 2^-24 survives on its own)"
);
}
#[test]
fn same_signature_sites_are_a_multiset_not_a_set() {
let sites = vanishing(TWO_SITE_LOG_SOFTMAX);
assert_eq!(
sites.len(),
2,
"two independent softmax->log sites must render as TWO findings, not collapse to one — got {sites:?}"
);
assert_eq!(
sites[0], sites[1],
"...and they share a signature, which is exactly what a set/dedup would fold away"
);
assert_eq!(sites[0], "log/fp16 guard=softmax->log eff=0e0");
}
#[test]
fn a_partial_parse_fails_the_audit_never_reports_clean() {
let audit = parse_mil(VALID_GUARD_PLUS_UNREADABLE_GUARD).audit();
assert!(
audit
.findings
.iter()
.any(|f| f.op == "log" && f.survives_fp16()),
"the valid mel-style guard must still be audited and survive"
);
assert!(
!audit.unresolved.is_empty(),
"an unreadable guard statement must fail the audit, not vanish — got no unresolved holes"
);
assert!(
audit
.unresolved
.iter()
.any(|u| u.contains("log") && u.contains("bad_log")),
"the unresolved report must quote the offending `log` statement: {:?}",
audit.unresolved
);
}
#[test]
fn a_norm_with_an_unreadable_epsilon_is_a_hole_not_a_skip() {
let audit = parse_mil(NORM_WITH_UNRESOLVABLE_EPSILON).audit();
assert!(
audit.findings.is_empty(),
"an unresolvable-epsilon norm yields no resolved finding, got: {:?}",
audit
.findings
.iter()
.map(Finding::render)
.collect::<Vec<_>>()
);
assert!(
audit
.unresolved
.iter()
.any(|u| u.contains("batch_norm") && u.contains("n_out")),
"...but it must be reported unresolved, quoting the statement — not dropped: {:?}",
audit.unresolved
);
}
#[test]
fn an_l2_norm_with_a_vanishing_epsilon_is_a_finding_not_a_wildcard_drop() {
assert_eq!(
vanishing(L2_NORM_VANISHING),
["l2_norm/fp16 guard=norm eff=9.99999993922529e-9"],
"l2_norm's epsilon is its whole divide guard; 1e-8 rounds to zero in fp16 and \
must surface as a vanishing finding, not drop through the wildcard"
);
}
#[test]
fn an_epsilon_bearing_unknown_op_is_a_hole_not_a_wildcard_drop() {
let audit = parse_mil(EPSILON_BEARING_UNKNOWN_OP).audit();
assert!(
audit.findings.is_empty(),
"an unrecognized op yields no resolved finding, got: {:?}",
audit
.findings
.iter()
.map(Finding::render)
.collect::<Vec<_>>()
);
assert!(
audit
.unresolved
.iter()
.any(|u| u.contains("some_future_norm") && u.contains("mystery_out")),
"an unknown op carrying an `epsilon` kwarg must be surfaced as unresolved, quoting the \
statement — not dropped through the wildcard: {:?}",
audit.unresolved
);
}
#[test]
fn follows_a_cast_wrapped_pooling_divisor_guard() {
assert_eq!(
vanishing(CAST_WRAPPED_POOLING_DIVISOR),
["real_div/fp16 guard=denom:add(+9.99999993922529e-9) eff=9.99999993922529e-9"],
"a `count + cast(1e-8)` divisor guard must be caught THROUGH the cast, \
rendering identically to the direct-const wespeaker pooling guard"
);
}
#[test]
fn an_unresolvable_add_guarded_divisor_is_a_hole_not_a_drop() {
let audit = parse_mil(DYNAMIC_UNRESOLVABLE_DIVISOR).audit();
assert!(
audit.findings.is_empty(),
"an unresolvable divisor guard yields no resolved finding, got: {:?}",
audit
.findings
.iter()
.map(Finding::render)
.collect::<Vec<_>>()
);
assert!(
audit
.unresolved
.iter()
.any(|u| u.contains("real_div") && u.contains("mean")),
"the unreadable `add`-guarded divisor must be surfaced as unresolved, quoting \
the statement — not dropped: {:?}",
audit.unresolved
);
}
#[test]
fn follows_a_cast_before_an_unresolvable_divisor_guard() {
let audit = parse_mil(CAST_WRAPPED_DYNAMIC_DIVISOR).audit();
assert!(
audit
.findings
.iter()
.any(|f| f.op == "log" && f.survives_fp16()),
"the valid mel-style guard must still be audited and survive"
);
assert!(
audit
.unresolved
.iter()
.any(|u| u.contains("real_div") && u.contains("mean")),
"the cast-wrapped unresolvable divisor guard must be surfaced as unresolved, \
quoting the statement — not dropped: {:?}",
audit.unresolved
);
}
#[test]
fn discover_propagates_read_dir_errors() {
let mut out = Vec::new();
let missing = models_dir().join("__no_such_subtree_for_the_walk__");
assert!(
discover(&missing, &mut out).is_err(),
"read_dir on a nonexistent path must return Err, not an empty Ok"
);
assert!(out.is_empty(), "a failed walk collects nothing");
}
fn models_dir() -> PathBuf {
workspace_root::models_root()
}
fn discover(root: &Path, out: &mut Vec<PathBuf>) -> io::Result<()> {
let entries = fs::read_dir(root)
.map_err(|e| io::Error::new(e.kind(), format!("read_dir {}: {e}", root.display())))?;
for entry in entries {
let entry = entry
.map_err(|e| io::Error::new(e.kind(), format!("dir entry under {}: {e}", root.display())))?;
let path = entry.path();
if !path.is_dir() {
continue;
}
if entry.file_name().to_string_lossy().starts_with('.') {
continue;
}
if path.extension().is_some_and(|e| e == "mlmodelc") {
out.push(path);
} else {
discover(&path, out)?;
}
}
Ok(())
}
fn vendor_of(path: &str) -> &str {
path.split('/').next().unwrap_or(path)
}
fn expected_vendors() -> BTreeSet<String> {
vendor_manifest(env::var("COREMLIT_FP16_SWEEP_VENDORS").ok().as_deref())
}
fn vendor_manifest(raw: Option<&str>) -> BTreeSet<String> {
let Some(raw) = raw else {
return pinned_paths()
.into_iter()
.map(|path| vendor_of(path).to_string())
.collect();
};
let vendors: BTreeSet<String> = raw
.split(',')
.map(str::trim)
.filter(|s| !s.is_empty())
.map(String::from)
.collect();
assert!(
!vendors.is_empty(),
"COREMLIT_FP16_SWEEP_VENDORS is set to {raw:?} but names no vendor — a present-but-empty \
override (empty, whitespace, or comma-only) would require NO vendor and silently re-open the \
whole-vendor-deletion escape the manifest exists to close. Unset it to require every \
KNOWN_DEFECTS vendor, or name the vendors to audit (e.g. whisperkit-coreml)."
);
vendors
}
struct SweepOutcome {
models_len: usize,
audited_sites: usize,
bands: BTreeMap<&'static str, usize>,
variance_channels: usize,
zero_variance_channels: usize,
failures: Vec<String>,
}
fn sweep_tree(root: &Path, expected_vendors: &BTreeSet<String>) -> io::Result<SweepOutcome> {
let mut models = Vec::new();
discover(root, &mut models)?;
models.sort();
let pins: BTreeMap<&str, &KnownDefect> = KNOWN_DEFECTS.iter().map(|d| (d.path, d)).collect();
let load_bearing: BTreeMap<&str, &LoadBearingNorm> =
LOAD_BEARING_NORMS.iter().map(|d| (d.path, d)).collect();
let mut audited_sites = 0_usize;
let mut bands: BTreeMap<&'static str, usize> = BTreeMap::new();
let mut variance_channels = 0_usize;
let mut zero_variance_channels = 0_usize;
let mut failures = Vec::new();
let mut seen = Vec::new();
for model in &models {
let rel = model
.strip_prefix(root)
.expect("discovered under root")
.to_string_lossy()
.replace('\\', "/");
seen.push(rel.clone());
let mil = model.join("model.mil");
let text = match fs::read_to_string(&mil) {
Ok(text) => text,
Err(e) => {
failures.push(format!("{rel}: .mlmodelc has no readable model.mil ({e})"));
continue;
}
};
let Audit {
findings,
unresolved,
const_variance_norms,
} = parse_mil(&text).audit();
if !unresolved.is_empty() {
failures.push(format!(
"{rel}: {} guard-looking statement(s) the reader could not resolve — a partial parse \
fails the sweep rather than dropping a guard. Re-convert with a readable guard or teach \
the reader this shape:\n {}",
unresolved.len(),
unresolved.join("\n ")
));
}
if findings.is_empty() {
failures.push(format!(
"{rel}: parsed zero guard sites from a {} byte graph — the parser has rotted",
text.len()
));
continue;
}
audited_sites += findings.len();
for finding in &findings {
*bands.entry(finding.band().label()).or_default() += 1;
}
let mut zero_variance: Vec<String> = Vec::new();
let mut model_channels = 0_usize;
let mut model_zero_channels = 0_usize;
for norm in &const_variance_norms {
match read_blob_as_fp16(model, &norm.blob) {
Ok(values) => {
let zeros = values.iter().filter(|v| v.to_f32() == 0.0).count();
model_channels += values.len();
model_zero_channels += zeros;
if zeros > 0 {
zero_variance.push(
ZeroVarianceSite {
dtype: norm.dtype.clone(),
eps: norm.eps,
channels: values.len(),
zeros,
}
.render(),
);
}
}
Err(e) => failures.push(format!(
"{rel}: batch_norm {} declares a constant `variance` this reader could not read \
({e}). What its epsilon is worth depends on that tensor, so an unreadable one is a \
hole, not a pass.",
norm.var
)),
}
}
zero_variance.sort();
variance_channels += model_channels;
zero_variance_channels += model_zero_channels;
let vanishing = vanishing_sites(&findings);
let decomposed: Vec<&Finding> = findings
.iter()
.filter(|f| f.is_decomposed_log_softmax() && f.survives_fp16())
.collect();
for f in decomposed {
failures.push(format!(
"{rel}: {} ({}) is a decomposed log_softmax. Its epsilon survives fp16, but the \
softmax output underflows to 0 BEFORE the log adds it, clamping the true log-prob \
at log(eps). Convert with a fused, stable log_softmax (x - logsumexp(x)).",
f.var,
f.render()
));
}
match pins.get(rel.as_str()) {
Some(pin) => {
let expected: Vec<String> = pin.sites.iter().map(|s| (*s).to_string()).collect();
if vanishing != expected {
if vanishing.is_empty() {
failures.push(format!(
"{rel}: PINNED KNOWN DEFECT IS FIXED.\n was: {expected:?}\n now: clean.\n \
If this model was re-converted, that is good news — but it must be seen: delete \
its KNOWN_DEFECTS entry and re-cut the parity goldens deliberately.\n Pin note: \
{}",
pin.note
));
} else {
failures.push(format!(
"{rel}: pinned defect CHANGED.\n expected: {expected:?}\n found: \
{vanishing:?}\n Pin note: {}",
pin.note
));
}
}
}
None if !vanishing.is_empty() => {
failures.push(format!(
"{rel}: NEW fp16-vanishing guard in an unpinned model: {vanishing:?}\n Every \
epsilon here is below fp16's smallest subnormal ({FP16_MIN_SUBNORMAL:e}), so it \
rounds to zero and the guard goes inert on the ANE/GPU. Re-convert with an epsilon \
>= 2^-24, or pin it in KNOWN_DEFECTS with a note saying what breaks."
));
}
None => {}
}
match load_bearing.get(rel.as_str()) {
Some(pin) => {
let expected: Vec<String> = pin.sites.iter().map(|s| (*s).to_string()).collect();
if (&zero_variance, model_channels, model_zero_channels)
!= (&expected, pin.channels, pin.zero_channels)
{
failures.push(format!(
"{rel}: pinned LOAD-BEARING epsilon sites CHANGED.\n expected: {expected:?} over \
{}/{} channels\n found: {zero_variance:?} over {model_zero_channels}/\
{model_channels} channels\n A zero-variance channel is guarded by its epsilon \
and by nothing else, so this changing in EITHER direction is a change in what the \
artifact depends on — a REPAIR included, which must be seen and the pin retired \
deliberately. Pin note: {}",
pin.zero_channels, pin.channels, pin.note
));
}
}
None if !zero_variance.is_empty() => {
failures.push(format!(
"{rel}: NEW load-bearing epsilon in an unpinned model: {zero_variance:?}\n These \
`batch_norm` sites hold channels whose stored variance is exactly 0.0 in fp16, so \
`sqrt(variance + epsilon)` reduces to `sqrt(epsilon)` and the epsilon is the entire \
denominator. That can clear the {FP16_MIN_SUBNORMAL:e} floor and still be a single \
point of failure. Measure what the site does with the epsilon set to zero, then pin \
it in LOAD_BEARING_NORMS with that measurement."
));
}
None => {}
}
}
for vendor in expected_vendors {
if !root.join(vendor).is_dir() {
failures.push(format!(
"expected vendor Models/{vendor}/ is MISSING — its pinned known-defect models cannot be \
verified, and deleting an entire vendor must never silently disable its pins. Restore \
the vendor tree, or narrow COREMLIT_FP16_SWEEP_VENDORS for a deliberately partial tree."
));
}
}
for path in pinned_paths() {
let vendor = vendor_of(path);
if root.join(vendor).is_dir() && !seen.iter().any(|s| s == path) {
failures.push(format!(
"{path}: pinned model is MISSING, but Models/{vendor}/ is present. The pin cannot be \
verified. Restore the model or remove the pin."
));
}
}
Ok(SweepOutcome {
models_len: models.len(),
audited_sites,
bands,
variance_channels,
zero_variance_channels,
failures,
})
}
#[cfg_attr(
not(models_present),
ignore = "no downloaded model tree on disk — nothing to sweep beyond the committed \
vadkit artifact (build.rs)"
)]
#[test]
fn every_shipped_model_graph_survives_fp16() {
let root = models_dir();
assert!(
root.is_dir(),
"Models/ vanished between build and run: {}",
root.display()
);
let outcome = sweep_tree(&root, &expected_vendors())
.unwrap_or_else(|e| panic!("walking Models/ failed instead of silently skipping: {e}"));
assert!(
outcome.models_len > 0,
"Models/ exists but contains no .mlmodelc — the sweep would be vacuous"
);
assert!(
outcome.audited_sites > 0,
"swept {} models and audited zero guard sites — vacuous",
outcome.models_len
);
assert_eq!(
outcome.bands.values().sum::<usize>(),
outcome.audited_sites,
"the band census {:?} does not account for all {} audited sites",
outcome.bands,
outcome.audited_sites
);
assert!(
outcome.failures.is_empty(),
"fp16 guard sweep failed over {} models / {} guard sites (bands: {:?}; constant batch_norm \
variance: {} of {} channels zero in fp16):\n\n{}\n",
outcome.models_len,
outcome.audited_sites,
outcome.bands,
outcome.zero_variance_channels,
outcome.variance_channels,
outcome.failures.join("\n\n")
);
}
struct TempTree(PathBuf);
impl TempTree {
fn new(tag: &str) -> Self {
let uniq = format!(
"coremlit_fp16_sweep_{tag}_{}_{}",
std::process::id(),
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.expect("clock after epoch")
.as_nanos()
);
let root = env::temp_dir().join(uniq);
fs::create_dir_all(&root).expect("create temp tree");
TempTree(root)
}
fn path(&self) -> &Path {
&self.0
}
}
impl Drop for TempTree {
fn drop(&mut self) {
let _ = fs::remove_dir_all(&self.0);
}
}
fn write_model(root: &Path, rel: &str, mil: &str) {
let dir = root.join(rel);
fs::create_dir_all(&dir).expect("create model dir");
fs::write(dir.join("model.mil"), mil).expect("write model.mil");
}
fn write_model_with_variance(root: &Path, rel: &str, mil: &str, offset: u64, values: &[f32]) {
write_model(root, rel, mil);
let payload: Vec<u8> = values
.iter()
.flat_map(|v| half::f16::from_f32(*v).to_le_bytes())
.collect();
let data_at = offset + 24;
let mut blob = vec![0_u8; usize::try_from(data_at).expect("test offset fits") + payload.len()];
let at = usize::try_from(offset).expect("test offset fits");
blob[at..at + 4].copy_from_slice(&BLOB_SENTINEL.to_le_bytes());
blob[at + 4..at + 8].copy_from_slice(&BLOB_DTYPE_FP16.to_le_bytes());
blob[at + 8..at + 16].copy_from_slice(&(payload.len() as u64).to_le_bytes());
blob[at + 16..at + 24].copy_from_slice(&data_at.to_le_bytes());
blob[usize::try_from(data_at).expect("test offset fits")..].copy_from_slice(&payload);
let weights = root.join(rel).join("weights");
fs::create_dir_all(&weights).expect("create weights dir");
fs::write(weights.join("weight.bin"), blob).expect("write weight.bin");
}
fn variances_with_zeros(zeros: usize) -> Vec<f32> {
(0..1024)
.map(|i| if i < zeros { 0.0 } else { 1.0 })
.collect()
}
#[test]
fn the_two_surviving_bands_are_classified_apart() {
assert_eq!(GuardBand::of(FP16_MIN_NORMAL), GuardBand::Normal);
assert_eq!(GuardBand::of(FP16_MIN_SUBNORMAL), GuardBand::SubnormalOnly);
assert_eq!(
GuardBand::of(FP16_MIN_NORMAL - FP16_MIN_SUBNORMAL),
GuardBand::SubnormalOnly
);
assert_eq!(
GuardBand::of(FP16_MIN_SUBNORMAL / 2.0),
GuardBand::Inert,
"half of the smallest subnormal is not representable in fp16"
);
assert_eq!(GuardBand::of(0.0), GuardBand::Inert);
assert_eq!(GuardBand::Normal.label(), "normal");
assert_eq!(GuardBand::SubnormalOnly.label(), "subnormal-only");
assert_eq!(GuardBand::Inert.label(), "inert");
assert_eq!(
GuardBand::of(1.000_165_939_331_054_7e-4),
GuardBand::Normal,
"speakerkit/Embedding's divisor guard is an ordinary fp16 number"
);
assert_eq!(
GuardBand::of(1.001_358_032_226_562_5e-5),
GuardBand::SubnormalOnly,
"0x1.5p-17, PyTorch's BatchNorm1d default rounded to fp16, is 168x the smallest subnormal \
and 0.164x the smallest NORMAL — it exists only as a subnormal"
);
assert_eq!(GuardBand::of(9.999_999_939_225_29e-9), GuardBand::Inert);
}
#[test]
fn a_variance_blob_is_read_from_the_offset_the_mil_names() {
let tree = TempTree::new("blob_read");
write_model_with_variance(
tree.path(),
"lid/probe.mlmodelc",
LID_BATCH_NORM,
LID_VARIANCE_BLOB_OFFSET,
&[1.0, 0.0, 2.0],
);
let bundle = tree.path().join("lid/probe.mlmodelc");
let blob = BlobRef {
path: "@model_path/weights/weight.bin".to_string(),
offset: LID_VARIANCE_BLOB_OFFSET,
};
let values = read_blob_as_fp16(&bundle, &blob).expect("the written blob reads back");
assert_eq!(
values.iter().map(|v| v.to_f32()).collect::<Vec<_>>(),
[1.0, 0.0, 2.0]
);
let wrong = BlobRef {
path: "@model_path/weights/weight.bin".to_string(),
offset: 0,
};
let err = read_blob_as_fp16(&bundle, &wrong).expect_err("offset 0 holds no blob record");
assert!(
err.contains("sentinel"),
"a bad offset must be reported as a missing sentinel, got {err:?}"
);
}
#[test]
fn an_unpinned_load_bearing_epsilon_fails_the_sweep() {
let tree = TempTree::new("load_bearing_new");
write_model_with_variance(
tree.path(),
"vadkit/synthetic.mlmodelc",
LID_BATCH_NORM,
LID_VARIANCE_BLOB_OFFSET,
&variances_with_zeros(7),
);
let expected = BTreeSet::from(["vadkit".to_string()]);
let outcome = sweep_tree(tree.path(), &expected).expect("walk the temp tree");
assert_eq!(
outcome.zero_variance_channels, 7,
"seven of the 1 024 stored variances are exactly zero in fp16"
);
assert_eq!(outcome.variance_channels, 1024);
assert!(
outcome.failures.iter().any(|f| {
f.contains("NEW load-bearing epsilon")
&& f.contains("band=subnormal-only")
&& f.contains("zero=7/1024")
}),
"an unpinned zero-variance batch_norm must fail the sweep, naming the band and the count — \
got {:?}",
outcome.failures
);
}
#[test]
fn a_repaired_load_bearing_pin_fails_the_sweep() {
let pin = LOAD_BEARING_NORMS
.iter()
.find(|p| p.path == "lid/SpeechBrainECAPAVoxLingua107.mlmodelc")
.expect("the lid artifact is pinned");
assert!(!pin.sites.is_empty(), "and it pins at least one site");
let tree = TempTree::new("load_bearing_repaired");
write_model_with_variance(
tree.path(),
pin.path,
LID_BATCH_NORM,
LID_VARIANCE_BLOB_OFFSET,
&variances_with_zeros(0),
);
let expected = BTreeSet::from(["lid".to_string()]);
let outcome = sweep_tree(tree.path(), &expected).expect("walk the temp tree");
assert_eq!(
outcome.zero_variance_channels, 0,
"nothing is zero any more"
);
assert!(
outcome
.failures
.iter()
.any(|f| f.contains("pinned LOAD-BEARING epsilon sites CHANGED") && f.contains("REPAIR")),
"a pinned load-bearing artifact whose zero-variance channels vanished must FAIL, so the \
repair is seen and the pin retired deliberately — got {:?}",
outcome.failures
);
}
#[test]
fn an_unreadable_variance_blob_is_a_hole_not_a_skip() {
let tree = TempTree::new("variance_hole");
write_model(tree.path(), "vadkit/synthetic.mlmodelc", LID_BATCH_NORM);
let expected = BTreeSet::from(["vadkit".to_string()]);
let outcome = sweep_tree(tree.path(), &expected).expect("walk the temp tree");
assert!(
outcome.failures.iter().any(|f| {
f.contains("constant `variance` this reader could not read") && f.contains("weight.bin")
}),
"a missing weight blob must be reported as a hole naming the file — got {:?}",
outcome.failures
);
}
#[test]
fn a_norm_that_computes_its_own_variance_makes_no_claim() {
for (label, mil) in [
("granite layer_norm", GRANITE_NORMS_CLEAN),
("clap layer_norm", CLAPKIT_NORMS_CLEAN),
("siglip layer_norm", SIGLIP_NORMS_CLEAN),
] {
let audit = parse_mil(mil).audit();
assert!(
audit.const_variance_norms.is_empty(),
"{label} has no constant `variance` operand, so the census must say nothing about it"
);
assert!(
audit.unresolved.is_empty(),
"{label} must not become a hole either: {:?}",
audit.unresolved
);
assert!(!audit.findings.is_empty(), "{label} still has guard sites");
}
}
#[test]
fn neither_register_pins_a_path_twice() {
let defects: BTreeSet<&str> = KNOWN_DEFECTS.iter().map(|d| d.path).collect();
assert_eq!(
defects.len(),
KNOWN_DEFECTS.len(),
"KNOWN_DEFECTS pins a path twice; the second entry would be unreachable"
);
let bearing: BTreeSet<&str> = LOAD_BEARING_NORMS.iter().map(|d| d.path).collect();
assert_eq!(
bearing.len(),
LOAD_BEARING_NORMS.len(),
"LOAD_BEARING_NORMS pins a path twice; the second entry would be unreachable"
);
assert_eq!(
pinned_paths(),
defects.union(&bearing).copied().collect::<BTreeSet<&str>>(),
"pinned_paths must be the union of both registers — it is what the vendor manifest and the \
missing-model check are built from"
);
for pin in LOAD_BEARING_NORMS {
assert!(
!pin.sites.is_empty(),
"{}: a LOAD_BEARING_NORMS entry with no sites pins nothing",
pin.path
);
assert!(
pin.zero_channels > 0 && pin.zero_channels <= pin.channels,
"{}: {} of {} channels zero is not a coherent census",
pin.path,
pin.zero_channels,
pin.channels
);
assert!(
!pin.note.trim().is_empty(),
"{}: the note is the declaration — an entry without one records nothing",
pin.path
);
}
}
#[test]
fn a_missing_pinned_vendor_fails_the_sweep_not_silently_skips() {
let tree = TempTree::new("missing_vendor");
write_model(
tree.path(),
"whisperkit-coreml/openai_whisper-tiny/MelSpectrogram.mlmodelc",
WHISPER_MEL,
);
let expected = BTreeSet::from(["whisperkit-coreml".to_string(), "speakerkit".to_string()]);
let outcome = sweep_tree(tree.path(), &expected).expect("walk the temp tree");
assert!(
outcome
.failures
.iter()
.any(|f| f.contains("speakerkit") && f.contains("MISSING")),
"a missing expected vendor must FAIL the sweep, naming the vendor — got {:?}",
outcome.failures
);
}
#[test]
fn the_ci_model_job_scope_sweeps_clean() {
let tree = TempTree::new("ci_model_job_scope");
write_model(
tree.path(),
"whisperkit-coreml/openai_whisper-tiny/MelSpectrogram.mlmodelc",
WHISPER_MEL,
);
write_model(
tree.path(),
"embedkit-granite/granite-97m-multilingual-r2/granite_97m_512.mlmodelc",
GRANITE_NORMS_CLEAN,
);
write_model(
tree.path(),
"vadkit/silero-vad-unified-256ms-v6.2.1.mlmodelc",
VADKIT_STFT_SQRT,
);
let expected = BTreeSet::from([
"whisperkit-coreml".to_string(),
"embedkit-granite".to_string(),
"vadkit".to_string(),
]);
let outcome = sweep_tree(tree.path(), &expected).expect("walk the temp tree");
assert!(
outcome.failures.is_empty(),
"the CI model job's scope must sweep clean (every expected vendor present + clean; the pins \
are the documented escape) — got {:?}",
outcome.failures
);
assert_eq!(outcome.models_len, 3, "all three synthetic models");
assert!(
outcome.audited_sites >= 3,
"the mel log site, a granite norm site and vadkit's STFT sqrt were audited"
);
}
#[test]
fn an_empty_vendor_override_hard_errors_not_disables_the_manifest() {
let full = vendor_manifest(None);
assert_eq!(
full,
KNOWN_DEFECTS
.iter()
.map(|d| vendor_of(d.path).to_string())
.collect::<BTreeSet<String>>(),
"unset must yield every KNOWN_DEFECTS vendor"
);
assert!(!full.is_empty(), "the pinned-vendor manifest is non-empty");
assert_eq!(
vendor_manifest(Some(" whisperkit-coreml , speakerkit ")),
BTreeSet::from(["whisperkit-coreml".to_string(), "speakerkit".to_string()]),
"a named override selects exactly those vendors"
);
let prev = std::panic::take_hook();
std::panic::set_hook(Box::new(|_| {}));
let mut leaked: Vec<&str> = Vec::new();
for bad in ["", " ", ",", ", ", " , , "] {
if std::panic::catch_unwind(|| vendor_manifest(Some(bad))).is_ok() {
leaked.push(bad);
}
}
std::panic::set_hook(prev);
assert!(
leaked.is_empty(),
"these present-but-empty overrides returned a manifest instead of hard-erroring: {leaked:?}"
);
}