ruda-runtime 0.1.22

Ruda portable host runtime.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
//! Common selection engine for tensor operations, fused graphs and complete inference plans.
//! GPU code is behind TrialRunner: the built-in runtime adapter uses completed device profiles
//! or explicit synchronization, while application adapters must honor that same contract.
use super::{cache::{DiskCache, Record, now_seconds}, policy::*};
use std::{cell::Cell, collections::{BTreeMap, BTreeSet, VecDeque}, path::PathBuf,
    string::{String, ToString}, sync::{Arc, Mutex, MutexGuard}, time::{Duration, Instant}, vec::Vec};

std::thread_local! { static DEPTH: Cell<usize> = const { Cell::new(0) }; }
/// Whether the current thread is inside a shared-controller trial.
/// This is nesting state, not a query of GPU activity or other processes.
pub fn is_tuning() -> bool { DEPTH.with(|d| d.get() != 0) }
pub(super) struct DepthGuard;
impl DepthGuard { pub(super) fn enter() -> Self { DEPTH.with(|d| d.set(d.get()+1)); Self } }
impl Drop for DepthGuard { fn drop(&mut self) { DEPTH.with(|d| d.set(d.get()-1)); } }

/// Benchmarks MUST use isolated state. A completed measurement includes all device work whose
/// cost belongs to the plan. A submit-only host timestamp is not a valid measurement.
pub trait TrialRunner {
    /// Compare isolated outputs for the candidate indices supplied to select.
    /// Unsupported comparison returns `Validation::Unsupported`; wrong output
    /// or unconfirmed completion returns an appropriate failure instead.
    fn validate(&mut self, reference: usize, candidate: usize, tolerance: Tolerance) -> Result<Validation, TuneFailure>;
    /// Return nonzero completed-work duration for an isolated candidate trial.
    /// Stateful requests must not be advanced by measuring their live state.
    fn measure(&mut self, candidate: usize) -> Result<Duration, TuneFailure>;
}
/// Why a reference or selected implementation was returned.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum DecisionSource {
    /// A fresh search completed, possibly choosing the reference.
    Tuned,
    /// Valid process-memory record, without a fresh synchronization or validation.
    MemoryCache,
    /// Persistent record validated once in this process before reuse.
    DiskCache,
    /// Cache-only miss; use the declared reference.
    CacheMiss,
    /// Reference-only mode.
    Disabled,
    /// Another trial owns the device/key or the parallel-trial budget is full.
    Busy,
    /// Nested trial without a usable memory record.
    Nested,
    /// No supported numerical validator; use the declared reference.
    ValidationUnavailable
}
/// Implementation selection, not a computed output or an instruction to replay a request.
#[derive(Debug, Clone)]
pub struct Decision {
    /// Index in the candidate slice passed to select for this call.
    pub index: usize,
    /// Caller-declared reference index in that same slice.
    pub reference_index: usize,
    /// Stable selected candidate name, independent of slice indexing in another build.
    pub name: String,
    /// Search/cache/reference provenance.
    pub source: DecisionSource,
    /// True only after a validator reported Passed, never inferred from successful launch.
    pub verified: bool,
    /// Paired selected/reference timing ratio when available; None on unmeasured bypass.
    pub ratio: Option<f64>,
    /// Full canonical key used by invalidation and regression observations.
    pub cache_key: String,
}
/// One candidate's local timing/validation metadata.
#[derive(Debug, Clone)]
pub struct CandidateReport {
    /// Candidate identity supplied by the adapter.
    pub name: String,
    /// Number of completed timing pairs; the reference row reports one baseline sample.
    pub samples: usize,
    /// Median selected/reference ratio, or None when no acceptable score exists.
    pub ratio: Option<f64>,
    /// Relative median absolute deviation of paired ratios, when measured.
    pub relative_mad: Option<f64>,
    /// Whether numerical validation passed for this trial.
    pub verified: bool,
    /// Trial selection/rejection reason for diagnostics.
    pub note: String,
}
/// Bounded in-memory diagnostics for one selection workload.
#[derive(Debug, Clone)]
pub struct TuneReport {
    /// Stable operator/graph/application identity.
    pub operation: String,
    /// Backend/device/driver/build signature supplied by the adapter.
    pub environment: String,
    /// Exact shapes, strides, precision and operation options.
    pub workload: String,
    /// Caller-supplied load/topology regime, not an automatically detected topology.
    pub execution_context: String,
    /// Selected candidate name, including a reference selected without usable validation.
    pub winner: String,
    /// Time spent in fresh search; unsupported-validation diagnostics may use zero.
    pub elapsed: Duration,
    /// Whether the soft budget elapsed; it does not interrupt in-flight kernels.
    pub budget_exhausted: bool,
    /// Candidate diagnostic rows considered by this report.
    pub candidates: Vec<CandidateReport>,
}
/// Controller-local counters; these are not GPU utilization or performance estimates.
#[derive(Debug, Clone, Default)]
pub struct Stats {
    /// Successful process-memory lookups.
    pub memory_hits: u64,
    /// Persistent records validated and reused in this process.
    pub disk_hits: u64,
    /// Calls admitted to the cold selection path, before disk/cache-only handling.
    pub misses: u64,
    /// Contending calls returning the declared reference rather than waiting.
    pub busy_fallbacks: u64,
    /// Successful fresh searches, including reference wins.
    pub tunes: u64,
    /// Disk read/write/removal warnings; normal execution can continue.
    pub cache_warnings: u64,
    /// Explicit or regression-triggered invalidation calls.
    pub invalidations: u64,
    /// Device completion failures recorded by this controller.
    pub device_failures: u64,
}
#[derive(Default)]
struct State {
    records: BTreeMap<String, Arc<Record>>, order: MemoryOrder,
    bypass: BTreeMap<String, Scope>,
    pending: BTreeSet<String>, devices: BTreeSet<String>, poisoned_devices: BTreeSet<String>,
    banned: BTreeMap<String, BTreeSet<String>>, regressions: BTreeMap<String, VecDeque<f64>>,
    reports: VecDeque<TuneReport>, stats: Stats,
}
enum MemoryOrder {
    Fifo(VecDeque<String>),
    Lru(LruOrder),
}
impl Default for MemoryOrder {
    fn default() -> Self { Self::Fifo(VecDeque::new()) }
}
impl MemoryOrder {
    fn new(eviction: MemoryEviction) -> Self {
        match eviction { MemoryEviction::Fifo => Self::default(), MemoryEviction::Lru => Self::Lru(LruOrder::default()) }
    }
    fn len(&self) -> usize {
        match self { Self::Fifo(order) => order.len(), Self::Lru(order) => order.indices.len() }
    }
    fn contains(&self, key: &String) -> bool {
        match self { Self::Fifo(order) => order.contains(key), Self::Lru(order) => order.indices.contains_key(key.as_str()) }
    }
    fn push_back(&mut self, key: String) {
        match self { Self::Fifo(order) => order.push_back(key), Self::Lru(order) => order.insert(key) }
    }
    fn pop_front(&mut self) -> Option<String> {
        match self { Self::Fifo(order) => order.pop_front(), Self::Lru(order) => order.pop_front() }
    }
    fn promote(&mut self, key: &str) {
        if let Self::Lru(order) = self { order.promote(key); }
    }
}
#[derive(Default)]
struct LruOrder {
    indices: BTreeMap<Arc<str>, usize>,
    slots: Vec<Option<LruEntry>>,
    vacant: Vec<usize>,
    head: Option<usize>,
    tail: Option<usize>,
}
struct LruEntry {
    key: Arc<str>,
    previous: Option<usize>,
    next: Option<usize>,
}
impl LruOrder {
    fn detach(&mut self, index: usize) {
        let entry = self.slots[index].as_ref().expect("retained LRU entry");
        let (previous, next) = (entry.previous, entry.next);
        match previous {
            Some(previous) => self.slots[previous].as_mut().expect("previous LRU entry").next = next,
            None => self.head = next,
        }
        match next {
            Some(next) => self.slots[next].as_mut().expect("next LRU entry").previous = previous,
            None => self.tail = previous,
        }
    }
    fn append(&mut self, index: usize) {
        let entry = self.slots[index].as_mut().expect("retained LRU entry");
        entry.previous = self.tail;
        entry.next = None;
        match self.tail {
            Some(tail) => self.slots[tail].as_mut().expect("last LRU entry").next = Some(index),
            None => self.head = Some(index),
        }
        self.tail = Some(index);
    }
    fn promote(&mut self, key: &str) {
        if let Some(&index) = self.indices.get(key) {
            if self.tail != Some(index) { self.detach(index); self.append(index); }
        }
    }
    fn insert(&mut self, key: String) {
        if self.indices.contains_key(key.as_str()) { self.promote(&key); return; }
        let key = Arc::<str>::from(key);
        let index = match self.vacant.pop() {
            Some(index) => index,
            None => { self.slots.push(None); self.slots.len() - 1 },
        };
        self.slots[index] = Some(LruEntry { key: key.clone(), previous: None, next: None });
        self.indices.insert(key, index);
        self.append(index);
    }
    fn pop_front(&mut self) -> Option<String> {
        let index = self.head?;
        self.detach(index);
        let entry = self.slots[index].take().expect("first LRU entry");
        self.indices.remove(entry.key.as_ref());
        self.vacant.push(index);
        Some(entry.key.to_string())
    }
}
/// No state mutex is held while running a candidate, validator, profile, or disk operation.
/// Contenders use the declared reference instead of waiting (also avoids nested-tuning deadlocks).
pub struct StackTuner { policy: StackPolicy, memory_eviction: MemoryEviction, disk: Option<DiskCache>, state: Mutex<State> }
struct Permit<'a> { tuner: &'a StackTuner, key: String, device: String }
impl Drop for Permit<'_> {
    fn drop(&mut self) { let mut s = self.tuner.lock(); s.pending.remove(&self.key); s.devices.remove(&self.device); }
}
impl StackTuner {
    /// Validate policy and create an independent controller.
    /// `None` disables disk storage. Construction performs no disk I/O and does
    /// not install this instance as the global runtime controller.
    pub fn new(policy: StackPolicy, cache_directory: Option<PathBuf>) -> Result<Self, TuneFailure> {
        Self::new_with_memory_eviction(policy, cache_directory, MemoryEviction::Fifo)
    }
    /// Select an explicit memory retention order without changing disk eviction or validation.
    pub fn new_with_memory_eviction(policy: StackPolicy, cache_directory: Option<PathBuf>, memory_eviction: MemoryEviction) -> Result<Self, TuneFailure> {
        policy.validate()?;
        let disk = cache_directory.map(|dir| DiskCache::new(dir, policy.capacity));
        Ok(Self { policy, memory_eviction, disk, state: Mutex::new(State { order: MemoryOrder::new(memory_eviction), ..State::default() }) })
    }
    fn lock(&self) -> MutexGuard<'_, State> { self.state.lock().unwrap_or_else(|e| e.into_inner()) }
    /// Borrow the immutable policy; runtime reconfiguration is not supported.
    pub fn policy(&self) -> &StackPolicy { &self.policy }
    /// Return the immutable process-memory retention order.
    pub fn memory_eviction(&self) -> MemoryEviction { self.memory_eviction }
    /// Snapshot this instance's counters without launching device work.
    pub fn stats(&self) -> Stats { self.lock().stats.clone() }
    /// Conservative dependency snapshot for complete pipelines. It includes all currently
    /// cached lower-level decisions in this controller, including other workloads/devices.
    /// That may over-invalidate a pipeline but never hides a changed lower-level choice.
    pub fn lower_level_fingerprint(&self) -> String {
        use std::fmt::Write;
        let s = self.lock();
        let mut parts = Vec::new();
        for (key, record) in &s.records {
            if record.scope != Scope::Pipeline as u8 && record.is_fresh(now_seconds(), self.policy.ttl.as_secs().max(1)) { parts.push(fields(&[key, &record.winner])); }
        }
        // Bypassed operators retain their declared references, never an unchecked winner.
        for (key, scope) in &s.bypass {
            if *scope != Scope::Pipeline { parts.push(fields(&[key, "unverified-reference"])); }
        }
        parts.sort();
        let mut encoded = String::new();
        for part in &parts { let _ = write!(encoded, "{}:{part}", part.len()); }
        super::cache::digest(encoded.as_bytes())
    }
    /// Clone retained search diagnostics, oldest first, bounded by min(capacity, 128).
    /// Cache hits are counters, not new search reports.
    pub fn reports(&self) -> Vec<TuneReport> { self.lock().reports.iter().cloned().collect() }
    fn promote(&self, state: &mut State, key: &str) {
        state.order.promote(key);
    }
    fn insert(&self, record: Record) {
        let mut s = self.lock();
        if !s.order.contains(&record.key) {
            while s.order.len() >= self.policy.capacity {
                if let Some(key) = s.order.pop_front() {
                    s.records.remove(&key); s.banned.remove(&key); s.regressions.remove(&key); s.bypass.remove(&key);
                }
            }
            s.order.push_back(record.key.clone());
        }
        self.promote(&mut s, &record.key);
        s.records.insert(record.key.clone(), Arc::new(record));
    }
    fn allowed(&self, key: &str, c: &impl CandidateSource) -> bool {
        let c = c.view();
        c.fits(&self.policy) && !self.lock().banned.get(key).is_some_and(|b| b.contains(c.name))
    }
    fn from_record<C: CandidateSource>(&self, r: &Record, candidates: &[C], reference: usize, source: DecisionSource) -> Option<Decision> {
        if !r.is_fresh(now_seconds(), self.policy.ttl.as_secs().max(1)) || (self.policy.require_validation && !r.verified) { return None; }
        let index = candidates.iter().position(|c| c.view().name == &r.winner && self.allowed(&r.key, c))?;
        Some(Decision { index, reference_index: reference, name: r.winner.clone(), source, verified: r.verified, ratio: Some(r.ratio), cache_key: r.key.clone() })
    }
    fn validation(&self, runner: &mut impl TrialRunner, reference: usize, candidate: usize) -> Result<bool, TuneFailure> {
        match runner.validate(reference, candidate, self.policy.tolerance)? {
            Validation::Passed => Ok(true),
            Validation::Unsupported if !self.policy.require_validation => Ok(false),
            Validation::Unsupported => Err(TuneFailure { kind: FailureKind::Unavailable, message: "no numerical validator for this output/size".into() }),
        }
    }
    fn fail_device(&self, environment: &str) {
        let mut s = self.lock(); s.poisoned_devices.insert(environment.to_string()); s.stats.device_failures += 1;
    }
    /// Reuse or select an eligible implementation for one exact workload.
    /// Candidate names must be unique/nonempty, count 1..=4096, and the reference
    /// index must exist and fit policy. Empty operation/environment/workload or an oversized
    /// key return InvalidInput. Trials may synchronize; memory hits do not.
    /// A returned decision does not execute the caller's live request.
    pub fn select(&self, problem: &Problem, candidates: &[Candidate], reference: usize, runner: &mut impl TrialRunner) -> Result<Decision, TuneFailure> {
        self.select_candidates(problem, candidates, reference, runner)
    }

    pub(super) fn select_candidates<C: CandidateSource>(&self, problem: &Problem, candidates: &[C], reference: usize, runner: &mut impl TrialRunner) -> Result<Decision, TuneFailure> {
        if candidates.is_empty() || candidates.len() > 4096 || reference >= candidates.len()
            || problem.operation.is_empty() || problem.environment.is_empty() || problem.workload.is_empty() {
            return Err(TuneFailure::invalid("a workload, environment and valid reference candidate are required"));
        }
        let mut names = BTreeSet::new();
        if candidates.iter().any(|c| {
            let c = c.view();
            c.name.is_empty() || c.name.len() > 4096 || !names.insert(c.name)
        }) {
            return Err(TuneFailure::invalid("candidate names must be nonempty, bounded and unique"));
        }
        if !candidates[reference].view().fits(&self.policy) { return Err(TuneFailure::invalid("reference does not satisfy eligibility/workspace policy")); }
        let key = cache_key_candidates(problem, candidates, reference, &self.policy);
        if key.len() > 384 * 1024 { return Err(TuneFailure::invalid("autotune key exceeds 384 KiB")); }
        let fallback = |source| Decision { index: reference, reference_index: reference, name: String::clone(candidates[reference].view().name), source, verified: false, ratio: None, cache_key: key.clone() };
        if self.lock().poisoned_devices.contains(&problem.environment) {
            return Err(TuneFailure { kind: FailureKind::Quarantined, message: "device tuning lane is quarantined after an unconfirmed device completion".into() });
        }
        if self.policy.mode == Mode::Disabled { return Ok(fallback(DecisionSource::Disabled)); }
        let memory = { self.lock().records.get(&key).cloned() };
        if let Some(record) = memory {
            if let Some(d) = self.from_record(&record, candidates, reference, DecisionSource::MemoryCache) {
                let mut state = self.lock();
                state.stats.memory_hits += 1;
                self.promote(&mut state, &key);
                return Ok(d);
            }
        }
        let bypass = { self.lock().bypass.contains_key(&key) };
        if bypass {
            if self.memory_eviction == MemoryEviction::Lru { self.promote(&mut self.lock(), &key); }
            return Ok(fallback(DecisionSource::ValidationUnavailable));
        }
        if is_tuning() { return Ok(fallback(DecisionSource::Nested)); }
        let _permit = {
            let mut s = self.lock();
            if s.pending.contains(&key) || s.devices.contains(&problem.environment) || s.pending.len() >= self.policy.max_parallel_tunes {
                s.stats.busy_fallbacks += 1; return Ok(fallback(DecisionSource::Busy));
            }
            s.pending.insert(key.clone()); s.devices.insert(problem.environment.clone()); s.stats.misses += 1;
            Permit { tuner: self, key: key.clone(), device: problem.environment.clone() }
        };
        let _depth = DepthGuard::enter();
        if problem.persistent {
            if let Some(disk) = &self.disk {
                match disk.load(&key) {
                    Ok(Some(record)) => {
                        if let Some(mut decision) = self.from_record(&record, candidates, reference, DecisionSource::DiskCache).filter(|_| record.scope == problem.scope as u8) {
                            // A disk record is not trusted as evidence for the current inputs until
                            // the candidate is checked once in this process. CacheOnly still validates.
                            match self.validation(runner, reference, decision.index) {
                                Ok(checked) => {
                                    decision.verified = checked;
                                    let mut record = record; record.verified = checked;
                                    self.insert(record); self.lock().stats.disk_hits += 1; return Ok(decision);
                                }
                                Err(e) if e.kind == FailureKind::Device => { self.fail_device(&problem.environment); return Err(e); }
                                Err(_) => { if disk.remove(&key).is_err() { self.lock().stats.cache_warnings += 1; } }
                            }
                        }
                    }
                    Err(_) => self.lock().stats.cache_warnings += 1,
                    Ok(None) => {}
                }
            }
        }
        if self.policy.mode == Mode::CacheOnly { return Ok(fallback(DecisionSource::CacheMiss)); }
        match self.explore(problem, candidates, reference, &key, runner) {
            Ok((decision, report, record)) => {
                if problem.persistent {
                    if let Some(disk) = &self.disk { if disk.save(&record).is_err() { self.lock().stats.cache_warnings += 1; } }
                }
                self.insert(record);
                let mut s = self.lock(); s.stats.tunes += 1;
                if s.reports.len() >= self.policy.capacity.min(128) { s.reports.pop_front(); }
                s.reports.push_back(report);
                Ok(decision)
            }
            Err(e) if e.kind == FailureKind::Unavailable => {
                // Reference benchmarking may exceed the validation/workspace budget. Preserve
                // normal execution on the declared reference; never promote an unchecked winner.
                let mut s = self.lock();
                if !s.order.contains(&key) { s.order.push_back(key.clone()); }
                s.bypass.insert(key.clone(), problem.scope);
                self.promote(&mut s, &key);
                while s.order.len() > self.policy.capacity {
                    if let Some(old) = s.order.pop_front() { s.records.remove(&old); s.banned.remove(&old); s.regressions.remove(&old); s.bypass.remove(&old); }
                }
                if s.reports.len() >= self.policy.capacity.min(128) { s.reports.pop_front(); }
                s.reports.push_back(TuneReport {
                    operation: problem.operation.clone(), environment: problem.environment.clone(),
                    workload: problem.workload.clone(), execution_context: problem.execution_context.clone(),
                    winner: String::clone(candidates[reference].view().name), elapsed: Duration::ZERO, budget_exhausted: false,
                    candidates: std::vec![CandidateReport { name: String::clone(candidates[reference].view().name), samples: 0,
                        ratio: None, relative_mad: None, verified: false, note: e.message }],
                });
                Ok(fallback(DecisionSource::ValidationUnavailable))
            }
            Err(e) => { if e.kind == FailureKind::Device { self.fail_device(&problem.environment); } Err(e) }
        }
    }
    fn explore<C: CandidateSource>(&self, problem: &Problem, candidates: &[C], reference: usize, key: &str, runner: &mut impl TrialRunner) -> Result<(Decision, TuneReport, Record), TuneFailure> {
        let started = Instant::now();
        // The mandatory reference is validated even if its compilation takes the whole budget.
        // Never treat a failed reference as permission to choose an unverified fast candidate.
        for _ in 0..self.policy.warmups { runner.measure(reference)?; }
        let reference_checked = self.validation(runner, reference, reference)?;
        let reference_time = runner.measure(reference)?;
        if reference_time.is_zero() { return Err(TuneFailure { kind: FailureKind::Unavailable, message: "reference has zero elapsed time".into() }); }
        let mut winner = reference; let mut winner_ratio = 1.0; let mut winner_time = reference_time;
        let mut winner_reference = reference_time; let mut winner_checked = reference_checked;
        let mut reports = Vec::new();
        reports.push(CandidateReport { name: String::clone(candidates[reference].view().name), samples: 1, ratio: Some(1.0), relative_mad: Some(0.0), verified: reference_checked, note: "reference; retained unless a stable measured improvement qualifies".into() });
        let mut attempted = 1; let mut exhausted = false;
        for (index, candidate) in candidates.iter().enumerate() {
            if index == reference { continue; }
            let mut report = CandidateReport { name: String::clone(candidate.view().name), samples: 0, ratio: None, relative_mad: None, verified: false, note: String::new() };
            if !self.allowed(key, candidate) { report.note = "ineligible, unknown/over-budget workspace, or regressed candidate".into(); reports.push(report); continue; }
            if attempted >= self.policy.max_candidates || started.elapsed() >= self.policy.budget {
                exhausted = true; report.note = "search budget exhausted".into(); reports.push(report); continue;
            }
            attempted += 1;
            let result = (|| -> Result<(bool, Vec<(Duration, Duration)>), TuneFailure> {
                for _ in 0..self.policy.warmups {
                    if started.elapsed() >= self.policy.budget { return Err(TuneFailure::rejected("budget exhausted during warmup")); }
                    runner.measure(index)?;
                }
                let checked = self.validation(runner, reference, index)?;
                let mut pairs = Vec::with_capacity(self.policy.samples);
                for sample in 0..self.policy.samples {
                    if started.elapsed() >= self.policy.budget { return Err(TuneFailure::rejected("budget exhausted before enough paired samples")); }
                    // Alternate order to reduce monotonic warmup/clock/thermal bias.
                    let pair = if (sample + index) % 2 == 0 {
                        let a = runner.measure(reference)?; let b = runner.measure(index)?; (a,b)
                    } else {
                        let b = runner.measure(index)?; let a = runner.measure(reference)?; (a,b)
                    };
                    if pair.0.is_zero() || pair.1.is_zero() { return Err(TuneFailure::rejected("zero duration cannot be scored")); }
                    pairs.push(pair);
                }
                Ok((checked, pairs))
            })();
            match result {
                Err(e) if e.kind == FailureKind::Device => return Err(e),
                Err(e) => { report.note = e.message; },
                Ok((checked, pairs)) => {
                    report.verified = checked; report.samples = pairs.len();
                    match paired_score(&pairs, &self.policy) {
                        None => { report.note = "timings too noisy or incomplete".into(); },
                        Some((ratio, mad)) => {
                            report.ratio = Some(ratio); report.relative_mad = Some(mad);
                            report.note = "paired synchronized measurements".into();
                            if ratio < winner_ratio && ratio * self.policy.min_speedup <= 1.0 {
                                winner = index; winner_ratio = ratio; winner_checked = checked;
                                let mut a: Vec<_> = pairs.iter().map(|p| p.0.as_secs_f64()).collect();
                                let mut b: Vec<_> = pairs.iter().map(|p| p.1.as_secs_f64()).collect();
                                winner_reference = Duration::from_secs_f64(median(&mut a).unwrap());
                                winner_time = Duration::from_secs_f64(median(&mut b).unwrap());
                            }
                        }
                    }
                }
            }
            reports.push(report);
        }
        exhausted |= started.elapsed() >= self.policy.budget;
        let record = Record { scope: problem.scope as u8, key: key.to_string(), winner: String::clone(candidates[winner].view().name), created: now_seconds(),
            reference_ns: winner_reference.as_nanos().min(u64::MAX as u128) as u64,
            winner_ns: winner_time.as_nanos().min(u64::MAX as u128) as u64, ratio: winner_ratio, verified: winner_checked };
        let decision = Decision { index: winner, reference_index: reference, name: record.winner.clone(), source: DecisionSource::Tuned, verified: winner_checked, ratio: Some(winner_ratio), cache_key: key.to_string() };
        let report = TuneReport { operation: problem.operation.clone(), environment: problem.environment.clone(), workload: problem.workload.clone(), execution_context: problem.execution_context.clone(), winner: record.winner.clone(), elapsed: started.elapsed(), budget_exhausted: exhausted, candidates: reports };
        Ok((decision, report, record))
    }
    /// Invalidate future selections; NEVER re-run the current stateful request after a failure.
    /// `ban_candidate=true` additionally bans a non-reference name for this key.
    /// Removes the persistent record when possible; disk failures increment cache warnings.
    pub fn invalidate(&self, decision: &Decision, ban_candidate: bool) {
        {
            let mut s = self.lock(); s.records.remove(&decision.cache_key); s.regressions.remove(&decision.cache_key); s.bypass.remove(&decision.cache_key);
            if ban_candidate && decision.index != decision.reference_index {
                s.banned.entry(decision.cache_key.clone()).or_default().insert(decision.name.clone());
            }
            // Retain one bounded FIFO slot while the key is banned.
            if !s.order.contains(&decision.cache_key) { s.order.push_back(decision.cache_key.clone()); }
            while s.order.len() > self.policy.capacity {
                if let Some(old) = s.order.pop_front() { s.records.remove(&old); s.banned.remove(&old); s.regressions.remove(&old); s.bypass.remove(&old); }
            }
            s.stats.invalidations += 1;
        }
        if let Some(disk) = &self.disk { if disk.remove(&decision.cache_key).is_err() { self.lock().stats.cache_warnings += 1; } }
    }
    /// Feed paired observations from a controlled replay of the SAME workload and load regime.
    /// Ordinary production request latency is NOT a comparable baseline measurement.
    /// Returns true only when the accumulated ratio window invalidates this decision.
    /// Missing records or changed winner names return false; zero or unchecked
    /// timings return InvalidInput. This observation does not revalidate TTL.
    pub fn record_comparison(&self, decision: &Decision, reference: Duration, selected: Duration, correctness_checked: bool) -> Result<bool, TuneFailure> {
        if !correctness_checked || reference.is_zero() || selected.is_zero() { return Err(TuneFailure::invalid("regression observations require nonzero, correctness-checked paired timings")); }
        let should_invalidate = {
            let mut s = self.lock();
            if !s.records.get(&decision.cache_key).is_some_and(|r| r.winner == decision.name) { return Ok(false); }
            let history = s.regressions.entry(decision.cache_key.clone()).or_default();
            if history.len() == self.policy.regression_pairs { history.pop_front(); }
            history.push_back(selected.as_secs_f64()/reference.as_secs_f64());
            if history.len() < self.policy.regression_pairs { false } else {
                let mut ratios: Vec<_> = history.iter().copied().collect(); median(&mut ratios).is_some_and(|r| r > self.policy.regression_ratio)
            }
        };
        if should_invalidate { self.invalidate(decision, true); }
        Ok(should_invalidate)
    }
}