Skip to main content

ferrotherm_gpu/
device.rs

1//! The GPU as a [`Device`], so the conformance machinery can score it.
2//!
3//! Until this existed, `ferrotherm`'s survey of its own capabilities carried a line it had earned:
4//! *"no `impl Device` for a GPU, so `conform` cannot even score the GPU path."* The fastest sampler
5//! in the stack was the one path the verification machinery could not reach — it could be run, and
6//! it could not be **checked against the fabric it claims to be**.
7//!
8//! What that check turns out to be about is precision. The CPU sampler is `f64` throughout and the
9//! shader is `f32`, because that is what WGSL storage buffers hold. Every other `Device` here
10//! declares its precision honestly — D-Wave as `Unstated`, the fixed-point fabric as
11//! `Fixed { bits }` — and the GPU had no declaration at all, so nothing downstream could reason
12//! about it. It is [`Precision::Float`] with a 24-bit mantissa, and saying so is what lets
13//! `conform` compare the two paths knowing which differences are the arithmetic and which are the
14//! sampler.
15//!
16//! ```no_run
17//! use ferrotherm::{fabric::Device, ising::lattice2d, schedule::Schedule, ftp::Program};
18//! use ferrotherm_gpu::GpuDevice;
19//!
20//! let Some(mut dev) = GpuDevice::open() else { return };
21//! let g = lattice2d(32, 1.0);
22//! let p = Program::from_graph(&g, &Schedule::default());
23//! assert!(dev.program(&p).is_empty());
24//! let state = dev.run(&Schedule::constant(0.6, 200), 7).unwrap();
25//! assert_eq!(state.len(), g.n);
26//! // The writes are charged, which is the term the ledger's thesis rests on.
27//! assert_eq!(dev.ledger().writes, g.n as u64);
28//! ```
29
30use ferrotherm::fabric::{Device, Fabric, Precision, Unsupported};
31use ferrotherm::ftp::Program;
32use ferrotherm::ledger::{Ledger, Prices};
33use ferrotherm::rng::Pcg;
34use ferrotherm::schedule::Schedule;
35use ferrotherm::graph::Graph;
36use ferrotherm::wgsl::GpuModel;
37
38/// A [`Device`] backed by the native WGSL sampler.
39///
40/// Separate from [`Gpu`](crate::Gpu) rather than implemented on it, because `Gpu` is a handle to an
41/// adapter and holds no problem: `Device` is a stateful loader-and-runner, and merging the two
42/// would make every `Gpu` carry a program it may never be given.
43pub struct GpuDevice {
44    gpu: crate::Gpu,
45    model: Option<GpuModel>,
46    /// Kept beside the `GpuModel` so a stage's result can be SCORED without a second lowering.
47    /// `GpuModel` is the device-side layout and carries no energy function.
48    graph: Option<Graph>,
49    state: Vec<i8>,
50    ledger: Ledger,
51}
52
53impl GpuDevice {
54    /// Open the default adapter, or `None` where this machine exposes none.
55    ///
56    /// `None` means **not found here**, never "impossible" — the same contract as
57    /// [`Gpu::new`](crate::Gpu::new), and worth preserving because a headless CI runner is the
58    /// common case and every test around this skips rather than fails on it.
59    pub fn open() -> Option<GpuDevice> {
60        Some(GpuDevice::with(crate::Gpu::new()?))
61    }
62
63    /// Wrap an adapter already opened, so a caller that enumerated once does not enumerate again.
64    pub fn with(gpu: crate::Gpu) -> GpuDevice {
65        GpuDevice { gpu, model: None, graph: None, state: Vec::new(), ledger: Ledger::default() }
66    }
67
68    /// What the driver reports, for a caller that needs to know whether this is real silicon.
69    pub fn adapter(&self) -> &wgpu::AdapterInfo {
70        self.gpu.adapter()
71    }
72
73    /// True when the adapter is hardware rather than a software rasteriser.
74    #[must_use = "false means a software rasteriser, whose timings say nothing about a GPU"]
75    pub fn is_hardware(&self) -> bool {
76        self.gpu.is_hardware()
77    }
78}
79
80impl Device for GpuDevice {
81    fn fabric(&self) -> Fabric {
82        // NOT Prices::UNSTATED by oversight -- by fact. A GPU vendor publishes board power, which
83        // is a rate for the whole card, not an energy per spin update. `ferrotherm-meter` derives
84        // the per-operation figure by measuring THIS machine, and that measured value is what
85        // belongs here; a datasheet number would be a different machine's.
86        let mut f = Fabric::unconstrained("gpu", Prices::UNSTATED);
87        // Lowers through `Program::to_graph`, which is pairwise.
88        f.max_arity = 2;
89        // The declaration that was missing. WGSL storage buffers hold f32: `GpuModel::w` and
90        // `GpuModel::h` are `Vec<f32>`, so every coupling and field is rounded to 24 mantissa bits
91        // on the way in. The CPU path keeps f64. Neither is wrong; an undeclared difference is.
92        f.coupling_precision = Precision::Float { mantissa: 24 };
93        f.field_precision = Precision::Float { mantissa: 24 };
94        f.unstated = &[
95            "per-operation energy: GPU vendors publish board power (a rate for the whole card), \
96             not joules per spin update. Measure it with ferrotherm-meter on the machine that ran \
97             the work.",
98        ];
99        f
100    }
101
102    fn program(&mut self, p: &Program) -> Vec<Unsupported> {
103        let bad = self.fabric().check(p);
104        if !bad.is_empty() {
105            return bad;
106        }
107        match p.to_graph() {
108            Ok(g) => {
109                self.model = Some(GpuModel::from_graph(&g));
110                self.state = vec![-1; g.n];
111                // The write, charged -- one node's couplings, bias and clamp state flashed. The
112                // CPU device charges it and the ledger's whole thesis rests on it; a GPU device
113                // that skipped it would make the GPU look free at exactly the term that is not.
114                self.ledger.writes += g.n as u64;
115                self.graph = Some(g);
116                Vec::new()
117            }
118            Err(e) => vec![Unsupported::Unplaceable { detail: e.to_string() }],
119        }
120    }
121
122    /// Run the schedule and return the BEST state seen, not the last one.
123    ///
124    /// The trait's wording says "the final state" and every implementation here returns the best:
125    /// `Cpu` delegates to `tempering::anneal_scheduled`, which tracks the minimum over every sweep,
126    /// and `sbm` calls the same thing a best-so-far readout. That is not pedantry about wording. An
127    /// anneal's last state is wherever the coldest stage happened to stop, and this returned it --
128    /// which `conform` caught the moment it could reach this path at all, scoring **-57 against
129    /// variable elimination's exact -59** while the CPU on the same ladder found -59. Nothing was
130    /// wrong with the sampler; it was being asked the wrong question at the end.
131    ///
132    /// Tracked per STAGE rather than per sweep, and that difference is real: scoring a state means
133    /// reading it back off the device, so per-sweep tracking would put a round trip between every
134    /// sweep and spend the throughput the GPU exists for. The conformance ladder is 80 stages, so
135    /// the minimum is taken over 80 checkpoints against the CPU's 3,200.
136    fn run(&mut self, schedule: &Schedule, seed: u64) -> Result<Vec<i8>, String> {
137        let m = self.model.as_ref().ok_or("no program loaded")?;
138        let g = self.graph.as_ref().ok_or("no program loaded")?;
139        if schedule.is_empty() {
140            return Err("an empty schedule runs nothing; give it at least one stage".into());
141        }
142        // A RUN STARTS FROM THE SEED, and does not inherit the last one's answer.
143        //
144        // Two differences from the reference `Cpu` device, both found by a test that could not
145        // otherwise have failed. `Cpu::run` builds a fresh `Sampler::new(g, beta, seed)`, whose
146        // initial state is drawn from the seed -- so it resets per run AND starts somewhere the
147        // seed chose. This carried `self.state` between calls and began at all-minus-one, which
148        // made a second `run` start from the first one's best: two different seeds then returned
149        // the same state, because the second was handed an answer it could not improve on and
150        // simply gave it back. Matching `Sampler::new` exactly also means CPU and GPU now start a
151        // given seed at the SAME configuration, which is what makes the two paths comparable.
152        let mut rng = Pcg::new(seed, 0x5EED);
153        self.state = (0..g.n).map(|_| rng.spin(0.5)).collect();
154        let mut best = self.state.clone();
155        let mut best_e = g.energy(&best);
156        // Stage by stage, because a schedule is a temperature ladder and the shader takes one beta
157        // per dispatch. The state carries across stages, which is what makes it an anneal rather
158        // than a sequence of independent runs.
159        for (i, st) in schedule.stages().iter().enumerate() {
160            // The seed varies per stage. Reusing one stream across stages would have every stage
161            // draw the same numbers at the same nodes, which is the failure the step counter
162            // already exists to prevent, one level up.
163            let stage_seed = seed.wrapping_add(i as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15);
164            self.gpu.sweep_seeded(m, &mut self.state, st.beta, st.sweeps as u32, stage_seed)?;
165            self.ledger.samples += m.n as u64 * st.sweeps as u64;
166            let e = g.energy(&self.state);
167            if e < best_e {
168                best_e = e;
169                best = self.state.clone();
170            }
171        }
172        self.state = best.clone();
173        Ok(best)
174    }
175
176    fn ledger(&self) -> Ledger {
177        self.ledger
178    }
179}
180
181#[cfg(test)]
182mod tests {
183    use super::*;
184    use ferrotherm::ising::lattice2d;
185
186    macro_rules! dev_or_skip {
187        () => {{
188            // Serialised for the same reason as `ferrotherm_gpu::tests::ADAPTER`: parallel
189            // Vulkan device creation segfaults the NVIDIA driver stack.
190            let own = crate::tests::ADAPTER.lock().unwrap_or_else(|e| e.into_inner());
191            match GpuDevice::open() {
192                Some(d) => (d, own),
193                None => {
194                    eprintln!("no GPU adapter on this machine; skipping");
195                    return;
196                }
197            }
198        }};
199    }
200
201    #[test]
202    fn conform_can_finally_score_the_gpu_path() {
203        // THE POINT OF THIS MODULE. `conform::run` takes a `&mut dyn Device`, so before this impl
204        // existed the fastest sampler in the stack was the one path the conformance suite could not
205        // reach -- runnable, and uncheckable against the fabric it claims to be.
206        let (mut d, _own) = dev_or_skip!();
207        if !d.is_hardware() {
208            eprintln!("software rasteriser; the physics is still checked, the timings mean nothing");
209        }
210        let report = ferrotherm::conform::run(&mut d);
211        assert!(
212            report.passed(),
213            "the GPU path fails conformance:\n{report}\n{}",
214            report.failures().map(|c| format!("  {} -- {}", c.name, c.detail)).collect::<Vec<_>>().join("\n")
215        );
216    }
217
218    #[test]
219    fn the_fabric_declares_f32_rather_than_leaving_it_unsaid() {
220        // The difference that was invisible. `GpuModel::{w,h}` are `Vec<f32>`, so every coupling and
221        // field is rounded to 24 mantissa bits going in while the CPU path keeps f64. Neither is
222        // wrong; an undeclared difference is, because nothing downstream can then tell an
223        // arithmetic gap from a sampler gap.
224        let (d, _own) = dev_or_skip!();
225        let f = d.fabric();
226        assert_eq!(f.coupling_precision, Precision::Float { mantissa: 24 });
227        assert_eq!(f.field_precision, Precision::Float { mantissa: 24 });
228        assert_eq!(f.max_arity, 2, "it lowers through to_graph, which is pairwise");
229        assert!(!f.prices.is_stated(), "a GPU publishes board power, not joules per spin update");
230        assert!(
231            f.unstated.iter().any(|u| u.contains("per-operation energy")),
232            "the gap has to be named, not merely left empty: {:?}",
233            f.unstated
234        );
235    }
236
237    #[test]
238    fn the_seed_selects_a_stream_rather_than_being_swallowed() {
239        // The trait hands `run` a seed and `Gpu::sweep` had none, so the obvious implementation
240        // takes the argument and drops it -- and a caller varying the seed to gauge spread gets one
241        // answer every time and reads a deaf sampler as a confident one. Note that `conform`'s
242        // determinism case CANNOT catch this: an ignored seed is perfectly reproducible.
243        //
244        // Checked at the sweep level, where it is unambiguous: identical starting state, identical
245        // beta, identical sweep count, two seeds.
246        let (d, _own) = dev_or_skip!();
247        let g = lattice2d(24, 1.0);
248        let m = GpuModel::from_graph(&g);
249        let hot = 0.15; // disordered, so two streams separate immediately
250
251        let mut a = vec![1i8; g.n];
252        let mut b = vec![1i8; g.n];
253        let mut a_again = vec![1i8; g.n];
254        d.gpu.sweep_seeded(&m, &mut a, hot, 40, 1).unwrap();
255        d.gpu.sweep_seeded(&m, &mut b, hot, 40, 2).unwrap();
256        d.gpu.sweep_seeded(&m, &mut a_again, hot, 40, 1).unwrap();
257        assert_ne!(a, b, "two seeds produced identical states; the seed is being ignored");
258        assert_eq!(a, a_again, "the same seed must reproduce; this is a seed, not just noise");
259
260        // And seed 0 has to leave the unseeded stream exactly where it was, or every Onsager number
261        // ever taken on this shader silently moved.
262        let mut viaseed = vec![1i8; g.n];
263        let mut unseeded = vec![1i8; g.n];
264        d.gpu.sweep_seeded(&m, &mut viaseed, hot, 40, 0).unwrap();
265        d.gpu.sweep(&m, &mut unseeded, hot, 40).unwrap();
266        assert_eq!(viaseed, unseeded, "seed 0 must be the stream that existed before seeding");
267    }
268
269    #[test]
270    fn the_device_threads_its_seed_through_to_the_sampler() {
271        // The same property one level up, where it is what the trait actually promises.
272        //
273        // On a FRUSTRATED instance, deliberately. A ferromagnet's all-minus-one start is already a
274        // ground state, so best-so-far never leaves it and both seeds return the initial state --
275        // which is correct behaviour and a completely blind test. This assertion only means
276        // something where the sampler has somewhere to go.
277        let (mut d, _own) = dev_or_skip!();
278        let inst = ferrotherm::planted::frustrated_loops(12, 24, 5);
279        let p = Program::from_graph(&inst.graph, &Schedule::default());
280        assert!(d.program(&p).is_empty());
281
282        let hot = Schedule::constant(0.25, 30);
283        let a = d.run(&hot, 11).unwrap();
284        let b = d.run(&hot, 22).unwrap();
285        assert!(
286            inst.graph.energy(&a) < 0.0 && inst.graph.energy(&b) < 0.0,
287            "both runs should have moved off the initial state at all"
288        );
289        assert_ne!(a, b, "two seeds gave the same trajectory; the device is not threading the seed");
290    }
291
292    #[test]
293    fn the_write_is_charged_because_that_is_the_term_the_ledger_rests_on() {
294        let (mut d, _own) = dev_or_skip!();
295        assert_eq!(d.ledger().writes, 0);
296        let g = lattice2d(16, 1.0);
297        let p = Program::from_graph(&g, &Schedule::default());
298        assert!(d.program(&p).is_empty());
299        assert_eq!(d.ledger().writes, g.n as u64, "one write per node flashed");
300        assert_eq!(d.ledger().samples, 0, "loading is not sampling");
301
302        d.run(&Schedule::constant(0.6, 10), 3).unwrap();
303        assert_eq!(d.ledger().samples, g.n as u64 * 10, "one sample per node per sweep");
304    }
305
306    #[test]
307    fn running_without_a_program_is_an_error_not_an_empty_state() {
308        let (mut d, _own) = dev_or_skip!();
309        let e = d.run(&Schedule::constant(0.6, 10), 1).unwrap_err();
310        assert!(e.contains("no program"), "{e}");
311        // And an empty schedule runs nothing, which is worth saying rather than returning the
312        // initial state as though it had been sampled.
313        let g = lattice2d(8, 1.0);
314        assert!(d.program(&Program::from_graph(&g, &Schedule::default())).is_empty());
315        let e2 = d.run(&Schedule::new(), 1).unwrap_err();
316        assert!(e2.contains("empty schedule"), "{e2}");
317    }
318}