axiolid_contracts/execution.rs
1//! Execution policy passed explicitly to every costly operation.
2
3use axiolid_core::{Scalar, Tolerance};
4
5use crate::cancel::CancellationToken;
6use crate::{BackendId, GeomError, GeomResult, Precision};
7
8/// Reproducibility requirement.
9///
10/// These are three genuinely different contracts, not degrees of one. A backend
11/// that reduces in a deterministic order satisfies [`Self::Topological`] while
12/// still failing [`Self::Bitwise`] against a differently-scheduled backend, so
13/// collapsing them into one flag lets two backends both claim "deterministic"
14/// and still disagree. Ordered weakest to strongest.
15#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
16pub enum Determinism {
17 /// No guarantee. The backend may reorder, re-associate, and reschedule.
18 BestEffort,
19 /// Same connectivity and same output ordering for the same inputs.
20 ///
21 /// Coordinates may differ within tolerance. This is what a clash result or
22 /// a topology commit needs; it does not promise identical floats.
23 Topological,
24 /// Topological determinism plus values within the operation's stated
25 /// numerical error bound.
26 NumericallyBounded,
27 /// Bit-for-bit identical output for the same inputs and options.
28 ///
29 /// The only contract that supports hashing a result or comparing artifacts
30 /// across machines.
31 Bitwise,
32}
33
34impl Determinism {
35 /// Whether this guarantee is at least as strong as `required`.
36 ///
37 /// Strength is the declaration order, so a backend offering
38 /// [`Self::Bitwise`] satisfies a [`Self::Topological`] request but never
39 /// the reverse. Routing must use this instead of equality, or a stronger
40 /// backend gets rejected for being too good.
41 pub const fn satisfies(self, required: Self) -> bool {
42 (self as u8) >= (required as u8)
43 }
44}
45
46/// CPU scheduling preference.
47#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
48pub enum Parallelism {
49 /// One worker.
50 Serial,
51 /// Backend chooses from available parallelism and workload size.
52 Auto,
53 /// Upper bound on worker count. Zero is rejected by the builder.
54 Threads(usize),
55}
56
57/// Device selection preference. `Auto` is a policy request, not permission to
58/// silently reduce precision.
59#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
60pub enum DevicePreference {
61 /// Select from compatible registered backends.
62 Auto,
63 /// Require a CPU backend.
64 Cpu,
65 /// Require a GPU backend.
66 Gpu,
67 /// Require one named backend.
68 Backend(BackendId),
69}
70
71/// Where an operation's data physically lives.
72///
73/// Device preference says *where to run*; residency says *where the bytes
74/// already are*. Routing needs both: a GPU-resident batch is cheap to run on
75/// the GPU and expensive to run on the CPU, and the reverse holds for a
76/// host-resident one. Without this, a planner cannot see a transfer that
77/// dominates the operation it is scheduling.
78#[non_exhaustive]
79#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
80pub enum Residency {
81 /// Host (CPU) memory.
82 Host,
83 /// Memory owned by the named device.
84 Device(BackendId),
85 /// Host-visible memory shared with the named device; no copy is needed.
86 Unified(BackendId),
87}
88
89impl Residency {
90 /// Whether `backend` can read this data without a host/device transfer.
91 pub fn is_local_to(self, backend: BackendId) -> bool {
92 match self {
93 Self::Host => false,
94 Self::Device(owner) | Self::Unified(owner) => owner == backend,
95 }
96 }
97
98 /// Whether the host can read this data without a transfer.
99 pub const fn is_host_readable(self) -> bool {
100 matches!(self, Self::Host | Self::Unified(_))
101 }
102}
103
104/// Where an operation's inputs live and where its outputs are wanted.
105///
106/// Kept as a pair because they genuinely differ: a GPU broad phase may consume
107/// device-resident geometry and still have to deliver host-readable results.
108#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
109pub struct DataResidency {
110 input: Residency,
111 output: Residency,
112}
113
114impl DataResidency {
115 /// Both inputs and outputs in host memory. The portable default.
116 pub const HOST: Self = Self {
117 input: Residency::Host,
118 output: Residency::Host,
119 };
120
121 /// Construct an explicit input/output residency plan.
122 pub const fn new(input: Residency, output: Residency) -> Self {
123 Self { input, output }
124 }
125
126 /// Where inputs currently live.
127 pub const fn input(self) -> Residency {
128 self.input
129 }
130
131 /// Where outputs are wanted.
132 pub const fn output(self) -> Residency {
133 self.output
134 }
135
136 /// Whether running on `backend` needs no host/device transfer either way.
137 pub fn is_transfer_free_on(self, backend: BackendId) -> bool {
138 self.input.is_local_to(backend) && self.output.is_local_to(backend)
139 }
140}
141
142/// Scratch memory an operation needs beyond its inputs and outputs.
143///
144/// Declared up front so a caller can budget, pre-reserve, or refuse before any
145/// work starts. `Unbounded` is deliberately representable and deliberately
146/// unpleasant: an operation that cannot bound its scratch must say so rather
147/// than allocating silently in a hot loop.
148#[non_exhaustive]
149#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
150pub enum ScratchRequirement {
151 /// Operates in place; no scratch beyond inputs and outputs.
152 None,
153 /// At most `bytes` of scratch, independent of input size.
154 Fixed {
155 /// Upper bound in bytes.
156 bytes: usize,
157 },
158 /// At most `bytes_per_element * elements` of scratch.
159 PerElement {
160 /// Upper bound per input element, in bytes.
161 bytes_per_element: usize,
162 },
163 /// Scratch cannot be bounded ahead of time.
164 ///
165 /// Callers must treat this as "may allocate arbitrarily"; a memory budget
166 /// cannot be enforced against it.
167 Unbounded,
168}
169
170impl ScratchRequirement {
171 /// Upper bound for `elements` inputs, or `None` when unbounded.
172 pub const fn upper_bound_bytes(self, elements: usize) -> Option<usize> {
173 match self {
174 Self::None => Some(0),
175 Self::Fixed { bytes } => Some(bytes),
176 Self::PerElement { bytes_per_element } => bytes_per_element.checked_mul(elements),
177 Self::Unbounded => None,
178 }
179 }
180
181 /// Whether this requirement fits `options`' memory budget for `elements`.
182 ///
183 /// An unbounded requirement never fits a declared budget: allowing it would
184 /// make the budget advisory, which is the failure this type exists to stop.
185 /// Whether this requirement fits the caller's memory budget.
186 ///
187 /// No longer `const`: it reads an [`ExecutionOptions`] that now owns a
188 /// shared cancellation handle, so it cannot be destructured at compile
189 /// time. Budget checks happen once per dispatch, not on a hot path.
190 pub fn fits_budget(self, options: &ExecutionOptions, elements: usize) -> bool {
191 match options.memory_budget_bytes() {
192 None => true,
193 Some(budget) => match self.upper_bound_bytes(elements) {
194 Some(needed) => needed <= budget,
195 None => false,
196 },
197 }
198 }
199}
200
201/// Upper bound on how many outputs an operation produces per input element.
202///
203/// Declared so a caller can size a destination buffer *before* the operation
204/// runs. That is what makes a scan-then-scatter implementation possible: with a
205/// per-element bound, exclusive-prefix-summing the per-element counts gives
206/// every worker a disjoint write offset, so no lock, no atomic counter, and no
207/// dynamically growing vector is needed on the hot path.
208#[non_exhaustive]
209#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
210pub enum OutputBound {
211 /// Exactly one output per input. The output index equals the input index.
212 OneToOne,
213 /// At most `max` outputs per input element.
214 AtMost {
215 /// Inclusive upper bound per element.
216 max: usize,
217 },
218 /// The output count cannot be bounded before running.
219 ///
220 /// Callers must fall back to a growable collection. Kept representable and
221 /// deliberately unpleasant so an operation that could declare a bound is
222 /// not tempted to shrug.
223 Unbounded,
224}
225
226impl OutputBound {
227 /// Worst-case total outputs for `elements` inputs, or `None` if unbounded
228 /// or the product would overflow.
229 pub const fn upper_bound(self, elements: usize) -> Option<usize> {
230 match self {
231 Self::OneToOne => Some(elements),
232 Self::AtMost { max } => max.checked_mul(elements),
233 Self::Unbounded => None,
234 }
235 }
236
237 /// Whether a caller can preallocate an exact destination for `elements`.
238 pub const fn is_preallocatable(self, elements: usize) -> bool {
239 self.upper_bound(elements).is_some()
240 }
241
242 /// Exclusive prefix sum of `counts`, plus the total.
243 ///
244 /// This is the scan that turns "how many outputs does each element make?"
245 /// into "where does each element write?". Returns `None` if any count
246 /// exceeds this bound (a provider contract violation) or the total
247 /// overflows. Allocation is the caller's single result buffer; the scan
248 /// itself is a running total with no per-element allocation.
249 pub fn write_offsets(self, counts: &[usize]) -> Option<(Vec<usize>, usize)> {
250 let per_element_max = match self {
251 Self::OneToOne => Some(1),
252 Self::AtMost { max } => Some(max),
253 Self::Unbounded => None,
254 };
255 let mut offsets = Vec::with_capacity(counts.len());
256 let mut running = 0usize;
257 for &count in counts {
258 if let Some(max) = per_element_max {
259 if count > max {
260 return None;
261 }
262 }
263 offsets.push(running);
264 running = running.checked_add(count)?;
265 }
266 Some((offsets, running))
267 }
268}
269
270/// Operation policy with explicit tolerance, precision, and cancellation.
271///
272/// Deliberately not `Copy`: it carries a shared [`CancellationToken`], and an
273/// implicitly copied cancellation handle is a footgun. Callers clone when they
274/// mean to share the token and construct fresh options when they do not.
275#[derive(Debug, Clone, PartialEq)]
276pub struct ExecutionOptions {
277 tolerance: Tolerance,
278 precision: Precision,
279 determinism: Determinism,
280 parallelism: Parallelism,
281 device: DevicePreference,
282 residency: DataResidency,
283 memory_budget_bytes: Option<usize>,
284 cancellation: Option<CancellationToken>,
285 chord_error: Option<Scalar>,
286}
287
288impl ExecutionOptions {
289 /// Start from the required model-aware tolerance.
290 pub const fn new(tolerance: Tolerance) -> Self {
291 Self {
292 tolerance,
293 precision: Precision::F64,
294 determinism: Determinism::NumericallyBounded,
295 parallelism: Parallelism::Auto,
296 device: DevicePreference::Auto,
297 residency: DataResidency::HOST,
298 memory_budget_bytes: None,
299 cancellation: None,
300 chord_error: None,
301 }
302 }
303
304 /// Replace the tolerance while preserving every other execution policy.
305 ///
306 /// Compilers use this when evaluating geometry in a transformed local
307 /// coordinate system. Cloning first keeps cancellation, budgets,
308 /// determinism, residency, and provider preferences intact.
309 pub fn with_tolerance(mut self, tolerance: Tolerance) -> Self {
310 self.tolerance = tolerance;
311 self
312 }
313
314 /// Attach a cooperative cancellation token.
315 ///
316 /// Absent a token, an operation runs to completion; there is no ambient
317 /// cancellation source. Providers declare how finely they poll via
318 /// [`crate::CancellationGranularity`], so a caller can see the real latency.
319 pub fn with_cancellation(mut self, token: CancellationToken) -> Self {
320 self.cancellation = Some(token);
321 self
322 }
323
324 /// The attached cancellation token, if any.
325 pub fn cancellation(&self) -> Option<&CancellationToken> {
326 self.cancellation.as_ref()
327 }
328
329 /// `Err(GeomError::Cancelled)` if a token is attached and cancelled.
330 ///
331 /// The call providers make at each poll point: `options.check_cancelled()?`.
332 /// Cheap (one relaxed load) and a no-op when no token is attached.
333 pub fn check_cancelled(&self) -> crate::GeomResult<()> {
334 match &self.cancellation {
335 Some(token) => token.check(),
336 None => Ok(()),
337 }
338 }
339
340 /// Set required precision.
341 pub fn with_precision(mut self, precision: Precision) -> Self {
342 self.precision = precision;
343 self
344 }
345
346 /// Set determinism requirement.
347 pub fn with_determinism(mut self, value: Determinism) -> Self {
348 self.determinism = value;
349 self
350 }
351
352 /// Set scheduling preference. Returns `None` for zero explicit threads.
353 pub fn with_parallelism(mut self, value: Parallelism) -> Option<Self> {
354 if matches!(value, Parallelism::Threads(0)) {
355 return None;
356 }
357 self.parallelism = value;
358 Some(self)
359 }
360
361 /// Set device preference.
362 pub fn with_device(mut self, value: DevicePreference) -> Self {
363 self.device = value;
364 self
365 }
366
367 /// Declare where inputs live and where outputs are wanted.
368 pub fn with_residency(mut self, value: DataResidency) -> Self {
369 self.residency = value;
370 self
371 }
372
373 /// Bound how far flattened curves may deviate from the exact curve.
374 ///
375 /// Curved geometry that a provider approximates with straight chords
376 /// (profile arcs, sweep directrices, curved B-rep faces) stays within this
377 /// distance of the exact curve. Without it the chord budget is the linear
378 /// tolerance, which is a coincidence tolerance and coarse for small radii:
379 /// at `Tolerance::MILLIMETRE` a 5 mm arc gets four chords per half turn.
380 /// Quantity take-off wants a tighter budget than display.
381 ///
382 /// Returns `None` for a non-finite or non-positive value. A provider may
383 /// still refuse a budget too fine to meet within its own work limits.
384 pub fn with_chord_error(mut self, value: Scalar) -> Option<Self> {
385 if !(value.is_finite() && value > 0.0) {
386 return None;
387 }
388 self.chord_error = Some(value);
389 Some(self)
390 }
391
392 /// The explicit chord budget, if one was set.
393 ///
394 /// `None` means the provider uses its default, the linear tolerance.
395 pub fn chord_error(&self) -> Option<Scalar> {
396 self.chord_error
397 }
398
399 /// Bound temporary allocation.
400 pub fn with_memory_budget(mut self, bytes: usize) -> Self {
401 self.memory_budget_bytes = Some(bytes);
402 self
403 }
404
405 /// Tolerance.
406 pub fn tolerance(&self) -> Tolerance {
407 self.tolerance
408 }
409
410 /// Precision.
411 pub fn precision(&self) -> Precision {
412 self.precision
413 }
414
415 /// Determinism requirement.
416 pub fn determinism(&self) -> Determinism {
417 self.determinism
418 }
419
420 /// Scheduling preference.
421 pub fn parallelism(&self) -> Parallelism {
422 self.parallelism
423 }
424
425 /// Device preference.
426 pub fn device(&self) -> DevicePreference {
427 self.device
428 }
429
430 /// Optional temporary-memory budget.
431 pub fn memory_budget_bytes(&self) -> Option<usize> {
432 self.memory_budget_bytes
433 }
434
435 /// Where inputs live and where outputs are wanted.
436 pub fn residency(&self) -> DataResidency {
437 self.residency
438 }
439
440 /// Charge `bytes` of scratch against the budget before allocating it.
441 ///
442 /// The budget is only real if something checks it, so this is the single
443 /// enforcement point every hot path routes through. Backends must call it
444 /// *before* the allocation, not after: reporting an overrun once the
445 /// allocation already succeeded defeats the purpose of a budget.
446 pub fn charge_scratch(&self, bytes: usize) -> GeomResult<()> {
447 match self.memory_budget_bytes {
448 Some(budget) if bytes > budget => Err(GeomError::BudgetExceeded { resource: "memory" }),
449 _ => Ok(()),
450 }
451 }
452}