axiolid_contracts/
execution.rs

1//! Execution policy passed explicitly to every costly operation.
2
3use axiolid_core::{Scalar, Tolerance};
4
5use crate::cancel::CancellationToken;
6use crate::{BackendId, GeomError, GeomResult, Precision};
7
8/// Reproducibility requirement.
9///
10/// These are three genuinely different contracts, not degrees of one. A backend
11/// that reduces in a deterministic order satisfies [`Self::Topological`] while
12/// still failing [`Self::Bitwise`] against a differently-scheduled backend, so
13/// collapsing them into one flag lets two backends both claim "deterministic"
14/// and still disagree. Ordered weakest to strongest.
15#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
16pub enum Determinism {
17    /// No guarantee. The backend may reorder, re-associate, and reschedule.
18    BestEffort,
19    /// Same connectivity and same output ordering for the same inputs.
20    ///
21    /// Coordinates may differ within tolerance. This is what a clash result or
22    /// a topology commit needs; it does not promise identical floats.
23    Topological,
24    /// Topological determinism plus values within the operation's stated
25    /// numerical error bound.
26    NumericallyBounded,
27    /// Bit-for-bit identical output for the same inputs and options.
28    ///
29    /// The only contract that supports hashing a result or comparing artifacts
30    /// across machines.
31    Bitwise,
32}
33
34impl Determinism {
35    /// Whether this guarantee is at least as strong as `required`.
36    ///
37    /// Strength is the declaration order, so a backend offering
38    /// [`Self::Bitwise`] satisfies a [`Self::Topological`] request but never
39    /// the reverse. Routing must use this instead of equality, or a stronger
40    /// backend gets rejected for being too good.
41    pub const fn satisfies(self, required: Self) -> bool {
42        (self as u8) >= (required as u8)
43    }
44}
45
46/// CPU scheduling preference.
47#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
48pub enum Parallelism {
49    /// One worker.
50    Serial,
51    /// Backend chooses from available parallelism and workload size.
52    Auto,
53    /// Upper bound on worker count. Zero is rejected by the builder.
54    Threads(usize),
55}
56
57/// Device selection preference. `Auto` is a policy request, not permission to
58/// silently reduce precision.
59#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
60pub enum DevicePreference {
61    /// Select from compatible registered backends.
62    Auto,
63    /// Require a CPU backend.
64    Cpu,
65    /// Require a GPU backend.
66    Gpu,
67    /// Require one named backend.
68    Backend(BackendId),
69}
70
71/// Where an operation's data physically lives.
72///
73/// Device preference says *where to run*; residency says *where the bytes
74/// already are*. Routing needs both: a GPU-resident batch is cheap to run on
75/// the GPU and expensive to run on the CPU, and the reverse holds for a
76/// host-resident one. Without this, a planner cannot see a transfer that
77/// dominates the operation it is scheduling.
78#[non_exhaustive]
79#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
80pub enum Residency {
81    /// Host (CPU) memory.
82    Host,
83    /// Memory owned by the named device.
84    Device(BackendId),
85    /// Host-visible memory shared with the named device; no copy is needed.
86    Unified(BackendId),
87}
88
89impl Residency {
90    /// Whether `backend` can read this data without a host/device transfer.
91    pub fn is_local_to(self, backend: BackendId) -> bool {
92        match self {
93            Self::Host => false,
94            Self::Device(owner) | Self::Unified(owner) => owner == backend,
95        }
96    }
97
98    /// Whether the host can read this data without a transfer.
99    pub const fn is_host_readable(self) -> bool {
100        matches!(self, Self::Host | Self::Unified(_))
101    }
102}
103
104/// Where an operation's inputs live and where its outputs are wanted.
105///
106/// Kept as a pair because they genuinely differ: a GPU broad phase may consume
107/// device-resident geometry and still have to deliver host-readable results.
108#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
109pub struct DataResidency {
110    input: Residency,
111    output: Residency,
112}
113
114impl DataResidency {
115    /// Both inputs and outputs in host memory. The portable default.
116    pub const HOST: Self = Self {
117        input: Residency::Host,
118        output: Residency::Host,
119    };
120
121    /// Construct an explicit input/output residency plan.
122    pub const fn new(input: Residency, output: Residency) -> Self {
123        Self { input, output }
124    }
125
126    /// Where inputs currently live.
127    pub const fn input(self) -> Residency {
128        self.input
129    }
130
131    /// Where outputs are wanted.
132    pub const fn output(self) -> Residency {
133        self.output
134    }
135
136    /// Whether running on `backend` needs no host/device transfer either way.
137    pub fn is_transfer_free_on(self, backend: BackendId) -> bool {
138        self.input.is_local_to(backend) && self.output.is_local_to(backend)
139    }
140}
141
142/// Scratch memory an operation needs beyond its inputs and outputs.
143///
144/// Declared up front so a caller can budget, pre-reserve, or refuse before any
145/// work starts. `Unbounded` is deliberately representable and deliberately
146/// unpleasant: an operation that cannot bound its scratch must say so rather
147/// than allocating silently in a hot loop.
148#[non_exhaustive]
149#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
150pub enum ScratchRequirement {
151    /// Operates in place; no scratch beyond inputs and outputs.
152    None,
153    /// At most `bytes` of scratch, independent of input size.
154    Fixed {
155        /// Upper bound in bytes.
156        bytes: usize,
157    },
158    /// At most `bytes_per_element * elements` of scratch.
159    PerElement {
160        /// Upper bound per input element, in bytes.
161        bytes_per_element: usize,
162    },
163    /// Scratch cannot be bounded ahead of time.
164    ///
165    /// Callers must treat this as "may allocate arbitrarily"; a memory budget
166    /// cannot be enforced against it.
167    Unbounded,
168}
169
170impl ScratchRequirement {
171    /// Upper bound for `elements` inputs, or `None` when unbounded.
172    pub const fn upper_bound_bytes(self, elements: usize) -> Option<usize> {
173        match self {
174            Self::None => Some(0),
175            Self::Fixed { bytes } => Some(bytes),
176            Self::PerElement { bytes_per_element } => bytes_per_element.checked_mul(elements),
177            Self::Unbounded => None,
178        }
179    }
180
181    /// Whether this requirement fits `options`' memory budget for `elements`.
182    ///
183    /// An unbounded requirement never fits a declared budget: allowing it would
184    /// make the budget advisory, which is the failure this type exists to stop.
185    /// Whether this requirement fits the caller's memory budget.
186    ///
187    /// No longer `const`: it reads an [`ExecutionOptions`] that now owns a
188    /// shared cancellation handle, so it cannot be destructured at compile
189    /// time. Budget checks happen once per dispatch, not on a hot path.
190    pub fn fits_budget(self, options: &ExecutionOptions, elements: usize) -> bool {
191        match options.memory_budget_bytes() {
192            None => true,
193            Some(budget) => match self.upper_bound_bytes(elements) {
194                Some(needed) => needed <= budget,
195                None => false,
196            },
197        }
198    }
199}
200
201/// Upper bound on how many outputs an operation produces per input element.
202///
203/// Declared so a caller can size a destination buffer *before* the operation
204/// runs. That is what makes a scan-then-scatter implementation possible: with a
205/// per-element bound, exclusive-prefix-summing the per-element counts gives
206/// every worker a disjoint write offset, so no lock, no atomic counter, and no
207/// dynamically growing vector is needed on the hot path.
208#[non_exhaustive]
209#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
210pub enum OutputBound {
211    /// Exactly one output per input. The output index equals the input index.
212    OneToOne,
213    /// At most `max` outputs per input element.
214    AtMost {
215        /// Inclusive upper bound per element.
216        max: usize,
217    },
218    /// The output count cannot be bounded before running.
219    ///
220    /// Callers must fall back to a growable collection. Kept representable and
221    /// deliberately unpleasant so an operation that could declare a bound is
222    /// not tempted to shrug.
223    Unbounded,
224}
225
226impl OutputBound {
227    /// Worst-case total outputs for `elements` inputs, or `None` if unbounded
228    /// or the product would overflow.
229    pub const fn upper_bound(self, elements: usize) -> Option<usize> {
230        match self {
231            Self::OneToOne => Some(elements),
232            Self::AtMost { max } => max.checked_mul(elements),
233            Self::Unbounded => None,
234        }
235    }
236
237    /// Whether a caller can preallocate an exact destination for `elements`.
238    pub const fn is_preallocatable(self, elements: usize) -> bool {
239        self.upper_bound(elements).is_some()
240    }
241
242    /// Exclusive prefix sum of `counts`, plus the total.
243    ///
244    /// This is the scan that turns "how many outputs does each element make?"
245    /// into "where does each element write?". Returns `None` if any count
246    /// exceeds this bound (a provider contract violation) or the total
247    /// overflows. Allocation is the caller's single result buffer; the scan
248    /// itself is a running total with no per-element allocation.
249    pub fn write_offsets(self, counts: &[usize]) -> Option<(Vec<usize>, usize)> {
250        let per_element_max = match self {
251            Self::OneToOne => Some(1),
252            Self::AtMost { max } => Some(max),
253            Self::Unbounded => None,
254        };
255        let mut offsets = Vec::with_capacity(counts.len());
256        let mut running = 0usize;
257        for &count in counts {
258            if let Some(max) = per_element_max {
259                if count > max {
260                    return None;
261                }
262            }
263            offsets.push(running);
264            running = running.checked_add(count)?;
265        }
266        Some((offsets, running))
267    }
268}
269
270/// Operation policy with explicit tolerance, precision, and cancellation.
271///
272/// Deliberately not `Copy`: it carries a shared [`CancellationToken`], and an
273/// implicitly copied cancellation handle is a footgun. Callers clone when they
274/// mean to share the token and construct fresh options when they do not.
275#[derive(Debug, Clone, PartialEq)]
276pub struct ExecutionOptions {
277    tolerance: Tolerance,
278    precision: Precision,
279    determinism: Determinism,
280    parallelism: Parallelism,
281    device: DevicePreference,
282    residency: DataResidency,
283    memory_budget_bytes: Option<usize>,
284    cancellation: Option<CancellationToken>,
285    chord_error: Option<Scalar>,
286}
287
288impl ExecutionOptions {
289    /// Start from the required model-aware tolerance.
290    pub const fn new(tolerance: Tolerance) -> Self {
291        Self {
292            tolerance,
293            precision: Precision::F64,
294            determinism: Determinism::NumericallyBounded,
295            parallelism: Parallelism::Auto,
296            device: DevicePreference::Auto,
297            residency: DataResidency::HOST,
298            memory_budget_bytes: None,
299            cancellation: None,
300            chord_error: None,
301        }
302    }
303
304    /// Replace the tolerance while preserving every other execution policy.
305    ///
306    /// Compilers use this when evaluating geometry in a transformed local
307    /// coordinate system. Cloning first keeps cancellation, budgets,
308    /// determinism, residency, and provider preferences intact.
309    pub fn with_tolerance(mut self, tolerance: Tolerance) -> Self {
310        self.tolerance = tolerance;
311        self
312    }
313
314    /// Attach a cooperative cancellation token.
315    ///
316    /// Absent a token, an operation runs to completion; there is no ambient
317    /// cancellation source. Providers declare how finely they poll via
318    /// [`crate::CancellationGranularity`], so a caller can see the real latency.
319    pub fn with_cancellation(mut self, token: CancellationToken) -> Self {
320        self.cancellation = Some(token);
321        self
322    }
323
324    /// The attached cancellation token, if any.
325    pub fn cancellation(&self) -> Option<&CancellationToken> {
326        self.cancellation.as_ref()
327    }
328
329    /// `Err(GeomError::Cancelled)` if a token is attached and cancelled.
330    ///
331    /// The call providers make at each poll point: `options.check_cancelled()?`.
332    /// Cheap (one relaxed load) and a no-op when no token is attached.
333    pub fn check_cancelled(&self) -> crate::GeomResult<()> {
334        match &self.cancellation {
335            Some(token) => token.check(),
336            None => Ok(()),
337        }
338    }
339
340    /// Set required precision.
341    pub fn with_precision(mut self, precision: Precision) -> Self {
342        self.precision = precision;
343        self
344    }
345
346    /// Set determinism requirement.
347    pub fn with_determinism(mut self, value: Determinism) -> Self {
348        self.determinism = value;
349        self
350    }
351
352    /// Set scheduling preference. Returns `None` for zero explicit threads.
353    pub fn with_parallelism(mut self, value: Parallelism) -> Option<Self> {
354        if matches!(value, Parallelism::Threads(0)) {
355            return None;
356        }
357        self.parallelism = value;
358        Some(self)
359    }
360
361    /// Set device preference.
362    pub fn with_device(mut self, value: DevicePreference) -> Self {
363        self.device = value;
364        self
365    }
366
367    /// Declare where inputs live and where outputs are wanted.
368    pub fn with_residency(mut self, value: DataResidency) -> Self {
369        self.residency = value;
370        self
371    }
372
373    /// Bound how far flattened curves may deviate from the exact curve.
374    ///
375    /// Curved geometry that a provider approximates with straight chords
376    /// (profile arcs, sweep directrices, curved B-rep faces) stays within this
377    /// distance of the exact curve. Without it the chord budget is the linear
378    /// tolerance, which is a coincidence tolerance and coarse for small radii:
379    /// at `Tolerance::MILLIMETRE` a 5 mm arc gets four chords per half turn.
380    /// Quantity take-off wants a tighter budget than display.
381    ///
382    /// Returns `None` for a non-finite or non-positive value. A provider may
383    /// still refuse a budget too fine to meet within its own work limits.
384    pub fn with_chord_error(mut self, value: Scalar) -> Option<Self> {
385        if !(value.is_finite() && value > 0.0) {
386            return None;
387        }
388        self.chord_error = Some(value);
389        Some(self)
390    }
391
392    /// The explicit chord budget, if one was set.
393    ///
394    /// `None` means the provider uses its default, the linear tolerance.
395    pub fn chord_error(&self) -> Option<Scalar> {
396        self.chord_error
397    }
398
399    /// Bound temporary allocation.
400    pub fn with_memory_budget(mut self, bytes: usize) -> Self {
401        self.memory_budget_bytes = Some(bytes);
402        self
403    }
404
405    /// Tolerance.
406    pub fn tolerance(&self) -> Tolerance {
407        self.tolerance
408    }
409
410    /// Precision.
411    pub fn precision(&self) -> Precision {
412        self.precision
413    }
414
415    /// Determinism requirement.
416    pub fn determinism(&self) -> Determinism {
417        self.determinism
418    }
419
420    /// Scheduling preference.
421    pub fn parallelism(&self) -> Parallelism {
422        self.parallelism
423    }
424
425    /// Device preference.
426    pub fn device(&self) -> DevicePreference {
427        self.device
428    }
429
430    /// Optional temporary-memory budget.
431    pub fn memory_budget_bytes(&self) -> Option<usize> {
432        self.memory_budget_bytes
433    }
434
435    /// Where inputs live and where outputs are wanted.
436    pub fn residency(&self) -> DataResidency {
437        self.residency
438    }
439
440    /// Charge `bytes` of scratch against the budget before allocating it.
441    ///
442    /// The budget is only real if something checks it, so this is the single
443    /// enforcement point every hot path routes through. Backends must call it
444    /// *before* the allocation, not after: reporting an overrun once the
445    /// allocation already succeeded defeats the purpose of a budget.
446    pub fn charge_scratch(&self, bytes: usize) -> GeomResult<()> {
447        match self.memory_budget_bytes {
448            Some(budget) if bytes > budget => Err(GeomError::BudgetExceeded { resource: "memory" }),
449            _ => Ok(()),
450        }
451    }
452}