Skip to main content

shape_jit/mir_compiler/
mod.rs

1//! MIR-to-Cranelift IR compiler (JIT v2).
2//!
3//! Compiles directly from Shape's MIR (Mid-level IR) to Cranelift IR,
4//! preserving CFG structure, ownership semantics (Move/Copy/Drop),
5//! liveness, and storage plans that are lost in the bytecode encoding.
6//!
7//! # Architecture
8//!
9//! ```text
10//! AST → MIR (existing) → BorrowAnalysis + Liveness + StoragePlan (existing)
11//!                      → MirToIR (this module) → Cranelift IR → native code
12//! ```
13//!
14//! # Key differences from BytecodeToIR
15//!
16//! - **1:1 block mapping**: MIR BasicBlocks map directly to Cranelift blocks
17//! - **Ownership-aware**: Move nulls the source, Copy retains, Drop releases
18//! - **~7 statement kinds** vs ~100 bytecode opcodes
19//! - **Explicit Drop points**: Scope cleanup from MIR, not heuristic
20
21mod blocks;
22pub mod bounds_elision;
23mod conversions;
24mod ownership;
25mod places;
26mod rvalues;
27mod statements;
28mod terminators;
29pub(crate) mod types;
30pub(crate) mod v2_array;
31pub(crate) mod v2_field;
32pub(crate) mod v2_int;
33pub(crate) mod v2_refcount;
34pub(crate) mod v2_string;
35pub(crate) mod v2_typed_map;
36
37// Heavy execution-path tests — gated behind the `deep-tests` feature.
38// Each test calls JITExecutor::execute_program, which JIT-compiles ~118
39// stdlib functions via MirToIR. Running them on the default Tier 1 path
40// at n-cpu parallelism makes the shape-jit test binary slow enough to
41// miss the summary line and racy enough to SIGILL in the JIT code cache.
42// See `just test-deep` to run.
43#[cfg(all(test, feature = "deep-tests"))]
44mod integration_tests;
45
46#[cfg(all(test, feature = "deep-tests"))]
47mod v2_array_tests;
48
49// Re-gated post-W11 reopen verification: with principled arc_retain/
50// release (W11-jit-new-array), the SIGABRT source for these tests is
51// confirmed to be `ffi/control/mod.rs:171::jit_call_value` whose body
52// is `todo!("phase-2c §2.7.10/Q11 + §2.7.11/Q12: JIT-side kinded
53// value-call ABI rebuild")` — NOT a retain/release issue. The
54// closure-dispatch tests exercise the §2.7.11 / Q12 value-call ABI
55// (callee/args kind-stamping) which is W11-jit-carrier-conversion's
56// territory; the W11-jit-new-array charter is limited to the array
57// FFI surface + arc_retain/release. Re-enable when the kinded
58// value-call ABI lands at `ffi/control/mod.rs:171`.
59#[cfg(all(test, feature = "deep-tests"))]
60mod closure_dispatch_regression_tests;
61
62// Phase 4b Round 5c-2-α jit-shortcircuit-eager soundness fix regression
63// tests (v0.3-gating per supervisor ratify 2026-05-19; sister-class to
64// LANG-9-spin-3-first VM/JIT divergence). Pin the t25 reproducer +
65// divzero / chained / nested short-circuit cases against the MIR-layer
66// short-circuit lowering at `crates/shape-vm/src/mir/lowering/expr.rs`
67// `lower_short_circuit_and_or`. Gated behind `deep-tests` per the
68// `closure_dispatch_regression_tests` precedent (full JIT pipeline
69// invocation; non-trivial compile time per test).
70#[cfg(all(test, feature = "deep-tests"))]
71mod short_circuit_regression_tests;
72
73// Phase 4b Round 5c-2-α jit-ref-param-chain-stamp regression tests
74// (ADR-006 §2.7.13 + §2.7.5; supervisor ratify 2026-05-19). Gated
75// behind `deep-tests` for the same reason as
76// `closure_dispatch_regression_tests` above — `JITExecutor::execute_program`
77// JIT-compiles the stdlib on every test, so default-parallelism CI runs
78// would race the JIT code cache.
79#[cfg(all(test, feature = "deep-tests"))]
80mod ref_param_regression_tests;
81
82// γ-CP4 jit-makefieldref regression tests (ADR-006 §2.7.13 + §2.3;
83// v0.3-gating NO-KNOWN-INCORRECTNESS). Pin the JIT codegen for
84// `MakeFieldRef` — `&`/`&mut` references projecting into a typed-object
85// field — against the field-address path in `rvalues.rs::Rvalue::Borrow`
86// + `places.rs::emit_typed_field_address`. Sister-class to
87// `ref_param_regression_tests` (the `Place::Local` ref-param chain).
88// Gated behind `deep-tests` for the same reason: `JITExecutor::
89// execute_program` JIT-compiles the stdlib on every test.
90#[cfg(all(test, feature = "deep-tests"))]
91mod field_ref_regression_tests;
92// v0.3 γ-CP3 jit-array-builder regression tests. Pin the array-spread +
93// destructure-rest reproducers against the honest surface-and-stop fix
94// (MIR slice-shape lowering of `...rest` + `emit_v2_array_aggregate`
95// heap-pointer-operand rejection). Gated behind `deep-tests` for the
96// same reason as the sibling regression modules above.
97#[cfg(all(test, feature = "deep-tests"))]
98mod array_builder_regression_tests;
99// v0.3 γ-CP9 jit-groupby-surface regression tests. Pin the array
100// `groupBy` / `count` / `group` reproducers against the honest
101// surface-and-stop fix (`try_emit_v2_array_method` compile-stage `Err`
102// + the `jit_call_method` defense-in-depth closure-arg guard). Gated
103// behind `deep-tests` for the same reason as the sibling regression
104// modules above — `JITExecutor::execute_program` JIT-compiles the
105// stdlib on every test.
106#[cfg(all(test, feature = "deep-tests"))]
107mod groupby_surface_regression_tests;
108
109// γ-CP5 jit-typedarray-ptr regression tests (v0.3-gating
110// NO-KNOWN-INCORRECTNESS). Pin two JIT bugs un-masked by the Family-α
111// TypedArray fix: 7a — `Place::Index` codegen on a `Place::Field` base
112// (`b.items[i]` for a struct field of type `Array<int>`) must use the
113// v2 `TypedArray` layout (data@8/len@16), recognised via the
114// schema-derived `field_array_elem_kinds` map; 7b — `jit_call_value`
115// must retain each heap-typed closure capture (kind-driven
116// `KindedSlot::clone`) before handing it to `jit_trampoline_call_closure`,
117// which builds a fresh `OwnedClosureBlock` whose `Drop` releases each
118// capture. Gated behind `deep-tests` for the same reason as
119// `field_ref_regression_tests`: `JITExecutor::execute_program`
120// JIT-compiles the stdlib on every test.
121#[cfg(all(test, feature = "deep-tests"))]
122mod typedarray_ptr_regression_tests;
123
124// v0.3 WS-7 jit-array-param-fix regression tests (v0.3-gating
125// NO-KNOWN-INCORRECTNESS). Pin the SIGSEGV crash where a named function
126// with an UNANNOTATED array parameter, indexed `xs[i]`, crashed in JIT
127// mode once tier-compiled — even on a valid in-bounds access. Root cause:
128// the inferred pass-by-reference optimization marked the param as a
129// reference (callee auto-deref) while the JIT/MIR caller passed the heap
130// pointer by value. Gated behind `deep-tests` for the same reason as
131// `typedarray_ptr_regression_tests`: `JITExecutor::execute_program`
132// JIT-compiles the stdlib on every test.
133#[cfg(all(test, feature = "deep-tests"))]
134mod jit_array_param_regression_tests;
135
136use cranelift::codegen::ir::{FuncRef, StackSlot};
137use cranelift::prelude::*;
138use std::collections::{HashMap, HashSet};
139use std::sync::Arc;
140
141use crate::ffi_refs::FFIFuncRefs;
142use shape_value::v2::closure_layout::ClosureLayout;
143use shape_value::v2::struct_layout::FieldKind;
144use shape_value::v2::ConcreteType;
145use shape_vm::bytecode::MirFunctionData;
146use shape_vm::mir::types::*;
147use shape_vm::type_tracking::NativeKind;
148
149/// Session 2: side-table entry for a non-escaping stack closure call.
150///
151/// Carries the function_id, per-capture byte offset, and per-capture
152/// Cranelift type recorded at `emit_stack_closure` time. The indirect
153/// `Call` terminator consults this when the callee operand resolves to
154/// a slot in `MirToIR::stack_closure_call_info` and emits a direct
155/// `user_func_refs[function_id]` call with captures loaded from the
156/// stack slot instead of routing through the `jit_call_value` FFI.
157#[derive(Debug, Clone)]
158pub(crate) struct StackClosureCallInfo {
159    /// Target function_id (matches the `StackClosure.function_id` field).
160    pub(crate) function_id: u16,
161    /// Per-capture byte offset inside the `StackSlot`.
162    pub(crate) capture_offsets: Vec<i32>,
163    /// Per-capture native Cranelift type (F64 / I64 / I32 / I16 / I8 / Bool).
164    pub(crate) capture_types: Vec<cranelift::prelude::Type>,
165}
166
167/// MIR-to-Cranelift IR compiler.
168///
169/// Each instance compiles a single MIR function. Reuses the JIT's existing
170/// FFI infrastructure (250+ function references) and type mapping.
171pub struct MirToIR<'a, 'b> {
172    /// Cranelift function builder.
173    pub(crate) builder: &'a mut FunctionBuilder<'b>,
174    /// JITContext pointer (passed as first function parameter).
175    pub(crate) ctx_ptr: Value,
176    /// FFI function references (arc_retain, arc_release, print, etc.).
177    pub(crate) ffi: FFIFuncRefs,
178    /// The caller's entry block (already created, with function params).
179    /// MIR bb0 maps to this block instead of creating a new one.
180    pub(crate) entry_block: Block,
181
182    // ── Block mapping ──────────────────────────────────────────────
183    /// MIR BasicBlockId → Cranelift Block.
184    pub(crate) block_map: HashMap<BasicBlockId, Block>,
185
186    // ── Local variables ────────────────────────────────────────────
187    /// MIR SlotId → Cranelift Variable.
188    pub(crate) locals: HashMap<SlotId, Variable>,
189    /// Type info for each local slot (from MIR's LocalTypeInfo).
190    pub(crate) local_types: Vec<LocalTypeInfo>,
191    /// Frame descriptor slot kinds (from bytecode Function.frame_descriptor),
192    /// enriched by MIR-level type inference. `None` per slot means the
193    /// inference pass left the kind undetermined — codegen consumers
194    /// surface-and-stop on `None` per ADR-006 §2.7.7 (no deleted
195    /// `NativeKind::Unknown` placeholder).
196    pub(crate) slot_kinds: Vec<Option<NativeKind>>,
197    /// v2: Per-slot fully-resolved `ConcreteType` from the bytecode compiler's
198    /// `function_local_concrete_types` / `top_level_local_concrete_types`
199    /// side-tables. Used by the v2 typed-array codegen path. Empty when the
200    /// bytecode compiler did not populate the side-table — callers fall back
201    /// to the legacy NaN-boxed path.
202    pub(crate) concrete_types: Vec<ConcreteType>,
203    /// Next Cranelift variable index.
204    pub(crate) next_var: usize,
205
206    // ── MIR data ───────────────────────────────────────────────────
207    /// The MIR function being compiled.
208    pub(crate) mir: &'a MirFunction,
209    /// Borrow analysis (for ownership decisions).
210    pub(crate) mir_data: &'a MirFunctionData,
211    /// String table for resolving StringId constants.
212    pub(crate) strings: &'a [String],
213    /// Function name → index mapping for resolving Call terminators.
214    pub(crate) function_indices: &'a HashMap<String, u16>,
215
216    // ── Direct call support ─────────────────────────────────────────
217    /// Function index → Cranelift FuncRef for direct calls (bypasses FFI).
218    pub(crate) user_func_refs: HashMap<u16, FuncRef>,
219    /// Function index → arity for call validation.
220    pub(crate) user_func_arities: HashMap<u16, u16>,
221
222    // ── Borrow support ──────────────────────────────────────────────
223    /// MIR SlotId → (Cranelift StackSlot, Cranelift Type) for references
224    /// created by `Rvalue::Borrow`. After calls, all referenced locals are
225    /// reloaded from their stack slots using the recorded native type.
226    ///
227    /// R4.2F: the type is tracked so `reload_referenced_locals` can issue a
228    /// native-width `stack_load` that matches both the `stack_store` width
229    /// and the declared variable type. Non-native slot kinds map to I64 via
230    /// `cranelift_type_for_slot`, collapsing to the legacy 8-byte cell.
231    pub(crate) ref_stack_slots: HashMap<SlotId, (StackSlot, Type)>,
232    /// Mapping from field name to byte offset within a TypedObject.
233    pub(crate) field_byte_offsets: HashMap<String, u16>,
234
235    /// W12-jit-binop-after-heap-read-kind-tracker (ADR-006 §2.7.5
236    /// stamp-at-compile-time): field-name → `NativeKind` map populated
237    /// by the producer-side MIR walk (`infer_field_native_kinds` in
238    /// `types.rs`). Every `StatementKind::ObjectStore { operands,
239    /// field_names, .. }` stamps each named operand's MIR-inferred kind
240    /// here, threading the producer's kind classification across the
241    /// `Place::Field` projection at consumer sites — specifically the
242    /// `Rvalue::BinaryOp` lowering in `rvalues.rs`, which needs proven
243    /// operand kinds at compile time per CLAUDE.md "Forbidden code"
244    /// (runtime tag_bits dispatch deleted with the W-series IC).
245    ///
246    /// Keying by field name (not `FieldIdx` or `StructLayoutId`) mirrors
247    /// the existing `field_byte_offsets` discipline; both maps have the
248    /// same structural caveat ("last-writer-wins on name collision
249    /// across distinct struct types") but cover every load-bearing
250    /// cluster-0 smoke. A schema-aware `(StructLayoutId, FieldIdx) →
251    /// NativeKind` registry is the principled long-term shape — out of
252    /// scope for this sub-cluster.
253    ///
254    /// Populated once at `MirToIR::new_with_closure_layouts` time so
255    /// the kind is available for cross-block field reads, mirroring how
256    /// `slot_kinds` is computed pre-codegen via `infer_slot_kinds`.
257    pub(crate) field_native_kinds: HashMap<String, NativeKind>,
258
259    /// γ-CP5 7a (jit-typedarray-ptr): field-name → v2 typed-array
260    /// **element** `NativeKind` for struct fields declared `Array<T>`
261    /// with a scalar element `T`.
262    ///
263    /// `field_native_kinds` (above) collapses every `Array<T>` field to
264    /// `NativeKind::Ptr(HeapKind::TypedArray)` — the element type is
265    /// erased. The v2 `Place::Index` fast path (`v2_typed_array_elem_kind`
266    /// in `v2_array.rs`) needs the element kind to pick the inline
267    /// `v2_array_get` codegen (v2 layout: data@8 / len@16) instead of the
268    /// legacy `inline_array_get` (v1 layout: data@+0 / len@+8 past an
269    /// 8-byte header). Without it a `b.items[i]` access where `items` is
270    /// a struct field of type `Array<int>` falls through to the v1 path
271    /// and reads the wrong element offset.
272    ///
273    /// Keyed by field name, matching the `field_byte_offsets` /
274    /// `field_native_kinds` discipline (same "last-writer-wins on name
275    /// collision across distinct struct types" caveat — benign for the
276    /// single-receiver-type field reads this fast path serves).
277    ///
278    /// Per ADR-006 §2.7.5 producer-side stamp: derived from the canonical
279    /// `TypeSchemaRegistry` `FieldType::Array(elem)` declaration at
280    /// `populate_field_byte_offsets_from_schemas` time — a compile-time
281    /// index, not a runtime decode.
282    pub(crate) field_array_elem_kinds: HashMap<String, NativeKind>,
283
284    // ── Closure Spec Phase E: stack-allocated closures ──────────────
285    /// Slots that hold a non-escaping closure value, per the MIR
286    /// storage plan's `non_escaping_closure_slots`. When a
287    /// `StatementKind::ClosureCapture` targets a slot in this set,
288    /// codegen allocates a Cranelift `StackSlot` shaped like
289    /// `StackClosure { function_id: u32, type_id: u32, captures... }`
290    /// instead of calling `jit_make_closure`. Cranelift's SROA then
291    /// eliminates the slot when Phase C has inlined the closure body
292    /// and the env pointer is dead.
293    pub(crate) non_escaping_closure_slots: HashSet<SlotId>,
294    /// MIR SlotId → Cranelift `StackSlot` backing a non-escaping
295    /// closure. Populated on `ClosureCapture`. Used by drop/release
296    /// paths to skip `arc_release` on stack-resident closure handles
297    /// and by other consumers that need to know the slot is stack-resident.
298    pub(crate) stack_closure_slots: HashMap<SlotId, StackSlot>,
299
300    /// Session 2: per-slot stack-closure call metadata captured alongside
301    /// `stack_closure_slots`. When an indirect `Call` whose `func` operand
302    /// resolves to a slot in this map dispatches the closure, the
303    /// terminator can bypass `jit_call_value` entirely — the function_id
304    /// and capture byte offsets/Cranelift types are baked into codegen.
305    ///
306    /// This closes the hole where a stack closure's callee bits are a raw
307    /// stack pointer (no NaN-box tag, no `HK_CLOSURE` header) that the
308    /// FFI dispatcher can't recognise — the fix is to not dispatch through
309    /// the FFI at all when the JIT itself built the closure.
310    pub(crate) stack_closure_call_info:
311        HashMap<SlotId, StackClosureCallInfo>,
312
313    // ── Phase 4b Round 5c-2-α jit-ref-param-chain-stamp ────────────
314    /// Param slots whose source-declaration carries a reference borrow kind
315    /// (`&x` / `&mut x`). Populated at JIT compile-entry from
316    /// `mir.param_reference_kinds`. Read/write/null sites for these slots
317    /// auto-dispatch through the cell-indirection path (load/store at the
318    /// referent address) instead of the slot's raw local variable.
319    ///
320    /// ADR-006 §2.7.13 ref-chain stamp + §2.7.5 producer-side stamp:
321    /// MIR-lowering at `crates/shape-vm/src/mir/lowering/mod.rs:617-628`
322    /// classifies reference parameters as `LocalTypeInfo::NonCopy` and
323    /// records `param_reference_kinds[i] = Some(BorrowKind::*)` but does NOT
324    /// emit `Place::Deref` projections for the body's `x = x + 1` style
325    /// reads/writes — the slot is treated as if it held the referenced
326    /// value directly (`crates/shape-vm/src/mir/lowering/expr.rs:24-25`
327    /// returns `Place::Local(slot)` for `Expr::Identifier`, and
328    /// `crates/shape-vm/src/mir/lowering/stmt.rs:307` assigns to
329    /// `Place::Local(slot)` on identifier-target assignments).
330    ///
331    /// The BYTECODE compile path handles this orthogonally by emitting
332    /// `DerefLoad` / `DerefStore` against the ref-slot at
333    /// `compiler/expressions/identifiers.rs:219-221` (read) and the
334    /// assignment lowering (write). The W14.2-G4 close at
335    /// `compiler/functions.rs:1331-1390` further fixes a producer-side
336    /// kind-stamping race on the bytecode side. The JIT-MIR consumer was
337    /// never updated to honor reference semantics — calling
338    /// `bump(&a); print(a)` returns the un-mutated `a` because the JIT
339    /// reads the slot's raw pointer bits, adds 1, and writes the result
340    /// back into the same local slot (never touching the caller's cell).
341    ///
342    /// W14.2-G4 was VM-only by composition: the `tools/shape-test`
343    /// harness's `BytecodeExecutor` (`tools/shape-test/src/shape_test.rs:
344    /// 237`) runs every assertion via the bytecode interpreter — JIT
345    /// divergence is not surfaced. The empirical W15.2-F SURFACE at HEAD
346    /// `989b18d6` and supervisor ratify 2026-05-19 promote this to
347    /// v0.3-gating soundness.
348    ///
349    /// Sister-class to LANG-9-spin-3-first / W14.2-E SURFACE-A. The fix
350    /// site here mirrors the bytecode-compiler's ref-slot handling shape:
351    /// reads auto-deref, writes auto-deref, no NEW cell allocation for
352    /// re-borrowing `&x` of an existing ref-param.
353    pub(crate) ref_param_slots: HashSet<SlotId>,
354
355    // ── Closure Spec Phase H1: heap-allocated closure codegen ──────
356    /// Map from closure body `function_id` to its `ClosureLayout`.
357    /// When present, `emit_heap_closure` uses the layout to emit inline
358    /// Cranelift code that allocates a `TypedClosureHeader`-shaped block
359    /// and writes captures at their natural-width offsets, replacing the
360    /// legacy `jit_make_closure` FFI call. Absent entries fall back to
361    /// the FFI path (e.g. when loading a cached program from disk, which
362    /// doesn't carry layout metadata).
363    pub(crate) closure_function_layouts: HashMap<u16, Arc<ClosureLayout>>,
364
365    // ── Track A.1D.2: OwnedMutable capture side-table ──────────────
366    /// Local slots whose Cranelift variable holds the raw `*mut ValueWord`
367    /// bits of an `OwnedMutable` capture cell (allocated by
368    /// `jit_alloc_owned_mut_cell` in `emit_heap_closure`). For a closure
369    /// compiled under this `MirToIR`, the leading `N` entries of
370    /// `MirFunction::param_slots` correspond to captures in the same
371    /// order as `ClosureLayout::capture_kinds`; each slot whose
372    /// `capture_storage_kind(i) == OwnedMutable` is recorded here.
373    ///
374    /// Effects on the lowering pipeline:
375    /// - `read_place(Local(s))` emits `load.i64 [cell_ptr, 0]` (matches
376    ///   the interpreter's `op_load_owned_mutable_capture` fresh read).
377    /// - `write_place(Local(s), v)` emits `store.i64 v, [cell_ptr, 0]`
378    ///   (matches the interpreter's `op_store_owned_mutable_capture`
379    ///   fresh write — no old-value release, no retain).
380    /// - `null_place` / `release_old_value_if_heap` / `emit_drop` all
381    ///   early-return for these slots: the cell pointer bits must
382    ///   survive for the entire frame so every read/write finds the
383    ///   right box, and the box is reclaimed exactly once by
384    ///   `release_typed_closure`'s `Box::from_raw` loop (see
385    ///   `ClosureLayout::owned_mutable_capture_mask`, A.1A).
386    ///
387    /// Empty when the function being compiled is not a closure body,
388    /// or has no OwnedMutable captures — non-closure functions then
389    /// behave identically to pre-A.1D.2.
390    ///
391    /// Wave C.2: the value carries the `FieldKind` of the cell's interior
392    /// payload (from `ClosureLayout::capture_inner_kind`). The Cranelift
393    /// codegen for `read_place` / `write_place` dispatches on this kind
394    /// to select the matching per-FieldKind FFI helper
395    /// (`jit_read_owned_mut_cell_<kind>` / `jit_write_owned_mut_cell_<kind>`),
396    /// so values cross the cell boundary as native Cranelift types
397    /// (i64/f64/i32/...) instead of NaN-boxed ValueWord bits.
398    pub(crate) owned_mutable_capture_slots: HashMap<SlotId, FieldKind>,
399
400    // ── Track A.1E: Shared capture side-table ─────────────────────
401    /// Local slots whose Cranelift variable holds the raw
402    /// `*const SharedCell` bits of a `Shared` capture cell (retained via
403    /// `jit_arc_shared_retain` in `emit_heap_closure`). Structurally
404    /// parallel to `owned_mutable_capture_slots`: the leading `N`
405    /// entries of `MirFunction::param_slots` are captures, and each slot
406    /// whose `capture_storage_kind(i) == Shared` is recorded here.
407    ///
408    /// Effects on the lowering pipeline:
409    /// - `read_place(Local(s))` emits the inline lock fast path (CAS
410    ///   state byte 0→1 with `Acquire` ordering; on failure, call
411    ///   `jit_shared_lock_contended`), then `load.i64 [cell_ptr,
412    ///   SHARED_CELL_VALUE_OFFSET]`, then inline unlock fast path
413    ///   (CAS 1→0 with `Release` ordering; on failure, call
414    ///   `jit_shared_unlock_contended`). Matches the interpreter's
415    ///   `op_load_shared_capture` handler semantics (take mutex, clone
416    ///   inner bits, drop guard).
417    /// - `write_place(Local(s), v)` emits the same lock fast path,
418    ///   then `store.i64 v, [cell_ptr, SHARED_CELL_VALUE_OFFSET]`,
419    ///   then the unlock fast path. Matches
420    ///   `op_store_shared_capture` (take mutex, write, drop guard).
421    /// - `null_place` / `release_old_value_if_heap` / `emit_drop` all
422    ///   early-return for these slots: the Arc pointer bits must
423    ///   survive for the entire frame so every read/write finds the
424    ///   right cell, and the share is reclaimed exactly once by
425    ///   `release_typed_closure`'s `Arc::from_raw` loop (see
426    ///   `ClosureLayout::shared_capture_mask`, A.1A).
427    ///
428    /// Mutually exclusive with `owned_mutable_capture_slots` per the
429    /// `ClosureLayout` invariant (the three capture-kind masks are
430    /// disjoint). Empty when the function being compiled is not a
431    /// closure body, or has no Shared captures.
432    ///
433    /// Wave C.2: like `owned_mutable_capture_slots`, the value carries
434    /// the inner `FieldKind` so the Cranelift load/store after the
435    /// inline `emit_shared_lock` dispatches to the correct native
436    /// width at `[cell_ptr + SHARED_CELL_VALUE_OFFSET]`. We keep the
437    /// inline lock/unlock — only the per-kind direct load/store is
438    /// per-FieldKind — to avoid double-locking through the
439    /// `read_shared_cell_<kind>` FFI on the JIT hot path.
440    pub(crate) shared_capture_slots: HashMap<SlotId, FieldKind>,
441
442    // ── Session 1 Commit 3: outer-scope Shared-cell slot side-table ─
443    /// Local slots whose `BindingStorageClass` is `SharedCow` — i.e.
444    /// outer-scope `var` bindings that escape into a closure and hence
445    /// get promoted to `Arc<SharedCell>` storage by the bytecode
446    /// compiler (`AllocSharedLocal` on promotion;
447    /// `Load/StoreSharedLocal` on every subsequent access;
448    /// `DropSharedLocal` at scope exit — see
449    /// `shape-vm/src/executor/variables/mod.rs`).
450    ///
451    /// MIR doesn't reflect that promotion directly — it emits plain
452    /// `Assign(Local(s), ...)` and `Drop(Local(s))` on the slot — so
453    /// the JIT must recognise SharedCow slots via this side-table and
454    /// dispatch read/write/drop to the lock-gated + Arc-lifecycle
455    /// lowering path.
456    ///
457    /// Effects on the lowering pipeline:
458    /// - `initialize_shared_local_slots` (called once at the start of
459    ///   `compile`) allocates a fresh `Arc<SharedCell>` per slot via
460    ///   `jit_alloc_shared_cell(NONE_BITS)` and stores the pointer
461    ///   bits into the slot's Cranelift variable.
462    /// - `read_place(Local(s))` emits the inline lock-gated
463    ///   `load.i64 [cell_ptr + SHARED_CELL_VALUE_OFFSET]` (same lowering
464    ///   as `shared_capture_slots` — see
465    ///   `emit_shared_lock`/`emit_shared_unlock`).
466    /// - `write_place(Local(s), v)` emits the matching lock-gated
467    ///   store.
468    /// - `emit_drop(Local(s))` calls `jit_arc_shared_release` to
469    ///   consume the slot's strong share.
470    /// - `compile_operand_for_shared_capture` (new) emits a raw
471    ///   pointer read — bypassing the lock — so `ClosureCapture`
472    ///   operands install the outer cell pointer into the closure's
473    ///   Shared capture slot without locking.
474    ///
475    /// Disjoint from `owned_mutable_capture_slots` and
476    /// `shared_capture_slots` — those are leading-capture param slots
477    /// of a closure BODY; `shared_local_slots` is a declaring-scope
478    /// slot in the outer function.
479    pub(crate) shared_local_slots: HashSet<SlotId>,
480
481    // ── JIT-side back-patch for unresolved ClosurePlaceholder ──────
482    /// Per-placeholder function_id, populated at construction by scanning the
483    /// MIR in block/statement order. Each entry corresponds to a
484    /// `ClosurePlaceholder` assign that the bytecode compiler's back-patcher
485    /// did NOT replace with `Function(name)` — typically because
486    /// `closure_function_ids` got cleared by a monomorphization-triggered
487    /// `compile_function` call before the top-level-MIR patching ran.
488    ///
489    /// When `compile_constant(MirConstant::ClosurePlaceholder)` is invoked,
490    /// we pop the head of this queue (via `next_closure_placeholder_idx`)
491    /// and NaN-box the corresponding function id so the stack carries a
492    /// proper `TAG_FUNCTION` bit pattern instead of literal 0. Without this
493    /// the JIT's `jit_call_value` sees `0x0` for a no-capture closure and
494    /// bails out with "callee is neither function nor closure", which is
495    /// the root cause of the gated `parity_array_map/filter/reduce`
496    /// failures.
497    ///
498    /// Empty for MIRs that have no unresolved placeholders, for closure
499    /// bodies themselves (their `ClosureCapture` statements carry the
500    /// resolved `function_id` directly), and for functions whose
501    /// back-patching already succeeded.
502    pub(crate) closure_placeholder_fids: Vec<u16>,
503    /// Cursor into `closure_placeholder_fids`. Incremented once per call to
504    /// `compile_constant(ClosurePlaceholder)`. Statement visit order during
505    /// `compile_body` matches the scan order used by
506    /// `scan_closure_placeholder_fids`, so this is a stable pairing.
507    pub(crate) next_closure_placeholder_idx: std::cell::Cell<usize>,
508    /// Bounds-check elision plan: pairs `(arr_slot, iv_slot)` for which
509    /// `Place::Index(Local(arr), Operand::*(Local(iv)))` accesses can skip
510    /// the inline bounds check. Populated by callers via
511    /// `set_bounds_elision_plan` after running
512    /// `bounds_elision::analyze(mir)`. Empty by default — falls back to
513    /// the bounds-checked path, preserving the v2_array_tests OOB
514    /// zero-default semantics.
515    pub(crate) bounds_elision: bounds_elision::BoundsElisionPlan,
516
517    // ── V3-S6c JIT method-monomorph routing side-table ─────────────
518    /// ADR-006 §2.7.5 V3-S6c-jit-method-monomorph-routing (PATH α-prime
519    /// per supervisor 2026-05-15 ratification): the V3-S6b side-table
520    /// `BytecodeProgram.monomorphized_method_call_sites` cloned into the
521    /// JIT MirToIR so the Call-terminator compile path can re-route
522    /// `MirConstant::Method` Call terminators to direct Cranelift FuncRef
523    /// calls via `user_func_refs[specialized_idx]`. Key is
524    /// `(call_site_span, caller_function_id)` where `caller_function_id`
525    /// is the bytecode compiler's `self.current_function` at
526    /// `try_monomorphize_method_call` success (matches the JIT-side
527    /// `caller_function_id` field below).
528    ///
529    /// Empty when the bytecode compiler did not specialize any method
530    /// call (no generic method calls in the program, or all monomorph
531    /// attempts bailed). JIT falls through to the existing
532    /// `jit_call_method` trampoline path for any miss — preserves V3-S6b
533    /// baseline behaviour.
534    pub(crate) monomorphized_method_call_sites:
535        HashMap<(shape_ast::ast::span::Span, Option<usize>), usize>,
536
537    /// V3-S6c routing: the caller function id used as the second
538    /// component of the `monomorphized_method_call_sites` composite key.
539    /// `None` for top-level (`__main__`) code per the same convention
540    /// the bytecode compiler uses (`self.current_function == None` when
541    /// compiling top-level statements). For user functions, this is the
542    /// post-monomorphization specialized FunctionId (matches the
543    /// `func_idx: usize` passed to `compile_function_with_user_funcs` at
544    /// `compiler/program.rs:236`).
545    pub(crate) caller_function_id: Option<usize>,
546
547    /// ADR-006 §2.7.5 W10 jit-call-method-user-trait-fix (2026-05-17):
548    /// per-binop/unop-site operator-trait-dispatch side-table cloned from
549    /// `BytecodeProgram.operator_trait_dispatch_sites`. Consumed by
550    /// `compile_rvalue`'s `Rvalue::BinaryOp` / `Rvalue::UnaryOp` arms to
551    /// re-emit the bytecode-time trait-dispatch as a method-call
552    /// equivalent. Keyed by the statement span (matches MIR lowering's
553    /// `expr.span()`). Empty when the program has no user-type operator
554    /// overloading — JIT falls through to the existing typed-arith /
555    /// typed-cmp / unop lowering paths.
556    pub(crate) operator_trait_dispatch_sites:
557        HashMap<shape_ast::ast::span::Span, (String, u16)>,
558}
559
560/// Result of MIR preflight check.
561pub struct MirPreflightResult {
562    /// Whether this function can be compiled via MirToIR.
563    pub can_compile: bool,
564    /// Reasons why compilation is not possible (empty if can_compile is true).
565    pub blockers: Vec<String>,
566}
567
568/// Check if a function's MIR can be compiled by MirToIR.
569///
570/// Returns detailed preflight results. Functions with unsupported MIR
571/// features (async, closures, complex places) fall back to BytecodeToIR.
572pub fn preflight(mir_data: &MirFunctionData) -> MirPreflightResult {
573    let mut blockers = Vec::new();
574
575    for block in &mir_data.mir.blocks {
576        for stmt in &block.statements {
577            match &stmt.kind {
578                StatementKind::Assign(place, rvalue) => {
579                    if !is_simple_place(place) {
580                        blockers.push(format!(
581                            "complex place in assignment at {:?}",
582                            stmt.span
583                        ));
584                    }
585                    match rvalue {
586                        // W15.2-LANG-5 (Phase 4b, 2026-05-18). MIR-level
587                        // marker for `Pattern::Typed` arms in `match`
588                        // expressions. JIT codegen is not yet wired
589                        // (`compile_rvalue` surfaces-and-stops on this
590                        // variant); preflight rejects so the W12 fall-
591                        // through routes the program to the bytecode
592                        // interpreter, which compiles typed patterns via
593                        // `OpCode::TypeCheck` in
594                        // `compiler/patterns/checking.rs`. ADR-006 §2.7.5
595                        // producer-side classification: the annotation is
596                        // carried verbatim from `ast::Pattern::Typed`.
597                        Rvalue::TypePatternTest { type_annotation, .. } => {
598                            blockers.push(format!(
599                                "TypePatternTest (W15.2-LANG-5): \
600                                 `Pattern::Typed` codegen pending, \
601                                 annotation = {:?} at {:?}",
602                                type_annotation, stmt.span
603                            ));
604                        }
605                        // W15.2-LANG-1 (Phase 4b, 2026-05-18). MIR-level
606                        // marker for non-trinity (user-defined) `Pattern::
607                        // Constructor` arms in `match` expressions
608                        // (e.g. `match Color::Red { Color::Red => ..., ...
609                        // }`). JIT codegen is not yet wired (`compile_rvalue`
610                        // surfaces-and-stops on this variant); preflight
611                        // rejects so the W12 fall-through routes the program
612                        // to the bytecode interpreter, which compiles user-
613                        // defined enum patterns via the typed-object
614                        // discriminant check at `compile_typed_enum_pattern_
615                        // check` in `compiler/patterns/checking.rs`
616                        // (emits `GetFieldTyped(__variant, I64)` +
617                        // `PushConst(variant_id)` + `EqInt`). ADR-006
618                        // §2.7.5 producer-side classification: the
619                        // (enum_name, variant_name) pair is carried verbatim
620                        // from `ast::Pattern::Constructor`. Mirrors the
621                        // LANG-5 `TypePatternTest` precedent.
622                        Rvalue::EnumDiscriminantTest {
623                            enum_name,
624                            variant_name,
625                            ..
626                        } => {
627                            blockers.push(format!(
628                                "EnumDiscriminantTest (W15.2-LANG-1): \
629                                 user-defined `Pattern::Constructor` codegen \
630                                 pending, enum = {:?}, variant = {:?} at {:?}",
631                                enum_name, variant_name, stmt.span
632                            ));
633                        }
634                        // R8 W9 G.2 Step 2 Bucket 2 EnumPayload SURFACE
635                        // (ADR-006 §2.7.14 / §2.7.17, supervisor 2026-05-25).
636                        // `Rvalue::EnumPayload { variant: Ok|Err|Some_ }`
637                        // is the MIR-level marker for the payload binder in
638                        // `Pattern::Constructor` arms like `Ok(path)`,
639                        // `Some(p)`, `Err(m)`. The JIT codegen at
640                        // `compile_rvalue` calls `jit_arc_*_payload` which
641                        // casts the operand bits to `*const ResultData` /
642                        // `*const OptionData`; when the operand is the
643                        // return slot of a user-defined fn whose return
644                        // shape doesn't actually carry the strict
645                        // `Arc<ResultData>` / `Arc<OptionData>` carrier
646                        // (i.e. the §2.7.17 receiver-recovery soundness rule
647                        // is violated at the call-site producer because the
648                        // return-kind track threads an `Arc<HeapValue>`
649                        // pointer instead), the cast is UB and produces
650                        // either silent-wrong-output (e.g. Result<int,int>
651                        // payload `42` returning `8589934634` — the i64
652                        // overlapped by a neighbouring slot's bits) or a
653                        // SIGSEGV (e.g. `Result<string,string>` payload
654                        // dereferencing a HeapValue::String through a
655                        // `*const ResultData` layout offset). Empirically
656                        // observed at HEAD on Option<string> → Result<string,
657                        // string> match-destruct (deterministic ec=139
658                        // SIGSEGV) and Result<int,int> match-destruct
659                        // (deterministic silent-wrong-output).
660                        //
661                        // Mirrors the W15.2-LANG-5 / LANG-1 preflight
662                        // precedent above + R8 W7 G.5 HashMap key-kind +
663                        // R8 W8 Cluster A imported-const-inline / aliased-
664                        // CoW typed-array-push surface-and-stop. Whole-
665                        // program deopt via W12 `[jit-fallback]` routes the
666                        // program to the bytecode interpreter where the
667                        // EnumPayload Rvalue is compiled to opcodes that
668                        // dispatch on the actual carrier shape (not the
669                        // JIT's strict `Arc<*Data>` cast). Root-cause fix
670                        // — extending §2.7.17 receiver-recovery to the
671                        // user-fn return-kind boundary so the producer at
672                        // the call-site stamps the strict carrier per
673                        // ADR-006 §2.7.5 — is v0.4 per
674                        // `docs/v0.3-close-summary.md` §5.16 JIT-lowering
675                        // followup workstream.
676                        Rvalue::EnumPayload { variant, .. } => {
677                            blockers.push(format!(
678                                "EnumPayload (R8 W9 G.2 Step 2 Bucket 2): \
679                                 `Pattern::Constructor` payload binder \
680                                 (`Ok(_)` / `Err(_)` / `Some(_)`) codegen \
681                                 has receiver-recovery soundness gap at the \
682                                 user-fn return-kind boundary per ADR-006 \
683                                 \u{a7}2.7.17; whole-program deopt via W12 \
684                                 `[jit-fallback]` routes to the bytecode \
685                                 interpreter (which compiles EnumPayload via \
686                                 the kind-aware opcode dispatch). \
687                                 variant = {:?} at {:?}. Tracked v0.4 per \
688                                 `docs/v0.3-close-summary.md` \u{a7}5.16 \
689                                 JIT-lowering followup workstream.",
690                                variant, stmt.span
691                            ));
692                        }
693                        // BinaryOp, UnaryOp, Use, Clone, Borrow, Aggregate,
694                        // EnumTest are supported
695                        _ => {}
696                    }
697                }
698                StatementKind::Drop(place) => {
699                    if !is_simple_place(place) {
700                        blockers.push(format!("complex place in drop at {:?}", stmt.span));
701                    }
702                }
703                StatementKind::TaskBoundary(_, _) => {
704                    // TaskBoundary is a borrow-checker annotation — no-op at codegen time.
705                }
706                StatementKind::ClosureCapture { function_id, .. } => {
707                    // ClosureCapture is supported when function_id has been patched
708                    if function_id.is_none() {
709                        blockers.push("ClosureCapture missing function_id".to_string());
710                    }
711                }
712                _ => {}
713            }
714        }
715
716        match &block.terminator.kind {
717            TerminatorKind::Goto(_)
718            | TerminatorKind::SwitchBool { .. }
719            | TerminatorKind::Return
720            | TerminatorKind::Unreachable => {}
721            TerminatorKind::Call { .. } => {
722                // Call terminators are now supported via FFI dispatch.
723            }
724        }
725    }
726
727    MirPreflightResult {
728        can_compile: blockers.is_empty(),
729        blockers,
730    }
731}
732
733/// Check if a Place is supported by MirToIR.
734/// Supports arbitrary nesting of Local, Field, and Index.
735/// Only Deref (references) is unsupported.
736fn is_simple_place(place: &Place) -> bool {
737    match place {
738        Place::Local(_) => true,
739        Place::Field(inner, _) | Place::Index(inner, _) => is_simple_place(inner),
740        Place::Deref(inner) => is_simple_place(inner),
741    }
742}
743
744impl<'a, 'b> MirToIR<'a, 'b> {
745    /// Create a new MIR-to-IR compiler.
746    ///
747    /// `entry_block` is the Cranelift block already created by the caller
748    /// (with function parameters appended). MIR bb0 maps to this block.
749    pub fn new(
750        builder: &'a mut FunctionBuilder<'b>,
751        ctx_ptr: Value,
752        ffi: FFIFuncRefs,
753        mir_data: &'a MirFunctionData,
754        slot_kinds: Vec<Option<NativeKind>>,
755        strings: &'a [String],
756        entry_block: Block,
757        function_indices: &'a HashMap<String, u16>,
758        user_func_refs: HashMap<u16, FuncRef>,
759        user_func_arities: HashMap<u16, u16>,
760    ) -> Self {
761        Self::new_with_concrete_types(
762            builder,
763            ctx_ptr,
764            ffi,
765            mir_data,
766            slot_kinds,
767            Vec::new(),
768            strings,
769            entry_block,
770            function_indices,
771            user_func_refs,
772            user_func_arities,
773        )
774    }
775
776    /// Same as `new` but also accepts a per-slot `ConcreteType` vector for
777    /// the v2 typed-array fast path. Empty vec → legacy NaN-boxed behaviour.
778    pub fn new_with_concrete_types(
779        builder: &'a mut FunctionBuilder<'b>,
780        ctx_ptr: Value,
781        ffi: FFIFuncRefs,
782        mir_data: &'a MirFunctionData,
783        slot_kinds: Vec<Option<NativeKind>>,
784        concrete_types: Vec<ConcreteType>,
785        strings: &'a [String],
786        entry_block: Block,
787        function_indices: &'a HashMap<String, u16>,
788        user_func_refs: HashMap<u16, FuncRef>,
789        user_func_arities: HashMap<u16, u16>,
790    ) -> Self {
791        Self::new_with_closure_layouts(
792            builder,
793            ctx_ptr,
794            ffi,
795            mir_data,
796            slot_kinds,
797            concrete_types,
798            strings,
799            entry_block,
800            function_indices,
801            user_func_refs,
802            user_func_arities,
803            HashMap::new(),
804        )
805    }
806
807    /// Closure-spec Phase H1 constructor: also accepts a
808    /// `function_id → ClosureLayout` map so `emit_heap_closure` can lay out
809    /// captures for escaping closures without going through the
810    /// `jit_make_closure` FFI. Passing an empty map degrades gracefully to
811    /// the legacy FFI path (same behaviour as `new_with_concrete_types`).
812    pub fn new_with_closure_layouts(
813        builder: &'a mut FunctionBuilder<'b>,
814        ctx_ptr: Value,
815        ffi: FFIFuncRefs,
816        mir_data: &'a MirFunctionData,
817        slot_kinds: Vec<Option<NativeKind>>,
818        concrete_types: Vec<ConcreteType>,
819        strings: &'a [String],
820        entry_block: Block,
821        function_indices: &'a HashMap<String, u16>,
822        user_func_refs: HashMap<u16, FuncRef>,
823        user_func_arities: HashMap<u16, u16>,
824        closure_function_layouts: HashMap<u16, Arc<ClosureLayout>>,
825    ) -> Self {
826        let local_types = mir_data.mir.local_types.clone();
827        // Slot-numbering correction: the bytecode compiler's
828        // `FrameDescriptor.slots` and the MIR's local slots use different
829        // numbering. MIR reserves `SlotId(0)` for the implicit return
830        // value (`__mir_return`) and numbers parameters starting at 1;
831        // the bytecode compiler puts the first parameter at slot 0 with
832        // no implicit return slot. Seeding MirToIR with bytecode
833        // frame_descriptor kinds thus misaligns every slot by +1. In the
834        // worst case this declares MIR's return slot with the bytecode
835        // param's `NativeKind`, so a `return 7.0` write gets narrowed
836        // (e.g. `F64 → Bool` via `ireduce`) and corrupts the return value.
837        // Regression case: `fn get_val(flag: bool) -> number? { if flag
838        // { return 7.0 } return None }` declared MIR slot 0 as `Bool`
839        // because the bytecode put `flag` at index 0; writing the `7.0`
840        // F64 through `ensure_kind(_, Bool)` truncated to 0 and
841        // `None ?? 42.0` then evaluated to 42.0 for every branch.
842        //
843        // Until the two tables share a slot-numbering convention, drop
844        // the bytecode seed and rely on MIR-level inference only.
845        let _ = slot_kinds;
846        // ADR-006 §2.7.7 / §2.7.11 kind-source seed: when the bytecode
847        // compiler has populated `concrete_types[slot]` with a precise
848        // `ConcreteType`, project it to `NativeKind` for the parallel-kind
849        // track. This is the load-bearing kind source for closure-bearing
850        // slots returned from function calls (e.g. `let add3 =
851        // make_adder(3)` where `make_adder` returns
852        // `Function<(int), int>` / `ConcreteType::Closure`), which
853        // `infer_slot_kinds` alone cannot derive from MIR-observable
854        // statements.
855        let concrete_seed: Vec<Option<NativeKind>> = concrete_types
856            .iter()
857            .map(|ct| types::native_kind_from_concrete_type(ct))
858            .collect();
859        // ADR-006 §2.7.5 producing-site classification: pass the per-
860        // slot `ConcreteType` map into the inference so two projections
861        // both work end-to-end —
862        //
863        // (1) W12-jit-binop-after-heap-read-kind-tracker (Round 5A):
864        // `Place::Field` reads stamp the destination kind from the
865        // FIELD's kind, not the base struct's heap kind (drives the
866        // Smoke 3 `p.x + p.y` int-add).
867        //
868        // (2) W12-jit-print-kind (Round 5C): `Place::Index` reads off
869        // typed-array slots stamp the destination kind from the
870        // element kind, not the array's pointer kind — same source the
871        // JIT codegen-side `place_native_kind` /
872        // `v2_typed_array_elem_kind` projection uses. Without this seed
873        // `print(xs[0])` on `xs: Array<int>` falls into the kind-blind
874        // print decoder.
875        let slot_kinds = types::infer_slot_kinds_with_concrete(
876            &mir_data.mir,
877            &concrete_seed,
878            &concrete_types,
879        );
880        // Phase E: pull the set of non-escaping closure slots out of the MIR
881        // storage plan so `ClosureCapture` lowering can pick the stack-slot
882        // fast path. Slots absent from this set fall back to the legacy
883        // `jit_make_closure` FFI path (Phase H will delete that).
884        let non_escaping_closure_slots =
885            mir_data.storage_plan.non_escaping_closure_slots.clone();
886
887        // Session 1 Commit 3: scan `storage_plan` for outer-scope
888        // local slots that actually get promoted to
889        // `Arc<SharedCell>` storage at runtime. The bytecode
890        // compiler emits `AllocSharedLocal` ONLY when a slot is
891        // captured by a closure AND gets the Shared capture kind —
892        // not for every SharedCow slot. The `SHAPE_V2_VAR_SHAREDCOW`
893        // default classifies every `var` binding as SharedCow even
894        // when it never escapes, so we cannot use the storage class
895        // alone.
896        //
897        // The authoritative signal is `slot_semantics[slot]
898        // .escape_status == Captured` AND
899        // `slot_classes[slot] == SharedCow`. Captured-by-closure +
900        // SharedCow is the exact condition under which the bytecode
901        // compiler emits `AllocSharedLocal` (see
902        // `expressions/closures.rs`'s `is_shared_local_slot` arm).
903        //
904        // Param slots (captures) are further excluded because they
905        // are governed by the capture-side-tables
906        // `owned_mutable_capture_slots` / `shared_capture_slots`.
907        //
908        // cell-identity #1: the storage-plan scan alone is NOT
909        // sufficient. The MIR's storage planner classifies a slot's
910        // ownership from `binding_semantics`, and on some pipelines
911        // a `var` binding arrives at the planner as
912        // `BindingOwnershipClass::OwnedImmutable` rather than
913        // `Flexible` — so Rule 1b (`SHAPE_V2_VAR_SHAREDCOW` +
914        // Flexible → SharedCow) does not fire and the slot lands as
915        // `Direct` / `LocalMutablePtr` even though the bytecode
916        // emits the `AllocSharedLocal` lifecycle against it. The
917        // second scan below covers the gap by picking up every slot
918        // that is an operand of a `ClosureCapture` whose layout
919        // declares a `CaptureKind::Shared` capture at that position.
920        use shape_vm::type_tracking::{BindingStorageClass, EscapeStatus};
921        let param_slot_set: HashSet<SlotId> =
922            mir_data.mir.param_slots.iter().copied().collect();
923        let mut shared_local_slots: HashSet<SlotId> = HashSet::new();
924        for (slot, class) in &mir_data.storage_plan.slot_classes {
925            if !matches!(class, BindingStorageClass::SharedCow) {
926                continue;
927            }
928            if param_slot_set.contains(slot) {
929                continue;
930            }
931            // Only slots captured by a closure get the cell
932            // promotion at the bytecode level. A `var` that never
933            // escapes into a closure stays plain-valued in the
934            // interpreter — the JIT must match that semantics or
935            // diverge from the interpreter's view of the same slot.
936            let is_captured = mir_data
937                .storage_plan
938                .slot_semantics
939                .get(slot)
940                .map(|sem| matches!(sem.escape_status, EscapeStatus::Captured))
941                .unwrap_or(false);
942            if !is_captured {
943                continue;
944            }
945            shared_local_slots.insert(*slot);
946        }
947
948        // cell-identity #1: augment `shared_local_slots` by scanning
949        // `ClosureCapture` statements whose `function_id` resolves to a
950        // `ClosureLayout` with `CaptureKind::Shared` captures. The MIR
951        // storage planner sometimes classifies `var` bindings as
952        // `LocalMutablePtr` (not `SharedCow`) when the ownership class
953        // for the slot is stored as `OwnedImmutable` in the MIR's
954        // `binding_semantics` table, so the storage-plan scan above
955        // misses them. The bytecode compiler still emits `AllocSharedLocal`
956        // / `LoadSharedLocal` / `StoreSharedLocal` / `DropSharedLocal`
957        // for those slots — and the closure body's JIT compilation
958        // treats its capture param slot as `shared_capture_slots`
959        // (it expects a `*const SharedCell` pointer). If the declaring
960        // frame's JIT doesn't allocate an `Arc<SharedCell>` and doesn't
961        // lock-gated route reads/writes through it, the closure gets a
962        // plain scalar bit pattern as its "cell pointer" — and the
963        // closure's first `jit_arc_shared_retain` on that value
964        // segfaults. Driving the side-table off the layout's
965        // `CaptureKind::Shared` mask closes the gap: any slot that is
966        // an operand of a Shared capture in a call to a layout-carrying
967        // function is promoted to the Arc<SharedCell> lowering path.
968        use shape_value::v2::closure_layout::CaptureKind;
969        use shape_vm::mir::types::{Operand as MirOperand, Place as MirPlace, StatementKind};
970        for block in &mir_data.mir.blocks {
971            for stmt in &block.statements {
972                let StatementKind::ClosureCapture {
973                    operands,
974                    function_id,
975                    ..
976                } = &stmt.kind
977                else {
978                    continue;
979                };
980                let Some(fid) = *function_id else {
981                    continue;
982                };
983                let Some(layout) = closure_function_layouts.get(&fid) else {
984                    continue;
985                };
986                for (i, op) in operands.iter().enumerate() {
987                    if i >= layout.capture_count() {
988                        break;
989                    }
990                    if !matches!(layout.capture_storage_kind(i), CaptureKind::Shared) {
991                        continue;
992                    }
993                    let root = match op {
994                        MirOperand::Copy(p)
995                        | MirOperand::Move(p)
996                        | MirOperand::MoveExplicit(p) => match p {
997                            MirPlace::Local(s) => Some(*s),
998                            _ => None,
999                        },
1000                        MirOperand::Constant(_) => None,
1001                    };
1002                    if let Some(slot) = root {
1003                        if param_slot_set.contains(&slot) {
1004                            // Capture-side slot: handled by the
1005                            // `shared_capture_slots` side-table via
1006                            // `register_owned_mutable_capture_slots`.
1007                            continue;
1008                        }
1009                        shared_local_slots.insert(slot);
1010                    }
1011                }
1012            }
1013        }
1014
1015        // JIT-side fallback for unresolved `ClosurePlaceholder` constants.
1016        // See the `closure_placeholder_fids` doc-comment on `MirToIR` for
1017        // why this is needed; in short, monomorphization's `compile_function`
1018        // clears `closure_function_ids` in the bytecode compiler before the
1019        // top-level MIR back-patching runs, so some placeholders leak into
1020        // the MIR we receive. This scan produces the same pairing the
1021        // bytecode's back-patcher would have, keyed on MIR traversal order.
1022        let closure_placeholder_fids =
1023            scan_closure_placeholder_fids(&mir_data.mir, function_indices);
1024
1025        // W12-jit-binop-after-heap-read-kind-tracker (ADR-006 §2.7.5):
1026        // pre-pass the MIR for every `StatementKind::ObjectStore` and
1027        // record each named operand's inferred kind. This makes
1028        // `Place::Field(_, field_idx)` reads available with a proven
1029        // kind at JIT compile time, so the downstream `Rvalue::BinaryOp`
1030        // lowering picks the typed inline arithmetic path instead of
1031        // surfacing `compile_binop_dynamic_arith`. Pre-pass placement
1032        // (rather than during `compile_statement`) makes the kind
1033        // available for cross-block field reads, mirroring how
1034        // `infer_slot_kinds` and the §2.7.5 conduit's
1035        // `infer_top_level_concrete_types_from_mir` already work.
1036        let field_native_kinds =
1037            types::infer_field_native_kinds(&mir_data.mir, &slot_kinds);
1038
1039        // Phase 4b Round 5c-2-α jit-ref-param-chain-stamp (ADR-006 §2.7.13
1040        // ref-chain stamp + §2.7.5 producer-side stamp; supervisor ratify
1041        // 2026-05-19). Populate `ref_param_slots` from MIR-lowering's
1042        // `param_reference_kinds` — entries with `Some(BorrowKind::_)` are
1043        // reference parameters whose slot holds the BORROWED CELL ADDRESS
1044        // (allocated by `Rvalue::Borrow` in the caller's frame), NOT the
1045        // referenced value directly. Read/write sites for these slots
1046        // dispatch through the cell-indirection path; see
1047        // `read_place` / `write_place` / `null_place` and the
1048        // `Rvalue::Borrow` short-circuit in `rvalues.rs`.
1049        let ref_param_slots: HashSet<SlotId> = mir_data
1050            .mir
1051            .param_slots
1052            .iter()
1053            .zip(mir_data.mir.param_reference_kinds.iter())
1054            .filter_map(|(slot, kind)| kind.as_ref().map(|_| *slot))
1055            .collect();
1056
1057        Self {
1058            builder,
1059            ctx_ptr,
1060            ffi,
1061            entry_block,
1062            block_map: HashMap::new(),
1063            locals: HashMap::new(),
1064            local_types,
1065            slot_kinds,
1066            concrete_types,
1067            next_var: 0,
1068            mir: &mir_data.mir,
1069            mir_data,
1070            strings,
1071            function_indices,
1072            user_func_refs,
1073            user_func_arities,
1074            ref_stack_slots: HashMap::new(),
1075            field_byte_offsets: HashMap::new(),
1076            field_native_kinds,
1077            field_array_elem_kinds: HashMap::new(),
1078            non_escaping_closure_slots,
1079            stack_closure_slots: HashMap::new(),
1080            stack_closure_call_info: HashMap::new(),
1081            closure_function_layouts,
1082            owned_mutable_capture_slots: HashMap::new(),
1083            shared_capture_slots: HashMap::new(),
1084            shared_local_slots,
1085            closure_placeholder_fids,
1086            next_closure_placeholder_idx: std::cell::Cell::new(0),
1087            bounds_elision: bounds_elision::BoundsElisionPlan::default(),
1088            // V3-S6c: side-table + caller-id default empty/None; populated
1089            // by `set_monomorph_routing_context` from the JIT compile
1090            // orchestration layer (`compiler/program.rs` per-function path +
1091            // `compiler/strategy.rs` top-level path).
1092            monomorphized_method_call_sites: HashMap::new(),
1093            caller_function_id: None,
1094            // W10 jit-call-method-user-trait-fix: operator-trait-dispatch
1095            // side-table default empty; populated by
1096            // `set_operator_trait_dispatch_sites` from the JIT orchestration
1097            // layer. Empty is sound — JIT falls through to typed-arith/cmp
1098            // lowering identically to pre-W10 behaviour.
1099            operator_trait_dispatch_sites: HashMap::new(),
1100            ref_param_slots,
1101        }
1102    }
1103
1104    /// W10 jit-call-method-user-trait-fix (2026-05-17): install the
1105    /// bytecode compiler's `operator_trait_dispatch_sites` side-table so
1106    /// `compile_rvalue`'s `Rvalue::BinaryOp` / `Rvalue::UnaryOp` arms can
1107    /// re-emit user-type operator overloading as a method call. Sibling
1108    /// of `set_monomorph_routing_context` — same threading pattern.
1109    pub fn set_operator_trait_dispatch_sites(
1110        &mut self,
1111        sites: HashMap<shape_ast::ast::span::Span, (String, u16)>,
1112    ) {
1113        self.operator_trait_dispatch_sites = sites;
1114    }
1115
1116    /// V3-S6c JIT method-monomorph routing: install the bytecode compiler's
1117    /// `monomorphized_method_call_sites` side-table + the caller function
1118    /// id used for the `(span, caller_function_id)` composite key. Callers
1119    /// normally clone `program.monomorphized_method_call_sites` and pass
1120    /// the post-monomorphization `func_idx: usize` (per-function path) or
1121    /// `None` (top-level path). An empty map / `None` caller is sound —
1122    /// every Method-call falls through to the existing `jit_call_method`
1123    /// trampoline path, preserving V3-S6b baseline behaviour.
1124    pub fn set_monomorph_routing_context(
1125        &mut self,
1126        sites: HashMap<(shape_ast::ast::span::Span, Option<usize>), usize>,
1127        caller_function_id: Option<usize>,
1128    ) {
1129        self.monomorphized_method_call_sites = sites;
1130        self.caller_function_id = caller_function_id;
1131    }
1132
1133    /// Install a precomputed bounds-elision plan so `Place::Index` codegen
1134    /// can skip the inline bounds check on trusted access pairs.
1135    ///
1136    /// Callers normally invoke `bounds_elision::analyze(&mir_data.mir)` and
1137    /// pass the result here. Leaving the plan empty (the default) is
1138    /// always sound — every access falls back to the bounds-checked path,
1139    /// matching pre-elision behaviour and preserving the v2_array_tests
1140    /// OOB zero-default semantics.
1141    pub fn set_bounds_elision_plan(&mut self, plan: bounds_elision::BoundsElisionPlan) {
1142        self.bounds_elision = plan;
1143    }
1144
1145    /// W14.2-E-followup-jit-trait-method-arity-soundness fix (SURFACE-A2,
1146    /// 2026-05-19, v0.3-gating SOUNDNESS BUG): pre-populate
1147    /// `field_byte_offsets` from the program's `type_schema_registry` for
1148    /// every field name visible in this function's MIR `field_name_table`.
1149    ///
1150    /// **Background.** The existing `field_byte_offsets` map is populated
1151    /// only by `StatementKind::ObjectStore` walks at codegen time
1152    /// (`mir_compiler/statements.rs:243`). Trait-impl method bodies (and
1153    /// generally any function that READS fields but does not CONSTRUCT
1154    /// typed objects) never emit `ObjectStore`, so the map stays empty
1155    /// and `try_resolve_field_byte_offset` returns `None`. Field reads
1156    /// then fall through to the `jit_get_prop(obj_bits, key_bits)` FFI
1157    /// (`places.rs:899-906`), whose `heap_kind(obj_bits)` discriminator
1158    /// (`ffi/value_ffi.rs:331-336`) requires `is_heap(bits)` — i.e.
1159    /// `is_tagged(bits) && get_tag(bits) == TAG_HEAP_BITS`. Under ADR-006
1160    /// §2.7.5 the JIT typed-object allocator (`jit_typed_object_alloc` at
1161    /// `ffi/typed_object/allocation.rs:83`) returns raw `Box::into_raw`
1162    /// pointers without NaN-box tag bits, so `is_heap` always returns
1163    /// false and `jit_get_prop` returns `TAG_NULL` for a TypedObject
1164    /// receiver — the empirical garbage NaN-bits at the
1165    /// `vm_trait_method_self_field_access_n0` reproducer.
1166    ///
1167    /// **Fix.** Use the program-wide schema registry to map each field
1168    /// name in the MIR to its position in the carrying schema. The JIT
1169    /// data layout (`typed_object_alloc(schema_id, field_count * 8)`)
1170    /// uses 8-byte slots per field regardless of declared field type, so
1171    /// `byte_offset = field_index * 8`. Same shape as the existing
1172    /// ObjectStore-walk at `statements.rs:243`.
1173    ///
1174    /// **Discriminator caveat.** This shares the "last-writer-wins on
1175    /// name collision across distinct struct types" caveat documented at
1176    /// `field_native_kinds`'s comment (mod.rs:170-180). For impl method
1177    /// bodies the receiver is one specific struct type, so the collision
1178    /// is benign in practice; the principled schema-aware
1179    /// `(StructLayoutId, FieldIdx) → offset` registry remains the
1180    /// long-term shape per that comment's "out of scope" note. The W12-
1181    /// jit-binop-after-heap-read-kind-tracker invariant is preserved:
1182    /// when a function contains both ObjectStore (local-populate at
1183    /// statements.rs:243) AND field reads on different types, the
1184    /// schema-pre-pass runs FIRST (here, at MirToIR construction time)
1185    /// and the local ObjectStore-walk overwrites for the constructed
1186    /// type — matching the existing single-name single-offset contract.
1187    ///
1188    /// Per ADR-006 §2.7.5 producer-side stamp: schema field positions
1189    /// are stamped at AST→bytecode-compile time (the canonical schema
1190    /// registry); the JIT's `field_byte_offsets` is a derived index, not
1191    /// a runtime decode.
1192    pub fn populate_field_byte_offsets_from_schemas(
1193        &mut self,
1194        registry: &shape_runtime::type_schema::TypeSchemaRegistry,
1195    ) {
1196        use shape_runtime::type_schema::FieldType;
1197        // Collect every field name referenced in this function's MIR.
1198        let referenced_names: std::collections::HashSet<&str> = self
1199            .mir
1200            .field_name_table
1201            .values()
1202            .map(|s| s.as_str())
1203            .collect();
1204
1205        // Schema FieldType → NativeKind projection. Mirrors the JIT
1206        // slot encoding for typed-object fields: every field occupies an
1207        // 8-byte slot regardless of the declared field width, so the
1208        // NativeKind classification follows the field's declared type
1209        // (Int64 for `int` / I64, Float64 for `number` / F64, Bool for
1210        // `bool`, String for `string`, etc.). Width-specific integers
1211        // project to Int64 since they are stored as raw i64 bits in the
1212        // 8-byte JIT slot (the alignment-clamped layout at
1213        // `mir_compiler/statements.rs:233`).
1214        fn field_type_to_native_kind(ft: &FieldType) -> Option<shape_value::NativeKind> {
1215            use shape_value::NativeKind;
1216            match ft {
1217                FieldType::F64 => Some(NativeKind::Float64),
1218                FieldType::I64 => Some(NativeKind::Int64),
1219                FieldType::Bool => Some(NativeKind::Bool),
1220                FieldType::String => Some(NativeKind::String),
1221                // Width-specific ints project to Int64 in the JIT slot —
1222                // they are stored as raw i64 bits per the typed-object
1223                // 8-byte slot encoding.
1224                FieldType::I8
1225                | FieldType::U8
1226                | FieldType::I16
1227                | FieldType::U16
1228                | FieldType::I32
1229                | FieldType::U32
1230                | FieldType::U64 => Some(NativeKind::Int64),
1231                FieldType::Timestamp => Some(NativeKind::Int64),
1232                // Object/Array/Decimal/Any/HashMap/Set: not projected —
1233                // leave as None so the downstream consumer falls back to
1234                // the existing surface-and-stop / FFI dispatch path. The
1235                // principled projection for nested-object fields
1236                // requires a typed pointer kind that the JIT-side
1237                // carrier discipline (`Ptr(HeapKind::TypedObject)`) does
1238                // not yet thread through schema-recovered reads (W10
1239                // jit-playbook §5). W17.3-4.1 adds HashMap/Set to the
1240                // same None-fallback shape as Array/Option — runtime
1241                // dispatch + JIT typed-pointer-kind threading for the
1242                // new containers lands at W17.3-4.3.
1243                FieldType::Object(_)
1244                | FieldType::Array(_)
1245                | FieldType::Option(_)
1246                | FieldType::Decimal
1247                | FieldType::Any
1248                | FieldType::HashMap { .. }
1249                | FieldType::Set(_) => None,
1250            }
1251        }
1252
1253        // γ-CP5 7a (jit-typedarray-ptr): project the *element* type of a
1254        // scalar `Array<T>` field declaration into the matching v2
1255        // typed-array element `NativeKind`. Mirrors
1256        // `mir_compiler/types.rs::elem_slot_kind_for_concrete` (which
1257        // operates on `ConcreteType`); here the source is the schema's
1258        // declared `FieldType`. Non-scalar element types
1259        // (`Object`/`Array`/`Option`/`Any`) and `Decimal` return `None`
1260        // — those need heap-element carrier discipline the inline
1261        // `v2_array_get` fast path does not provide, so the consumer
1262        // falls back to the legacy NaN-boxed array path. `String`
1263        // elements likewise return `None`: `Array<string>` reads through
1264        // the v2-raw `*const StringObj` carrier need a retain-on-read
1265        // that the scalar `v2_array_get` does not emit.
1266        fn array_elem_to_native_kind(ft: &FieldType) -> Option<shape_value::NativeKind> {
1267            use shape_value::NativeKind;
1268            let FieldType::Array(elem) = ft else {
1269                return None;
1270            };
1271            match elem.as_ref() {
1272                FieldType::F64 => Some(NativeKind::Float64),
1273                FieldType::I64 | FieldType::Timestamp => Some(NativeKind::Int64),
1274                FieldType::Bool => Some(NativeKind::Bool),
1275                FieldType::I8 | FieldType::U8 => Some(NativeKind::Int8),
1276                FieldType::I16 | FieldType::U16 => Some(NativeKind::Int16),
1277                FieldType::I32 | FieldType::U32 => Some(NativeKind::Int32),
1278                FieldType::U64 => Some(NativeKind::UInt64),
1279                FieldType::String
1280                | FieldType::Decimal
1281                | FieldType::Object(_)
1282                | FieldType::Array(_)
1283                | FieldType::Option(_)
1284                | FieldType::Any
1285                // W17.3-4.1 — HashMap<K, V> / Set<T> elements are
1286                // not scalar element types; fall back to legacy
1287                // NaN-boxed array path (matches Array/Option shape).
1288                | FieldType::HashMap { .. }
1289                | FieldType::Set(_) => None,
1290            }
1291        }
1292
1293        // Walk every registered schema; map name → position (i*8).
1294        // Existing `field_byte_offsets` / `field_native_kinds` entries
1295        // from the local ObjectStore-walk (statements.rs:243 +
1296        // `infer_field_native_kinds` at types.rs:1534, populated at
1297        // construction / codegen time) take precedence — we only insert
1298        // when absent so the local-walk's per-constructor stamp wins on
1299        // collision.
1300        for type_name in registry.type_names().collect::<Vec<_>>() {
1301            let Some(schema) = registry.get(type_name) else {
1302                continue;
1303            };
1304            for (i, field) in schema.fields.iter().enumerate() {
1305                if !referenced_names.contains(field.name.as_str()) {
1306                    continue;
1307                }
1308                // 8-byte JIT slot offsets per `typed_object_alloc(
1309                // schema_id, field_count * 8)` at
1310                // `mir_compiler/statements.rs:233`.
1311                let byte_off = (i as u16) * 8;
1312                self.field_byte_offsets
1313                    .entry(field.name.clone())
1314                    .or_insert(byte_off);
1315                // Per ADR-006 §2.7.5 producer-side stamp: project the
1316                // schema's declared FieldType into the matching
1317                // NativeKind so `place_native_kind(Place::Field(...))`
1318                // can return a precise kind for impl-method-body field
1319                // reads. The downstream `compile_rvalue::BinaryOp`
1320                // picker reads this kind via `operand_slot_kind` →
1321                // `place_native_kind` → `field_native_kinds.get(name)`
1322                // (see `mir_compiler/rvalues.rs:496-516`) to select the
1323                // typed inline arithmetic path (`compile_binop_int64` /
1324                // `compile_binop_f64`) instead of surfacing
1325                // `compile_binop_dynamic_arith`. Without this stamp
1326                // `self.value * 2` in an impl method body falls into
1327                // the dynamic-arith surface-and-stop arm — the empirical
1328                // SURFACE-A2 garbage NaN-bits root cause.
1329                if let Some(kind) = field_type_to_native_kind(&field.field_type) {
1330                    self.field_native_kinds
1331                        .entry(field.name.clone())
1332                        .or_insert(kind);
1333                }
1334                // γ-CP5 7a: stamp the v2 typed-array element kind for
1335                // scalar `Array<T>` fields so `v2_typed_array_elem_kind`
1336                // can recognise a `Place::Field` base and emit the v2
1337                // inline `v2_array_get` codegen (v2 layout data@8/len@16).
1338                if let Some(elem_kind) = array_elem_to_native_kind(&field.field_type) {
1339                    self.field_array_elem_kinds
1340                        .entry(field.name.clone())
1341                        .or_insert(elem_kind);
1342                }
1343            }
1344        }
1345    }
1346
1347    /// Track A.1D.2: register the leading capture param slots that back
1348    /// an `OwnedMutable` capture cell for the closure body currently being
1349    /// compiled.
1350    ///
1351    /// `captures_count` is the number of leading entries in
1352    /// `MirFunction::param_slots` that correspond to closure captures
1353    /// (the caller ABI stores captures before user params: `[ctx_ptr,
1354    /// capture_0..N, user_param_0..M]`). `layout` is the
1355    /// `ClosureLayout` for this function's `function_id`, so
1356    /// `layout.capture_storage_kind(i)` reports the per-capture
1357    /// `CaptureKind`. Slots whose kind is `OwnedMutable` are flagged —
1358    /// `read_place` and `write_place` then emit a pointer-deref load /
1359    /// store through the raw `*mut ValueWord` bits, matching the A.1B
1360    /// interpreter handlers.
1361    ///
1362    /// Also patches `self.slot_kinds` for each capture param slot using
1363    /// the layout's `capture_types[i]`. Closure params are untyped at
1364    /// the bytecode compiler level (see `compile_expr_closure` in
1365    /// `expressions/closures.rs` — capture params are synthesised with
1366    /// `type_annotation: None`), so MIR-level inference leaves them
1367    /// `Unknown`. Without per-capture kinds the `Rvalue::BinaryOp`
1368    /// lowering falls through to the dynamic-binop path, which
1369    /// unconditionally errors out (see `compile_binop` at
1370    /// `rvalues.rs::~411`). Patching the slot kind here lets the
1371    /// typed binop pickers (`compile_binop_int64`, `compile_binop_f64`,
1372    /// etc.) engage for `x + 1`-style closure-body arithmetic. For
1373    /// OwnedMutable slots, `read_place` always emits `load.i64` through
1374    /// the cell — the kind informs the binop picker about the inner
1375    /// value's representation (NaN-boxed int, NaN-boxed float, etc.),
1376    /// not the width of the slot itself.
1377    ///
1378    /// No-op for non-closure functions (`captures_count == 0`) and for
1379    /// closures whose layout marks every capture as `Immutable`. A.1E
1380    /// extends this registration to populate `shared_capture_slots`
1381    /// alongside `owned_mutable_capture_slots`; both side-tables are
1382    /// parallel in structure but drive different lowering paths (see
1383    /// their doc-comments on `MirToIR`).
1384    pub fn register_owned_mutable_capture_slots(
1385        &mut self,
1386        captures_count: u16,
1387        layout: &ClosureLayout,
1388    ) {
1389        use shape_value::v2::closure_layout::CaptureKind;
1390        let captures_count = captures_count as usize;
1391        if captures_count == 0 {
1392            return;
1393        }
1394        // Defensive: the layout must have a capture_kinds entry per
1395        // declared capture. A mismatch indicates a compiler bug upstream
1396        // (e.g. the layout was minted against a different signature); we
1397        // clamp to the smaller of the two so no out-of-bounds panics
1398        // slip into release builds.
1399        let len = captures_count.min(layout.capture_kinds.len());
1400        for (i, &param_slot) in self
1401            .mir
1402            .param_slots
1403            .iter()
1404            .take(len)
1405            .enumerate()
1406        {
1407            let capture_kind = layout.capture_storage_kind(i);
1408            let is_cell_capture = matches!(
1409                capture_kind,
1410                CaptureKind::OwnedMutable | CaptureKind::Shared
1411            );
1412            if !is_cell_capture {
1413                continue;
1414            }
1415            // Wave C.2: capture the cell's interior FieldKind alongside
1416            // the slot id so `read_place`/`write_place` can pick the
1417            // matching per-kind FFI helper instead of NaN-boxing through
1418            // the legacy ValueWord-bits path.
1419            let inner_kind = layout.capture_inner_kind(i);
1420            match capture_kind {
1421                CaptureKind::OwnedMutable => {
1422                    self.owned_mutable_capture_slots
1423                        .insert(param_slot, inner_kind);
1424                }
1425                CaptureKind::Shared => {
1426                    self.shared_capture_slots.insert(param_slot, inner_kind);
1427                }
1428                CaptureKind::Immutable => unreachable!(),
1429            }
1430            // Propagate the layout's known concrete type onto the
1431            // slot kind vector so `Rvalue::BinaryOp` lowering can
1432            // pick the typed arithmetic path. Only patch when the
1433            // slot was previously `Unknown`; a non-Unknown kind
1434            // from the bytecode frame descriptor wins. Same
1435            // treatment applies to OwnedMutable and Shared — in
1436            // both cases `read_place` returns an I64 ValueWord-
1437            // shaped value and the downstream binop picker keys on
1438            // the kind, not the cell-pointer width itself.
1439            if let Some(concrete) = layout.capture_types.get(i) {
1440                if let Some(kind) = types::elem_slot_kind_for_concrete(concrete) {
1441                    let idx = param_slot.0 as usize;
1442                    if idx < self.slot_kinds.len() && self.slot_kinds[idx].is_none() {
1443                        self.slot_kinds[idx] = Some(kind);
1444                    }
1445                }
1446            }
1447        }
1448    }
1449
1450    /// Compile the MIR function to Cranelift IR.
1451    ///
1452    /// Returns Ok(()) on success. The actual return instructions are emitted
1453    /// by compile_terminator for TerminatorKind::Return blocks.
1454    /// Full compilation: create blocks, declare locals, initialize, compile body.
1455    /// Used when the caller hasn't set up blocks/locals externally.
1456    pub fn compile(&mut self) -> Result<(), String> {
1457        self.create_blocks();
1458        self.declare_locals();
1459        // Session 1 Commit 3: eagerly materialise Arc<SharedCell>s for
1460        // every SharedCow local slot. No-op when the set is empty.
1461        self.initialize_shared_local_slots();
1462        self.compile_body()
1463    }
1464
1465    /// Compile the MIR function body (blocks already created, locals already declared).
1466    /// Called after the caller has optionally stored function params to local variables.
1467    /// `param_count` indicates how many leading slots are function params (skip init).
1468    pub fn compile_body(&mut self) -> Result<(), String> {
1469        // Cluster-2 closure-wave-F tracing-crate migration (2026-05-16):
1470        // replaces SHAPE_JIT_MIR_TRACE env-var. CLI selector is
1471        // `--trace-jit=shape_jit::mir=trace`. The enabled-check gates the
1472        // entire MIR-walk so feature-OFF builds skip the iteration cost.
1473        if tracing::enabled!(target: "shape_jit::mir", tracing::Level::TRACE) {
1474            for (bi, block) in self.mir.blocks.iter().enumerate() {
1475                tracing::trace!(
1476                    target: "shape_jit::mir",
1477                    bb = bi,
1478                    stmts = block.statements.len(),
1479                    term = ?block.terminator.kind,
1480                    "mir-trace block",
1481                );
1482                for (si, stmt) in block.statements.iter().enumerate() {
1483                    tracing::trace!(
1484                        target: "shape_jit::mir",
1485                        bb = bi,
1486                        s = si,
1487                        stmt = ?stmt.kind,
1488                        "mir-trace statement",
1489                    );
1490                }
1491            }
1492        }
1493        for block_idx in 0..self.mir.blocks.len() {
1494            let block = &self.mir.blocks[block_idx];
1495            let cl_block = self.block_map[&block.id];
1496
1497            // bb0 is the caller's entry block — already switched to and sealed.
1498            // For other blocks, switch to the new block.
1499            if block_idx != 0 {
1500                self.builder.switch_to_block(cl_block);
1501            }
1502
1503            // DON'T initialize locals here — the caller has already:
1504            // 1. Called declare_locals() (all vars declared)
1505            // 2. Stored function params to their slots (params have real values)
1506            // Cranelift's SSA handles undefined variables as 0/default.
1507
1508            // Compile statements.
1509            for stmt in &block.statements {
1510                self.compile_statement(stmt)?;
1511            }
1512
1513            // Compile terminator.
1514            self.compile_terminator(&block.terminator)?;
1515        }
1516
1517        // Seal all blocks after all code is emitted. Sealing before all
1518        // predecessors are known causes Cranelift assertion failures.
1519        self.builder.seal_all_blocks();
1520
1521        Ok(())
1522    }
1523
1524    /// Reload all locals that have been borrowed via Rvalue::Borrow.
1525    ///
1526    /// After a function call, the callee may have mutated values through
1527    /// shared references. We conservatively reload all referenced locals
1528    /// from their StackSlots to keep Cranelift variables in sync.
1529    ///
1530    /// R4.2F: stack cells are now native-sized/aligned (matching the root
1531    /// local's Cranelift type), so `stack_load` directly produces a value
1532    /// of the declared variable's type — no NaN-box unboxing needed.
1533    pub(crate) fn reload_referenced_locals(&mut self) {
1534        let refs: Vec<_> = self
1535            .ref_stack_slots
1536            .iter()
1537            .map(|(&slot_id, &(stack_slot, cl_ty))| (slot_id, stack_slot, cl_ty))
1538            .collect();
1539        for (slot_id, stack_slot, cl_ty) in refs {
1540            let reloaded = self.builder.ins().stack_load(cl_ty, stack_slot, 0);
1541            if let Some(&var) = self.locals.get(&slot_id) {
1542                self.builder.def_var(var, reloaded);
1543            }
1544        }
1545    }
1546}
1547
1548pub(crate) mod v2_call_abi;
1549
1550/// Reconstruct the bytecode compiler's back-patch pairing for unresolved
1551/// `ClosurePlaceholder` constants in a MIR function, keyed on statement
1552/// traversal order.
1553///
1554/// # Why this exists
1555///
1556/// The bytecode compiler's `closure_function_ids` vector — the list of
1557/// `("__closure_N", function_id)` pairs used to patch
1558/// `MirConstant::ClosurePlaceholder` into `MirConstant::Function(name)` —
1559/// is cleared by `compile_function` (see
1560/// `shape-vm/src/compiler/functions.rs:510`). Monomorphization triggered
1561/// during a top-level call (e.g. `arr.map(|x| x*2)`) goes through
1562/// `ensure_monomorphic_function` → `compile_function`, clearing the
1563/// vector even though the top-level MIR's back-patching hasn't yet run.
1564/// The result: any no-capture closure literal in top-level code leaks
1565/// into the top-level MIR as a raw `ClosurePlaceholder`.
1566///
1567/// The downstream effect is that `compile_constant` lowers the
1568/// placeholder to a literal `iconst 0`, the callsite pushes `0x0` as the
1569/// callee, and `jit_call_value` BAILs with "callee is neither function
1570/// nor closure". The gated `parity_array_map/filter/reduce` tests all
1571/// hit this exact path.
1572///
1573/// # Scan semantics
1574///
1575/// Mirrors the bytecode compiler's loop in
1576/// `functions.rs::compile_function` (and the top-level analogue in
1577/// `compiler_impl_reference_model.rs`): walk every block's statement
1578/// list; a `ClosureCapture` with a resolved `function_id` "claims" the
1579/// immediately-following `ClosurePlaceholder` (which the patcher rewrites
1580/// to `Nop`), so the placeholder does NOT need independent resolution.
1581/// An unpaired placeholder consumes the next `__closure_<idx>` name.
1582///
1583/// # Name-to-id lookup
1584///
1585/// We can't consult `closure_function_ids` from the JIT, but
1586/// `function_indices` (built from `BytecodeProgram::functions`) has every
1587/// `__closure_N` entry in the same global ordering the bytecode compiler
1588/// assigned. Looking up `__closure_<idx>` directly produces the correct
1589/// function_id for each unpaired placeholder, in the same sequence the
1590/// unpatched back-patcher would have produced.
1591///
1592/// # Graceful degradation
1593///
1594/// If the lookup misses (e.g. a malformed program where `__closure_<idx>`
1595/// is absent), we push `u16::MAX` as a sentinel. `compile_constant`'s
1596/// fallback path then emits the legacy `iconst 0` so behaviour is no
1597/// worse than before this scan.
1598fn scan_closure_placeholder_fids(
1599    mir: &shape_vm::mir::types::MirFunction,
1600    function_indices: &std::collections::HashMap<String, u16>,
1601) -> Vec<u16> {
1602    use shape_vm::mir::types::{MirConstant, Operand, Rvalue, StatementKind};
1603
1604    let mut result: Vec<u16> = Vec::new();
1605    let mut closure_idx: u32 = 0;
1606    let mut has_capture = false;
1607    for block in &mir.blocks {
1608        for stmt in &block.statements {
1609            let is_placeholder = matches!(
1610                &stmt.kind,
1611                StatementKind::Assign(
1612                    _,
1613                    Rvalue::Use(Operand::Constant(MirConstant::ClosurePlaceholder))
1614                )
1615            );
1616            if is_placeholder {
1617                if has_capture {
1618                    // Paired with a preceding ClosureCapture — the
1619                    // bytecode back-patcher would have turned this into
1620                    // a Nop. compile_constant still receives the
1621                    // placeholder for this slot (the patched MIR would
1622                    // not), so we record u16::MAX and let the fallback
1623                    // iconst(0) fire; ClosureCapture's function_id
1624                    // drives the actual closure allocation in
1625                    // `emit_heap_closure` / `emit_stack_closure`, and
1626                    // the subsequent Assign(slot, placeholder) is a
1627                    // dead store that write_place discards.
1628                    result.push(u16::MAX);
1629                    has_capture = false;
1630                } else {
1631                    let name = format!("__closure_{}", closure_idx);
1632                    let fid = function_indices.get(&name).copied().unwrap_or(u16::MAX);
1633                    result.push(fid);
1634                    closure_idx = closure_idx.saturating_add(1);
1635                }
1636                continue;
1637            }
1638            if let StatementKind::ClosureCapture {
1639                function_id: Some(_),
1640                ..
1641            } = &stmt.kind
1642            {
1643                // The bytecode patcher consumes one closure_id for the
1644                // capture itself — advance the counter so the unpaired
1645                // placeholder counter stays aligned with the compiler's.
1646                closure_idx = closure_idx.saturating_add(1);
1647                has_capture = true;
1648            }
1649        }
1650    }
1651    result
1652}